From 8270fab38c473c98814ef3ee306543cb2e7ea76a Mon Sep 17 00:00:00 2001 From: Tony Ketcham Date: Sat, 18 Jul 2026 20:49:44 -0700 Subject: [PATCH] feat(effort-graph): ship agent runtime, skills, and planning graph Promote the Effort Graph from write-only memory into an agent-facing runtime: digests/read APIs, `flatbread effort` CLI, versioned skill packaging with sync/pack CI gates, and a dogfooded `.flatbread-efforts` planning authority that replaces the ADR/roadmap docs stack. Also hardens publish/bump versioning for skill release metadata, updates onboarding to the unified watch loop, and adds the root Flatbread config used to query this repo's own effort graph. ## Note Follow-up graph update after the initial push: accept **Crumb Trail** as the product/brand name for agent memory (supersedes Crumb Graph as the product name). Crumb Graph stays as the datamodel explainer; Effort Graph remains an allowed technical alias. Package/CLI/path renames stay deferred under a new implementation issue. ## Summary - Add effort digests, generation-aware reads, and `flatbread effort` CLI - Package effort-graph / effort-modeling / grill-with-efforts skills with sync + pack checks in CI - Commit the dogfooded `.flatbread-efforts` graph; retire ADRs, roadmap, and stale proposal docs - Improve bump/publish scripts for skill release pinning; refresh CONTRIBUTING and public READMEs - Brand decision: Crumb Trail (product) vs Crumb Graph (datamodel explainer) ## Test plan - [ ] `pnpm skills:check && pnpm skills:pack-check` - [ ] `pnpm verify` (or at least build + effort-graph / flatbread effort CLI tests) - [ ] `pnpm exec flatbread effort list` / `get` against root `flatbread.config.js` - [ ] Spot-check CONTRIBUTING + package READMEs for watch-loop / effort skill install paths - [ ] Confirm Crumb Trail / Crumb Graph branding records resolve cleanly via `flatbread effort` Co-authored-by: Cursor Change-Id: Ibb417555cd91d868417671c56fe4d4613f932041 --- .agents/skills/effort-graph/SKILL.md | 155 ++++ .agents/skills/effort-graph/glossary.md | 64 ++ .agents/skills/effort-graph/reference.md | 166 ++++ .agents/skills/effort-graph/release.json | 6 + .agents/skills/effort-graph/setup.md | 86 +++ .../skills/effort-modeling/CONTEXT-FORMAT.md | 14 + .../skills/effort-modeling/DECISION-BODY.md | 29 + .agents/skills/effort-modeling/SKILL.md | 60 ++ .agents/skills/grill-with-efforts/SKILL.md | 15 + ...rface-not-the-product--4966nqjsnsmwq1fs.md | 9 + ...ys-deliberately-small--45v1ae3neq26g1rz.md | 26 + ...-markdown-yaml-writer--wrawg3b0hp779hd8.md | 11 + ...e-the-source-of-truth--pnd7w2bs2zvk7ye9.md | 9 + ...cli-and-unified-watch--k9dpxj57yrcv0pbz.md | 15 + ...fing-fast-path-for-ef--kcw0rw39g3b2ym2h.md | 31 + ...urface-as-crumb-graph--fvskcvagx3a7sybe.md | 41 + ...urface-as-crumb-trail--tngncepdbwjkh9jc.md | 46 ++ ...tions-into-live-reads--mc3728t4w1kyqcqq.md | 37 + ...-journaled-save-or-un--sv9x93svkfz4a98r.md | 48 ++ ...read-write-extraction--xxa1ge2chw6x3xx0.md | 39 + ...-as-a-versioned-agent--8as4sybr34zcqyqh.md | 34 + ...nclude-the-full-recor--xs7rnbha3zx8qdj9.md | 37 + ...ction-and-nullability--0khw2zkc51hy230t.md | 15 + ...xtract-validation-api--e96yngmrjqcjdxcn.md | 11 + ...inement-as-bounded-lo--9ptqq1ck3sax00fq.md | 32 + ...-with-a-primary-wedge--estattvqnhffm2dc.md | 15 + ...n-first-content-layer--rc3pc0vczp6qhp3t.md | 15 + ...he-planning-authority--gq4xegmc0hhbqjf7.md | 18 + ...before-replacing-tool--q7xpn1g7rk65xvzj.md | 32 + ...he-flatbread-query-en--476qb9qk878yfg62.md | 57 ++ ...on-and-display-planes--spm2ckxvdsch6h9m.md | 37 + ...ver-effort-scoped-edg--pxwaenyra0aa365j.md | 30 + ...ts-in-repo-by-default--xxpgwm9sv25j07z7.md | 41 + ...ories-without-a-cache--8bn823dg29yfvbzv.md | 33 + ...with-reviewable-slugs--xdg764ejat1kacd2.md | 43 ++ ...-not-separate-schemas--27p0wakkyxj0kfc1.md | 11 + ...d-a-standalone-writer--2d0m3tkqhad4yyhr.md | 45 ++ ...emory-and-agent-wedge--szeqvmgqjqnhd002.md | 9 + ...read-product-branding--zt7b35sa05kyvhdz.md | 9 + ...me-and-ownership-loop--g28gfbb0kdrnpe2t.md | 9 + ...utor-operating-system--ahhgtafvdhg4dfve.md | 9 + ...al-content-foundation--8a8332x4cazgf2k0.md | 9 + ...nd-planning-authority--j1waeg8qh900sqee.md | 9 + ...tocol-halves-effort-g--ht1jd1hsssf7mv2y.md | 18 + ...faced-three-real-defe--g21hg77x72hdty56.md | 9 + ...p-but-lack-external-v--fx2qtjpcm8mtpw4f.md | 11 + ...is-94-percent-smaller--qpvz4vch5hygye20.md | 11 + ...ul-but-not-fully-type--vmr23k0t6qx3hk1k.md | 9 + ...same-excerpt-as-brows--q5rbnhpxc697x2dv.md | 15 + ...eeds-mapping-profiles--k2yvq5e9rcpsm1s1.md | 11 + ...but-retains-test-gaps--91h2ncmytj8mq40r.md | 20 + ...ped-validation-and-wa--p04gd8xfknwvz2pe.md | 21 + ...avor-trail-over-graph--zzpgaw0kvqtaqmxt.md | 15 + ...passes-the-unified-wa--qqvckprax45hyd2y.md | 19 + ...ts-users-to-watch-mod--cwgqyjjvr6j9gx8n.md | 22 + ...seded-live-reload-gui--naafmpvt3pcdes4p.md | 20 + ...y-is-roadmap-critical--2ss712xpmsfh77xf.md | 9 + ...-reaches-a-typed-read--8anm4wxbjm9f348z.md | 9 + ...ce-is-already-removed--w56t43aa10v311hj.md | 15 + ...o-the-deleted-roadmap--k38zmv7pqv7z95ts.md | 19 + ...d-per-build-isolation--vdjfmtqb0dbjfcqk.md | 9 + ...ntended-runtime-contr--t9ghag8yqxgf3p5t.md | 9 + ...n-passes-full-reposit--hm5cy9j2zts96dg4.md | 11 + ...rumb-graph-over-yeast--tv3ydt3wfb3n4y0g.md | 16 + ...record-bodies-through--ekfpcg6hrkwgy287.md | 15 + ...ng-decision-semantics--3bj9ph7ppab2c7g2.md | 16 + ...e-across-packages-cli--dp6jvt2kafab7m4t.md | 18 + ...e-across-packages-cli--316rdzpt80sdxkw1.md | 18 + ...xport-cli-remain-gaps--xe19gzqf654xeyg4.md | 12 + ...ollow-ups-remain-open--4fkjb93v3pey43y0.md | 12 + ...-example-over-constra--vd0gcnpc9cm6jzjh.md | 14 + ...remains-a-product-gap--4zawc913n4gnyfcw.md | 12 + ...ow-prompts-and-leak-s--3p7ybk07jqe5w593.md | 11 + .github/workflows/pipeline.yml | 6 + .gitignore | 3 +- CONTRIBUTING.md | 62 +- docs/data-ownership.md | 7 +- docs/edit-file-see-query-update-demo.md | 20 +- docs/effort-graph/CONTEXT.md | 80 -- .../adr/0001-effort-graph-memory-location.md | 28 - .../0002-semantic-mutation-write-surface.md | 28 - ...e-path-and-read-after-write-consistency.md | 37 - .../adr/0004-multi-file-honesty.md | 31 - .../adr/0005-v1-semantic-mutation-enum.md | 39 - .../adr/0006-id-and-slug-strategy.md | 28 - docs/effort-graph/adr/0007-agent-read-shim.md | 41 - .../adr/0008-committed-generation-bridge.md | 50 -- docs/glossary.md | 31 +- docs/isolated-schema-factory.md | 9 - docs/json-export.md | 2 +- docs/local-dev-loop.md | 17 +- docs/pmf-decision-rubric.md | 99 +-- docs/positioning.md | 62 +- .../proof-bounded-convergence-loops.md | 206 ----- .../proposals/proof-output-retention-judge.md | 92 --- docs/proposals/proof-output-retention-plan.md | 419 ---------- .../proof-output-retention-review.md | 131 ---- docs/roadmap.md | 151 ---- docs/tooling-modernization.md | 181 ----- examples/nextjs/README.md | 60 +- examples/nextjs/app/components/BlogIndex.tsx | 4 +- examples/nextjs/package.json | 2 +- flatbread-flow-pmf-audit.md | 4 + flatbread.config.js | 13 + package.json | 6 +- packages/config/src/load.test.ts | 34 + packages/config/src/load.ts | 26 +- packages/effort-graph/README.md | 34 +- packages/effort-graph/package.json | 8 +- packages/effort-graph/scripts/pack-skills.mjs | 161 ++++ packages/effort-graph/scripts/sync-skills.mjs | 154 ++++ .../effort-graph/scripts/watch-skills.mjs | 99 +++ .../effort-graph/skills/effort-graph/SKILL.md | 155 ++++ .../skills/effort-graph/glossary.md | 64 ++ .../skills/effort-graph/reference.md | 166 ++++ .../skills/effort-graph/release.json | 6 + .../effort-graph/skills/effort-graph/setup.md | 86 +++ .../skills/effort-modeling/CONTEXT-FORMAT.md | 14 + .../skills/effort-modeling/DECISION-BODY.md | 29 + .../skills/effort-modeling/SKILL.md | 60 ++ .../skills/grill-with-efforts/SKILL.md | 15 + .../effort-graph/src/__tests__/digest.test.ts | 201 +++++ .../src/__tests__/pack-skills.test.js | 82 ++ .../effort-graph/src/__tests__/read.test.ts | 49 ++ .../effort-graph/src/__tests__/skills.test.ts | 61 ++ packages/effort-graph/src/digest.ts | 373 +++++++++ packages/effort-graph/src/index.ts | 17 + packages/effort-graph/src/read.ts | 222 ++++++ packages/effort-graph/src/schemas.ts | 2 +- packages/flatbread/README.md | 107 ++- packages/flatbread/src/cli/effort.test.ts | 465 ++++++++++++ packages/flatbread/src/cli/effort.ts | 513 +++++++++++++ packages/flatbread/src/cli/index.ts | 3 + packages/flatbread/src/effort/read.ts | 714 ++++++++++++++++++ packages/flatbread/src/index.ts | 1 + packages/source-filesystem/README.md | 2 +- packages/transformer-markdown/README.md | 2 +- pnpm-lock.yaml | 3 + scripts/bumpVersions.test.ts | 78 ++ scripts/bumpVersions.ts | 182 ++++- scripts/publish.test.ts | 111 +++ scripts/publish.ts | 234 +++++- scripts/utils/packageManifest.ts | 19 +- 143 files changed, 6578 insertions(+), 1816 deletions(-) create mode 100644 .agents/skills/effort-graph/SKILL.md create mode 100644 .agents/skills/effort-graph/glossary.md create mode 100644 .agents/skills/effort-graph/reference.md create mode 100644 .agents/skills/effort-graph/release.json create mode 100644 .agents/skills/effort-graph/setup.md create mode 100644 .agents/skills/effort-modeling/CONTEXT-FORMAT.md create mode 100644 .agents/skills/effort-modeling/DECISION-BODY.md create mode 100644 .agents/skills/effort-modeling/SKILL.md create mode 100644 .agents/skills/grill-with-efforts/SKILL.md create mode 100644 .flatbread-efforts/constraints/con-graphql-is-a-read-interface-not-the-product--4966nqjsnsmwq1fs.md create mode 100644 .flatbread-efforts/constraints/con-mutation-enum-stays-deliberately-small--45v1ae3neq26g1rz.md create mode 100644 .flatbread-efforts/constraints/con-prettier-remains-the-markdown-yaml-writer--wrawg3b0hp779hd8.md create mode 100644 .flatbread-efforts/constraints/con-repo-files-are-the-source-of-truth--pnd7w2bs2zvk7ye9.md create mode 100644 .flatbread-efforts/decisions/dec-add-export-cli-and-unified-watch--k9dpxj57yrcv0pbz.md create mode 100644 .flatbread-efforts/decisions/dec-adopt-a-bounded-status-briefing-fast-path-for-ef--kcw0rw39g3b2ym2h.md create mode 100644 .flatbread-efforts/decisions/dec-brand-the-agent-memory-surface-as-crumb-graph--fvskcvagx3a7sybe.md create mode 100644 .flatbread-efforts/decisions/dec-brand-the-agent-memory-surface-as-crumb-trail--tngncepdbwjkh9jc.md create mode 100644 .flatbread-efforts/decisions/dec-bridge-committed-generations-into-live-reads--mc3728t4w1kyqcqq.md create mode 100644 .flatbread-efforts/decisions/dec-canonical-forward-edges-and-journaled-save-or-un--sv9x93svkfz4a98r.md create mode 100644 .flatbread-efforts/decisions/dec-defer-shared-flatbread-write-extraction--xxa1ge2chw6x3xx0.md create mode 100644 .flatbread-efforts/decisions/dec-distribute-the-effort-graph-as-a-versioned-agent--8as4sybr34zcqyqh.md create mode 100644 .flatbread-efforts/decisions/dec-effort-get-digests-always-include-the-full-recor--xs7rnbha3zx8qdj9.md create mode 100644 .flatbread-efforts/decisions/dec-iterate-typed-selection-and-nullability--0khw2zkc51hy230t.md create mode 100644 .flatbread-efforts/decisions/dec-keep-and-extract-validation-api--e96yngmrjqcjdxcn.md create mode 100644 .flatbread-efforts/decisions/dec-keep-dags-acyclic-model-refinement-as-bounded-lo--9ptqq1ck3sax00fq.md create mode 100644 .flatbread-efforts/decisions/dec-keep-effort-graph-secondary-with-a-primary-wedge--estattvqnhffm2dc.md create mode 100644 .flatbread-efforts/decisions/dec-keep-relation-first-content-layer--rc3pc0vczp6qhp3t.md create mode 100644 .flatbread-efforts/decisions/dec-make-the-effort-graph-the-planning-authority--gq4xegmc0hhbqjf7.md create mode 100644 .flatbread-efforts/decisions/dec-modernize-reachable-checks-before-replacing-tool--q7xpn1g7rk65xvzj.md create mode 100644 .flatbread-efforts/decisions/dec-route-agent-reads-through-the-flatbread-query-en--476qb9qk878yfg62.md create mode 100644 .flatbread-efforts/decisions/dec-separate-execution-and-display-planes--spm2ckxvdsch6h9m.md create mode 100644 .flatbread-efforts/decisions/dec-specify-blockingdecisions-over-effort-scoped-edg--pxwaenyra0aa365j.md create mode 100644 .flatbread-efforts/decisions/dec-store-graph-artifacts-in-repo-by-default--xxpgwm9sv25j07z7.md create mode 100644 .flatbread-efforts/decisions/dec-use-isolated-schema-factories-without-a-cache--8bn823dg29yfvbzv.md create mode 100644 .flatbread-efforts/decisions/dec-use-prefixed-random-ids-with-reviewable-slugs--xdg764ejat1kacd2.md create mode 100644 .flatbread-efforts/decisions/dec-use-profiles-not-separate-schemas--27p0wakkyxj0kfc1.md create mode 100644 .flatbread-efforts/decisions/dec-use-semantic-mutations-and-a-standalone-writer--2d0m3tkqhad4yyhr.md create mode 100644 .flatbread-efforts/efforts/eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002.md create mode 100644 .flatbread-efforts/efforts/eff-flatbread-product-branding--zt7b35sa05kyvhdz.md create mode 100644 .flatbread-efforts/efforts/eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t.md create mode 100644 .flatbread-efforts/efforts/eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve.md create mode 100644 .flatbread-efforts/efforts/eff-relational-content-foundation--8a8332x4cazgf2k0.md create mode 100644 .flatbread-efforts/findings/fnd-adrs-created-a-second-planning-authority--j1waeg8qh900sqee.md create mode 100644 .flatbread-efforts/findings/fnd-bounded-status-briefing-protocol-halves-effort-g--ht1jd1hsssf7mv2y.md create mode 100644 .flatbread-efforts/findings/fnd-dogfooding-the-read-cli-surfaced-three-real-defe--g21hg77x72hdty56.md create mode 100644 .flatbread-efforts/findings/fnd-exports-strengthen-ownership-but-lack-external-v--fx2qtjpcm8mtpw4f.md create mode 100644 .flatbread-efforts/findings/fnd-filtered-retrieval-is-94-percent-smaller--qpvz4vch5hygye20.md create mode 100644 .flatbread-efforts/findings/fnd-generated-read-api-is-useful-but-not-fully-type--vmr23k0t6qx3hk1k.md create mode 100644 .flatbread-efforts/findings/fnd-getrecord-digests-call-the-same-excerpt-as-brows--q5rbnhpxc697x2dv.md create mode 100644 .flatbread-efforts/findings/fnd-one-canonical-schema-needs-mapping-profiles--k2yvq5e9rcpsm1s1.md create mode 100644 .flatbread-efforts/findings/fnd-phase-1-has-no-blockers-but-retains-test-gaps--91h2ncmytj8mq40r.md create mode 100644 .flatbread-efforts/findings/fnd-pmf-rubric-understates-shipped-validation-and-wa--p04gd8xfknwvz2pe.md create mode 100644 .flatbread-efforts/findings/fnd-postable-brand-names-favor-trail-over-graph--zzpgaw0kvqtaqmxt.md create mode 100644 .flatbread-efforts/findings/fnd-primary-onboarding-still-bypasses-the-unified-wa--qqvckprax45hyd2y.md create mode 100644 .flatbread-efforts/findings/fnd-public-onboarding-now-directs-users-to-watch-mod--cwgqyjjvr6j9gx8n.md create mode 100644 .flatbread-efforts/findings/fnd-public-readmes-retain-superseded-live-reload-gui--naafmpvt3pcdes4p.md create mode 100644 .flatbread-efforts/findings/fnd-reference-integrity-is-roadmap-critical--2ss712xpmsfh77xf.md create mode 100644 .flatbread-efforts/findings/fnd-relation-first-starter-reaches-a-typed-read--8anm4wxbjm9f348z.md create mode 100644 .flatbread-efforts/findings/fnd-root-readme-roadmap-reference-is-already-removed--w56t43aa10v311hj.md create mode 100644 .flatbread-efforts/findings/fnd-root-readme-still-links-to-the-deleted-roadmap--k38zmv7pqv7z95ts.md create mode 100644 .flatbread-efforts/findings/fnd-schemas-need-per-build-isolation--vdjfmtqb0dbjfcqk.md create mode 100644 .flatbread-efforts/findings/fnd-unified-watch-loop-is-the-intended-runtime-contr--t9ghag8yqxgf3p5t.md create mode 100644 .flatbread-efforts/findings/fnd-versioned-skill-distribution-passes-full-reposit--hm5cy9j2zts96dg4.md create mode 100644 .flatbread-efforts/findings/fnd-workshop-shortlist-favors-crumb-graph-over-yeast--tv3ydt3wfb3n4y0g.md create mode 100644 .flatbread-efforts/issues/iss-agents-cannot-obtain-full-record-bodies-through--ekfpcg6hrkwgy287.md create mode 100644 .flatbread-efforts/issues/iss-define-blocking-decision-semantics--3bj9ph7ppab2c7g2.md create mode 100644 .flatbread-efforts/issues/iss-implement-crumb-graph-rename-across-packages-cli--dp6jvt2kafab7m4t.md create mode 100644 .flatbread-efforts/issues/iss-implement-crumb-trail-rename-across-packages-cli--316rdzpt80sdxkw1.md create mode 100644 .flatbread-efforts/issues/iss-live-reload-and-export-cli-remain-gaps--xe19gzqf654xeyg4.md create mode 100644 .flatbread-efforts/issues/iss-proof-output-retention-follow-ups-remain-open--4fkjb93v3pey43y0.md create mode 100644 .flatbread-efforts/issues/iss-skill-scoped-records-filter-example-over-constra--vd0gcnpc9cm6jzjh.md create mode 100644 .flatbread-efforts/issues/iss-typed-selection-builder-remains-a-product-gap--4zawc913n4gnyfcw.md create mode 100644 .flatbread-efforts/risks/rsk-full-transcripts-can-overflow-prompts-and-leak-s--3p7ybk07jqe5w593.md delete mode 100644 docs/effort-graph/CONTEXT.md delete mode 100644 docs/effort-graph/adr/0001-effort-graph-memory-location.md delete mode 100644 docs/effort-graph/adr/0002-semantic-mutation-write-surface.md delete mode 100644 docs/effort-graph/adr/0003-write-path-and-read-after-write-consistency.md delete mode 100644 docs/effort-graph/adr/0004-multi-file-honesty.md delete mode 100644 docs/effort-graph/adr/0005-v1-semantic-mutation-enum.md delete mode 100644 docs/effort-graph/adr/0006-id-and-slug-strategy.md delete mode 100644 docs/effort-graph/adr/0007-agent-read-shim.md delete mode 100644 docs/effort-graph/adr/0008-committed-generation-bridge.md delete mode 100644 docs/isolated-schema-factory.md delete mode 100644 docs/proposals/proof-bounded-convergence-loops.md delete mode 100644 docs/proposals/proof-output-retention-judge.md delete mode 100644 docs/proposals/proof-output-retention-plan.md delete mode 100644 docs/proposals/proof-output-retention-review.md delete mode 100644 docs/roadmap.md delete mode 100644 docs/tooling-modernization.md create mode 100644 flatbread.config.js create mode 100644 packages/effort-graph/scripts/pack-skills.mjs create mode 100644 packages/effort-graph/scripts/sync-skills.mjs create mode 100644 packages/effort-graph/scripts/watch-skills.mjs create mode 100644 packages/effort-graph/skills/effort-graph/SKILL.md create mode 100644 packages/effort-graph/skills/effort-graph/glossary.md create mode 100644 packages/effort-graph/skills/effort-graph/reference.md create mode 100644 packages/effort-graph/skills/effort-graph/release.json create mode 100644 packages/effort-graph/skills/effort-graph/setup.md create mode 100644 packages/effort-graph/skills/effort-modeling/CONTEXT-FORMAT.md create mode 100644 packages/effort-graph/skills/effort-modeling/DECISION-BODY.md create mode 100644 packages/effort-graph/skills/effort-modeling/SKILL.md create mode 100644 packages/effort-graph/skills/grill-with-efforts/SKILL.md create mode 100644 packages/effort-graph/src/__tests__/digest.test.ts create mode 100644 packages/effort-graph/src/__tests__/pack-skills.test.js create mode 100644 packages/effort-graph/src/__tests__/read.test.ts create mode 100644 packages/effort-graph/src/__tests__/skills.test.ts create mode 100644 packages/effort-graph/src/digest.ts create mode 100644 packages/effort-graph/src/read.ts create mode 100644 packages/flatbread/src/cli/effort.test.ts create mode 100644 packages/flatbread/src/cli/effort.ts create mode 100644 packages/flatbread/src/effort/read.ts create mode 100644 scripts/bumpVersions.test.ts create mode 100644 scripts/publish.test.ts diff --git a/.agents/skills/effort-graph/SKILL.md b/.agents/skills/effort-graph/SKILL.md new file mode 100644 index 00000000..41f65698 --- /dev/null +++ b/.agents/skills/effort-graph/SKILL.md @@ -0,0 +1,155 @@ +--- +name: effort-graph +description: Journal reasoning (decisions, findings, issues, constraints, risks) into a Flatbread Effort Graph and recall it with bounded reads. Use when starting or resuming a thread of work, recording a decision or finding, resolving an issue, checking what is blocking or still open on an effort, or when the user mentions effort graph, journaling, blocking decisions, or agent memory. +--- + +# Effort Graph — agent journaling and recall + +The Effort Graph is persistent, queryable memory for long-horizon work, stored +as markdown records in the repo. Six primitives: **Effort** (the anchor thread +of work), **Issue**, **Finding**, **Decision**, **Constraint**, **Risk**. +Every record belongs to exactly one Effort. You write through 13 typed +mutations and read through 5 bounded queries — never by hand-editing record +frontmatter (bodies may be edited freely). + +Read [glossary.md](./glossary.md) for the primitive and edge semantics before +inventing a new record kind or relation. + +All commands run from the project root via the `flatbread` CLI (`pnpm exec flatbread`, `npm exec -- flatbread`, `yarn flatbread`, or `bunx flatbread`). +Commands print one JSON object to stdout; +errors print JSON to stderr and exit 1. + +## First activation + +Read [setup.md](./setup.md), make the reviewed config and gitignore edits, then +run `flatbread effort bootstrap` followed by `flatbread effort bootstrap --verify`. Bootstrap is report-only and never edits project files. + +## Prerequisites + +Your `flatbread.config.*` must include the preset: + +```js +import { + defineConfig, + sourceFilesystem, + transformerMarkdown, + effortGraphContent, +} from 'flatbread'; + +export default defineConfig({ + source: sourceFilesystem(), + transformer: transformerMarkdown(), + content: [...effortGraphContent()], +}); +``` + +Records live under `/{efforts,issues,findings,decisions,constraints,risks}/`. +The write journal is `/.journal/`; read digests cache under +`.flatbread/effort-graph/read-cache/` (both gitignored). + +## Writing (journaling) + +One command for all 13 mutations — pass the payload as a single JSON argument: + +```bash +flatbread effort write '{"type":"WriteDecision","effort":"","title":"...","body":"...","derives_from":[""]}' +``` + +Response: `{"generation":"","artifacts":[{"id","path","operation"}],"touched":[...]}`. +**Capture `artifacts[0].id`** to wire later edges, and **keep `generation`** +for strict read-your-writes. + +Full payload shapes for all 13 mutations: read [reference.md](./reference.md). +Critical semantics: + +- Creates always start in the initial lifecycle state: `WriteDecision` → + `proposed`, `WriteIssue` → `open`, `WriteRisk` → `open`. You cannot pass a + state; use lifecycle mutations (`AcceptDecision`, `ResolveIssue`, + `MitigateRisk`, `SetRiskState`) to transition. +- `AcceptDecision` defaults `rejectSiblings: true`, which rejects ALL other + proposed Decisions in the same Effort. Pass `"rejectSiblings": false` + unless you deliberately want the competing proposals closed. +- Edges are forward-only in payloads (`derives_from`, `supersedes`, + `invalidates`); back-edges are materialized automatically. +- When superseding, open the new record's body with a short rollup of what + changed and why — reads render ancestors only as one-line checkpoints. +- For a hard-to-reverse, surprising decision made after a real trade-off, use + the Decision body as the durable rationale: include context, alternatives, + consequences, and reversal criteria. Do not create a parallel ADR; use + [effort-modeling](../effort-modeling/SKILL.md) when the decision is still + being grilled. + +## Reading (recall) + +Every read returns a bounded envelope, not records: a ≤160-token `summary`, +an `artifact_path` to a rendered markdown digest (the evidence — spend one +Read on it, or grep it), `served_generation`, page info, and ≤10 executable +`hints`. Digests cap at 25 records / one-hop expansion / 50 edges / 64 KiB. + +Browse digests (`list`, `records`, `relations`, `blocking-decisions`) excerpt +each body at 600 chars / 12 lines (`[…truncated]`). **`effort get` digests +always include the full record body** (still subject to the 64 KiB digest +byte cap). Zoom in with `get`, then Read/grep that digest — do not open +`.flatbread-efforts/**/*.md` for normal full-body recall. + +```bash +# What's gating this effort? (proposed Decisions deriving from open blocker Issues) +flatbread effort blocking-decisions + +# Resume: discover active Efforts first +flatbread effort list --status active + +# Scoped listing with filters (AND across flags, OR within comma lists). +# --status filters Issues and --state filters Decisions, so combining them in +# one call ANDs across kinds and matches nothing — query each kind separately. +flatbread effort records --kinds issue --status open --since 2026-07-01T00:00:00Z --limit 10 +flatbread effort records --kinds decision --state proposed --limit 10 + +# One-hop neighbors of a record +flatbread effort relations --relations derives_from,superseded_by + +# Single record with full body; --resolve head follows supersession to the tip +flatbread effort get [--resolve head] +``` + +Flags shared by reads: `--strict-min-generation ` (with optional +`--timeout-ms `, default 3000) and, on `list`/`records`/`relations`, `--limit` +(≤25) and `--cursor` (opaque `next_cursor` from a prior page; only valid for +the same query at the same generation). + +`effort list` is bounded Effort discovery. It defaults to `active`; valid +statuses are exactly `active`, `paused`, `completed`, and `abandoned`. +Comma-separated statuses are ORed. Results use the shared `created_at`, then +`id` ordering. After discovery, use bounded effort-scoped reads. + +**Consistency:** reads are eventual by default. Immediately after a write, +pass the returned generation as `--strict-min-generation` — you get either +fresh data or an `EFFORT_GRAPH_GENERATION_WAIT_TIMEOUT` error (exit 1), +never silently stale results. Do not build polling loops; the wait is +server-side. + +## Recommended session workflow + +1. **Resume / status briefing (bounded fast-path):** `effort list --status active` + and trust the returned digest. For each active Effort, run + `effort records --kinds issue,decision` and read each record's + status/state from that one digest. Run `effort blocking-decisions ` + only for an Effort whose digest shows an open `blocker` Issue — skip it + otherwise. Do not open raw `.flatbread-efforts/**/*.md` for briefing; + browse digests are authoritative for status/state. Budget ≈ (1 + number + of active Efforts) digest reads. A 12-run experiment across three model + families showed this roughly halves recall tool calls with no loss of + answer quality (Decision + `dec-adopt-a-bounded-status-briefing-fast-path-for-ef--kcw0rw39g3b2ym2h`). +2. **When a browse digest shows `[…truncated]` and you need the body:** run + `flatbread effort get `, then Read/grep that digest (`artifact_path`) + for the full body. Reserve opening `.flatbread-efforts/**/*.md` for rare + cases (e.g. digest byte-cap miss on an oversized record), not normal + zoom-in. +3. **During work:** journal Findings as evidence lands; open Issues for real + gaps/blockers; record Decisions with `derives_from` citing the Findings, + Constraints, and Issues they respond to. +4. **On commitment:** `AcceptDecision` (mind `rejectSiblings`), `ResolveIssue` + with `resolvedBy` citing the closing Decision/Findings. +5. Maintenance: `flatbread effort cache prune` deletes digests older than + 24h / over the 100 MiB ceiling. diff --git a/.agents/skills/effort-graph/glossary.md b/.agents/skills/effort-graph/glossary.md new file mode 100644 index 00000000..b3b605bd --- /dev/null +++ b/.agents/skills/effort-graph/glossary.md @@ -0,0 +1,64 @@ +# Effort Graph glossary + +The Effort Graph is persistent, queryable memory for long-horizon software +work. It builds on Flatbread's content vocabulary: each primitive is a +Collection, its instances are Records, and cross-primitive references are +Relations in frontmatter. + +It is not a CMS, authoring UI, hosted memory product, or general task tracker. +Operational provenance (session, agent, model, DAG run) belongs in record +frontmatter; durable run transcripts live with Proof artifacts. + +## Primitives + +### Effort + +The anchor for one coherent thread of work: a feature, migration, spike, +research investigation, or refactor. Every other primitive belongs to exactly +one Effort. It scopes bounded reads but carries only a short description; the +reasoning belongs in the related records. + +### Issue + +A tracked item needing attention: a question, defect, gap, or blocker. An Issue +is reactive. Decisions and Findings resolve it through lifecycle edges. + +### Finding + +A grounded observation about code, users, literature, or runtime behavior. +Findings cite evidence, resolve Issues, inform Decisions, surface Risks, and +may invalidate past Findings or Decisions. A retrospective Finding is evidence +gathered after a decision shipped. + +### Decision + +A commitment among alternatives. A proposed Decision is an active alternative; +an accepted Decision is committed; rejected, superseded, and deprecated +Decisions retain their lifecycle history. A Decision cites the Findings, +Constraints, and Risks it weighed rather than duplicating them. + +### Constraint + +A sticky hard or soft boundary that limits the decision space. Constraints are +known limits; they are not prospective negative outcomes. + +### Risk + +A prospective negative outcome with likelihood and severity. It is open, +mitigated by an accepted Decision, realized with evidence, or explicitly +accepted. + +## Edges + +`derives_from` is causal upstream evidence or context. `supersedes` replaces a +record of the same primitive, while `invalidates` says a record was wrong. +Those forward edges are authoritative; `superseded_by` and `invalidated_by` are +writer-materialized reverse projections. New edge vocabulary needs a +dogfooded query the existing vocabulary cannot express. + +## Intentional non-models + +Session, Run, Plan, Artifact, Agent, Investigation, Question, Proposal, +Retrospective, and Branch are not collections. Use provenance fields for +operational data; represent questions as Issues, proposals as proposed +Decisions, retrospectives as Findings, and branch history through Git. diff --git a/.agents/skills/effort-graph/reference.md b/.agents/skills/effort-graph/reference.md new file mode 100644 index 00000000..1e7bf053 --- /dev/null +++ b/.agents/skills/effort-graph/reference.md @@ -0,0 +1,166 @@ +# Effort Graph — full API reference + +Ground truth: the installed `flatbread` CLI and this reference (mutations), +reads, and configuration examples. Repository implementation files are not +consumer ground truth. + +## IDs + +Generated as `---<16-char-crockford>` with prefixes `eff`, +`iss`, `fnd`, `dec`, `con`, `rsk`. Filenames never define identity. Let the +writer generate ids; capture them from mutation results (`artifacts[0].id` +for creates). + +## The 13 mutations (`flatbread effort write ''`) + +Common optional fields on all creates: `id`, `created_at` (ISO with offset), +`produced_in`, `created_by` (opaque provenance strings). Forward edge fields +on all creates except `CreateEffort`: `derives_from[]`, `supersedes[]`, +`invalidates[]` (arrays of existing ids; targets are validated). + +### Effort lifecycle + +```json +{"type":"CreateEffort","title":"...","body":"...","slug":"optional"} +{"type":"SetEffortStatus","effortId":"","status":"active|paused|completed|abandoned"} +``` + +### Creation (required: effort, title, body; initial state is derived) + +```json +{"type":"WriteIssue","effort":"","title":"...","body":"...","kind":"question|defect|gap|blocker|"} +{"type":"WriteFinding","effort":"","title":"...","body":"...","kind":"measurement|survey|dead-end|retrospective|"} +{"type":"WriteDecision","effort":"","title":"...","body":"..."} +{"type":"WriteConstraint","effort":"","title":"...","body":"...","kind":"hard|soft"} +{"type":"WriteRisk","effort":"","title":"...","body":"...","likelihood":"low|medium|high","severity":"low|medium|high"} +``` + +Initial states: Issue `status: open`; Decision `state: proposed`; Risk +`state: open`. + +### Edge retro-linking (records must already exist) + +```json +{"type":"Supersede","supersederId":"","targetId":""} +{"type":"Invalidate","findingId":"","targetId":""} +``` + +`Supersede` is same-primitive only and rejects an already-superseded target. +`Invalidate` asserts the target was wrong (stronger than superseded). + +### Lifecycle transitions + +```json +{"type":"ResolveIssue","issueId":"","resolution":"resolved|deferred|wontfix","resolvedBy":[""]} +{"type":"AcceptDecision","decisionId":"","rejectSiblings":false} +{"type":"MitigateRisk","riskId":"","decisionId":""} +{"type":"SetRiskState","riskId":"","state":"realized|accepted","evidence":[""]} +``` + +`AcceptDecision` with `rejectSiblings: true` (the default!) also sets every +other `proposed` Decision in the Effort to `rejected` with a back-pointer. +All mutations run in one journal transaction (save-or-undo). + +### Mutation result + +```json +{ + "generation": "57", + "artifacts": [ + { "id": "...", "path": "decisions/....md", "operation": "created|updated" } + ], + "touched": [{ "id": "...", "path": "..." }] +} +``` + +`generation` is a durable, monotonic journal token — the input to strict reads. + +## The 5 read queries + +All reads execute through Flatbread's query engine (in-process GraphQL over +the generated schema) and return a `ReadEnvelope`: + +```json +{ + "summary": "2 records; proposed 2; complete", + "artifact_path": ".flatbread/effort-graph/read-cache//.md", + "artifact_sha256": "...", + "served_generation": "55", + "consistency": { "mode": "eventual|strict", "min_generation": null }, + "page": { "returned": 2, "has_more": false, "next_cursor": null }, + "hints": ["getRecord(\"dec-...\")"] +} +``` + +The digest at `artifact_path` is deterministic markdown: YAML query header, +anchor index, per-record sections (selected frontmatter, body, relation +lists), one-hop related records, and an edge table. Body policy: + +- **`effort get`:** full record body (the normal zoom-in path). +- **`list` / `records` / `relations` / `blocking-decisions`:** body excerpt + capped at 600 chars / 12 lines (`[…truncated]`). + +Caps: 25 primary records, one hop, 50 edges, 64 KiB; hitting a cap sets +`complete: false` with named `cap_reasons` — narrow the query or page rather +than expecting more. If a `get` body alone exceeds the 64 KiB digest byte +cap, the digest fails closed with a byte-cap banner (it does **not** fake a +full body via the 600/12 excerpt). + +### Commands + +```bash +flatbread effort get [--resolve exact|head] [consistency flags] +flatbread effort list [--status active,paused,...] [--limit n] [--cursor c] [consistency flags] +flatbread effort records [--kinds k1,k2] [--state s1,s2] [--status s1,s2] [--kind k1,k2] [--since iso] [--until iso] [--limit n] [--cursor c] [consistency flags] +flatbread effort relations --relations r1,r2 [--limit n] [--cursor c] [consistency flags] +flatbread effort blocking-decisions [consistency flags] +flatbread effort cache prune +``` + +- `--kinds`: `effort|issue|finding|decision|constraint|risk` (records: + default all non-effort kinds). +- `list --status`: defaults to `active`; valid values are exactly `active`, + `paused`, `completed`, and `abandoned`. Values are ORed and results are + ordered by `created_at` ascending, then `id`. +- Filter semantics: AND across different flags, OR within a comma list. + `--since`/`--until` bound `created_at` (gte/lte, ISO strings). +- `--relations` values: `derives_from`, `supersedes`, `superseded_by`, + `invalidates`, `invalidated_by`, `rejected_by`, `mitigated_by`, + `resolved_by`, `evidence` (one hop, explicit only). +- `--resolve head`: follow `superseded_by` to the current tip; ancestors + render as checkpoint lines (max 5, then a count). +- `blocking-decisions` membership (frozen): Decision in the effort with + `state: proposed` whose `derives_from` directly contains an Issue in the + same effort with `kind: blocker` and `status: open`. For "what blockers + are open at all", use + `records --kinds issue --kind blocker --status open`. + +### Consistency flags + +- `--strict-min-generation `: serve at or after that journal + generation, or fail. `--timeout-ms ` bounds the wait (default 3000). +- Errors (stderr JSON, exit 1): `EFFORT_GRAPH_GENERATION_WAIT_TIMEOUT`, + `EFFORT_GRAPH_INVALID_CURSOR` (cursor reused across a different query or + generation). + +## Configuration surface + +| Option | Where | Default | Notes | +| ---------------- | --------------------------------------------------- | ------------------------------------------- | ------------------------------------------------------------------------------------------------------------- | +| Graph root | `effortGraphContent(root)` in `flatbread.config.js` | `.flatbread-efforts` | All six collection paths + refs derive from it; the preset must appear complete and unmodified for detection. | +| Config discovery | cwd of the CLI invocation | — | Exactly one `flatbread.config.*` must exist in cwd. | +| Digest cache | fixed | `/.flatbread/effort-graph/read-cache/` | Generation-keyed; gitignore it. `cache prune`: >24h old deleted, then oldest-first to ≤100 MiB. | +| Journal | fixed | `/.journal/` | Writer-owned; gitignored. Never edit. | +| Strict timeout | `--timeout-ms` per read | 3000 ms | | +| Page limit | `--limit` per read | 25 | Hard max 25. | + +## What not to do + +- Do not hand-edit record frontmatter or `.journal/`; bodies are freely + editable (the reindexer validates and repairs projections). +- Do not parse digest files as data feeds for other programs — they are + evidence for you to Read/grep; the envelope is the machine surface. +- Do not build polling loops around generations; strict reads wait + server-side. +- Do not model sessions/plans/agents as records — put provenance in + `produced_in` / `created_by` fields. diff --git a/.agents/skills/effort-graph/release.json b/.agents/skills/effort-graph/release.json new file mode 100644 index 00000000..a0d4235f --- /dev/null +++ b/.agents/skills/effort-graph/release.json @@ -0,0 +1,6 @@ +{ + "format": 1, + "flatbreadVersion": "1.0.0-alpha.22", + "effortGraphVersion": "0.1.0-alpha.0", + "gitTag": "v1.0.0-alpha.22" +} diff --git a/.agents/skills/effort-graph/setup.md b/.agents/skills/effort-graph/setup.md new file mode 100644 index 00000000..71a13793 --- /dev/null +++ b/.agents/skills/effort-graph/setup.md @@ -0,0 +1,86 @@ +# Effort Graph setup + +The canonical skill files live in this package. The repository +`.agents/skills/effort-graph/` directory is an exclusively generated +projection: do not edit it directly, and stale projected files are deleted by +`pnpm skills:sync`. + +## 1. Choose the package manager + +Use the nearest `package.json`'s `packageManager` field first. If it is absent, +inspect lockfiles. Exactly one of `package-lock.json`, `pnpm-lock.yaml`, +`yarn.lock`, or `bun.lock`/`bun.lockb` must exist. If multiple conflicting +lockfiles exist, ask the user which manager owns the project. + +For an end-user release, read `release.json` next to this file. It is the +canonical package and tag authority: use its `flatbreadVersion` and `gitTag` +values exactly. `skills-lock.json` is installation provenance/restore data only; +do not use its optional ref or version fields as release identity: + +```bash +npx skills add https://github.com/FlatbreadLabs/flatbread/tree//packages/effort-graph/skills/effort-graph --skill effort-graph +npm install --save-dev flatbread@ +``` + +Equivalent commands are: + +```bash +pnpm dlx skills add https://github.com/FlatbreadLabs/flatbread/tree//packages/effort-graph/skills/effort-graph --skill effort-graph +pnpm add -D flatbread@ + +yarn dlx skills add https://github.com/FlatbreadLabs/flatbread/tree//packages/effort-graph/skills/effort-graph --skill effort-graph +yarn add -D flatbread@ + +bunx skills add https://github.com/FlatbreadLabs/flatbread/tree//packages/effort-graph/skills/effort-graph --skill effort-graph +bun add -d flatbread@ +``` + +Do not use a floating branch, `latest`, or a guessed version. When dogfooding +the Flatbread monorepo, use its workspace `flatbread` binary and do not install +Flatbread from npm. + +## 2. Review the configuration + +Add the exports through the public `flatbread` facade and preserve existing +content entries: + +```js +import { + defineConfig, + sourceFilesystem, + transformerMarkdown, + effortGraphContent, +} from 'flatbread'; + +export default defineConfig({ + source: sourceFilesystem(), + transformer: transformerMarkdown(), + content: [ + // existing entries + ...effortGraphContent(), // or effortGraphContent('path/to/graph') + ], +}); +``` + +Add these entries to `.gitignore` (using the selected graph root): + +```gitignore +**/.flatbread-efforts/.journal/ +**/.flatbread/effort-graph/read-cache/ +``` + +For a custom root, replace `.flatbread-efforts` with that root. Review both +edits before saving; the bootstrap command never creates or rewrites them. + +## 3. Verify activation + +```bash +flatbread effort bootstrap +flatbread effort bootstrap --verify +``` + +The second command must print `{"status":"ready",...}` and exit successfully. +On resume, begin with `flatbread effort list --status active`, then use bounded +effort-scoped reads. Capture mutation `generation` tokens and use +`--strict-min-generation` for immediate read-after-write checks; never implement +client polling loops. Semantic changes go through `flatbread effort write`. diff --git a/.agents/skills/effort-modeling/CONTEXT-FORMAT.md b/.agents/skills/effort-modeling/CONTEXT-FORMAT.md new file mode 100644 index 00000000..6f0a6461 --- /dev/null +++ b/.agents/skills/effort-modeling/CONTEXT-FORMAT.md @@ -0,0 +1,14 @@ +# Context glossary format + +`CONTEXT.md` is a glossary, not a design document. Add terms as concise, +domain-specific definitions: + +```md +### Canonical term + +A precise definition in the project's language. State what it excludes when +that prevents a common ambiguity. +``` + +Do not put implementation plans, alternatives, or commitments here. Journal +those in Effort Graph records. diff --git a/.agents/skills/effort-modeling/DECISION-BODY.md b/.agents/skills/effort-modeling/DECISION-BODY.md new file mode 100644 index 00000000..4286ba4c --- /dev/null +++ b/.agents/skills/effort-modeling/DECISION-BODY.md @@ -0,0 +1,29 @@ +# Long-form Decision bodies + +Use this structure only when the decision needs durable rationale. Omit empty +sections; the title and body should stay readable in a source markdown file. + +```md +## Context + +What makes this decision necessary now? + +## Decision + +What are we committing to? + +## Alternatives considered + +- **Option:** Why it was not chosen. + +## Consequences + +What becomes easier, harder, required, or intentionally deferred? + +## Reversal criteria + +What evidence would justify revisiting this? +``` + +The body belongs to the Decision record. Cite related Findings, Constraints, +Risks, and Issues through `derives_from` when creating it. diff --git a/.agents/skills/effort-modeling/SKILL.md b/.agents/skills/effort-modeling/SKILL.md new file mode 100644 index 00000000..38a29955 --- /dev/null +++ b/.agents/skills/effort-modeling/SKILL.md @@ -0,0 +1,60 @@ +--- +name: effort-modeling +description: Sharpen a project's vocabulary and planning through one-question-at-a-time grilling, then journal Decisions, Constraints, Findings, Issues, and Risks into the Flatbread Effort Graph. Use when a plan needs durable reasoning instead of ADRs. +disable-model-invocation: true +--- + +# Effort modeling + +Use this discipline while a plan or design is being shaped. Store planning +records in the Effort Graph. Keep project terms in a glossary such as +`CONTEXT.md` or `docs/glossary.md`. + +## Resume before asking + +From the project root, use the `effort-graph` skill's bounded reads: + +1. `flatbread effort list --status active` +2. For the relevant Effort, inspect open Issues and blocking Decisions. +3. Read a record or digest when a prior conclusion affects the question. + +Look up facts in code instead of asking. Put decisions to the user; do not +answer them autonomously. + +## During the grill + +Ask one question at a time. Offer a recommendation, wait for the user's +answer, and resolve prerequisite choices before dependent ones. + +- Challenge a term that conflicts with the glossary. +- Sharpen vague or overloaded terms. +- Use concrete edge cases to test relationships and scope. +- Compare a claim about behavior with the code and surface contradictions. +- Update the relevant project glossary when vocabulary resolves. Keep it free + of implementation details and planning rationale. + +## Journal the right speech act + +Use `flatbread effort write` for the current Effort, following +[the Effort Graph reference](../effort-graph/reference.md): + +- **Finding** — evidence about code, users, or runtime behavior. +- **Issue** — a question, defect, gap, or blocker needing attention. +- **Constraint** — a sticky hard or soft boundary. +- **Risk** — a prospective negative outcome with likelihood and severity. +- **Decision** — a proposed or accepted commitment among alternatives. + +Record a Decision when it is hard to reverse, surprising without context, and +the result of a real trade-off. Create it as proposed while the user is still +deciding; call `AcceptDecision` only after they commit. Always pass +`"rejectSiblings": false` unless deliberately closing every competing proposal. + +Use the long-form body template in [DECISION-BODY.md](./DECISION-BODY.md) when +the rationale would otherwise be lost. The body is the durable explanation; +do not create an ADR alongside it. + +## Finish + +Capture accepted Decisions, unresolved Issues, and material Findings before +ending the session. For an immediate verification read, use the generation +returned by the mutation with `--strict-min-generation`. diff --git a/.agents/skills/grill-with-efforts/SKILL.md b/.agents/skills/grill-with-efforts/SKILL.md new file mode 100644 index 00000000..18d024f4 --- /dev/null +++ b/.agents/skills/grill-with-efforts/SKILL.md @@ -0,0 +1,15 @@ +--- +name: grill-with-efforts +description: Run a relentless one-question-at-a-time planning interview that sharpens vocabulary and journals durable reasoning into the Flatbread Effort Graph. Use when a plan is fuzzy and needs an Effort Graph trail instead of ADRs. +disable-model-invocation: true +--- + +# Grill with efforts + +Run a one-question-at-a-time grilling session using +[effort-modeling](../effort-modeling/SKILL.md). + +Offer a recommended answer for each decision, wait for the user's response, +and resolve dependent choices in order. Explore the codebase for facts, but +leave choices to the user. Do not implement the plan until shared understanding +is confirmed. diff --git a/.flatbread-efforts/constraints/con-graphql-is-a-read-interface-not-the-product--4966nqjsnsmwq1fs.md b/.flatbread-efforts/constraints/con-graphql-is-a-read-interface-not-the-product--4966nqjsnsmwq1fs.md new file mode 100644 index 00000000..55909012 --- /dev/null +++ b/.flatbread-efforts/constraints/con-graphql-is-a-read-interface-not-the-product--4966nqjsnsmwq1fs.md @@ -0,0 +1,9 @@ +--- +id: con-graphql-is-a-read-interface-not-the-product--4966nqjsnsmwq1fs +effort: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: 'GraphQL is a read interface, not the product' +kind: hard +created_at: '2026-07-18T19:42:12.650Z' +--- + +GraphQL may serve the repo-backed graph, but Flatbread identity remains files to model to typed reads. diff --git a/.flatbread-efforts/constraints/con-mutation-enum-stays-deliberately-small--45v1ae3neq26g1rz.md b/.flatbread-efforts/constraints/con-mutation-enum-stays-deliberately-small--45v1ae3neq26g1rz.md new file mode 100644 index 00000000..5ec4931c --- /dev/null +++ b/.flatbread-efforts/constraints/con-mutation-enum-stays-deliberately-small--45v1ae3neq26g1rz.md @@ -0,0 +1,26 @@ +--- +id: con-mutation-enum-stays-deliberately-small--45v1ae3neq26g1rz +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Mutation enum stays deliberately small +kind: hard +created_at: '2026-07-18T19:42:47.759Z' +derives_from: + - dec-use-semantic-mutations-and-a-standalone-writer--2d0m3tkqhad4yyhr +--- + +V1 has exactly thirteen named mutations. Every operation has a Zod schema, +validates against a committed index generation, and owns a defined semantic +transition. + +The surface consists of Effort lifecycle (`CreateEffort`, `SetEffortStatus`); +one creation mutation for each primitive (`WriteIssue`, `WriteFinding`, +`WriteDecision`, `WriteConstraint`, `WriteRisk`); edge retro-linking +(`Supersede`, `Invalidate`); and lifecycle transitions (`ResolveIssue`, +`AcceptDecision`, `MitigateRisk`, `SetRiskState`). + +No generic frontmatter patch, delete/archive operation, standalone +`RejectDecision`, or body-edit mutation is part of v1. Git is the undo story; +Decision sibling rejection is part of accepting an alternative; and bodies +remain ordinary editable markdown while the platform owns frontmatter +semantics. Additive mutations require dogfood evidence; removing or reshaping +one is a breaking migration. diff --git a/.flatbread-efforts/constraints/con-prettier-remains-the-markdown-yaml-writer--wrawg3b0hp779hd8.md b/.flatbread-efforts/constraints/con-prettier-remains-the-markdown-yaml-writer--wrawg3b0hp779hd8.md new file mode 100644 index 00000000..2d1e0e67 --- /dev/null +++ b/.flatbread-efforts/constraints/con-prettier-remains-the-markdown-yaml-writer--wrawg3b0hp779hd8.md @@ -0,0 +1,11 @@ +--- +id: con-prettier-remains-the-markdown-yaml-writer--wrawg3b0hp779hd8 +effort: eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve +title: Prettier remains the Markdown/YAML writer +kind: hard +created_at: '2026-07-18T19:43:45.526Z' +derives_from: + - dec-modernize-reachable-checks-before-replacing-tool--q7xpn1g7rk65xvzj +--- + +A partial Biome migration must not create two writers for repository Markdown, YAML, snapshots, or framework examples. diff --git a/.flatbread-efforts/constraints/con-repo-files-are-the-source-of-truth--pnd7w2bs2zvk7ye9.md b/.flatbread-efforts/constraints/con-repo-files-are-the-source-of-truth--pnd7w2bs2zvk7ye9.md new file mode 100644 index 00000000..6c48b5ba --- /dev/null +++ b/.flatbread-efforts/constraints/con-repo-files-are-the-source-of-truth--pnd7w2bs2zvk7ye9.md @@ -0,0 +1,9 @@ +--- +id: con-repo-files-are-the-source-of-truth--pnd7w2bs2zvk7ye9 +effort: eff-relational-content-foundation--8a8332x4cazgf2k0 +title: Repo files are the source of truth +kind: hard +created_at: '2026-07-18T19:41:45.236Z' +--- + +Flatbread remains a Git-native flat-file content layer owned by the project. diff --git a/.flatbread-efforts/decisions/dec-add-export-cli-and-unified-watch--k9dpxj57yrcv0pbz.md b/.flatbread-efforts/decisions/dec-add-export-cli-and-unified-watch--k9dpxj57yrcv0pbz.md new file mode 100644 index 00000000..bde17b25 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-add-export-cli-and-unified-watch--k9dpxj57yrcv0pbz.md @@ -0,0 +1,15 @@ +--- +id: dec-add-export-cli-and-unified-watch--k9dpxj57yrcv0pbz +effort: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: Add export CLI and unified watch +state: proposed +created_at: '2026-07-18T19:42:27.717Z' +derives_from: + - iss-live-reload-and-export-cli-remain-gaps--xe19gzqf654xeyg4 +--- + +Prioritize `flatbread export json/csv` and the unified `flatbread start --watch` +path. The watcher should classify config, content, and documents; validate and +hot-swap valid generations atomically; refresh codegen; and retain the prior +valid graph when a candidate is invalid. Export commands remain open until the +developer-facing CLI workflow is delivered. diff --git a/.flatbread-efforts/decisions/dec-adopt-a-bounded-status-briefing-fast-path-for-ef--kcw0rw39g3b2ym2h.md b/.flatbread-efforts/decisions/dec-adopt-a-bounded-status-briefing-fast-path-for-ef--kcw0rw39g3b2ym2h.md new file mode 100644 index 00000000..5e66838e --- /dev/null +++ b/.flatbread-efforts/decisions/dec-adopt-a-bounded-status-briefing-fast-path-for-ef--kcw0rw39g3b2ym2h.md @@ -0,0 +1,31 @@ +--- +id: dec-adopt-a-bounded-status-briefing-fast-path-for-ef--kcw0rw39g3b2ym2h +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Adopt a bounded status-briefing fast-path for Effort Graph recall +state: proposed +created_at: '2026-07-19T02:03:45.865Z' +derives_from: + - fnd-bounded-status-briefing-protocol-halves-effort-g--ht1jd1hsssf7mv2y + - iss-skill-scoped-records-filter-example-over-constra--vd0gcnpc9cm6jzjh +--- + +## Context + +Recall-style questions ("what is open / in flight / resume") are a common agent entry point. A controlled 12-run experiment (see the derives_from Finding) showed the current skill lets agents roughly double their tool calls versus a bounded protocol, with no quality gain and complete distribution separation. + +## Decision + +Add an explicit "status briefing / resume" fast-path to the effort-graph skill: (1) `list --status active` and trust the returned digest; (2) for each active Effort, `records --kinds issue,decision` and read status/state from the digest without opening source markdown; (3) `blocking-decisions` only for an Effort that has an open blocker Issue; (4) do not open raw .flatbread-efforts/\*_/_.md unless a digest truncated a body that must be quoted. Also correct the over-constrained `--status open --state proposed` filter example (see the derives_from Issue). + +## Alternatives considered + +- Leave guidance as-is: rejected; the experiment shows a consistent ~52 percent tool-call waste from blocking-decision fan-out and re-querying. +- Build a new aggregate CLI command (e.g. `effort briefing`): deferred; a guidance-only change captured the full effect with no new code surface. Revisit only if guidance proves insufficient. + +## Consequences + +Recall becomes cheaper and lower-latency with unchanged answer quality, and the prescriptive path doubles as a forcing function (Treatment tool-call variance was near zero). Requires editing the canonical skill in packages/effort-graph/skills and re-running `pnpm skills:sync`. + +## Reversal criteria + +Revisit if answer quality regresses on richer recall questions, if the fast-path causes agents to miss records that need source-body detail, or if a measured task type shows no benefit. diff --git a/.flatbread-efforts/decisions/dec-brand-the-agent-memory-surface-as-crumb-graph--fvskcvagx3a7sybe.md b/.flatbread-efforts/decisions/dec-brand-the-agent-memory-surface-as-crumb-graph--fvskcvagx3a7sybe.md new file mode 100644 index 00000000..fbf3950a --- /dev/null +++ b/.flatbread-efforts/decisions/dec-brand-the-agent-memory-surface-as-crumb-graph--fvskcvagx3a7sybe.md @@ -0,0 +1,41 @@ +--- +id: dec-brand-the-agent-memory-surface-as-crumb-graph--fvskcvagx3a7sybe +effort: eff-flatbread-product-branding--zt7b35sa05kyvhdz +title: Brand the agent memory surface as Crumb Graph +state: accepted +created_at: '2026-07-19T03:45:45.157Z' +derives_from: + - fnd-workshop-shortlist-favors-crumb-graph-over-yeast--tv3ydt3wfb3n4y0g +superseded_by: + - dec-brand-the-agent-memory-surface-as-crumb-trail--tngncepdbwjkh9jc +--- + +## Context + +The persistent, queryable agent-memory system is currently marketed and documented as the Effort Graph. That name is accurate for the Effort-anchored reasoning graph but reads as a subsystem label, not a Flatbread-native product surface alongside Proof. A branding workshop shortlisted names that keep longform memory + relational structure while fitting unleavened Flatbread metaphors. + +## Decision + +Adopt **Crumb Graph** as the product/brand name for Flatbread's longform agent memory (journaled Efforts, Issues, Findings, Decisions, Constraints, Risks with bounded recall). + +Keep **Effort Graph** as an allowed technical descriptor and disambiguation alias for the same system — useful when emphasizing Effort-scoped primitives, the graph of reasoning records, or continuity with existing dogfood language — not as a competing product name. + +Code, package, CLI, path, and skill renames are **intentionally deferred**. This Decision commits branding and messaging only; implementation tracks as follow-up work. + +## Alternatives considered + +- **Leaven / Sourdough / Starter:** Strong living-culture metaphor and peer to Proof, rejected because Flatbread is unleavened — yeast metaphors fight the brand. +- **Pantry / Bake Log / Flat Memory:** On-brand storage or flat-files framing, but weaker at signaling trail + relational graph. +- **Keep Effort Graph only:** Clear and already shipped in dogfood, but underplays brandability and product-surface identity next to Proof. +- **Reasoning Graph / Memory Graph:** Accurate and sober, too generic and not Flatbread-native. + +## Consequences + +- New docs, skills copy, and product messaging should prefer Crumb Graph; Effort Graph may appear in parentheses or glossary-style disambiguation. +- The Effort primitive name stays; branding rename does not require renaming the Effort collection. +- Avoid yeast metaphors in Crumb Graph marketing. +- Package `@flatbread/effort-graph`, `flatbread effort`, `.flatbread-efforts/`, and related identifiers remain until a separate implementation Decision/Issue lands. + +## Reversal criteria + +Revisit if Crumb Graph confuses users (crumb ≈ UI breadcrumbs or leftover scraps), collides with another Flatbread surface, or fails external comprehension tests versus Effort Graph. Also revisit if a stronger unleavened, Proof-peer name emerges from broader launch naming. diff --git a/.flatbread-efforts/decisions/dec-brand-the-agent-memory-surface-as-crumb-trail--tngncepdbwjkh9jc.md b/.flatbread-efforts/decisions/dec-brand-the-agent-memory-surface-as-crumb-trail--tngncepdbwjkh9jc.md new file mode 100644 index 00000000..551729c5 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-brand-the-agent-memory-surface-as-crumb-trail--tngncepdbwjkh9jc.md @@ -0,0 +1,46 @@ +--- +id: dec-brand-the-agent-memory-surface-as-crumb-trail--tngncepdbwjkh9jc +effort: eff-flatbread-product-branding--zt7b35sa05kyvhdz +title: Brand the agent memory surface as Crumb Trail +state: accepted +created_at: '2026-07-19T03:54:22.436Z' +derives_from: + - fnd-postable-brand-names-favor-trail-over-graph--zzpgaw0kvqtaqmxt + - fnd-workshop-shortlist-favors-crumb-graph-over-yeast--tv3ydt3wfb3n4y0g +supersedes: + - dec-brand-the-agent-memory-surface-as-crumb-graph--fvskcvagx3a7sybe +--- + +Supersedes Crumb Graph as the product name. Same system and deferred implementation; product brand is now Crumb Trail, with Crumb Graph retained as the datamodel explainer. + +## Context + +We accepted Crumb Graph as the product/brand name with Effort Graph as a technical descriptor. A follow-up pass weighed postable brandability: names that land online are metaphor-first and easy to say. That pressure favors Trail for the product surface while Graph still names the relational memory model accurately. + +## Decision + +Adopt **Crumb Trail** as the product/brand name for Flatbread's longform agent memory. + +Use **Crumb Graph** as the explainer for the datamodel (Effort-scoped relational records, edges, bounded digests) — glossary/docs language, not a competing product name. + +Keep **Effort Graph** as an allowed technical descriptor and disambiguation alias for continuity with dogfood language and Effort-anchored primitives. + +Code, package, CLI, path, and skill renames remain **intentionally deferred**. This Decision commits branding and messaging only. + +## Alternatives considered + +- **Crumb Graph as product name (prior Decision):** Strong datamodel honesty; weaker as a postable brand next to metaphor-first agent tools. Retained as explainer instead. +- **Crumb Trail only, drop Graph language:** Maximizes brand simplicity; loses a clear noun for the relational structure. Rejected — keep Graph as explainer. +- **Effort Graph only:** Already shipped in dogfood; underplays brandability. +- **Yeast metaphors (Leaven / Starter):** Still rejected — Flatbread is unleavened. + +## Consequences + +- Product messaging, launch copy, and skill intros prefer Crumb Trail. +- Architecture/glossary copy may say Crumb Graph when explaining the relational model; Effort Graph remains valid disambiguation. +- The Effort primitive name stays. +- Package/CLI/path identifiers stay until a separate implementation plan lands. + +## Reversal criteria + +Revisit if Crumb Trail collapses into UI-breadcrumb confusion, if Graph-as-explainer creates two-name fatigue, or if external comprehension tests prefer a single public noun. diff --git a/.flatbread-efforts/decisions/dec-bridge-committed-generations-into-live-reads--mc3728t4w1kyqcqq.md b/.flatbread-efforts/decisions/dec-bridge-committed-generations-into-live-reads--mc3728t4w1kyqcqq.md new file mode 100644 index 00000000..69bd7c6c --- /dev/null +++ b/.flatbread-efforts/decisions/dec-bridge-committed-generations-into-live-reads--mc3728t4w1kyqcqq.md @@ -0,0 +1,37 @@ +--- +id: dec-bridge-committed-generations-into-live-reads--mc3728t4w1kyqcqq +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Bridge committed generations into live reads +state: accepted +created_at: '2026-07-18T19:43:15.357Z' +derives_from: + - dec-canonical-forward-edges-and-journaled-save-or-un--sv9x93svkfz4a98r +--- + +## Context + +The writer's journal generation is durable per graph root, while live schema +generations are process-local and also advance for unrelated content. Treating +them as the same counter produces false strict-read guarantees after restart +or ordinary reloads. + +## Decision + +Use `CommittedGenerationPublisher` as the publish gate. It maps changed paths, +awaits the live reloader's accepted schema candidate, and privately associates +durable journal token J with live generation L. Publish J only after both the +journal transaction and live schema commit succeed; strict readers require the +durable publication and live commit, with a bounded timeout. + +A disk-backed, journal-aware `ReindexBarrier` defers paths named by +uncommitted journal intents, releases after commit or rollback, fails closed +for malformed intent, and bounds its wait so an orphaned transaction cannot +stall the reindex queue. + +## Consequences + +A same-process mutation token supports strict read-your-writes. An +out-of-process writer uses the no-op publisher, so a server watcher sees its +files eventually rather than claiming strictness. The composition root detects +the complete Effort Graph content shape, attaches the bridge, performs +non-fatal recovery before listening, and keeps journal mechanics out of core. diff --git a/.flatbread-efforts/decisions/dec-canonical-forward-edges-and-journaled-save-or-un--sv9x93svkfz4a98r.md b/.flatbread-efforts/decisions/dec-canonical-forward-edges-and-journaled-save-or-un--sv9x93svkfz4a98r.md new file mode 100644 index 00000000..644bd956 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-canonical-forward-edges-and-journaled-save-or-un--sv9x93svkfz4a98r.md @@ -0,0 +1,48 @@ +--- +id: dec-canonical-forward-edges-and-journaled-save-or-un--sv9x93svkfz4a98r +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Canonical forward edges and journaled save-or-undo +state: accepted +created_at: '2026-07-18T19:42:42.756Z' +derives_from: + - dec-use-semantic-mutations-and-a-standalone-writer--2d0m3tkqhad4yyhr +--- + +## Context + +Records need to answer “am I current?” from a single file, but treating both +directions of `supersedes` and `invalidates` as authoritative creates needless +two-file conflicts. Some lifecycle operations are still irreducibly +multi-record, such as accepting one Decision and rejecting competing proposed +siblings. + +## Decision + +Forward `supersedes` and `invalidates` edges are authoritative. Their reverse +edges are derived, materialized projections: the writer writes both sides, +and the reindexer repairs drift after hand edits, merge damage, or recovery. + +Use a writer-level save-or-undo journal as the correctness boundary. It records +intent and before-images durably, writes each target through a same-directory +temporary file and rename, marks the transaction committed, then runs one +journal-aware reindex batch. Publish a generation only after projection repair +and the schema swap succeed. Serialize concurrent writers with a per-graph +lock. Git history is opt-in and follows, never determines, semantic commit. + +## Alternatives considered + +- **Bidirectional authoritative edges:** rejected because reverse-only edits + cannot establish intent and make every edge update an authoritative + two-file transaction. +- **Git commit as the transaction boundary:** rejected because it touches + contributor history, complicates rebases, and cannot be correctness + infrastructure. Deliberate session checkpoints remain possible. + +## Consequences + +Raw disk readers may briefly see an in-progress rename and materialized +projections may be stale until repair; indexed reads never observe a partial +committed mutation. Merge repair reconciles forward edges then regenerates +reverse projections. Reversal requires evidence that raw-file readers need +atomic cross-file visibility or that per-mutation commits remain low-friction +in real concurrent workflows. diff --git a/.flatbread-efforts/decisions/dec-defer-shared-flatbread-write-extraction--xxa1ge2chw6x3xx0.md b/.flatbread-efforts/decisions/dec-defer-shared-flatbread-write-extraction--xxa1ge2chw6x3xx0.md new file mode 100644 index 00000000..f1c9e3cd --- /dev/null +++ b/.flatbread-efforts/decisions/dec-defer-shared-flatbread-write-extraction--xxa1ge2chw6x3xx0.md @@ -0,0 +1,39 @@ +--- +id: dec-defer-shared-flatbread-write-extraction--xxa1ge2chw6x3xx0 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Defer shared Flatbread write extraction +state: accepted +created_at: '2026-07-18T19:43:20.362Z' +derives_from: + - dec-bridge-committed-generations-into-live-reads--mc3728t4w1kyqcqq + - dec-use-semantic-mutations-and-a-standalone-writer--2d0m3tkqhad4yyhr +--- + +## Context + +Flatbread remains valuable as a read-only query layer for static content. +Effort Graph already proves a narrowly scoped semantic write path, but +generalizing it before another writable system needs it would speculate on +sources, lifecycle, permissions, and GraphQL mutation behavior. + +## Decision + +Finish and dogfood the Effort Graph writer as its own system. Do not add +general-purpose GraphQL create/update/delete operations or redesign Flatbread +around writes yet. Treat its writer, recovery, and committed-generation bridge +as the first working example of a future shared save-and-refresh layer. + +Begin extraction before copying this machinery when Flatbread needs a second +writable source, a CMS needs drafts/revisions/conflicts/media/permissions, +another feature would duplicate writer recovery or live refresh, or ordinary +collections need GraphQL mutations. Sources remain the persistence authority; +transformers must explain how to save writable records; shared writes must +commit multi-record changes atomically and publish only validated graphs. + +## Consequences + +Effort Graph commands remain the supported write API and GraphQL stays +read-only for ordinary collections. Any future Effort Graph GraphQL surface +delegates to the existing writer. The current read-only static use case stays +simple, while the writer's journal and generation tests become behavioral +contracts for a later extraction. diff --git a/.flatbread-efforts/decisions/dec-distribute-the-effort-graph-as-a-versioned-agent--8as4sybr34zcqyqh.md b/.flatbread-efforts/decisions/dec-distribute-the-effort-graph-as-a-versioned-agent--8as4sybr34zcqyqh.md new file mode 100644 index 00000000..19bee40b --- /dev/null +++ b/.flatbread-efforts/decisions/dec-distribute-the-effort-graph-as-a-versioned-agent--8as4sybr34zcqyqh.md @@ -0,0 +1,34 @@ +--- +id: dec-distribute-the-effort-graph-as-a-versioned-agent--8as4sybr34zcqyqh +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Distribute the Effort Graph as a versioned Agent Skill +state: accepted +created_at: '2026-07-18T23:55:43.863Z' +--- + +## Context + +The original local skill taught monorepo-only commands and could not be +released as a consumer contract. A floating skill branch can also drift from a +separately versioned runtime. + +## Decision + +`@flatbread/effort-graph` owns canonical skill assets under +`packages/effort-graph/skills/`; `.agents/skills/` is a generated, +byte-identical local projection for dogfooding and discovery. Deterministic +sync, watch, CI, and package-payload verification detect projection or +publication drift. + +Consumers install a canonical skill from an immutable `v` +tag. Its checked-in release manifest binds the tag to exact Flatbread and +Effort Graph versions. The first-activation flow supports npm, pnpm, Yarn, +and Bun; bootstrap reports requirements and verifies reviewed configuration +edits rather than rewriting a project silently. + +## Consequences + +The monorepo uses the same skill payload that consumers receive. Releases tie +runtime, manifest, package payload, and git tag to one committed identity. +Session recall starts with bounded active-Effort discovery, followed by +scoped reads; it does not depend on GraphQL being the product surface. diff --git a/.flatbread-efforts/decisions/dec-effort-get-digests-always-include-the-full-recor--xs7rnbha3zx8qdj9.md b/.flatbread-efforts/decisions/dec-effort-get-digests-always-include-the-full-recor--xs7rnbha3zx8qdj9.md new file mode 100644 index 00000000..c57afbf4 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-effort-get-digests-always-include-the-full-recor--xs7rnbha3zx8qdj9.md @@ -0,0 +1,37 @@ +--- +id: dec-effort-get-digests-always-include-the-full-recor--xs7rnbha3zx8qdj9 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: effort get digests always include the full record body +state: accepted +created_at: '2026-07-19T03:44:51.061Z' +derives_from: + - dec-route-agent-reads-through-the-flatbread-query-en--476qb9qk878yfg62 + - fnd-getrecord-digests-call-the-same-excerpt-as-brows--q5rbnhpxc697x2dv + - iss-agents-cannot-obtain-full-record-bodies-through--ekfpcg6hrkwgy287 +--- + +## Context + +Browse digests must stay excerpted for the bounded status-briefing budget. Agents still need a reliable zoom-in path for complete Decision (and other) bodies without opening `.flatbread-efforts/**/*.md` for normal recall. The accepted read-routing Decision already promised full bodies via single-record lookup; digests did not deliver it. + +## Decision + +`flatbread effort get` digests always render the full primary record body via the existing ReadEnvelope + cached digest at `artifact_path`. Browse reads (`list`, `records`, `relations`, `blocking-decisions`) remain excerpted at 600 chars / 12 lines. Agent zoom-in is: Shell JSON → Read/grep the get digest. Digest-level caps (64 KiB) still apply; an oversized body fails closed with a byte-cap banner and `complete: false` — never a fake-full 600/12 excerpt. + +## Alternatives considered + +- **Opt-in `--full` flag:** Rejected — full body is the point of single-record lookup; an extra flag invites agents to keep missing context. +- **Inline full body in the JSON envelope:** Rejected — keep unrestricted content on the digest artifact path, not the machine surface. +- **Separate `effort body` / tmp-file command:** Rejected — unnecessary surface; get + artifact_path already exists. +- **Make list/records full-body:** Rejected — would blow the status-briefing token budget. + +## Consequences + +- Normal-sized `get` digests contain complete source bodies; agents should not open source markdown for zoom-in. +- Skill/reference must teach: when browse digests show `[…truncated]` and the body is needed, run `effort get` then Read that digest. +- Single-record lookup is now the real full-body path promised by dec-route-agent-reads-through-the-flatbread-query-en--476qb9qk878yfg62. +- Oversized bodies that exceed the digest byte cap surface a visible miss (source path in banner), not silent excerpting. + +## Reversal criteria + +Revisit if agents routinely need bodies larger than the digest byte cap, or if full-body get digests prove too large for the Read/grep workflow and a different artifact strategy is required. diff --git a/.flatbread-efforts/decisions/dec-iterate-typed-selection-and-nullability--0khw2zkc51hy230t.md b/.flatbread-efforts/decisions/dec-iterate-typed-selection-and-nullability--0khw2zkc51hy230t.md new file mode 100644 index 00000000..8bb7db0c --- /dev/null +++ b/.flatbread-efforts/decisions/dec-iterate-typed-selection-and-nullability--0khw2zkc51hy230t.md @@ -0,0 +1,15 @@ +--- +id: dec-iterate-typed-selection-and-nullability--0khw2zkc51hy230t +effort: eff-relational-content-foundation--8a8332x4cazgf2k0 +title: Iterate typed selection and nullability +state: accepted +created_at: '2026-07-18T19:41:57.959Z' +derives_from: + - fnd-generated-read-api-is-useful-but-not-fully-type--vmr23k0t6qx3hk1k +--- + +Generated helpers prove typed consumption is plausible, but selection typing +and nullability need hardening before stable type-safety positioning. Prioritize +a typed selection/projection API and sharper relation helpers; keep GraphQL as +a useful schema and introspection interface rather than Flatbread's product +definition. diff --git a/.flatbread-efforts/decisions/dec-keep-and-extract-validation-api--e96yngmrjqcjdxcn.md b/.flatbread-efforts/decisions/dec-keep-and-extract-validation-api--e96yngmrjqcjdxcn.md new file mode 100644 index 00000000..3ff7e786 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-keep-and-extract-validation-api--e96yngmrjqcjdxcn.md @@ -0,0 +1,11 @@ +--- +id: dec-keep-and-extract-validation-api--e96yngmrjqcjdxcn +effort: eff-relational-content-foundation--8a8332x4cazgf2k0 +title: Keep and extract validation API +state: accepted +created_at: '2026-07-18T19:42:07.632Z' +derives_from: + - fnd-reference-integrity-is-roadmap-critical--2ss712xpmsfh77xf +--- + +Keep ID, ref, and cardinality validation as a first-class foundation and extract a reusable validation API. diff --git a/.flatbread-efforts/decisions/dec-keep-dags-acyclic-model-refinement-as-bounded-lo--9ptqq1ck3sax00fq.md b/.flatbread-efforts/decisions/dec-keep-dags-acyclic-model-refinement-as-bounded-lo--9ptqq1ck3sax00fq.md new file mode 100644 index 00000000..3d63f674 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-keep-dags-acyclic-model-refinement-as-bounded-lo--9ptqq1ck3sax00fq.md @@ -0,0 +1,32 @@ +--- +id: dec-keep-dags-acyclic-model-refinement-as-bounded-lo--9ptqq1ck3sax00fq +effort: eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve +title: Keep DAGs acyclic; model refinement as bounded loops +state: accepted +created_at: '2026-07-18T19:43:25.417Z' +--- + +## Context + +Dependency edges express static causality, readiness, skip behavior, and +parallelism. Allowing cycles would make those properties ambiguous, while +review-and-refine workflows still need controlled repetition. + +## Decision + +Keep `depends_on` acyclic. Model research → critique → refine as declared, +bounded convergence loops in the DAG. A loop names its convergence task, +iteration ceiling, and either the ancestor cone or an explicit, +dependency-closed re-execution subset. + +Multiple loops execute in declaration order, must have distinct convergence +tasks and disjoint re-execution sets, and cannot be combined with the legacy +`--converge-on` CLI flag. The CLI flag remains supported by synthesizing one +loop for ad-hoc use. + +## Consequences + +The existing blockers/high-severity-findings parser stays the v1 stop +predicate. Nested loops, alternate predicates, and cross-loop coordination +remain out of scope until concrete runtime demand exists. Existing DAG JSON +without loops remains valid. diff --git a/.flatbread-efforts/decisions/dec-keep-effort-graph-secondary-with-a-primary-wedge--estattvqnhffm2dc.md b/.flatbread-efforts/decisions/dec-keep-effort-graph-secondary-with-a-primary-wedge--estattvqnhffm2dc.md new file mode 100644 index 00000000..d5d165c7 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-keep-effort-graph-secondary-with-a-primary-wedge--estattvqnhffm2dc.md @@ -0,0 +1,15 @@ +--- +id: dec-keep-effort-graph-secondary-with-a-primary-wedge--estattvqnhffm2dc +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Keep Effort Graph secondary with a primary-wedge gate +state: accepted +created_at: '2026-07-18T19:43:02.802Z' +derives_from: + - fnd-filtered-retrieval-is-94-percent-smaller--qpvz4vch5hygye20 +--- + +Keep the graph as a strong candidate secondary vertical rather than displacing +Flatbread's relational-content wedge. Promote it only after real multi-session +retrieval, token-based benchmark, read-surface parity, and external +rediscovery evidence. Continue to use the graph to dogfood planning rather +than treating fixture results as product validation. diff --git a/.flatbread-efforts/decisions/dec-keep-relation-first-content-layer--rc3pc0vczp6qhp3t.md b/.flatbread-efforts/decisions/dec-keep-relation-first-content-layer--rc3pc0vczp6qhp3t.md new file mode 100644 index 00000000..3e802f58 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-keep-relation-first-content-layer--rc3pc0vczp6qhp3t.md @@ -0,0 +1,15 @@ +--- +id: dec-keep-relation-first-content-layer--rc3pc0vczp6qhp3t +effort: eff-relational-content-foundation--8a8332x4cazgf2k0 +title: Keep relation-first content layer +state: accepted +created_at: '2026-07-18T19:41:50.089Z' +derives_from: + - con-repo-files-are-the-source-of-truth--pnd7w2bs2zvk7ye9 + - fnd-relation-first-starter-reaches-a-typed-read--8anm4wxbjm9f348z +--- + +Keep the relation-first content layer: the canonical starter reaches a typed +posts/authors/tags read quickly, and files → model → typed reads remains the +first-success path. Continue improving example polish and validation coverage; +do not pivot toward a hosted CMS or a general database replacement. diff --git a/.flatbread-efforts/decisions/dec-make-the-effort-graph-the-planning-authority--gq4xegmc0hhbqjf7.md b/.flatbread-efforts/decisions/dec-make-the-effort-graph-the-planning-authority--gq4xegmc0hhbqjf7.md new file mode 100644 index 00000000..d5a6b453 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-make-the-effort-graph-the-planning-authority--gq4xegmc0hhbqjf7.md @@ -0,0 +1,18 @@ +--- +id: dec-make-the-effort-graph-the-planning-authority--gq4xegmc0hhbqjf7 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Make the Effort Graph the planning authority +state: accepted +created_at: '2026-07-19T00:41:19.716Z' +--- + +## Decision + +For Flatbread product and engineering work, Effort Graph records are the sole durable authority for decisions, constraints, findings, risks, and open issues. Keep CONTEXT.md only as a glossary. Decision bodies carry the context, alternatives, consequences, and reversal criteria previously preserved in ADRs. + +## Consequences + +Migrate the former ADR rationale into the corresponding graph records, retire +the duplicate documents, and replace ADR-oriented agent guidance with Effort +Graph journaling. Working notes may remain when useful, but must cite graph +record IDs and cannot establish a competing decision authority. diff --git a/.flatbread-efforts/decisions/dec-modernize-reachable-checks-before-replacing-tool--q7xpn1g7rk65xvzj.md b/.flatbread-efforts/decisions/dec-modernize-reachable-checks-before-replacing-tool--q7xpn1g7rk65xvzj.md new file mode 100644 index 00000000..156c48d0 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-modernize-reachable-checks-before-replacing-tool--q7xpn1g7rk65xvzj.md @@ -0,0 +1,32 @@ +--- +id: dec-modernize-reachable-checks-before-replacing-tool--q7xpn1g7rk65xvzj +effort: eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve +title: Modernize reachable checks before replacing tooling +state: accepted +created_at: '2026-07-18T19:43:40.522Z' +--- + +## Context + +Vitest suites were not reachable from the root test command, CI installs were +mutable, lint and typecheck coverage was incomplete, and the SvelteKit +integration job exercised the wrong example. Replacing the toolchain before +making existing checks reachable would add overlapping authority without +improving confidence. + +## Decision + +Make the existing AVA, Vitest, and Prettier stack deterministic and reachable +first: root tests build then run both test runners; CI uses frozen installs, +lint, typecheck, build, and tests; and SvelteKit integration builds SvelteKit. +Keep TypeScript modernization scoped to packages that need it. + +Defer Biome and Oxc. Prettier remains the whole-repository Markdown/YAML +writer; framework ESLint boundaries stay intact. Revisit a Biome or Oxlint +pilot only after root lint ownership is explicit. + +## Consequences + +Follow-ups cover dormant root ESLint, AVA/Vitest convergence, monorepo +typecheck architecture, coverage thresholds, deprecated runtime dependencies, +unused workspace dependencies, and the remaining SvelteKit typing issue. diff --git a/.flatbread-efforts/decisions/dec-route-agent-reads-through-the-flatbread-query-en--476qb9qk878yfg62.md b/.flatbread-efforts/decisions/dec-route-agent-reads-through-the-flatbread-query-en--476qb9qk878yfg62.md new file mode 100644 index 00000000..2a0f400a --- /dev/null +++ b/.flatbread-efforts/decisions/dec-route-agent-reads-through-the-flatbread-query-en--476qb9qk878yfg62.md @@ -0,0 +1,57 @@ +--- +id: dec-route-agent-reads-through-the-flatbread-query-en--476qb9qk878yfg62 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Route agent reads through the Flatbread query engine +state: accepted +created_at: '2026-07-18T20:42:15.423Z' +derives_from: + - fnd-dogfooding-the-read-cli-surfaced-three-real-defe--g21hg77x72hdty56 + - iss-define-blocking-decision-semantics--3bj9ph7ppab2c7g2 +--- + +## Context + +Agents need compact, evidence-backed recall without putting a complete +reasoning graph in context. A second bespoke record-filtering implementation +would drift from Flatbread's own ID, relation, and filter semantics. + +## Decision + +Agent reads return a bounded envelope: deterministic summary, digest path and +hash, served generation, consistency echo, paging information, and executable +hints. The rendered digest is the evidence surface. It is atomically cached by +generation and canonical query hash, contains selected frontmatter, relations, +and a 600-character/12-line body excerpt, and never inlines unrestricted graph +content. + +The product surface is five named, effort-scoped reads: record lookup, effort +records, one-hop relations, blocking decisions, and active-effort discovery. +The closed filters compile to Flatbread filter objects and execute through +`FlatbreadProvider` against the generated schema. The digest renderer, cache, +and envelope remain engine-agnostic; CLI and future MCP are thin transports. + +Reads are eventual by default. Strict reads wait for the caller's durable +generation token or fail explicitly; they never silently degrade. + +## Alternatives considered + +- **A general agent query language:** rejected because canonical cache keys, + cap enforcement, and a teachable API require a closed vocabulary. +- **Bespoke filesystem snapshot filtering:** rejected because it recreates + Flatbread semantics and accumulates a second query-engine debt. +- **Full graph responses:** rejected because they overflow context and expose + unrelated reasoning. + +## Consequences + +Digests cap primary records, relation expansion, displayed edges, and bytes; +callers narrow or page incomplete responses. Full bodies remain available from +the source record or single-record lookup. The exact read contract lives in +this Decision and its supporting findings, not in a parallel ADR. + +`get`, effort-scoped records, relations, blocking decisions, and active-Effort +discovery are the complete named v1 reads. Their digests cap at 25 primary +records, one hop, 50 edges, and 64 KiB; summaries cap at 160 tokens and +excerpts at 600 characters or 12 lines. Supersession reads resolve to the head +and render prior records as deterministic checkpoints; semantic rollups belong +in the superseding record's body, never in an LLM read path. diff --git a/.flatbread-efforts/decisions/dec-separate-execution-and-display-planes--spm2ckxvdsch6h9m.md b/.flatbread-efforts/decisions/dec-separate-execution-and-display-planes--spm2ckxvdsch6h9m.md new file mode 100644 index 00000000..4b133604 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-separate-execution-and-display-planes--spm2ckxvdsch6h9m.md @@ -0,0 +1,37 @@ +--- +id: dec-separate-execution-and-display-planes--spm2ckxvdsch6h9m +effort: eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve +title: Separate execution and display planes +state: proposed +created_at: '2026-07-18T19:43:33.002Z' +derives_from: + - rsk-full-transcripts-can-overflow-prompts-and-leak-s--3p7ybk07jqe5w593 +--- + +## Context + +Proof's bounded stream buffer served as the canvas view, upstream prompt +context, convergence input, findings sidecar source, and persisted resume +state. It dropped the leading stream content, so execution decisions could +silently lose evidence while removing the cap would make canvases, prompts, +state files, and accidental sharing unsafe. + +## Decision + +Separate the execution plane from the display plane for `kind: "task"` output. +An execution-authoritative transcript drives upstream context, convergence +finding extraction and feedback, findings sidecars, artifacts, and resumed +runs. Canvas and persisted display state use explicitly bounded views with +visible truncation banners. Prompt policy remains named and bounded rather +than inheriting a display cap. + +Artifact-backed restarts reconstruct transcripts from the same pinned +`--full-output-dir`; legacy bounded `resultText` is only a compatibility +fallback. Oracle stdout/stderr and advanced prompt-budget policy remain +separate follow-up work. + +## Consequences + +The runner must test transcript-to-prompt, convergence, sidecar, artifact, and +resume paths independently. Canvas size and privacy remain bounded; execution +logic no longer treats a UX cap as its source of truth. diff --git a/.flatbread-efforts/decisions/dec-specify-blockingdecisions-over-effort-scoped-edg--pxwaenyra0aa365j.md b/.flatbread-efforts/decisions/dec-specify-blockingdecisions-over-effort-scoped-edg--pxwaenyra0aa365j.md new file mode 100644 index 00000000..b17e8c1a --- /dev/null +++ b/.flatbread-efforts/decisions/dec-specify-blockingdecisions-over-effort-scoped-edg--pxwaenyra0aa365j.md @@ -0,0 +1,30 @@ +--- +id: dec-specify-blockingdecisions-over-effort-scoped-edg--pxwaenyra0aa365j +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Specify blockingDecisions over effort-scoped edges +state: accepted +created_at: '2026-07-18T19:42:57.783Z' +derives_from: + - iss-define-blocking-decision-semantics--3bj9ph7ppab2c7g2 +--- + +## Decision + +A Decision is blocking for effort E when it belongs to E, remains `proposed`, +and its `derives_from` contains an Issue that also belongs to E with +`kind: blocker` and `status: open`. The relation is direct, results are +deduplicated and ordered by `created_at` then ID, and declared lifecycle state +is authoritative. + +Accepted, rejected, superseded, or deprecated Decisions are excluded. So are +resolved, deferred, or non-blocker Issues, transitive dependencies, +cross-effort references, and prose mentions. The query deliberately answers +which proposed Decisions are gated by blockers; an open blocker with no +proposed response is discovered through the ordinary effort-records query. + +## Consequences + +The predicate is a frozen v1 contract. Widening it with transitive edges or +new edge semantics requires an explicit successor Decision with evidence from +dogfooding. Bounded one-hop digest expansion includes the matching blockers so +agents can inspect the gate without loading an entire Effort. diff --git a/.flatbread-efforts/decisions/dec-store-graph-artifacts-in-repo-by-default--xxpgwm9sv25j07z7.md b/.flatbread-efforts/decisions/dec-store-graph-artifacts-in-repo-by-default--xxpgwm9sv25j07z7.md new file mode 100644 index 00000000..6f600760 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-store-graph-artifacts-in-repo-by-default--xxpgwm9sv25j07z7.md @@ -0,0 +1,41 @@ +--- +id: dec-store-graph-artifacts-in-repo-by-default--xxpgwm9sv25j07z7 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Store graph artifacts in-repo by default +state: accepted +created_at: '2026-07-18T19:42:32.811Z' +--- + +## Context + +Effort Graph artifacts must branch with the code they explain, yet teams also +need an escape hatch when exploration spans repositories or abandoned branches +are common. + +## Decision + +Use branch-coupled, in-repository storage under `.flatbread-efforts/` by +default. Keep the schema and write API identical for all storage modes so a +team can use a sibling repository by changing the configured `path`, without +application code changes. + +Defer an automated promote-on-close workflow. Until dogfooding proves it +necessary, teams can cherry-pick artifacts and transition their lifecycle +state when promoting an exploration. + +## Alternatives considered + +- **In-repo with promote-on-close tooling:** adds brittle semantics around + squash merges, rebases, deleted branches, and what it means for an + exploration to close. +- **Sibling repository or submodule:** preserves reasoning independently of a + branch and remains the supported alternative, but should not be the default. +- **Cross-branch record refs:** rejected because indexing another branch + requires mutating the worktree or maintaining per-branch indexes, obscures + staleness, and weakens reviewability. + +## Consequences + +Reasoning on an unmerged branch is intentionally lost unless promoted. The +schema does not encode branch or git-ref fields: branching is git behavior, +while the graph records the epistemic lifecycle and causal edges. diff --git a/.flatbread-efforts/decisions/dec-use-isolated-schema-factories-without-a-cache--8bn823dg29yfvbzv.md b/.flatbread-efforts/decisions/dec-use-isolated-schema-factories-without-a-cache--8bn823dg29yfvbzv.md new file mode 100644 index 00000000..a804930a --- /dev/null +++ b/.flatbread-efforts/decisions/dec-use-isolated-schema-factories-without-a-cache--8bn823dg29yfvbzv.md @@ -0,0 +1,33 @@ +--- +id: dec-use-isolated-schema-factories-without-a-cache--8bn823dg29yfvbzv +effort: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: Use isolated schema factories without a cache +state: accepted +created_at: '2026-07-18T19:42:17.673Z' +derives_from: + - fnd-schemas-need-per-build-isolation--vdjfmtqb0dbjfcqk +--- + +## Context + +A configuration-keyed schema cache reused resolver closures captured from the +first content snapshot, producing stale reads after content changed. It also +required validation-before-cache ordering and forced the watch demo to carry a +cache-busting field. + +## Decision + +Create a GraphQL composer and schema for every build. Each returned schema owns +its resolver closures and content snapshot. Keep per-build composers isolated, +and retain the owned JSON-to-type parser because the upstream JSON composer +leaks nested types onto a global composer even when passed an instance. + +Do not add a content-keyed cache without profiling evidence, an explicit seam, +and correctness tests for changing content. + +## Consequences + +Schema construction is simpler and correct under rebuilds. AVA remains +serialized until complete-suite parallel safety is demonstrated. A measured +cache may return only after it shows meaningful leverage without recreating +snapshot-lifetime coupling. diff --git a/.flatbread-efforts/decisions/dec-use-prefixed-random-ids-with-reviewable-slugs--xdg764ejat1kacd2.md b/.flatbread-efforts/decisions/dec-use-prefixed-random-ids-with-reviewable-slugs--xdg764ejat1kacd2.md new file mode 100644 index 00000000..46c0eed5 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-use-prefixed-random-ids-with-reviewable-slugs--xdg764ejat1kacd2.md @@ -0,0 +1,43 @@ +--- +id: dec-use-prefixed-random-ids-with-reviewable-slugs--xdg764ejat1kacd2 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Use prefixed random IDs with reviewable slugs +state: accepted +created_at: '2026-07-18T19:42:50.359Z' +derives_from: + - dec-canonical-forward-edges-and-journaled-save-or-un--sv9x93svkfz4a98r +--- + +## Context + +Record references must survive title changes, renames, moves, and concurrent +creation on branches that later merge. Filenames need to remain reviewable but +cannot safely define identity. + +## Decision + +Use `---<16 lowercase Crockford-base32 characters>`. +The permanent prefixes are `eff`, `iss`, `fnd`, `dec`, `con`, and `rsk`; slugs +are capped at 48 characters. The 80-bit random suffix makes uncoordinated +creation effectively collision-free, while prefix and slug keep references +readable. `created_at`, not the identifier, expresses recency. + +`id` is required frontmatter and is the sole identity. The writer normally +names a file after the ID, but filename/path mismatches are advisory. Efforts +may have an editable title and human-facing slug alias without changing their +ID. Collection-first directories organize files; `effort` refs define +membership. + +## Alternatives considered + +ULIDs and UUIDv7s were rejected as noisier identifiers. Pure human slugs were +rejected because similarly named efforts created concurrently become merge +conflicts. Typed wrapper refs such as `decision:dec-…` duplicate the prefix +and add adapter work. + +## Consequences + +The prefixes are permanently reserved and mutation-time validation enforces ID +shape and uniqueness. A future union-reference migration can dispatch on the +prefix without rewriting stored IDs, although Flatbread's collection-ref model +will still need its own schema migration. diff --git a/.flatbread-efforts/decisions/dec-use-profiles-not-separate-schemas--27p0wakkyxj0kfc1.md b/.flatbread-efforts/decisions/dec-use-profiles-not-separate-schemas--27p0wakkyxj0kfc1.md new file mode 100644 index 00000000..8c639579 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-use-profiles-not-separate-schemas--27p0wakkyxj0kfc1.md @@ -0,0 +1,11 @@ +--- +id: dec-use-profiles-not-separate-schemas--27p0wakkyxj0kfc1 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: 'Use profiles, not separate schemas' +state: accepted +created_at: '2026-07-18T19:43:10.329Z' +derives_from: + - fnd-one-canonical-schema-needs-mapping-profiles--k2yvq5e9rcpsm1s1 +--- + +Ship layout profiles and invest in ref-integrity diagnostics; defer automatic branch merge semantics until a human policy exists. diff --git a/.flatbread-efforts/decisions/dec-use-semantic-mutations-and-a-standalone-writer--2d0m3tkqhad4yyhr.md b/.flatbread-efforts/decisions/dec-use-semantic-mutations-and-a-standalone-writer--2d0m3tkqhad4yyhr.md new file mode 100644 index 00000000..eb4fd917 --- /dev/null +++ b/.flatbread-efforts/decisions/dec-use-semantic-mutations-and-a-standalone-writer--2d0m3tkqhad4yyhr.md @@ -0,0 +1,45 @@ +--- +id: dec-use-semantic-mutations-and-a-standalone-writer--2d0m3tkqhad4yyhr +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Use semantic mutations and a standalone writer +state: accepted +created_at: '2026-07-18T19:42:37.770Z' +derives_from: + - dec-store-graph-artifacts-in-repo-by-default--xxpgwm9sv25j07z7 +--- + +## Context + +An edge such as `supersedes` changes more than one file once its materialized +reverse projection is updated. Asking an agent to patch generic frontmatter +would make it responsible for maintaining that invariant and would leave +partial writes as a corruption mode. Flatbread's core is deliberately a +read-only, in-memory GraphQL projection over source files. + +## Decision + +Expose a small set of typed, semantic mutations. Each mutation owns its Zod +validation, expands a conceptual edit into every required file change, and +completes the group or leaves it unchanged. The standalone Effort Graph writer +owns this logic; Flatbread's GraphQL read layer remains read-only. A future +GraphQL mutation facade may delegate to this writer, but is not its home. + +The writer returns affected IDs and paths plus a monotonic generation token. +Reads are eventual by default; callers can request a strict read at that +generation where a concurrent workflow requires it. + +## Alternatives considered + +- **Create-only mutations plus a generic patch:** rejected because agents + would need to coordinate both sides of semantic edits and recover from + partial application. +- **GraphQL mutations in core:** rejected because it would impose write-back + and cache-coherence concerns on Flatbread's read-only core. + +## Consequences + +The agent-facing API is a versioned set of named operations, not a YAML editing +protocol. Multi-file transaction and recovery semantics are writer concerns. +The mutation enum remains deliberately small, while watch/reindex work and +strict read support are tracked independently rather than being silently +assumed complete. diff --git a/.flatbread-efforts/efforts/eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002.md b/.flatbread-efforts/efforts/eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002.md new file mode 100644 index 00000000..3b240483 --- /dev/null +++ b/.flatbread-efforts/efforts/eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002.md @@ -0,0 +1,9 @@ +--- +id: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Effort Graph memory and agent wedge +slug: effort-graph-memory-and-agent-wedge +status: active +created_at: '2026-07-18T19:41:40.070Z' +--- + +Finish semantic writes, generation-aware reads, layout mapping, and evidence-backed blocking-decision retrieval before promoting the wedge. diff --git a/.flatbread-efforts/efforts/eff-flatbread-product-branding--zt7b35sa05kyvhdz.md b/.flatbread-efforts/efforts/eff-flatbread-product-branding--zt7b35sa05kyvhdz.md new file mode 100644 index 00000000..b884ce80 --- /dev/null +++ b/.flatbread-efforts/efforts/eff-flatbread-product-branding--zt7b35sa05kyvhdz.md @@ -0,0 +1,9 @@ +--- +id: eff-flatbread-product-branding--zt7b35sa05kyvhdz +title: Flatbread product branding +slug: flatbread-product-branding +status: active +created_at: '2026-07-19T03:45:33.828Z' +--- + +Product-facing names, metaphors, and disambiguation language for Flatbread surfaces. Keeps branding commitments out of subsystem implementation Efforts while still binding rename and messaging work. diff --git a/.flatbread-efforts/efforts/eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t.md b/.flatbread-efforts/efforts/eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t.md new file mode 100644 index 00000000..e4c013b4 --- /dev/null +++ b/.flatbread-efforts/efforts/eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t.md @@ -0,0 +1,9 @@ +--- +id: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: Local runtime and ownership loop +slug: local-runtime-and-ownership-loop +status: active +created_at: '2026-07-18T19:41:37.694Z' +--- + +Make watch and rebuild behavior, isolated schemas, and JSON/CSV exit artifacts reliable and honest. diff --git a/.flatbread-efforts/efforts/eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve.md b/.flatbread-efforts/efforts/eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve.md new file mode 100644 index 00000000..a2e93127 --- /dev/null +++ b/.flatbread-efforts/efforts/eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve.md @@ -0,0 +1,9 @@ +--- +id: eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve +title: Proof and contributor operating system +slug: proof-and-contributor-operating-system +status: active +created_at: '2026-07-18T19:41:42.974Z' +--- + +Keep proof convergence/output behavior and repository verification deterministic while deferring broad tooling rewrites. diff --git a/.flatbread-efforts/efforts/eff-relational-content-foundation--8a8332x4cazgf2k0.md b/.flatbread-efforts/efforts/eff-relational-content-foundation--8a8332x4cazgf2k0.md new file mode 100644 index 00000000..34b40ee3 --- /dev/null +++ b/.flatbread-efforts/efforts/eff-relational-content-foundation--8a8332x4cazgf2k0.md @@ -0,0 +1,9 @@ +--- +id: eff-relational-content-foundation--8a8332x4cazgf2k0 +title: Relational content foundation +slug: relational-content-foundation +status: active +created_at: '2026-07-18T19:41:35.689Z' +--- + +Stabilize the Git-native files to model to typed-read path, integrity validation, generated types, and canonical example. diff --git a/.flatbread-efforts/findings/fnd-adrs-created-a-second-planning-authority--j1waeg8qh900sqee.md b/.flatbread-efforts/findings/fnd-adrs-created-a-second-planning-authority--j1waeg8qh900sqee.md new file mode 100644 index 00000000..c5bd3ab6 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-adrs-created-a-second-planning-authority--j1waeg8qh900sqee.md @@ -0,0 +1,9 @@ +--- +id: fnd-adrs-created-a-second-planning-authority--j1waeg8qh900sqee +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: ADRs created a second planning authority +kind: retrospective +created_at: '2026-07-19T00:41:18.477Z' +--- + +Dogfooding exposed a split-brain record of Flatbread planning: ADRs retained full rationale while matching Effort Graph Decisions were terse mirrors. The stale blocking-decision state and external ADR-writing skills made agents treat docs/effort-graph as authoritative. Decision bodies support unrestricted Markdown, so the fragmentation is workflow and guidance rather than a schema limitation. diff --git a/.flatbread-efforts/findings/fnd-bounded-status-briefing-protocol-halves-effort-g--ht1jd1hsssf7mv2y.md b/.flatbread-efforts/findings/fnd-bounded-status-briefing-protocol-halves-effort-g--ht1jd1hsssf7mv2y.md new file mode 100644 index 00000000..7e3e3c80 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-bounded-status-briefing-protocol-halves-effort-g--ht1jd1hsssf7mv2y.md @@ -0,0 +1,18 @@ +--- +id: fnd-bounded-status-briefing-protocol-halves-effort-g--ht1jd1hsssf7mv2y +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Bounded status-briefing protocol halves Effort Graph recall tool calls +kind: measurement +created_at: '2026-07-19T02:03:43.496Z' +--- + +Controlled experiment on how agents engage with the Effort Graph for the recall question "What open work is being done right now?". Design: 3 model families (Composer 2.5, GPT-5.4, GPT-5.6-Luna) x 2 conditions x 2 reps = 12 isolated, read-only subagents. + +- Control: follow the effort-graph skill freely. +- Treatment: a bounded status-briefing protocol - `list --status active` (trust the digest) -> per-effort `records --kinds issue,decision` (read status/state from the digest, never open source markdown) -> `blocking-decisions` only when an effort has an open blocker Issue. + +Results (tool_calls_total): Control median 21 (range 12-25); Treatment median 10 (range 7-11). Distributions do not overlap (Control min 12 > Treatment max 11): Mann-Whitney U=0, two-tailed p ~= 0.0022, Cliff delta = 1.0, roughly 52 percent fewer tool calls, consistent across all three model families. Answer quality scored 5/5 against ground truth (4 efforts, 3 open issues, 2 proposed decisions, 0 blocking) in all 12 runs; files_opened_from_source = 0 in every run. + +The excess Control cost came from three anti-patterns the protocol removes: (1) running blocking-decisions on every effort when none had an open blocker Issue; (2) copying the skill's over-constrained `--status open --state proposed` filter example, getting 0 records, then re-querying; (3) exploratory drift (effort --help, config globbing, get on a resolved Issue, risk queries). + +Caveat: self-reported estimated_bytes_ingested was roughly flat across arms (~22-34k) because digest reads dominate payload. The win is in round-trips / tool-call count and latency, not raw bytes ingested. Scope: one recall question, one graph state (gen 69-70), self-reported counts. [session: effort-graph agent-engagement experiment, 2026-07-18; 12 isolated Cursor subagents] diff --git a/.flatbread-efforts/findings/fnd-dogfooding-the-read-cli-surfaced-three-real-defe--g21hg77x72hdty56.md b/.flatbread-efforts/findings/fnd-dogfooding-the-read-cli-surfaced-three-real-defe--g21hg77x72hdty56.md new file mode 100644 index 00000000..4ff27927 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-dogfooding-the-read-cli-surfaced-three-real-defe--g21hg77x72hdty56.md @@ -0,0 +1,9 @@ +--- +id: fnd-dogfooding-the-read-cli-surfaced-three-real-defe--g21hg77x72hdty56 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Dogfooding the read CLI surfaced three real defects +kind: retrospective +created_at: '2026-07-18T20:42:04.325Z' +--- + +Transferring docs/ onto the graph via the new CLI caught: (1) one-shot CLI commands never exited because config bundling left esbuild watch:true alive; (2) --strict-min-generation was silently dropped by kebab-case option mapping, degrading strict reads to eventual; (3) the engine relation eq comparator matches materialized objects rather than ids, forcing client-side effort-ownership intersection in the projection layer. [session: effort-graph agent runtime, 2026-07-18] diff --git a/.flatbread-efforts/findings/fnd-exports-strengthen-ownership-but-lack-external-v--fx2qtjpcm8mtpw4f.md b/.flatbread-efforts/findings/fnd-exports-strengthen-ownership-but-lack-external-v--fx2qtjpcm8mtpw4f.md new file mode 100644 index 00000000..22d07709 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-exports-strengthen-ownership-but-lack-external-v--fx2qtjpcm8mtpw4f.md @@ -0,0 +1,11 @@ +--- +id: fnd-exports-strengthen-ownership-but-lack-external-v--fx2qtjpcm8mtpw4f +effort: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: Exports strengthen ownership but lack external validation +kind: retrospective +created_at: '2026-07-18T19:42:30.677Z' +derives_from: + - iss-live-reload-and-export-cli-remain-gaps--xe19gzqf654xeyg4 +--- + +JSON preserves normalized IDs and refs and CSV provides a flat view, but trust evidence is a product self-review. diff --git a/.flatbread-efforts/findings/fnd-filtered-retrieval-is-94-percent-smaller--qpvz4vch5hygye20.md b/.flatbread-efforts/findings/fnd-filtered-retrieval-is-94-percent-smaller--qpvz4vch5hygye20.md new file mode 100644 index 00000000..dedea806 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-filtered-retrieval-is-94-percent-smaller--qpvz4vch5hygye20.md @@ -0,0 +1,11 @@ +--- +id: fnd-filtered-retrieval-is-94-percent-smaller--qpvz4vch5hygye20 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Filtered retrieval is 94 percent smaller +kind: measurement +created_at: '2026-07-18T19:43:00.348Z' +derives_from: + - dec-specify-blockingdecisions-over-effort-scoped-edg--pxwaenyra0aa365j +--- + +The representative blocking-decision query retrieved 2,225 bytes versus 37,101 bytes of cold context stuffing. diff --git a/.flatbread-efforts/findings/fnd-generated-read-api-is-useful-but-not-fully-type--vmr23k0t6qx3hk1k.md b/.flatbread-efforts/findings/fnd-generated-read-api-is-useful-but-not-fully-type--vmr23k0t6qx3hk1k.md new file mode 100644 index 00000000..7cb0acf0 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-generated-read-api-is-useful-but-not-fully-type--vmr23k0t6qx3hk1k.md @@ -0,0 +1,9 @@ +--- +id: fnd-generated-read-api-is-useful-but-not-fully-type--vmr23k0t6qx3hk1k +effort: eff-relational-content-foundation--8a8332x4cazgf2k0 +title: Generated read API is useful but not fully type-safe +kind: survey +created_at: '2026-07-18T19:41:55.100Z' +--- + +Generated model helpers and the prototype read API work, but string selections and partial/nullability behavior remain gaps. diff --git a/.flatbread-efforts/findings/fnd-getrecord-digests-call-the-same-excerpt-as-brows--q5rbnhpxc697x2dv.md b/.flatbread-efforts/findings/fnd-getrecord-digests-call-the-same-excerpt-as-brows--q5rbnhpxc697x2dv.md new file mode 100644 index 00000000..90ec68cb --- /dev/null +++ b/.flatbread-efforts/findings/fnd-getrecord-digests-call-the-same-excerpt-as-brows--q5rbnhpxc697x2dv.md @@ -0,0 +1,15 @@ +--- +id: fnd-getrecord-digests-call-the-same-excerpt-as-brows--q5rbnhpxc697x2dv +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: getRecord digests call the same excerpt() as browse reads +kind: measurement +created_at: '2026-07-19T03:44:49.738Z' +derives_from: + - iss-agents-cannot-obtain-full-record-bodies-through--ekfpcg6hrkwgy287 +--- + +Inspected `packages/effort-graph/src/digest.ts`: `renderRecord()` always called `excerpt(record.body_excerpt)` (12 lines / 600 chars) for every query type, including `getRecord`. + +Decision dec-route-agent-reads-through-the-flatbread-query-en--476qb9qk878yfg62 consequences state that full bodies remain available from single-record lookup. Unit coverage in `digest.test.ts` asserted `[…truncated]` on a `getRecord` digest — locking in the contradiction. + +`body_excerpt` on `ReadRecord` already carries the full `_content.raw` from the engine; only the digest renderer truncated it. diff --git a/.flatbread-efforts/findings/fnd-one-canonical-schema-needs-mapping-profiles--k2yvq5e9rcpsm1s1.md b/.flatbread-efforts/findings/fnd-one-canonical-schema-needs-mapping-profiles--k2yvq5e9rcpsm1s1.md new file mode 100644 index 00000000..d0b7e2be --- /dev/null +++ b/.flatbread-efforts/findings/fnd-one-canonical-schema-needs-mapping-profiles--k2yvq5e9rcpsm1s1.md @@ -0,0 +1,11 @@ +--- +id: fnd-one-canonical-schema-needs-mapping-profiles--k2yvq5e9rcpsm1s1 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: One canonical schema needs mapping profiles +kind: survey +created_at: '2026-07-18T19:43:07.822Z' +derives_from: + - dec-keep-effort-graph-secondary-with-a-primary-wedge--estattvqnhffm2dc +--- + +Core nouns remain stable across layouts; identity, partial graphs, noise, and branch lifecycle require explicit profiles. diff --git a/.flatbread-efforts/findings/fnd-phase-1-has-no-blockers-but-retains-test-gaps--91h2ncmytj8mq40r.md b/.flatbread-efforts/findings/fnd-phase-1-has-no-blockers-but-retains-test-gaps--91h2ncmytj8mq40r.md new file mode 100644 index 00000000..c216b14d --- /dev/null +++ b/.flatbread-efforts/findings/fnd-phase-1-has-no-blockers-but-retains-test-gaps--91h2ncmytj8mq40r.md @@ -0,0 +1,20 @@ +--- +id: fnd-phase-1-has-no-blockers-but-retains-test-gaps--91h2ncmytj8mq40r +effort: eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve +title: Phase 1 has no blockers but retains test gaps +kind: retrospective +created_at: '2026-07-18T19:43:35.665Z' +derives_from: + - dec-separate-execution-and-display-planes--spm2ckxvdsch6h9m +--- + +Phase 1 has no merge blocker: in-process output, pinned-resume transcript +reconstruction, convergence feedback, and downstream budget-exceeded skipping +all use the execution transcript where available. + +Residual evidence gaps remain: empty transcripts should not repeatedly read +from disk; missing resumed mirrors should warn instead of silently falling back +to bounded display output; and behavioral tests should cover stitched prompts, +post-loop blocker detection, artifact/mirror parity, sidecar parity, and +budget-exceeded child skipping. Oracle full evidence, stream mirror durability, +and prompt-budget policy remain phased follow-ups. diff --git a/.flatbread-efforts/findings/fnd-pmf-rubric-understates-shipped-validation-and-wa--p04gd8xfknwvz2pe.md b/.flatbread-efforts/findings/fnd-pmf-rubric-understates-shipped-validation-and-wa--p04gd8xfknwvz2pe.md new file mode 100644 index 00000000..ee9a0e5e --- /dev/null +++ b/.flatbread-efforts/findings/fnd-pmf-rubric-understates-shipped-validation-and-wa--p04gd8xfknwvz2pe.md @@ -0,0 +1,21 @@ +--- +id: fnd-pmf-rubric-understates-shipped-validation-and-wa--p04gd8xfknwvz2pe +effort: eff-relational-content-foundation--8a8332x4cazgf2k0 +title: PMF rubric understates shipped validation and watch behavior +kind: retrospective +created_at: '2026-07-19T01:30:56.615Z' +derives_from: + - fnd-reference-integrity-is-roadmap-critical--2ss712xpmsfh77xf + - fnd-unified-watch-loop-is-the-intended-runtime-contr--t9ghag8yqxgf3p5t +--- + +## Evidence + +- `docs/pmf-decision-rubric.md` describes reliable content hot reload as not yet a pillar and treats ordinary content edits requiring a full restart as a no-go signal. +- `docs/local-dev-loop.md` documents `flatbread start --watch` as the supported unified path: valid content/config edits hot-swap the GraphQL schema without restarting the framework. +- `packages/flatbread/src/cli/index.ts` exposes `start --watch`, and live-server tests cover filesystem watch to schema hot-swap. +- The same rubric describes configured reference integrity as uneven and suggests silent query-time null chains, while `validateRecords` runs before schema generation and reports duplicate IDs, missing targets, and invalid reference shapes. + +## Implication + +A buyer-facing comparison page understates current behavior and conflicts with the retained product docs. Refresh its current-vs-target language; retain only documented limitations such as package-code rebuilds, framework-owned refresh, and prototype read-surface gaps. diff --git a/.flatbread-efforts/findings/fnd-postable-brand-names-favor-trail-over-graph--zzpgaw0kvqtaqmxt.md b/.flatbread-efforts/findings/fnd-postable-brand-names-favor-trail-over-graph--zzpgaw0kvqtaqmxt.md new file mode 100644 index 00000000..36994382 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-postable-brand-names-favor-trail-over-graph--zzpgaw0kvqtaqmxt.md @@ -0,0 +1,15 @@ +--- +id: fnd-postable-brand-names-favor-trail-over-graph--zzpgaw0kvqtaqmxt +effort: eff-flatbread-product-branding--zt7b35sa05kyvhdz +title: Postable brand names favor trail over graph +kind: survey +created_at: '2026-07-19T03:54:16.130Z' +derives_from: + - fnd-workshop-shortlist-favors-crumb-graph-over-yeast--tv3ydt3wfb3n4y0g +--- + +Follow-up on the Crumb Graph vs Crumb Trail split. + +Online brandability (sayable in a post, sticky after one glance, metaphor-first like peer agent-tool brands) favors **Crumb Trail**. Agents and humans experience resume/recall as following a trail; that is the phrase that travels. + +**Crumb Graph** remains the accurate explainer for the underlying datamodel (Effort-scoped relational records, edges, digests) and should not compete as the product name. diff --git a/.flatbread-efforts/findings/fnd-primary-onboarding-still-bypasses-the-unified-wa--qqvckprax45hyd2y.md b/.flatbread-efforts/findings/fnd-primary-onboarding-still-bypasses-the-unified-wa--qqvckprax45hyd2y.md new file mode 100644 index 00000000..a2e547e3 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-primary-onboarding-still-bypasses-the-unified-wa--qqvckprax45hyd2y.md @@ -0,0 +1,19 @@ +--- +id: fnd-primary-onboarding-still-bypasses-the-unified-wa--qqvckprax45hyd2y +effort: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: Primary onboarding still bypasses the unified watch path +kind: retrospective +created_at: '2026-07-19T01:32:26.238Z' +derives_from: + - fnd-unified-watch-loop-is-the-intended-runtime-contr--t9ghag8yqxgf3p5t +--- + +## Evidence + +- `packages/flatbread/README.md` directs the Next example to `pnpm dev` and later says live reload is unreliable, although the same README and `docs/local-dev-loop.md` identify `flatbread start --watch` as the supported live content/config path. +- `examples/nextjs/README.md` recommends standalone `flatbread codegen --watch` then says the GraphQL server needs a restart for content/schema changes. The local loop explicitly says not to run standalone codegen watch alongside unified watch. +- `examples/nextjs/app/components/BlogIndex.tsx` displays `npx flatbread dev`, but the CLI has no `dev` subcommand. + +## Implication + +The canonical example steers users toward split, restart-dependent workflows and contains an invalid visible recovery command. Promote `flatbread start --watch -- next dev --turbopack` as the canonical path, describe standalone codegen watch only as a non-unified alternative, and replace the UI command with a valid example-local command. diff --git a/.flatbread-efforts/findings/fnd-public-onboarding-now-directs-users-to-watch-mod--cwgqyjjvr6j9gx8n.md b/.flatbread-efforts/findings/fnd-public-onboarding-now-directs-users-to-watch-mod--cwgqyjjvr6j9gx8n.md new file mode 100644 index 00000000..55797ddb --- /dev/null +++ b/.flatbread-efforts/findings/fnd-public-onboarding-now-directs-users-to-watch-mod--cwgqyjjvr6j9gx8n.md @@ -0,0 +1,22 @@ +--- +id: fnd-public-onboarding-now-directs-users-to-watch-mod--cwgqyjjvr6j9gx8n +effort: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: Public onboarding now directs users to watch mode +kind: retrospective +created_at: '2026-07-19T01:53:00.252Z' +derives_from: + - fnd-pmf-rubric-understates-shipped-validation-and-wa--p04gd8xfknwvz2pe + - fnd-primary-onboarding-still-bypasses-the-unified-wa--qqvckprax45hyd2y + - fnd-public-readmes-retain-superseded-live-reload-gui--naafmpvt3pcdes4p +--- + +## Evidence + +- The Next.js `dev` script now starts `flatbread start --watch`, and contributor and example guides use the same development command. +- The main README, example README, and comparison page now explain that valid content and config changes reload with watch mode. They keep the real limits for package rebuilds and framework refreshes. +- The example empty state now shows the valid `pnpm dev` command instead of the unsupported `flatbread dev`. +- Public guides now use plainer language, move internal planning details out of the first-use path, and link to the glossary when readers need product terms. + +## Verification + +Formatting, type checks, and generated-skill checks completed after the documentation changes. diff --git a/.flatbread-efforts/findings/fnd-public-readmes-retain-superseded-live-reload-gui--naafmpvt3pcdes4p.md b/.flatbread-efforts/findings/fnd-public-readmes-retain-superseded-live-reload-gui--naafmpvt3pcdes4p.md new file mode 100644 index 00000000..3c43e2ab --- /dev/null +++ b/.flatbread-efforts/findings/fnd-public-readmes-retain-superseded-live-reload-gui--naafmpvt3pcdes4p.md @@ -0,0 +1,20 @@ +--- +id: fnd-public-readmes-retain-superseded-live-reload-gui--naafmpvt3pcdes4p +effort: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: Public READMEs retain superseded live-reload guidance +kind: retrospective +created_at: '2026-07-19T01:30:04.460Z' +derives_from: + - fnd-unified-watch-loop-is-the-intended-runtime-contr--t9ghag8yqxgf3p5t +--- + +## Evidence + +- `README.md` says reliable live reload is unsupported and tells readers to restart after edits. +- `packages/flatbread/README.md` repeats the same claim and links GitHub issue #65. +- `docs/local-dev-loop.md` documents `flatbread start --watch` as the current unified path: valid content/config edits hot-swap the GraphQL schema while the framework process remains running. +- `packages/flatbread/src/cli/index.ts` exposes `start --watch` as "Hot-swap content and reload config". + +## Implication + +First-time adopters receive conflicting setup guidance and may choose a restart-only workflow even though the documented supported watcher exists. Update the two README passages to point to the local development loop and preserve only the real package-code/framework restart boundary. diff --git a/.flatbread-efforts/findings/fnd-reference-integrity-is-roadmap-critical--2ss712xpmsfh77xf.md b/.flatbread-efforts/findings/fnd-reference-integrity-is-roadmap-critical--2ss712xpmsfh77xf.md new file mode 100644 index 00000000..c33c3cd9 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-reference-integrity-is-roadmap-critical--2ss712xpmsfh77xf.md @@ -0,0 +1,9 @@ +--- +id: fnd-reference-integrity-is-roadmap-critical--2ss712xpmsfh77xf +effort: eff-relational-content-foundation--8a8332x4cazgf2k0 +title: Reference integrity is roadmap-critical +kind: measurement +created_at: '2026-07-18T19:42:05.122Z' +--- + +Missing refs, duplicate IDs, and invalid relation shapes should fail clearly before query-time null chains. diff --git a/.flatbread-efforts/findings/fnd-relation-first-starter-reaches-a-typed-read--8anm4wxbjm9f348z.md b/.flatbread-efforts/findings/fnd-relation-first-starter-reaches-a-typed-read--8anm4wxbjm9f348z.md new file mode 100644 index 00000000..bb01595f --- /dev/null +++ b/.flatbread-efforts/findings/fnd-relation-first-starter-reaches-a-typed-read--8anm4wxbjm9f348z.md @@ -0,0 +1,9 @@ +--- +id: fnd-relation-first-starter-reaches-a-typed-read--8anm4wxbjm9f348z +effort: eff-relational-content-foundation--8a8332x4cazgf2k0 +title: Relation-first starter reaches a typed read +kind: measurement +created_at: '2026-07-18T19:41:47.615Z' +--- + +The fresh-worktree benchmark completed install, build, codegen, and first query in 49 seconds; this is not network-cold evidence. diff --git a/.flatbread-efforts/findings/fnd-root-readme-roadmap-reference-is-already-removed--w56t43aa10v311hj.md b/.flatbread-efforts/findings/fnd-root-readme-roadmap-reference-is-already-removed--w56t43aa10v311hj.md new file mode 100644 index 00000000..9c54c77c --- /dev/null +++ b/.flatbread-efforts/findings/fnd-root-readme-roadmap-reference-is-already-removed--w56t43aa10v311hj.md @@ -0,0 +1,15 @@ +--- +id: fnd-root-readme-roadmap-reference-is-already-removed--w56t43aa10v311hj +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Root README roadmap reference is already removed +kind: retrospective +created_at: '2026-07-19T01:31:20.259Z' +invalidates: + - fnd-root-readme-still-links-to-the-deleted-roadmap--k38zmv7pqv7z95ts +--- + +## Correction + +The prior audit finding was based on a stale read of the root symlink. `README.md` resolves to `packages/flatbread/README.md`; the current working-tree version has already replaced the old `docs/roadmap.md` link with the Effort Graph planning section. + +The previous finding is therefore invalid for this worktree. Separate relative-link-base concerns in the symlinked README remain an audit item. diff --git a/.flatbread-efforts/findings/fnd-root-readme-still-links-to-the-deleted-roadmap--k38zmv7pqv7z95ts.md b/.flatbread-efforts/findings/fnd-root-readme-still-links-to-the-deleted-roadmap--k38zmv7pqv7z95ts.md new file mode 100644 index 00000000..aea34658 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-root-readme-still-links-to-the-deleted-roadmap--k38zmv7pqv7z95ts.md @@ -0,0 +1,19 @@ +--- +id: fnd-root-readme-still-links-to-the-deleted-roadmap--k38zmv7pqv7z95ts +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Root README still links to the deleted roadmap +kind: retrospective +created_at: '2026-07-19T01:30:05.841Z' +derives_from: + - fnd-adrs-created-a-second-planning-authority--j1waeg8qh900sqee +invalidated_by: + - fnd-root-readme-roadmap-reference-is-already-removed--w56t43aa10v311hj +--- + +## Evidence + +`README.md` has a public "Roadmap" sentence linking to `docs/roadmap.md`. That file was intentionally removed when planning state moved into `.flatbread-efforts/`, so the repository-relative link is broken in a fresh clone and on GitHub. + +## Implication + +The main project entrypoint contradicts the new canonical planning location. Replace the stale roadmap copy with the bounded Effort Graph entrypoint or remove it from product-facing onboarding. diff --git a/.flatbread-efforts/findings/fnd-schemas-need-per-build-isolation--vdjfmtqb0dbjfcqk.md b/.flatbread-efforts/findings/fnd-schemas-need-per-build-isolation--vdjfmtqb0dbjfcqk.md new file mode 100644 index 00000000..cde8da9b --- /dev/null +++ b/.flatbread-efforts/findings/fnd-schemas-need-per-build-isolation--vdjfmtqb0dbjfcqk.md @@ -0,0 +1,9 @@ +--- +id: fnd-schemas-need-per-build-isolation--vdjfmtqb0dbjfcqk +effort: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: Schemas need per-build isolation +kind: retrospective +created_at: '2026-07-18T19:42:15.139Z' +--- + +A fresh composer/schema per build prevents resolver closures from reading a prior content snapshot. diff --git a/.flatbread-efforts/findings/fnd-unified-watch-loop-is-the-intended-runtime-contr--t9ghag8yqxgf3p5t.md b/.flatbread-efforts/findings/fnd-unified-watch-loop-is-the-intended-runtime-contr--t9ghag8yqxgf3p5t.md new file mode 100644 index 00000000..95d697ea --- /dev/null +++ b/.flatbread-efforts/findings/fnd-unified-watch-loop-is-the-intended-runtime-contr--t9ghag8yqxgf3p5t.md @@ -0,0 +1,9 @@ +--- +id: fnd-unified-watch-loop-is-the-intended-runtime-contr--t9ghag8yqxgf3p5t +effort: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: Unified watch loop is the intended runtime contract +kind: measurement +created_at: '2026-07-18T19:42:22.692Z' +--- + +The watcher classifies content, config, and doc events, serializes rebuilds, and hot-swaps only validated generations. diff --git a/.flatbread-efforts/findings/fnd-versioned-skill-distribution-passes-full-reposit--hm5cy9j2zts96dg4.md b/.flatbread-efforts/findings/fnd-versioned-skill-distribution-passes-full-reposit--hm5cy9j2zts96dg4.md new file mode 100644 index 00000000..e876de49 --- /dev/null +++ b/.flatbread-efforts/findings/fnd-versioned-skill-distribution-passes-full-reposit--hm5cy9j2zts96dg4.md @@ -0,0 +1,11 @@ +--- +id: fnd-versioned-skill-distribution-passes-full-reposit--hm5cy9j2zts96dg4 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Versioned skill distribution passes full repository verification +kind: retrospective +created_at: '2026-07-19T00:01:36.105Z' +derives_from: + - dec-distribute-the-effort-graph-as-a-versioned-agent--8as4sybr34zcqyqh +--- + +The package-owned Agent Skill, generated .agents projection, first-activation bootstrap, active-Effort list query, release manifest, pack gate, topological publisher, and dependent bump closure were implemented together. `pnpm verify` passed, including 316 AVA tests, 7 utils tests, 45 codegen tests, build, lint, typecheck, skill projection, and packed-payload checks. Adversarial review also surfaced and closed strict-token, pagination cache/completeness, config temp-file concurrency, watcher portability, and partial-release retry defects. diff --git a/.flatbread-efforts/findings/fnd-workshop-shortlist-favors-crumb-graph-over-yeast--tv3ydt3wfb3n4y0g.md b/.flatbread-efforts/findings/fnd-workshop-shortlist-favors-crumb-graph-over-yeast--tv3ydt3wfb3n4y0g.md new file mode 100644 index 00000000..9502bc3d --- /dev/null +++ b/.flatbread-efforts/findings/fnd-workshop-shortlist-favors-crumb-graph-over-yeast--tv3ydt3wfb3n4y0g.md @@ -0,0 +1,16 @@ +--- +id: fnd-workshop-shortlist-favors-crumb-graph-over-yeast--tv3ydt3wfb3n4y0g +effort: eff-flatbread-product-branding--zt7b35sa05kyvhdz +title: Workshop shortlist favors Crumb Graph over yeast metaphors +kind: survey +created_at: '2026-07-19T03:45:38.089Z' +--- + +Naming workshop for longform agent memory (today: Effort Graph) against Flatbread brand fit. + +- Yeast metaphors (Leaven, Sourdough, Starter) clash with Flatbread as unleavened bread. +- Proof is already taken as the DAG execution surface; the memory product needs a peer-level name, not another subsystem label alone. +- Crumb Graph landed as cute, brandable, and honest: agents leave reasoning crumbs in-repo; recall follows the trail; Graph keeps the relational structure visible. +- Effort Graph remains useful as a technical/descriptor name for the same system (primitives anchored by Effort, graph of reasoning records). + +Kind: qualitative brand workshop, not a measurement. diff --git a/.flatbread-efforts/issues/iss-agents-cannot-obtain-full-record-bodies-through--ekfpcg6hrkwgy287.md b/.flatbread-efforts/issues/iss-agents-cannot-obtain-full-record-bodies-through--ekfpcg6hrkwgy287.md new file mode 100644 index 00000000..a24200af --- /dev/null +++ b/.flatbread-efforts/issues/iss-agents-cannot-obtain-full-record-bodies-through--ekfpcg6hrkwgy287.md @@ -0,0 +1,15 @@ +--- +id: iss-agents-cannot-obtain-full-record-bodies-through--ekfpcg6hrkwgy287 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Agents cannot obtain full record bodies through CLI reads +kind: defect +status: resolved +created_at: '2026-07-19T03:44:48.333Z' +resolved_by: + - dec-effort-get-digests-always-include-the-full-recor--xs7rnbha3zx8qdj9 + - fnd-getrecord-digests-call-the-same-excerpt-as-brows--q5rbnhpxc697x2dv +--- + +Bounded digests excerpt every record body to 12 lines / 600 chars, including `effort get`. Accepted Decision dec-route-agent-reads-through-the-flatbread-query-en--476qb9qk878yfg62 claims full bodies remain available from single-record lookup, but implementation never honored that. + +Agents doing status briefing or zoom-in invent missing Decision sections or invent source paths because digests show `[…truncated]` and `_path` is never exposed. This causes hallucination and missed context on long Decision bodies. diff --git a/.flatbread-efforts/issues/iss-define-blocking-decision-semantics--3bj9ph7ppab2c7g2.md b/.flatbread-efforts/issues/iss-define-blocking-decision-semantics--3bj9ph7ppab2c7g2.md new file mode 100644 index 00000000..00e19e1a --- /dev/null +++ b/.flatbread-efforts/issues/iss-define-blocking-decision-semantics--3bj9ph7ppab2c7g2.md @@ -0,0 +1,16 @@ +--- +id: iss-define-blocking-decision-semantics--3bj9ph7ppab2c7g2 +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Define blocking decision semantics +kind: blocker +status: resolved +created_at: '2026-07-18T19:42:55.276Z' +derives_from: + - con-mutation-enum-stays-deliberately-small--45v1ae3neq26g1rz +resolved_by: + - dec-specify-blockingdecisions-over-effort-scoped-edg--pxwaenyra0aa365j +--- + +The bounded agent-read design needed a precise definition against the current +edge vocabulary. `dec-specify-blockingdecisions-over-effort-scoped-edg--pxwaenyra0aa365j` +now resolves this blocker. diff --git a/.flatbread-efforts/issues/iss-implement-crumb-graph-rename-across-packages-cli--dp6jvt2kafab7m4t.md b/.flatbread-efforts/issues/iss-implement-crumb-graph-rename-across-packages-cli--dp6jvt2kafab7m4t.md new file mode 100644 index 00000000..520ea1b1 --- /dev/null +++ b/.flatbread-efforts/issues/iss-implement-crumb-graph-rename-across-packages-cli--dp6jvt2kafab7m4t.md @@ -0,0 +1,18 @@ +--- +id: iss-implement-crumb-graph-rename-across-packages-cli--dp6jvt2kafab7m4t +effort: eff-flatbread-product-branding--zt7b35sa05kyvhdz +title: 'Implement Crumb Graph rename across packages, CLI, paths, and skills' +kind: gap +status: open +created_at: '2026-07-19T03:45:50.712Z' +derives_from: + - dec-brand-the-agent-memory-surface-as-crumb-graph--fvskcvagx3a7sybe +superseded_by: + - iss-implement-crumb-trail-rename-across-packages-cli--316rdzpt80sdxkw1 +--- + +Branding Decision accepts Crumb Graph as the product name with Effort Graph as descriptor. Implementation is pending and out of scope for the branding commit. + +Likely touchpoints (non-exhaustive): `@flatbread/effort-graph`, `effortGraphContent`, `flatbread effort` CLI, `.flatbread-efforts/`, `.flatbread/effort-graph/`, agent skill name/paths, docs/README copy, error codes (`EFFORT_GRAPH_*`), and dogfood references. + +Do not start the mechanical rename until an implementation plan chooses which identifiers move vs stay for compatibility. diff --git a/.flatbread-efforts/issues/iss-implement-crumb-trail-rename-across-packages-cli--316rdzpt80sdxkw1.md b/.flatbread-efforts/issues/iss-implement-crumb-trail-rename-across-packages-cli--316rdzpt80sdxkw1.md new file mode 100644 index 00000000..c3179aaf --- /dev/null +++ b/.flatbread-efforts/issues/iss-implement-crumb-trail-rename-across-packages-cli--316rdzpt80sdxkw1.md @@ -0,0 +1,18 @@ +--- +id: iss-implement-crumb-trail-rename-across-packages-cli--316rdzpt80sdxkw1 +effort: eff-flatbread-product-branding--zt7b35sa05kyvhdz +title: 'Implement Crumb Trail rename across packages, CLI, paths, and skills' +kind: gap +status: open +created_at: '2026-07-19T03:54:27.962Z' +derives_from: + - dec-brand-the-agent-memory-surface-as-crumb-trail--tngncepdbwjkh9jc +supersedes: + - iss-implement-crumb-graph-rename-across-packages-cli--dp6jvt2kafab7m4t +--- + +Supersedes the Crumb Graph rename Issue. Branding now targets Crumb Trail as the product name (Crumb Graph = datamodel explainer; Effort Graph = technical descriptor). + +Implementation remains pending. Likely touchpoints (non-exhaustive): `@flatbread/effort-graph`, `effortGraphContent`, `flatbread effort` CLI, `.flatbread-efforts/`, `.flatbread/effort-graph/`, agent skill name/paths, docs/README copy, error codes (`EFFORT_GRAPH_*`), and dogfood references. + +Do not start the mechanical rename until an implementation plan chooses which identifiers move vs stay for compatibility, and how Crumb Trail / Crumb Graph / Effort Graph map onto public vs internal names. diff --git a/.flatbread-efforts/issues/iss-live-reload-and-export-cli-remain-gaps--xe19gzqf654xeyg4.md b/.flatbread-efforts/issues/iss-live-reload-and-export-cli-remain-gaps--xe19gzqf654xeyg4.md new file mode 100644 index 00000000..59b99020 --- /dev/null +++ b/.flatbread-efforts/issues/iss-live-reload-and-export-cli-remain-gaps--xe19gzqf654xeyg4.md @@ -0,0 +1,12 @@ +--- +id: iss-live-reload-and-export-cli-remain-gaps--xe19gzqf654xeyg4 +effort: eff-local-runtime-and-ownership-loop--g28gfbb0kdrnpe2t +title: Live reload and export CLI remain gaps +kind: gap +status: open +created_at: '2026-07-18T19:42:25.205Z' +derives_from: + - fnd-unified-watch-loop-is-the-intended-runtime-contr--t9ghag8yqxgf3p5t +--- + +Watch design and JSON/CSV APIs exist, but CLI export and reliable long-running integration remain follow-ups. diff --git a/.flatbread-efforts/issues/iss-proof-output-retention-follow-ups-remain-open--4fkjb93v3pey43y0.md b/.flatbread-efforts/issues/iss-proof-output-retention-follow-ups-remain-open--4fkjb93v3pey43y0.md new file mode 100644 index 00000000..8a129381 --- /dev/null +++ b/.flatbread-efforts/issues/iss-proof-output-retention-follow-ups-remain-open--4fkjb93v3pey43y0.md @@ -0,0 +1,12 @@ +--- +id: iss-proof-output-retention-follow-ups-remain-open--4fkjb93v3pey43y0 +effort: eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve +title: Proof output-retention follow-ups remain open +kind: gap +status: open +created_at: '2026-07-18T19:43:38.051Z' +derives_from: + - fnd-phase-1-has-no-blockers-but-retains-test-gaps--91h2ncmytj8mq40r +--- + +The review tracks hardening, prompt policy, resume ergonomics, oracle evidence, and documentation follow-ups. diff --git a/.flatbread-efforts/issues/iss-skill-scoped-records-filter-example-over-constra--vd0gcnpc9cm6jzjh.md b/.flatbread-efforts/issues/iss-skill-scoped-records-filter-example-over-constra--vd0gcnpc9cm6jzjh.md new file mode 100644 index 00000000..73cf26fe --- /dev/null +++ b/.flatbread-efforts/issues/iss-skill-scoped-records-filter-example-over-constra--vd0gcnpc9cm6jzjh.md @@ -0,0 +1,14 @@ +--- +id: iss-skill-scoped-records-filter-example-over-constra--vd0gcnpc9cm6jzjh +effort: eff-effort-graph-memory-and-agent-wedge--szeqvmgqjqnhd002 +title: Skill scoped-records filter example over-constrains and returns nothing +kind: defect +status: open +created_at: '2026-07-19T02:03:44.673Z' +derives_from: + - fnd-bounded-status-briefing-protocol-halves-effort-g--ht1jd1hsssf7mv2y +--- + +The effort-graph skill and reference document the scoped-listing example `flatbread effort records --kinds issue,decision --status open --state proposed`. Because filter flags AND across kinds, this requires a single record to satisfy both `status: open` (an Issue field) and `state: proposed` (a Decision field), which no record can. Verified at generation 70 on eff-local-runtime-and-ownership-loop: the combined form returns "0 records", while `--kinds issue,decision` returns 3, `--kinds issue --status open` returns 1, and `--kinds decision --state proposed` returns 1. + +In the experiment, agents that followed the documented example got nothing back and then re-ran separate queries, inflating tool calls. Fix the documented example (use two separate scoped queries, or clarify that --status and --state target disjoint kinds) in packages/effort-graph/skills (canonical) and re-sync the generated .agents copy. diff --git a/.flatbread-efforts/issues/iss-typed-selection-builder-remains-a-product-gap--4zawc913n4gnyfcw.md b/.flatbread-efforts/issues/iss-typed-selection-builder-remains-a-product-gap--4zawc913n4gnyfcw.md new file mode 100644 index 00000000..c7676c9f --- /dev/null +++ b/.flatbread-efforts/issues/iss-typed-selection-builder-remains-a-product-gap--4zawc913n4gnyfcw.md @@ -0,0 +1,12 @@ +--- +id: iss-typed-selection-builder-remains-a-product-gap--4zawc913n4gnyfcw +effort: eff-relational-content-foundation--8a8332x4cazgf2k0 +title: Typed selection builder remains a product gap +kind: gap +status: open +created_at: '2026-07-18T19:42:02.615Z' +derives_from: + - fnd-generated-read-api-is-useful-but-not-fully-type--vmr23k0t6qx3hk1k +--- + +The current string-selection escape hatch is not compile-time checked; add a typed projection/selection builder or isolate it. diff --git a/.flatbread-efforts/risks/rsk-full-transcripts-can-overflow-prompts-and-leak-s--3p7ybk07jqe5w593.md b/.flatbread-efforts/risks/rsk-full-transcripts-can-overflow-prompts-and-leak-s--3p7ybk07jqe5w593.md new file mode 100644 index 00000000..6dab451f --- /dev/null +++ b/.flatbread-efforts/risks/rsk-full-transcripts-can-overflow-prompts-and-leak-s--3p7ybk07jqe5w593.md @@ -0,0 +1,11 @@ +--- +id: rsk-full-transcripts-can-overflow-prompts-and-leak-s--3p7ybk07jqe5w593 +effort: eff-proof-and-contributor-operating-system--ahhgtafvdhg4dfve +title: Full transcripts can overflow prompts and leak secrets +state: open +created_at: '2026-07-18T19:43:30.825Z' +likelihood: medium +severity: high +--- + +An execution-authoritative transcript can exceed model context or expose sensitive output; mitigation requires retention and hygiene policies. diff --git a/.github/workflows/pipeline.yml b/.github/workflows/pipeline.yml index 45ec81e4..71b3fede 100644 --- a/.github/workflows/pipeline.yml +++ b/.github/workflows/pipeline.yml @@ -49,6 +49,12 @@ jobs: - name: Build run: pnpm build + - name: Check canonical skill projection + run: pnpm skills:check + + - name: Check packed skill payload + run: pnpm skills:pack-check + lint: runs-on: ${{ matrix.os }} diff --git a/.gitignore b/.gitignore index c069b1c4..1f2d5756 100644 --- a/.gitignore +++ b/.gitignore @@ -14,4 +14,5 @@ # logs yarn-error.log .pnpm-debug.log -**/.flatbread-efforts/.journal/ \ No newline at end of file +**/.flatbread-efforts/.journal/ +**/.flatbread/effort-graph/read-cache/ \ No newline at end of file diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 43604fc4..956ddee6 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -2,9 +2,12 @@ Thanks for your interest in contributing! This guide covers local development and the release process (bumping versions and publishing packages). -**Flatbread** is **relational, Git-tracked content for TypeScript apps**: flat files in the repo become a typed content graph. **GraphQL is one consumer** of that graph (see `docs/glossary.md`), not the whole product story. +**Flatbread** turns related content files in Git into typed data for TypeScript +apps. **GraphQL is one way to read that data** (see `docs/glossary.md`); it is +not the whole product. -For the **canonical posts / authors / tags** onboarding narrative (collections, `refs`, codegen, then GraphQL), see the [Flatbread package README quickstart](https://github.com/FlatbreadLabs/flatbread/blob/main/packages/flatbread/README.md#quickstart-posts-authors-and-tags) (traceability: **files → config → query interface**, tied to **`docs/glossary.md`**). +For a first project with posts, authors, and tags, see the +[Flatbread package README quickstart](https://github.com/FlatbreadLabs/flatbread/blob/main/packages/flatbread/README.md#quickstart-posts-authors-and-tags). ## Prerequisites @@ -14,14 +17,15 @@ For the **canonical posts / authors / tags** onboarding narrative (collections, ## Recommended onboarding (try Flatbread in the Next.js example) -Use this single path first; it matches how CI and most contributors exercise the stack (**shared content** under `examples/content`, symlinked from the Next app as `content/`): +Use this path first. The Next.js app reads shared content from +`examples/content` through its `content/` symlink: 1. From the **monorepo root**: `pnpm install` then `pnpm build` (builds all packages except `examples/*`). 2. `cd examples/nextjs` 3. One-shot codegen: `pnpm exec flatbread codegen --verbose` (output: `generated/graphql.ts`; globs and dirs come from `flatbread.config.js`). 4. Run the app **and** Flatbread together with **`flatbread start`** (there is **no** `flatbread dev` subcommand): - - **`pnpm dev`** — Next dev with local HTTPS + Flatbread (GraphQL on **5057**, Next on **3000**). - - Headless / no HTTPS: `pnpm exec flatbread start -- next dev --turbopack`. + - **`pnpm dev`** — starts Next with local HTTPS and watches Flatbread content, config, and GraphQL documents. GraphQL runs on **5057** and Next on **3000**. + - Headless / no HTTPS: `pnpm exec flatbread start --watch -- next dev --turbopack`. Optional **`pnpm play`** from the repo root is a shortcut for **`cd examples/nextjs && pnpm dev`** — same as step 4 above, not a separate product command. @@ -101,7 +105,8 @@ What the script does: - Comparing git commits in `packages/` since that time - Ignoring commits that only change the `version` field in `package.json` - Skipping packages that are not yet published on npm -- Preselects only changed packages for you to bump +- Preselects changed packages and their workspace dependents for you to bump +- Required workspace dependents must remain selected when a changed dependency is selected - Runs `pnpm bumpp --no-commit --no-push --no-tag` in each selected package directory Notes: @@ -125,6 +130,21 @@ Notes: > Note: you must have access permissions on NPM +When changing the Effort Graph skill, edit the source files under +`packages/effort-graph/skills/effort-graph/`, then run these checks in order: + +```bash +pnpm skills:sync +pnpm skills:check +pnpm skills:pack-check +``` + +Bump and publish `@flatbread/effort-graph` and `flatbread` together when the +skill and runtime need matching versions. The publish script checks the copied +skill files and package contents first. It then publishes ordinary packages, +`@flatbread/effort-graph`, and finally `flatbread`, stopping at the first +failure. + Publish all public packages (the script builds first and then attempts to publish each package): ```bash @@ -134,25 +154,47 @@ pnpm publish:ci Details: - Builds the repo: `pnpm run build` -- Iterates public packages in `packages/*` and runs: +- Iterates public packages in dependency-safe deterministic order and runs: ```bash pnpm publish --access public --no-git-checks ``` -- If a package's version was not changed, the publish for that package will error and the script will move on to the next +- Before each publish, checks `npm view @ version --json`. An + exact version already in the registry is reported as already published and + skipped; npm not-found responses proceed to publish, while authentication, + network, and other errors abort before that package is published. +- If a release stops after some packages publish, rerun `pnpm publish:ci` + safely. Exact versions already published are skipped, and the script resumes + with the first package that still needs publishing. - Unpublished packages will be published for the first time - Dist-tags (alpha/beta) are currently disabled in the script. If you need them, bump with a pre-release version (`x.y.z-alpha.n`) and add tagging logic in `scripts/publish.ts` ### Post-publish -- Push your commits: +- Only after every package publishes successfully, create an annotated, + immutable `v` Git tag at the exact release commit SHA + printed by `pnpm publish:ci`, then push the release commit and tag: ```bash + git tag -a v -m "Release v" git push + git push origin v ``` -- Tag a new release via Github and include a set of changes with Dev Experience in mind + Protect release tags in the repository settings so they cannot be moved or + deleted after publication. + +End users install the skill from that release tag and install the matching +`flatbread` version: + +```bash +npx skills add https://github.com/FlatbreadLabs/flatbread/tree/vX/packages/effort-graph/skills/effort-graph --skill effort-graph +npm install --save-dev flatbread@X +``` + +`skills update` does not advance an immutable tag. To upgrade deliberately, +install a newer release tag and its matching `flatbread` version. ## Troubleshooting diff --git a/docs/data-ownership.md b/docs/data-ownership.md index 28ceb243..785e9012 100644 --- a/docs/data-ownership.md +++ b/docs/data-ownership.md @@ -82,9 +82,10 @@ durable exit surfaces. ## Current limitations -- JSON/CSV export is currently an API surface, not a first-class CLI command. +- JSON/CSV export is available through the API, not a CLI command. - CSV is a flat view: nested object fields are omitted, and relation fields are exported as reference IDs rather than expanded records. - Generated TypeScript read helpers execute through the GraphQL layer today. -- Live content reload for the long-running `flatbread start` server remains a - separate watch-loop effort; see [local dev loop boundaries](./local-dev-loop.md). +- Live content/config reload is available through `flatbread start --watch`; + package-code changes still require their own rebuild or restart. See + [local dev loop boundaries](./local-dev-loop.md). diff --git a/docs/edit-file-see-query-update-demo.md b/docs/edit-file-see-query-update-demo.md index 3679d604..17786582 100644 --- a/docs/edit-file-see-query-update-demo.md +++ b/docs/edit-file-see-query-update-demo.md @@ -1,15 +1,16 @@ # Edit file → see query update demo This is a single-process demo harness, not the long-running `flatbread start` -server. Production live-editing still requires the unified watch design -described in [local-dev-loop.md](./local-dev-loop.md). +server. Production live-editing uses the unified watch path described in +[local-dev-loop.md](./local-dev-loop.md). -This demo is the current reproducible path for issue #158. It proves the core -edit/query loop for the canonical **posts → authors + tags** model without -requiring a manual process restart in this focused demo path. +This demo is the current reproducible path for issue #158. It shows the +edit/query loop for posts, authors, and tags without requiring a manual +restart. -The full `flatbread start` GraphQL server still builds its schema at startup -(see [local dev loop boundaries](./local-dev-loop.md)). This demo therefore +One-shot `flatbread start` builds its schema at startup; use +`flatbread start --watch` for live content/config updates (see +[local dev loop boundaries](./local-dev-loop.md)). This demo therefore uses a tiny watcher script that rebuilds the Flatbread schema per file event and executes the same posts/authors/tags query shape the generated TypeScript read API uses in the Next.js example. @@ -82,6 +83,5 @@ pnpm --filter nextjs run demo:restore - ✅ The query includes post fields, tag facets, resolved Markdown author records, and resolved YAML author records. - ✅ The demo is reproducible from the monorepo root with pnpm commands. -- ⚠️ The long-running `flatbread start` server still needs restart for content - and schema changes today. -- ⚠️ The watcher script is a demo harness, not the final `flatbread start --watch` implementation described in [local-dev-loop.md](./local-dev-loop.md). +- ⚠️ The demo watcher is a focused harness, not a replacement for the unified + `flatbread start --watch` path. diff --git a/docs/effort-graph/CONTEXT.md b/docs/effort-graph/CONTEXT.md deleted file mode 100644 index aee3aae5..00000000 --- a/docs/effort-graph/CONTEXT.md +++ /dev/null @@ -1,80 +0,0 @@ -# Effort Graph glossary — reasoning primitives for `@flatbread/proof` - -This glossary defines the **epistemic primitives** that make up the Effort Graph — a persistent, queryable memory layer over the reasoning and planning that happens during long-horizon software work, single-agent or multi-agent. - -The Effort Graph is **built on top of** Flatbread's content-layer vocabulary (see [Flatbread glossary](../glossary.md) for `Collection`, `Record`, `Relation`, `Refs`, `ID`). Each primitive here is a Flatbread **Collection**; instances are **Records**; cross-primitive references are **Relations** wired through frontmatter `refs`. - -**What this is not:** a CMS, an authoring UI, a hosted memory product, or a general task tracker. It is a relational substrate for capturing the **gray-area reasoning** that an ADR-only record loses: open questions, considered alternatives, sticky constraints, prospective risks, and post-hoc invalidations. - -**Operational provenance** (which session produced this, which agent, which model, which DAG run) is captured as **frontmatter fields** on these primitives, not as peer collections. The durable transcript record lives next to the graph under `.flatbread/artifacts/` (see [`packages/proof/README.md`](../../packages/proof/README.md) §Artifact Output). - -**Committed generation.** The opaque journal generation token returned by an Effort Graph mutation. It is published only after the full save group has produced a committed live schema (the `CommittedGenerationPublisher` seam — not to be confused with `EffortGraphIndex`, the plan-time read interface). It is a different counter from any process-local live-schema generation; the committed-generation bridge maps the former to the latter for strict readers. - ---- - -### Effort - -The **anchor** of the graph. One Effort represents a coherent, named thread of work — a feature, a migration, a spike, a research investigation, a refactor. Every epistemic primitive belongs to exactly one Effort. - -An Effort is the stable filter (`effort: { eq: "" }`) that scopes every "what's still open / what did we conclude / what are we considering?" query. Loss of the Effort anchor is the failure mode that vault MCPs and flat memory stores cannot avoid; preserving it is the central wedge. - -An Effort has its own lifecycle (active, paused, completed, abandoned) but carries **no reasoning content of its own** — its body is a short description; the reasoning lives in the primitives that ref back to it. - -### Issue - -A **tracked unit needing attention within an Effort**, in the GitHub-issue sense — broader than "something is wrong." Issues span open questions, observed defects, identified gaps, and explicit blockers. Each Issue carries a `kind` field that names the speech act (`question`, `defect`, `gap`, `blocker`, …) and a status (`open`, `resolved`, `deferred`, `wontfix`). - -An Issue is resolved by a Decision (we'll do X) and/or one or more Findings (here's what we learned that closes this). The `kind` is open-ended (free-form string) so common values emerge from dogfooding rather than from schema enforcement. - -Feature _proposals_ are not Issues — they are `Decision{state: proposed}`. Issues are reactive (something exists that needs attention); proposed Decisions are proactive (let's commit to doing X). - -### Finding - -A **grounded observation** — a claim about reality (the codebase, the user, the literature, the runtime) backed by cited evidence. Findings resolve Issues, support or contradict Decisions, surface Risks, and invalidate prior Findings or Decisions when reality refutes a prior belief. - -The `Finding{kind: retrospective}` variant carries the additional semantic that the Finding was produced **after a Decision shipped** and may invalidate that Decision in light of new evidence. Other Finding kinds (e.g. `measurement`, `survey`, `dead-end`) may emerge from usage but are not load-bearing in the schema. - -### Decision - -A **commitment** — a chosen path among alternatives. Has a `state`: `proposed` (under consideration), `accepted` (committed), `rejected` (an alternative we chose not to take), `superseded` (replaced by a later Decision), or `deprecated` (no longer current but not replaced). - -Multiple `state: proposed` Decisions under the same Effort represent **competing directions under exploration**. When one is `accepted`, the others should transition to `rejected` with a back-pointer to the accepted Decision. This is the schema's substitute for a separate `Proposal` primitive. - -A Decision cites the Findings, Constraints, and Risks it weighed; it does not duplicate their content. - -### Constraint - -A **sticky boundary** that scopes the decision space for an Effort. May be hard (license incompatibility, regulatory rule, irreversible upstream choice) or soft (team preference, budget envelope, performance target). Constraints typically outlive individual Decisions and apply to many of them. - -A Constraint is not a Risk: a Constraint is a known limit you must design within; a Risk is a possible outcome you might suffer. - -### Risk - -A **prospective negative outcome** with a likelihood and a severity. Risks attach to Decisions as part of the rationale for choosing among them. A Risk has a lifecycle: `open` (live, unmitigated), `mitigated` (an accepted Decision exists to reduce likelihood or severity), `realized` (it happened — usually triggers a Finding and possibly a retrospective Finding), or `accepted` (we knowingly proceed despite it). - ---- - -## Cross-cutting edge vocabulary - -These edges are **ubiquitous** — they live on every epistemic primitive. They are the type-agnostic semantic graph that lets a reader trace causality, evolution, and disagreement. - -| Edge | Description | -| -------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `derives_from` | Causal upstream — what this artifact is responding to or built on. (A Finding `derives_from` an Issue; a Decision `derives_from` Findings + Constraints; a retrospective Finding `derives_from` the original Decision.) | -| `supersedes` / `superseded_by` | Replaces an earlier artifact of the same primitive. The forward edge (`supersedes`) is canonical; `superseded_by` is a derived projection materialized to disk so any single record can answer "am I current?" in one access (see ADR-0004). | -| `invalidates` / `invalidated_by` | Stronger than `supersedes` — asserts the targeted artifact was _wrong_, not just outdated. Used primarily by retrospective Findings against shipped Decisions. Same forward-canonical / materialized-back-edge rule as `supersedes`. | - -**The edge vocabulary is permitted to grow** as dogfooding surfaces real omissions. Candidate additions to watch for: `refines` (soft non-replacing evolution), `contradicts` (explicit disagreement that doesn't yet rise to invalidation), `blocks` (an open Issue gating progress on another). Any addition must justify itself with a query the existing vocabulary cannot answer. - ---- - -## What is intentionally not modeled - -- **Session, Run, Plan, Artifact, Agent** as collections. These are operational provenance, captured as opaque-string frontmatter fields (`produced_in`, `created_by`, etc.) on the epistemic primitives above. Their durable log-grade record lives under `.flatbread/artifacts/` from `@flatbread/proof` runs. -- **Investigation** as a collection. An investigation is a Session-grouping of Findings (and possibly an Issue with `status: investigating`), not a noun in its own right. -- **Question** as a collection. Collapsed into `Issue{kind: question}` — the speech-act distinction does not warrant a separate primitive. -- **Proposal** as a collection. A Proposal is a `Decision{state: proposed}`. -- **Retrospective** as a collection. A Retrospective is a `Finding{kind: retrospective}`. -- **Branch** as a frontmatter field. Speculative exploration lives on git branches; cross-branch reasoning is preserved by promoting artifacts to the integration branch when an exploration closes (rejected or merged). - -These collapses may be revisited if real usage proves the host primitive cannot carry the missing semantics; Flatbread's `refs` model permits later splitting without ID breakage. diff --git a/docs/effort-graph/adr/0001-effort-graph-memory-location.md b/docs/effort-graph/adr/0001-effort-graph-memory-location.md deleted file mode 100644 index d3cfed52..00000000 --- a/docs/effort-graph/adr/0001-effort-graph-memory-location.md +++ /dev/null @@ -1,28 +0,0 @@ -# 0001 — Effort Graph memory location - -Status: Accepted - -## Context - -The Effort Graph stores epistemic artifacts (Issues, Findings, Decisions, Constraints, Risks) as flat markdown files indexed by Flatbread. Where those files live relative to the project repo determines whether reasoning branches with the code, and whether reasoning survives when a speculative branch is abandoned. - -Three modes were considered: - -- **A. In-repo, branch-coupled** — files under `/.flatbread-efforts/`. Reasoning branches with the code. Abandoned branch ⇒ abandoned reasoning. -- **B. In-repo with promote-on-close tooling** — same location, plus a helper that lifts artifacts onto the integration branch when an exploration closes. Requires a robust definition of "close a branch" across squash-merge, rebase-merge, PR-closed-unmerged, branch-deleted-then-resurrected. -- **C. Sibling repo / submodule** — memory has its own git history independent of project branches. Cross-branch reasoning loss disappears because memory commits to memory `main`. - -A cross-branch `ref` mechanism (point a ref at reasoning on another branch) was rejected: refs must resolve at index time against a single on-disk tree; resolving across branches requires either mutating the working tree or a per-branch index (a source-plugin rewrite), breaks the self-contained-artifact review story, and makes staleness invisible on rebase/force-push/delete. - -## Decision - -Ship **mode A as the default**. Make the schema and write API **identical across all three modes** so that **mode C works by repointing `path` in `flatbread.config.ts`** with no code change; document mode C as the supported alternative for teams that abandon exploration often or span multiple repos. - -**Defer mode B.** The promote-on-close ergonomics can be replaced day one by a documented `git cherry-pick ` + status-flip convention (`Decision: rejected_explored`, `Issue: wontfix`, `Finding: archived-from-exploration`). A `flatbread efforts promote-branch-artifacts` helper may land later. - -## Consequences - -- The default (A) accepts that reasoning on a never-merged branch is lost unless the team promotes it. This is acceptable because most efforts merge. -- Teams that care about preserving rejected exploration have two escape hatches without new platform code: cherry-pick promotion (still mode A) or mode C (config-only). -- The schema must not encode git internals (no `branch:`, no `git_ref:` field). Branching is emergent from git itself plus `Decision.state` and `derives_from` edges. -- Deferring B leaves a documented manual convention as the only path for promote-on-close until demand justifies the helper. diff --git a/docs/effort-graph/adr/0002-semantic-mutation-write-surface.md b/docs/effort-graph/adr/0002-semantic-mutation-write-surface.md deleted file mode 100644 index 6944e6f1..00000000 --- a/docs/effort-graph/adr/0002-semantic-mutation-write-surface.md +++ /dev/null @@ -1,28 +0,0 @@ -# 0002 — Semantic-mutation write surface - -Status: Accepted - -## Context - -Storing edges bidirectionally (`supersedes`/`superseded_by`, `invalidates`/`invalidated_by`) so that any single record can answer "am I current?" in one access (ADR-adjacent decision captured in `CONTEXT.md`) means a single conceptual change — "Decision X supersedes Decision A" — must update **two files atomically**: write X with `supersedes: [A]`, and patch A with `superseded_by: X`. A half-applied edge (X claims supersession, A does not know) is a corruption mode. - -Two stances for the write API the agent calls: - -- **(α) Narrow create-only surface.** Ship only `WriteIssue` / `WriteFinding` / `WriteDecision` / `WriteConstraint` / `WriteRisk` create mutations plus a generic frontmatter patch. The agent is responsible for issuing both halves of a bidirectional edge; the read shim warns about half-applied edges during a validation pass. -- **(β) Semantic-mutation surface.** Each schema-level concept (supersede, invalidate, resolve-issue, mitigate-risk) gets a dedicated typed mutation that atomically updates both ends of the edge. The platform owns bidirectional consistency; the agent does not. - -## Decision - -Adopt **β — a semantic-mutation write surface**. Mutations are validated and expanded at the level of **semantic edits**, not file edits. Each mutation: - -- has its own Zod schema and validation rules (e.g. `SupersedeDecision` requires the target Decision to exist in the index and not already be `superseded`), -- expands into the full set of file writes its semantics imply, and -- completes all of those writes or none (transaction semantics — partial-failure handling is specified separately). - -## Consequences - -- The agent-facing write contract is a set of named mutations, not "validate this YAML and write it." This is a larger design and implementation surface than α. -- Bidirectional-edge integrity is guaranteed by the platform, mirroring how `@flatbread/proof` put convergence-loop semantics in the DAG primitive rather than in agent prompts. -- A transaction/rollback primitive is now required (what happens when the second of two writes fails). This is a new capability for the write path. -- Flatbread has no mutation support today; β raises the question of whether mutations are exposed through GraphQL resolvers in core or through a standalone write library that does not touch the read layer. That fork is decided separately. -- The set of semantic mutations must be enumerated and kept small; each new mutation is API surface that must be versioned and taught to agents. diff --git a/docs/effort-graph/adr/0003-write-path-and-read-after-write-consistency.md b/docs/effort-graph/adr/0003-write-path-and-read-after-write-consistency.md deleted file mode 100644 index 9347eaf9..00000000 --- a/docs/effort-graph/adr/0003-write-path-and-read-after-write-consistency.md +++ /dev/null @@ -1,37 +0,0 @@ -# 0003 — Write-path architecture and read-after-write consistency - -Status: Accepted - -## Context - -ADR-0002 adopted a semantic-mutation write surface (β). Flatbread today is read-only and in-memory: `FlatbreadProvider` (`packages/core/src/providers/base.ts`) builds the GraphQL schema once at construction and exposes only `query()`. There is no `Mutation` type in `packages/core` and no write-back. The filesystem is the source of truth; the GraphQL graph is a projection built once. Live reload is explicitly unsupported today ([issue #65](https://github.com/FlatbreadLabs/flatbread/issues/65); `docs/positioning.md`). - -Two write-path architectures were considered: - -- **(1) Mutations inside core's GraphQL.** Add a `Mutation` type and resolvers that write files; the provider gains `mutate()`. Forces an in-process cache-coherence subsystem (every successful write must patch/invalidate the cached `EntryNode` graph or the next `query()` is stale) into core, which it does not have today. -- **(2) Standalone writer + read-only GraphQL.** A separate library owns the Zod mutation schemas, file expansion, and transaction semantics, and writes the source-of-truth files directly. The GraphQL read layer stays read-only and re-projects from disk. - -A separate question is the **read-after-write consistency model**: tool-call-boundary re-index (write returns touched ids/paths; next read re-indexes) vs instantaneous in-process read-after-write. - -## Decision - -Adopt **architecture (2): a standalone semantic writer; GraphQL stays read-only.** Mutation logic (validation, multi-file expansion, transactions) lives outside core's resolver layer, consistent with Flatbread's existing files-are-source-of-truth model. A thin GraphQL mutation _facade_ that delegates to the writer may land later, but GraphQL mutations are not the home of the logic. - -Adopt **live-reindex (watch mode) as a v1 dependency** for the consistency model, satisfied by implementing the **"Draft unified watch design"** already specified in `docs/local-dev-loop.md` (reload records → re-run ID/ref/cardinality validation → rebuild schema → hot-swap the live schema only if the new graph validates, else keep the prior schema and log). - -### Consistency contract (v1) - -The in-memory graph that reads are served from is rebuilt after a write; there is a brief window between "file saved" and "rebuild complete." During that window a read does not block and never returns a torn/half-built graph, but a _concurrent_ reader may observe the prior graph (briefly out of date, never wrong-shaped). The window widens with corpus size because rebuilds are full today. The contract that bounds this: - -1. **Read-your-own-writes (mandatory).** A mutation's return payload includes the written/changed artifacts, so the writing agent never re-queries to see its own write. This removes the window entirely for single-agent write-then-read, which is the dominant case. -2. **Default eventual, opt-in strict for concurrent readers (Q7.i → Option 1).** A second, concurrent reader may be a beat behind by default. When a workflow cannot tolerate this (e.g. a downstream `@flatbread/proof` task that must observe an upstream task's writes), the reader opts into a strong read that waits until the index has caught up to the depended-on write before answering. **This relaxed-by-default behavior, and how to request a strict read, must be called out in user-facing documentation when implemented.** -3. **Incremental reindex (v1 dependency, Q7.ii → Option A).** Because the writer knows exactly which files it touched, reindex re-reads only the changed files plus their ref-affected neighbors and patches the in-memory graph, instead of rebuilding the whole graph. This keeps the staleness window small regardless of corpus size (thousands+ of memories) and pairs with retiring the per-resolver `cloneDeep(contentNodesByCollection[...])` cost in `packages/core/src/generators/schema.ts`. - -## Consequences - -- The Effort Graph spec now has a **hard dependency on shipping the unified watch / live schema-swap seam** in Flatbread, **including incremental (changed-files-only) reindex**. This is scoped (a documented design contract and an existing codegen watch loop to factor from), not greenfield, but it is on the critical path and must be sequenced before the write story is considered done. -- Writes cannot destabilize core's read path, since transactional file-writing lives in a separate package. -- The writer must return the ids and file paths it touched, both for read-your-own-writes payloads and so the incremental reindex layer can refresh exactly the affected collections. -- A reader opting into a strict read needs a way to name the write generation it depends on; the writer must therefore expose a monotonic generation/version token in its return payload. -- Transaction/rollback semantics for multi-file mutations remain to be specified (see follow-up). -- If the watch seam slips, the fallback is tool-call-boundary re-index (read shim re-indexes affected collections per invocation); this is a degraded mode, not the target. diff --git a/docs/effort-graph/adr/0004-multi-file-honesty.md b/docs/effort-graph/adr/0004-multi-file-honesty.md deleted file mode 100644 index 5a956fef..00000000 --- a/docs/effort-graph/adr/0004-multi-file-honesty.md +++ /dev/null @@ -1,31 +0,0 @@ -# 0004 — Multi-file honesty: edge authority and the atomicity boundary - -Status: Accepted - -## Context - -ADR-0002 promised transaction semantics for multi-file mutations ("completes all writes or none") but deferred the mechanism; ADR-0003 deferred it again. Two sub-questions remained. - -**Edge authority.** `CONTEXT.md` stores `supersedes`/`superseded_by` and `invalidates`/`invalidated_by` bidirectionally so any single record can answer "am I current?" in one access — a human opening the raw file in a PR or grep, without an index. Naively that makes every edge write a two-file transaction, since both directions are authoritative and a half-applied edge is corruption. - -**Irreducibly multi-authoritative mutations.** Some mutations change authoritative state on several files even after edge authority is settled: `AcceptDecision(A)` sets A to `accepted` and its sibling `proposed` Decisions to `rejected` — each rejection is that file's own state, not derivable from A. Two atomicity boundaries were considered: **(a)** the git commit (writer stages and commits every mutation; a crash leaves a dirty tree that git recovers; bonus: free, high-fidelity evolution history of the reasoning graph — on-theme for a git-native product), and **(b)** a writer-level save-or-undo transaction independent of git. - -## Decision - -**Forward edges are canonical; back-edges are derived, materialized projections.** Only `supersedes` and `invalidates` carry authoritative intent. `superseded_by`/`invalidated_by` are mechanically determined projections that are still written to disk: the writer materializes both sides in the same save group, and the incremental reindexer validates and repairs any drift (hand edits, merge damage, crash residue). On conflict the forward edge wins; a reverse-only manual edit is non-authoritative and will be corrected, with a diagnostic naming the repaired file. "Derived" describes ownership and repair direction, not an optional disk cache — the raw file keeps its one-access honesty after convergence, preserving ADR-0001's self-contained-artifact review story. - -**Writer-level save-or-undo is the default atomicity boundary; git commit is opt-in history, never correctness.** The writer implements a small write-ahead journal: fsync an intent record (transaction id, paths, before-images, target generation), apply each file via same-directory temp-file + rename, write a durable committed marker, then trigger one journal-aware incremental reindex batch; the generation token is published only after reindex and schema swap succeed. Startup recovery is idempotent: uncommitted journal ⇒ roll back; committed journal ⇒ complete and reindex. A per-graph writer lock serializes concurrent writers (two agents in a proof DAG fail/retry rather than interleave). - -The opt-in git mode creates one isolated memory commit per successful mutation using a dedicated temporary index (never touching the user's staged work), and runs only **after** the journal transaction commits. If the commit fails, the mutation stands — report "committed locally, history commit unavailable"; never roll back a committed semantic mutation to preserve git symmetry. - -**The "free reasoning history" of (a) is recovered by session-level checkpoints, not per-mutation commits.** The writer records a session/transaction id and touched paths; `flatbread efforts checkpoint` creates one deliberate commit for a session's coherent reasoning evolution, and `@flatbread/proof` may checkpoint at successful DAG completion — never per node. Per-mutation commits in mode A would interleave dozens of tiny memory commits with code history, complicate rebase/squash, and let agents commit without being asked; mode C (sibling repo) softens those costs and may document a checkpoint-on-session profile, but does not change the default. - -## Consequences - -- Forward-authoritative edges shrink the true multi-authoritative set to lifecycle transitions (sibling rejection, issue resolution), keeping the journal small and simple. -- The transactional guarantee covers the writer API and the indexed read contract. Raw-disk readers can observe a partially applied group mid-rename; between a hand edit and reindex a file's back-edge may be stale. Both windows are bounded by the journal protocol and reindex repair, and are recorded as the honest limit of file-level atomicity. -- The reindexer must recognize active journals and defer affected paths, so the watch seam (ADR-0003) cannot validate a half-applied group. Reindex write-back of repaired back-edges uses the same journaled write path as mutations. -- Generation-token semantics tighten: a generation is not "committed" until back-edge projection repair and the live schema swap complete; strict reads (ADR-0003) wait on committed generations only. -- Merge resolution gets a deterministic rule: reconcile forward edges, regenerate reverse projections. Reverse-edge fields are canonicalized (sorted, dedicated frontmatter keys) to minimize conflict noise; a repair command must exist. -- `CONTEXT.md`'s edge-vocabulary language changes from "stored bidirectionally" to "forward edge canonical, back-edge materialized projection." -- Reversal criteria: 6.i reverses only if raw-file readers demonstrably require atomic cross-file edge visibility without writer/indexer involvement; 6.ii reverses only if dogfooding shows teams want every reasoning transition independently cherry-pickable and per-mutation commits stay low-friction under rebases and concurrent agents. diff --git a/docs/effort-graph/adr/0005-v1-semantic-mutation-enum.md b/docs/effort-graph/adr/0005-v1-semantic-mutation-enum.md deleted file mode 100644 index 5b4f1caf..00000000 --- a/docs/effort-graph/adr/0005-v1-semantic-mutation-enum.md +++ /dev/null @@ -1,39 +0,0 @@ -# 0005 — v1 semantic mutation enumeration - -Status: Accepted - -## Context - -ADR-0002 requires the set of semantic mutations to be enumerated and kept small — each mutation is versioned API surface that must be taught to agents. ADR-0004 settled edge authority (forward edges canonical) and the atomicity boundary (journaled save-or-undo), so each mutation's file-expansion footprint is now specifiable. - -## Decision - -Ship exactly thirteen mutations in v1. Each has its own Zod schema; validation runs against a committed generation of the index (targets must exist, state transitions must be legal, e.g. `Supersede` rejects an already-`superseded` target). - -**Effort lifecycle (2)** - -- `CreateEffort` — creates the anchor record. -- `SetEffortStatus` — `active | paused | completed | abandoned`. - -**Creation (5)** — one per epistemic primitive; single-file writes that accept forward edges at creation (`effort` is required; `derives_from`, `supersedes`, `invalidates` as applicable), expanding to back-edge materialization per ADR-0004: - -- `WriteIssue`, `WriteFinding`, `WriteDecision`, `WriteConstraint`, `WriteRisk`. - -**Edge retro-linking (2)** — for wiring records that already exist: - -- `Supersede(supersederId, targetId)` — same-primitive only (per `CONTEXT.md`); validates the target is not already superseded. -- `Invalidate(findingId, targetId)` — a Finding asserting a prior Finding or Decision was wrong. - -**Lifecycle transitions (4)** - -- `ResolveIssue(issueId, resolution: resolved | deferred | wontfix, resolvedBy: refs)` — resolvedBy cites the closing Decision and/or Findings. -- `AcceptDecision(decisionId, rejectSiblings = true)` — the irreducibly multi-authoritative mutation: sets the target `accepted` and sibling `proposed` Decisions under the same Effort to `rejected` with a back-pointer to the accepted Decision (the `CONTEXT.md` proposal-collapse contract). Runs inside one journal transaction. -- `MitigateRisk(riskId, decisionId)` — flips the Risk to `mitigated`, citing the accepted Decision. -- `SetRiskState(riskId, state: realized | accepted, evidence: refs)` — the remaining Risk transitions; `realized` should cite the triggering Finding. - -## Consequences - -- Deliberately absent from v1: generic frontmatter patch (reintroduces the α surface ADR-0002 rejected), delete/archive mutations (git is the undo story), `RejectDecision` as a standalone (covered by `AcceptDecision` sibling-reject; a lone rejection without an accepted alternative is `SetRiskState`-style scope creep until dogfooding demands it), and body-edit mutations (edit the markdown body directly; only frontmatter semantics are platform-owned). -- Freeform body edits and hand edits to forward edges remain legal — the reindexer validates and repairs projections per ADR-0004. The mutation surface is the supported path, not the only physical path. -- Every mutation returns the RYW payload from ADR-0003: written/changed artifacts, touched ids and paths, and the generation token. -- Adding a mutation later is additive API surface; removing or reshaping one is a breaking change subject to the major-migration process. diff --git a/docs/effort-graph/adr/0006-id-and-slug-strategy.md b/docs/effort-graph/adr/0006-id-and-slug-strategy.md deleted file mode 100644 index cf8169dc..00000000 --- a/docs/effort-graph/adr/0006-id-and-slug-strategy.md +++ /dev/null @@ -1,28 +0,0 @@ -# 0006 — ID and slug strategy - -Status: Accepted - -## Context - -Effort Graph records need refs that remain stable when files move, titles change, or multiple agents create records concurrently. Filenames are useful for review, but cannot safely serve as identity across branches. - -## Decision - -Use the ID format `---<16 lowercase Crockford-base32 random chars>`. Prefixes are permanent and reserved: `eff` (Effort), `iss` (Issue), `fnd` (Finding), `dec` (Decision), `con` (Constraint), and `rsk` (Risk). For example: `dec-use-standalone-writer--r6dt3vp7k4m9q2x8`. The slug is capped at 48 characters. - -The 80-bit random suffix makes concurrent creation by multiple agents, including agents on branches that later merge, effectively collision-free without coordination. The slug keeps refs reviewable in YAML frontmatter and PRs; the suffix makes same-title records safe; the prefix is a semantic type discriminator. ULID and UUIDv7 are rejected as visual noise. Recency comes only from the required explicit `created_at` field, never from an ID. - -`id` is required frontmatter and the sole identity. Filename and path never participate in identity, so renames and moves preserve refs. The writer defaults the filename to the ID (`dec-use-standalone-writer--r6dt3vp7k4m9q2x8.md`), but a mismatch is advisory, not invalid. - -Efforts also use hybrid IDs rather than pure human slugs. Pure slugs make similarly named Efforts created by uncoordinated agents a duplicate-ID merge failure. Efforts carry an editable `title` and may carry a unique human-facing `slug` alias for CLI lookup; renaming an Effort changes only its title or slug, never its ID or dependent refs. - -The directory layout is collection-first and organizational only: `.flatbread-efforts/efforts/`, `issues/`, `findings/`, `decisions/`, `constraints/`, and `risks/`. Membership is expressed by the `effort: eff-…` ref, not by nesting under an effort directory, which would make Effort renames operationally noisy. - -Refs remain scalar IDs. The mandatory kind prefix already encodes the target primitive type, so future union or multi-collection refs such as `invalidates: [dec-…, fnd-…]` can dispatch by prefix without rewriting stored frontmatter values. Reject `decision:dec-…` wrapper syntax: it duplicates the prefix and forces adapter work immediately. - -## Consequences - -- Union refs still require a core/config/schema migration because Flatbread `refs` config currently targets exactly one collection; this is a `flatbread-major-migration` concern, but requires no content migration. -- The six prefixes must be reserved and documented permanently. -- The promised GitHub issue for union/multi-collection refs remains unfiled (a repo search found none) and is a follow-up. -- The writer must validate ID shape and uniqueness at mutation time. diff --git a/docs/effort-graph/adr/0007-agent-read-shim.md b/docs/effort-graph/adr/0007-agent-read-shim.md deleted file mode 100644 index a2773f8d..00000000 --- a/docs/effort-graph/adr/0007-agent-read-shim.md +++ /dev/null @@ -1,41 +0,0 @@ -# 0007 — Agent read shim - -Status: Accepted - -## Context - -Agents querying the Effort Graph through MCP tools or the SDK must not dump whole reasoning graphs into their context window. The first planned agent-facing read surface is `blockingDecisions(effortId)` (roadmap). ADR-0003 defined generation tokens and opt-in strict reads. - -## Decision - -Every read tool returns a bounded envelope. The response is navigation; the rendered markdown file is the evidence. Reading it costs the agent one Read tool call, and it can be grepped: - -```json -{ - "summary": "2 results; 1 accepted, 1 proposed; complete", - "artifact_path": ".flatbread/effort-graph/read-cache/42/abc123.md", - "artifact_sha256": "…", - "served_generation": "…", - "consistency": { "mode": "eventual", "min_generation": null }, - "page": { "returned": 2, "has_more": false, "next_cursor": null }, - "hints": ["getRecord(\"dec-…\")"] -} -``` - -`summary` is deterministic and at most 160 tokens, covering result count, material states, and truncation. `hints` contains at most 10 executable follow-up query calls, not prose. Digests are written atomically to `.flatbread/effort-graph/read-cache//.md` (gitignored). Generation in the path prevents a stale projection from being served under a current-looking filename; identical query and generation reuse the cached file. On startup or `flatbread effort cache prune`, prune files older than 24 hours and enforce a 100 MiB ceiling, oldest first. - -Each digest contains a query header (query, served generation, result counts, completeness), an index of anchor links, per-record sections with selected frontmatter, normalized relation lists, and a bounded body excerpt (600 characters / 12 lines), plus an explicit edge table. Full bodies are never inlined; `getRecord(id)` renders a single-record digest. - -V1 caps are 25 primary records, one-hop relation expansion, 50 displayed edges, and a 64 KiB digest. At any cap the digest is marked incomplete and returns an opaque cursor. Scope and body length never silently expand. - -Every response echoes `served_generation`. Reads are eventual by default. Strict callers pass `{consistency: {mode: "strict", min_generation: ""}}`; the server waits for the projection to reach that generation or returns an explicit consistency error, never a stale result labelled strict. Generation tokens are opaque to clients. - -Use one shared read/query-render layer for MCP and CLI: query projection → bounded result model → digest renderer/cache writer. MCP is a thin transport adapter, so agent and human read semantics cannot diverge. - -V1 supports effort-scoped structured predicates, known relation traversal, pagination, record lookup, generation-aware reads, and `blockingDecisions(effortId)`. Defer semantic or embedding search, cross-effort traversal, arbitrary graph queries, rendering templates, durable query artifacts, and cross-machine cache sync. - -## Consequences - -- `blocking decision` needs a precise definition against the current edge vocabulary before implementation. -- Caps should be re-benchmarked in tokens against real Efforts. -- The cache directory must be added to gitignore when implemented. diff --git a/docs/effort-graph/adr/0008-committed-generation-bridge.md b/docs/effort-graph/adr/0008-committed-generation-bridge.md deleted file mode 100644 index 9007e4d3..00000000 --- a/docs/effort-graph/adr/0008-committed-generation-bridge.md +++ /dev/null @@ -1,50 +0,0 @@ -# ADR-0008: Committed-generation bridge - -Status: Accepted - -## Context - -ADR-0003 requires the writer to expose a monotonic generation token and an -opt-in strict read. ADR-0004 requires generation publication only after -reindex and live schema swap. Both halves existed (the journal protocol and -the live reloader), but the wire between them did not: `.journal/generation.json` -could advance while `LiveSchemaReloader.generation` never moved. Journal tokens -are durable per root; live generations are process-local across all content. -Equating their values is false after restart and whenever unrelated content -changes. - -## Decision - -The writer now uses `CommittedGenerationPublisher`, renamed from -`EffortGraphIndexer` to separate it from the plan-time `EffortGraphIndex`. -The live adapter maps relative paths to absolute paths, awaits -`notifyChanged({ source: 'writer' })`, and throws on rejected candidates. This -existing callback gate is the publish gate; the journal protocol is unchanged. -The bridge privately maps journal token J to live generation L, and strict -readers require both a live commit and durable publication, with an explicit -timeout escape hatch. - -A disk-backed, journal-aware `ReindexBarrier` defers watcher paths named by -uncommitted intents, releases on the committed marker or rollback removal, -never waits on the reloader or publisher, fails closed on malformed intents, -and bounds its wait so an orphaned transaction cannot stall the serialized -reindex queue forever. - -The Flatbread composition root activates on structural detection of the -complete six-entry `effortGraphContent(root)` shape (paths and refs), attaches -the bridge, attempts non-fatal boot recovery before listening, and exposes -`RunningGraphqlServer.effortGraph`. Flatbread takes a runtime workspace -dependency on effort-graph; effort-graph keeps core type-only, and core learns -no journal semantics. - -## Consequences - -A returned mutation token names a generation whose live schema commit already -completed, providing strict same-process read-your-writes. An out-of-process -writer publishes normally with the no-op publisher; the server watcher observes -its files after commit or rollback and those reads are EVENTUAL, not strict. -A dead external writer can defer intersecting watcher work until lease-safe -recovery; bounded barrier waits convert that to a logged rejected batch rather -than a stalled queue. The watcher may rebuild files the writer just published; -whole-file reads and serialization make that safe. This completes the -ADR-0003/0004 committed-generation contract. diff --git a/docs/glossary.md b/docs/glossary.md index 03b189ad..1acc4d97 100644 --- a/docs/glossary.md +++ b/docs/glossary.md @@ -1,10 +1,14 @@ # Flatbread glossary — relational content primitives -This page defines vocabulary for Flatbread’s **Git-native, flat-file relational content layer** for TypeScript apps. Flatbread turns files in your repo into a coherent **content graph** you can read from your application; it is **not** a hosted CMS, a full authoring product, or a general-purpose database. +This page defines words used by Flatbread. Flatbread turns files in your +repository into related data that your TypeScript app can read. It is **not** a +hosted CMS, a full writing product, or a general-purpose database. -**[GraphQL](https://graphql.org/)** is often the **default query interface** in typical setups, but it is one way to read the graph—not the product’s whole identity. +**[GraphQL](https://graphql.org/)** is often the default way to query the data, +but it is only one option. -See also: [Flatbread positioning](./positioning.md); [PMF decision rubric](./pmf-decision-rubric.md) (comparative criteria and agent-wedge signals). +See also: [Flatbread positioning](./positioning.md) and +[Comparing Flatbread with other tools](./pmf-decision-rubric.md). --- @@ -12,7 +16,7 @@ See also: [Flatbread positioning](./positioning.md); [PMF decision rubric](./pmf How many related items a field connects—whether a relation resolves to **one** related entry or **many** (for example, a single author versus a list of tag strings on a post). Cardinality shapes how the graph is exposed to your app (including generated GraphQL fields); it does **not** imply a SQL-style database engine. -Current relation cardinality rules are intentionally small: +Flatbread supports these relation shapes: - **One-to-one:** a `refs` field whose content value is a single ID (`author: 2a3e`) resolves to one related record. - **One-to-many:** a `refs` field whose content value is a list of IDs (`authors: [2a3e, 40s3]`) resolves to a list of related records. @@ -35,24 +39,29 @@ Current normalization rule: IDs may be **non-empty strings** or **finite numbers ### Query interface -The **API surface your application uses to read** the built content graph. In many projects today that surface is **GraphQL** (schema plus operations, often with codegen), meaning GraphQL is **an interface**, not the definition of Flatbread. Other ways to consume the same graph may exist in your stack alongside it. +The way your application reads Flatbread data. In many projects today, that is +**GraphQL**: a schema and operations, often with codegen. GraphQL is one way to +read the data, not the definition of Flatbread. ### Generated schema and operation types (GraphQL) When GraphQL is your **query interface**, the **generated GraphQL schema** describes how **collections** and fields are exposed at read time: list fields such as `allPosts` / `allAuthors` correspond to **collections**; nested selections follow **`refs`** (**relations**) and resolve to related **records**; scalar list fields that come from frontmatter (for example **`tags`** on a post) align with **Tag (facet)** in this glossary—not a **`Tag` collection** unless you add one. -**Generated TypeScript** from GraphQL document codegen (for example operation result types such as `GetPostsAuthorsAndTagsQuery`) types **that read path only**. It does not redefine Flatbread’s domain model: the **records** and **relations** still originate in repo files and config. A future non-GraphQL generated TypeScript read surface, if shipped, would be documented separately so it does not blur this boundary. +**Generated TypeScript** from GraphQL document codegen (for example, an +operation result type such as `GetPostsAuthorsAndTagsQuery`) types that way of +reading data only. Records and relations still come from repository files and +config. Any future generated TypeScript reader without GraphQL would be +documented separately. ### Record **One loaded item** in a collection: the structured result of reading a file (metadata, body, derived fields) that your app treats as a single unit. “Record” here means **a document-shaped object in memory**, not a row in a remote database. -### Record production +### Reading records -The transformation of collection-grouped source files into records, with core -stamping source context (`_path`, `_filename`) last. Record production excludes -validation and path ownership, which are handled by `validateRecords` and -`classifyPath`. +Flatbread reads source files into records and adds source details such as +`_path` and `_filename`. It then checks records and decides which configured +content path owns each file. ### Relation diff --git a/docs/isolated-schema-factory.md b/docs/isolated-schema-factory.md deleted file mode 100644 index 7c4cd22b..00000000 --- a/docs/isolated-schema-factory.md +++ /dev/null @@ -1,9 +0,0 @@ -# Isolated schema factory - -`generateSchema` now creates a new GraphQL composer and schema for every build. A returned schema therefore owns the resolver closures for that build’s content snapshot; a later build cannot replace its types, fields, or reads. - -We removed the config-keyed schema cache after applying the deletion test. Deleting it removed more complexity than it exposed: the cache keyed schemas from configuration while resolver closures captured content from the first build, producing stale reads after content changed. It also required validation-before-cache ordering and forced the Next.js watch demo to add a `__demoCacheBust` field solely to avoid a cache hit. - -The process-global composer was a separate shared-state issue. Per-build composers isolate type registration and make schemas independently usable; this is not evidence that a schema cache is needed. Because `graphql-compose-json` registers nested object types on the global composer even when given a composer instance, core now owns a small JSON→type parser that threads the per-build composer through every recursion while reproducing the upstream semantics and type naming exactly. AVA remains configured with concurrency `1` until a follow-up validates parallel safety across the complete test suite. - -We did not re-key the cache by content-snapshot identity. There is no profiling evidence that schema construction is the dominant cost in a relevant workload, while a content-keyed cache would add identity, eviction, and lifecycle complexity without demonstrated leverage. If profiling later proves a need, introduce a measured cache behind an explicit seam with correctness tests for changing content. diff --git a/docs/json-export.md b/docs/json-export.md index c93c6966..3f8b1d34 100644 --- a/docs/json-export.md +++ b/docs/json-export.md @@ -11,7 +11,7 @@ stable collection snapshots and `exportCollectionsAsCsv(configResult, options)` for flat collection views. They are currently API surfaces rather than CLI commands. -## Stability contract +## What stays the same - Selected collection names are sorted by Unicode codepoint order. - Records are sorted by normalized record ID. diff --git a/docs/local-dev-loop.md b/docs/local-dev-loop.md index 800c4bed..214aaba0 100644 --- a/docs/local-dev-loop.md +++ b/docs/local-dev-loop.md @@ -73,7 +73,7 @@ Expected behavior: edits received during an in-flight generation are queued for the next serialized batch. -## Unified watch coordinator contract +## How watch mode works The unified loop is started with: @@ -81,7 +81,7 @@ The unified loop is started with: pnpm exec flatbread start --watch -- next dev --turbopack ``` -The coordinator contract is: +Watch mode does the following: 1. A single coordinator classifies config, content, and document events, then serializes all rebuild and codegen phases. @@ -100,13 +100,14 @@ The coordinator contract is: ## Known limitations -- `flatbread start --watch` hot-swaps valid content/config generations; invalid - candidates leave the prior schema active. -- Unified watch mode is a long-running process; do not use it in CI or one-shot +- `flatbread start --watch` replaces the running schema after a valid content + or config change. If a change is invalid, it keeps the previous schema. +- Watch mode is a long-running process; do not use it in CI or one-shot scripts. - The Next.js example `pnpm dev` includes `--https` for local convenience, but the Flatbread GraphQL endpoint remains documented as HTTP on `5057`. In - headless environments prefer `pnpm exec flatbread start -- next dev --turbopack`. + headless environments prefer + `pnpm exec flatbread start --watch -- next dev --turbopack`. - Codegen failures are logged and do not undo a committed schema generation. - Watch mode requires a source plugin with `fetchPaths`; sources without it fail fast at startup. @@ -114,5 +115,5 @@ The coordinator contract is: - Port `5057` collisions are not resolved automatically; stop the old Flatbread process before starting another server. -Framework restarts remain explicit: Flatbread keeps the framework child process -running and does not attempt to restart or control its own refresh behavior. +Flatbread keeps the framework process running. It does not restart the +framework or control how the framework refreshes its pages. diff --git a/docs/pmf-decision-rubric.md b/docs/pmf-decision-rubric.md index 8bc9acfc..63fed5ba 100644 --- a/docs/pmf-decision-rubric.md +++ b/docs/pmf-decision-rubric.md @@ -1,79 +1,44 @@ -# PMF decision rubric — Flatbread vs adjacent workflows +# Comparing Flatbread with other tools -This page supports product and positioning decisions (e.g. [issue #144](https://github.com/FlatbreadLabs/flatbread/issues/144)) by comparing **Flatbread** to four **buyer-recognizable** workflow families. Use it to avoid mixing “Flatbread vs SQLite” with “Flatbread vs Notion” in the same breath without naming who you are selling to. +This page helps explain where Flatbread fits. It compares Flatbread with tools +that people often consider for the same job. -**Flatbread in one line:** Git-native **relational content** for TypeScript apps, **backed by flat files** in the repo. **[GraphQL](https://graphql.org/)** is a common **read interface** and codegen driver; it is **not** the whole product identity. The core artifact is the **modeled content graph** (collections, fields, relations, validation). +**Flatbread in one sentence:** it turns related content files in a TypeScript +project into data your app can read. GraphQL and codegen are common ways to +read that data, but they are not the product itself. ---- +## How to use this table -## How to read the matrix +Each column describes a group of tools, not every product in that group. -Each row names a **workflow category** a buyer might already use. Columns are **decision criteria** aligned with validation experiments and near-term PMF work. Cells summarize typical tradeoffs **for that category**, not a single vendor scorecard. +- **Strong** means the group usually handles the need with little extra work. +- **Medium** means it works, but needs some setup or care. +- **Weak** means the group is often a poor fit for that need. -**Legend (qualitative):** +## Comparison -- **Strong** — category usually excels here with little extra work. -- **Medium** — workable with discipline, tooling, or conventions; gaps are predictable. -- **Weak** — common pain or structural mismatch for this criterion in typical setups. -- **N/A** — criterion does not apply the same way (call out explicitly). +| Need | Flatbread | SQL database | Hosted CMS | Content files with build tools | Agent notes and project-memory tools | +| ------------------------------ | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------- | --------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------ | +| **Setup time** | **Medium** — install packages, add config and content paths, then optionally run GraphQL and codegen. | **Medium** — define a schema, connect a client, and manage migrations. | **Medium–High** — create an account, define content types, and set up API access. | **Medium** — set up a build plugin, content rules, and file layout. | **Varies** — each tool has its own folder and metadata rules. | +| **Type safety** | **Medium** — generated query types help, while config and raw file types still have limits. | **Strong** with a schema and a typed query library. | **Medium** — SDK and API types help, but drafts and flexible fields can weaken them. | **Strong** when the content schema is defined. | **Weak–Medium** — many systems store mostly free-form Markdown. | +| **Links between records** | **Strong** — `refs` connect collections, and nested reads follow those links. | **Strong** — joins and database constraints handle links. | **Medium–Strong** — reference fields are common, but deeper queries depend on the API. | **Medium** — links usually focus on site content. | **Weak** — links and search often replace structured references. | +| **Bad IDs and references** | **Strong for configured `refs`** — loading checks duplicate IDs, missing targets, and invalid reference values before Flatbread builds the schema. | **Strong** with constraints and transactions. | **Medium–Strong** — many CMSs block invalid publishing, but imports can still go wrong. | **Medium** — support differs by tool. | **Weak** — broken links and missing notes are common. | +| **Keeping your data** | **Strong** — files stay in Git, and the core API can create JSON and CSV exports. | **Strong** — dumps, backups, and SQL files are common. | **Medium** — exports and APIs vary by provider. | **Strong** — content stays in the repository. | **Strong** — notes are usually files, though moving their meaning to another tool can take work. | +| **Local development** | **Medium–Strong** — `flatbread start --watch` reloads valid content and config changes. Package code and app refresh behavior still need their own rebuild or restart. | **Strong** — local databases and migration tools are well established. | **Varies** — offline work and previews depend on the provider. | **Medium–Strong** — many tools rebuild when files change. | **Strong for saving files** — structured data updates need extra tooling. | +| **Reading data from an agent** | **Medium** — GraphQL and generated TypeScript can read related data; more direct agent tools are still developing. | **Strong** when the agent can use SQL safely. | **Medium** — HTTP APIs work, but authentication and rate limits add steps. | **Medium** — build-time access is simple; asking new questions at run time is harder. | **Weak–Medium** — search is common, but structured filtering is less common. | -Where Flatbread is **targeting** behavior that is not fully shipped yet (for example, first-class reference integrity at load time), the cell notes **current vs target** honestly. +## What to emphasize ---- - -## Comparative matrix - -| Criterion | Flatbread (relational flat files) | SQLite-style database workflows | Hosted / headless CMS workflows | Contentlayer-like content workflows | Agent artifact / Effort Graph workflows | -| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------- | -| **Setup time** | **Medium** — deps, config, content paths, optional GraphQL server/codegen; goal is ~10 minutes to a typed read for a `posts → authors → tags` starter. | **Medium** — schema/migrations, client, connection; very fast for experienced DB users. | **Medium–High** — account, schema/content model, API keys, webhooks; low ops if fully hosted. | **Medium** — build plugin, schemas, content layout; familiar to static/SSG teams. | **High variance** — conventions differ (`AGENTS.md`, `.handoff/`, vaults); **relational** effort graphs rarely work out of the box. | -| **Type safety** | **Medium (moving target)** — generated types help at the query boundary; config and raw content surfaces may still be looser until model-first typing lands end-to-end. | **Strong** with SQL builders/ORMs; schema is the source of truth. | **Medium** — SDK/OpenAPI/GraphQL types help; CMS field types and draft content can weaken guarantees. | **Strong** for defined content schemas; weaker if everything is MDX/adhoc. | **Weak–Medium** — lots of markdown prose; typed edges often missing unless encoded manually. | -| **Relation modeling** | **Strong intent** — `refs`, nested reads, filters over a content graph; cardinality must stay **documented** (implied behavior is an audit gap). | **Strong** — joins and constraints are the database’s job. | **Medium–Strong** — reference fields and UI; deeper graph queries depend on API. | **Medium** — relations exist but are optimized for site content, not arbitrary graphs. | **Weak** — links and search, not always **foreign-key-style** relations across tool boundaries. | -| **Reference integrity** | **Target: Strong / Today: uneven** — buyers expect missing refs, duplicate IDs, and bad shapes to **fail with clear diagnostics at load/validate**, not silent GraphQL `null` chains; full guarantee is **roadmap-critical**, not optional polish. | **Strong** with constraints and transactions (or app-enforced). | **Medium–Strong** — CMS often blocks bad publishes; export/sync paths can still drift. | **Medium** — build fails on schema errors; cross-file refs vary by stack. | **Weak** — broken links and orphan artifacts are common; validation is not standardized. | -| **Portability** | **Strong (raw files + Git)** — export story should include **JSON/CSV per collection** as a deliberate trust lever; contrast with ad-hoc “query and save.” | **Strong** via `dump`, backups, SQL files; binary portability has ops nuance. | **Medium** — APIs and export formats; lock-in depends on vendor. | **Strong** — content lives in repo; migration is folder moves + schema rewrites. | **Strong** — everything is files; **semantic** portability across tools is harder than byte portability. | -| **Local dev loop** | **Medium (honest)** — file-backed by nature; **reliable hot reload of content is not a pillar yet**; expect restarts or manual steps today where examples require a server/codegen refresh. | **Strong** — migrations + local DB; ORM dev UX mature. | **Variable** — offline editing depends on sync; preview stacks add latency. | **Medium–Strong** — dev servers often rebuild on file change; watch modes vary. | **Strong for “save file”** — weak for “typed graph updates everywhere” without extra tooling. | -| **Agent query ergonomics** | **Medium (directional)** — predicate-rich filters and nested reads suit **structured** agent queries; today’s path often touches **GraphQL** or codegen; **MCP / generated TS** as first-class agent surfaces is PMF leverage, not a nice-to-have. | **Strong** — SQL is the universal agent substrate when access is allowed. | **Medium** — HTTP APIs; auth and rate limits add friction for agents. | **Medium** — build-time access is easy; **runtime** ad-hoc queries less natural. | **Weak–Medium** today — keyword/vault MCP and search; **Effort Graph**-style queries want relational filters + integrity. | - ---- - -## Named contrasts (avoid category mixing) - -When writing positioning or issues, **name the buyer** and **one primary alternative**: - -| If the buyer is deciding against… | Lead with… | -| ---------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **SQLite / Postgres + app** | Versioned **content** and **review in Git** vs operational DB ergonomics; Flatbread is **not** replacing transactions or multi-writer DB semantics. | -| **Notion / Contentful / Sanity / etc.** | **Repo ownership** and **flat files** vs editorial APIs and hosted workflows; relations without standing up CMS infrastructure. | -| **Contentlayer / Velite / similar** | **Cross-collection references and graph reads** in TypeScript vs site-generation-first content pipelines. | -| **Handoff folders / vault MCP / memory tools** | **Typed relations and validation** over agent artifacts vs search-only or narrative-memory layouts—only after core integrity and watch/export bars are credible. | - ---- - -## Agent artifacts: secondary vertical vs primary wedge - -**Secondary vertical (default posture today)** fits when Flatbread’s **near-term bar** is still about relational **content** for apps—schemas, IDs, validation, exports, watch—and agent use inherits the same graph primitives without a bespoke **Effort Graph** product bundle. - -**Go signals for treating agent artifacts as primary wedge** - -- Reference integrity and diagnostics at **load/validate** are **trusted** on real repos (missing refs, duplicate IDs, invalid shapes fail loudly and actionably). -- **Model-first** onboarding reaches a **typed** query without demanding GraphQL literacy on day one; generated TypeScript and/or MCP cover the **agent-shaped** query path. -- **Watch** or an honest, low-friction loop makes **file edit → graph update** usable for harnesses that emit many small artifacts. -- At least one **reference layout** (for example `.agents/` or handoff-oriented trees) is documented as **indexed and validated** incremental adoption, not a migration cliff. -- Evaluation buyers consistently compare Flatbread to **vault/handoff/GCC** workflows—not only to CMS or Contentlayer—_and_ the graph answers queries like _blocking decisions for effort X with plan title_ without bespoke glue per repo. - -**No-go / hold signals (keep agent artifacts secondary)** - -- Broken links and duplicate IDs still **silently** degrade query results; buyers cannot distinguish “no data” from “bad graph.” -- The **only** documented happy path assumes a running **GraphQL** mental model for authors and agents. -- Local iteration still **requires full process restart** for ordinary content edits in the primary examples, with no credible watch/export story. -- Positioning drifts into **database replacement** or **hosted CMS** parity; agent narrative distracts from the core **TypeScript + Git relational content** promise. - -**Decision summary:** Agent artifacts are a **credible strategic option** because they amplify demand for the same integrity, typing, and query surfaces the core product needs; they become a **primary wedge** only when those properties are **proven in production-shaped workflows**, not declared in roadmap language alone. - ---- +| If someone is comparing Flatbread with… | Explain that Flatbread offers… | +| ------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------- | +| **SQLite or Postgres** | Content files that stay in Git and can be reviewed in pull requests. It does not replace database transactions or many-writer systems. | +| **Notion, Contentful, or Sanity** | Repository ownership and file-based content instead of a hosted editing service. | +| **Contentlayer, Velite, or similar tools** | References between collections and related reads in TypeScript. | +| **Handoff folders or note tools** | Structured links and validation when simple search across files is not enough. | ## Related docs -- [Flatbread positioning](./positioning.md) — canonical product framing. -- [Glossary](./glossary.md) — collections, relations, IDs, validation, query interfaces. -- [Flatbread Flow PMF Audit](../flatbread-flow-pmf-audit.md) — evidence-backed gaps and near-term experiments. -- [Agent artifact opportunity](../flatbread-agent-artifact-opportunity.md) — Effort Graph and adjacent landscape (deeper than this rubric). +- [Flatbread positioning](./positioning.md) +- [Glossary](./glossary.md) +- [Local development loop](./local-dev-loop.md) +- [Data ownership](./data-ownership.md) diff --git a/docs/positioning.md b/docs/positioning.md index 4512628a..2408eb25 100644 --- a/docs/positioning.md +++ b/docs/positioning.md @@ -1,21 +1,45 @@ # Flatbread positioning -Flatbread positions itself the same way across the repo; this page is a stable link target. For install and usage, see the [main README](../README.md). For vocabulary used across docs and config—**collections**, **relations**, **IDs**, and how a **query interface** fits in—see the [glossary](./glossary.md). For **buyer-aware comparisons** (SQLite-style workflows, CMSs, Contentlayer-like stacks, agent artifact graphs) across setup time, typing, integrity, and related criteria—plus **go / no-go** guidance for an agent-artifact wedge—see the [PMF decision rubric](./pmf-decision-rubric.md). For portability and exit paths, see [data ownership](./data-ownership.md). - -Turn flat files in Git into typed, relational content for your TypeScript app. The core artifact is an in-repo **content graph** (collections, records, **`refs`**). **Generated types plus [GraphQL](https://graphql.org/) operations** layer on top today as the most common **read interface** — they describe how many apps consume that graph at build/run time; they do not redefine what Flatbread **is**. - -**Flatbread** is a Git-native relational flat-file content layer for TypeScript apps. Your repo and filesystem are the source of truth; plugins (sources, transformers, and resolvers) extend how content is loaded and shaped. - -**Who it's for:** Teams shipping TypeScript sites, internal tools, and starters who want **versioned, reviewable content** and **relationships between entries**—without standing up a CMS database or giving up ownership of where content lives. - -**Non-goals:** - -- Not a hosted CMS, dashboard, or authoring UI: Flatbread is a library and local workflow, not a full content-management product you log into. -- Not a general-purpose GraphQL platform or a substitute for a general-purpose database (transactions, granular access control, and high-scale multi-writer workloads are out of scope). -- Reliable live reload of content while the dev server runs is [not a supported pillar yet](https://github.com/FlatbreadLabs/flatbread/issues/65); expect to restart to pick up file changes. - -**GraphQL:** In the default setup, GraphQL is a primary **interface** for reading an already-loaded content graph (`schema → operations → codegen`). Prefer thinking **files → model → typed read path** rather than treating GraphQL alone as Flatbread. For **traceability** from **backing files** (posts, authors, tag facets on posts) through **config** to generated schema and operation types—aligned with the [glossary](./glossary.md)—see the **Quickstart** and **Traceability** sections of [`packages/flatbread/README.md`](../packages/flatbread/README.md#quickstart-posts-authors-and-tags). - -**Portability and exit:** Raw files stay in Git, so content can be branched, reviewed, reverted, and migrated without asking a hosted CMS for a dump. JSON and CSV exports provide reviewable snapshots with normalized IDs and refs; GraphQL introspection and generated operation types preserve the read shapes your app used. The prototype generated read API is convenient inside Flatbread, while raw files, snapshots, GraphQL documents, and operation types are the durable exit surfaces. - -**Skimming from GraphQL-first experience:** Jump to **`refs` + relations** in [glossary](./glossary.md), then codegen and your app’s **`flatbread codegen`** docs — the relational layer is upstream of the queries you write. +For installation and usage, see the [main README](../README.md). For +definitions used in the docs and config, see the [glossary](./glossary.md). +To compare Flatbread with databases, CMSs, and other file-based tools, see +[Comparing Flatbread with other tools](./pmf-decision-rubric.md). For keeping +and moving your data, see [data ownership](./data-ownership.md). + +Turn files in Git into typed, related content for your TypeScript app. A +Flatbread project has collections, records, and `refs` that link records. +Generated types and [GraphQL](https://graphql.org/) operations are common ways +for an app to read that data; they do not define what Flatbread is. + +**Flatbread** reads content from your repository and file system. Plugins +control how it reads files and turns them into data. + +**Who it is for:** Teams building TypeScript sites, internal tools, and starter +projects that want versioned, reviewable content and links between entries +without setting up a CMS database. + +**What Flatbread does not do:** + +- It is not a hosted CMS, dashboard, or writing UI. +- It is not a general-purpose GraphQL platform or database. Transactions, + detailed access control, and many concurrent writers are outside its scope. +- [`flatbread start --watch`](./local-dev-loop.md) reloads valid content and + config changes. Changes to Flatbread packages still need their own rebuild or + restart. + +**GraphQL:** In the default setup, GraphQL reads data that Flatbread has already +loaded (`schema → operations → codegen`). Start with files and configuration, +then choose how your app reads the data. The +[Quickstart](../packages/flatbread/README.md#quickstart-posts-authors-and-tags) +shows posts, authors, and tags from files through generated types. + +**Keeping your data:** Raw files stay in Git, so you can branch, review, +revert, and move content without asking a hosted CMS for an export. JSON and +CSV exports make reviewable snapshots. GraphQL documents and generated +operation types show the read shapes your app used. The generated read API is a +convenience layer; files, snapshots, documents, and operation types are easier +to take to another tool. + +**If you already use GraphQL:** Read about [`refs` and relations](./glossary.md#relation), +then use your app's `flatbread codegen` documentation. The files and +configuration come before the queries you write. diff --git a/docs/proposals/proof-bounded-convergence-loops.md b/docs/proposals/proof-bounded-convergence-loops.md deleted file mode 100644 index b516cff1..00000000 --- a/docs/proposals/proof-bounded-convergence-loops.md +++ /dev/null @@ -1,206 +0,0 @@ -# Proposal: First-class bounded convergence loops in `@flatbread/proof` - -Status: Implementation -Tracking: branch `toeknee/proof-bounded-loop-cde0` stacked on PR #177 - -## Why - -The earlier discussion on cyclic vs acyclic task graphs (cursor agent -`bc-0ff9d782-…`, run `run-3cc886ad-…`) settled on the position: - -- The dependency graph should stay acyclic (DAG `depends_on` edges are - about static causality and parallelism — letting `depends_on` form a - cycle destroys readiness, skip, and rank semantics for no benefit). -- "Cyclic flow" is real and useful — research → critique → refine, fix - → test → fix-until-oracle, write → review → patch — but it is - bounded refinement, not a back-edge in the dependency graph. - -`proof` already implements the right shape, just at the CLI: - -- `--converge-on ` + `--max-iterations ` re-executes the - named task plus its transitive ancestors with the previous result - stitched into ancestor prompts as `extraContext`. -- The loop body parses `## Blockers` and `## High-severity findings` - and exits when both are empty, otherwise marks the convergence task - `BUDGET-EXCEEDED` after exhausting the iteration cap. - -Three real limitations: - -1. **Only one convergence task per run.** The CLI flag is a singleton; - you cannot stack a code-review loop and a docs-review loop in the - same DAG. -2. **The "what to re-execute" set is hardcoded** to "all transitive - ancestors". For wide DAGs you often want to re-run only a focused - subset (say, the implementation task and the reviewer, not the - six independent research tasks at the root). -3. **The convergence config lives outside the DAG JSON.** A DAG - author who wants reproducible convergence has to remember to pass - the right CLI flags every run, and tooling that emits DAGs has no - way to declare loop intent. - -This proposal adds a first-class, DAG-native bounded loop primitive -that subsumes the CLI flag without breaking it. - -## What - -Add an optional top-level `DAG.loops` array. Each entry is a -`DAGConvergenceLoop`: - -```jsonc -{ - "title": "implementation + adversarial review", - "loops": [ - { - "id": "review-loop", - "convergeOn": "review", - "maxIterations": 3, - "reexecute": { "kind": "ancestors" } - } - ], - "tasks": [ - /* … */ - ] -} -``` - -### Schema - -```ts -export type LoopReexecute = - | { kind: 'ancestors' } - | { kind: 'tasks'; tasks: string[] }; - -export interface DAGConvergenceLoop { - /** Stable id for canvas/log display. Defaults to `loop-${convergeOn}`. */ - id?: string; - /** Task whose `## Blockers` / `## High-severity findings` drive the loop. */ - convergeOn: string; - /** Iteration ceiling. Iteration 0 is the original main-rank run. */ - maxIterations: number; - /** What to re-execute on each iteration. Defaults to `{ kind: 'ancestors' }`. */ - reexecute?: LoopReexecute; -} -``` - -### Validation rules - -- `convergeOn` must be a known task id. -- `maxIterations` must be a positive integer. -- For `reexecute.kind === 'tasks'`: every entry must be a known task - id; the set must be a subset of `transitiveAncestors(convergeOn) ∪ {convergeOn}` - (re-executing tasks outside the convergence ancestor cone breaks - topological re-execution order — explicit error rather than silent - divergence). Every non-`convergeOn` task in the list must also bring along - its own transitive ancestors so the rerun subset is dependency-closed. -- Two loops cannot share the same `convergeOn` (avoids ambiguous - iteration counter ownership). -- `id` must be unique across loops after defaults are applied, so an explicit - `id: "loop-review"` cannot collide with another loop whose defaulted id would - also be `loop-review`. -- Two loops must have disjoint re-execution sets. If they overlap, the parser - rejects the DAG rather than letting a later loop silently invalidate an - earlier loop's converged outcome. -- The CLI `--converge-on` flag is mutually exclusive with `DAG.loops` - — supplying both is an error rather than a silent precedence rule. - -### Runner behavior - -The existing `runConvergenceLoop` function generalizes: - -- Caller supplies an explicit `reExecIds` set instead of computing - `transitiveAncestors(convergeOn) ∪ {convergeOn}` inside the loop. -- The CLI flag synthesizes a single-element `loops` array so the same - code path covers both entry points. -- Multiple loops run sequentially (in declaration order). Each loop's - `BUDGET-EXCEEDED` propagates to the run-level outcome the same way - the single CLI loop does today. -- Runner restarts resume from the persisted convergence iteration counter - instead of replaying iteration numbers from `1`. -- `dag.budget.maxIterations` continues to work and applies to each - loop independently — it is a hard cap on the per-loop iteration - counter, not a global counter. - -### What is intentionally out of scope (this PR) - -- Alternate loop stop predicates. The existing parser - (`extractConvergenceFindings` in `converge_loop.ts`) is the only stop rule; - richer predicates (oracle-pass, numeric thresholds) can land in follow-ups - once there is a concrete runtime need. -- Nested loops (loop inside loop). The flat array is enough for - every workflow we have today. -- Cross-loop coordination (loop A waits on loop B's iteration N). - Same reasoning — no real demand and would force a bigger - scheduler rewrite. - -## Backward compatibility - -- DAG JSON without `loops` keeps parsing untouched. -- The CLI flags `--converge-on` and `--max-iterations` keep working - end-to-end. Their behavior is reimplemented as a synthesized - single-element loops array. -- `DAG.budget.maxIterations` keeps the same meaning (per-loop hard - cap) and the same `BUDGET-EXCEEDED` terminal status. -- The `extractConvergenceFindings` parser, the `findings-dir` - sidecar contract, and the `extraContext` stitching format are - unchanged. Existing reviewer prompts keep working. - -## Test plan - -Focused AVA tests (`packages/proof/src/__tests__/loops.test.ts`, runnable via -`pnpm -F @flatbread/proof test` and included in root `pnpm test`): - -- `parseDAG` accepts `loops` with default `reexecute`. -- `parseDAG` rejects `convergeOn` referencing an unknown task id. -- `parseDAG` rejects two loops with the same `convergeOn`. -- `parseDAG` rejects two loops with the same materialized `id` (including - defaulted `loop-${convergeOn}` collisions). -- `parseDAG` rejects `reexecute.tasks` containing unknown ids or ids - outside the convergence ancestor cone, and rejects non-closed subsets. -- `parseDAG` rejects overlapping loop re-execution sets. -- `parseDAG` rejects non-positive `maxIterations`. -- `resolveLoopReexecuteIds` returns the right id set for both - `'ancestors'` and explicit `tasks` modes. -- Re-execution rank filtering preserves topological order for the - filtered subset. - -Backward-compat smoke: - -- A DAG with no `loops` and no CLI `--converge-on` runs zero - convergence iterations (existing behavior). -- A DAG with no `loops` plus CLI `--converge-on` synthesizes one - loop and runs it. -- A DAG with `loops` plus CLI `--converge-on` errors at startup. - -Self-review via `/proof` is the user-facing acceptance test; this -PR's test plan above is what gates the merge. Contributor-facing command: - -```bash -pnpm -F @flatbread/proof test -``` - -## Migration - -No code changes required for existing DAG JSON. Authors who want -DAG-native convergence can move from: - -```bash -proof --dag run.json --converge-on review --max-iterations 3 -``` - -…to: - -```jsonc -// run.json -{ - "loops": [{ "convergeOn": "review", "maxIterations": 3 }], - "tasks": [ - /* … */ - ] -} -``` - -```bash -proof --dag run.json -``` - -The CLI form stays valid for ad-hoc runs. diff --git a/docs/proposals/proof-output-retention-judge.md b/docs/proposals/proof-output-retention-judge.md deleted file mode 100644 index 86116fa3..00000000 --- a/docs/proposals/proof-output-retention-judge.md +++ /dev/null @@ -1,92 +0,0 @@ -# Adversarial review: `proof-output-retention-plan.md` - -Author: Opus 4.7 (replaces the prior judge draft). -Scope: read-only review of the plan against the actual code in `packages/proof/src/**`. - -This review is grounded in spot checks of `run_dag.ts`, `canvas_writer.ts`, `findings_sidecar.ts`, `converge_loop.ts`, `oracle_task.ts`, `self_hosting.ts`, `pause_task.ts`, and `dag.ts` at HEAD. Every finding cites the file/symbol it relies on. - ---- - -## Blockers - -None. - -The plan is implementable as written. The objections below are correctness and scope risks that should reshape Phase 1–3, not gates that prevent the work from starting. - ---- - -## High-severity findings - -1. **The forward-compatibility claim for `version: 2` `PersistedRunState` is false as stated.** Phase 3 says: "the `version: 1` reader is kept long enough that an older runner can still read newer files via the legacy fallback (though the converse is not guaranteed and is documented)." `self_hosting.ts` `readPersistedRunState` is hard-gated on `obj.version !== 1` and throws `Resume state ${path} has unsupported version.` Any already-shipped runner reading a `version: 2` file will fail at startup — it cannot fall back to anything because the payload schema changes shape (`state.tasks[].resultText` becomes optional, supplanted by `transcript: { kind, ... }`). This matters for the supervisor + `EXIT_RUNNER_RESTART` (75) flow: a mid-rolling-upgrade scenario where an older runner reads a state file written by a newer process will hard-fail instead of resume. The plan needs to either (a) drop the forward-compat sentence, document v2 as a one-way cutover gated by a release boundary, and add an explicit migration test that the _new_ runner reads v1 (already covered) but not the inverse; or (b) ship a Phase 0 / point-release patch that loosens the existing v1 reader to tolerate unknown future versions and degrade gracefully. (a) is the honest path; (b) requires lead time the plan does not budget. - -2. **`BUDGET-EXCEEDED` parents do not propagate to dependent children, and Phase 2 turns that latent gap into a hot path.** In `run_dag.ts`, `runOne` skips a task only when an upstream is `'ERROR'`: - - ```ts - const failedDeps = task.depends_on.filter((depId) => { - const dep = stateById.get(depId); - return dep !== undefined && dep.status === 'ERROR'; - }); - ``` - - `BUDGET-EXCEEDED` is intentionally not in that set (see the comment in `markRunTerminated`: "BUDGET-EXCEEDED is a terminal status the convergence loop sets explicitly; do not stomp it into a generic ERROR on shutdown"). Today this is mostly invisible because the only producer of `BUDGET-EXCEEDED` is the convergence task at the tail of a loop — it has no DAG-typed children. Phase 2 of the plan introduces a new producer: `runTask`'s pre-dispatch overflow check, which can mark _any_ mid-DAG task `BUDGET-EXCEEDED` because its stitched prompt exceeds `outputPolicy.maxPromptChars`. Once that lands, every task downstream of the budget-exceeded one will: - - - be considered runnable by `runOne` (no `failedDeps` hit), and - - call `buildUpstreamContext` against a parent whose `resultText` is undefined (the task never dispatched), so the dependency renders as `'(no output)'` and the child runs against a silently-broken DAG. - - The plan should either extend `failedDeps` (and `isResumeTerminalStatus`) to treat `'BUDGET-EXCEEDED'` as a skip-trigger for descendants, or document a different propagation policy. Either way, this is a Phase 2 prerequisite, not a follow-up. - -3. **The new `--restart-on-runner-change` precondition is a silent compatibility break for today's supervisor.** The constraint matrix under "Resume / supervisor" adds: "The supervisor and runner both refuse to start if `--restart-on-runner-change` is set without either a pinned `--full-output-dir` or a non-default `--max-in-memory-output-bytes`." That is a behavior change. The current `parseArgs` defaults `statePath` to `.proof/run-state.json` _because_ `--restart-on-runner-change` was set (see the `restartOnRunnerChange ? resumeState ?? '.proof/run-state.json'` branch), with no requirement to pass `--full-output-dir`. Existing supervisor invocations that rely on the timestamped default artifact directory (`.flatbread/artifacts/dag--/` per `defaultArtifactsDir`) would start failing once Phase 3 lands. The plan calls this out as a docs change ("New guidance in `README.md`…") but treats the refusal as load-bearing for the resume-input-parity acceptance criterion. The refusal should be either downgraded to a warning + automatic fallback to gzipped inline transcripts (`kind: 'inline'`), or staged behind an explicit opt-in flag with a deprecation window. As written, "refuse to start" plus "supervisor pin advice" plus "default still timestamped" is internally inconsistent and will trip real users on the first deploy. - -4. **The plan misdescribes how today's truncation banner travels through `buildUpstreamContext`, and that mistake hides a real correctness gap.** Ground-truth section 2 says: "the tail buffer's own `[...truncated N earlier chars...]` banner does travel into the upstream block when present, so the child does see a signal that the parent was capped." The code in `truncateUpstreamSnippet` / `parseUpstreamSections` says otherwise for the most common case. `BoundedTextBuffer.render()` returns `[...truncated ${droppedChars} earlier chars...]\n${data}`, i.e. the banner is the _first_ line, _before_ `data`. When the parent is a reviewer-style task whose output begins with `## Blockers` / `## High-severity findings` (the very pattern `--converge-on` keys on), `parseUpstreamSections` reaches the heading on the second line, and the banner falls into the documented "Lines before the first `## ` heading are intentionally dropped" branch. The child therefore sees a section-aware excerpt with no truncation banner at all — a child prompt that today silently conceals the fact the parent's prefix was lost. This is an _existing_ correctness bug (and an instance of the plan's thesis), but the plan's "Ground truth" describes it the wrong way and the Phase 1 design relies on the wrong model: it routes the visible-banner fix only through the _new_ `applyUpstreamPolicy` boundary, not through preserving the `BoundedTextBuffer` banner inside `parseUpstreamSections` for the legacy default. Phase 1 acceptance criterion 7 ("a grep-for-`…` test fails the legacy silent-ellipsis path") will pass even when the new `summarize` default still drops the banner inside section-aware mode, because the banner survives only on the `truncate(text, cap)` fall-through. The plan should (a) correct the ground-truth description, and (b) require the section-aware path to preserve any `[...truncated N earlier chars...]` preamble as a synthetic kept section even when it appears before the first `## ` heading. - ---- - -## Medium-severity findings - -1. **The post-loop final-findings extraction in `runConvergenceLoop` is not in Phase 1's deliverables.** Phase 1 lists "`runConvergenceLoop` updated so **both** `extractConvergenceFindings` and `buildConvergenceContext` read from the transcript store." There is a third call site in the same function: after the iteration loop exits, `runConvergenceLoop` re-extracts `extractConvergenceFindings(finalSidecarText ?? convergeTs.resultText)` to decide whether to flip the convergence task to `BUDGET-EXCEEDED`. If the transcript-store rewire skips this call, a long final-iteration reviewer whose `## Blockers` lands past the legacy 4000-char cap will silently terminate as `FINISHED`-then-clean even though the underlying evidence still has blockers. Phase 1's criterion 2 covers detection _during_ the loop; it does not cover the post-loop check. - -2. **The sidecar fallback is lossy in ways that defeat its use as a `buildConvergenceContext` source.** Phase 1's source-of-truth table lists "the `--findings-dir` sidecar continues to be written from the same source so external tooling has a stable JSON form, but it is no longer required for parser correctness inside the runner — it is required only when the **runner process** has restarted." Phase 3 then folds the sidecar into the cross-process fallback chain for the transcript store ("falling back to sidecar (across-process boundary, e.g. resume), falling back to `resultText` (legacy resume)"). The sidecar is not a lossless mirror of the transcript: `findings_sidecar.ts` `parseSections` keys on `## ` headings and discards every line before the first heading, then trims trailing whitespace per section (`out[currentHeading] = currentLines.join('\n').trim()`); and `readFindingsSidecarAsText` reconstructs `## Heading\n${body}` joined by `\n\n`, which is not byte-identical to the original transcript even when the original was perfectly heading-shaped. Routing `buildConvergenceContext` through this fallback after a resume will produce ancestor prompts that differ from the in-process path on whitespace and on any preamble content (including, per high-sev #4, the truncation banner). Phase 3's "input parity across restart" acceptance criterion will only catch this if the parity test explicitly forces the sidecar fallback path on the resumed process. - -3. **The Phase 1 canvas-size envelope test will be flaky on benign DAGs.** Phase 1 criterion 6 asserts: "for a fixture of 5 tasks × 12 000-char outputs, the generated `.canvas.tsx` file size remains below `5 * CANVAS_DISPLAY_CAP + 64 KiB`." The static template hardcoded in `canvas_writer.ts` (`HEADER + BODY` template strings around line 209 onward) is already comfortably over 20 KiB on its own, and `JSON.stringify(state, null, 2)` embeds every `subtask_prompt`, `depends_on` array, model selection, and oracle command for every task — none of which are bounded by `STREAM_CAP` today and none of which are bounded by the new `CANVAS_DISPLAY_CAP`. A DAG with 5 tasks whose `subtask_prompt` is, say, 8 KiB each (entirely realistic for the instruction-heavy tasks the proof package targets) blows past `5 * 4000 + 64 KiB` purely from prompt content. Either rebase the envelope on `O(static_template + Σ subtask_prompt + Σ CANVAS_DISPLAY_CAP) + slack`, or add a separate `CANVAS_PROMPT_DISPLAY_CAP` for `subtask_prompt` and document it. - -4. **Phase 4's "oracle in `--no-artifacts` mode marks `BUDGET-EXCEEDED` on in-memory overflow" is not implementable without changing `oracle_task.ts` `execShell`.** `execShell` accumulates `stdout` and `stderr` as plain `string` concatenations on each `'data'` event, with no size accounting and no cancellation hook. By the time `runOracleTask` finishes and would consult the `--max-in-memory-output-bytes` ceiling, the memory was already consumed. To honor the ceiling pre-emptively, `execShell` must (a) track running byte counts per stream, and (b) `kill('SIGTERM')`/escalate on threshold crossing. The plan does not budget that change. The acceptance criterion "the oracle is marked BUDGET-EXCEEDED" can only be honored _after_ the oracle command completed — i.e. after the OOM risk it was meant to prevent already happened. - -5. **Per-chunk `appendFile` to `${taskId}.stream.txt` will dominate the stream loop on real workloads.** Phase 1's "Append-only mirror to `${fullOutputAbsoluteDir}/${taskId}.stream.txt`" runs from inside the `runTask` `while (true)` stream loop, where today the only persistent work is `buffer.append(block.text)`, `fullStreamChunks.push(block.text)`, and a throttled `publishIfDue`. The SDK assistant stream emits text blocks at sub-millisecond intervals during long generations; per-block `fs.appendFile` invocations are `open + write + close` per call (three syscalls per text block, plus a per-call `Promise` allocation in the hot path). On a 100k-block stream that is 300k syscalls plus 100k `await` points dropping into the event loop. The plan's risk-table mitigation ("best-effort … failures are logged and the task is flagged but not aborted") covers correctness on write failure; it does not cover throughput. The deliverable should specify a coalesced strategy — open a `FileHandle` once, accumulate chunks behind the same `streamPublishMs` throttle as `publishIfDue`, and write on flush — and the design must reckon with what happens to the in-memory tail when the file write is slower than the stream. - -6. **`writeFindingsSidecar`'s call site does not have access to the new transcript store.** Phase 1 says: "`findings_sidecar.ts` `writeFindingsSidecar` reads from the transcript store and emits `sections` keyed identically to today." `writeFindingsSidecar` is currently called from the `dispatchTask` closure in `run_dag.ts`'s `main()` with the signature `writeFindingsSidecar(findingsAbsoluteDir, ts)` — it gets a `TaskState`, not a transcript store handle. The plan does not specify the access pattern (DI parameter, module-level singleton, or pass-through at call sites) and the choice has knock-on effects on Phase 3 (a singleton must be re-hydrated before `runConvergenceLoop` runs in a resumed process). This needs to land in Phase 0's "decision document", not be deferred to Phase 1 implementation. - -7. **`applyUpstreamPolicy`'s `summarize` mode silently drops freeform preamble — a regression that becomes more visible once transcripts are no longer pre-truncated.** `parseUpstreamSections` documents the pre-heading drop: "Lines before the first `## ` heading are intentionally dropped — the section-aware truncate only applies to outputs that lead with a heading; freeform preludes fall through to `truncate()`." Today this is masked because `STREAM_CAP=4000` already discarded most of the prefix. After Phase 1, the parent's transcript is the full stream, so a parent that emits (say) 6 KiB of preamble before its first `## Heading` will silently drop all 6 KiB on the section-aware path, with no banner accounting for it. The plan's "visible counted banner" mitigates the _cap-driven_ drop but not this _structural_ drop. Either preserve the preamble as a synthetic "(preamble)" section, or change the heuristic so a preamble larger than N% of the cap routes to the slice fall-through. - -8. **The plan does not budget for the test infrastructure cost of the stitched-prompt assertions.** Phase 1 acceptance criteria 1, 3, 4, and 7 all assert against the stitched prompt string passed to `agent.send(stitched)` in `runTask`. The current bounded-loop suite under `packages/proof/src/__tests__/loops.test.ts` exercises pure helpers (`extractConvergenceFindings`, etc.) — there is no fake `Agent.create` in the repo today, no scripted async-iterator harness, and no recorded fixture format for stream chunks. Phase 1's deliverables should include a "harness" deliverable (a fake `@cursor/sdk` `Agent` surface plus a fixture-driven `RunnerTaskRun` factory) so the acceptance criteria are wired to runnable tests rather than aspirations. - -9. **`enforceTokenBudget` is not the right place to absorb the new prompt overflow path.** Phase 2 says "this reuses the existing `BUDGET-EXCEEDED` exit path". The existing path runs after every rank's `Promise.all` completes (`enforceTokenBudget(state, dag.budget)`), throws `BudgetExceededError`, and unwinds. The new prompt-overflow path runs _inside_ `runTask` _before_ `agent.send`, must mark a single task `BUDGET-EXCEEDED` without throwing (so siblings continue), and depends on per-task accounting that has nothing to do with `maxTokensTotal`. The plan's "reuses the existing path" framing collapses two distinct control-flow lanes. Either the implementation needs a new per-task short-circuit that mirrors the convergence loop's `convergeTs.status = 'BUDGET-EXCEEDED'` pattern, or the plan should commit to a separate `EXIT_PROMPT_BUDGET_EXCEEDED` and own the precedence rules vs. token-budget exit code 4. - -10. **The Phase 0 consumer inventory is missing the `_index.md` / `persistTaskMarkdownFile` writer chain.** Phase 0 lists every read site of `TaskState.resultText`, but `persistTaskMarkdownFile` is called from `runOne` for `kind: 'pause'` / `kind: 'oracle'` with `ts.resultText ?? ''` _and_ from `runTask`'s `finally` for `kind: 'task'` with `fullStreamChunks.join('')` — i.e. there are _two_ artifact-writing branches today, each reading from a different source. After Phase 1 retires `fullStreamChunks`, the call-site contract for `persistTaskMarkdownFile` must be unified, and the inventory should name both branches up front so the unification is intentional rather than incidental. - ---- - -## Recommended adjustments - -1. **Drop the v1↔v2 "older runner can read newer files" sentence in Phase 3 and replace it with an explicit cutover policy.** Add a one-line acceptance criterion that the v1 reader is still strict about `version: 1` (so future readers are forced to broaden the check intentionally), and document the supervisor-restart implication for rolling deploys. - -2. **Make `BUDGET-EXCEEDED` propagate to dependents before Phase 2 lands.** Concretely: extend `failedDeps` in `runOne` to include `'BUDGET-EXCEEDED'` upstreams, and add a Phase 1 acceptance criterion asserting that a child of a `BUDGET-EXCEEDED` parent ends in `'ERROR'` with a "skipped: upstream budget exceeded" message. This is also the right time to revisit `isResumeTerminalStatus` (already includes `'BUDGET-EXCEEDED'`) so resume semantics line up with the new skip semantics. - -3. **Reframe the `--restart-on-runner-change` change as additive.** Default behavior should remain "start regardless"; add a new opt-in (e.g. `--require-pinned-artifacts-on-restart`, or fold it into a future `--strict` flag). The plan can keep its supervisor pin _recommendation_ without making the absence of `--full-output-dir` a startup failure. - -4. **Correct the "Ground truth" description of the truncation banner's path through `buildUpstreamContext`, and add a Phase 1 deliverable that preserves the banner as a synthetic kept section** (or routes section-aware truncation to a code path that prepends the banner unconditionally). This is a one-symbol fix in `parseUpstreamSections` / `renderUpstreamSections` with outsized signal value for the rest of the plan. - -5. **Add the post-loop final-findings extraction to Phase 1's `runConvergenceLoop` rewire.** Either name it explicitly in the deliverables list, or add an acceptance criterion that the `BUDGET-EXCEEDED` decision is made against the transcript store, not `convergeTs.resultText`. - -6. **Either commit to a lossless on-disk transcript format for cross-process fallback, or restrict the sidecar fallback to `extractConvergenceFindings` only.** The Phase 3 input-parity test should explicitly include a leg that forces the resumed process to use the sidecar (not the in-memory reconstruction) and asserts byte-identical stitched prompts. If that leg cannot be made to pass without a separate raw-transcript file, add the raw-transcript file (e.g. `${taskId}.transcript.txt`) to the Phase 1 artifact set and have the sidecar continue to be a derived index, not a fallback source. - -7. **Replace the canvas-size envelope formula with one that includes `subtask_prompt` mass.** A safe formulation: `static_template_bytes + Σ_t (CANVAS_DISPLAY_CAP + |subtask_prompt_t| + per_task_metadata_overhead) + slack`. Pick `slack` empirically against a real DAG fixture and fail the test only on regressions vs. that envelope. - -8. **Specify the stream-file write strategy explicitly.** A concrete sketch the plan can adopt: open a `FileHandle` per task at first chunk; accumulate chunks in a small in-memory buffer; flush on the same `streamPublishMs` cadence as `publishIfDue`; close in the `runTask` `finally`. Document the failure mode when the FS cannot keep up (in-memory buffer grows; trip `--max-in-memory-output-bytes` like any other store). - -9. **Push the oracle full-evidence work into Phase 4 _with_ the `execShell` streaming-cap change, or split it into two phases.** A defensible split: Phase 4a writes `${taskId}.stdout.log` / `${taskId}.stderr.log` from the existing `outcome.stdout`/`stderr` strings (preserves today's memory bound, gains forensic file). Phase 4b adds streaming size accounting in `execShell` with `SIGTERM` on threshold crossing, _only then_ honors `--max-in-memory-output-bytes` for oracles. Without 4b, the `--no-artifacts` ceiling is misleading. - -10. **Disambiguate the prompt-overflow `BUDGET-EXCEEDED` lane from the token-budget lane.** Either name the new exit code (e.g. `EXIT_PROMPT_BUDGET_EXCEEDED = 5`) and add precedence rules to the main-run tally, or explicitly fold the new lane into the existing `EXIT_BUDGET_EXCEEDED = 4` and document the wrapper-script impact. - -11. **Move the `writeFindingsSidecar` access-pattern decision into Phase 0.** Pick singleton vs. DI; do not defer to implementation time. Whichever choice you make has direct implications for Phase 3 resume reconstruction. - -12. **Add the `_index.md` / `persistTaskMarkdownFile` branches to the Phase 0 consumer inventory.** They are a third execution-plane consumer alongside `buildUpstreamContext`, `extractConvergenceFindings`, `buildConvergenceContext`, the sidecar, and the persisted state. The inventory is otherwise complete. diff --git a/docs/proposals/proof-output-retention-plan.md b/docs/proposals/proof-output-retention-plan.md deleted file mode 100644 index 54847d01..00000000 --- a/docs/proposals/proof-output-retention-plan.md +++ /dev/null @@ -1,419 +0,0 @@ -# Proposal: Proof rank/task output retention — fix execution fidelity without breaking the canvas - -Status: Planning — replaces the prior draft. -Author: Opus 4.7 -Scope: `@flatbread/proof` (`packages/proof`) - ---- - -## GitHub issue tracking - -These follow-ups were created from the Cursor cloud-agent review of PR -[#199](https://github.com/FlatbreadLabs/flatbread/pull/199). The issue bodies -link back to this proposal, the adversarial review, and the judge artifact. - -- [#200 — Phase 0/1 follow-up hardening](https://github.com/FlatbreadLabs/flatbread/issues/200) -- [#201 — Phase 2: upstream prompt policy and budget preflight](https://github.com/FlatbreadLabs/flatbread/issues/201) -- [#202 — Phase 3: resume, supervisor, and disk ergonomics](https://github.com/FlatbreadLabs/flatbread/issues/202) -- [#203 — Phase 4: oracle evidence alignment](https://github.com/FlatbreadLabs/flatbread/issues/203) -- [#204 — Phase 5: documentation and skill refresh](https://github.com/FlatbreadLabs/flatbread/issues/204) - ---- - -## TL;DR - -Today, the runner in `packages/proof/src/run_dag.ts` stores each task's assistant output through a `BoundedTextBuffer(STREAM_CAP=4000)` that **drops the leading characters** as the stream grows past the cap. That bounded string is the **only** copy used for: the canvas `STATE` literal, the parent context stitched into child prompts (capped a second time at `UPSTREAM_SNIPPET_CAP=2000` by `buildUpstreamContext` / `truncateUpstreamSnippet`), the `--findings-dir` JSON sidecar payload, the convergence `extraContext` re-injection, and the persisted `--state-path` snapshot. Only the per-task `${taskId}.md` artifact, written from a separate uncapped `fullStreamChunks` array in `runTask`, ever retains the complete stream — and only when artifacts are enabled. - -This is the "rank/task output truncation limitation" we are paying down. The fix is **not** "remove all caps". The fix is to **split the execution plane from the display plane** so that: - -- Decisions the runner makes on a user's behalf — what to put in a child's prompt, what counts as a `## Blockers` finding, what convergence re-runs see, what resume hands back to a relaunched process — read from an **execution-authoritative full transcript** that the runner persists for the duration of the run. -- The `.canvas.tsx` file the IDE hot-recompiles, the persisted state JSON the supervisor reloads, and the `extraContext` that lands inside a model prompt each consume **explicit, named excerpts** of that transcript with documented size policies and visible truncation banners. No layer is permitted to feed a downstream consumer a silently-truncated string and pretend the rest never existed. - -The work is staged across five phases. Phase 0 just nails the contracts. Phase 1 (the load-bearing one) rebuilds the runner's per-task storage and rewires the **four** existing consumers of `ts.resultText` so that "complete" sources stay complete and "bounded" sources are explicitly bounded with banners. Phase 2 handles upstream-prompt budgets honestly. Phase 3 covers resume/supervisor schema growth. Phase 4 aligns oracle evidence. Phase 5 refreshes docs. - ---- - -## Ground truth (what the code actually does today) - -These are the load-bearing facts the rest of this document is built on. Every claim points to a file/symbol in the package. - -### 1. Two parallel buffers in `runTask`, only one is bounded - -`packages/proof/src/run_dag.ts` `runTask` (the `kind: 'task'` path) maintains both: - -- `const buffer = new BoundedTextBuffer(STREAM_CAP);` (`STREAM_CAP = 4000`). On `append`, when the cumulative chunk length exceeds the cap, `BoundedTextBuffer` does `this.data = this.data.slice(overflow)` and tracks `droppedChars`. `render()` returns either the raw data or `[...truncated ${droppedChars} earlier chars...]\n${data}`. This buffer feeds `ts.resultText` via `publishIfDue` (live) and the final assignments in the success and error branches of `runTask`. -- `const fullStreamChunks: string[] = [];` — every `block.text` from the assistant stream is appended verbatim. This array is only joined in the `finally` of `runTask` and written to `${taskId}.md` via `persistTaskMarkdownFile` when `options.fullOutputAbsoluteDir` is set. - -Consequence: `ts.resultText` is **always the tail** (with an explicit banner when the prefix was dropped). The complete stream exists only as in-memory chunks for the lifetime of `runTask`, then on disk in `${taskId}.md` (and only when artifacts are not suppressed by `--no-artifacts`). - -### 2. `buildUpstreamContext` reads the bounded buffer, then truncates it again - -`buildUpstreamContext` (same file) walks `task.depends_on`, fetches each parent's `TaskState` from `stateById`, and inlines `dep.resultText` after passing it through `truncateUpstreamSnippet(text, UPSTREAM_SNIPPET_CAP)` with `UPSTREAM_SNIPPET_CAP = 2000`. `truncateUpstreamSnippet` is section-aware: when the text has two or more `## ` headings, it drops sections in `SECTION_DROP_PRIORITY` order; otherwise it falls back to `truncate(text, cap)` which does `s.slice(0, n - 1) + '…'`. The section-aware path can also fall through to that final hard slice when no eligible section is droppable. - -Consequence: a child task's prompt is `framing + buildUpstreamContext(...) + extraContext + subtask_prompt`, where the upstream block is **a 2000-char view of the 4000-char tail of the parent's full stream**, glued together with **no visible banner** at the prompt level. The tail buffer's own `[...truncated N earlier chars...]` banner does travel into the upstream block when present, so the child does see a signal that the parent was capped — but the second 2000-char truncate that `truncateUpstreamSnippet` performs ends with `'…'` and no count, which a model will not reliably interpret as "the prompt above is itself truncated". - -### 3. `findings_sidecar.ts` parses the bounded buffer, not the full stream - -`writeFindingsSidecar(findingsDir, ts)` builds `sections: parseSections(ts.resultText ?? '')`. There is no path that reads `fullStreamChunks` or the artifact file. The header comment on `findings_sidecar.ts` correctly states "sidecar is captured at task completion", but "task completion" means after `BoundedTextBuffer` has already dropped the prefix. - -Consequence: the sidecar **cannot** repair `STREAM_CAP` prefix loss. It can only stabilize parsing against in-flight canvas updates: by writing once at `dispatchTask` completion, it avoids the race where `extractConvergenceFindings` reads `ts.resultText` mid-stream. The earlier draft's "sidecars compensate for truncation" framing was incorrect on this point and is dropped here. - -### 4. `runConvergenceLoop` reads the sidecar **for findings extraction only**, not for `extraContext` - -In `run_dag.ts`, `runConvergenceLoop` calls `readFindingsSidecarAsText(...)` and feeds the result (falling back to `convergeTs.resultText`) into `extractConvergenceFindings`. But the very next call, `buildConvergenceContext(convergeOn, iter, convergeTs.resultText)`, is unconditionally passed `convergeTs.resultText`. The resulting string is threaded through `dispatchTask(task, { extraContext: convergenceContext })` for every re-executed ancestor. - -Consequence: even with `--findings-dir` set, **ancestor re-runs still see only the bounded tail** of the reviewer's output as their "Convergence feedback from … (iteration N-1)" preamble. This is a second, independent truncation surface beyond the prompt-stitch issue in (2), and it is the precise bug the adversarial review identified. - -### 5. Canvas inlines the full `RunState` - -`canvas_writer.ts` `renderCanvasSource` builds the canvas with `const STATE: RunState = ${JSON.stringify(state, null, 2)};`. There is no compression, no externalization, no opt-out. Every `TaskState.resultText` value is embedded verbatim in the `.canvas.tsx` file the IDE recompiles. Today, this is tolerable specifically because `STREAM_CAP = 4000` caps each `resultText` value. Any plan that wants to make `ts.resultText` carry full streams must also redesign what goes into `STATE`, or the canvas will balloon to megabytes and stall the IDE. - -### 6. Resume serializes whatever is in `state.tasks[].resultText` - -`writePersistedRunState` (`self_hosting.ts`) does `JSON.stringify(payload, null, 2)` on `{ version: 1, writtenAt, reason, state }`. The state's task list includes `resultText`. `loadResumedRunState` (`run_dag.ts`) refreshes static metadata from the live DAG but leaves `resultText` untouched. So whatever the bounded buffer happens to hold at rank-boundary persistence — typically the tail of a finished task or the running tail of a `RUNNING` task that gets re-queued to `PENDING` — is what a relaunched process inherits. - -### 7. Oracle evidence is bounded by the same number - -`oracle_task.ts` declares `const ORACLE_TAIL_CAP = 4000;` and stamps `tail(outcome.stdout, ORACLE_TAIL_CAP)` / `tail(outcome.stderr, ORACLE_TAIL_CAP)` into the `## Stdout (tail):` / `## Stderr (tail):` sections of `ts.resultText`. There is no uncapped capture analog — full stdout/stderr exist only in the local strings inside `execShell` and are dropped at function exit. Unlike `kind: 'task'`, there is **no artifact path that preserves them**: `persistTaskMarkdownFile` writes `ts.resultText`, which is already tail-truncated for oracles. - -### 8. The docs already describe `STREAM_CAP` and `UPSTREAM_SNIPPET_CAP` - -`packages/proof/README.md` (Artifact Output, `dag.budget`, supervisor sections) and `.cursor/skills/proof/SKILL.md` ("Caveats": "Per-task streamed text is capped at `STREAM_CAP = 4000` chars to keep the canvas file modest. Upstream context passed to child tasks is capped at 2000 chars per parent, with section-aware truncation …") tell operators about both caps. They do **not** explain that those caps are reused as the execution-plane source of truth for sidecars, convergence `extraContext`, and resumed prompts. That gap is part of the limitation: callers who read the docs and reach for `--findings-dir` reasonably believe it is a backstop, when in fact it shares the same upstream loss. - ---- - -## The split: execution plane vs display plane - -The whole plan reduces to one rule: - -> A consumer that influences what the runner does next must never read from a buffer that another consumer is bounding for size or UX reasons. - -Concretely, after this work lands, each consumer of per-task output has a single, documented source: - -| Consumer | Source after the project completes | Plane | -| ------------------------------------------------------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------- | -| `buildUpstreamContext` (parent → child prompts) | A new execution-authoritative per-task transcript (in-memory `string` during the run, mirrored to disk when artifacts are enabled), explicitly trimmed by an **upstream prompt policy** with a visible, counted banner when trimmed. | Execution | -| `extractConvergenceFindings` (`## Blockers` parsing) | Same authoritative transcript. The `--findings-dir` sidecar continues to be written from the same source so external tooling has a stable JSON form, but it is no longer required for parser correctness inside the runner — it is required only when the **runner process** has restarted. | Execution | -| `buildConvergenceContext` (reviewer feedback into ancestors) | Same authoritative transcript, trimmed by the **same** upstream prompt policy as `buildUpstreamContext` (so a convergence iteration is governed by one policy, not two implicit caps). | Execution | -| `--findings-dir` JSON sidecar contents | Same authoritative transcript. Schema field `sectionsTruncated?: boolean` plus per-section length is added so consumers can detect when **policy** trimmed evidence, distinct from "no content". | Execution-mirrored-to-disk | -| `${taskId}.md` artifact | Authoritative transcript, identical bytes (modulo header). The artifact is the canonical on-disk form of the execution-plane truth for the run. | Execution-mirrored-to-disk | -| `--state-path` / `--resume-state` payload | Either a pointer into the artifact directory (when artifacts are enabled and the directory is stable across restarts — see Phase 3) or an inline-but-compressed body when artifacts are suppressed. Either way, the relaunched process reconstructs the same authoritative transcript before resuming. | Execution | -| Canvas `STATE.tasks[].resultText` | Display-only: a bounded tail (today's `STREAM_CAP` semantics, but renamed `CANVAS_DISPLAY_CAP`) with a visible `[...truncated N earlier chars...]` banner. The canvas may additionally surface a path/hash pointer to the full transcript so a user can open it from the IDE. | Display | -| Canvas `
` block (today already `maxHeight: 320`)        | Same display string; UI continues to virtualize.                                                                                                                                                                                                                                                                  | Display                                        |
-| Oracle `## Stdout (tail) / Stderr (tail)` in `resultText`    | Bounded by `ORACLE_TAIL_CAP` for display **and** the inline sidecar value, but a separate full-evidence path (`${taskId}.stdout.log` / `${taskId}.stderr.log`) is written under the artifact dir for forensics. Convergence and downstream tasks that need oracle evidence pull from artifacts, not `resultText`. | Mixed (display bounded, full evidence on disk) |
-
-The five rows under "Execution" all read from one place. No more drift.
-
-### Why not "just remove the caps"
-
-A naïve "stream everything into `ts.resultText` and inline it in the canvas" approach breaks four things observed today in the code:
-
-1. **Canvas reload UX**: `renderCanvasSource` writes the whole `RunState` JSON every debounce window (default 200ms). A multi-megabyte `resultText` per task × a 10-task DAG would push the file to tens of megabytes and re-trigger an IDE hot-recompile on every `publishIfDue` (default every 500ms). `debounce` and `stream-publish-ms` were tuned around a 4000-char ceiling.
-2. **Prompt overflow**: `buildUpstreamContext` glues every parent's text into the child's prompt. Removing `UPSTREAM_SNIPPET_CAP` without a model-context-aware policy causes silent SDK rejections at runtime whose error messages do not mention "your DAG outputs grew too large".
-3. **State file bloat**: `writePersistedRunState` writes one JSON file per rank boundary and per convergence iteration via `persistState(...)`. Resume reads the entire file synchronously. Long runs would dominate disk and slow restart.
-4. **Privacy**: the canvas lives under `~/.cursor/projects//canvases/` (per `.cursor/skills/proof/SKILL.md` Step 1 conventions). Casual sharing of a canvas TSX today exposes 4000 chars per task; uncapped, it could trivially exfiltrate secrets emitted by a misbehaving subagent (e.g. an oracle dumping env). The privacy posture is a function of "what's inlined in the canvas", not "what the runner saw".
-
-The plan therefore treats each of these as a first-class layer with its own policy, not a side effect of a single shared buffer.
-
----
-
-## Constraint matrix (where each cap lives after the project)
-
-### Canvas safety (`canvas_writer.ts`)
-
-| Constraint                   | Today                                                                                           | After                                                                                                                                                                                                                                                                                                                 |
-| ---------------------------- | ----------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `.canvas.tsx` file size      | Bounded indirectly by `STREAM_CAP` × tasks.                                                     | Bounded directly by `CANVAS_DISPLAY_CAP` per task (default = 4000, same as today) plus an optional pointer field. No path emits a canvas larger than `CANVAS_DISPLAY_CAP × tasks + framing/header bytes`. A regression test asserts the size envelope on a "large output" fixture.                                    |
-| Streaming write churn        | Every assistant text block triggers a `publishIfDue`.                                           | Unchanged. Canvas writes continue to consume the display buffer, which is appended to in the streaming loop. The new execution buffer is appended in the same loop but never triggers a canvas write on its own.                                                                                                      |
-| Truncation banner visibility | `BoundedTextBuffer.render()` prepends `[...truncated N earlier chars...]`.                      | Preserved verbatim. The canvas template will be extended to optionally render a "View full transcript" affordance using a relative path inside the artifact dir (feasibility-gated — see Phase 1). If the IDE canvas runtime cannot fetch, the link degrades to a copy-able path. The plan never assumes fetch works. |
-| Privacy posture              | Canvas inlines up to 4000 chars/task of raw stream — already a leak risk for secrets-in-stdout. | Display cap is unchanged in size; banner unchanged. The new execution transcripts live alongside the existing `${taskId}.md` artifacts and inherit their `.gitignore` story (already covered by `.flatbread/artifacts/` convention). Docs gain a "what gets persisted where" section.                                 |
-
-### Prompt budget (`run_dag.ts`)
-
-| Constraint                                   | Today                                                                                                     | After                                                                                                                                                                                                                                                                                                                                         |
-| -------------------------------------------- | --------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| Per-parent excerpt in `buildUpstreamContext` | Hard `UPSTREAM_SNIPPET_CAP=2000` with section-aware-then-slice truncation, silent at the prompt boundary. | Explicit **upstream prompt policy** with three modes: `full`, `summarize` (today's section-aware behavior, but with a counted banner emitted into the prompt itself), and `maxChars: N` (operator-controlled). Default remains conservative (the current 2000-char section-aware path) but is now named, surfaced in logs, and tested.        |
-| Convergence `extraContext`                   | `buildConvergenceContext(convergeOn, iter, convergeTs.resultText)` — bounded tail unconditionally.        | Reads the same authoritative transcript as `extractConvergenceFindings`. Trims via the same policy as `buildUpstreamContext`. The judge's finding #2 is the test case: a reviewer whose `## Blockers` lines appear past byte 4000 must produce ancestor prompts that contain those blockers.                                                  |
-| `dag.framing`                                | Prepended verbatim. Counts against model context but not against any Proof cap.                           | Unchanged. Documented as "part of the prompt budget — author it deliberately."                                                                                                                                                                                                                                                                |
-| Stitched prompt overflow handling            | None. The SDK may reject; the task ends as `ERROR` with whatever message comes back.                      | When `outputPolicy.upstream` is `full` and the policy estimator predicts a stitched prompt larger than `outputPolicy.maxPromptChars` (new), the runner marks the task `BUDGET-EXCEEDED` **before** dispatch with an actionable message naming the offending parent ids and char counts. This reuses the existing `BUDGET-EXCEEDED` exit path. |
-
-### Artifact storage (README, `--full-output-dir`, `--no-artifacts`)
-
-| Constraint                           | Today                                                                                   | After                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                  |
-| ------------------------------------ | --------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| Default location                     | `/.flatbread/artifacts/dag--/` (timestamped per run).                    | Unchanged.                                                                                                                                                                                                                                                                                                                                                                                                                                                                                             |
-| `--no-artifacts`                     | Suppresses transcripts, `_index.md`, `_dag.json`. Findings sidecars are independent.    | Unchanged for the user-facing flag. Internally, when the runner is in execution-complete mode and artifacts are suppressed, the authoritative transcript is held in memory for the lifetime of the run with an **explicit RAM ceiling** (`--max-in-memory-output-bytes`, default 64 MiB across all tasks). Crossing the ceiling produces `BUDGET-EXCEEDED` on the next task to overflow, not silent loss. This resolves the judge's "policy decision deferred" finding by picking the policy up front. |
-| Supervisor + timestamped directories | Each child runner picks a new timestamp unless the supervisor pins `--full-output-dir`. | Same. New guidance in `README.md` Self-Hosting Mode: when self-hosting is enabled, **pin** `--full-output-dir` so artifact-backed resume reads find the same transcripts the prior process wrote. The supervisor and runner both refuse to start if `--restart-on-runner-change` is set without either a pinned `--full-output-dir` or a non-default `--max-in-memory-output-bytes`.                                                                                                                   |
-| Atomicity                            | Per-task `${taskId}.md` is written via `writeFile` once per terminal state.             | Unchanged for `${taskId}.md`. New per-task `${taskId}.stream.txt` (or `.bin` if we go content-addressed in a later phase) is written incrementally during the run via append-only writes from the stream loop, then closed in `runTask`'s `finally`. Failure to write the stream file does not abort the task; it is logged and the task is marked with a `streamPersisted: false` flag in state.                                                                                                      |
-
-### Privacy
-
-| Constraint                            | Today                                                            | After                                                                                                                                                                                                                                                                                                                                                                                             |
-| ------------------------------------- | ---------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| Canvas as accidentally-shared surface | Up to 4000 chars/task of raw stream content.                     | Same display ceiling; the only new privacy surface is the pointer (relative path) to the transcript file. The pointer's path lives under `.flatbread/artifacts/`, which is already covered by Flatbread's convention to gitignore the artifacts root. The plan does **not** add new redaction hooks — that is deferred work, called out under Risks.                                              |
-| Oracle stdout/stderr                  | Already tail-bounded at 4000 chars in `resultText`.              | Display unchanged. Full stdout/stderr written to `${taskId}.stdout.log` and `${taskId}.stderr.log` under the artifact dir **only when artifacts are enabled**. With `--no-artifacts`, oracle full evidence stays in memory subject to the same `--max-in-memory-output-bytes` ceiling.                                                                                                            |
-| Persisted state JSON                  | Embeds `resultText` (bounded today; would be huge if unbounded). | Embeds the display string (small) plus a `transcript` discriminated union: either `{ kind: 'artifact'; path: string }` or `{ kind: 'inline'; gzippedBase64: string }`. The `inline` form is reserved for `--no-artifacts` runs and carries the new `streamCompression: 'gzip'` versioned field. Schema version bumps to `2` with a documented migration path from `1` (legacy `resultText`-only). |
-
-### Resume / supervisor (`self_hosting.ts`, `loadResumedRunState`)
-
-| Constraint                      | Today                                                                                                               | After                                                                                                                                                                                                                                                                                                                                      |
-| ------------------------------- | ------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| `version: 1` schema             | Hard rejects anything else.                                                                                         | Adds `version: 2`. Reader accepts `1` and `2`; on `1`, the `resultText` value is promoted to both the display string and the inline transcript (because that's all the legacy run captured). Writer always emits `2`. The version bump is the migration.                                                                                   |
-| `RUNNING → PENDING` on resume   | Re-queues the task; existing behavior.                                                                              | Unchanged. The relaunched task gets a fresh stream and a fresh transcript file (the prior partial transcript is preserved at `${taskId}.iter${N}.partial.stream.txt` for forensics).                                                                                                                                                       |
-| Prompt parity across restart    | A relaunched run reconstructs `extraContext` and `buildUpstreamContext` from the bounded `resultText` it inherited. | A relaunched run reconstructs the **same stitched prompt string** that the original process would have produced for that rank, given the same DAG and the same authoritative transcripts. This is **input parity**, not output parity — LLM outputs are not deterministic, and the plan does not claim they are. Phase 3 test guards this. |
-| `RUNNER_RUNTIME_FILES` snapshot | Triggers `EXIT_RUNNER_RESTART` (75) when `run_dag.ts` / `canvas_writer.ts` / etc. change.                           | Unchanged set. The new module(s) added in Phase 1 join `RUNNER_RUNTIME_FILES`.                                                                                                                                                                                                                                                             |
-
-### Convergence semantics
-
-| Constraint                            | Today                                                                             | After                                                                                                                                                                                                                                           |
-| ------------------------------------- | --------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `extractConvergenceFindings` source   | `readFindingsSidecarAsText(...)` or `convergeTs.resultText`.                      | Authoritative transcript, falling back to sidecar (across-process boundary, e.g. resume), falling back to `resultText` (legacy resume). The fallback chain is documented; tests cover each leg.                                                 |
-| Reviewer payload size                 | Implicitly bounded by `STREAM_CAP`. Blockers past byte 4000 are invisible.        | Bounded only by `outputPolicy.maxPromptChars` for prompt-side stitching; bounded only by `--max-in-memory-output-bytes` or disk for storage. A reviewer can emit `## Blockers` near the end of a long evidence dump and the loop will see them. |
-| `## Blockers` placeholder semantics   | `extractConvergenceFindings.filterMeaningful` already drops `(none)`, `n/a`, etc. | Unchanged.                                                                                                                                                                                                                                      |
-| Mutual exclusion with `--converge-on` | Today's "no `DAG.loops` AND `--converge-on`" guard remains.                       | Unchanged.                                                                                                                                                                                                                                      |
-
-### Oracle evidence
-
-| Constraint                                 | Today                                    | After                                                                                                                                                                                                                                                                                                                                                   |
-| ------------------------------------------ | ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `ORACLE_TAIL_CAP = 4000`                   | Applied to stdout and stderr separately. | Renamed `ORACLE_DISPLAY_TAIL_CAP` (semantically still 4000 by default) for clarity. The display-bound tail continues to land in `resultText` so the canvas and sidecar render identically to today.                                                                                                                                                     |
-| Full oracle stdout/stderr persistence      | None.                                    | When artifacts are enabled, `${taskId}.stdout.log` and `${taskId}.stderr.log` capture full output. Sidecar gains optional `stdoutPath` / `stderrPath` fields when present so external tooling can resolve to the full evidence. `formatOracleResult` references these paths in a new `## Evidence` footer.                                              |
-| `extraContext` references to oracle output | Bounded tail.                            | If a downstream task lists an oracle in `depends_on`, `buildUpstreamContext` follows the same policy as for `kind: 'task'`. Long stderr dumps that today are silently dropped will now either flow through to the child (`policy: 'full'`) or trigger `BUDGET-EXCEEDED` (`policy: 'summarize'` exceeding `maxPromptChars`). Either way, no silent drop. |
-
----
-
-## Phased plan with acceptance criteria
-
-Each phase is independently shippable. Phases 1 and 2 together carry the "remove the silent truncation" promise; the rest harden it.
-
-### Phase 0 — Contracts, inventory, and feasibility spikes (no behavior change)
-
-Tracking issue: [#200](https://github.com/FlatbreadLabs/flatbread/issues/200).
-
-**Deliverables**
-
-- **Consumer inventory document** committed alongside this proposal (or inlined as a stable section of the README's "Internal layout" appendix) that explicitly names every read site of `TaskState.resultText` in `packages/proof/src/**`:
-  - `canvas_writer.ts` → canvas `STATE` literal.
-  - `run_dag.ts` `buildUpstreamContext` → child prompts.
-  - `run_dag.ts` `runConvergenceLoop` → `extractConvergenceFindings` source, `buildConvergenceContext` source.
-  - `findings_sidecar.ts` `writeFindingsSidecar` → sidecar `sections`.
-  - `self_hosting.ts` `writePersistedRunState` (indirectly via `state.tasks[].resultText`) → resume payload.
-  - The implicit consumers in `persistTaskMarkdownFile` (uses `fullStreamChunks`, not `resultText` — noted to disambiguate).
-- **Decision document** picking the canonical execution-authoritative store shape: in-memory `Map` plus per-task append-only stream files under the artifact dir. (The plan does not pursue a content-addressed blob store in Phase 1 — that is called out in "Risks" as deferred work; per-task markdown remains the human-friendly form.)
-- **Feasibility spike for canvas pointer affordance.** Open question from the prior draft: can a `.canvas.tsx` invoke runtime fetch of a workspace-relative file? A short spike against `cursor/canvas` answers yes/no. If no, the canvas affordance degrades to a click-to-copy path string. The Phase 1 design does not block on the result — the pointer is always written; only the UI shape changes.
-- **Policy default decision.** `--no-artifacts` + execution-complete mode is resolved here: in-memory ceiling with `BUDGET-EXCEEDED` on overflow (see constraint matrix). Documented in this proposal already; Phase 0 lifts it into the package docs.
-
-**Acceptance criteria**
-
-- A `docs/proposals/proof-output-retention-plan.md` (this file) and a one-page consumer inventory appendix land. No code under `packages/proof/src/**` changes in Phase 0.
-- `pnpm verify` continues to pass (it should, because nothing changed).
-- A short feasibility note answers the canvas fetch question with either "feasible — example PR" or "not feasible — fall back to copy-path". The note is referenced from Phase 1's design.
-
-### Phase 1 — Execution-authoritative transcript and rewired consumers
-
-This is the load-bearing phase.
-
-**Deliverables**
-
-- New module `packages/proof/src/task_transcript.ts` (or equivalent — naming is implementation-detail) that owns the per-task authoritative store. Responsibilities:
-  - In-memory `string` per task id, appended chunk-by-chunk from the stream loop.
-  - Append-only mirror to `${fullOutputAbsoluteDir}/${taskId}.stream.txt` when artifacts are enabled (best-effort; errors logged but never abort the task).
-  - A `read(taskId)` accessor that returns the in-memory string when present, falls back to reading the stream file (used by resume), and falls back to the legacy `ts.resultText` on Phase-1-pre-existing state.
-- `runTask` updated so the stream loop pushes each `block.text` into both the existing `BoundedTextBuffer` (renamed semantically to "display buffer", same `CANVAS_DISPLAY_CAP = 4000`) **and** the new transcript store. `fullStreamChunks` is removed (the transcript store subsumes it). `persistTaskMarkdownFile` reads from the transcript store.
-- `buildUpstreamContext` rewritten to read from the transcript store for each `depends_on` parent. The result passes through a new `applyUpstreamPolicy(text, policy)` that today defaults to the legacy section-aware-then-slice behavior at 2000 chars but emits a prompt-visible counted banner (`[...upstream excerpt: kept last 2000 of N chars, sections dropped: X, Y...]`) instead of a bare `…`.
-- `runConvergenceLoop` updated so **both** `extractConvergenceFindings` and `buildConvergenceContext` read from the transcript store. The sidecar continues to be written, and continues to be used as a cross-process fallback (e.g. resumed runs), but is no longer the primary in-process source.
-- Canvas `STATE.tasks[].resultText` continues to be the bounded display string. A new optional `STATE.tasks[].transcriptPath` field carries the relative path to the stream file when one was written. The canvas template renders a "View full transcript" affordance per the Phase 0 feasibility result.
-- `findings_sidecar.ts` `writeFindingsSidecar` reads from the transcript store and emits `sections` keyed identically to today. A new sibling field `sectionsRaw?: Record` is reserved for a future phase if downstream tooling asks for it; Phase 1 ships only `sections` to keep the on-disk schema stable.
-
-**Acceptance criteria**
-
-Each of the following is a named test that must pass; the test names below are mnemonic, not literal file paths.
-
-1. **Late-region prompt content (no `--findings-dir`).** Fixture: a synthetic stream of 12 000 deterministic chunks for parent task `a` containing the marker string `MARKER_LATE` at byte ~10 000. Child task `b` depends on `a`. After the runner executes, the stitched prompt for `b` (captured via a test-only hook on `applyUpstreamPolicy`) contains `MARKER_LATE` **when** `outputPolicy.upstream` is `full`, and **does not** when `summarize` (with a visible counted banner in either case when trimmed).
-
-2. **Convergence detects late blockers (no `--findings-dir`).** Fixture: reviewer task `r` emits a long preamble followed by `## Blockers\n- still broken` past byte 5000. The legacy runner would miss this because `STREAM_CAP=4000` drops the prefix and `## Blockers` lands at the cap boundary. The new runner detects it and schedules a re-run, regardless of `--findings-dir`.
-
-3. **Convergence `extraContext` carries late reviewer content (no `--findings-dir`).** Fixture as in (2). When the ancestor `a` is re-executed, the captured `extraContext` substring of `a`'s stitched prompt contains the `## Blockers` line. This is the judge's explicit recommendation — pair the parsing test with a stitched-prompt assertion.
-
-4. **Same with `--findings-dir` set.** Both (2) and (3) pass when `--findings-dir` is also set, proving that the sidecar pathway is consistent with the in-memory authoritative source (no drift between in-process and resumed extraction).
-
-5. **Artifact `${taskId}.md` matches the stream file.** For a non-trivial fixture, `${taskId}.md` bytes equal the transcript store bytes (modulo the meta header). This catches a regression where `fullStreamChunks` was retired but `persistTaskMarkdownFile` was missed.
-
-6. **Canvas size envelope.** For a fixture of 5 tasks × 12 000-char outputs, the generated `.canvas.tsx` file size remains below `5 * CANVAS_DISPLAY_CAP + 64 KiB` (the constant captures header, layout, types, and a generous slack). Today's behavior is preserved at the display layer.
-
-7. **Visible truncation banner in the prompt.** When `applyUpstreamPolicy` trims, the **stitched prompt string** (not just the upstream block) contains a banner with a real character count. A grep-for-`…` test fails the legacy silent-ellipsis path.
-
-8. **Existing bounded-loop suite.** `pnpm -F @flatbread/proof test` continues to pass. The plan does not invalidate `BoundedTextBuffer` tests — that helper still exists for the display buffer.
-
-### Phase 2 — Upstream prompt policy with honest budgets
-
-Tracking issue: [#201](https://github.com/FlatbreadLabs/flatbread/issues/201).
-
-**Deliverables**
-
-- `DAG.outputPolicy` (top-level, optional) added to `packages/proof/src/dag.ts` with `parseDAG` validation:
-  ```ts
-  outputPolicy?: {
-    upstream?: 'full' | 'summarize' | { maxChars: number };
-    maxPromptChars?: number;
-  }
-  ```
-  Defaults: `upstream: 'summarize'` with the existing 2000-char section-aware policy (preserves today's behavior); `maxPromptChars: 200_000` (well under typical model context windows but generous enough that benign DAGs do not trip).
-- `runTask` preflight: before `agent.send(stitched)`, compute the stitched prompt length and compare against `maxPromptChars`. On overflow, the task is marked `BUDGET-EXCEEDED` with a message listing the offending parents and their contribution.
-- New CLI knobs (kept narrow — these are mostly DAG-driven):
-  - `--output-policy-upstream `: overrides `outputPolicy.upstream` for ad-hoc runs.
-  - `--max-prompt-chars `: overrides `outputPolicy.maxPromptChars`.
-- `BUDGET-EXCEEDED` exit code path (`EXIT_BUDGET_EXCEEDED = 4`) is reused so wrapper scripts already keyed on it continue to work.
-
-**Acceptance criteria**
-
-1. **Default is byte-for-byte the legacy behavior.** A DAG without `outputPolicy` produces identical stitched prompts to today (excluding the new visible-banner change from Phase 1, which is in effect from Phase 1 onward).
-2. **`policy: 'full'` passes the late marker through** (covered already by Phase 1 test (1) when the fixture sets `upstream: 'full'`).
-3. **`maxPromptChars` overflow surfaces `BUDGET-EXCEEDED` with an actionable message.** Fixture: two parents each contributing 60 000 chars with `policy: 'full'` and `maxPromptChars: 80 000`. The child task ends `BUDGET-EXCEEDED` and the message names both parent ids and total char count.
-4. **CLI/DAG precedence test.** `--max-prompt-chars` overrides `outputPolicy.maxPromptChars`; `outputPolicy` in the DAG overrides defaults; `--models-file`-style precedence is documented and tested.
-
-### Phase 3 — Resume, supervisor, and disk ergonomics
-
-Tracking issue: [#202](https://github.com/FlatbreadLabs/flatbread/issues/202).
-
-**Deliverables**
-
-- `PersistedRunState` schema bumped to `version: 2` in `self_hosting.ts`:
-  ```ts
-  state.tasks[].transcript: { kind: 'artifact'; path: string }
-                          | { kind: 'inline'; encoding: 'gzip-base64'; data: string }
-                          | { kind: 'legacy'; resultText: string };
-  ```
-  - `kind: 'artifact'` is used when `--full-output-dir` is set and the stream file exists.
-  - `kind: 'inline'` is used when `--no-artifacts` is set (or the stream file is missing); the body is gzipped to bound state size.
-  - `kind: 'legacy'` is what `version: 1` readers see and what writers emit for the migration grace period; documented as removable in a future release.
-- `loadResumedRunState` reconstructs the in-memory transcript store from `state.tasks[].transcript` before any rank executes, so subsequent ranks see the same `applyUpstreamPolicy` inputs the prior process would have produced.
-- Supervisor (`run_dag_supervisor.ts`, not modified in source by this proposal but referenced) gains an early validation: if `--restart-on-runner-change` is forwarded and the runner is configured for `--no-artifacts` without `--max-in-memory-output-bytes` overrides, the supervisor logs a clear warning that resumed runs will pay the gzip cost for every transcript. (This is documentation + a log line, not a refusal.)
-- README "Self-Hosting Mode" gains a "pin `--full-output-dir` for resumable runs" paragraph.
-
-**Acceptance criteria**
-
-1. **Input parity across restart.** Fixture: a 4-task DAG that finishes the first two ranks, persists state, exits via `EXIT_RUNNER_RESTART`, and resumes. The stitched prompt strings handed to the SDK in rank 3 are **bytewise identical** between (a) a non-restart full run and (b) the resumed second process, when the underlying assistant streams from rank 1–2 are deterministically replayed via fixtures. Output parity is **not** asserted (LLM determinism is out of scope; the test uses a fake `Agent.send` that replays canned chunks).
-2. **State file size stays bounded.** For the same fixture above, `state-path` JSON size after rank 2 stays below `O(tasks × CANVAS_DISPLAY_CAP)` when artifacts are enabled (because transcripts are pointers), and below `O(tasks × gzipped_transcript_size)` when artifacts are disabled. Both ceilings are asserted as soft thresholds in the test.
-3. **`version: 1` → `version: 2` migration.** A hand-crafted `version: 1` state file (i.e. legacy `resultText`-only) is loaded successfully and its `resultText` populates both the display buffer and `transcript: { kind: 'legacy', resultText: ... }`. The runner continues from there without error. Documented as a one-release grace period.
-4. **Supervisor pin advice.** Running the supervisor with `--restart-on-runner-change` and no pinned `--full-output-dir` emits the warning line; the test greps stdout/stderr for it.
-
-### Phase 4 — Oracle evidence alignment
-
-Tracking issue: [#203](https://github.com/FlatbreadLabs/flatbread/issues/203).
-
-**Deliverables**
-
-- `oracle_task.ts` writes `${taskId}.stdout.log` and `${taskId}.stderr.log` under the artifact directory when artifacts are enabled. These are written atomically at task completion (oracle commands are short-lived; we can capture into in-memory strings, as today, and flush once).
-- `formatOracleResult` adds an optional `## Evidence` footer listing the absolute paths when present; the canvas template renders them as "Open stdout / Open stderr" pointers analogous to the `kind: 'task'` "View full transcript" affordance.
-- The new `--max-in-memory-output-bytes` ceiling (introduced in Phase 1's `--no-artifacts` handling) applies uniformly to oracle full evidence in `--no-artifacts` mode. Overflow produces `BUDGET-EXCEEDED` on the oracle task with an actionable message; this is consistent with how Phase 2 treats prompt overflow.
-- `buildUpstreamContext` is unchanged for oracles in shape: if a child task depends on an oracle, the upstream excerpt follows `outputPolicy.upstream`. The `## Stdout (tail) / ## Stderr (tail)` headings in `resultText` survive untouched.
-
-**Acceptance criteria**
-
-1. **Oracle full evidence persists.** Fixture: an oracle command that emits 10 000 lines to stdout. After completion, `${taskId}.stdout.log` contains all 10 000 lines and `ts.resultText`'s `## Stdout (tail):` body matches today's tail-capped string.
-2. **Oracle in `--no-artifacts` mode**. With `--no-artifacts` and a 10 000-line oracle, the in-memory store holds the full output; if the run's cumulative held bytes exceed `--max-in-memory-output-bytes`, the oracle is marked `BUDGET-EXCEEDED` (a deterministic test fixture sets the ceiling low to trip this on a small payload).
-3. **Downstream task depending on oracle sees policy-bounded upstream.** With `policy: 'full'` and a long oracle stderr, the child task's prompt contains stderr lines past the legacy 4000-char tail. With `policy: 'summarize'`, the prompt sees the existing tail behavior and a visible banner.
-
-### Phase 5 — Documentation and skill refresh
-
-Tracking issue: [#204](https://github.com/FlatbreadLabs/flatbread/issues/204).
-
-**Deliverables**
-
-- `packages/proof/README.md` gains a new section "Where output lives" with a copy of the consumer-source table from this document, adapted for operator audience. The "Caveats" mentions of `STREAM_CAP = 4000` and "upstream context capped at 2000 chars" are rewritten to point at `outputPolicy` and `CANVAS_DISPLAY_CAP` / `ORACLE_DISPLAY_TAIL_CAP`.
-- `.cursor/skills/proof/SKILL.md` Caveats section is rewritten in the same way. The "DAG quality bar" section is unchanged.
-- The CLI options table in `SKILL.md` and `README.md` gains the new flags from Phase 2 (`--output-policy-upstream`, `--max-prompt-chars`) and Phase 1's `--max-in-memory-output-bytes`. The existing flag rows are unchanged.
-- A short "What changed" migration paragraph in the README's release notes for the version that ships Phase 1/2.
-
-**Acceptance criteria**
-
-- A grep across `README.md` and `.cursor/skills/proof/SKILL.md` for the literal strings `STREAM_CAP`, `4000`, and `2000` finds them only where they document the **display** caps `CANVAS_DISPLAY_CAP` and `ORACLE_DISPLAY_TAIL_CAP`, not as execution-plane behavior.
-- `pnpm verify` passes.
-
----
-
-## Test strategy
-
-The proof package's existing test infrastructure is the AVA bounded-loop suite plus the ava+vitest matrix surfaced by root `pnpm test` (per `AGENTS.md`). The plan adds tests at three layers.
-
-### Unit-level (vitest under `packages/proof/__tests__/` or equivalent existing location)
-
-- **`task_transcript`** (new module): append behavior, read fallback chain (in-memory → stream file → legacy `resultText`), write-failure logging without abort.
-- **`applyUpstreamPolicy`**: full/summarize/maxChars modes; banner contents; section-aware drop order preserved when `summarize` is selected; counted banner replaces the bare `'…'`.
-- **`extractConvergenceFindings`**: regression coverage for placeholder detection (existing) and a new case with blockers past the legacy 4000-char boundary.
-- **`writeFindingsSidecar`** + **`readFindingsSidecarAsText`**: round-trip, including the case where the sidecar is written from the new authoritative transcript and consumed across a simulated process boundary.
-- **Schema migration**: `version: 1` → `version: 2` `PersistedRunState` reader. The migration is a pure function and easy to test.
-
-### Integration / golden-fixture (AVA bounded-loop suite, expanded)
-
-These tests do not call the live Cursor SDK. They drive `runTask` through a fake `Agent.create` whose `send` returns a `RunnerTaskRun` with a scripted async iterator. Fixtures are checked-in JSON files describing scripted chunks per task plus expected stitched-prompt substrings.
-
-- **Late marker prompt content** (Phase 1, criteria 1, 7).
-- **Convergence detects late blockers** (Phase 1, criterion 2).
-- **Convergence `extraContext` carries late blockers** (Phase 1, criterion 3) — explicitly asserts the substring of the captured stitched prompt for the re-executed ancestor task. This addresses the adversarial review's "tests should cover `buildConvergenceContext`, with and without `--findings-dir`" recommendation.
-- **`--findings-dir` ↔ in-memory parity** (Phase 1, criterion 4).
-- **Artifact / transcript byte equality** (Phase 1, criterion 5).
-- **Canvas size envelope** (Phase 1, criterion 6) — reads the generated `.canvas.tsx` on disk after a fixture run and asserts the file size and the presence of the new `transcriptPath` field per task.
-- **`maxPromptChars` overflow** (Phase 2, criterion 3) — uses the same scripted-chunks harness with deliberately large fixtures.
-- **Resume input parity** (Phase 3, criterion 1) — runs the fake-agent harness through `EXIT_RUNNER_RESTART`, persists state, re-instantiates a second runner with `--resume-state`, and asserts identical stitched prompts in rank 3 between the resumed second process and a non-restart full run with the same fixtures.
-- **Oracle full evidence persistence** (Phase 4, criterion 1) — `execShell` is exercised against `node -e 'for (let i=0;i<10000;i++) console.log(i)'` so the test does not depend on a system command beyond Node itself.
-
-### Privacy and operator-facing smoke
-
-- Canvas snapshot test: for a fixture that streams a synthetic "secret-shaped" token at byte 0, byte 3500, byte 4500, and byte 10 000, assert that the `.canvas.tsx` `STATE` contains only the byte-3500 and byte-4500 occurrences (i.e. the display tail), with the explicit `[...truncated N earlier chars...]` banner. The transcript file contains all four. Future redaction work plugs in here.
-- A `pnpm verify` invocation continues to pass at every phase boundary. Failing `pnpm verify` blocks the phase.
-
-### CI commands (from `AGENTS.md`)
-
-- `pnpm -F @flatbread/proof test` — focused bounded-loop suite; should remain fast even with the added fixtures (no live SDK).
-- `pnpm verify` — full lint + typecheck + build + test before any phase is considered shipped.
-
----
-
-## Risks and mitigations
-
-| Risk                                                                                              | Mitigation                                                                                                                                                                                                                                                                                                     |
-| ------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| Canvas pointer affordance is not feasible inside `cursor/canvas` runtime.                         | Phase 0 spike. Degrade to a copy-able path string with no fetch. The plan's execution-plane fix does not depend on the canvas displaying full output; it depends on the runner reading from a separate store.                                                                                                  |
-| `outputPolicy: 'full'` users blow past the model context window and see SDK rejections.           | `maxPromptChars` preflight produces `BUDGET-EXCEEDED` with an actionable message before dispatch. Default policy remains `summarize`, so this only affects users who opt in.                                                                                                                                   |
-| Append-only stream files multiply small writes and stress slow filesystems.                       | Writes are best-effort and behind the existing artifact dir gate; failures are logged and the task is flagged but not aborted. The display path continues to work without the stream file. We can add a "write-every-N-chunks" coalesce later if profiling warrants.                                           |
-| State file growth even with pointers (the canvas state itself plus task metadata can still grow). | `--state-path` writes are already once per rank boundary plus per convergence iteration, not per chunk. The new `transcript` field is a tiny discriminated union when artifacts are enabled. Phase 3 acceptance test asserts a soft size ceiling.                                                              |
-| Privacy regression via persisted `${taskId}.stream.txt` files capturing secrets.                  | Files live under the same `.flatbread/artifacts/` directory as today's `${taskId}.md`; the same `.gitignore` and review hygiene apply. A redaction hook (e.g. opt-in regex-based stripping in the stream loop) is **out of scope** for this proposal and called out as future work.                            |
-| Resumed runs find a stale `${taskId}.stream.txt` written by a prior runner with different source. | The supervisor already restarts at rank boundaries — partial in-flight tasks are re-queued `PENDING` and get a new stream file (`${taskId}.iter${N}.partial.stream.txt` is preserved for forensics, new run writes `${taskId}.iter${N+1}.stream.txt`). The transcript store reads honor the iteration counter. |
-| Schema version churn breaks downstream tooling that consumed `--findings-dir` sidecars by shape.  | The sidecar `sections` shape is preserved. New fields (`stdoutPath`, `stderrPath`, `sectionsRaw`) are optional. Existing schemas keep working.                                                                                                                                                                 |
-| `--restart-on-runner-change` triggers mid-Phase-1 development as engineers edit the source.       | The new module(s) are added to `RUNNER_RUNTIME_FILES` only after their first stable commit. During development, contributors use ad-hoc CLI runs (without the supervisor) per `AGENTS.md`'s commands.                                                                                                          |
-| Bounded-loop test suite slows down with the added fixtures.                                       | Fixtures use scripted-chunk fake agents (no LLM calls). Each fixture is at most a few tens of milliseconds. The aggregate budget for the proof test suite is monitored in CI; this proposal's tests target +20 cases at well under 1 s each.                                                                   |
-
----
-
-## Backout / partial-ship considerations
-
-Each phase is independently revertable:
-
-- Phase 1 is the most invasive. If the new transcript module is rolled back, `BoundedTextBuffer` and `fullStreamChunks` are restored exactly as today and the consumer rewires unwind. The schema bump and CLI flags from later phases do not depend on Phase 1 source files existing.
-- Phase 2 adds CLI flags and a parse step. Reverting them returns the defaults that match today's behavior.
-- Phase 3 schema bump must be backed out alongside any field additions to `PersistedRunState`; the `version: 1` reader is kept long enough that an older runner can still read newer files via the legacy fallback (though the converse is not guaranteed and is documented).
-- Phases 4 and 5 are nearly pure additions.
-
----
-
-## Open questions deferred (and why)
-
-1. **Redaction hooks in the stream loop.** Important enough to call out, not enough scoped runway to bundle in. Tracked as a follow-up that consumes the same `task_transcript` write boundary.
-2. **Content-addressed blob storage.** Per-task markdown is human-friendly and gets reused by `_index.md`. A single content-addressed store (e.g. `/blobs/.txt` plus a manifest) is a clean Phase 6 if disk duplication becomes painful, but premature today.
-3. **Lazy canvas fetch vs path-only pointers.** Phase 0 spike decides. Either answer is consistent with the rest of the plan.
-4. **Per-task summaries generated by a cheap model.** Some operators may want the canvas to show a 200-char LLM summary instead of a raw tail. That is purely a display-plane question and orthogonal to the execution-fidelity fix. Listed for future work.
-
----
-
-## References (code, with line-ish landmarks)
-
-- `packages/proof/src/run_dag.ts`: `BoundedTextBuffer`, `STREAM_CAP`, `UPSTREAM_SNIPPET_CAP`, `buildUpstreamContext`, `truncateUpstreamSnippet`, `runTask` (display buffer + `fullStreamChunks` + `persistTaskMarkdownFile`), `runConvergenceLoop` (sidecar fallback + `buildConvergenceContext` call site), `persistState`, `loadResumedRunState`.
-- `packages/proof/src/canvas_writer.ts`: `renderCanvasSource`, `STATE` literal layout, the `
` rendering for streaming output.
-- `packages/proof/src/findings_sidecar.ts`: `writeFindingsSidecar` (reads `ts.resultText`), `readFindingsSidecarAsText`, `parseSections`, `FindingsSidecar`.
-- `packages/proof/src/converge_loop.ts`: `extractConvergenceFindings`, `buildConvergenceContext`, `filterMeaningful`, `PLACEHOLDER_WORDS`.
-- `packages/proof/src/oracle_task.ts`: `ORACLE_TAIL_CAP`, `tail`, `formatOracleResult`, `execShell`.
-- `packages/proof/src/self_hosting.ts`: `PersistedRunState` (`version: 1`), `RUNNER_RUNTIME_FILES`, `EXIT_RUNNER_RESTART`.
-- `packages/proof/README.md`: artifact defaults, supervisor pinning notes, the existing `dag.budget` story.
-- `.cursor/skills/proof/SKILL.md`: operator caveats explicitly naming the 4000/2000 caps that this plan reframes.
diff --git a/docs/proposals/proof-output-retention-review.md b/docs/proposals/proof-output-retention-review.md
deleted file mode 100644
index 1f928bab..00000000
--- a/docs/proposals/proof-output-retention-review.md
+++ /dev/null
@@ -1,131 +0,0 @@
-# Adversarial review: Proof output retention — Phase 1 after latest fixes
-
-Author: Opus 4.7 (replaces the prior review draft).
-Scope: read-only audit of the currently uncommitted diff in `packages/proof/src/**`, `packages/proof/README.md`, and `.cursor/skills/proof/SKILL.md`, plus the new modules `task_transcript.ts` and `upstream_policy.ts` and the new test file `packages/proof/src/__tests__/output-retention-phase1.test.ts`, measured against [`docs/proposals/proof-output-retention-plan.md`](./proof-output-retention-plan.md) and [`docs/proposals/proof-output-retention-judge.md`](./proof-output-retention-judge.md).
-
-The latest fixes close the prior review's three high-severity items: `TaskTranscriptStore.get` now reads `${taskId}.stream.txt` from disk when the in-memory map is empty, `main()` registers existing mirror paths from `loadResumedRunState` for any task whose persisted `transcriptPath` is set, `runConvergenceLoop` threads `includeSidecar: false` into the `buildConvergenceContext` source, and the README / SKILL.md docs are now scoped to `kind: 'task'` with an explicit "resumed runs can reconstruct that transcript when the same `--full-output-dir` is reused" clause. This pass focuses on what those fixes still leave open and on residual structural issues.
-
-Follow-up issues:
-
-- [#200 — Phase 0/1 follow-up hardening](https://github.com/FlatbreadLabs/flatbread/issues/200) covers the residual medium/low findings in this review.
-- [#201 — Phase 2: upstream prompt policy and budget preflight](https://github.com/FlatbreadLabs/flatbread/issues/201) covers prompt budgets and the planned `{ maxChars }` policy shape.
-- [#202 — Phase 3: resume, supervisor, and disk ergonomics](https://github.com/FlatbreadLabs/flatbread/issues/202) covers versioned resume state and supervisor restart ergonomics.
-- [#203 — Phase 4: oracle evidence alignment](https://github.com/FlatbreadLabs/flatbread/issues/203) covers full oracle stdout/stderr evidence.
-- [#204 — Phase 5: documentation and skill refresh](https://github.com/FlatbreadLabs/flatbread/issues/204) covers the final operator-facing docs pass.
-
----
-
-## Blockers
-
-None.
-
-Phase 1's in-process happy path is correct; the resumed-process happy path is now correct when `--full-output-dir` is pinned across restarts; the disk fallback in `TaskTranscriptStore.get` is wired into both `buildUpstreamContext` and `resolveConvergenceReviewerSource`; `BUDGET-EXCEEDED` propagation skips downstream children. Nothing below prevents merging Phase 1.
-
----
-
-## High-severity findings
-
-None.
-
-The three high-severity items from the prior review are all addressed in source:
-
-1. **Disk fallback for resumed runs.** `task_transcript.ts:116-128` (`get`) reads `${taskId}.stream.txt` synchronously when the in-memory entry is missing/empty and a registered path exists, then caches the result. `run_dag.ts:678-688` wires this up by calling `transcriptStore.registerExistingMirror(...)` for every task whose persisted `transcriptPath` is non-empty before any rank runs. The fallback chain is the one the plan promised: in-memory → stream file → legacy `resultText`.
-2. **`buildConvergenceContext` no longer reads the sidecar.** `run_dag.ts:1828-1834` constructs the `buildConvergenceContext` source via `resolveConvergenceReviewerSource({ ..., includeSidecar: false })`. Only the findings-extraction call sites (`run_dag.ts:1777-1783`, `run_dag.ts:1883-1889`) pass `includeSidecar: true`, so the lossy sidecar shape (heading-only, body-trimmed) can no longer enter the stitched prompt. The judge's medium #2 fix landed.
-3. **Docs no longer overstate.** `packages/proof/README.md:155-160` scopes the execution-authoritative claim to `kind: "task"` and explicitly says "Resumed runs can reconstruct that transcript when the same `--full-output-dir` is reused and `transcriptPath` points at `${task-id}.stream.txt`; otherwise legacy bounded `resultText` remains the fallback." `.cursor/skills/proof/SKILL.md:260` carries the same scoping. The earlier unconditional claim is gone.
-
----
-
-## Medium-severity findings
-
-1. **`TaskTranscriptStore.get` cannot distinguish "definitively empty" from "not yet read"; every empty-transcript access re-reads from disk.** `task_transcript.ts:116-128`: the early-return guard is `if (v !== undefined && v !== '') return v;`. After the disk fallback returns `''` (e.g., a task with no streamed text, or a freshly-truncated mirror file), the cached `text.set(taskId, '')` does not suppress the disk read on the next call — `v === ''` still falls through to the `readFileSync` branch. For a `kind: 'task'` invocation that legitimately emits no assistant text, every downstream `buildUpstreamContext` → `transcripts.getJoined(depId)` call performs a synchronous filesystem read. Functionally correct (returns `''` on every call), but it is the kind of hot-path footgun that the plan's "execution plane vs display plane" framing should have eliminated, not introduced. Add a `triedDisk: Set` (or store `null` vs `''` semantically) so a definitively-empty result is cached.
-
-2. **`registerExistingMirror` does not verify the file exists; the disk fallback silently degrades when the supervisor uses a fresh timestamped directory.** `task_transcript.ts:65-71` just stores the absolute path; it does not `stat` the file. `run_dag.ts:678-688` calls `registerExistingMirror(taskId, fullOutputAbsoluteDir, taskState.transcriptPath)`, where `fullOutputAbsoluteDir` is the **current process's** artifact directory. When `proof-supervisor` relaunches without a pinned `--full-output-dir`, `defaultArtifactsDir(...)` (`run_dag.ts:450-459`) picks a new timestamp, so the registered path points into a directory that does not contain the prior process's stream files. `get`'s `try { readFileSync(...) } catch { return v; }` then silently falls back to bounded `resultText` for every consumer (upstream, convergence, sidecar). The README documents the pinning requirement (`packages/proof/README.md:204`), and `.cursor/skills/proof/SKILL.md:254` repeats it, but the runner neither warns at startup ("registered transcript path missing on disk") nor surfaces the degradation on the canvas. A single `console.warn` in `main()` (after the `registerExistingMirror` loop, comparing how many `transcriptPath`s pointed to extant files) and a one-line `runMessage` augmentation when any registered path is missing would convert a silent fidelity regression into a noisy diagnostic.
-
-3. **The new `task transcript store reads existing mirror files after resume` test exercises the store in isolation, not the runner consumers that depend on it.** `output-retention-phase1.test.ts:190-201` writes a fixture file, calls `store.registerExistingMirror(...)`, and asserts `store.getJoined(...)` returns the bytes. It does not assert that (a) `buildUpstreamContext` surfaces those bytes into the upstream-context block for a downstream task, nor (b) `resolveConvergenceReviewerSource` surfaces them into either `extractConvergenceFindings` or `buildConvergenceContext`. Both wirings are present at `run_dag.ts:2017` and `run_dag.ts:1706` respectively, but a regression that swapped `transcripts.getJoined(depId)` for `dep.resultText` in `buildUpstreamContext` would not fail any current test. The plan's Phase 1 criteria 1, 3, and 4 require end-to-end stitched-prompt assertions; the current diff still does not provide a fake `Agent.create` harness that would let those criteria be wired to runnable tests. Same family as prior review medium #2.
-
-4. **The post-loop `BUDGET-EXCEEDED` decision now sources from the transcript store, but no test asserts that.** `run_dag.ts:1875-1890` wires `extractConvergenceFindings(finalReviewerSource)` to `resolveConvergenceReviewerSource(..., { includeSidecar: true })`, which is the fix the judge's medium #1 asked for. There is still no fixture asserting that a final-iteration reviewer whose `## Blockers` lands past byte 4000 (i.e., outside the legacy display tail) drives `convergeTs.status = 'BUDGET-EXCEEDED'`. The closest test is `convergence extract sees late section beyond legacy STREAM cap window` (`output-retention-phase1.test.ts:100-105`), which exercises `extractConvergenceFindings` directly on a fabricated string — it does not go through `resolveConvergenceReviewerSource` or the loop's terminal branch. Add a fixture that seeds `transcriptStore` with late-region blockers, drives `runConvergenceLoop` with `maxIterations: 1`, and asserts `convergeTs.status === 'BUDGET-EXCEEDED'`. Same gap as prior review medium #7.
-
-5. **The `runOne skips children when upstream is BUDGET-EXCEEDED` test is still a source-string grep, not a behavioral assertion.** `output-retention-phase1.test.ts:243-250` reads `run_dag.ts` from disk, slices 450 chars after `failedDeps = task.depends_on.filter`, and asserts the snippet contains `'BUDGET-EXCEEDED'`. The runtime change at `run_dag.ts:832-838` (and the message update in `skipTask` at `run_dag.ts:1940-1942`) is correct, but a behavioral assertion — a `TaskState` for an upstream marked `'BUDGET-EXCEEDED'` drives the `failedDeps` path through `skipTask`, the child ends `'ERROR'` with the expected `errorMessage` substring — would cost a few lines and is the form the judge's recommendation #2 actually asked for. Carried forward from prior review medium #1.
-
-6. **`${taskId}.md` ↔ `${taskId}.stream.txt` byte parity is unasserted.** `run_dag.ts:1367-1377` calls `persistTaskMarkdownFile(..., options.transcriptStore.getJoined(task.id))` after the awaited final mirror flush, so the artifact body and the mirror file should agree modulo the meta header. Plan Phase 1 criterion 5 names this explicitly: _"For a non-trivial fixture, `${taskId}.md` bytes equal the transcript store bytes (modulo the meta header)."_ The diff has zero coverage. A regression that swapped `transcriptStore.getJoined(task.id)` for `ts.resultText` in the `persistTaskMarkdownFile` call would not fail any test. Carried forward from prior review medium #3.
-
-7. **`--findings-dir` ↔ in-memory parity claim is asserted in one direction only.** `output-retention-phase1.test.ts:114-134` proves the writer respects `parseSource` over `ts.resultText`. The reverse claim — that the convergence loop, with `--findings-dir` set, derives the same `extraContext` it derives without `--findings-dir` — is not exercised. After fix #2 (sidecar removed from `buildConvergenceContext`), the parity is structural in source; testing it would lock the invariant. Carried forward from prior review medium #4.
-
-8. **`beginMirroredAppend` truncates `${taskId}.stream.txt` on every entry, including convergence re-runs and `runTask`-re-queued tasks on resume.** `task_transcript.ts:37-63`: `await writeFile(absPath, '', 'utf8')` overwrites whatever the prior process / prior iteration wrote. On a convergence loop iteration the prior iteration's stream evidence on disk is gone; on a resumed `RUNNING` → `PENDING` task (`run_dag.ts:497-507`), the prior process's partial transcript is gone the moment the new process's `runTask` reaches `beginMirroredAppend`. The plan's Phase 3 risk-table notes the same forensic concern and proposes `${taskId}.iter${N}.partial.stream.txt` preservation; Phase 1 does not implement it. `${taskId}.md` is rewritten with the new content at the end of the new run so the artifact survives, but `.stream.txt` is the only authoritative real-time capture and it is destroyed in place. Carried forward from prior review medium #5.
-
-9. **Mirror flush is bounded to the canvas publish cadence; an unscheduled exit drops up to `streamPublishMs` of bytes from disk.** `run_dag.ts:1234-1244` (`publishIfDue`) returns early when `now - lastPublishAt < streamPublishMs`. The `void options.transcriptStore.flushStreamMirror(...)` call lives inside the non-early-return branch. With `--stream-publish-ms 500` (default), a task crashing 450ms after the last publish loses up to 450ms of unflushed bytes from `m.pendingBuf` from the mirror file. The signal handlers (`run_dag.ts:736-744`) call `failAndExit` without flushing mirrors first; `uncaughtException` similarly skips the flush. Add an `await options.transcriptStore.flushStreamMirror(...)` (or a parallel `flushAllStreamMirrors`) to `failAndExit`'s try block before the canvas flush, and similar drainage inside `onSignal` / `onUncaughtException`. Carried forward from prior review medium #6.
-
-10. **`outputPolicy` validation does not anticipate the Phase 2 shape (`{ maxChars: N }`), and the rejection error does not name the reserved object form.** `dag.ts:287-310` rejects any `upstream` value not equal to `'full'` or `'summarize'`. The test in `output-retention-phase1.test.ts:44-61` only exercises the `'everything'` rejection. When Phase 2 lands and authors copy a `{ maxChars: 2000 }` shape from an internal doc or an LLM-generated DAG, the error will be the same generic string with no signal that the object form is reserved for a later release. An `{ maxChars: 2000 }` rejection test plus a clearer error message — "DAG.outputPolicy.upstream must be 'full' or 'summarize'; the object form `{ maxChars }` is reserved for a future Proof release" — would future-proof the contract. Carried forward from prior review medium #10.
-
-11. **`parseUpstreamSections`' synthetic-heading text is not namespaced and can collide with parent-authored sections.** `upstream_policy.ts:78-92`: when a parent transcript has pre-heading content, the parser injects a section with `heading: 'Upstream truncation notice'` (when the canvas truncation banner is present) or `'Upstream preamble'`. Neither name is unique to the synthetic context. A parent task that legitimately emits `## Upstream preamble` of its own would produce two sections with the same heading; `summarizeWithinCap`'s `findIndex((s, idx2) => idx2 > 0 && s.normalized === dropTarget)` (`upstream_policy.ts:146-154`) would treat them as candidates for the same drop key, and `renderUpstreamSections` would emit both side-by-side. Neither name appears in `SECTION_DROP_PRIORITY`, so the practical attack surface is small, but the collision is structural. Prefixing with a non-Markdown sentinel (e.g. `[proof] Upstream preamble`) — or attaching the preamble to the first authored section's `bodyLines` instead of synthesizing a new heading — closes this without adding state. Carried forward from prior review medium #8.
-
-12. **The README's "Execution vs canvas" bullet says "in-process convergence parsing" — but the disk fallback now serves cross-process resumed convergence too.** `packages/proof/README.md:157`: _"For `kind: "task"` only, stitched prompts, in-process convergence parsing (`--converge-on` / `DAG.loops`), `${task-id}.findings.json` payloads (`--findings-dir`), and `.md` derive from an **execution-authoritative** transcript."_ The qualifier "in-process" is now imprecise: after the disk fallback fix, a resumed process's convergence loop reads the prior process's transcript from `${taskId}.stream.txt` via `TaskTranscriptStore.get` and then runs `extractConvergenceFindings` / `buildConvergenceContext` against that. The next sentence about "Resumed runs can reconstruct that transcript when the same `--full-output-dir` is reused" partly clarifies the picture, but the first sentence walks an operator into the wrong model of when the execution-authoritative source is consulted. Drop "in-process" or rewrite as: "stitched prompts, convergence parsing, findings sidecars, and `.md` derive from an execution-authoritative transcript that is reconstructed from `${task-id}.stream.txt` after a runner restart when `--full-output-dir` is pinned."
-
-13. **`TaskTranscriptStore.get` uses `readFileSync`, which blocks the runner's event loop for large transcripts on resume.** `task_transcript.ts:116-128`: the sync read is called from `buildUpstreamContext` (`run_dag.ts:1995-2027`, synchronous), which itself is called from the async `runTask`. The synchronous-read decision was likely deliberate — `buildUpstreamContext` cannot today produce an `await` without restructuring — but a 5 MiB transcript file read synchronously is tens of ms of event-loop pause per resumed dispatch. The fix is non-trivial (thread `await` through `buildUpstreamContext` → `runTask` → `runOne`) but the cost should be measured before deferring. For Phase 1, a comment naming the constraint and a follow-up note would be sufficient.
-
-14. **`resolveConvergenceReviewerSource` falls through to `convergeTs.resultText` silently when the transcript store is empty and `includeSidecar: false`.** `run_dag.ts:1699-1716`: in the path used by `buildConvergenceContext` (line 1828, `includeSidecar: false`), if `transcriptStore.getJoined(convergeOn).trim().length === 0` — e.g., a resumed process where the convergence task's mirror file was lost per medium #2 — the function returns `opts.resultText ?? ''`. That is the bounded display string, exactly the silent loss the plan was paid down to retire. The structural fix landed for the common case; the residual gap is the edge case where the disk fallback also failed. Either log the fallback at the call site (`buildConvergenceContext` reading bounded `resultText` because no authoritative source was reachable), or attach a synthetic `[...convergence reviewer transcript unavailable; reading display tail...]` banner to the returned string so the model can interpret what it is seeing.
-
----
-
-## Low-severity findings
-
-1. **Stray JSDoc fragment in `TaskTranscriptStore`.** `task_transcript.ts:24` carries `/** When false, omit mirror writes entirely (logged once per task via callback). */` directly above `mirrorEnabledForTask(taskId)`. The comment describes a boolean field that does not exist (the method consults the `mirrors` map and returns `true`/`false` based on presence). Either delete the orphan or rewrite as a proper `@returns` comment on the method.
-
-2. **`RUNNER_RUNTIME_FILES` ordering is no longer alphabetical.** `self_hosting.ts:23-35` appends `'task_transcript.ts'` and `'upstream_policy.ts'` after `'self_hosting.ts'`. The list feeds a hash-and-compare loop, so order does not matter functionally, but the surrounding entries are alphabetical. Minor consistency lapse.
-
-3. **`renderCanvasSource` is exported solely so the new test can call it.** `canvas_writer.ts:210` changed from `function renderCanvasSource` to `export function renderCanvasSource`. Fine for testability, but it broadens the public surface of `canvas_writer.ts`; consider adding a JSDoc `@internal` marker or moving the canvas-size test inside a separate `__internal__` import path.
-
-4. **Canvas-size envelope test slack is generous to the point of being load-insensitive.** `output-retention-phase1.test.ts:234`: `t.true(cappedLen < baselineLen + 5 * CANVAS_DISPLAY_CAP + 96000)`. The plan's Phase 1 criterion 6 specified `5 * CANVAS_DISPLAY_CAP + 64 KiB`. The test allows ~94 KiB of slack — a 5× canvas-render growth of arbitrary metadata would still pass. The complementary assertion `uncappedLen - cappedLen > 35000` does catch a "leaked uncapped resultText" regression, so the test is sufficient as a "cap is doing something" gate, but it is not the tight envelope the plan specified. Tighten the upper bound (e.g. `+ 32000`) so the test fails on 2× growth, not 5×.
-
-5. **`buildConvergenceContext` defaults `upstreamMode` to `'summarize'`.** `converge_loop.ts:164-168`. The runner always passes a value explicitly from `runConvergenceLoop` (`run_dag.ts:1834`), but the default is a footgun for downstream library callers: a tool that calls `buildConvergenceContext(reviewer, iter, text)` without thinking about policy silently gets the legacy 2000-char excerpt. Either require the policy parameter (`UpstreamPolicyMode` is exported, so callers can pass it) or document the default in the JSDoc so the contract is obvious to a downstream consumer.
-
-6. **The README's "Execution vs canvas" block is a four-bullet list now (good) but the bullets do not call out the silent-fallback edge case.** `packages/proof/README.md:155-160`. After the latest fix, the bullets describe the common cases but do not mention what happens when `--full-output-dir` is not reused on the supervisor: the runner does not warn, the canvas does not surface degradation, and downstream prompts silently read bounded `resultText`. A single sentence — "When the artifact directory is not preserved across restarts, the runner silently falls back to the bounded display string for previously-completed tasks; pin `--full-output-dir` on the supervisor to avoid this." — would close the operator-trust gap.
-
-7. **No consumer-inventory document landed.** The plan's Phase 0 deliverable was: _"Consumer inventory document committed alongside this proposal (or inlined as a stable section of the README's 'Internal layout' appendix) that explicitly names every read site of `TaskState.resultText` in `packages/proof/src/**`."_ The diff does not contain such a document, and the README's prose does not enumerate the consumers. Future contributors editing `run_dag.ts` will have to re-derive the list from scratch. Low-impact for Phase 1 itself, but a one-page appendix would pay back the next refactor.
-
-8. **`writeFindingsSidecar`'s call site in `dispatchTask` gates `parseSource` on `effectiveTaskKind(task) === 'task' && transcriptBody.length > 0`** (`run_dag.ts:930-937`). For a `kind: 'task'` that finished with no streamed output, `parseSource` is `undefined` and the sidecar falls back to `ts.resultText` — also empty. Fine in practice. For resumed runs where the prior process completed the task and the current process has an empty in-memory transcript but a non-empty disk transcript, the `transcriptBody.length > 0` check would now succeed because the disk fallback fills `transcriptStore.getJoined(task.id)` on demand — good. But the current runner does not actually call `dispatchTask` for previously-FINISHED tasks (the rank loop's `runnableRank` filter at `run_dag.ts:951-954` skips them), so the sidecar is never re-emitted on resume. That keeps the on-disk sidecar stale relative to the new process's view of the transcript. A comment naming the invariant would prevent a future "always re-emit sidecars on resume" change from regressing.
-
-9. **`fullStreamChunks` removal leaves a stale JSDoc comment in `runTask`.** `run_dag.ts:1230` carries `/** Uncapped execution transcript is accumulated in options.transcriptStore. */` directly above `let run: RunnerTaskRun | undefined;`. The comment attaches to the `let run` declaration, which has nothing to do with the transcript. Reflow onto `options.transcriptStore.append(task.id, block.text);` at line 1282, or delete.
-
-10. **`registerExistingMirror` is a void-return method, so a registration that points at a non-existent file is indistinguishable at the call site from a successful registration.** `task_transcript.ts:65-71`. Return `boolean` (or a discriminated union) so `main()` can count missing files and emit a single summary warning. Pairs naturally with medium #2.
-
-11. **`oracle_task.ts` still tail-bounds stdout/stderr at `ORACLE_TAIL_CAP = 4000` for the inline `resultText` and the sidecar, with no parallel transcript or evidence file.** Phase 4 territory and explicitly out of Phase 1 scope. The README's "Execution vs canvas" bullets correctly carve out "For `kind: "task"` only" so the docs do not promise oracle full-evidence. No action needed beyond keeping the carve-out intact in any future doc rewrite.
-
----
-
-## Verification
-
-1. **Documents consulted end-to-end.** [`docs/proposals/proof-output-retention-plan.md`](./proof-output-retention-plan.md) (all five phases, the constraint matrix, and Phase 1 acceptance criteria 1–8) and [`docs/proposals/proof-output-retention-judge.md`](./proof-output-retention-judge.md) (all four high-severity findings plus the ten medium-severity items) re-read in full. Verified the prior review draft's items 1, 2, 3 (high-severity) are addressed in this diff and that mediums 1, 2, 3, 4, 5, 6, 7, 8, 10 from the prior review are still open.
-
-2. **Resumed-transcript fallback specifically.** Traced `loadResumedRunState` (`run_dag.ts:461-516`) → `main()` registration loop (`run_dag.ts:678-688`) → `transcriptStore.registerExistingMirror` (`task_transcript.ts:65-71`) → `transcriptStore.get` (`task_transcript.ts:116-128`, sync `readFileSync`) → `buildUpstreamContext` (`run_dag.ts:1995-2027`, calls `transcripts.getJoined(depId)`) → `resolveConvergenceReviewerSource` (`run_dag.ts:1699-1716`, calls `opts.transcriptStore.getJoined(opts.convergeOn)`). The chain works when `fullOutputAbsoluteDir` is the same path the prior process wrote to; it silently degrades to legacy `resultText` when the path differs (medium #2).
-
-3. **Sidecar use for findings vs convergence context specifically.** Confirmed the three `resolveConvergenceReviewerSource` call sites at `run_dag.ts:1777`, `run_dag.ts:1828`, and `run_dag.ts:1883`. The middle call (feeding `buildConvergenceContext`) passes `includeSidecar: false`, so the lossy sidecar reconstruction can no longer enter ancestor prompts. The other two calls feed `extractConvergenceFindings`, where the sidecar's heading-only shape is sufficient (`## Blockers` / `## High-severity findings` round-trip cleanly).
-
-4. **Canvas boundedness specifically.** `canvas_writer.ts:210-213` still embeds the full `RunState` JSON via `JSON.stringify(state, null, 2)`. Each `TaskState.resultText` is bounded at `CANVAS_DISPLAY_CAP=4000` chars in `runTask`'s `publishIfDue` (the rendered `BoundedTextBuffer` output is assigned). The new `transcriptPath` field is a short relative path string. `subtask_prompt` remains uncapped — the canvas-size test (`output-retention-phase1.test.ts:203-241`) implicitly accepts this by using `'prompt:'.repeat(200)` (1.4 KiB per task) as the fixture; a larger prompt fixture would expand the assertion's slack.
-
-5. **Budget-exceeded downstream skip specifically.** `run_dag.ts:832-838` (`failedDeps`) includes both `'ERROR'` and `'BUDGET-EXCEEDED'`. `skipTask` (`run_dag.ts:1926-1959`) writes `Skipped: upstream task(s) … blocked this task (upstream ERROR or BUDGET-EXCEEDED)` and marks the child `'ERROR'`. The only test (`output-retention-phase1.test.ts:243-250`) is the source-grep noted in medium #5.
-
-6. **`outputPolicy` validation specifically.** `dag.ts:287-310`: rejects `null`, non-objects, arrays, unknown top-level keys (offending key named in the error), and non-`'full' | 'summarize'` `upstream`. Returns `{}` when `upstream === undefined` so downstream `dag.outputPolicy?.upstream === 'full'` evaluations stay clean. The three tests at `output-retention-phase1.test.ts:28-80` cover the accept-`full`, reject-unknown-key, and reject-unknown-value paths; not the array, non-object, or `{ maxChars }` paths (medium #10).
-
-7. **Stream-mirror ordering specifically.** Re-traced `flushStreamMirror` (`task_transcript.ts:86-110`) against `append` (`task_transcript.ts:73-81`). The synchronous portion of each flush call captures `payload = m.pendingBuf` and clears `m.pendingBuf` before scheduling `m.flushing = m.flushing.then(...)`. Two interleaved `append` + `flush` calls always serialize their `appendFile` payloads in append order via the `flushing` chain. The fire-and-forget `void options.transcriptStore.flushStreamMirror(...)` in `publishIfDue` does not violate this because the synchronous capture still happens before `void` returns. The awaited final flush in `runTask`'s `finally` (`run_dag.ts:1363-1365`) drains all prior scheduled appends before `finalizeTaskMirrorsDone`. The corresponding test at `output-retention-phase1.test.ts:169-188` exercises this with two overlapping flushes and asserts `raw === 'ab'`. Sound.
-
-8. **Docs accuracy specifically.** `packages/proof/README.md:143-180` ("Artifact Output", "Execution vs canvas") now scopes the execution-authoritative claim to `kind: "task"` and names the `--full-output-dir` pinning requirement. `.cursor/skills/proof/SKILL.md:251-263` ("Caveats") matches. The "in-process" qualifier in the bullet at README:157 is now slightly imprecise after the disk fallback fix (medium #12). The Self-Hosting Mode block (`packages/proof/README.md:191-211`) explicitly tells operators to pin `--full-output-dir ` on the supervisor invocation; matches the runtime behavior.
-
-9. **Tests specifically.** All 15 tests in `output-retention-phase1.test.ts` read and mapped to plan acceptance criteria:
-
-   - Criterion 1 (late-region prompt content) — partially covered by helper-level `full upstream excerpt includes late marker past multi-kchar parents`. End-to-end stitched-prompt assertion still missing (medium #3).
-   - Criterion 2 (late blockers) — covered by `convergence extract sees late section beyond legacy STREAM cap window`.
-   - Criterion 3 (convergence extraContext) — covered by `convergence extraContext carries late blockers under full upstream excerpt mode` (helper-level).
-   - Criterion 4 (`--findings-dir` parity) — partially covered by `findings sidecar uses parseSource (full transcript) over bounded resultText` (writer direction only — medium #7).
-   - Criterion 5 (`.md` ↔ `.stream.txt`) — not covered (medium #6).
-   - Criterion 6 (canvas envelope) — covered, slack noted (low #4).
-   - Criterion 7 (counted banner) — covered by `summarize upstream attaches counted excerpt banner instead of omitting rationale`.
-   - Criterion 8 (existing bounded-loop suite) — out-of-scope for this diff.
-
-10. **Suggested next-step verification (not executed in this review).**
-    - `pnpm -F @flatbread/proof typecheck` — confirms TypeScript still compiles. The diff adds a `readFileSync` import to `task_transcript.ts`, exports `renderCanvasSource` and `taskStreamArtifactRelPath`, and threads `UpstreamPolicyMode` through new call sites; all should be statically clean.
-    - `pnpm -F @flatbread/proof test` — exercises both `loops.test.ts` and `output-retention-phase1.test.ts`.
-    - `pnpm verify` — Phase 1 acceptance gate.
-    - A manual fixture run with a >12 KiB parent transcript and `--restart-on-runner-change` triggered between rank 2 and the convergence loop, with `--full-output-dir` pinned, asserting late-region `## Blockers` survive into the convergence ancestor's stitched prompt across the restart. This is the missing acceptance test that pins down high-severity items #1 and #2 in their cross-process form.
-    - The same fixture run without `--full-output-dir` pinning, asserting the runner either warns (the recommendation in medium #2) or at minimum still completes without surfacing stale bounded `resultText` as if it were authoritative.
-    - A manual fixture comparing `${taskId}.stream.txt` bytes against the `## Agent output` body of `${taskId}.md` for a non-trivial streaming task (medium #6).
diff --git a/docs/roadmap.md b/docs/roadmap.md
deleted file mode 100644
index e10937f3..00000000
--- a/docs/roadmap.md
+++ /dev/null
@@ -1,151 +0,0 @@
-# Flatbread roadmap update from validation work
-
-This roadmap reflects the PMF audit, implementation work, and experiment
-reports completed through the current project-board sequence. All verdicts below
-rest primarily on internal evidence; external validation gates promotion past
-**Iterate** on user-facing claims.
-
-## Evidence inputs
-
-- [PMF decision rubric](./pmf-decision-rubric.md)
-- [PMF audit](../flatbread-flow-pmf-audit.md)
-- [Agent artifact opportunity](../flatbread-agent-artifact-opportunity.md)
-- [Positioning](./positioning.md)
-- [Relational starter benchmark](./experiments/issue-162-relational-starter-benchmark.md)
-- [TypeScript safety test](./experiments/issue-163-typescript-safety-test.md)
-- [Export trust experiment](./experiments/issue-164-export-trust-experiment.md)
-- [Effort Graph wire-up](./experiments/issue-167-effort-graph-layout-mapping.md)
-- [Adversarial Effort Graph schema test](./experiments/issue-168-adversarial-multi-layout-schema.md)
-- [Agent artifact retrieval benchmark](./experiments/issue-169-agent-artifact-retrieval-benchmark.md)
-
-## Keep / kill / iterate decisions
-
-| Initiative                          | Decision                             | Rationale                                                                                                                                                    | Next action                                                                                                              |
-| ----------------------------------- | ------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------ |
-| Relation-first content layer        | **Keep**                             | Starter path reaches install/build/codegen/demo query under 10 minutes in a fresh worktree; docs now lead with files → model → typed reads.                  | Polish example content and resolve Next.js ESLint warning.                                                               |
-| ID/ref/cardinality validation       | **Keep**                             | Normalized IDs, duplicate diagnostics, missing-ref validation, cardinality docs/tests, and snapshots now make integrity first-class.                         | Extract reusable validation API and add current/live server integration tests.                                           |
-| Generated TypeScript model/read API | **Iterate**                          | Generated model helpers and read API prove typed consumption is plausible, but selection typing and nullability need hardening before stable positioning.    | Build typed selection/projection API and refine relation helper docs.                                                    |
-| GraphQL interface                   | **Keep, repositioned**               | GraphQL remains useful as schema/introspection/client interface, but docs now frame it as one read surface over the model.                                   | Add schema/SDL export command or documented introspection artifact.                                                      |
-| Local dev loop/watch                | **Iterate**                          | Current codegen watch is useful, but `flatbread start` still needs restart for live content/schema changes.                                                  | Implement `flatbread start --watch` after design/test seams are pinned.                                                  |
-| JSON/CSV portability exports        | **Iterate**                          | Core APIs validate and export stable JSON/CSV views; trust story improves, but CLI and non-developer workflow are not complete.                              | Add `flatbread export json/csv` CLI and fixture outputs.                                                                 |
-| Agent artifact / Effort Graph       | **Iterate — strong candidate wedge** | #167/#168 show schema+mapping is viable; #169 shows large context reduction for blocking-decision retrieval. Evidence is promising but still fixture-driven. | Build MCP query for blocking decisions and run a multi-session real-effort benchmark before making it the primary wedge. |
-| Append/deposit write API            | **Deferred**                         | Effort Graph may need append-oriented writes, but write scope is not validated enough to broaden beyond read/export surfaces.                                | Revisit only if Effort Graph moves toward primary wedge.                                                                 |
-| Hosted CMS / authoring UI           | **Kill for now**                     | No validation required a hosted dashboard; it conflicts with the ownership/local-first wedge.                                                                | Do not schedule until core filesystem workflow is excellent.                                                             |
-| General database replacement        | **Kill for now**                     | Validation work strengthens content integrity but not transactions, auth, multi-writer, or operational DB semantics.                                         | Keep non-goal language prominent.                                                                                        |
-
-## Updated priority order
-
-1. **Ship validation + type-safety foundation** — stabilize IDs, refs,
-   cardinality, snapshots, and generated model helpers.
-2. **Make the canonical example excellent** — keep posts/authors/tags as the
-   first-success path; resolve example lint noise; ensure docs and generated
-   artifacts never drift.
-3. **Add export CLI** — turn JSON/CSV APIs into copy-pasteable commands for the
-   ownership story.
-4. **Implement unified watch loop** — move from documented restart boundaries to
-   `flatbread start --watch` with tests; this is also a precondition for
-   promoting Effort Graph beyond secondary vertical because agent artifact
-   folders change continuously.
-5. **Prototype MCP / agent query surface** — start with blocking decisions by
-   effort ID and reuse the Effort Graph fixture.
-6. **Run external validation** — repeat starter, type-safety, export-trust, and
-   agent retrieval experiments with humans or real multi-session efforts.
-
-## Agent artifact opportunity status
-
-**Decision:** keep Effort Graph as a **secondary vertical with a path to primary
-wedge**.
-
-Reasoning:
-
-- The opportunity aligns with core Flatbread primitives instead of inventing a
-  separate product.
-- The adversarial schema test found fragmentation in mapping profiles, not in
-  the core nouns.
-- Filtered retrieval for blocking decisions was dramatically smaller than
-  context stuffing in the representative benchmark.
-- Evidence is not yet external or multi-session enough to displace the broader
-  TypeScript relational content wedge.
-
-Gate to primary wedge:
-
-- MCP blocking-decision query works against a real multi-session effort.
-- A token-based benchmark (not just bytes) confirms retrieval leverage.
-- MCP and generated-TypeScript read paths reach parity with the GraphQL filter
-  shape on the #167 fixture.
-- At least one external user/team records a saved rediscovery pass on a real
-  multi-session effort relative to its current vault/handoff/search workflow.
-
-## Follow-up issue drafts
-
-The current automation cannot create or close GitHub issues directly. These
-drafts should be turned into issues/project notes by a maintainer:
-
-1. **Add export CLI for JSON/CSV snapshots**
-   - `flatbread export json --collections Post,Author --out snapshots/`
-   - `flatbread export csv --collections Post --out snapshots/`
-2. **Implement `flatbread start --watch`**
-   - schema/content reload, codegen refresh, and failure semantics from
-     `docs/local-dev-loop.md`.
-3. **Add MCP Effort Graph query**
-   - `blockingDecisions(effortId)` returning decision + plan + session context.
-4. **Run network-cold starter benchmark**
-   - fresh clone/container with cold pnpm store.
-5. **Run external export trust interviews**
-   - at least two TypeScript/static-site developers.
-6. **Typed selection builder for generated read API**
-   - remove or isolate the string-selection escape hatch.
-7. **Schema/introspection export artifact**
-   - check in or command-print GraphQL SDL/introspection for exit workflows.
-8. **Harness mapping profiles**
-   - ship Claude-oriented, Cursor-oriented, and GCC-style Effort Graph mapping
-     profiles as configuration rather than separate schemas.
-9. **External validation interview set**
-   - starter, export trust, TypeScript safety, and Effort Graph retrieval runs
-     with non-maintainer users.
-
-## Maintainer action checklist
-
-1. Create follow-up issues/project notes from the drafts above.
-2. Close or split project-board issues according to the traceability table
-   below.
-3. Confirm whether any issue should remain open because acceptance requires
-   external validation this branch could only draft.
-4. Update project-board priority lanes to match the "Updated priority order"
-   section.
-
-## Closed / completed project-board issues in this stack
-
-| Issue | Evidence artifact / commit area                   | Proposed status                                                 |
-| ----- | ------------------------------------------------- | --------------------------------------------------------------- |
-| #142  | `docs/positioning.md`                             | Close                                                           |
-| #143  | `docs/glossary.md`                                | Close                                                           |
-| #144  | `docs/pmf-decision-rubric.md`                     | Close                                                           |
-| #145  | Root quickstart in `packages/flatbread/README.md` | Close                                                           |
-| #146  | README/command guidance updates                   | Close                                                           |
-| #147  | Relation-first traceability docs                  | Close                                                           |
-| #148  | ID normalization helpers/tests                    | Close                                                           |
-| #149  | Missing-reference validation/tests                | Close                                                           |
-| #150  | Duplicate-ID diagnostics/tests                    | Close                                                           |
-| #151  | Cardinality docs/tests                            | Close                                                           |
-| #152  | Validation snapshot fixtures                      | Close                                                           |
-| #153  | Generated content-model types                     | Close                                                           |
-| #154  | Prototype generated TypeScript read API           | Close as prototype; iterate follow-ups                          |
-| #155  | Read-interface docs                               | Close                                                           |
-| #156  | Narrowed core type surfaces                       | Close; iterate follow-ups                                       |
-| #157  | `docs/local-dev-loop.md` design                   | Close design slice; implementation remains follow-up            |
-| #158  | Edit/query demo docs/scripts                      | Close                                                           |
-| #159  | JSON export API/docs/tests                        | Close API slice; CLI remains follow-up                          |
-| #160  | CSV export API/docs/tests                         | Close API slice; CLI remains follow-up                          |
-| #161  | `docs/data-ownership.md`                          | Close                                                           |
-| #162  | Starter benchmark report                          | Close; network-cold benchmark remains follow-up                 |
-| #163  | TypeScript safety report                          | Close; external tests remain follow-up                          |
-| #164  | Export trust report                               | Close product self-review; external interviews remain follow-up |
-| #165  | This roadmap                                      | Close after maintainer review                                   |
-| #166  | Already merged separately                         | Excluded                                                        |
-| #167  | Effort Graph wire-up report/fixtures              | Close                                                           |
-| #168  | Adversarial schema report/fixtures                | Close                                                           |
-| #169  | Artifact retrieval benchmark                      | Close; token/multi-session benchmark remains follow-up          |
-
-Human maintainers still need to confirm, split, or close issues in GitHub
-because this environment cannot mutate issues directly.
diff --git a/docs/tooling-modernization.md b/docs/tooling-modernization.md
deleted file mode 100644
index 2885e790..00000000
--- a/docs/tooling-modernization.md
+++ /dev/null
@@ -1,181 +0,0 @@
-# Tooling modernization notes
-
-This note records the modernization decisions from the proof DAG run used to
-scope this stack. The changes are intentionally conservative: make current
-checks reachable and deterministic first, then defer broad lint/test runner
-replacement until the repository has fewer overlapping toolchain generations.
-
-## Stack entries
-
-### 1. Package test toolchain and root scripts
-
-Objective:
-
-- Modernize the packages that already use Vitest:
-  `@flatbread/codegen` and `@flatbread/utils`.
-- Add root scripts that expose typechecking, split test runners, and a full
-  local verification path.
-- Pin the package manager and Node floor used by the modern Vite/Vitest stack.
-
-Rationale:
-
-- The Vitest suites existed but were not reachable from `pnpm test`.
-- The updated Vitest packages need aligned local `vite`, `typescript`,
-  `@types/node`, and `tsup` versions to avoid peer drift.
-- Package-local `tsconfig.json` files keep TS 6 options scoped to the updated
-  packages instead of forcing every package off the older shared TS 4.7 path at
-  once.
-
-Migration notes:
-
-- `pnpm test` now builds the workspace, runs AVA, then runs the Vitest suites.
-- Root `typecheck` is a pilot scoped to `@flatbread/proof`, the only package
-  with a dedicated typecheck script today.
-- `prepublish:ci` now uses a frozen install plus declaration generation instead
-  of recursively updating dependency ranges.
-
-Rollback:
-
-- Revert `package.json`, `tsconfig.json`,
-  `packages/codegen/package.json`, `packages/codegen/tsconfig.json`,
-  `packages/utils/package.json`, `packages/utils/tsconfig.json`, and
-  `pnpm-lock.yaml`.
-
-### 2. CI verification hardening
-
-Objective:
-
-- Make installs deterministic.
-- Enforce lint, typecheck, build, and tests in CI.
-- Fix integration coverage so the SvelteKit job exercises the SvelteKit example.
-
-Rationale:
-
-- The prior workflow used mutable `pnpm i` installs.
-- `lint` only ran Prettier, `typecheck` was absent, and package Vitest suites
-  were not part of the root test script.
-- `integration-sveltekit` previously ran `pnpm play:build`, which builds the
-  Next.js example.
-
-Migration notes:
-
-- The workflow pins pnpm to `10.33.0` and uses `pnpm install --frozen-lockfile`.
-- `pnpm-workspace.yaml` explicitly approves native build scripts that are
-  required by the current toolchain and examples, including `esbuild`, `sharp`,
-  and `sqlite3`.
-- `FLATBREAD_CI` is set once at workflow scope.
-- `permissions: contents: read` and cancellation concurrency reduce the default
-  token scope and cancel superseded PR runs.
-- SvelteKit integration now runs root build and `examples/sveltekit` build.
-  `svelte-check` remains deferred because the existing route data types report
-  a pre-existing `data.allPostCategories` error after `svelte-kit sync`.
-
-Rollback:
-
-- Revert `.github/workflows/pipeline.yml`. The local scripts from the previous
-  entry remain independently useful.
-
-### 3. Config hygiene from adversarial audit
-
-Objective:
-
-- Remove pnpm configuration that package manifests cannot enforce.
-- Make workspace path aliases and example package metadata explicit.
-
-Rationale:
-
-- `pnpm.peerDependencyRules` only applies from the workspace root, so package
-  local copies in `@flatbread/codegen` and `flatbread` created warning noise
-  without changing install behavior.
-- The Next.js example invoked the `flatbread` CLI without declaring the
-  workspace dependency it relies on.
-- Root path aliases should cover workspace packages consistently for editor and
-  package-local TS config consumers.
-
-Rollback:
-
-- Revert the config hygiene commit to restore the previous package metadata,
-  path aliases, and lockfile entries.
-
-### 4. Decision record and deferred follow-ups
-
-Objective:
-
-- Document major modernization choices, especially the decision not to adopt
-  Biome or Oxc in this pass.
-- Leave reviewers with clear follow-up seams.
-
-Rollback:
-
-- Revert this document only.
-
-## Biome and Oxc evaluation
-
-Biome and Oxc were evaluated as possible replacements or partial replacements
-for ESLint and Prettier. They are not adopted in this stack.
-
-Why not adopt Biome now:
-
-- Prettier currently checks the whole repository, including Markdown and YAML.
-  A partial Biome migration would require careful single-writer boundaries to
-  avoid formatting the same files with two tools.
-- The examples already use framework-specific ESLint stacks:
-  Next.js uses `eslint-config-next`, and SvelteKit uses `eslint-plugin-svelte`.
-  Biome would not replace those framework rules cleanly.
-- A packages-only Biome pilot may still be worthwhile, but it should be a
-  dedicated PR with explicit excludes in `.prettierignore`.
-
-Why not adopt Oxc now:
-
-- Oxlint is best as a fast lint accelerator, not a complete replacement for
-  framework ESLint rules and TypeScript-aware project policy.
-- Root ESLint is not currently enforced by `pnpm lint`, and its dependency graph
-  is already skewed. Adding Oxlint before deciding whether to repair or demote
-  root ESLint would increase the number of lint surfaces.
-
-Recommended future pilot:
-
-1. Repair root linting first: either migrate root packages to ESLint 9 flat
-   config with explicit `typescript-eslint` dependencies, or remove the dormant
-   root ESLint script and rely on compiler plus formatter checks.
-2. If the repo still wants a Rust-based formatter/linter, pilot Biome only for
-   `packages/**/src/**/*.{ts,tsx,js,mjs,cjs}` and keep Prettier responsible for
-   Markdown, YAML, snapshots, and framework examples.
-3. Consider Oxlint only after the ESLint boundary is explicit, as an additional
-   fast lint signal rather than the primary rule authority.
-
-## Runtime impact
-
-Measured locally in this cloud workspace:
-
-- Proof audit DAG: 5/5 tasks completed in about 1 minute 9 seconds.
-- Adversarial follow-up DAG: 5/5 tasks completed in about 55 seconds.
-- Updated `@flatbread/codegen` + `@flatbread/utils` builds completed in about
-  4 seconds after the package-local TS configs were added.
-- `pnpm typecheck` for the current proof pilot completed in about 2.5 seconds.
-
-CI impact must be measured from GitHub Actions after merge or on the draft PR:
-
-- Frozen installs should reduce nondeterministic lockfile drift, not necessarily
-  raw runtime.
-- Cancellation concurrency should reduce wasted runner minutes on superseded PR
-  pushes.
-- The old SvelteKit integration job duplicated Next.js coverage. The new job
-  spends those runner cells on actual SvelteKit build coverage instead of
-  duplicate Next.js validation.
-
-## Deferred recommendations
-
-- Migrate or remove dormant root ESLint in a dedicated linting PR.
-- Unify AVA and Vitest after deciding whether package-local tests should all
-  move to Vitest, or keep both with the contributor guide's current split.
-- Add package-level `typecheck` scripts and move toward project references or a
-  monorepo `tsc -b` flow.
-- Add coverage collection and thresholds around critical paths after test runner
-  boundaries are settled.
-- Audit deprecated runtime dependencies such as Apollo Server v3 and
-  Express-GraphQL separately from this tooling stack.
-- Confirm whether `@nrwl/workspace` is still used; remove it in its own PR if
-  it is dead weight.
-- Resolve the existing SvelteKit route data typing issue, then add
-  `svelte-check` to CI.
diff --git a/examples/nextjs/README.md b/examples/nextjs/README.md
index ccb8651e..a9b80494 100644
--- a/examples/nextjs/README.md
+++ b/examples/nextjs/README.md
@@ -1,8 +1,13 @@
 # Flatbread Next.js Example with TypeScript Codegen
 
-This example is the repo’s **default first success path**: **relational Git-backed markdown** ( **`Post`** ↔ **`Author`** via `refs`; **`tags`** as string arrays on posts) compiled into a typed shape. **GraphQL plus codegen** are one read path baked into this demo, and the generated TypeScript read API gives simple app reads a collection-shaped interface over the same typed model; see [Choosing a read interface](../../packages/flatbread/README.md#choosing-a-read-interface).
+This example turns Markdown files in Git into typed data for a Next.js app.
+Posts link to authors through `refs`, and each post can have a list of string
+tags. GraphQL and codegen are one way to read that data. The generated
+TypeScript API is another; see
+[Choosing a read interface](../../packages/flatbread/README.md#choosing-a-read-interface).
 
-**Note:** `flatbread.config.js` here also declares **PostCategory**, **OverrideTest**, **YamlAuthor**, etc. for integration tests. Treat those as **secondary**; the onboarding narrative is **posts + authors + tags** on the **`Post`** row.
+**Note:** `flatbread.config.js` also includes collections used by integration
+tests. For this guide, focus on posts, authors, and tags.
 
 ## Quick start (from monorepo root)
 
@@ -27,10 +32,14 @@ This example is the repo’s **default first success path**: **relational Git-ba
 
    Add or edit **`.graphql`** files under `queries/` (or globs in config), then rerun codegen so **`tags`**, **`authors`**, and other fields stay in sync. Codegen also emits the prototype **generated TypeScript read API** in `generated/graphql.ts`; see `lib/read.ts` for the posts/authors/tags example that calls `createFlatbreadReadApi()`, and see [Choosing a read interface](../../packages/flatbread/README.md#choosing-a-read-interface) for when to use each read path.
 
-4. **Serve the GraphQL read interface alongside Next** (**there is no `flatbread dev`** — use **`flatbread start`**):
+4. **Start Flatbread and Next** (**there is no `flatbread dev`** — use
+   **`flatbread start`**):
 
-   - **Recommended / headless-safe:** `pnpm exec flatbread start -- next dev --turbopack`.
-   - **Package shortcut:** `pnpm dev` — currently passes `--https` for local convenience, but the Flatbread GraphQL endpoint remains documented as HTTP on `5057`.
+   - **With local HTTPS:** `pnpm dev`. This runs watch mode and starts Next.
+   - **Headless or no HTTPS:** `pnpm exec flatbread start --watch -- next dev --turbopack`.
+
+   With `--watch`, Flatbread reloads valid content and config changes and
+   refreshes generated types. You do not need a second codegen watcher.
 
 5. Open **[http://localhost:3000](http://localhost:3000)** for the app. Flatbread defaults to **`http://localhost:5057/graphql`** (not the Next port).
 
@@ -38,25 +47,28 @@ This example is the repo’s **default first success path**: **relational Git-ba
 
 | Script                                    | Purpose                                                                                             |
 | ----------------------------------------- | --------------------------------------------------------------------------------------------------- |
-| `pnpm dev`                                | **`flatbread start`** + Next dev (HTTPS). GraphQL on **5057**, Next on **3000**.                    |
+| `pnpm dev`                                | **`flatbread start --watch`** + Next dev (HTTPS). GraphQL on **5057**, Next on **3000**.            |
 | `pnpm build`                              | **`flatbread start`** wrapping **`next build`** so schema/codegen paths resolve during build.       |
 | `pnpm start`                              | **`next start` only** — production Next; does **not** run Flatbread unless you arrange it.          |
-| `pnpm run codegen`                        | **Watch-only:** `flatbread codegen --watch` — regenerate types when config/content/documents change. |
+| `pnpm run codegen`                        | Optional separate type watcher. Use it only when `flatbread start --watch` is not running.           |
 | `pnpm run demo:watch-query`               | Watch `example-post.md` and print updated posts/authors/tags query results.                         |
 | `pnpm run demo:edit` / `demo:restore`     | Edit and restore the watched post title for the demo loop.                                          |
 
-### Watch-only codegen
+### Separate codegen watcher
+
+`pnpm dev` already watches content, config, and GraphQL documents. Do not run
+this command beside it.
 
-For iterative work, run the watcher in a second terminal:
+Use this command only when Flatbread is started without `--watch` and you want
+generated types to update:
 
 ```bash
 pnpm run codegen
 ```
 
-For the full loader/schema/codegen/framework boundary contract, see
-[`docs/local-dev-loop.md`](../../docs/local-dev-loop.md). In short: codegen can
-watch content/config/document files, but the running GraphQL server still needs
-a restart for schema or content changes today.
+For details about what changes reload automatically, see
+[`docs/local-dev-loop.md`](../../docs/local-dev-loop.md). When Flatbread starts
+without `--watch`, restart it after content or config changes.
 
 To see a Markdown/YAML edit/query loop without manually restarting a server, run
 the focused demo watcher:
@@ -76,7 +88,10 @@ Markdown and YAML for this demo live under **`examples/content`**; this package
 - **Posts:** `examples/content/markdown/posts/` (`tags` in frontmatter → `[String]` on **`Post`** in the schema.)
 - **Authors:** `examples/content/markdown/authors/` (referenced by id from **`Post`** **`authors`**.)
 
-Canonical layout, **backing files for tags** (facet on each post), **traceability** (same **relation model** from files through config to read interfaces and illustrative query JSON), and guidance on GraphQL versus the generated TypeScript read API are documented in the [Flatbread README quickstart](../../packages/flatbread/README.md#quickstart-posts-authors-and-tags), [Choosing a read interface](../../packages/flatbread/README.md#choosing-a-read-interface), and [glossary](../../docs/glossary.md).
+The [Flatbread README quickstart](../../packages/flatbread/README.md#quickstart-posts-authors-and-tags)
+explains the file layout, tags, and how files become app data. See
+[Choosing a read interface](../../packages/flatbread/README.md#choosing-a-read-interface)
+and the [glossary](../../docs/glossary.md) for more detail.
 
 ## Project structure
 
@@ -137,13 +152,22 @@ const authorNames = posts[0]?.authors?.map((author) => author.name);
 const tags = posts[0]?.tags;
 ```
 
-That path queries **posts**, **authors**, and **tags** through the generated TypeScript API while GraphQL remains the underlying execution layer. The lower-level generated methods still accept an optional GraphQL selection string for experimentation, but the canonical example uses the generated default selection so the call site does not hand-write a GraphQL document. For custom selections, persisted operations, or direct GraphQL clients, use operation documents instead; the root [Choosing a read interface](../../packages/flatbread/README.md#choosing-a-read-interface) section is the canonical contract.
+That path queries **posts**, **authors**, and **tags** through the generated
+TypeScript API while GraphQL handles the request underneath. Lower-level
+generated methods can also take a GraphQL selection string. This example uses
+the default selection, so the call site does not need a GraphQL document. For
+custom selections, persisted operations, or direct GraphQL clients, use
+operation documents instead; see
+[Choosing a read interface](../../packages/flatbread/README.md#choosing-a-read-interface).
 
 ## Troubleshooting
 
 ### "No posts found" or network errors
 
-Ensure something is serving Flatbread at **`http://localhost:5057/graphql`** — typically by running **`pnpm dev`** or **`pnpm exec flatbread start -- next dev --turbopack`** from this directory, not `pnpm start` alone.
+Ensure Flatbread is serving **`http://localhost:5057/graphql`**. From this
+directory, run **`pnpm dev`** or
+**`pnpm exec flatbread start --watch -- next dev --turbopack`**. `pnpm start`
+runs Next alone.
 
 ### TypeScript errors after schema changes
 
@@ -152,7 +176,7 @@ Run **`pnpm exec flatbread codegen --clear-cache --verbose`**.
 ## Learn more
 
 - [Flatbread package README](../../packages/flatbread/README.md) — quickstart, install, **`flatbread start`**, and choosing GraphQL or the generated TypeScript read API
-- [Glossary](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/glossary.md) — collections, relations; GraphQL as one surface
-- [Contributing / monorepo workflow](https://github.com/FlatbreadLabs/flatbread/blob/main/CONTRIBUTING.md)
+- [Glossary](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/glossary.md) — collections, relations, and GraphQL as one way to read data
+- [Contributing guide](https://github.com/FlatbreadLabs/flatbread/blob/main/CONTRIBUTING.md)
 - [GraphQL Code Generator](https://www.the-guild.dev/graphql/codegen)
 - [Next.js Documentation](https://nextjs.org/docs)
diff --git a/examples/nextjs/app/components/BlogIndex.tsx b/examples/nextjs/app/components/BlogIndex.tsx
index 93b8c249..ad789e91 100644
--- a/examples/nextjs/app/components/BlogIndex.tsx
+++ b/examples/nextjs/app/components/BlogIndex.tsx
@@ -68,10 +68,10 @@ export default function BlogIndex({ posts }: BlogIndexProps) {
             No Posts Found
           
           

- Start the Flatbread server to load content + From examples/nextjs, start Flatbread to load content

- npx flatbread dev + pnpm dev diff --git a/examples/nextjs/package.json b/examples/nextjs/package.json index 1fad3030..a88a368a 100644 --- a/examples/nextjs/package.json +++ b/examples/nextjs/package.json @@ -3,7 +3,7 @@ "version": "0.1.0", "private": true, "scripts": { - "dev": "flatbread start --https -- next dev --turbopack", + "dev": "flatbread start --watch --https -- next dev --turbopack", "codegen": "flatbread codegen --watch", "demo:edit": "node scripts/demo-edit.mjs", "demo:restore": "node scripts/demo-restore.mjs", diff --git a/flatbread-flow-pmf-audit.md b/flatbread-flow-pmf-audit.md index 7ed8438d..1ba55403 100644 --- a/flatbread-flow-pmf-audit.md +++ b/flatbread-flow-pmf-audit.md @@ -2,6 +2,10 @@ Generated from the DAG task runner audit on May 7, 2026. +> This report records the repository as it was in May 2026. It is background +> research, not current setup instructions. For the current development steps +> and watch behavior, see [Local development loop](./docs/local-dev-loop.md). + **Buyer-facing comparison rubric** (SQLite, CMS, Contentlayer-like, agent-artifact workflows; issue #144 acceptance-style criteria): [docs/pmf-decision-rubric.md](./docs/pmf-decision-rubric.md). Canvas: `file:///Users/tonyketcham/.cursor/projects/Users-tonyketcham-Code-Github-personal-flatbread/canvases/dag-flatbread-pmf-audit.canvas.tsx` diff --git a/flatbread.config.js b/flatbread.config.js new file mode 100644 index 00000000..cee007df --- /dev/null +++ b/flatbread.config.js @@ -0,0 +1,13 @@ +import { source as filesystem } from '@flatbread/source-filesystem'; +import { transformer as markdownTransformer } from '@flatbread/transformer-markdown'; +import { defineConfig } from '@flatbread/config'; +import { effortGraphContent } from '@flatbread/effort-graph'; + +/** + * For development, we use Flatbread's Effort Graph on itself as a long-running agent memory layer. + */ +export default defineConfig({ + source: filesystem(), + transformer: markdownTransformer(), + content: effortGraphContent(), +}); diff --git a/package.json b/package.json index 74949735..999f0ade 100644 --- a/package.json +++ b/package.json @@ -18,6 +18,9 @@ "build:examples": "pnpm -r --filter {examples/*} build", "build:types": "pnpm -r --filter !{examples/*} exec -- tsup --dts-only", "dev": "pnpm -r --parallel --filter !{examples/*} dev", + "skills:sync": "pnpm --filter @flatbread/effort-graph skills:sync", + "skills:check": "pnpm --filter @flatbread/effort-graph skills:check", + "skills:pack-check": "pnpm --filter @flatbread/effort-graph skills:pack-check", "lint:eslint": "eslint packages/**/src", "lint:prettier": "prettier --check --plugin-search-dir=. .", "lint": "pnpm lint:prettier", @@ -33,7 +36,7 @@ "test:ava": "ava", "test:vitest": "pnpm --filter @flatbread/codegen --filter @flatbread/utils test", "test": "pnpm build && pnpm test:ava && pnpm test:vitest", - "verify": "pnpm lint && pnpm typecheck && pnpm build && pnpm test", + "verify": "pnpm skills:check && pnpm skills:pack-check && pnpm lint && pnpm typecheck && pnpm build && pnpm test", "cursor:fetch-cloud-agent": "pnpm --filter @flatbread/proof exec node scripts/fetch-cloud-agent-conversation.mjs", "dev:test": "ava --watch --verbose", "prepare": "husky install" @@ -60,6 +63,7 @@ }, "devDependencies": { "@ava/typescript": "3.0.1", + "@flatbread/effort-graph": "workspace:*", "@nrwl/workspace": "14.4.3", "@types/inquirer": "8.2.1", "@types/node": "16.11.47", diff --git a/packages/config/src/load.test.ts b/packages/config/src/load.test.ts index 4fab39f5..1f8c5864 100644 --- a/packages/config/src/load.test.ts +++ b/packages/config/src/load.test.ts @@ -68,3 +68,37 @@ test('loadConfig returns an initialized config', async (t) => { } ); }); + +test('loadConfig isolates concurrent temporary ESM modules', async (t) => { + await withTempConfig( + { + 'flatbread.config.js': ` + export default { + source: { fetch: async () => ({}) }, + transformer: { extensions: ['.md'], inspect: (input) => String(input) }, + content: [], + }; + `, + }, + async (cwd) => { + const results = await Promise.all( + Array.from({ length: 32 }, () => loadConfig({ cwd })) + ); + + t.is(results.length, 32); + t.true( + results.every( + (result) => result.filepath === path.join(cwd, 'flatbread.config.js') + ) + ); + + const remainingFiles = await fs.readdir(cwd); + t.deepEqual( + remainingFiles.filter((filename) => + filename.startsWith('.flatbread.config.js.timestamp-') + ), + [] + ); + } + ); +}); diff --git a/packages/config/src/load.ts b/packages/config/src/load.ts index fa8a5612..48dd6136 100644 --- a/packages/config/src/load.ts +++ b/packages/config/src/load.ts @@ -5,6 +5,7 @@ import { initializeConfig, } from '@flatbread/core'; import { build } from 'esbuild'; +import { randomUUID } from 'node:crypto'; import fs from 'node:fs/promises'; import path from 'node:path'; import { pathToFileURL } from 'node:url'; @@ -39,9 +40,13 @@ async function loadConfigFromBundledFile( // for esm, before we can register loaders without requiring users to run node // with --experimental-loader themselves, we have to do a hack here: // write it to disk, load it with native Node ESM, then delete the file. - const fileBase = `${fileName}.timestamp-${Date.now()}`; - const fileNameTmp = `${fileBase}.mjs`; - const fileUrl = `${pathToFileURL(fileBase)}.mjs`; + const fileNameTmp = path.join( + path.dirname(fileName), + `.${path.basename(fileName)}.timestamp-${Date.now()}-${ + process.pid + }-${randomUUID()}.mjs` + ); + const fileUrl = pathToFileURL(fileNameTmp).href; await fs.writeFile(fileNameTmp, bundledCode); try { @@ -50,7 +55,17 @@ async function loadConfigFromBundledFile( return configModule.default; } finally { - await fs.unlink(fileNameTmp); + try { + await fs.unlink(fileNameTmp); + } catch (error) { + const errorCode = + error instanceof Error + ? (error as Error & { code?: string }).code + : undefined; + if (errorCode !== 'ENOENT') { + throw error; + } + } } } @@ -83,7 +98,7 @@ export async function loadConfig({ cwd = process.cwd() } = {}): Promise< const configFilePath = path.join(cwd, configFileName); const { code } = await bundleConfigFile(cwd, configFileName); - const rawConfig = await esmLoader(configFileName, code); + const rawConfig = await esmLoader(configFilePath, code); const config = initializeConfig(rawConfig); return { @@ -118,7 +133,6 @@ async function bundleConfigFile( outExtension: { '.js': '.mjs', }, - watch: true, plugins: [ { // diff --git a/packages/effort-graph/README.md b/packages/effort-graph/README.md index c0964573..cc2c425e 100644 --- a/packages/effort-graph/README.md +++ b/packages/effort-graph/README.md @@ -1,7 +1,35 @@ # `@flatbread/effort-graph` -The standalone semantic writer for Flatbread's git-native Effort Graph. It stores typed reasoning primitives as markdown and uses a journal for multi-file mutations. +This package writes Flatbread Effort Graph records to Markdown files in your +repository. It uses a journal so a change that affects several files either +finishes completely or is undone. -The v1 mutation surface is exactly: `CreateEffort`, `SetEffortStatus`, `WriteIssue`, `WriteFinding`, `WriteDecision`, `WriteConstraint`, `WriteRisk`, `Supersede`, `Invalidate`, `ResolveIssue`, `AcceptDecision`, `MitigateRisk`, and `SetRiskState`. +Version 1 supports these actions: `CreateEffort`, `SetEffortStatus`, +`WriteIssue`, `WriteFinding`, `WriteDecision`, `WriteConstraint`, `WriteRisk`, +`Supersede`, `Invalidate`, `ResolveIssue`, `AcceptDecision`, `MitigateRisk`, +and `SetRiskState`. -The journal lives at `/.journal` and is ignored by git. Use `effortGraphContent()` to add the six collections to a Flatbread configuration. Cross-collection union fields remain writer-validated rather than Flatbread `refs`. +The journal is stored in `/.journal` and is ignored by Git. Use +`effortGraphContent()` to add Efforts, Issues, Findings, Decisions, +Constraints, and Risks to a Flatbread configuration. The writer checks links +between those record types. + +Read [`skills/effort-graph/glossary.md`](./skills/effort-graph/glossary.md) for +the portable Effort Graph domain model. + +The packaged Agent Skill is in `skills/effort-graph/`. The repository copy in +`.agents/skills/effort-graph/` is generated from those files. Run +`pnpm skills:sync` from the repository root after changing the skill. + +## Install the Effort Graph skill + +Install from a release tag, then activate the skill for setup: + +```bash +npx skills add https://github.com/FlatbreadLabs/flatbread/tree/vX/packages/effort-graph/skills/effort-graph --skill effort-graph +npm install --save-dev flatbread@X +``` + +Replace the placeholders with `gitTag` and `flatbreadVersion` from +`skills/effort-graph/release.json`. See `skills/effort-graph/setup.md` for +equivalent `pnpm`, `yarn`, and `bun` commands. diff --git a/packages/effort-graph/package.json b/packages/effort-graph/package.json index 0601e513..3fcac788 100644 --- a/packages/effort-graph/package.json +++ b/packages/effort-graph/package.json @@ -5,8 +5,11 @@ "type": "module", "scripts": { "build": "tsup", - "dev": "tsup --watch src", - "test": "pnpm --dir ../.. exec ava \"packages/effort-graph/src/__tests__/**/*.test.ts\"", + "dev": "node scripts/watch-skills.mjs", + "skills:sync": "node scripts/sync-skills.mjs", + "skills:check": "node scripts/sync-skills.mjs --check", + "skills:pack-check": "node scripts/pack-skills.mjs", + "test": "pnpm --dir ../.. exec ava \"packages/effort-graph/src/__tests__/**/*.test.{js,ts}\"", "typecheck": "tsc -p tsconfig.json --noEmit" }, "repository": { @@ -28,6 +31,7 @@ "types": "dist/index.d.ts", "files": [ "dist", + "skills", "*.d.ts" ], "dependencies": { diff --git a/packages/effort-graph/scripts/pack-skills.mjs b/packages/effort-graph/scripts/pack-skills.mjs new file mode 100644 index 00000000..6f0794f2 --- /dev/null +++ b/packages/effort-graph/scripts/pack-skills.mjs @@ -0,0 +1,161 @@ +import { execFileSync } from 'node:child_process'; +import { promises as fs } from 'node:fs'; +import { relative, resolve } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +export const packageRoot = resolve( + fileURLToPath(new URL('..', import.meta.url)) +); +export const canonicalSkillRoot = resolve(packageRoot, 'skills'); +export const forbiddenInvocation = 'node packages/flatbread/bin/flatbread.js'; + +export function verifyReleaseIdentity(canonicalTexts, packageVersions) { + const entry = canonicalTexts.find( + ({ path }) => path === 'skills/effort-graph/release.json' + ); + if (!entry) + throw new Error('Canonical skill payload is missing release.json'); + let release; + try { + release = JSON.parse(entry.text); + } catch (error) { + throw new Error( + `Canonical release.json is invalid JSON: ${ + error instanceof Error ? error.message : String(error) + }` + ); + } + const expectedTag = `v${packageVersions.flatbreadVersion}`; + if ( + release.format !== 1 || + release.flatbreadVersion !== packageVersions.flatbreadVersion || + release.effortGraphVersion !== packageVersions.effortGraphVersion || + release.gitTag !== expectedTag + ) { + throw new Error( + 'Canonical release.json must contain format 1, both current package versions, and gitTag=v' + ); + } +} + +export function verifyPackPayload( + payload, + canonicalFiles, + canonicalTexts, + packageVersions +) { + const files = Array.isArray(payload) ? payload[0]?.files : payload?.files; + if (!Array.isArray(files)) { + throw new Error('npm pack --dry-run returned no files list'); + } + + const packagedPaths = new Set( + files + .map((file) => + typeof file === 'string' + ? file + : file && typeof file === 'object' && 'path' in file + ? file.path + : undefined + ) + .filter((file) => typeof file === 'string') + ); + const missing = canonicalFiles.filter((file) => !packagedPaths.has(file)); + if (missing.length > 0) { + throw new Error( + [ + 'Effort Graph package payload is missing canonical skill files:', + ...missing.map((file) => ` ${file}`), + 'Check the package "files" configuration and run pnpm skills:sync.', + ].join('\n') + ); + } + + const forbiddenFiles = canonicalTexts + .filter((entry) => entry.text.includes(forbiddenInvocation)) + .map((entry) => entry.path); + if (forbiddenFiles.length > 0) { + throw new Error( + [ + `Canonical skill text contains the monorepo-only invocation "${forbiddenInvocation}":`, + ...forbiddenFiles.map((file) => ` ${file}`), + 'Use the installed flatbread CLI instead.', + ].join('\n') + ); + } + verifyReleaseIdentity(canonicalTexts, packageVersions); +} + +async function readCanonicalSkill() { + const files = []; + async function visit(directory) { + const names = await fs.readdir(directory, { withFileTypes: true }); + for (const entry of names) { + const path = resolve(directory, entry.name); + if (entry.isDirectory()) await visit(path); + else if (entry.isFile()) + files.push({ + path: relative(packageRoot, path), + text: await fs.readFile(path, 'utf8'), + }); + } + } + await visit(canonicalSkillRoot); + return files.sort((left, right) => left.path.localeCompare(right.path)); +} + +export async function verifyPackagePayload() { + const canonicalTexts = await readCanonicalSkill(); + const canonicalFiles = canonicalTexts.map((entry) => entry.path); + const flatbreadPackage = JSON.parse( + await fs.readFile(resolve(packageRoot, '../flatbread/package.json'), 'utf8') + ); + const effortGraphPackage = JSON.parse( + await fs.readFile(resolve(packageRoot, 'package.json'), 'utf8') + ); + let output; + try { + output = execFileSync('npm', ['pack', '--dry-run', '--json'], { + cwd: packageRoot, + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'pipe'], + }); + } catch (error) { + const detail = + error && typeof error === 'object' && 'stderr' in error + ? String(error.stderr) + : error instanceof Error + ? error.message + : String(error); + throw new Error(`npm pack --dry-run failed:\n${detail}`); + } + + let payload; + try { + payload = JSON.parse(output); + } catch (error) { + throw new Error( + `npm pack --dry-run returned invalid JSON: ${ + error instanceof Error ? error.message : String(error) + }` + ); + } + verifyPackPayload(payload, canonicalFiles, canonicalTexts, { + flatbreadVersion: flatbreadPackage.version, + effortGraphVersion: effortGraphPackage.version, + }); + return canonicalFiles; +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) { + verifyPackagePayload() + .then((files) => { + console.log( + `Effort Graph package payload verified (${files.length} skill files).` + ); + }) + .catch((error) => { + console.error(error instanceof Error ? error.message : String(error)); + process.exitCode = 1; + }); +} diff --git a/packages/effort-graph/scripts/sync-skills.mjs b/packages/effort-graph/scripts/sync-skills.mjs new file mode 100644 index 00000000..2b9e7371 --- /dev/null +++ b/packages/effort-graph/scripts/sync-skills.mjs @@ -0,0 +1,154 @@ +import { promises as fs } from 'node:fs'; +import { dirname, relative, resolve } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const scriptDirectory = dirname(fileURLToPath(import.meta.url)); +export const repositoryRoot = resolve(scriptDirectory, '../../..'); +export const defaultSource = resolve( + repositoryRoot, + 'packages/effort-graph/skills/effort-graph' +); +export const defaultDestination = resolve( + repositoryRoot, + '.agents/skills/effort-graph' +); +export const managedSkillNames = [ + 'effort-graph', + 'effort-modeling', + 'grill-with-efforts', +]; +export const managedSkillSources = managedSkillNames.map((name) => + resolve(repositoryRoot, 'packages/effort-graph/skills', name) +); + +async function entries(root) { + const result = []; + async function visit(directory) { + for (const entry of await fs.readdir(directory, { withFileTypes: true })) { + const path = resolve(directory, entry.name); + if (entry.isDirectory()) { + await visit(path); + } else if (entry.isFile()) { + result.push(relative(root, path)); + } + } + } + await visit(root); + return result.sort(); +} + +async function exists(path) { + try { + await fs.access(path); + return true; + } catch { + return false; + } +} + +export async function inspectProjection( + source = defaultSource, + destination = defaultDestination +) { + const sourceFiles = await entries(source); + const destinationFiles = (await exists(destination)) + ? await entries(destination) + : []; + const sourceSet = new Set(sourceFiles); + const destinationSet = new Set(destinationFiles); + const missing = sourceFiles.filter((file) => !destinationSet.has(file)); + const stale = destinationFiles.filter((file) => !sourceSet.has(file)); + const different = []; + + for (const file of sourceFiles) { + if ( + destinationSet.has(file) && + !Buffer.from(await fs.readFile(resolve(source, file))).equals( + await fs.readFile(resolve(destination, file)) + ) + ) { + different.push(file); + } + } + + return { missing, different, stale }; +} + +export async function syncProjection({ + source = defaultSource, + destination = defaultDestination, + check = false, +} = {}) { + const drift = await inspectProjection(source, destination); + if (check) return drift; + + await fs.mkdir(destination, { recursive: true }); + for (const file of [...drift.missing, ...drift.different]) { + const target = resolve(destination, file); + await fs.mkdir(dirname(target), { recursive: true }); + await fs.copyFile(resolve(source, file), target); + } + for (const file of drift.stale) { + await fs.rm(resolve(destination, file)); + } + + return inspectProjection(source, destination); +} + +export async function syncManagedProjections({ check = false } = {}) { + const drift = { missing: [], different: [], stale: [] }; + + for (const [index, name] of managedSkillNames.entries()) { + const source = managedSkillSources[index]; + const destination = resolve(repositoryRoot, '.agents/skills', name); + const result = await syncProjection({ source, destination, check }); + for (const key of Object.keys(drift)) { + drift[key].push(...result[key].map((file) => `${name}/${file}`)); + } + } + + return drift; +} + +function parseArguments(arguments_) { + const options = {}; + for (let index = 0; index < arguments_.length; index += 1) { + const argument = arguments_[index]; + if (argument === '--check') options.check = true; + else if (argument === '--source') + options.source = resolve(arguments_[++index]); + else if (argument === '--destination') + options.destination = resolve(arguments_[++index]); + else throw new Error(`Unknown argument: ${argument}`); + } + return options; +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) { + try { + const options = parseArguments(process.argv.slice(2)); + const hasCustomProjection = options.source || options.destination; + if (Boolean(options.source) !== Boolean(options.destination)) { + throw new Error('--source and --destination must be provided together'); + } + const drift = hasCustomProjection + ? await syncProjection(options) + : await syncManagedProjections(options); + const changed = + drift.missing.length + drift.different.length + drift.stale.length; + if (options.check && changed > 0) { + console.error( + [ + 'Effort Graph skill projection is out of date:', + ...drift.missing.map((file) => ` missing: ${file}`), + ...drift.different.map((file) => ` different: ${file}`), + ...drift.stale.map((file) => ` stale: ${file}`), + ].join('\n') + ); + process.exitCode = 1; + } + } catch (error) { + console.error(error instanceof Error ? error.message : error); + process.exitCode = 1; + } +} diff --git a/packages/effort-graph/scripts/watch-skills.mjs b/packages/effort-graph/scripts/watch-skills.mjs new file mode 100644 index 00000000..e1c541d2 --- /dev/null +++ b/packages/effort-graph/scripts/watch-skills.mjs @@ -0,0 +1,99 @@ +import { watch } from 'node:fs'; +import { promises as fs } from 'node:fs'; +import { spawn } from 'node:child_process'; +import { managedSkillSources, syncManagedProjections } from './sync-skills.mjs'; + +const watchers = new Map(); +let timer; +let syncing = false; +let queued = false; +let stopping = false; +let tsup; + +async function directories(root) { + const result = [root]; + for (const entry of await fs.readdir(root, { withFileTypes: true })) { + if (entry.isDirectory()) { + result.push(...(await directories(`${root}/${entry.name}`))); + } + } + return result; +} + +function closeWatchers() { + for (const watcher of watchers.values()) watcher.close(); + watchers.clear(); +} + +function stop(signal, error) { + if (stopping) return; + stopping = true; + clearTimeout(timer); + closeWatchers(); + if (tsup && !tsup.killed) tsup.kill(signal); + if (error) { + console.error(error instanceof Error ? error.message : String(error)); + process.exitCode = 1; + } +} + +async function refreshWatchers() { + const current = new Set( + (await Promise.all(managedSkillSources.map(directories))).flat() + ); + for (const directory of current) { + if (watchers.has(directory)) continue; + const watcher = watch(directory, { recursive: false }, () => { + clearTimeout(timer); + timer = setTimeout(() => void sync(), 100); + }); + watcher.on('error', (error) => stop('SIGTERM', error)); + watchers.set(directory, watcher); + } + for (const [directory, watcher] of watchers) { + if (!current.has(directory)) { + watcher.close(); + watchers.delete(directory); + } + } +} + +async function sync() { + if (stopping) return; + if (syncing) { + queued = true; + return; + } + syncing = true; + try { + await syncManagedProjections(); + await refreshWatchers(); + } catch (error) { + stop('SIGTERM', error); + } finally { + syncing = false; + if (queued && !stopping) { + queued = false; + await sync(); + } + } +} + +async function main() { + await sync(); + if (stopping) return; + tsup = spawn('pnpm', ['exec', 'tsup', '--watch', 'src'], { + stdio: 'inherit', + }); + tsup.on('error', (error) => stop('SIGTERM', error)); + tsup.once('exit', (code, signal) => { + if (!stopping) { + stop(signal ?? 'SIGTERM'); + process.exitCode = code ?? 1; + } + }); +} + +process.once('SIGINT', () => stop('SIGINT')); +process.once('SIGTERM', () => stop('SIGTERM')); +main().catch((error) => stop('SIGTERM', error)); diff --git a/packages/effort-graph/skills/effort-graph/SKILL.md b/packages/effort-graph/skills/effort-graph/SKILL.md new file mode 100644 index 00000000..41f65698 --- /dev/null +++ b/packages/effort-graph/skills/effort-graph/SKILL.md @@ -0,0 +1,155 @@ +--- +name: effort-graph +description: Journal reasoning (decisions, findings, issues, constraints, risks) into a Flatbread Effort Graph and recall it with bounded reads. Use when starting or resuming a thread of work, recording a decision or finding, resolving an issue, checking what is blocking or still open on an effort, or when the user mentions effort graph, journaling, blocking decisions, or agent memory. +--- + +# Effort Graph — agent journaling and recall + +The Effort Graph is persistent, queryable memory for long-horizon work, stored +as markdown records in the repo. Six primitives: **Effort** (the anchor thread +of work), **Issue**, **Finding**, **Decision**, **Constraint**, **Risk**. +Every record belongs to exactly one Effort. You write through 13 typed +mutations and read through 5 bounded queries — never by hand-editing record +frontmatter (bodies may be edited freely). + +Read [glossary.md](./glossary.md) for the primitive and edge semantics before +inventing a new record kind or relation. + +All commands run from the project root via the `flatbread` CLI (`pnpm exec flatbread`, `npm exec -- flatbread`, `yarn flatbread`, or `bunx flatbread`). +Commands print one JSON object to stdout; +errors print JSON to stderr and exit 1. + +## First activation + +Read [setup.md](./setup.md), make the reviewed config and gitignore edits, then +run `flatbread effort bootstrap` followed by `flatbread effort bootstrap --verify`. Bootstrap is report-only and never edits project files. + +## Prerequisites + +Your `flatbread.config.*` must include the preset: + +```js +import { + defineConfig, + sourceFilesystem, + transformerMarkdown, + effortGraphContent, +} from 'flatbread'; + +export default defineConfig({ + source: sourceFilesystem(), + transformer: transformerMarkdown(), + content: [...effortGraphContent()], +}); +``` + +Records live under `/{efforts,issues,findings,decisions,constraints,risks}/`. +The write journal is `/.journal/`; read digests cache under +`.flatbread/effort-graph/read-cache/` (both gitignored). + +## Writing (journaling) + +One command for all 13 mutations — pass the payload as a single JSON argument: + +```bash +flatbread effort write '{"type":"WriteDecision","effort":"","title":"...","body":"...","derives_from":[""]}' +``` + +Response: `{"generation":"","artifacts":[{"id","path","operation"}],"touched":[...]}`. +**Capture `artifacts[0].id`** to wire later edges, and **keep `generation`** +for strict read-your-writes. + +Full payload shapes for all 13 mutations: read [reference.md](./reference.md). +Critical semantics: + +- Creates always start in the initial lifecycle state: `WriteDecision` → + `proposed`, `WriteIssue` → `open`, `WriteRisk` → `open`. You cannot pass a + state; use lifecycle mutations (`AcceptDecision`, `ResolveIssue`, + `MitigateRisk`, `SetRiskState`) to transition. +- `AcceptDecision` defaults `rejectSiblings: true`, which rejects ALL other + proposed Decisions in the same Effort. Pass `"rejectSiblings": false` + unless you deliberately want the competing proposals closed. +- Edges are forward-only in payloads (`derives_from`, `supersedes`, + `invalidates`); back-edges are materialized automatically. +- When superseding, open the new record's body with a short rollup of what + changed and why — reads render ancestors only as one-line checkpoints. +- For a hard-to-reverse, surprising decision made after a real trade-off, use + the Decision body as the durable rationale: include context, alternatives, + consequences, and reversal criteria. Do not create a parallel ADR; use + [effort-modeling](../effort-modeling/SKILL.md) when the decision is still + being grilled. + +## Reading (recall) + +Every read returns a bounded envelope, not records: a ≤160-token `summary`, +an `artifact_path` to a rendered markdown digest (the evidence — spend one +Read on it, or grep it), `served_generation`, page info, and ≤10 executable +`hints`. Digests cap at 25 records / one-hop expansion / 50 edges / 64 KiB. + +Browse digests (`list`, `records`, `relations`, `blocking-decisions`) excerpt +each body at 600 chars / 12 lines (`[…truncated]`). **`effort get` digests +always include the full record body** (still subject to the 64 KiB digest +byte cap). Zoom in with `get`, then Read/grep that digest — do not open +`.flatbread-efforts/**/*.md` for normal full-body recall. + +```bash +# What's gating this effort? (proposed Decisions deriving from open blocker Issues) +flatbread effort blocking-decisions + +# Resume: discover active Efforts first +flatbread effort list --status active + +# Scoped listing with filters (AND across flags, OR within comma lists). +# --status filters Issues and --state filters Decisions, so combining them in +# one call ANDs across kinds and matches nothing — query each kind separately. +flatbread effort records --kinds issue --status open --since 2026-07-01T00:00:00Z --limit 10 +flatbread effort records --kinds decision --state proposed --limit 10 + +# One-hop neighbors of a record +flatbread effort relations --relations derives_from,superseded_by + +# Single record with full body; --resolve head follows supersession to the tip +flatbread effort get [--resolve head] +``` + +Flags shared by reads: `--strict-min-generation ` (with optional +`--timeout-ms `, default 3000) and, on `list`/`records`/`relations`, `--limit` +(≤25) and `--cursor` (opaque `next_cursor` from a prior page; only valid for +the same query at the same generation). + +`effort list` is bounded Effort discovery. It defaults to `active`; valid +statuses are exactly `active`, `paused`, `completed`, and `abandoned`. +Comma-separated statuses are ORed. Results use the shared `created_at`, then +`id` ordering. After discovery, use bounded effort-scoped reads. + +**Consistency:** reads are eventual by default. Immediately after a write, +pass the returned generation as `--strict-min-generation` — you get either +fresh data or an `EFFORT_GRAPH_GENERATION_WAIT_TIMEOUT` error (exit 1), +never silently stale results. Do not build polling loops; the wait is +server-side. + +## Recommended session workflow + +1. **Resume / status briefing (bounded fast-path):** `effort list --status active` + and trust the returned digest. For each active Effort, run + `effort records --kinds issue,decision` and read each record's + status/state from that one digest. Run `effort blocking-decisions ` + only for an Effort whose digest shows an open `blocker` Issue — skip it + otherwise. Do not open raw `.flatbread-efforts/**/*.md` for briefing; + browse digests are authoritative for status/state. Budget ≈ (1 + number + of active Efforts) digest reads. A 12-run experiment across three model + families showed this roughly halves recall tool calls with no loss of + answer quality (Decision + `dec-adopt-a-bounded-status-briefing-fast-path-for-ef--kcw0rw39g3b2ym2h`). +2. **When a browse digest shows `[…truncated]` and you need the body:** run + `flatbread effort get `, then Read/grep that digest (`artifact_path`) + for the full body. Reserve opening `.flatbread-efforts/**/*.md` for rare + cases (e.g. digest byte-cap miss on an oversized record), not normal + zoom-in. +3. **During work:** journal Findings as evidence lands; open Issues for real + gaps/blockers; record Decisions with `derives_from` citing the Findings, + Constraints, and Issues they respond to. +4. **On commitment:** `AcceptDecision` (mind `rejectSiblings`), `ResolveIssue` + with `resolvedBy` citing the closing Decision/Findings. +5. Maintenance: `flatbread effort cache prune` deletes digests older than + 24h / over the 100 MiB ceiling. diff --git a/packages/effort-graph/skills/effort-graph/glossary.md b/packages/effort-graph/skills/effort-graph/glossary.md new file mode 100644 index 00000000..b3b605bd --- /dev/null +++ b/packages/effort-graph/skills/effort-graph/glossary.md @@ -0,0 +1,64 @@ +# Effort Graph glossary + +The Effort Graph is persistent, queryable memory for long-horizon software +work. It builds on Flatbread's content vocabulary: each primitive is a +Collection, its instances are Records, and cross-primitive references are +Relations in frontmatter. + +It is not a CMS, authoring UI, hosted memory product, or general task tracker. +Operational provenance (session, agent, model, DAG run) belongs in record +frontmatter; durable run transcripts live with Proof artifacts. + +## Primitives + +### Effort + +The anchor for one coherent thread of work: a feature, migration, spike, +research investigation, or refactor. Every other primitive belongs to exactly +one Effort. It scopes bounded reads but carries only a short description; the +reasoning belongs in the related records. + +### Issue + +A tracked item needing attention: a question, defect, gap, or blocker. An Issue +is reactive. Decisions and Findings resolve it through lifecycle edges. + +### Finding + +A grounded observation about code, users, literature, or runtime behavior. +Findings cite evidence, resolve Issues, inform Decisions, surface Risks, and +may invalidate past Findings or Decisions. A retrospective Finding is evidence +gathered after a decision shipped. + +### Decision + +A commitment among alternatives. A proposed Decision is an active alternative; +an accepted Decision is committed; rejected, superseded, and deprecated +Decisions retain their lifecycle history. A Decision cites the Findings, +Constraints, and Risks it weighed rather than duplicating them. + +### Constraint + +A sticky hard or soft boundary that limits the decision space. Constraints are +known limits; they are not prospective negative outcomes. + +### Risk + +A prospective negative outcome with likelihood and severity. It is open, +mitigated by an accepted Decision, realized with evidence, or explicitly +accepted. + +## Edges + +`derives_from` is causal upstream evidence or context. `supersedes` replaces a +record of the same primitive, while `invalidates` says a record was wrong. +Those forward edges are authoritative; `superseded_by` and `invalidated_by` are +writer-materialized reverse projections. New edge vocabulary needs a +dogfooded query the existing vocabulary cannot express. + +## Intentional non-models + +Session, Run, Plan, Artifact, Agent, Investigation, Question, Proposal, +Retrospective, and Branch are not collections. Use provenance fields for +operational data; represent questions as Issues, proposals as proposed +Decisions, retrospectives as Findings, and branch history through Git. diff --git a/packages/effort-graph/skills/effort-graph/reference.md b/packages/effort-graph/skills/effort-graph/reference.md new file mode 100644 index 00000000..1e7bf053 --- /dev/null +++ b/packages/effort-graph/skills/effort-graph/reference.md @@ -0,0 +1,166 @@ +# Effort Graph — full API reference + +Ground truth: the installed `flatbread` CLI and this reference (mutations), +reads, and configuration examples. Repository implementation files are not +consumer ground truth. + +## IDs + +Generated as `---<16-char-crockford>` with prefixes `eff`, +`iss`, `fnd`, `dec`, `con`, `rsk`. Filenames never define identity. Let the +writer generate ids; capture them from mutation results (`artifacts[0].id` +for creates). + +## The 13 mutations (`flatbread effort write ''`) + +Common optional fields on all creates: `id`, `created_at` (ISO with offset), +`produced_in`, `created_by` (opaque provenance strings). Forward edge fields +on all creates except `CreateEffort`: `derives_from[]`, `supersedes[]`, +`invalidates[]` (arrays of existing ids; targets are validated). + +### Effort lifecycle + +```json +{"type":"CreateEffort","title":"...","body":"...","slug":"optional"} +{"type":"SetEffortStatus","effortId":"","status":"active|paused|completed|abandoned"} +``` + +### Creation (required: effort, title, body; initial state is derived) + +```json +{"type":"WriteIssue","effort":"","title":"...","body":"...","kind":"question|defect|gap|blocker|"} +{"type":"WriteFinding","effort":"","title":"...","body":"...","kind":"measurement|survey|dead-end|retrospective|"} +{"type":"WriteDecision","effort":"","title":"...","body":"..."} +{"type":"WriteConstraint","effort":"","title":"...","body":"...","kind":"hard|soft"} +{"type":"WriteRisk","effort":"","title":"...","body":"...","likelihood":"low|medium|high","severity":"low|medium|high"} +``` + +Initial states: Issue `status: open`; Decision `state: proposed`; Risk +`state: open`. + +### Edge retro-linking (records must already exist) + +```json +{"type":"Supersede","supersederId":"","targetId":""} +{"type":"Invalidate","findingId":"","targetId":""} +``` + +`Supersede` is same-primitive only and rejects an already-superseded target. +`Invalidate` asserts the target was wrong (stronger than superseded). + +### Lifecycle transitions + +```json +{"type":"ResolveIssue","issueId":"","resolution":"resolved|deferred|wontfix","resolvedBy":[""]} +{"type":"AcceptDecision","decisionId":"","rejectSiblings":false} +{"type":"MitigateRisk","riskId":"","decisionId":""} +{"type":"SetRiskState","riskId":"","state":"realized|accepted","evidence":[""]} +``` + +`AcceptDecision` with `rejectSiblings: true` (the default!) also sets every +other `proposed` Decision in the Effort to `rejected` with a back-pointer. +All mutations run in one journal transaction (save-or-undo). + +### Mutation result + +```json +{ + "generation": "57", + "artifacts": [ + { "id": "...", "path": "decisions/....md", "operation": "created|updated" } + ], + "touched": [{ "id": "...", "path": "..." }] +} +``` + +`generation` is a durable, monotonic journal token — the input to strict reads. + +## The 5 read queries + +All reads execute through Flatbread's query engine (in-process GraphQL over +the generated schema) and return a `ReadEnvelope`: + +```json +{ + "summary": "2 records; proposed 2; complete", + "artifact_path": ".flatbread/effort-graph/read-cache//.md", + "artifact_sha256": "...", + "served_generation": "55", + "consistency": { "mode": "eventual|strict", "min_generation": null }, + "page": { "returned": 2, "has_more": false, "next_cursor": null }, + "hints": ["getRecord(\"dec-...\")"] +} +``` + +The digest at `artifact_path` is deterministic markdown: YAML query header, +anchor index, per-record sections (selected frontmatter, body, relation +lists), one-hop related records, and an edge table. Body policy: + +- **`effort get`:** full record body (the normal zoom-in path). +- **`list` / `records` / `relations` / `blocking-decisions`:** body excerpt + capped at 600 chars / 12 lines (`[…truncated]`). + +Caps: 25 primary records, one hop, 50 edges, 64 KiB; hitting a cap sets +`complete: false` with named `cap_reasons` — narrow the query or page rather +than expecting more. If a `get` body alone exceeds the 64 KiB digest byte +cap, the digest fails closed with a byte-cap banner (it does **not** fake a +full body via the 600/12 excerpt). + +### Commands + +```bash +flatbread effort get [--resolve exact|head] [consistency flags] +flatbread effort list [--status active,paused,...] [--limit n] [--cursor c] [consistency flags] +flatbread effort records [--kinds k1,k2] [--state s1,s2] [--status s1,s2] [--kind k1,k2] [--since iso] [--until iso] [--limit n] [--cursor c] [consistency flags] +flatbread effort relations --relations r1,r2 [--limit n] [--cursor c] [consistency flags] +flatbread effort blocking-decisions [consistency flags] +flatbread effort cache prune +``` + +- `--kinds`: `effort|issue|finding|decision|constraint|risk` (records: + default all non-effort kinds). +- `list --status`: defaults to `active`; valid values are exactly `active`, + `paused`, `completed`, and `abandoned`. Values are ORed and results are + ordered by `created_at` ascending, then `id`. +- Filter semantics: AND across different flags, OR within a comma list. + `--since`/`--until` bound `created_at` (gte/lte, ISO strings). +- `--relations` values: `derives_from`, `supersedes`, `superseded_by`, + `invalidates`, `invalidated_by`, `rejected_by`, `mitigated_by`, + `resolved_by`, `evidence` (one hop, explicit only). +- `--resolve head`: follow `superseded_by` to the current tip; ancestors + render as checkpoint lines (max 5, then a count). +- `blocking-decisions` membership (frozen): Decision in the effort with + `state: proposed` whose `derives_from` directly contains an Issue in the + same effort with `kind: blocker` and `status: open`. For "what blockers + are open at all", use + `records --kinds issue --kind blocker --status open`. + +### Consistency flags + +- `--strict-min-generation `: serve at or after that journal + generation, or fail. `--timeout-ms ` bounds the wait (default 3000). +- Errors (stderr JSON, exit 1): `EFFORT_GRAPH_GENERATION_WAIT_TIMEOUT`, + `EFFORT_GRAPH_INVALID_CURSOR` (cursor reused across a different query or + generation). + +## Configuration surface + +| Option | Where | Default | Notes | +| ---------------- | --------------------------------------------------- | ------------------------------------------- | ------------------------------------------------------------------------------------------------------------- | +| Graph root | `effortGraphContent(root)` in `flatbread.config.js` | `.flatbread-efforts` | All six collection paths + refs derive from it; the preset must appear complete and unmodified for detection. | +| Config discovery | cwd of the CLI invocation | — | Exactly one `flatbread.config.*` must exist in cwd. | +| Digest cache | fixed | `/.flatbread/effort-graph/read-cache/` | Generation-keyed; gitignore it. `cache prune`: >24h old deleted, then oldest-first to ≤100 MiB. | +| Journal | fixed | `/.journal/` | Writer-owned; gitignored. Never edit. | +| Strict timeout | `--timeout-ms` per read | 3000 ms | | +| Page limit | `--limit` per read | 25 | Hard max 25. | + +## What not to do + +- Do not hand-edit record frontmatter or `.journal/`; bodies are freely + editable (the reindexer validates and repairs projections). +- Do not parse digest files as data feeds for other programs — they are + evidence for you to Read/grep; the envelope is the machine surface. +- Do not build polling loops around generations; strict reads wait + server-side. +- Do not model sessions/plans/agents as records — put provenance in + `produced_in` / `created_by` fields. diff --git a/packages/effort-graph/skills/effort-graph/release.json b/packages/effort-graph/skills/effort-graph/release.json new file mode 100644 index 00000000..a0d4235f --- /dev/null +++ b/packages/effort-graph/skills/effort-graph/release.json @@ -0,0 +1,6 @@ +{ + "format": 1, + "flatbreadVersion": "1.0.0-alpha.22", + "effortGraphVersion": "0.1.0-alpha.0", + "gitTag": "v1.0.0-alpha.22" +} diff --git a/packages/effort-graph/skills/effort-graph/setup.md b/packages/effort-graph/skills/effort-graph/setup.md new file mode 100644 index 00000000..71a13793 --- /dev/null +++ b/packages/effort-graph/skills/effort-graph/setup.md @@ -0,0 +1,86 @@ +# Effort Graph setup + +The canonical skill files live in this package. The repository +`.agents/skills/effort-graph/` directory is an exclusively generated +projection: do not edit it directly, and stale projected files are deleted by +`pnpm skills:sync`. + +## 1. Choose the package manager + +Use the nearest `package.json`'s `packageManager` field first. If it is absent, +inspect lockfiles. Exactly one of `package-lock.json`, `pnpm-lock.yaml`, +`yarn.lock`, or `bun.lock`/`bun.lockb` must exist. If multiple conflicting +lockfiles exist, ask the user which manager owns the project. + +For an end-user release, read `release.json` next to this file. It is the +canonical package and tag authority: use its `flatbreadVersion` and `gitTag` +values exactly. `skills-lock.json` is installation provenance/restore data only; +do not use its optional ref or version fields as release identity: + +```bash +npx skills add https://github.com/FlatbreadLabs/flatbread/tree//packages/effort-graph/skills/effort-graph --skill effort-graph +npm install --save-dev flatbread@ +``` + +Equivalent commands are: + +```bash +pnpm dlx skills add https://github.com/FlatbreadLabs/flatbread/tree//packages/effort-graph/skills/effort-graph --skill effort-graph +pnpm add -D flatbread@ + +yarn dlx skills add https://github.com/FlatbreadLabs/flatbread/tree//packages/effort-graph/skills/effort-graph --skill effort-graph +yarn add -D flatbread@ + +bunx skills add https://github.com/FlatbreadLabs/flatbread/tree//packages/effort-graph/skills/effort-graph --skill effort-graph +bun add -d flatbread@ +``` + +Do not use a floating branch, `latest`, or a guessed version. When dogfooding +the Flatbread monorepo, use its workspace `flatbread` binary and do not install +Flatbread from npm. + +## 2. Review the configuration + +Add the exports through the public `flatbread` facade and preserve existing +content entries: + +```js +import { + defineConfig, + sourceFilesystem, + transformerMarkdown, + effortGraphContent, +} from 'flatbread'; + +export default defineConfig({ + source: sourceFilesystem(), + transformer: transformerMarkdown(), + content: [ + // existing entries + ...effortGraphContent(), // or effortGraphContent('path/to/graph') + ], +}); +``` + +Add these entries to `.gitignore` (using the selected graph root): + +```gitignore +**/.flatbread-efforts/.journal/ +**/.flatbread/effort-graph/read-cache/ +``` + +For a custom root, replace `.flatbread-efforts` with that root. Review both +edits before saving; the bootstrap command never creates or rewrites them. + +## 3. Verify activation + +```bash +flatbread effort bootstrap +flatbread effort bootstrap --verify +``` + +The second command must print `{"status":"ready",...}` and exit successfully. +On resume, begin with `flatbread effort list --status active`, then use bounded +effort-scoped reads. Capture mutation `generation` tokens and use +`--strict-min-generation` for immediate read-after-write checks; never implement +client polling loops. Semantic changes go through `flatbread effort write`. diff --git a/packages/effort-graph/skills/effort-modeling/CONTEXT-FORMAT.md b/packages/effort-graph/skills/effort-modeling/CONTEXT-FORMAT.md new file mode 100644 index 00000000..6f0a6461 --- /dev/null +++ b/packages/effort-graph/skills/effort-modeling/CONTEXT-FORMAT.md @@ -0,0 +1,14 @@ +# Context glossary format + +`CONTEXT.md` is a glossary, not a design document. Add terms as concise, +domain-specific definitions: + +```md +### Canonical term + +A precise definition in the project's language. State what it excludes when +that prevents a common ambiguity. +``` + +Do not put implementation plans, alternatives, or commitments here. Journal +those in Effort Graph records. diff --git a/packages/effort-graph/skills/effort-modeling/DECISION-BODY.md b/packages/effort-graph/skills/effort-modeling/DECISION-BODY.md new file mode 100644 index 00000000..4286ba4c --- /dev/null +++ b/packages/effort-graph/skills/effort-modeling/DECISION-BODY.md @@ -0,0 +1,29 @@ +# Long-form Decision bodies + +Use this structure only when the decision needs durable rationale. Omit empty +sections; the title and body should stay readable in a source markdown file. + +```md +## Context + +What makes this decision necessary now? + +## Decision + +What are we committing to? + +## Alternatives considered + +- **Option:** Why it was not chosen. + +## Consequences + +What becomes easier, harder, required, or intentionally deferred? + +## Reversal criteria + +What evidence would justify revisiting this? +``` + +The body belongs to the Decision record. Cite related Findings, Constraints, +Risks, and Issues through `derives_from` when creating it. diff --git a/packages/effort-graph/skills/effort-modeling/SKILL.md b/packages/effort-graph/skills/effort-modeling/SKILL.md new file mode 100644 index 00000000..38a29955 --- /dev/null +++ b/packages/effort-graph/skills/effort-modeling/SKILL.md @@ -0,0 +1,60 @@ +--- +name: effort-modeling +description: Sharpen a project's vocabulary and planning through one-question-at-a-time grilling, then journal Decisions, Constraints, Findings, Issues, and Risks into the Flatbread Effort Graph. Use when a plan needs durable reasoning instead of ADRs. +disable-model-invocation: true +--- + +# Effort modeling + +Use this discipline while a plan or design is being shaped. Store planning +records in the Effort Graph. Keep project terms in a glossary such as +`CONTEXT.md` or `docs/glossary.md`. + +## Resume before asking + +From the project root, use the `effort-graph` skill's bounded reads: + +1. `flatbread effort list --status active` +2. For the relevant Effort, inspect open Issues and blocking Decisions. +3. Read a record or digest when a prior conclusion affects the question. + +Look up facts in code instead of asking. Put decisions to the user; do not +answer them autonomously. + +## During the grill + +Ask one question at a time. Offer a recommendation, wait for the user's +answer, and resolve prerequisite choices before dependent ones. + +- Challenge a term that conflicts with the glossary. +- Sharpen vague or overloaded terms. +- Use concrete edge cases to test relationships and scope. +- Compare a claim about behavior with the code and surface contradictions. +- Update the relevant project glossary when vocabulary resolves. Keep it free + of implementation details and planning rationale. + +## Journal the right speech act + +Use `flatbread effort write` for the current Effort, following +[the Effort Graph reference](../effort-graph/reference.md): + +- **Finding** — evidence about code, users, or runtime behavior. +- **Issue** — a question, defect, gap, or blocker needing attention. +- **Constraint** — a sticky hard or soft boundary. +- **Risk** — a prospective negative outcome with likelihood and severity. +- **Decision** — a proposed or accepted commitment among alternatives. + +Record a Decision when it is hard to reverse, surprising without context, and +the result of a real trade-off. Create it as proposed while the user is still +deciding; call `AcceptDecision` only after they commit. Always pass +`"rejectSiblings": false` unless deliberately closing every competing proposal. + +Use the long-form body template in [DECISION-BODY.md](./DECISION-BODY.md) when +the rationale would otherwise be lost. The body is the durable explanation; +do not create an ADR alongside it. + +## Finish + +Capture accepted Decisions, unresolved Issues, and material Findings before +ending the session. For an immediate verification read, use the generation +returned by the mutation with `--strict-min-generation`. diff --git a/packages/effort-graph/skills/grill-with-efforts/SKILL.md b/packages/effort-graph/skills/grill-with-efforts/SKILL.md new file mode 100644 index 00000000..18d024f4 --- /dev/null +++ b/packages/effort-graph/skills/grill-with-efforts/SKILL.md @@ -0,0 +1,15 @@ +--- +name: grill-with-efforts +description: Run a relentless one-question-at-a-time planning interview that sharpens vocabulary and journals durable reasoning into the Flatbread Effort Graph. Use when a plan is fuzzy and needs an Effort Graph trail instead of ADRs. +disable-model-invocation: true +--- + +# Grill with efforts + +Run a one-question-at-a-time grilling session using +[effort-modeling](../effort-modeling/SKILL.md). + +Offer a recommended answer for each decision, wait for the user's response, +and resolve dependent choices in order. Explore the codebase for facts, but +leave choices to the user. Do not implement the plan until shared understanding +is confirmed. diff --git a/packages/effort-graph/src/__tests__/digest.test.ts b/packages/effort-graph/src/__tests__/digest.test.ts new file mode 100644 index 00000000..2ccbd5ac --- /dev/null +++ b/packages/effort-graph/src/__tests__/digest.test.ts @@ -0,0 +1,201 @@ +import test from 'ava'; +import { mkdtemp, readFile, stat } from 'node:fs/promises'; +import { join } from 'node:path'; +import { tmpdir } from 'node:os'; +import { renderDigest } from '../digest.js'; + +test('renderDigest is deterministic and reuses the atomic cache artifact', async (t) => { + const cacheRoot = await mkdtemp(join(tmpdir(), 'eg-digest-')); + const longBody = Array.from({ length: 20 }, (_, i) => `line ${i}`).join('\n'); + const input = { + query: { type: 'getRecord', id: 'dec-one--0123456789abcdef' }, + queryHash: 'hash', + generation: '4', + consistency: { mode: 'eventual' as const, min_generation: null }, + cacheRoot, + fullBody: true, + records: [ + { + id: 'dec-one--0123456789abcdef', + kind: 'decision' as const, + path: 'decisions/one.md', + frontmatter: { + id: 'dec-one--0123456789abcdef', + title: 'One', + state: 'proposed', + }, + body_excerpt: longBody, + relations: { derives_from: ['iss-one--0123456789abcdef'] }, + }, + ], + edges: [ + { + from_id: 'dec-one--0123456789abcdef', + relation: 'derives_from' as const, + to_id: 'iss-one--0123456789abcdef', + }, + ], + }; + const first = await renderDigest(input); + const bytes = await readFile(first.artifact_path); + const second = await renderDigest(input); + t.deepEqual(first, second); + t.is(await stat(first.artifact_path).then((x) => x.isFile()), true); + const digest = bytes.toString(); + t.true(digest.includes(longBody)); + t.false(digest.includes('[…truncated]')); +}); + +test('renderDigest excerpts long bodies for non-getRecord queries', async (t) => { + const cacheRoot = await mkdtemp(join(tmpdir(), 'eg-digest-excerpt-')); + const longBody = Array.from({ length: 20 }, (_, i) => `line ${i}`).join('\n'); + const result = await renderDigest({ + query: { type: 'listRecords', effort: 'eff-one--0123456789abcdef' }, + queryHash: 'list', + generation: '4', + consistency: { mode: 'eventual' as const, min_generation: null }, + cacheRoot, + records: [ + { + id: 'dec-one--0123456789abcdef', + kind: 'decision' as const, + path: 'decisions/one.md', + frontmatter: { + id: 'dec-one--0123456789abcdef', + title: 'One', + state: 'proposed', + }, + body_excerpt: longBody, + relations: {}, + }, + ], + edges: [], + }); + const digest = await readFile(result.artifact_path, 'utf8'); + t.true(digest.includes('[…truncated]')); + t.false(digest.includes('line 19')); +}); + +test('renderDigest fullBody byte-cap miss does not fake-full with excerpt', async (t) => { + const cacheRoot = await mkdtemp(join(tmpdir(), 'eg-digest-full-bytes-')); + const hugeBody = 'x'.repeat(70 * 1024); + const result = await renderDigest({ + query: { type: 'getRecord', id: 'dec-huge--0123456789abcdef' }, + queryHash: 'huge', + generation: '4', + consistency: { mode: 'eventual' as const, min_generation: null }, + cacheRoot, + fullBody: true, + records: [ + { + id: 'dec-huge--0123456789abcdef', + kind: 'decision' as const, + path: 'decisions/huge.md', + frontmatter: { title: 'Huge' }, + body_excerpt: hugeBody, + relations: {}, + }, + ], + edges: [], + }); + const digest = await readFile(result.artifact_path, 'utf8'); + t.true(digest.includes('complete: false')); + t.true(digest.includes('cap_reasons')); + t.true(digest.includes('body exceeded digest byte cap')); + t.true(digest.includes('decisions/huge.md')); + t.false(digest.includes('[…truncated]')); + t.false(digest.includes(hugeBody.slice(0, 100))); +}); + +test('renderDigest refreshes same-generation cache bytes when records change', async (t) => { + const cacheRoot = await mkdtemp(join(tmpdir(), 'eg-digest-refresh-')); + const input = { + query: { type: 'getRecord', id: 'dec-one--0123456789abcdef' }, + queryHash: 'same-query', + generation: '4', + consistency: { mode: 'eventual' as const, min_generation: null }, + cacheRoot, + edges: [], + records: [ + { + id: 'dec-one--0123456789abcdef', + kind: 'decision' as const, + path: 'decisions/one.md', + frontmatter: { title: 'One' }, + body_excerpt: 'original body', + relations: {}, + }, + ], + }; + const first = await renderDigest(input); + const second = await renderDigest({ + ...input, + records: [{ ...input.records[0], body_excerpt: 'updated body' }], + }); + t.not(first.artifact_sha256, second.artifact_sha256); + t.true( + (await readFile(second.artifact_path, 'utf8')).includes('updated body') + ); +}); + +test('pagination is incomplete without adding a cap reason', async (t) => { + const cacheRoot = await mkdtemp(join(tmpdir(), 'eg-digest-page-')); + const result = await renderDigest({ + query: { type: 'listEfforts', page: { limit: 1 } }, + queryHash: 'page', + generation: '4', + consistency: { mode: 'eventual' as const, min_generation: null }, + cacheRoot, + edges: [], + records: [ + { + id: 'eff-one--0123456789abcdef', + kind: 'effort' as const, + path: 'efforts/one.md', + frontmatter: { title: 'One' }, + body_excerpt: '', + relations: {}, + }, + ], + totalKnown: 2, + hasMore: true, + nextCursor: 'next', + }); + const digest = await readFile(result.artifact_path, 'utf8'); + t.true(digest.includes('complete: false')); + t.true(digest.includes('"total_known":2')); + t.false(digest.includes('cap_reasons')); + t.true(result.summary.includes('incomplete: pagination')); + t.true(result.page.has_more); + t.is(result.page.next_cursor, 'next'); +}); + +test('byte caps keep complete unicode sections and rows', async (t) => { + const cacheRoot = await mkdtemp(join(tmpdir(), 'eg-digest-bytes-')); + const result = await renderDigest({ + query: { type: 'getRecord', id: 'dec-large--0123456789abcdef' }, + queryHash: 'large', + generation: '4', + consistency: { mode: 'eventual' as const, min_generation: null }, + cacheRoot, + records: Array.from({ length: 25 }, (_, index) => ({ + id: `dec-large-${index}--0123456789abcdef`, + kind: 'decision' as const, + path: `decisions/large-${index}.md`, + frontmatter: { title: '巨大'.repeat(3000) }, + body_excerpt: '界'.repeat(600), + relations: {}, + })), + edges: [ + { + from_id: 'dec-large--0123456789abcdef', + relation: 'derives_from' as const, + to_id: 'iss-large--0123456789abcdef', + }, + ], + }); + const bytes = await readFile(result.artifact_path); + t.true(bytes.byteLength <= 64 * 1024); + t.true(bytes.toString().endsWith('\n')); + t.true(bytes.toString().includes('cap_reasons')); +}); diff --git a/packages/effort-graph/src/__tests__/pack-skills.test.js b/packages/effort-graph/src/__tests__/pack-skills.test.js new file mode 100644 index 00000000..3fd2a324 --- /dev/null +++ b/packages/effort-graph/src/__tests__/pack-skills.test.js @@ -0,0 +1,82 @@ +import test from 'ava'; +import { + verifyPackPayload, + verifyReleaseIdentity, +} from '../../scripts/pack-skills.mjs'; + +const canonicalFiles = [ + 'skills/effort-graph/release.json', + 'skills/effort-graph/SKILL.md', + 'skills/effort-graph/reference.md', + 'skills/effort-graph/setup.md', +]; +const release = JSON.stringify({ + format: 1, + flatbreadVersion: '1.0.0-alpha.22', + effortGraphVersion: '0.1.0-alpha.0', + gitTag: 'v1.0.0-alpha.22', +}); +const canonicalTexts = canonicalFiles.map((path) => ({ + path, + text: path.endsWith('release.json') ? release : 'safe', +})); +const packageVersions = { + flatbreadVersion: '1.0.0-alpha.22', + effortGraphVersion: '0.1.0-alpha.0', +}; + +test('pack verification accepts canonical skills and release identity', (t) => { + t.notThrows(() => + verifyPackPayload( + [{ files: canonicalFiles.map((path) => ({ path })) }], + canonicalFiles, + canonicalTexts, + packageVersions + ) + ); +}); + +test('pack verification rejects missing canonical skill files', (t) => { + const error = t.throws(() => + verifyPackPayload( + [{ files: [{ path: canonicalFiles[0] }] }], + canonicalFiles, + canonicalTexts, + packageVersions + ) + ); + t.true(error.message.includes(canonicalFiles[1])); +}); + +test('pack verification rejects monorepo-only CLI invocations', (t) => { + const error = t.throws(() => + verifyPackPayload( + [{ files: canonicalFiles.map((path) => ({ path })) }], + canonicalFiles, + [ + ...canonicalTexts, + { + path: canonicalFiles[1], + text: 'node packages/flatbread/bin/flatbread.js', + }, + ], + packageVersions + ) + ); + t.true(error.message.includes(canonicalFiles[1])); +}); + +test('pack verification rejects release identity drift', (t) => { + const error = t.throws(() => + verifyReleaseIdentity( + [ + { + path: canonicalFiles[0], + text: release.replace('alpha.22', 'alpha.23'), + }, + ], + packageVersions + ) + ); + t.regex(error.message, /release\.json/); +}); diff --git a/packages/effort-graph/src/__tests__/read.test.ts b/packages/effort-graph/src/__tests__/read.test.ts new file mode 100644 index 00000000..5d0192fe --- /dev/null +++ b/packages/effort-graph/src/__tests__/read.test.ts @@ -0,0 +1,49 @@ +import test from 'ava'; +import { + canonicalizeReadQuery, + parseGenerationToken, + readQueryHash, +} from '../index.js'; + +test('canonicalizeReadQuery removes undefined values and normalizes sets', (t) => { + t.deepEqual( + canonicalizeReadQuery({ + z: ['b', 'a', 'b'], + nested: { optional: undefined, value: 1 }, + }), + { nested: { value: 1 }, z: ['a', 'b'] } + ); +}); + +test('readQueryHash is stable for equivalent query shapes', (t) => { + t.is( + readQueryHash({ kinds: ['risk', 'issue', 'risk'] }), + readQueryHash({ kinds: ['issue', 'risk'] }) + ); +}); + +test('readQueryHash includes page cursors while keeping set arrays stable', (t) => { + const base = { type: 'listEfforts', status: ['paused', 'active'] }; + t.not( + readQueryHash({ ...base, page: { limit: 1, cursor: 'page-1' } }), + readQueryHash({ ...base, page: { limit: 1, cursor: 'page-2' } }) + ); + t.is( + readQueryHash({ ...base, page: { limit: 1, cursor: 'page-1' } }), + readQueryHash({ + type: 'listEfforts', + status: ['active', 'paused', 'active'], + page: { cursor: 'page-1', limit: 1 }, + }) + ); +}); + +test('strict generation tokens are canonical safe non-negative integers', (t) => { + for (const value of ['', '-1', '1.5', '1e3', '01', '9007199254740992']) { + t.throws(() => parseGenerationToken(value), { + message: /canonical non-negative safe integer string/, + }); + } + t.is(parseGenerationToken('0'), 0); + t.is(parseGenerationToken('42'), 42); +}); diff --git a/packages/effort-graph/src/__tests__/skills.test.ts b/packages/effort-graph/src/__tests__/skills.test.ts new file mode 100644 index 00000000..171051f7 --- /dev/null +++ b/packages/effort-graph/src/__tests__/skills.test.ts @@ -0,0 +1,61 @@ +import { execFile } from 'node:child_process'; +import { mkdtemp, mkdir, readFile, rm, writeFile } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { resolve } from 'node:path'; +import { promisify } from 'node:util'; +import test from 'ava'; + +const run = promisify(execFile); +const script = resolve('packages/effort-graph/scripts/sync-skills.mjs'); + +test('skill projection syncs, removes stale files, and checks without writing', async (t) => { + const root = await mkdtemp(resolve(tmpdir(), 'flatbread-skills-')); + const source = resolve(root, 'source'); + const destination = resolve(root, 'destination'); + try { + await mkdir(resolve(source, 'nested'), { recursive: true }); + await mkdir(resolve(destination, 'nested'), { recursive: true }); + await writeFile(resolve(source, 'nested/skill.md'), 'canonical\n'); + await writeFile(resolve(destination, 'nested/skill.md'), 'old\n'); + await writeFile(resolve(destination, 'stale.md'), 'remove me\n'); + + const check = (await t.throwsAsync( + run(process.execPath, [ + script, + '--check', + '--source', + source, + '--destination', + destination, + ]) + )) as Error & { stderr: string }; + t.regex(check.stderr, /different: nested\/skill\.md/); + t.regex(check.stderr, /stale: stale\.md/); + t.is( + await readFile(resolve(destination, 'nested/skill.md'), 'utf8'), + 'old\n' + ); + + await run(process.execPath, [ + script, + '--source', + source, + '--destination', + destination, + ]); + t.is( + await readFile(resolve(destination, 'nested/skill.md'), 'utf8'), + 'canonical\n' + ); + await run(process.execPath, [ + script, + '--check', + '--source', + source, + '--destination', + destination, + ]); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); diff --git a/packages/effort-graph/src/digest.ts b/packages/effort-graph/src/digest.ts new file mode 100644 index 00000000..5c58d9d0 --- /dev/null +++ b/packages/effort-graph/src/digest.ts @@ -0,0 +1,373 @@ +import { createHash, randomUUID } from 'node:crypto'; +import { mkdir, open, readFile, rename, writeFile } from 'node:fs/promises'; +import { dirname, join } from 'node:path'; +import type { PrimitiveKind } from './types.js'; + +export type ReadRelation = + | 'derives_from' + | 'supersedes' + | 'superseded_by' + | 'invalidates' + | 'invalidated_by' + | 'rejected_by' + | 'mitigated_by' + | 'resolved_by' + | 'evidence'; + +export interface ReadRecord { + id: string; + kind: PrimitiveKind; + path: string; + frontmatter: Record; + body_excerpt: string; + relations: Partial>; +} + +export interface ReadEdge { + from_id: string; + relation: ReadRelation; + to_id: string; +} + +export interface ReadEnvelope { + summary: string; + artifact_path: string; + artifact_sha256: string; + served_generation: string; + consistency: { mode: 'eventual' | 'strict'; min_generation: string | null }; + page: { returned: number; has_more: boolean; next_cursor: string | null }; + hints: string[]; +} + +export interface DigestInput { + query: Record; + queryHash: string; + cursor?: string | null; + nextCursor?: string | null; + generation: string; + consistency: ReadEnvelope['consistency']; + records: readonly ReadRecord[]; + edges: readonly ReadEdge[]; + cacheRoot: string; + hasMore?: boolean; + totalKnown?: number; + hints?: string[]; + checkpointLines?: string[]; + anomaly?: string; + relatedRecords?: readonly ReadRecord[]; + /** When true, primary record bodies are rendered in full (getRecord path). */ + fullBody?: boolean; +} + +const CAP_RECORDS = 25; +const CAP_EDGES = 50; +const CAP_BYTES = 64 * 1024; +const FRONTMATTER_KEYS = [ + 'id', + 'effort', + 'title', + 'kind', + 'status', + 'state', + 'created_at', + 'slug', + 'produced_in', + 'created_by', + 'derives_from', + 'supersedes', + 'superseded_by', + 'invalidates', + 'invalidated_by', + 'resolved_by', + 'rejected_by', + 'mitigated_by', + 'evidence', +]; + +function scalar(value: unknown): string { + if (value === null || value === undefined) return 'null'; + if (typeof value === 'string') return JSON.stringify(value); + return JSON.stringify(value); +} + +function yamlHeader(input: DigestInput, complete: boolean, reasons: string[]) { + const query = JSON.stringify(input.query); + return [ + '---', + `query: ${query}`, + `query_hash: ${input.queryHash}`, + `cursor: ${scalar(input.cursor ?? null)}`, + `served_generation: ${JSON.stringify(input.generation)}`, + `consistency: ${JSON.stringify(input.consistency)}`, + `primary: ${JSON.stringify({ + returned: Math.min(input.records.length, CAP_RECORDS), + total_known: input.totalKnown ?? input.records.length, + has_more: Boolean(input.hasMore), + })}`, + `complete: ${complete}`, + `caps: ${JSON.stringify({ + primary_records: CAP_RECORDS, + relation_hops: 1, + displayed_edges: CAP_EDGES, + bytes: CAP_BYTES, + })}`, + ...(reasons.length + ? [`cap_reasons: ${JSON.stringify(reasons.sort())}`] + : []), + '---', + ].join('\n'); +} + +function normalizeBody(body: string): string { + return body.replace(/\r\n?/g, '\n'); +} + +function excerpt(body: string): string { + const normalized = normalizeBody(body); + const lines = normalized.split('\n').slice(0, 12); + let result = lines.join('\n'); + let truncated = lines.length < normalized.split('\n').length; + if ([...result].length > 600) { + result = [...result].slice(0, 600).join(''); + truncated = true; + } + return truncated ? `${result}[…truncated]` : result; +} + +function byteCapBodyBanner(record: ReadRecord): string { + return [ + `> Body exceeded digest byte cap (${CAP_BYTES} bytes).`, + `Source path: \`${record.path}\`.`, + 'Open the source record if the full body is required.', + ].join(' '); +} + +type RecordBodyMode = 'excerpt' | 'full' | 'byte_cap_miss'; + +function renderRecordBody( + record: ReadRecord, + bodyMode: RecordBodyMode +): string { + if (bodyMode === 'byte_cap_miss') return byteCapBodyBanner(record); + if (bodyMode === 'full') return normalizeBody(record.body_excerpt); + return excerpt(record.body_excerpt); +} + +function renderRecord( + record: ReadRecord, + options: { bodyMode?: RecordBodyMode } = {} +): string { + const bodyMode = options.bodyMode ?? 'excerpt'; + const frontmatter = Object.fromEntries( + FRONTMATTER_KEYS.filter((key) => record.frontmatter[key] !== undefined).map( + (key) => [key, record.frontmatter[key]] + ) + ); + const relations = Object.keys(record.relations) + .sort() + .map( + (key) => + `- ${key}: ${JSON.stringify( + [...(record.relations[key as ReadRelation] ?? [])].sort() + )}` + ) + .join('\n'); + return [ + `### ${record.id}`, + '```yaml', + ...Object.keys(frontmatter) + .sort() + .map((key) => `${key}: ${scalar(frontmatter[key])}`), + '```', + '', + renderRecordBody(record, bodyMode), + ...(relations ? ['', 'Relations', relations] : []), + '', + ].join('\n'); +} + +function summary( + records: readonly ReadRecord[], + complete: boolean, + reasons: string[], + hasMore: boolean +): string { + const states = new Map(); + const statuses = new Map(); + for (const record of records) { + for (const [map, key] of [ + [states, 'state'], + [statuses, 'status'], + ] as const) { + const value = record.frontmatter[key]; + if (typeof value === 'string') map.set(value, (map.get(value) ?? 0) + 1); + } + } + const values = [...states, ...statuses] + .sort(([a], [b]) => a.localeCompare(b)) + .map(([key, count]) => `${key} ${count}`); + const incompleteReasons = [...reasons, ...(hasMore ? ['pagination'] : [])]; + const text = `${records.length} record${records.length === 1 ? '' : 's'}${ + values.length ? `; ${values.join(', ')}` : '' + }; ${ + complete && !hasMore + ? 'complete' + : `incomplete: ${incompleteReasons.sort().join(', ')}` + }`; + if ([...text].length <= 640) return text; + const clipped = [...text].slice(0, 640).join(''); + return `${clipped.slice(0, clipped.lastIndexOf(' '))}…`; +} + +async function durableWrite(path: string, bytes: Buffer): Promise { + await mkdir(dirname(path), { recursive: true }); + const temporary = `${path}.tmp-${randomUUID()}`; + await writeFile(temporary, bytes); + const handle = await open(temporary, 'r+'); + await handle.sync(); + await handle.close(); + await rename(temporary, path); +} + +export async function renderDigest(input: DigestInput): Promise { + const primaryBodyMode: RecordBodyMode = input.fullBody ? 'full' : 'excerpt'; + const records = [...input.records].sort((a, b) => + `${String(a.frontmatter.created_at ?? '')}\0${a.id}`.localeCompare( + `${String(b.frontmatter.created_at ?? '')}\0${b.id}` + ) + ); + const visible = records.slice(0, CAP_RECORDS); + const edges = [...input.edges] + .sort((a, b) => + `${a.relation}\0${a.from_id}\0${a.to_id}`.localeCompare( + `${b.relation}\0${b.from_id}\0${b.to_id}` + ) + ) + .slice(0, CAP_EDGES); + const reasons = [ + ...(records.length > CAP_RECORDS ? ['primary_records'] : []), + ...(input.edges.length > CAP_EDGES ? ['displayed_edges'] : []), + ]; + const checkpoints = input.checkpointLines?.length + ? ['## Lineage checkpoints', ...input.checkpointLines, ''] + : []; + const renderPrimary = ( + record: ReadRecord, + bodyMode: RecordBodyMode = primaryBodyMode + ) => renderRecord(record, { bodyMode }); + // Related / browse neighbors stay excerpted even when primary bodies are full. + const renderRelated = (record: ReadRecord) => + renderRecord(record, { bodyMode: 'excerpt' }); + let markdown = [ + yamlHeader(input, reasons.length === 0 && !input.hasMore, reasons), + ...(input.anomaly ? [`> anomaly: ${input.anomaly}`, ''] : []), + '# Effort Graph read', + '## Index', + ...visible.map((record) => `- [\`${record.id}\`](#${record.id})`), + '## Records', + ...visible.map((record) => renderPrimary(record)), + ...(input.relatedRecords?.length + ? ['## Related records', ...input.relatedRecords.map(renderRelated)] + : []), + ...checkpoints, + '## Displayed edges', + '| From | Relation | To |', + '| --- | --- | --- |', + ...edges.map( + (edge) => `| ${edge.from_id} | ${edge.relation} | ${edge.to_id} |` + ), + '', + ].join('\n'); + if (Buffer.byteLength(markdown) > CAP_BYTES) { + reasons.push('bytes'); + // Full-body digests must not silently fall back to the 600/12 excerpt. + // Prefer a visible byte-cap miss banner over a fake "full" body. + const overflowBodyMode: RecordBodyMode = input.fullBody + ? 'byte_cap_miss' + : 'excerpt'; + const anomaly = + input.fullBody && input.anomaly + ? `${input.anomaly}; body exceeded digest byte cap` + : input.fullBody + ? 'body exceeded digest byte cap' + : input.anomaly; + const header = [ + yamlHeader(input, false, reasons), + ...(anomaly ? [`> anomaly: ${anomaly}`, ''] : []), + '# Effort Graph read', + '## Index', + ...visible.map((record) => `- [\`${record.id}\`](#${record.id})`), + '## Records', + ]; + const sections: string[] = []; + for (const record of visible) { + const section = renderPrimary(record, overflowBodyMode); + const candidate = [...header, ...sections, section, ''].join('\n'); + if ( + Buffer.byteLength([...candidate, '## Displayed edges'].join('\n')) > + CAP_BYTES + ) + break; + sections.push(section); + } + const edgeHeader = [ + ...header, + ...sections, + '## Displayed edges', + '| From | Relation | To |', + '| --- | --- | --- |', + ]; + const edgeRows: string[] = []; + for (const edge of edges) { + const candidate = [ + ...edgeHeader, + ...edgeRows, + `| ${edge.from_id} | ${edge.relation} | ${edge.to_id} |`, + '', + ].join('\n'); + if (Buffer.byteLength(candidate) > CAP_BYTES) break; + edgeRows.push(`| ${edge.from_id} | ${edge.relation} | ${edge.to_id} |`); + } + markdown = [...edgeHeader, ...edgeRows, ''].join('\n'); + } + const path = join( + input.cacheRoot, + 'read-cache', + input.generation, + `${input.queryHash}.md` + ); + let bytes: Buffer; + const renderedBytes = Buffer.from(markdown); + try { + bytes = await readFile(path); + if (!bytes.equals(renderedBytes)) { + bytes = renderedBytes; + await durableWrite(path, bytes); + } + } catch { + bytes = renderedBytes; + await durableWrite(path, bytes); + } + const ids = visible.map((record) => record.id); + const envelope: ReadEnvelope = { + summary: summary( + visible, + reasons.length === 0 && !input.hasMore, + reasons, + Boolean(input.hasMore) + ), + artifact_path: path, + artifact_sha256: createHash('sha256').update(bytes).digest('hex'), + served_generation: input.generation, + consistency: input.consistency, + page: { + returned: visible.length, + has_more: Boolean(input.hasMore || records.length > CAP_RECORDS), + next_cursor: input.hasMore ? input.nextCursor ?? null : null, + }, + hints: ( + input.hints ?? ids.slice(0, 10).map((id) => `getRecord("${id}")`) + ).slice(0, 10), + }; + return envelope; +} diff --git a/packages/effort-graph/src/index.ts b/packages/effort-graph/src/index.ts index b83804bb..9e97bced 100644 --- a/packages/effort-graph/src/index.ts +++ b/packages/effort-graph/src/index.ts @@ -12,3 +12,20 @@ export { acquireWriterLock } from './lock.js'; export { recoverJournal } from './journal.js'; export * from './live.js'; export * from './journalBarrier.js'; +export * from './digest.js'; +export { + READ_RELATIONS, + EffortGraphConsistencyError, + EffortGraphInvalidCursorError, + EffortGraphReadValidationError, + canonicalizeReadQuery, + parseGenerationToken, + pruneReadCache, + readQueryHash, +} from './read.js'; +export type { + ConsistencyErrorShape, + EffortStatus, + ReadOptions, + ReadQuery, +} from './read.js'; diff --git a/packages/effort-graph/src/read.ts b/packages/effort-graph/src/read.ts new file mode 100644 index 00000000..ff35248b --- /dev/null +++ b/packages/effort-graph/src/read.ts @@ -0,0 +1,222 @@ +import { readFile, readdir, stat, unlink } from 'node:fs/promises'; +import { createHash } from 'node:crypto'; +import { join } from 'node:path'; +export const READ_RELATIONS = [ + 'derives_from', + 'supersedes', + 'superseded_by', + 'invalidates', + 'invalidated_by', + 'rejected_by', + 'mitigated_by', + 'resolved_by', + 'evidence', +] as const; + +export interface ReadOptions { + cacheRoot: string; + consistency?: + | { mode: 'eventual' } + | { mode: 'strict'; min_generation: string; timeout_ms?: number }; +} + +export type EffortStatus = 'active' | 'paused' | 'completed' | 'abandoned'; + +export type ReadQuery = + | { type: 'getRecord'; id: string; resolve?: 'exact' | 'head' } + | { + type: 'effortRecords'; + effort_id: string; + kinds?: string[]; + where?: Record; + page?: { cursor?: string; limit?: number }; + } + | { + type: 'relations'; + effort_id: string; + from_id: string; + relations: string[]; + page?: { cursor?: string; limit?: number }; + } + | { + type: 'blockingDecisions'; + effort_id: string; + page?: { cursor?: string; limit?: number }; + } + | { + type: 'listEfforts'; + status: EffortStatus[]; + page?: { cursor?: string; limit?: number }; + }; + +export interface ConsistencyErrorShape { + error: { + code: + | 'EFFORT_GRAPH_GENERATION_WAIT_TIMEOUT' + | 'EFFORT_GRAPH_LIVE_SCHEMA_REJECTED' + | 'EFFORT_GRAPH_INVALID_GENERATION'; + message: string; + requested_generation: string; + timeout_ms?: number; + }; +} + +export class EffortGraphReadValidationError extends Error { + readonly shape: { + error: { + code: 'EFFORT_GRAPH_INVALID_GENERATION' | 'EFFORT_GRAPH_INVALID_ARGUMENT'; + message: string; + }; + }; + constructor( + code: 'EFFORT_GRAPH_INVALID_GENERATION' | 'EFFORT_GRAPH_INVALID_ARGUMENT', + message: string + ) { + super(message); + this.name = 'EffortGraphReadValidationError'; + this.shape = { error: { code, message } }; + } +} + +export class EffortGraphInvalidCursorError extends Error { + readonly shape = { + error: { code: 'EFFORT_GRAPH_INVALID_CURSOR' as const, message: '' }, + }; + constructor(message: string) { + super(message); + this.name = 'EffortGraphInvalidCursorError'; + this.shape.error.message = message; + } +} + +export class EffortGraphConsistencyError extends Error { + readonly shape: ConsistencyErrorShape; + constructor(shape: ConsistencyErrorShape) { + super(shape.error.message); + this.name = 'EffortGraphConsistencyError'; + this.shape = shape; + } +} + +async function generation(rootDir: string): Promise { + try { + const data = JSON.parse( + await readFile(join(rootDir, '.journal', 'generation.json'), 'utf8') + ); + return Number(data.generation) || 0; + } catch { + return 0; + } +} + +async function servedGeneration( + rootDir: string, + consistency: ReadOptions['consistency'] +): Promise { + const strict = consistency?.mode === 'strict' ? consistency : undefined; + const wanted = strict + ? parseGenerationToken(strict.min_generation) + : undefined; + const timeout = strict ? strict.timeout_ms ?? 3000 : 0; + const started = Date.now(); + let current = await generation(rootDir); + while ( + wanted !== undefined && + current < wanted && + Date.now() - started < timeout + ) { + await new Promise((resolve) => setTimeout(resolve, 25)); + current = await generation(rootDir); + } + if (wanted !== undefined && current < wanted) { + const shape: ConsistencyErrorShape = { + error: { + code: 'EFFORT_GRAPH_GENERATION_WAIT_TIMEOUT', + message: `Durable Effort Graph generation ${wanted} was not observed (current: ${current})`, + requested_generation: strict!.min_generation, + timeout_ms: timeout, + }, + }; + throw new EffortGraphConsistencyError(shape); + } + return String(current); +} + +export function parseGenerationToken(value: unknown): number { + if ( + typeof value !== 'string' || + !/^(0|[1-9]\d*)$/.test(value) || + !Number.isSafeInteger(Number(value)) + ) + throw new EffortGraphReadValidationError( + 'EFFORT_GRAPH_INVALID_GENERATION', + 'strict-min-generation must be a canonical non-negative safe integer string' + ); + return Number(value); +} + +function canonical(value: unknown): unknown { + if (Array.isArray(value)) return [...new Set(value.map(String))].sort(); + if (value && typeof value === 'object') + return Object.fromEntries( + Object.entries(value as Record) + .filter(([, item]) => item !== undefined) + .sort(([a], [b]) => a.localeCompare(b)) + .map(([key, item]) => [key, canonical(item)]) + ); + return value; +} + +export function canonicalizeReadQuery( + query: Record +): Record { + return canonical(query) as Record; +} + +export function readQueryHash(query: Record): string { + return createHash('sha256') + .update(JSON.stringify(canonical(query))) + .digest('hex'); +} + +export async function pruneReadCache( + cacheRoot: string, + options: { maxAgeMs?: number; maxBytes?: number } = {} +): Promise<{ deleted: number; bytes: number }> { + const maxAge = options.maxAgeMs ?? 24 * 60 * 60 * 1000; + const maxBytes = options.maxBytes ?? 100 * 1024 * 1024; + const files: { path: string; mtime: number; size: number }[] = []; + async function walk(dir: string): Promise { + let entries: any[]; + try { + entries = await readdir(dir, { withFileTypes: true }); + } catch { + return; + } + for (const entry of entries) { + const path = join(dir, entry.name); + if (entry.isDirectory()) await walk(path); + else if (entry.name.endsWith('.md')) { + const info = await stat(path); + files.push({ path, mtime: info.mtimeMs, size: info.size }); + } + } + } + await walk(join(cacheRoot, 'read-cache')); + const now = Date.now(); + let deleted = 0; + let bytes = files.reduce((sum, file) => sum + file.size, 0); + for (const file of files + .filter((item) => now - item.mtime > maxAge) + .concat( + files + .filter((item) => now - item.mtime <= maxAge) + .sort((a, b) => a.mtime - b.mtime) + )) { + if (now - file.mtime <= maxAge && bytes <= maxBytes) break; + await unlink(file.path); + deleted++; + bytes -= file.size; + } + return { deleted, bytes }; +} diff --git a/packages/effort-graph/src/schemas.ts b/packages/effort-graph/src/schemas.ts index d600d9b6..48d0172e 100644 --- a/packages/effort-graph/src/schemas.ts +++ b/packages/effort-graph/src/schemas.ts @@ -11,7 +11,7 @@ const common = { produced_in: z.string().optional(), created_by: z.string().optional(), }; -// Forward edges only; back-edges are writer-materialized projections (ADR-0004). +// Forward edges are canonical; the writer materializes reverse projections. const edges = { derives_from: id.array().optional(), supersedes: id.array().optional(), diff --git a/packages/flatbread/README.md b/packages/flatbread/README.md index 73fbfc36..c22b959f 100644 --- a/packages/flatbread/README.md +++ b/packages/flatbread/README.md @@ -16,39 +16,57 @@

-Turn flat files in Git into typed, relational content for your TypeScript app. **[GraphQL](https://graphql.org/)** and codegen are a common **read interface** for that content graph—not the only surface you can build; see [docs/positioning.md](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/positioning.md). +Turn files in Git into typed, related content for your TypeScript app. +**[GraphQL](https://graphql.org/)** and codegen are common ways to read that +content, but they are not the only options. See +[Flatbread positioning](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/positioning.md). -**Flatbread** is a Git-native relational flat-file content layer for TypeScript apps. Your repo and filesystem are the source of truth; plugins (sources, transformers, and resolvers) extend how content is loaded and shaped. +**Flatbread** reads content from your repository and file system. Plugins control +how it reads files and turns them into data. -**Who it's for:** Teams shipping TypeScript sites, internal tools, and starters who want **versioned, reviewable content** and **relationships between entries**—without standing up a CMS database or giving up ownership of where content lives. +**Who it is for:** Teams building TypeScript sites, internal tools, and starter +projects that want versioned, reviewable content and links between entries +without setting up a CMS database. -**Non-goals:** +**What Flatbread does not do:** -- Not a hosted CMS, dashboard, or authoring UI: Flatbread is a library and local workflow, not a full content-management product you log into. -- Not a general-purpose GraphQL platform or a substitute for a general-purpose database (transactions, granular access control, and high-scale multi-writer workloads are out of scope). -- Reliable live reload of content while the dev server runs is [not a supported pillar yet](https://github.com/FlatbreadLabs/flatbread/issues/65); expect to restart to pick up file changes. +- It is not a hosted CMS, dashboard, or writing UI. +- It is not a general-purpose GraphQL platform or database. Transactions, + detailed access control, and many concurrent writers are outside its scope. +- Run `flatbread start --watch` to update valid content and config changes + while you work. Changes to Flatbread packages still need their own rebuild or + restart. See the + [local development loop](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/local-dev-loop.md). -**GraphQL:** In the default toolkit, GraphQL is a common **read interface** for the content graph—schema generation and codegen are how many apps reach the data, not the definition of the product. More detail: [docs/positioning.md](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/positioning.md). +**GraphQL:** GraphQL is a common way to read Flatbread data. It does not define +the product. For more detail, see +[Flatbread positioning](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/positioning.md). -**Glossary:** Quick definitions for **collection**, **relation**, **ID**, **cardinality**, **validation**, **query interface**, and how the **generated GraphQL schema / operation types** map to those terms (GraphQL as one read path, not the whole product)—see [docs/glossary.md](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/glossary.md). +**Glossary:** Definitions for collections, relations, IDs, validation, and the +generated GraphQL types are in the +[glossary](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/glossary.md). -**Local dev loop:** Codegen watch, schema rebuild, content reload, and framework restart boundaries are documented in [docs/local-dev-loop.md](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/local-dev-loop.md). +**Local development:** Learn what updates automatically and what needs a +restart in the +[local development loop](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/local-dev-loop.md). -**Portability:** Stable JSON snapshot export is available as a core API and documented in [docs/json-export.md](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/json-export.md). +**Export:** The core API can create stable JSON snapshots. See +[JSON export](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/json-export.md). -**Ownership and exit:** Raw files, Git history, JSON/CSV exports, GraphQL introspection, and generated TypeScript all fit one portability story in [docs/data-ownership.md](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/data-ownership.md). - -**Roadmap:** Current keep/kill/iterate decisions from validation work live in [docs/roadmap.md](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/roadmap.md). +**Keeping your data:** Your files, Git history, JSON/CSV exports, GraphQL +introspection, and generated TypeScript remain available when you move away +from Flatbread. See +[data ownership](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/data-ownership.md). For contributing to this monorepo, use Node 20.19+ with pnpm 10.33.x. Runtime support for published packages is tracked by each package's own metadata. -Born out of a desire to [Gridsome](https://gridsome.org/) (or [Gatsby](https://www.gatsbyjs.com/)) anything, this project harnesses a plugin architecture to be easily customizable to fit your use cases. - ## Quickstart (posts, authors, and tags) 🚧 This project is experimental; the API may change before `v1.0`. -This repo’s **canonical first success path** is the **Next.js example** (`examples/nextjs`). It reads shared markdown under **`examples/content`** (mounted in that app as `content/` via symlink). Commands below are exact for that layout. +Start with the **Next.js example** in `examples/nextjs`. It reads shared +Markdown from `examples/content` through its `content/` symlink. The commands +below use that layout. ### 1 · What you are modeling @@ -123,7 +141,8 @@ Add **`tags`** (and any other fields) to your **`.graphql`** documents and rerun ### 3 · Run it from the repo root -Prerequisites: **Node 20.19+**, **pnpm 10.33.x** (see [CONTRIBUTING.md](CONTRIBUTING.md)). +Prerequisites: **Node 20.19+** and **pnpm 10.33.x**. See +[CONTRIBUTING.md](https://github.com/FlatbreadLabs/flatbread/blob/main/CONTRIBUTING.md). ```bash pnpm install @@ -160,7 +179,10 @@ The generated file also exposes a prototype **TypeScript read API** derived from #### Choosing a read interface -Flatbread starts with **Git-native relational content** for TypeScript apps: flat files define records, frontmatter fields, ids, and refs; `flatbread.config.js` tells Flatbread how those files become typed collections. **GraphQL is one interface over that typed model**, and the generated TypeScript read API is another app-facing surface generated from the same model. +Flatbread starts with related content files for TypeScript apps. Files define +records, frontmatter fields, IDs, and `refs`; `flatbread.config.js` tells +Flatbread how to group them into typed collections. **GraphQL** and the +generated TypeScript API are two ways for your app to read the same data. Use **GraphQL operations** when your app needs explicit query documents, custom selections, Apollo or other GraphQL clients, persisted operations, or direct access to the GraphQL endpoint. Add `.graphql` documents, include fields like **`tags`** and **`authors`**, and rerun codegen so operation types such as **`GetPostsAuthorsAndTagsQuery`** match the posts/authors/tags graph. @@ -205,21 +227,29 @@ Wire your framework so the CLI wraps dev/build (**`flatbread start`** passes thr // package.json scripts (adapt the part after `--` to your framework) { "scripts": { - "dev": "flatbread start -- next dev --turbopack", + "dev": "flatbread start --watch -- next dev --turbopack", "build": "flatbread start -- next build" } } ``` -In the Next example from **`examples/nextjs`**, **`pnpm dev`** enables HTTPS locally and pairs Next with Flatbread. The GraphQL HTTP endpoint defaults to **`http://localhost:5057/graphql`**; the Next app is on **`3000`**. **`pnpm run dev`** here is distinct from **`next start`** alone (production Next without Flatbread unless you arrange serving yourself). +In the Next.js example, **`pnpm dev`** starts Next with local HTTPS and starts +Flatbread in watch mode. The GraphQL endpoint is +**`http://localhost:5057/graphql`** and the Next app is on **`3000`**. +**`pnpm start`** runs production Next without Flatbread. ```bash pnpm dev ``` -If the server starts cleanly, Flatbread prints the **`graphql`** URL. Opening it launches Apollo Studio against the generated schema—you can iterate on queries there, then freeze them into **`.graphql`** files and rerun **`flatbread codegen`**. +When the server starts, Flatbread prints the **`graphql`** URL. Open it to use +Apollo Studio with the generated schema. You can then save queries in +**`.graphql`** files and run **`flatbread codegen`** again. -Live reload of markdown while the process runs is **[not reliable yet](https://github.com/FlatbreadLabs/flatbread/issues/65)**—restart dev after content changes. +With `--watch`, valid content and config changes update the running GraphQL +server. See the +[local development loop](https://github.com/FlatbreadLabs/flatbread/blob/main/docs/local-dev-loop.md) +for the cases that still need a rebuild or restart. ## Install Flatbread in your own repo @@ -237,7 +267,8 @@ pnpm exec flatbread init Point **`content`** entries at **your** `posts/` and **`authors/`** folders, reuse the relational ideas above, and add **`codegen`** in config when you want **`generated/graphql.ts`**. Browse [`packages`](https://github.com/FlatbreadLabs/flatbread/tree/main/packages) for plugins and resolver helpers. -More detail on the bundled example (scripts, codegen watch, troubleshooting): **`examples/nextjs/README.md`**. +More detail on the bundled example is in the +[Next.js example README](https://github.com/FlatbreadLabs/flatbread/blob/main/examples/nextjs/README.md). ## Query arguments (GraphQL read interface) @@ -367,7 +398,10 @@ Limits the number of returned entries to the specified amount. Accepts an intege ## Query from your app -Follow [Quickstart (posts, authors, and tags)](#quickstart-posts-authors-and-tags) for the relational model, **codegen**, and typed results. For framework wiring and scripts, use **[examples/nextjs](https://github.com/FlatbreadLabs/flatbread/tree/main/examples/nextjs)** or [other examples](https://github.com/FlatbreadLabs/flatbread/tree/main/examples) (for example SvelteKit). +Follow [Quickstart (posts, authors, and tags)](#quickstart-posts-authors-and-tags) +to model related content, run codegen, and get typed results. For scripts and +framework setup, use the +[Next.js example](https://github.com/FlatbreadLabs/flatbread/tree/main/examples/nextjs). ## Field overrides @@ -376,22 +410,20 @@ Field overrides allow you to define custom GraphQL types or resolvers on top of ### Example ```js -{ +const config = { content: { - ... overrides: [ { - // using the field name - field: 'name' - // the resulting type is string - // this can be a custom gql type + // The source field name. + field: 'name', + // The GraphQL type to expose. type: 'String', - // capitalize the name - resolve: name => capitalize(name) + // Capitalize the value before returning it. + resolve: (name) => capitalize(name), }, - ] - } -} + ], + }, +}; ``` ### Supported syntax for field @@ -427,4 +459,5 @@ Accepts a function which takes in field names and transforms them for the GraphQ # ☀️ Contributing -See [CONTRIBUTING.md](https://github.com/FlatbreadLabs/flatbread/blob/main/CONTRIBUTING.md) for the release workflow (bumping versions and publishing). +See [CONTRIBUTING.md](https://github.com/FlatbreadLabs/flatbread/blob/main/CONTRIBUTING.md) +for release steps, including version bumps and publishing. diff --git a/packages/flatbread/src/cli/effort.test.ts b/packages/flatbread/src/cli/effort.test.ts new file mode 100644 index 00000000..4b2ea42d --- /dev/null +++ b/packages/flatbread/src/cli/effort.test.ts @@ -0,0 +1,465 @@ +import test from 'ava'; +import { spawn } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; +import { + lstat, + mkdir, + mkdtemp, + readFile, + rm, + symlink, + unlink, + writeFile, +} from 'node:fs/promises'; +import { join, relative } from 'node:path'; +import { tmpdir } from 'node:os'; +import { effortGraphContent } from '@flatbread/effort-graph'; +import { effortGraphContent as publicEffortGraphContent } from '../index.js'; +import { + handleEffortBlockingDecisions, + handleEffortBootstrap, + handleEffortGet, + handleEffortList, + handleEffortRecords, + handleEffortWrite, + inspectEffortBootstrap, + mapEffortCliOptions, +} from './effort.js'; + +type TeardownContext = { + teardown(callback: () => void | Promise): void; +}; + +const repositoryNodeModules = fileURLToPath( + new URL('../../../../node_modules/', import.meta.url) +); + +async function createTempProject( + prefix: string, + t: TeardownContext +): Promise { + const cwd = await mkdtemp(join(tmpdir(), prefix)); + const nodeModules = join(cwd, 'node_modules'); + await symlink(repositoryNodeModules, nodeModules, 'junction'); + t.teardown(async () => { + if ((await lstat(nodeModules).catch(() => null))?.isSymbolicLink()) { + await unlink(nodeModules); + } + await rm(cwd, { recursive: true, force: true }); + }); + return cwd; +} + +function runCli( + cwd: string, + ...args: string[] +): Promise<{ code: number | null; stdout: string; stderr: string }> { + return new Promise((resolve, reject) => { + const child = spawn( + process.execPath, + [ + '--no-deprecation', + fileURLToPath(new URL('../../bin/flatbread.js', import.meta.url)), + ...args, + ], + { cwd } + ); + let stdout = ''; + let stderr = ''; + child.stdout.on('data', (chunk) => (stdout += chunk)); + child.stderr.on('data', (chunk) => (stderr += chunk)); + child.on('error', reject); + child.on('close', (code) => resolve({ code, stdout, stderr })); + }); +} + +test('public flatbread facade exposes the Effort Graph preset', (t) => { + t.deepEqual(publicEffortGraphContent(), effortGraphContent()); +}); + +test.serial( + 'spawned CLI emits JSON bootstrap errors without traces', + async (t) => { + const cwd = await createTempProject('flatbread-bootstrap-process-', t); + const result = await runCli(cwd, 'effort', 'bootstrap', '--verify'); + t.not(result.code, 0); + t.is(JSON.parse(result.stdout).status, 'action_required'); + t.is(result.stderr, ''); + } +); + +test.serial( + 'spawned CLI emits typed JSON validation errors without traces', + async (t) => { + for (const args of [ + ['effort', 'list', '--status', 'invalid'], + ['effort', 'list', '--limit', '0'], + ['effort', 'list', '--strict-min-generation', '1.5'], + ]) { + const result = await runCli(process.cwd(), ...args); + t.not(result.code, 0); + const payload = JSON.parse(result.stderr); + t.regex(payload.error.code, /^EFFORT_GRAPH_INVALID_/); + t.false(result.stderr.includes(' at ')); + t.is(result.stdout, ''); + } + } +); + +test.serial('maps sade kebab-case effort options to handler options', (t) => { + t.deepEqual( + mapEffortCliOptions({ + 'strict-min-generation': 53, + 'timeout-ms': '1250', + }), + { + strictMinGeneration: '53', + timeoutMs: 1250, + } + ); +}); + +test.serial( + 'effort handlers write and read the configured graph root', + async (t) => { + const cwd = await createTempProject('flatbread-effort-cli-', t); + for (const directory of [ + 'efforts', + 'issues', + 'findings', + 'decisions', + 'constraints', + 'risks', + ]) + await mkdir(join(cwd, '.flatbread-efforts', directory), { + recursive: true, + }); + await writeFile( + join(cwd, 'flatbread.config.js'), + `import { source } from '@flatbread/source-filesystem'; +import { transformer } from '@flatbread/transformer-markdown'; +import { effortGraphContent } from '@flatbread/effort-graph'; +export default { + source: source(), + transformer: transformer(), + content: effortGraphContent(${JSON.stringify( + relative(cwd, join(cwd, '.flatbread-efforts')) + )}), +};` + ); + const effort = await handleEffortWrite( + JSON.stringify({ type: 'CreateEffort', title: 'E', body: '' }), + { cwd } + ); + const effortId = effort.artifacts[0].id; + const issue = await handleEffortWrite( + JSON.stringify({ + type: 'WriteIssue', + effort: effortId, + title: 'Blocker', + body: '', + kind: 'blocker', + }), + { cwd } + ); + const pausedStatus = await handleEffortWrite( + JSON.stringify({ + type: 'WriteDecision', + effort: effortId, + title: 'D', + body: '', + derives_from: [issue.artifacts[0].id], + }), + { cwd } + ); + const ordinary = await handleEffortWrite( + JSON.stringify({ + type: 'WriteIssue', + effort: effortId, + title: 'Ordinary', + body: '', + kind: 'question', + }), + { cwd } + ); + const excluded = await handleEffortWrite( + JSON.stringify({ + type: 'WriteDecision', + effort: effortId, + title: 'Excluded', + body: '', + derives_from: [ordinary.artifacts[0].id], + }), + { cwd } + ); + const otherEffort = await handleEffortWrite( + JSON.stringify({ type: 'CreateEffort', title: 'Other', body: '' }), + { cwd } + ); + const foreign = await handleEffortWrite( + JSON.stringify({ + type: 'WriteDecision', + effort: otherEffort.artifacts[0].id, + title: 'Foreign', + body: '', + }), + { cwd } + ); + const envelope = await handleEffortBlockingDecisions(effortId, { + cwd, + strictMinGeneration: '7', + }); + t.is(envelope.served_generation, '7'); + t.truthy(await readFile(envelope.artifact_path, 'utf8')); + t.is(envelope.page.returned, 1); + const digest = await readFile(envelope.artifact_path, 'utf8'); + t.true(digest.includes(issue.artifacts[0].id)); + t.false(digest.includes(excluded.artifacts[0].id)); + t.false(digest.includes(foreign.artifacts[0].id)); + } +); + +test.serial( + 'engine-backed get returns an empty eventual envelope for a missing record', + async (t) => { + const cwd = await createTempProject('flatbread-effort-get-', t); + for (const directory of [ + 'efforts', + 'issues', + 'findings', + 'decisions', + 'constraints', + 'risks', + ]) + await mkdir(join(cwd, '.flatbread-efforts', directory), { + recursive: true, + }); + await writeFile( + join(cwd, 'flatbread.config.js'), + `import { source } from '@flatbread/source-filesystem'; +import { transformer } from '@flatbread/transformer-markdown'; +import { effortGraphContent } from '@flatbread/effort-graph'; +export default { + source: source(), + transformer: transformer(), + content: effortGraphContent('.flatbread-efforts'), +};` + ); + const result = await handleEffortGet('dec-missing--0123456789abcdef', { + cwd, + }); + t.is(result.page.returned, 0); + t.is(result.consistency.mode, 'eventual'); + } +); + +test.serial( + 'effort get digests include the full body while records digests stay excerpted', + async (t) => { + const cwd = await createTempProject('flatbread-effort-get-full-', t); + for (const directory of [ + 'efforts', + 'issues', + 'findings', + 'decisions', + 'constraints', + 'risks', + ]) + await mkdir(join(cwd, '.flatbread-efforts', directory), { + recursive: true, + }); + await writeFile( + join(cwd, 'flatbread.config.js'), + `import { source } from '@flatbread/source-filesystem'; +import { transformer } from '@flatbread/transformer-markdown'; +import { effortGraphContent } from '@flatbread/effort-graph'; +export default { + source: source(), + transformer: transformer(), + content: effortGraphContent('.flatbread-efforts'), +};` + ); + const longBody = [ + '## Context', + ...Array.from({ length: 20 }, (_, i) => `Context line ${i}.`), + '', + '## Decision', + 'Ship full-body get digests.', + '', + '## Reversal criteria', + 'If get digests truncate normal Decision bodies again.', + ].join('\n'); + const effort = await handleEffortWrite( + JSON.stringify({ type: 'CreateEffort', title: 'Full body', body: '' }), + { cwd } + ); + const decision = await handleEffortWrite( + JSON.stringify({ + type: 'WriteDecision', + effort: effort.artifacts[0].id, + title: 'Full body decision', + body: longBody, + }), + { cwd } + ); + const getEnvelope = await handleEffortGet(decision.artifacts[0].id, { + cwd, + }); + const getDigest = await readFile(getEnvelope.artifact_path, 'utf8'); + t.true(getDigest.includes('## Reversal criteria')); + t.true(getDigest.includes('Context line 19.')); + t.false(getDigest.includes('[…truncated]')); + + const recordsEnvelope = await handleEffortRecords(effort.artifacts[0].id, { + cwd, + kinds: ['decision'], + }); + const recordsDigest = await readFile(recordsEnvelope.artifact_path, 'utf8'); + t.true(recordsDigest.includes('[…truncated]')); + t.false(recordsDigest.includes('Context line 19.')); + } +); + +test.serial( + 'effort list defaults to active and supports explicit statuses and cursors', + async (t) => { + const cwd = await createTempProject('flatbread-effort-list-', t); + for (const directory of [ + 'efforts', + 'issues', + 'findings', + 'decisions', + 'constraints', + 'risks', + ]) + await mkdir(join(cwd, '.flatbread-efforts', directory), { + recursive: true, + }); + await writeFile( + join(cwd, 'flatbread.config.js'), + `import { source } from '@flatbread/source-filesystem'; +import { transformer } from '@flatbread/transformer-markdown'; +import { effortGraphContent } from '@flatbread/effort-graph'; +export default { source: source(), transformer: transformer(), content: effortGraphContent() };` + ); + const active = await handleEffortWrite( + JSON.stringify({ type: 'CreateEffort', title: 'Active', body: '' }), + { cwd } + ); + const paused = await handleEffortWrite( + JSON.stringify({ type: 'CreateEffort', title: 'Paused', body: '' }), + { cwd } + ); + const pausedStatus = await handleEffortWrite( + JSON.stringify({ + type: 'SetEffortStatus', + effortId: paused.artifacts[0].id, + status: 'paused', + }), + { cwd } + ); + const defaults = await handleEffortList({ cwd, limit: 1 }); + t.is(defaults.page.returned, 1); + const defaultDigest = await readFile(defaults.artifact_path, 'utf8'); + t.true(defaultDigest.includes(active.artifacts[0].id)); + t.false(defaultDigest.includes(paused.artifacts[0].id)); + const explicit = await handleEffortList({ + cwd, + status: ['paused'], + }); + t.is(explicit.page.returned, 1); + t.true( + (await readFile(explicit.artifact_path, 'utf8')).includes( + paused.artifacts[0].id + ) + ); + await t.throwsAsync(() => handleEffortList({ cwd, status: ['invalid'] }), { + message: /Invalid effort status: invalid/, + }); + const firstPage = await handleEffortList({ + cwd, + limit: 1, + status: ['active', 'paused'], + }); + t.truthy(firstPage.page.next_cursor); + const secondPage = await handleEffortList({ + cwd, + limit: 1, + cursor: firstPage.page.next_cursor ?? undefined, + status: ['active', 'paused'], + }); + t.is(secondPage.page.returned, 1); + t.not(firstPage.artifact_path, secondPage.artifact_path); + const firstDigest = await readFile(firstPage.artifact_path, 'utf8'); + const secondDigest = await readFile(secondPage.artifact_path, 'utf8'); + t.not(firstDigest, secondDigest); + t.not( + firstDigest.includes(active.artifacts[0].id), + secondDigest.includes(active.artifacts[0].id) + ); + t.not( + firstDigest.includes(paused.artifacts[0].id), + secondDigest.includes(paused.artifacts[0].id) + ); + const strict = await handleEffortList({ + cwd, + status: ['active'], + strictMinGeneration: pausedStatus.generation, + }); + t.is(strict.served_generation, pausedStatus.generation); + t.truthy(active.artifacts[0].id); + } +); + +test.serial( + 'bootstrap reports missing config and preserves missing-preset config bytes', + async (t) => { + const missing = await createTempProject('flatbread-bootstrap-missing-', t); + const missingReport = await inspectEffortBootstrap(missing); + t.is(missingReport.status, 'action_required'); + t.is(missingReport.config_path, null); + const cwd = await createTempProject('flatbread-bootstrap-preset-', t); + const configPath = join(cwd, 'flatbread.config.js'); + const config = `import { source } from '@flatbread/source-filesystem'; +import { transformer } from '@flatbread/transformer-markdown'; +export default { source: source(), transformer: transformer(), content: [{ collection: 'Post', path: 'content' }] };`; + await writeFile(configPath, config); + const report = await inspectEffortBootstrap(cwd); + t.is(report.requirements[0]?.code, 'EFFORT_BOOTSTRAP_PRESET_MISSING'); + t.is(await readFile(configPath, 'utf8'), config); + } +); + +test.serial( + 'bootstrap detects a ready custom root and verify returns action-required JSON state', + async (t) => { + const cwd = await createTempProject('flatbread-bootstrap-ready-', t); + await writeFile( + join(cwd, 'flatbread.config.js'), + `import { source } from '@flatbread/source-filesystem'; +import { transformer } from '@flatbread/transformer-markdown'; +import { effortGraphContent } from '@flatbread/effort-graph'; +export default { source: source(), transformer: transformer(), content: effortGraphContent('memory/graph') };` + ); + await writeFile( + join(cwd, '.gitignore'), + '**/memory/graph/.journal/\n**/.flatbread/effort-graph/read-cache/\n' + ); + const ready = await inspectEffortBootstrap(cwd); + t.deepEqual(ready, { + status: 'ready', + config_path: 'flatbread.config.js', + graph_root: 'memory/graph', + requirements: [], + }); + const previous = process.exitCode; + process.exitCode = undefined; + const action = await handleEffortBootstrap({ + cwd: await createTempProject('flatbread-bootstrap-verify-', t), + verify: true, + }); + t.is(action.status, 'action_required'); + t.true(process.exitCode === 1); + process.exitCode = previous; + } +); diff --git a/packages/flatbread/src/cli/effort.ts b/packages/flatbread/src/cli/effort.ts new file mode 100644 index 00000000..f642b9d0 --- /dev/null +++ b/packages/flatbread/src/cli/effort.ts @@ -0,0 +1,513 @@ +import { readFile, readdir } from 'node:fs/promises'; +import { resolve } from 'node:path'; +import { loadConfig } from '@flatbread/config'; +import { + pruneReadCache, + createEffortGraphWriter, + EffortGraphMutationSchema, + EffortGraphReadValidationError, + parseGenerationToken, + findEffortGraphContentRoot, + type EffortStatus, + type ReadEnvelope, +} from '@flatbread/effort-graph'; +import { + blockingDecisions, + effortRecords, + relations, + getRecord, + listEfforts, +} from '../effort/read.js'; +import type { PrimitiveKind, ReadRelation } from '@flatbread/effort-graph'; + +export interface EffortCliOptions { + cwd?: string; + strictMinGeneration?: string; + timeoutMs?: number; + resolve?: 'exact' | 'head'; + kinds?: string[]; + state?: string[]; + status?: string[]; + kind?: string[]; + since?: string; + until?: string; + limit?: number; + cursor?: string; + relations?: string[]; + verify?: boolean; +} + +export function mapEffortCliOptions( + options: Record +): EffortCliOptions { + return Object.fromEntries( + Object.entries({ + strictMinGeneration: + typeof options['strict-min-generation'] === 'string' + ? options['strict-min-generation'] + : typeof options['strict-min-generation'] === 'number' + ? String(options['strict-min-generation']) + : undefined, + timeoutMs: + typeof options['timeout-ms'] === 'string' + ? Number(options['timeout-ms']) + : typeof options['timeout-ms'] === 'number' + ? options['timeout-ms'] + : undefined, + resolve: options.resolve === 'head' ? 'head' : undefined, + kinds: split(options.kinds), + state: split(options.state), + status: split(options.status), + kind: split(options.kind), + since: typeof options.since === 'string' ? options.since : undefined, + until: typeof options.until === 'string' ? options.until : undefined, + limit: numberOption(options.limit), + cursor: typeof options.cursor === 'string' ? options.cursor : undefined, + relations: split(options.relations), + verify: options.verify === true ? true : undefined, + }).filter(([, value]) => value !== undefined) + ) as EffortCliOptions; +} +function split(value: unknown): string[] | undefined { + return typeof value === 'string' + ? value + .split(',') + .map((item) => item.trim()) + .filter(Boolean) + : undefined; +} +function numberOption(value: unknown): number | undefined { + const result = + typeof value === 'string' + ? Number(value) + : typeof value === 'number' + ? value + : undefined; + return result; +} + +async function rootFor(cwd: string): Promise { + const loaded = await loadConfig({ cwd }); + const root = findEffortGraphContentRoot(loaded.config?.content ?? []); + if (!root) + throw new Error( + 'No complete effort-graph preset found in flatbread.config; add effortGraphContent() to content.' + ); + return resolve(cwd, root); +} + +function consistency(options: EffortCliOptions) { + if (options.strictMinGeneration !== undefined) + parseGenerationToken(options.strictMinGeneration); + return options.strictMinGeneration !== undefined + ? { + mode: 'strict' as const, + min_generation: options.strictMinGeneration, + timeout_ms: options.timeoutMs, + } + : { mode: 'eventual' as const }; +} + +export async function handleEffortWrite( + json: string, + options: EffortCliOptions = {} +): Promise<{ + generation: string; + artifacts: { id: string; path: string; operation: string }[]; + touched: { id: string; path: string }[]; +}> { + const input = EffortGraphMutationSchema.parse(JSON.parse(json)); + const cwd = options.cwd ?? process.cwd(); + const writer = createEffortGraphWriter({ rootDir: await rootFor(cwd) }); + const result = await writer.mutate(input); + return { + generation: result.generation, + artifacts: result.artifacts.map(({ id, path, operation }) => ({ + id, + path, + operation, + })), + touched: result.touched, + }; +} + +export async function handleEffortGet( + id: string, + options: EffortCliOptions = {} +): Promise { + const cwd = options.cwd ?? process.cwd(); + return getRecord(id, { + cwd, + rootDir: await rootFor(cwd), + cacheRoot: resolve(cwd, '.flatbread/effort-graph'), + consistency: consistency(options), + resolve: options.resolve, + }); +} + +export async function handleEffortRecords( + effortId: string, + options: EffortCliOptions = {} +): Promise { + validateLimit(options.limit); + const cwd = options.cwd ?? process.cwd(); + return effortRecords(effortId, { + cwd, + rootDir: await rootFor(cwd), + cacheRoot: resolve(cwd, '.flatbread/effort-graph'), + consistency: consistency(options), + kinds: options.kinds as PrimitiveKind[] | undefined, + where: { + ...(options.state ? { state: options.state } : {}), + ...(options.status ? { status: options.status } : {}), + ...(options.kind ? { kind: options.kind } : {}), + ...(options.since || options.until + ? { created_at: { gte: options.since, lte: options.until } } + : {}), + }, + page: { limit: options.limit, cursor: options.cursor }, + }); +} + +export async function handleEffortRelations( + effortId: string, + fromId: string, + options: EffortCliOptions = {} +): Promise { + validateLimit(options.limit); + const cwd = options.cwd ?? process.cwd(); + return relations( + effortId, + fromId, + (options.relations ?? []) as ReadRelation[], + { + cwd, + rootDir: await rootFor(cwd), + cacheRoot: resolve(cwd, '.flatbread/effort-graph'), + consistency: consistency(options), + page: { limit: options.limit, cursor: options.cursor }, + } + ); +} + +export async function handleEffortBlockingDecisions( + effortId: string, + options: EffortCliOptions = {} +): Promise { + const cwd = options.cwd ?? process.cwd(); + return blockingDecisions(effortId, { + cwd, + rootDir: await rootFor(cwd), + cacheRoot: resolve(cwd, '.flatbread/effort-graph'), + consistency: consistency(options), + }); +} + +export interface EffortBootstrapReport { + readonly status: 'ready' | 'action_required'; + readonly config_path: string | null; + readonly graph_root: string; + readonly requirements: readonly { + code: string; + message: string; + recipe: string; + }[]; +} + +const CONFIG_PATTERN = /^flatbread\.config\.[mc]?[jt]s$/; +const DEFAULT_GRAPH_ROOT = '.flatbread-efforts'; + +function requirement( + code: string, + message: string, + recipe: string +): EffortBootstrapReport['requirements'][number] { + return { code, message, recipe }; +} + +function ignored( + lines: readonly string[], + candidates: readonly string[] +): boolean { + return lines.some((line) => { + const value = line.trim().replace(/^\/+/, ''); + return candidates.some( + (candidate) => + value === candidate || + value === `**/${candidate}` || + value === `/${candidate}` + ); + }); +} + +export async function inspectEffortBootstrap( + cwd = process.cwd() +): Promise { + const files = (await readdir(cwd)).filter((file) => + CONFIG_PATTERN.test(file) + ); + if (files.length > 1) + throw new Error( + JSON.stringify({ + error: { + code: 'EFFORT_BOOTSTRAP_MULTIPLE_CONFIGS', + message: `Expected exactly one valid flatbread.config.*; found ${files.join( + ', ' + )}.`, + }, + }) + ); + if (files.length === 0) { + return { + status: 'action_required', + config_path: null, + graph_root: DEFAULT_GRAPH_ROOT, + requirements: [ + requirement( + 'EFFORT_BOOTSTRAP_CONFIG_MISSING', + 'No valid flatbread.config.* exists in the working directory.', + 'Create flatbread.config.js (or .mjs/.cjs/.ts/.mts/.cts) and add the Flatbread configuration.' + ), + ], + }; + } + const configPath = files[0]; + let loaded; + try { + loaded = await loadConfig({ cwd }); + } catch (error) { + throw new Error( + JSON.stringify({ + error: { + code: 'EFFORT_BOOTSTRAP_INVALID_CONFIG', + message: error instanceof Error ? error.message : String(error), + config_path: configPath, + }, + }) + ); + } + const graphRoot = + findEffortGraphContentRoot(loaded.config?.content ?? []) ?? + DEFAULT_GRAPH_ROOT; + const requirements: EffortBootstrapReport['requirements'][number][] = []; + if (!findEffortGraphContentRoot(loaded.config?.content ?? [])) + requirements.push( + requirement( + 'EFFORT_BOOTSTRAP_PRESET_MISSING', + 'The config does not contain a complete effortGraphContent preset.', + "Import { effortGraphContent } from 'flatbread' and preserve existing content with content: [...(existingContent ?? []), ...effortGraphContent()]." + ) + ); + let gitignore = ''; + try { + gitignore = await readFile(resolve(cwd, '.gitignore'), 'utf8'); + } catch { + // Missing .gitignore is reported as both missing required entries. + } + const lines = gitignore.split(/\r?\n/); + if ( + !ignored(lines, [ + `${graphRoot}/.journal/`, + `${graphRoot}/.journal`, + `**/${graphRoot}/.journal/`, + ]) + ) + requirements.push( + requirement( + 'EFFORT_BOOTSTRAP_JOURNAL_IGNORE_MISSING', + `The Effort Graph journal for ${graphRoot} is not ignored.`, + `Add **/${graphRoot}/.journal/ to .gitignore.` + ) + ); + if ( + !ignored(lines, [ + '.flatbread/effort-graph/read-cache/', + '.flatbread/effort-graph/read-cache', + ]) + ) + requirements.push( + requirement( + 'EFFORT_BOOTSTRAP_CACHE_IGNORE_MISSING', + 'The derived Effort Graph read cache is not ignored.', + 'Add **/.flatbread/effort-graph/read-cache/ to .gitignore.' + ) + ); + return { + status: requirements.length ? 'action_required' : 'ready', + config_path: configPath, + graph_root: graphRoot, + requirements, + }; +} + +export async function handleEffortBootstrap( + options: EffortCliOptions = {} +): Promise { + const report = await inspectEffortBootstrap(options.cwd ?? process.cwd()); + if (options.verify && report.status === 'action_required') { + process.exitCode = 1; + } + return report; +} + +export async function handleEffortList( + options: EffortCliOptions = {} +): Promise { + validateLimit(options.limit); + const cwd = options.cwd ?? process.cwd(); + const statuses = options.status?.length + ? options.status + : (['active'] as string[]); + return listEfforts(statuses as EffortStatus[], { + cwd, + rootDir: await rootFor(cwd), + cacheRoot: resolve(cwd, '.flatbread/effort-graph'), + consistency: consistency(options), + page: { limit: options.limit, cursor: options.cursor }, + }); +} + +function validateLimit(limit: number | undefined): void { + if ( + limit === undefined || + (Number.isInteger(limit) && limit >= 1 && limit <= 25) + ) + return; + throw new EffortGraphReadValidationError( + 'EFFORT_GRAPH_INVALID_ARGUMENT', + 'page.limit must be an integer between 1 and 25' + ); +} + +export function registerEffortCommands(prog: any): void { + const hasShape = ( + value: unknown + ): value is { shape: Record } => { + if (value === null || typeof value !== 'object' || !('shape' in value)) + return false; + const candidate = value as { shape: unknown }; + return typeof candidate.shape === 'object' && candidate.shape !== null; + }; + const printResult = async (result: Promise): Promise => { + try { + console.log(JSON.stringify(await result)); + } catch (error) { + let payload: unknown; + if (hasShape(error)) payload = error.shape; + else if (error instanceof Error) { + try { + payload = JSON.parse(error.message); + } catch { + payload = { + error: { code: 'EFFORT_CLI_ERROR', message: error.message }, + }; + } + } else { + payload = { + error: { code: 'EFFORT_CLI_ERROR', message: String(error) }, + }; + } + console.error(JSON.stringify(payload)); + process.exitCode = 1; + } + }; + + prog + .command('effort write ', 'Write a validated Effort Graph mutation') + .action(async (json: string, options: Record) => + printResult(handleEffortWrite(json, mapEffortCliOptions(options))) + ); + prog + .command('effort get ', 'Read one Effort Graph record') + .option( + '--strict-min-generation ', + 'Require a durable journal generation' + ) + .option('--timeout-ms ', 'Strict-read wait timeout') + .option('--resolve ', 'Resolve exact record or supersession head') + .action(async (id: string, options: Record) => + printResult(handleEffortGet(id, mapEffortCliOptions(options))) + ); + prog + .command( + 'effort blocking-decisions ', + 'Read proposed decisions linked to open blockers' + ) + .option( + '--strict-min-generation ', + 'Require a durable journal generation' + ) + .option('--timeout-ms ', 'Strict-read wait timeout') + .action(async (effortId: string, options: Record) => + printResult( + handleEffortBlockingDecisions(effortId, mapEffortCliOptions(options)) + ) + ); + const consistencyOptions = (command: any) => + command + .option( + '--strict-min-generation ', + 'Require a durable journal generation' + ) + .option('--timeout-ms ', 'Strict-read wait timeout'); + const paginationOptions = (command: any) => + command + .option('--limit ', 'Page size') + .option('--cursor ', 'Opaque page cursor'); + const recordsOptions = (command: any) => + paginationOptions( + consistencyOptions(command) + .option('--kinds ', 'Comma-separated primitive kinds') + .option('--state ', 'Comma-separated states') + .option('--status ', 'Comma-separated statuses') + .option('--kind ', 'Comma-separated record kinds') + .option('--since ', 'Created-at lower bound') + .option('--until ', 'Created-at upper bound') + ); + recordsOptions( + prog.command('effort records ', 'Read records in an Effort') + ).action(async (effortId: string, options: Record) => + printResult(handleEffortRecords(effortId, mapEffortCliOptions(options))) + ); + paginationOptions( + consistencyOptions( + prog.command('effort list', 'List Efforts by lifecycle status') + ) + ) + .option('--status ', 'Comma-separated Effort statuses', 'active') + .action(async (options: Record) => + printResult(handleEffortList(mapEffortCliOptions(options))) + ); + prog + .command('effort bootstrap', 'Inspect Effort Graph activation requirements') + .option('--verify', 'Exit nonzero when activation is incomplete', false) + .action(async (options: Record) => + printResult(handleEffortBootstrap(mapEffortCliOptions(options))) + ); + paginationOptions( + consistencyOptions( + prog.command( + 'effort relations ', + 'Read one-hop relations' + ) + ) + ) + .option('--relations ', 'Comma-separated relation names') + .action( + async ( + effortId: string, + fromId: string, + options: Record + ) => + printResult( + handleEffortRelations(effortId, fromId, mapEffortCliOptions(options)) + ) + ); + prog + .command('effort cache prune', 'Prune derived read cache') + .action(async () => + printResult( + pruneReadCache(resolve(process.cwd(), '.flatbread/effort-graph')) + ) + ); +} diff --git a/packages/flatbread/src/cli/index.ts b/packages/flatbread/src/cli/index.ts index 75879155..5385feea 100644 --- a/packages/flatbread/src/cli/index.ts +++ b/packages/flatbread/src/cli/index.ts @@ -6,6 +6,7 @@ import { networkInterfaces, release } from 'node:os'; import orchestrateProcesses from './runner'; import initConfig from './initConfig'; import { createCodegenCommand } from '@flatbread/codegen'; +import { registerEffortCommands } from './effort'; const GRAPHQL_ENDPOINT = '/graphql'; @@ -93,6 +94,8 @@ prog }); }); +registerEffortCommands(prog); + prog.parse(process.argv, { unknown: (arg) => `Unknown option: ${arg}` }); /** diff --git a/packages/flatbread/src/effort/read.ts b/packages/flatbread/src/effort/read.ts new file mode 100644 index 00000000..df6f0bd1 --- /dev/null +++ b/packages/flatbread/src/effort/read.ts @@ -0,0 +1,714 @@ +import { FlatbreadProvider, type LoadedFlatbreadConfig } from '@flatbread/core'; +import { + canonicalizeReadQuery, + EffortGraphConsistencyError, + EffortGraphInvalidCursorError, + EffortGraphReadValidationError, + READ_RELATIONS, + readQueryHash, + renderDigest, + type ReadEdge, + type ReadEnvelope, + type ReadRecord, + type ReadRelation, + type ConsistencyErrorShape, + type PrimitiveKind, +} from '@flatbread/effort-graph'; +import { loadConfig } from '@flatbread/config'; +import { relative, resolve } from 'node:path'; + +export interface EffortProjectionOptions { + readonly cwd: string; + readonly rootDir: string; + readonly cacheRoot: string; + readonly consistency?: + | { mode: 'eventual' } + | { mode: 'strict'; min_generation: string; timeout_ms?: number }; +} + +type RawNode = Record; +type Collection = + | 'Effort' + | 'Issue' + | 'Finding' + | 'Decision' + | 'Constraint' + | 'Risk'; + +const COLLECTIONS: readonly Collection[] = [ + 'Effort', + 'Issue', + 'Finding', + 'Decision', + 'Constraint', + 'Risk', +]; +const KIND_TO_COLLECTION: Record = { + effort: 'Effort', + issue: 'Issue', + finding: 'Finding', + decision: 'Decision', + constraint: 'Constraint', + risk: 'Risk', +}; +const FRONTMATTER_FIELDS = [ + 'effort', + 'title', + 'kind', + 'status', + 'state', + 'created_at', + 'slug', + 'produced_in', + 'created_by', + 'derives_from', + 'supersedes', + 'superseded_by', + 'invalidates', + 'invalidated_by', + 'resolved_by', + 'rejected_by', + 'mitigated_by', + 'evidence', +] as const; +const RELATION_FIELDS = new Set([ + 'derives_from', + 'supersedes', + 'superseded_by', + 'invalidates', + 'invalidated_by', + 'rejected_by', + 'mitigated_by', + 'resolved_by', + 'evidence', +]); + +function plural(collection: Collection): string { + return collection === 'Effort' + ? 'Efforts' + : collection === 'Finding' + ? 'Findings' + : collection === 'Risk' + ? 'Risks' + : `${collection}s`; +} + +function rawKey(field: string): string { + return field; +} + +function collectionForId(id: string): Collection | undefined { + const prefix = id.split('-', 1)[0]; + return prefix === 'eff' + ? 'Effort' + : prefix === 'iss' + ? 'Issue' + : prefix === 'fnd' + ? 'Finding' + : prefix === 'dec' + ? 'Decision' + : prefix === 'con' + ? 'Constraint' + : prefix === 'rsk' + ? 'Risk' + : undefined; +} + +function relationIds(value: unknown): string[] { + if (Array.isArray(value)) + return value.flatMap((item) => + typeof item === 'string' + ? [item] + : item && + typeof item === 'object' && + typeof (item as RawNode).id === 'string' + ? [(item as RawNode).id as string] + : [] + ); + return typeof value === 'string' + ? [value] + : value && + typeof value === 'object' && + typeof (value as RawNode).id === 'string' + ? [(value as RawNode).id as string] + : []; +} + +function toRecord(node: RawNode, collection: Collection): ReadRecord { + const frontmatter: Record = {}; + const relations: Partial> = {}; + for (const field of FRONTMATTER_FIELDS) { + if (node[field] === undefined) continue; + const key = rawKey(field); + const ids = relationIds(node[field]); + if (field === 'effort') frontmatter[key] = ids[0]; + else if (RELATION_FIELDS.has(field)) relations[key as ReadRelation] = ids; + else frontmatter[key] = node[field]; + } + return { + id: String(node.id), + kind: collection.toLowerCase() as PrimitiveKind, + path: typeof node._path === 'string' ? node._path : '', + frontmatter, + body_excerpt: + node._content && + typeof node._content === 'object' && + typeof (node._content as RawNode).raw === 'string' + ? ((node._content as RawNode).raw as string) + : '', + relations, + }; +} + +function sortRecords(records: ReadRecord[]): ReadRecord[] { + return records.sort((a, b) => + `${String(a.frontmatter.created_at ?? '')}\0${a.id}`.localeCompare( + `${String(b.frontmatter.created_at ?? '')}\0${b.id}` + ) + ); +} + +function encodeCursor(value: Record): string { + return Buffer.from(JSON.stringify(value)).toString('base64url'); +} + +function baseCursorQuery( + query: Record +): Record { + const page = + query.page && typeof query.page === 'object' + ? (query.page as RawNode) + : undefined; + return { + ...query, + ...(page ? { page: { ...page, cursor: undefined } } : {}), + }; +} + +function decodeCursor( + cursor: string, + hash: string, + generation: string +): number { + try { + const value = JSON.parse( + Buffer.from(cursor, 'base64url').toString('utf8') + ) as RawNode; + if ( + value.v !== 1 || + value.query_hash !== hash || + value.served_generation !== generation || + !Number.isInteger(value.offset) || + (value.offset as number) < 0 + ) + throw new Error('cursor does not match query or served generation'); + return value.offset as number; + } catch (error) { + if (error instanceof EffortGraphInvalidCursorError) throw error; + throw new EffortGraphInvalidCursorError(`Invalid cursor: ${String(error)}`); + } +} + +async function generation(rootDir: string): Promise { + try { + const response = await import('node:fs/promises'); + const data = JSON.parse( + await response.readFile( + resolve(rootDir, '.journal/generation.json'), + 'utf8' + ) + ) as { generation?: number }; + return Number(data.generation) || 0; + } catch { + return 0; + } +} + +async function servedGeneration( + rootDir: string, + consistency: EffortProjectionOptions['consistency'] +): Promise { + const strict = consistency?.mode === 'strict' ? consistency : undefined; + const wanted = strict + ? parseGenerationToken(strict.min_generation) + : undefined; + const timeout = strict?.timeout_ms ?? 3000; + const started = Date.now(); + let current = await generation(rootDir); + while ( + wanted !== undefined && + current < wanted && + Date.now() - started < timeout + ) { + await new Promise((resolveWait) => setTimeout(resolveWait, 25)); + current = await generation(rootDir); + } + if (wanted !== undefined && current < wanted) { + const requestedGeneration = strict?.min_generation ?? String(wanted); + const shape: ConsistencyErrorShape = { + error: { + code: 'EFFORT_GRAPH_GENERATION_WAIT_TIMEOUT', + message: `Durable Effort Graph generation ${wanted} was not observed (current: ${current})`, + requested_generation: requestedGeneration, + timeout_ms: timeout, + }, + }; + throw new EffortGraphConsistencyError(shape); + } + return String(current); +} + +function parseGenerationToken(value: unknown): number { + if ( + typeof value !== 'string' || + !/^(0|[1-9]\d*)$/.test(value) || + !Number.isSafeInteger(Number(value)) + ) + throw new EffortGraphReadValidationError( + 'EFFORT_GRAPH_INVALID_GENERATION', + 'strict-min-generation must be a canonical non-negative safe integer string' + ); + return Number(value); +} + +function configForCwd( + config: LoadedFlatbreadConfig, + cwd: string +): LoadedFlatbreadConfig { + return { + ...config, + content: config.content.map((entry) => ({ + ...entry, + ...(entry.path + ? { + path: relative( + process.cwd(), + entry.path.startsWith('/') + ? entry.path + : entry.path.startsWith('Users/') + ? `/${entry.path}` + : resolve(cwd, entry.path) + ), + } + : {}), + })), + }; +} + +class EngineProjection { + readonly provider: FlatbreadProvider; + private fieldsPromise?: Promise>>; + + constructor(config: LoadedFlatbreadConfig) { + this.provider = new FlatbreadProvider(config); + } + + private async fields(): Promise>> { + if (!this.fieldsPromise) + this.fieldsPromise = Promise.all( + COLLECTIONS.map(async (collection) => { + const result = await this.provider.query({ + source: `{ __type(name: "${collection}") { fields { name } } }`, + }); + const names = + ( + result.data as + | { __type?: { fields?: Array<{ name: string }> } } + | undefined + )?.__type?.fields?.map(({ name }) => name) ?? []; + return [collection, new Set(names)] as const; + }) + ).then((items) => new Map(items)); + return this.fieldsPromise; + } + + async query( + collection: Collection, + filter?: Record + ): Promise { + const available = + (await this.fields()).get(collection) ?? new Set(); + const fields = [ + 'id', + '_path', + 'title', + 'kind', + 'status', + 'state', + 'created_at', + 'slug', + 'produced_in', + 'created_by', + 'derives_from', + 'invalidates', + 'invalidated_by', + 'resolved_by', + 'evidence', + ...((available.has('_content') ? ['_content { raw }'] : []) as string[]), + ].filter((field) => available.has(field.split(' ', 1)[0])); + for (const relation of [ + 'effort', + 'supersedes', + 'superseded_by', + 'rejected_by', + 'mitigated_by', + ]) + if (available.has(relation) && !fields.includes(relation)) + fields.push(`${relation} { id }`); + if (!fields.length) return []; + const document = `query($filter: JSON) { all${plural( + collection + )}(filter: $filter) { ${fields.join(' ')} } }`; + const result = await this.provider.query({ + source: document, + variableValues: { filter }, + }); + if (result.errors?.length) + throw new Error(result.errors.map((error) => error.message).join('; ')); + const nodes = + (result.data as Record | undefined)?.[ + `all${plural(collection)}` + ] ?? []; + return nodes.map((node) => toRecord(node, collection)); + } + + async one( + collection: Collection, + id: string + ): Promise { + const records = await this.query(collection, { id: { eq: id } }); + return records.find((record) => record.id === id); + } +} + +async function makeProjection( + options: EffortProjectionOptions +): Promise { + const loaded = await loadConfig({ cwd: options.cwd }); + if (!loaded.config) throw new Error('Flatbread config is not defined'); + return new EngineProjection(configForCwd(loaded.config, options.cwd)); +} + +async function render( + options: EffortProjectionOptions, + query: Record, + records: ReadRecord[], + edges: ReadEdge[], + hints: string[] = [], + extra: { + checkpointLines?: string[]; + anomaly?: string; + relatedRecords?: ReadRecord[]; + fullBody?: boolean; + } = {} +): Promise { + const served = await servedGeneration(options.rootDir, options.consistency); + const normalized = canonicalizeReadQuery({ + ...query, + page: { + ...(query.page as RawNode | undefined), + limit: (query.page as RawNode | undefined)?.limit ?? 25, + }, + consistency: options.consistency ?? { mode: 'eventual' }, + }); + const hash = readQueryHash(normalized); + const cursorHash = readQueryHash(baseCursorQuery(normalized)); + const page = query.page as { cursor?: string; limit?: number } | undefined; + const limit = page?.limit ?? 25; + if (!Number.isInteger(limit) || limit < 1 || limit > 25) + throw new EffortGraphReadValidationError( + 'EFFORT_GRAPH_INVALID_ARGUMENT', + 'page.limit must be an integer between 1 and 25' + ); + const offset = page?.cursor + ? decodeCursor(page.cursor, cursorHash, served) + : 0; + const selected = records.slice(offset, offset + limit); + const hasMore = offset + selected.length < records.length; + return renderDigest({ + query: normalized, + queryHash: hash, + generation: served, + consistency: + options.consistency?.mode === 'strict' + ? { mode: 'strict', min_generation: options.consistency.min_generation } + : { mode: 'eventual', min_generation: null }, + records: selected, + totalKnown: records.length, + edges, + cacheRoot: options.cacheRoot, + hasMore, + cursor: page?.cursor ?? null, + nextCursor: hasMore + ? encodeCursor({ + v: 1, + query_hash: cursorHash, + served_generation: served, + offset: offset + selected.length, + }) + : null, + hints, + ...extra, + }); +} + +export async function getRecord( + id: string, + options: EffortProjectionOptions & { resolve?: 'exact' | 'head' } +): Promise { + const projection = await makeProjection(options); + const collection = collectionForId(id); + if (!collection) return render(options, { type: 'getRecord', id }, [], []); + const requested = await projection.one(collection, id); + let record = requested; + const checkpoints: string[] = []; + let anomaly: string | undefined; + if (record && options.resolve === 'head') { + const seen = new Set([record.id]); + while (true) { + const currentRecord: ReadRecord = record; + const nextId: string | undefined = + currentRecord.relations.superseded_by?.[0]; + if (!nextId) break; + if ((currentRecord.relations.superseded_by?.length ?? 0) > 1) { + anomaly = 'supersession fork detected'; + break; + } + const next: ReadRecord | undefined = await projection.one( + collection, + nextId + ); + if (!next || seen.has(next.id)) { + anomaly = next ? 'supersession cycle detected' : undefined; + break; + } + seen.add(next.id); + checkpoints.unshift( + `${currentRecord.id} | ${String( + currentRecord.frontmatter.title ?? '' + )} | ${String(currentRecord.frontmatter.state ?? '')} | ${String( + currentRecord.frontmatter.created_at ?? '' + )} | superseded_by ${next.id}` + ); + record = next; + } + } + return render( + options, + { + type: 'getRecord', + id, + ...(options.resolve ? { resolve: options.resolve } : {}), + }, + requested && anomaly ? [requested] : record ? [record] : [], + [], + undefined, + { checkpointLines: checkpoints.slice(-5), anomaly, fullBody: true } + ); +} + +export async function effortRecords( + effortId: string, + options: EffortProjectionOptions & { + kinds?: PrimitiveKind[]; + where?: { + state?: string[]; + status?: string[]; + kind?: string[]; + created_at?: { gte?: string; lte?: string }; + }; + page?: { cursor?: string; limit?: number }; + } +): Promise { + const projection = await makeProjection(options); + const kinds = options.kinds?.length + ? [...new Set(options.kinds)].sort() + : ([ + 'issue', + 'finding', + 'decision', + 'constraint', + 'risk', + ] as PrimitiveKind[]); + const where = options.where ?? {}; + // The generated relation field materializes as an object, so its `eq` + // comparator does not match the stored identifier. Scalar predicates still + // execute in Flatbread; effort ownership is normalized and intersected here. + const filter: Record = {}; + if (where.state?.length) filter.state = { in: where.state }; + if (where.status?.length) filter.status = { in: where.status }; + if (where.kind?.length) filter.kind = { in: where.kind }; + if (where.created_at?.gte) filter.created_at = { gte: where.created_at.gte }; + if (where.created_at?.lte) + filter.created_at = { + ...(filter.created_at as RawNode), + lte: where.created_at.lte, + }; + const records = sortRecords( + ( + await Promise.all( + kinds.map((kind) => projection.query(KIND_TO_COLLECTION[kind], filter)) + ) + ) + .flat() + .filter((record) => record.frontmatter.effort === effortId) + ); + return render( + options, + { + type: 'effortRecords', + effort_id: effortId, + kinds, + ...(Object.keys(where).length ? { where } : {}), + page: options.page, + }, + records, + [], + [ + `blockingDecisions("${effortId}")`, + ...kinds + .slice(0, 8) + .map((kind) => `effortRecords("${effortId}", { kinds: ["${kind}"] })`), + ] + ); +} + +export async function listEfforts( + statuses: string[], + options: EffortProjectionOptions & { + page?: { cursor?: string; limit?: number }; + } +): Promise { + const allowed = new Set(['active', 'paused', 'completed', 'abandoned']); + const normalizedStatuses = [...new Set(statuses)]; + const invalid = normalizedStatuses.filter((status) => !allowed.has(status)); + if (invalid.length) + throw new EffortGraphReadValidationError( + 'EFFORT_GRAPH_INVALID_ARGUMENT', + `Invalid effort status: ${invalid.join( + ', ' + )}. Expected active, paused, completed, or abandoned.` + ); + const projection = await makeProjection(options); + const records = sortRecords( + ( + await projection.query('Effort', { + status: { in: normalizedStatuses }, + }) + ).filter((record) => + normalizedStatuses.includes(String(record.frontmatter.status)) + ) + ); + return render( + options, + { + type: 'listEfforts', + status: normalizedStatuses.sort(), + page: options.page, + }, + records, + [], + records.flatMap((record) => [ + `effortRecords("${record.id}")`, + `blockingDecisions("${record.id}")`, + ]) + ); +} + +export async function relations( + effortId: string, + fromId: string, + relationNames: ReadRelation[], + options: EffortProjectionOptions & { + page?: { cursor?: string; limit?: number }; + } +): Promise { + if ( + !relationNames.length || + relationNames.some( + (name) => !(READ_RELATIONS as readonly string[]).includes(name) + ) + ) + throw new Error('relations must contain only valid ReadRelation values'); + const projection = await makeProjection(options); + const collection = collectionForId(fromId); + const source = collection + ? await projection.one(collection, fromId) + : undefined; + if (!source || source.frontmatter.effort !== effortId) + throw new Error(`Record ${fromId} does not exist in effort ${effortId}`); + const selected = new Map(); + for (const relation of relationNames) { + for (const targetId of source.relations[relation] ?? []) { + const targetCollection = collectionForId(targetId); + if (!targetCollection) continue; + const target = await projection.one(targetCollection, targetId); + if (target?.frontmatter.effort === effortId) + selected.set(target.id, target); + } + } + const records = sortRecords([...selected.values()]); + const edges = records.flatMap((record) => + relationNames + .filter((relation) => + (source.relations[relation] ?? []).includes(record.id) + ) + .map((relation) => ({ from_id: fromId, relation, to_id: record.id })) + ); + return render( + options, + { + type: 'relations', + effort_id: effortId, + from_id: fromId, + relations: [...new Set(relationNames)].sort(), + page: options.page, + }, + records, + edges, + [`getRecord("${fromId}")`] + ); +} + +export async function blockingDecisions( + effortId: string, + options: EffortProjectionOptions & { + page?: { cursor?: string; limit?: number }; + } +): Promise { + const projection = await makeProjection(options); + const issues = ( + await projection.query('Issue', { + kind: { eq: 'blocker' }, + status: { eq: 'open' }, + }) + ).filter((issue) => issue.frontmatter.effort === effortId); + const blockerIds = new Set(issues.map((issue) => issue.id)); + const decisions = sortRecords( + ( + await projection.query('Decision', { + state: { eq: 'proposed' }, + }) + ).filter( + (decision) => + decision.frontmatter.effort === effortId && + (decision.relations.derives_from ?? []).some((id) => blockerIds.has(id)) + ) + ); + return render( + options, + { type: 'blockingDecisions', effort_id: effortId, page: options.page }, + decisions, + [], + [ + ...decisions.slice(0, 10).map((record) => `getRecord("${record.id}")`), + ...issues + .slice(0, 10) + .map( + () => + `effortRecords("${effortId}", { kinds: ["issue"], where: { kind: ["blocker"], status: ["open"] } })` + ), + ], + { relatedRecords: issues } + ); +} diff --git a/packages/flatbread/src/index.ts b/packages/flatbread/src/index.ts index f56bfe6c..4a3de262 100644 --- a/packages/flatbread/src/index.ts +++ b/packages/flatbread/src/index.ts @@ -5,3 +5,4 @@ export { default as defineConfig, loadConfig } from '@flatbread/config'; export { source as sourceFilesystem } from '@flatbread/source-filesystem'; export { transformer as transformerMarkdown } from '@flatbread/transformer-markdown'; export { transformer as transformerYaml } from '@flatbread/transformer-yaml'; +export { effortGraphContent } from '@flatbread/effort-graph'; diff --git a/packages/source-filesystem/README.md b/packages/source-filesystem/README.md index 800e06f7..e8d8433e 100644 --- a/packages/source-filesystem/README.md +++ b/packages/source-filesystem/README.md @@ -16,7 +16,7 @@ Add the source as a property of the default export within your `flatbread.config ```js // flatbread.config.js -import defineConfig from '@flatbread/config'; +import { defineConfig } from 'flatbread'; import transformer from '@flatbread/transformer-markdown'; import filesystem from '@flatbread/source-filesystem'; diff --git a/packages/transformer-markdown/README.md b/packages/transformer-markdown/README.md index bf08b264..d7ca38f6 100644 --- a/packages/transformer-markdown/README.md +++ b/packages/transformer-markdown/README.md @@ -16,7 +16,7 @@ Pair this with a compatible source plugin in your `flatbread.config.js` file: ```js // flatbread.config.js -import defineConfig from '@flatbread/config'; +import { defineConfig } from 'flatbread'; import transformer from '@flatbread/transformer-markdown'; import filesystem from '@flatbread/source-filesystem'; diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index f9ded765..72901fac 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -36,6 +36,9 @@ importers: '@ava/typescript': specifier: 3.0.1 version: 3.0.1 + '@flatbread/effort-graph': + specifier: workspace:* + version: link:packages/effort-graph '@nrwl/workspace': specifier: 14.4.3 version: 14.4.3(@swc/core@1.13.3)(eslint@7.32.0)(prettier@2.7.1)(ts-node@10.9.1(@swc/core@1.13.3)(@types/node@16.11.47)(typescript@4.7.4))(typescript@4.7.4) diff --git a/scripts/bumpVersions.test.ts b/scripts/bumpVersions.test.ts new file mode 100644 index 00000000..026fabdb --- /dev/null +++ b/scripts/bumpVersions.test.ts @@ -0,0 +1,78 @@ +import test from 'ava'; +import { + getWorkspaceDependentClosure, + markWorkspaceDependentsChanged, + validateBumpSelection, +} from './bumpVersions'; +import type { WorkspaceBumpPackage } from './bumpVersions'; + +type TestPackage = WorkspaceBumpPackage & { + changedSinceLastPublish: boolean; +}; + +const packages: TestPackage[] = [ + { + name: '@flatbread/effort-graph', + changedSinceLastPublish: true, + }, + { + name: '@flatbread/codegen', + dependencies: { '@flatbread/utils': 'workspace:*' }, + changedSinceLastPublish: true, + }, + { + name: '@flatbread/utils', + changedSinceLastPublish: false, + }, + { + name: 'flatbread', + dependencies: { + '@flatbread/effort-graph': 'workspace:*', + '@flatbread/codegen': 'workspace:^', + }, + devDependencies: { '@flatbread/utils': 'workspace:*' }, + changedSinceLastPublish: false, + } as TestPackage & { devDependencies: Record }, +]; + +test('workspace dependent closure follows production workspace dependencies', (t) => { + t.deepEqual( + [ + ...getWorkspaceDependentClosure(packages, ['@flatbread/effort-graph']), + ].sort(), + ['@flatbread/effort-graph', 'flatbread'] + ); + t.deepEqual( + [...getWorkspaceDependentClosure(packages, ['@flatbread/utils'])].sort(), + ['@flatbread/codegen', '@flatbread/utils', 'flatbread'] + ); +}); + +test('changed package propagation includes transitive public dependents', (t) => { + t.deepEqual( + markWorkspaceDependentsChanged(packages) + .filter((pkg) => pkg.changedSinceLastPublish) + .map((pkg) => pkg.name) + .sort(), + ['@flatbread/codegen', '@flatbread/effort-graph', 'flatbread'] + ); +}); + +test('selection validation reports omitted required dependents', (t) => { + t.deepEqual( + validateBumpSelection( + packages, + ['@flatbread/effort-graph', '@flatbread/codegen', 'flatbread'], + ['@flatbread/effort-graph'] + ), + ['flatbread'] + ); + t.deepEqual( + validateBumpSelection( + packages, + ['@flatbread/effort-graph', '@flatbread/codegen', 'flatbread'], + ['flatbread'] + ), + [] + ); +}); diff --git a/scripts/bumpVersions.ts b/scripts/bumpVersions.ts index 2324b77d..6a334251 100644 --- a/scripts/bumpVersions.ts +++ b/scripts/bumpVersions.ts @@ -1,7 +1,9 @@ import { execSync } from 'child_process'; +import { promises as fs } from 'node:fs'; import inquirer from 'inquirer'; import colors from 'kleur'; import path from 'node:path'; +import { fileURLToPath, pathToFileURL } from 'node:url'; import { getMonorepoPublicPackages, PathedFlatbreadPackage, @@ -12,6 +14,13 @@ type PackageChangeInfo = PathedFlatbreadPackage & { lastPublishedAt?: string | null; }; +export interface WorkspaceBumpPackage { + name: string; + dependencies?: Record; + optionalDependencies?: Record; + peerDependencies?: Record; +} + const DEBUG = Boolean(process.env.FLATBREAD_BUMP_DEBUG); function getPkgName(pkg: PathedFlatbreadPackage): string { @@ -226,35 +235,152 @@ async function detectChangedPackages(): Promise { }); } -// Discover changed packages since last publish and prompt to bump only those -const allPackages = await detectChangedPackages(); -const changedPackages = allPackages.filter((p) => p.changedSinceLastPublish); +function getWorkspaceDependencies(pkg: WorkspaceBumpPackage): string[] { + return [ + ...Object.entries(pkg.dependencies ?? {}), + ...Object.entries(pkg.optionalDependencies ?? {}), + ...Object.entries(pkg.peerDependencies ?? {}), + ] + .filter(([, range]) => range.startsWith('workspace:')) + .map(([name]) => name); +} + +export function getWorkspaceDependentClosure( + packages: readonly WorkspaceBumpPackage[], + packageNames: readonly string[] +): Set { + const closure = new Set(packageNames); + let changed = true; + + while (changed) { + changed = false; + for (const pkg of packages) { + if ( + !closure.has(pkg.name) && + getWorkspaceDependencies(pkg).some((dependency) => + closure.has(dependency) + ) + ) { + closure.add(pkg.name); + changed = true; + } + } + } + + return closure; +} -if (changedPackages.length === 0) { - console.log( - colors.bold().green('No package changes since last publish detected.') +export function markWorkspaceDependentsChanged< + T extends WorkspaceBumpPackage & { changedSinceLastPublish: boolean } +>(packages: readonly T[]): T[] { + const changedNames = packages + .filter((pkg) => pkg.changedSinceLastPublish) + .map((pkg) => pkg.name); + const closure = getWorkspaceDependentClosure(packages, changedNames); + + return packages.map((pkg) => + closure.has(pkg.name) ? { ...pkg, changedSinceLastPublish: true } : pkg ); - process.exit(0); } -const { selectedPackages }: Record = - await inquirer.prompt([ - { - type: 'checkbox', - name: 'selectedPackages', - message: 'Packages changed since last publish. Select which to bump:', - choices: changedPackages.map((pkg) => ({ - name: `${getPkgName(pkg)}`, - value: pkg, - checked: true, - })), - }, - ]); - -selectedPackages.forEach((selectedPackage) => { - console.log(colors.bold(colors.yellow(`Bumping ${selectedPackage.name}`))); - execSync('pnpm bumpp --no-commit --no-push --no-tag', { - stdio: 'inherit', - cwd: path.resolve(path.join('packages', selectedPackage.dirName)), - }); -}); +export function validateBumpSelection( + packages: readonly WorkspaceBumpPackage[], + changedPackageNames: readonly string[], + selectedPackageNames: readonly string[] +): string[] { + const changedNames = new Set(changedPackageNames); + const required = new Set(); + for (const selectedName of selectedPackageNames) { + for (const dependentName of getWorkspaceDependentClosure(packages, [ + selectedName, + ])) { + if (dependentName !== selectedName && changedNames.has(dependentName)) { + required.add(dependentName); + } + } + } + + return [...required] + .filter((name) => !selectedPackageNames.includes(name)) + .sort(); +} + +async function main(): Promise { + const allPackages = markWorkspaceDependentsChanged( + await detectChangedPackages() + ); + const changedPackages = allPackages.filter((p) => p.changedSinceLastPublish); + + if (changedPackages.length === 0) { + console.log( + colors.bold().green('No package changes since last publish detected.') + ); + return; + } + + const { selectedPackages }: Record = + await inquirer.prompt([ + { + type: 'checkbox', + name: 'selectedPackages', + message: + 'Packages changed since last publish (including workspace dependents). Select which to bump:', + choices: changedPackages.map((pkg) => ({ + name: `${getPkgName(pkg)}`, + value: pkg, + checked: true, + })), + }, + ]); + const missingDependents = validateBumpSelection( + allPackages, + changedPackages.map(getPkgName), + selectedPackages.map(getPkgName) + ); + if (missingDependents.length > 0) { + throw new Error( + `Cannot deselect required workspace dependents: ${missingDependents.join( + ', ' + )}` + ); + } + + for (const selectedPackage of selectedPackages) { + console.log(colors.bold(colors.yellow(`Bumping ${selectedPackage.name}`))); + execSync('pnpm bumpp --no-commit --no-push --no-tag', { + stdio: 'inherit', + cwd: path.resolve(path.join('packages', selectedPackage.dirName)), + }); + } + + const flatbreadManifest = JSON.parse( + await fs.readFile('packages/flatbread/package.json', 'utf8') + ); + const effortGraphManifest = JSON.parse( + await fs.readFile('packages/effort-graph/package.json', 'utf8') + ); + await fs.writeFile( + 'packages/effort-graph/skills/effort-graph/release.json', + `${JSON.stringify( + { + format: 1, + flatbreadVersion: flatbreadManifest.version, + effortGraphVersion: effortGraphManifest.version, + gitTag: `v${flatbreadManifest.version}`, + }, + null, + 2 + )}\n` + ); + execSync('pnpm skills:sync', { stdio: 'inherit' }); +} + +const invokedScript = process.argv[1] + ? pathToFileURL(path.resolve(process.argv[1])).href + : ''; +if ( + import.meta.url === invokedScript || + fileURLToPath(import.meta.url) === process.argv[1] +) { + void main(); +} diff --git a/scripts/publish.test.ts b/scripts/publish.test.ts new file mode 100644 index 00000000..f5792c28 --- /dev/null +++ b/scripts/publish.test.ts @@ -0,0 +1,111 @@ +import test from 'ava'; +import { + classifyNpmViewResult, + parseNpmViewVersion, + sortPackages, +} from './publish'; + +test('publish ordering is a stable topological sort', (t) => { + const sorted = sortPackages([ + { + name: 'flatbread', + dirName: 'flatbread', + dependencies: { + '@flatbread/effort-graph': 'workspace:*', + '@flatbread/codegen': 'workspace:*', + }, + }, + { + name: '@flatbread/codegen', + dirName: 'codegen', + dependencies: { '@flatbread/utils': 'workspace:*' }, + }, + { name: '@flatbread/utils', dirName: 'utils' }, + { name: '@flatbread/effort-graph', dirName: 'effort-graph' }, + { name: '@flatbread/config', dirName: 'config' }, + ]); + + t.deepEqual( + sorted.map((pkg) => pkg.name), + [ + '@flatbread/config', + '@flatbread/effort-graph', + '@flatbread/utils', + '@flatbread/codegen', + 'flatbread', + ] + ); +}); + +test('publish ordering rejects local dependency cycles', (t) => { + const error = t.throws(() => + sortPackages([ + { + name: '@flatbread/a', + dirName: 'a', + dependencies: { '@flatbread/b': 'workspace:*' }, + }, + { + name: '@flatbread/b', + dirName: 'b', + dependencies: { '@flatbread/a': 'workspace:*' }, + }, + ]) + ); + t.regex(error?.message ?? '', /dependency cycle/); + t.regex(error?.message ?? '', /@flatbread\/a/); +}); + +test('npm view preflight recognizes an exact published version', (t) => { + t.is(parseNpmViewVersion('"1.0.0-alpha.1"\n'), '1.0.0-alpha.1'); + t.is( + classifyNpmViewResult({ stdout: '"1.0.0-alpha.1"\n' }, '1.0.0-alpha.1'), + 'already-published' + ); +}); + +test('npm view preflight treats E404 as needing publish', (t) => { + t.is( + classifyNpmViewResult( + { + error: { + code: 'E404', + stderr: 'npm ERR! code E404\nnpm ERR! 404 Not Found', + }, + }, + '1.0.0' + ), + 'publish' + ); +}); + +test('npm view preflight aborts on non-not-found failures', (t) => { + t.throws( + () => + classifyNpmViewResult( + { + error: { + code: 'E401', + stderr: 'npm ERR! code E401\nUnable to authenticate', + }, + }, + '1.0.0' + ), + { message: /npm view failed/ } + ); +}); + +test('npm view preflight aborts on ambiguous not-found text', (t) => { + t.throws( + () => + classifyNpmViewResult( + { error: { code: 'E503', message: 'temporary package not found' } }, + '1.0.0' + ), + { message: /npm view failed/ } + ); +}); + +test('npm view preflight accepts numeric 404 status', (t) => { + t.is(classifyNpmViewResult({ error: { status: 404 } }, '1.0.0'), 'publish'); +}); diff --git a/scripts/publish.ts b/scripts/publish.ts index 2970f11e..bad3c0a7 100644 --- a/scripts/publish.ts +++ b/scripts/publish.ts @@ -1,36 +1,218 @@ -import { execSync } from 'child_process'; +import { execFileSync, execSync } from 'child_process'; +import { fileURLToPath } from 'url'; import path from 'path'; import colors from 'kleur'; import { getMonorepoPublicPackages } from './utils/packageManifest'; // import { version } from '../package.json'; -execSync('pnpm run build', { stdio: 'inherit' }); +export type NpmViewResult = { + stdout?: string; + error?: unknown; +}; -// Start building the npm registry publish command -let command = 'pnpm publish --access public --no-git-checks'; +export type PreflightStatus = 'publish' | 'already-published'; -// Get the list of public packages in the monorepo -const packages = await getMonorepoPublicPackages(); +type PublishPackage = { + name: string; + dirName: string; + version?: string; + dependencies?: Record; + optionalDependencies?: Record; + peerDependencies?: Record; +}; -// -// For each package, release it. -// This will publish the package to the npm registry if the version has changed; otherwise it will error and move on. -// -for (const { dirName, name, version } of packages) { +export function parseNpmViewVersion(stdout: string): string | undefined { try { - // Disabled for now as we are just publishing alpha/beta as the latest version before v1.0.0 - // if ((version as string | undefined)?.includes('alpha')) { - // command += ' --tag alpha'; - // } else if ((version as string | undefined)?.includes('beta')) { - // command += ' --tag beta'; - // } - - execSync(command, { - stdio: 'inherit', - cwd: path.resolve(path.join('packages', dirName)), - }); - } catch (_) { - console.log(colors.red(`${name} ${version} failed to publish`)); - } - console.log(colors.bold().green(`Published ${name} v${version}`)); + const value: unknown = JSON.parse(stdout); + return typeof value === 'string' ? value : undefined; + } catch { + return undefined; + } +} + +export function classifyNpmViewResult( + result: NpmViewResult, + expectedVersion: string +): PreflightStatus { + const version = parseNpmViewVersion(result.stdout ?? ''); + if (version === expectedVersion) return 'already-published'; + + const error = result.error; + const errorRecord = + error && typeof error === 'object' + ? (error as Record) + : undefined; + if ( + errorRecord && + (errorRecord.code === 'E404' || + errorRecord.status === 404 || + errorRecord.statusCode === 404) + ) { + return 'publish'; + } + + const details = errorRecord + ? [ + errorRecord.code, + errorRecord.status, + errorRecord.stderr, + errorRecord.message, + ] + .filter(Boolean) + .map(String) + .join(' ') + : String(error ?? ''); + + if (!error && version === undefined) { + throw new Error('npm view returned an unexpected response'); + } + throw new Error(`npm view failed: ${details || 'unknown error'}`); +} + +export function sortPackages(packages: T[]): T[] { + const byName = new Map(packages.map((pkg) => [pkg.name, pkg])); + const outgoing = new Map>(); + const indegree = new Map(packages.map((pkg) => [pkg.name, 0])); + const dependencySections = [ + 'dependencies', + 'optionalDependencies', + 'peerDependencies', + ] as const; + + for (const pkg of packages) { + for (const section of dependencySections) { + for (const [dependency, range] of Object.entries(pkg[section] ?? {})) { + if (!byName.has(dependency) || typeof range !== 'string') continue; + const dependents = outgoing.get(dependency) ?? new Set(); + if (!dependents.has(pkg.name)) { + dependents.add(pkg.name); + outgoing.set(dependency, dependents); + indegree.set(pkg.name, (indegree.get(pkg.name) ?? 0) + 1); + } + } + } + } + + const ready = packages + .filter((pkg) => indegree.get(pkg.name) === 0) + .sort(comparePackageNames) + .map((pkg) => pkg.name); + const result: T[] = []; + while (ready.length > 0) { + const name = ready.shift()!; + result.push(byName.get(name)!); + for (const dependent of [...(outgoing.get(name) ?? [])].sort()) { + const next = (indegree.get(dependent) ?? 0) - 1; + indegree.set(dependent, next); + if (next === 0) insertSorted(ready, dependent); + } + } + + if (result.length !== packages.length) { + const remaining = packages + .filter((pkg) => !result.some((item) => item.name === pkg.name)) + .map((pkg) => pkg.name) + .sort(); + throw new Error( + `Cannot determine publish order: local package dependency cycle detected among ${remaining.join( + ', ' + )}. Remove the cycle before publishing.` + ); + } + return result; +} + +function comparePackageNames( + left: PublishPackage, + right: PublishPackage +): number { + return ( + left.name.localeCompare(right.name) || + left.dirName.localeCompare(right.dirName) + ); +} + +function insertSorted(values: string[], value: string): void { + const index = values.findIndex((entry) => entry.localeCompare(value) > 0); + values.splice(index === -1 ? values.length : index, 0, value); +} + +export function assertCleanRelease(): string { + const status = execSync('git status --porcelain', { + encoding: 'utf8', + }).trim(); + if (status) { + throw new Error( + 'Release requires a clean working tree; commit or otherwise resolve these changes first:\n' + + status + ); + } + const sha = execSync('git rev-parse HEAD', { encoding: 'utf8' }).trim(); + if (!sha) throw new Error('Unable to determine the release commit SHA'); + console.log(colors.bold().green(`Release commit: ${sha}`)); + return sha; +} + +export function preflightPackage( + name: string, + version: string +): PreflightStatus { + try { + return classifyNpmViewResult( + { + stdout: execFileSync( + 'npm', + ['view', `${name}@${version}`, 'version', '--json'], + { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] } + ), + }, + version + ); + } catch (error) { + return classifyNpmViewResult({ error }, version); + } +} + +export async function publishPackages(): Promise { + const releaseSha = assertCleanRelease(); + execSync('pnpm run build', { stdio: 'inherit' }); + execSync('pnpm run skills:check', { stdio: 'inherit' }); + execSync('pnpm run skills:pack-check', { stdio: 'inherit' }); + + const packages = sortPackages( + (await getMonorepoPublicPackages()) as unknown as PublishPackage[] + ); + + for (const { dirName, name, version } of packages) { + try { + if (!version) throw new Error(`Package ${name} has no version`); + const preflight = preflightPackage(name, version); + if (preflight === 'already-published') { + console.log( + colors + .bold() + .yellow(`Already published ${name} v${version}; skipping`) + ); + continue; + } + + execSync('pnpm publish --access public --no-git-checks', { + stdio: 'inherit', + cwd: path.resolve(path.join('packages', dirName)), + }); + console.log(colors.bold().green(`Published ${name} v${version}`)); + } catch (error) { + console.error(colors.red(`${name} ${version} failed to publish`)); + if (error instanceof Error) console.error(error.message); + process.exitCode = 1; + break; + } + } + if (process.exitCode === undefined) { + console.log(colors.bold().green(`Published release commit: ${releaseSha}`)); + } +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) { + await publishPackages(); } diff --git a/scripts/utils/packageManifest.ts b/scripts/utils/packageManifest.ts index 23c19484..e1b6daf4 100644 --- a/scripts/utils/packageManifest.ts +++ b/scripts/utils/packageManifest.ts @@ -1,4 +1,4 @@ -import { readdir } from 'fs/promises'; +import { readFile, readdir } from 'fs/promises'; import colors from 'kleur'; import { join } from 'path'; import flatbreadPackage from '../../packages/flatbread/package.json'; @@ -24,15 +24,18 @@ export async function getPackagesManifest( const pkgManifest = 'package.json'; const pkgs = await Promise.all( - dirs.map(async (dir) => ({ - ...(await import(join(process.cwd() + '/packages', dir, pkgManifest))), - dirName: dir, - })) + dirs.map(async (dir) => { + const manifest = JSON.parse( + await readFile( + join(process.cwd() + '/packages', dir, pkgManifest), + 'utf8' + ) + ); + return { ...manifest, dirName: dir }; + }) ); - return pkgs.filter( - (pkg) => !(pkg.default.private && pkg.default.private === true) - ); + return pkgs.filter((pkg) => !(pkg.private && pkg.private === true)); } export async function getMonorepoPublicPackages(): Promise<