From e2f9dcae0554ff63921df618a819fd5e6afe80d2 Mon Sep 17 00:00:00 2001 From: Chris Phillipson Date: Mon, 28 Sep 2026 23:05:04 -0700 Subject: [PATCH 01/10] chore: establish develop remediation integration (#272) * docs(plan): confirm develop remediation execution * ci: validate develop branch pushes --- .github/workflows/ci.yml | 2 +- .../2026-09-28-remediation-program-v2.md | 42 ++- ...-09-28-remediation-v2-develop-execution.md | 253 ++++++++++++++++++ 3 files changed, 293 insertions(+), 4 deletions(-) create mode 100644 docs/plans/2026-09-28-remediation-v2-develop-execution.md diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 80141851..b5dfc6d8 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -2,7 +2,7 @@ name: ci on: push: - branches: [main, npm-kit] + branches: [main, develop, npm-kit] pull_request: workflow_dispatch: diff --git a/docs/plans/2026-09-28-remediation-program-v2.md b/docs/plans/2026-09-28-remediation-program-v2.md index 9ce59f9a..3da32f49 100644 --- a/docs/plans/2026-09-28-remediation-program-v2.md +++ b/docs/plans/2026-09-28-remediation-program-v2.md @@ -4,9 +4,45 @@ ## Status -Active, not yet started. The §1 decision batch is awaiting the maintainer's answers. D-2's first -release (`4.0.0-alpha.60`, carrying #263's breaking changes) is being cut ahead of branch V1, per -D-2 option A. +Active under the maintainer-confirmed [develop execution plan](2026-09-28-remediation-v2-develop-execution.md) +(2026-09-28). V1 #267 and V2 #264/#266/#269 are delivered; #262 still needs residual +Windows timing evidence. #251 is done, and the alpha.60 release commit is on `main`. +Publication and global installation were not verified in the execution-plan baseline. +The historical decision batch and schedule below remain as scope/evidence references; +the approved execution section governs wherever their authority or timing differs. + +### Approved execution and precedence + +The [confirmed execution plan](2026-09-28-remediation-v2-develop-execution.md) governs +branch sources, integration, releases, operations and completion. All Appendix A and B +rows remain in scope. In particular: + +- V1 and both V2 taxonomy PRs are already delivered; use their evidence and do not + replay their implementation. #262 remains open until its ten-run before/after + evidence is recovered or the missing evidence is explicitly reported. +- Bootstrap establishes `develop` from a verified current `main`. Every new feature + branch starts from current `develop`; every feature PR targets `develop`. The + controller may squash-merge only after required CI, including Windows, and + independent review pass. The aggregate `develop` → `main` PR is opened for human + review and left unmerged. References below to cutting branches from, merging into, + or fast-forwarding `main` for intermediate work mean `develop` for new work. +- D-1's cleanup language grants no automatic deletion. Preserve existing work and + unit-commit/evidence records. Real store operations, upstream submissions, paid + runs, and destructive cleanup retain their explicit approval gates. +- The alpha.60 commit is already on `main`; verify its actual published state before + any release decision. D-2's intermediate and close-out release schedule, including + alpha.61, and §2's install prerequisite are deferred until after human approval of + the final main PR and an independent release gate. Do not publish, install or sync + a new artifact merely to satisfy the historical schedule. +- D-3 through D-19 follow the confirmed execution plan's dispositions and current + evidence. D-3 and D-8 have landed. Execution uses bounded concurrency and exact + worktree ownership, rather than the original five-way Wave 2 schedule. +- For review readiness, reconcile every Appendix row to delivered evidence, an + approved disposition, a named open issue or a separately gated operation. Keep + the final main PR open. DoD §6's publication, global install, live-store work, + machine cleanup, future release dispatch and archive/memory operations are + operational completion gates after human review, not prerequisites to opening + that PR. Codex personal-memory updates require a direct user request. **Goal:** Finish everything left over from [remediation program v1](2026-09-26-remediation-program.md) in seven branches. The maintainer's attention goes into one decision sitting up front and a short list of named interrupts. v1 took about 28 attended hours; v2 aims for under 4 (§4 adds it up). diff --git a/docs/plans/2026-09-28-remediation-v2-develop-execution.md b/docs/plans/2026-09-28-remediation-v2-develop-execution.md new file mode 100644 index 00000000..567dec18 --- /dev/null +++ b/docs/plans/2026-09-28-remediation-v2-develop-execution.md @@ -0,0 +1,253 @@ +# Remediation v2 develop execution plan + +> **For agentic workers:** Use superpowers:subagent-driven-development after maintainer +> confirmation. Write each branch's code-level plan with superpowers:writing-plans against +> its actual starting revision. This document governs program sequencing and authority. + +## Status + +**Active and confirmed** — the maintainer approved this execution plan on 2026-09-28. +Bootstrap reconciliation is in progress. Approval covers isolated implementation work, +unit commits, feature PRs into `develop`, and conditional squash integration; the final +`develop` → `main` PR remains open for human review. Releases, installation, real-data +operations and cleanup remain separately gated. Baseline inspected: `main@94890a00`. + +**Goal:** Complete all remaining v2 remediation through feature PRs into `develop`, then +open one aggregate `develop` → `main` PR for human review. + +**Architecture:** One integration controller owns the merge queue and shared records. +Ruflo records coordination; bounded Codex workers implement in isolated worktrees. +Agentic QE supplies scoped quality work through its real installed tools. + +**Tech stack:** Node.js ES modules, existing zero-runtime-dependency CLI, GitHub Actions. + +**Spec:** [Remediation program v2](2026-09-28-remediation-program-v2.md), including every +Appendix A/B row and its referenced v1 plans, rulings and evidence. + +## Confirmed decisions and authority + +- Feature PRs target `develop`; the controller may squash-merge them after all required + CI and independent review pass. The final `develop` → `main` merge is human-owned. +- Make one conventional unit commit per independently verifiable task on feature branches. + Squashing deliberately produces one integration commit per PR. Preserve the unit commit + list and evidence mapping in the PR and ledger before any eventual branch cleanup. +- Defer all new releases and global installation until final main approval and the release + gate. The existing alpha.60 commit is already on main; publication and local installation + were not verified in this planning pass. Do not repeat or overwrite that release. +- Use recommendations D-3 through D-19, subject to current evidence. D-3 and D-8's work has + already landed. D-4 B uses successful version-bound memory-route evidence; D-5 A defers + the four unscoped #239 items; D-6 A drafts the AQE init issue; D-7 A fixes stray discovery. + D-9 A preserves both v5 branches and reserved ADR numbers. D-10–D-17 use recommended + deferrals/acceptances; D-18 A remains conditional on totals being unchanged; D-19 A labels + structured live events experimental. +- Real store merges, file deletion, upstream submissions and paid runs retain their + approval gates. Upstream approval is for exact sanitized text. No automatic worktree + deletion is inferred from approval to squash-merge PRs. +- Use program records for execution continuity. Updating Codex's personal memory requires + a direct user request; the original V7 auto-memory line does not override that boundary. +- This plan replaces the original program's main-targeted branch flow, intermediate + releases, unlimited Wave 2 fan-out, and cleanup-dependent pre-review completion criteria. + All other scope and gates remain in force unless explicitly reconciled by evidence. + +## Verified baseline and remaining uncertainty + +| Item | Current observation | Treatment | +| --- | --- | --- | +| V1 | #267 merged as `ab2fc5cb` | Verify residual timing evidence; do not repeat implementation | +| Windows stability | #262 open; #267 documents two passing runs at its head | Recover current run history; retain missing ten-run evidence as an open gate | +| V2 A/B | #264 and #266 merged; taxonomy plan marked Implemented; #269 archived it | Verify layouts/links and reuse completed work | +| D-8 | #251 merged as `ec749717` | Complete by existing evidence | +| Release | `94890a00` is the alpha.60 release commit | Verify status only; defer further release actions | +| GitHub PR inventory | No open PRs returned by the API | Recheck immediately before dispatch | +| Develop | No local or remote `develop` at planning baseline | Create under confirmed Phase A authority; verify current refs first | +| Existing work | Separate docs archive worktree; untracked `.harness/` in main | Preserve; inspect ownership and changes before dispatch | +| CI | `pull_request` enabled; push branches are `main` and `npm-kit` | Add develop push validation in the bootstrap PR | +| Ruflo | Guidance and memory calls succeeded; active claims empty | Prove scoped claims/runtime behavior before worker dispatch | +| AQE | Fleet status reports healthy, zero active agents/tasks | Verify each requested tool's real output against exact source | + +The original program's "not yet started" status and main-only flow are stale. Reconcile +them in the bootstrap PR, retaining historical evidence rather than replaying completed work. + +ADR-0063 is **Accepted**, updated 2026-09-28, with CLI delivery recorded; V3 completes its +dashboard changes and V4 its offline retry limitation. ADR-0048 is **Accepted**, updated +2026-09-28, with human evaluation gates outstanding; V3 records their approved v5 deferral. +ADR-0060 is **Proposed**, updated 2026-09-27, with discovery partly implemented; V6 implements +its approved remaining scope and records acceptance and the actual delivered subset. + +## Phase A: establish the integration baseline + +Owner: controller. Dependencies: plan confirmation. Size: S, high confidence. + +- [ ] Refresh GitHub and local refs; inventory worktrees, dirty paths, claims and open PRs. + Preserve unrelated work. Check the ignored v1 reconciliation, N-5 list, reviews and ledger + referenced by the spec are available; copy their relevant facts into scoped worker briefs. +- [ ] Create `develop` from verified current main in a dedicated integration worktree. + Record base commit, user authority, controller and allowed actions in the execution ledger. +- [ ] Create a bootstrap feature branch from develop. Commit this plan and reconcile the + source program's Status, decisions, branch targets and definition of done. +- [ ] Add `develop` to `.github/workflows/ci.yml` push validation. Review other workflow + filters and concurrency keys so develop integration is tested without enabling publishing. + Check repository rules/check requirements; proposed changes to repository protection must + be explicit. Enforce the same merge gate in the controller even if develop is unprotected. +- [ ] Validate docs layout, Markdown, links and workflow syntax; open bootstrap PR to develop. + Merge only after review and CI. Subsequent feature branches start at this integrated base. +- [ ] Reconcile V1/V2 completion and #262 run evidence. Capture source revision, run URL, + OS/Node, job duration and result. Do not substitute a two-run observation for ten runs or + claim the old three-consecutive-PR-run rule was historically met without its evidence. + +Acceptance: develop exists; bootstrap checks pass; source scope is reconciled; remaining +rows have owners; existing work is preserved. Rollback: close an unmerged bootstrap PR; +after merge, use a reviewed revert PR rather than resetting shared history. + +## Phase B: execute independent workstreams + +All feature branches start from current `develop`, not main. Each has one writing owner, +one absolute worktree path, a code-level plan, exact path claims, dependency list and +acceptance evidence. Each plan re-reads source and inherited task references before coding. + +| Stream / branch | Deliverable | Dependencies and exclusive boundaries | Acceptance | +| --- | --- | --- | --- | +| V3 `feat/dashboard-refresh` | 6c-1–6c-5; POST refresh, single Refresh control, read-only GETs, vocabulary and client corrections; #256 decisions; #254 only with trace | Bootstrap; owns dashboard server/refresh contracts and client changes first | GETs cause no refresh side effects; refresh contracts, live-view behavior, UI and applicable contrast checks pass | +| V4 `fix/follow-ups-v2` | All A.1–A.4, B.1–B.13 and C.1–C.6, with evidence-based conditional dispositions | Bootstrap; CLI/status/setup/memory/discovery/upstream watch; shared dashboard or test helper paths require explicit handoff | JSON/exit contracts, offline retries, path handling, file IDs, cancellation on Windows, store discovery and upstream conformance proven | +| V5 `test/runner-hygiene` | Branch 9 Tasks 5, 7, 10–13; LQ-1/LQ-4; focused runner and reviewed cleanup inventory | V1 already integrated; bootstrap; owns `scripts/run-tests.mjs`, environment helpers and Chrome launch helper | Clean real-state tripwire and suite temp root; environment isolation, focused runner and cleanup ownership proven | +| V6 `fix/usage-accuracy` | Branch 7 items 1–5; B7-D1–D3; per-turn import exclusion; UA-5; all listed Branch 8 capture fixes; UA-4 | Bootstrap for core work; V3 integration for UI work; owns usage parsers/cache, session vocabulary and census counting | Reproductions using enumerated metadata/counts; imported turns excluded and later genuine turns counted; unknown provider stays unknown | + +Each stream includes every item named in the source program, not just the table summary. +V4 conditional B.13 uses a read-only plan/preview or a disposable-copy reproduction before +proposing changes to the user's configuration. Product fixes need no real-store mutation. +V4's temporary busy-rule/memory-routing CI probe is sandboxed and removed before its PR merges. +V6 reads the actual cache schema before making exactly one migration; the old plan's 25 → 26 +number must not overwrite a schema bump that has already landed. + +### Scheduling and team shape + +The session has four slots total: controller plus at most three workers. Ruflo or AQE +dispatch must not create a second, uncounted fleet or expand spend or delegation limits. + +Start V3 and V6 core work, then V4 when exact file claims are disjoint. Queue V5 for the +next free slot. When a task is ready for review, release/pause that worker's writing scope +and use a free slot for an independent reviewer or AQE specialist. Do not keep three +implementers running while launching an additional reviewer. Queued streams may prepare +read-only briefs only when a slot is available. + +Use the real Ruflo coordination tools for task ownership/dependencies and actual Codex +workers for implementation. Use AQE for scoped test planning, risk/coverage assessment and +quality-gate work when its verified tool schemas can bind the worktree and source state. +A registration, success envelope or numerical score alone is not execution evidence. +If routing or isolation cannot be demonstrated, report the limitation before using a fallback. + +### Conflict prevention + +- One writer per worktree. Claims are exact paths plus resources/ports, not broad overlapping + globs. Recheck the work graph at each dispatch and integration boundary. +- V3 owns `src/lib/dashboard/client/*` first. V6 core excludes these files until V3 merges; + then V6 incorporates develop, reacquires paths and runs its UI task/review cycle. +- V3/V4 serialize changes to `src/lib/dashboard-server.mjs`, any shared Discovery modules, + shared tests and ADR-0063. V3 lands the refresh contract first wherever a consumer needs it. +- V1 → V5 orders `scripts/run-tests.mjs`. V5's helper changes require V4/V6 consumers to + incorporate develop and rerun affected tests. Stop dependent work if a contract changes. +- The controller alone integrates `docs/adr/README.md`, the shared decision log, program + ledger, manifests and lockfiles. Workers submit task-specific handoff text for these paths; + claim transfer happens only after their writing session ends. +- User guides and ADR-0063 get a named owner per task; different sections do not count as + separate file ownership. Update accepted ADR status/date/delivery notes in the same PR. +- Never message a running workflow agent. Use bounded task briefs and ledger handoffs. + Stop dependents on failed gates; independent work may continue in its own scope. + +## Phase C: integrate and finish V6's UI + +- [ ] Review every unit commit after its RED/GREEN evidence. Allow at most five task fix + rounds, then raise the specific unresolved issue rather than looping indefinitely. +- [ ] Freeze the candidate branch, integrate the latest develop and resolve conflicts in + its own worktree with no other writer. Run required full gates and whole-branch review. +- [ ] Queue one feature PR merge at a time. Required evidence covers exact head/base SHAs; + a changed base invalidates the previous integration result and requires appropriate reruns. +- [ ] Squash-merge only after required CI, including Windows, and independent review pass. + Verify the squash commit's tree equals the reviewed, develop-integrated feature tree. + Run/check develop CI before releasing dependent streams. +- [ ] After V3 integrates, complete V6's labels across Usage, Projects, Maintenance and + Intelligence plus imported-copy counts in UI and `ak system`; keep these in the V6 PR. + +Task validation uses the narrowest meaningful tests first. Before V5 introduces `focus`, +use the existing guarded runner interface, for example +`node scripts/run-tests.mjs exec -- --test tests/kit/.test.mjs`. +After its integration use the verified `focus` interface in new briefs. + +Branch gates include the guarded unit/legacy suites, required UI suites, typecheck, lint, +complexity/Markdown/link checks, build and applicable policy/security checks. Use the +package scripts' actual underlying commands; no pnpm in symlinked-node_modules worktrees. +Keep 70/70/70 enforcement and report measured coverage against the 80% line target. +Fixtures and child processes use sandboxed homes and temp roots; state drift or leftovers +fail the gate. Preserve FORCE_COLOR scrubbing and platform redirects. Provider calls that +incur cost are excluded unless explicitly approved. + +## Phase D: V7 close-out and final human review + +Dependencies: V3–V6 merged and develop green. Branch: `chore/v2-close-out`. Size: S/M. + +- [ ] Apply the tree-wide comment-label guard after competing code edits finish. +- [ ] Reconcile every original Appendix A/B row to an exact PR/evidence record, explicit + declined/superseded ruling, named open issue, or approved post-merge operation. No row + disappears because V1/V2 had already landed. +- [ ] Complete #262's ten-run before/after analysis or keep its evidence gate visibly open. + Classify source/runtime changes in samples; never cherry-pick green runs into a false series. +- [ ] Verify upstream watch behavior on the applicable branch/runtime. A first real release + dispatch may remain observation-pending until an actual release event; do not fabricate one. +- [ ] Prepare issue updates (#239/#240/#254/#257/#262) with evidence. Distinguish integrated + on develop from shipped on main; close only when the issue's own completion condition holds. +- [ ] Run docs alignment and archive completed branch plans with `scripts/docs-relocate.mjs`. + Keep this program's plan active while main approval and operational work remain pending. +- [ ] Open final `develop` → `main` PR with scope/decision matrix, feature PRs and unit commit + mappings, exact final source, tests, Windows evidence, release notes and known limitations. + Refresh main into develop first if necessary, review resulting changes, and rerun gates. + Leave the final PR open for human review; do not merge it. + +Review-ready means all code and documentation changes are integrated and validated, with +remaining human/external actions named. It does not claim publication, machine cleanup, +live-store consolidation or a future upstream event has happened. + +## Phase E: separately gated operational completion + +After human approval and merge to main, apply the independent release gate before tagging, +publishing or installing. Verify the existing published versions first; choose the next +version from current state, not this plan's stale alpha.61 schedule. Include breaking CLI +and dashboard changes in release notes. + +Complete approved install/sync/footprint verification against the released artifact. +For the real AQE stray store: exact preview, holder checks, explicit seed/import decision, +backup, disposable-copy import rehearsal, approved application/archive and receipt. +V5 supplies the reviewed literal-path temp inventory; deletion awaits approval. +Remove only approved program-owned worktrees/branches after confirming integration and +preserving unit-commit/evidence records. Keep develop for final review and until its retention +is decided; do not enforce the old "only main and v5 branches" rule during this workflow. + +## Review focus and failure handling + +1. Shared helpers and dashboard contracts changing under a concurrent consumer: exact claims, + explicit dependencies, serialized handoffs and consumer reruns. +2. A feature passing while its develop integration fails: latest-base validation, serialized + merges and develop push CI; stop dependents on failure. +3. Tests or real probes touching user data: guarded runner, isolated stores/homes, read-only + metadata/count observations and explicit operational gates. +4. Session relabeling or import fixes silently changing totals: fixture contracts plus + counts-only real-data reproductions and one controlled cache migration. +5. Completion claims exceeding evidence: source-bound results, no score-only approval, + explicit deferred human gates and a separate operational completion phase. + +On an unmerged failure retain the branch and evidence. After an integrated regression, +stop downstream promotion and submit a corrective or revert PR to develop. Never force-reset +develop/main, delete unrelated work, or weaken a gate to keep the schedule moving. + +## Planning receipt + +The maintainer confirmed this execution plan on 2026-09-28 and authorized isolated +implementation, unit commits, feature PRs into `develop`, and conditional squash +integration after required checks and review. The final `develop` → `main` PR is for +human review and remains unmerged. Release, installation, real-data operations and +cleanup retain their separate gates. The planning worktree started from `94890a00`; +the statements below describe the planning pass, before bootstrap implementation. +Current-state evidence came from local source/history and read-only GitHub API queries. +Ruflo memory search returned historical commands, not a current ownership decision. +Ruflo guidance/claims and AQE fleet status responded; no workers were launched. +Source grounding included `ruflo/plugins/ruflo-swarm/agents/coordinator.md` and +`agentic-qe/kb/capability-cards.md#agentic-qe` (the latter is a summary card, not runtime proof). From bd6b4f0fc33b68812497922ab9bc8467e3d4d443 Mon Sep 17 00:00:00 2001 From: Chris Phillipson Date: Mon, 28 Sep 2026 23:44:59 -0700 Subject: [PATCH 02/10] fix(paths): ignore relative XDG locations consistently (#273) * docs(plan): map V4 follow ups v2 branch * fix(paths): ignore a relative XDG_* value, as the XDG Base Directory spec requires * docs(plan): specify V4 follow-up mappings and decisions * fix(footprint): align deep runtime log root with state base --- docs/plans/2026-09-28-follow-ups-v2.md | 35 ++++++ src/commands/uninstall.mjs | 4 +- src/lib/footprint/consumers.mjs | 6 +- src/lib/footprint/index.mjs | 6 +- src/lib/footprint/install.mjs | 4 +- .../footprint/storage-reclaim-detectors.mjs | 6 +- src/lib/footprint/storage.mjs | 9 +- src/lib/hook-audit/providers/opencode.mjs | 3 +- src/lib/host-readiness-local.mjs | 7 +- src/lib/live/process-sessions.mjs | 4 +- src/lib/paths.mjs | 24 ++-- src/lib/usage-opencode.mjs | 3 +- tests/kit/xdg-relative.test.mjs | 111 ++++++++++++++++++ 13 files changed, 190 insertions(+), 32 deletions(-) create mode 100644 docs/plans/2026-09-28-follow-ups-v2.md create mode 100644 tests/kit/xdg-relative.test.mjs diff --git a/docs/plans/2026-09-28-follow-ups-v2.md b/docs/plans/2026-09-28-follow-ups-v2.md new file mode 100644 index 00000000..7c1bc897 --- /dev/null +++ b/docs/plans/2026-09-28-follow-ups-v2.md @@ -0,0 +1,35 @@ +# Follow ups v2: V4 branch plan + +## Status + +**Active.** Branch `fix/follow-ups-v2`; exact base `e2f9dcae0554ff63921df618a819fd5e6afe80d2` (develop bootstrap #272). B1 is complete in `fb54f02b`; other rows remain unimplemented. The controller reviews and assigns later rows. One test-first unit commit per row. + +The [remediation program V4](2026-09-28-remediation-program-v2.md#v4-fixfollow-ups-v2-every-small-product-cli-and-upstream-item) defines scope. The [archived Branch 9 plan](../archive/2026-09-28-superpowers-plan-branch-9-follow-ups.md) supplies task details. Paths below name current source seams and focused test targets. After an explicit directory prefix, subsequent bare filenames in the same cell use that directory. A new test named below is a proposed file. Later implementers must verify dependencies before editing. + +| Row | Source or artifact mapping | Focused proof and prerequisite | +| --- | --- | --- | +| A1 | `bin/agentic-kit.mjs`; `src/commands/usage.mjs`, `models.mjs`, `audit.mjs`, `heal.mjs`, `telemetry.mjs`, `x/host.mjs` | `tests/kit/cli-json-honesty.test.mjs`, `usage-cli.test.mjs`, `models-command.test.mjs`, `telemetry-cli.test.mjs`, `status-command.test.mjs`; include unknown models verb and status positional | +| A2 | `src/commands/x/host.mjs`; `bin/agentic-kit.mjs` | `tests/kit/host-dry-run.test.mjs`, `host-cli-migration.test.mjs`; pick refusal, off, reset-routes under `--dry-run --json` | +| A3 | `src/lib/versions.mjs`; `docs/adr/0063-evidence-store-and-refresh-vocabulary.md` | `tests/kit/version-lookup-record.test.mjs`, `drift-freshness.test.mjs`; offline tried-at TTL and ADR wording | +| A4 | `src/commands/status.mjs`; `src/lib/refresh.mjs` | `tests/kit/refresh.test.mjs`, `status-version-drift-refresh.test.mjs`; injected `refreshStages` plus `service` builds no collector | +| B1 | `src/lib/paths.mjs`; `src/lib/footprint/index.mjs`, `storage.mjs`, `consumers.mjs`, `storage-reclaim-detectors.mjs`, `install.mjs`; `src/lib/host-readiness-local.mjs`, `live/process-sessions.mjs`, `hook-audit/providers/opencode.mjs`, `usage-opencode.mjs`; `src/commands/uninstall.mjs` | `tests/kit/xdg-relative.test.mjs` and specified regressions; exact-head CI gate passed before edit; preserve nullable OpenCode fallback | +| B2 | `src/commands/x/daemon-gc.mjs`, `src/commands/x/host.mjs`, `src/commands/setup.mjs` | New `tests/kit/daemon-gc-rerecord.test.mjs`, `setup-host-rerecord.test.mjs`, `host-pick-rerecord.test.mjs`; Branch 9 Task 8 plus deferred host pick; compare `sync-host-repair.test.mjs` | +| B3 | `src/lib/ruflo-memory.mjs`, `paths.mjs` | `tests/kit/ruflo-memory-location.test.mjs`, `project-memory-status.test.mjs`; compose both unsuitable reasons and make `inside()` exclude equality | +| B4 | `src/commands/status/sections/project-memory.mjs`; `src/lib/ruflo-memory-contract.mjs`, `live-check-evidence.mjs` | D-4 **B approved**: `tests/kit/project-memory-status.test.mjs`, `live-check-evidence.test.mjs`, `verify-memory-routes.test.mjs`; info only after successful installed-version `memory-routes` evidence, warn on upgrade or failure | +| B5 | `src/lib/project-memory.mjs`, `aqe-readiness.mjs`; `src/commands/status/sections/project-memory.mjs`, `aqe.mjs` | `tests/kit/project-memory.test.mjs`, `aqe-readiness.test.mjs`, `project-memory-status.test.mjs`; dot-folder scan, real `~/.agentic-qe`, and agentic-qe#757 hint | +| B6 | `src/commands/status/sections/ruflo-components.mjs`; `src/lib/ruflo-components/states.mjs` | `tests/kit/ruflo-components-status.test.mjs`; applied-but-unverified row; hooks fix line requires #3419 answer first | +| B7 | `src/lib/ruflo-daemon-config.mjs`; `src/commands/sync.mjs`, `sync/plan-versions.mjs` | `tests/kit/sync-daemon-repair.test.mjs`, `sync-dry-run-preview.test.mjs`, `sync-skip-versions.test.mjs`; F6 hidden YAML keys and F7 versions-only preview parity | +| B8 | `src/lib/maintenance/discovery/orchestrator.mjs`, `history.mjs` | `tests/kit/maintenance-discovery-orchestrator.test.mjs`, `maintenance-recovery.test.mjs`; restart after pause shows paused history | +| B9 | `src/lib/exec.mjs`, `execution/process-tree.mjs` | `tests/kit/process-tree.test.mjs`; abort kills descendants; Windows CI required | +| B10 | `src/lib/maintenance/discovery/partitions.mjs`; inventory `src/lib/live/jsonl-tailer.mjs`, `live/transcript-streams.mjs`, `telemetry/store.mjs`, `maintenance/management/service-store.mjs` for additional persisted IDs | `tests/kit/file-identity-bigint.test.mjs`; distinguish IDs above `2^53`; enumerate the exact sites before edit | +| B11 | `src/lib/live-checks.mjs` | `tests/kit/live-checks.test.mjs`; skipped deja-vu check says skipped and check-created temp folders are cleaned | +| B12 | `src/commands/setup.mjs`; `src/lib/memory-probe-cleanup.mjs` | `tests/kit/setup-memory-probe.test.mjs`; disposable real Ruflo reproduction first, fix only if unused `agentdb-memory.db` appears | +| B13 | `src/commands/sync.mjs`; `src/lib/aqe-project-pin.mjs` | `tests/kit/sync-command.test.mjs`, `aqe-project-pin.test.mjs`; only if program §2 step 2 shows sync omitted the AQE pin | +| C1 | `.github/workflows/ci.yml`; `src/lib/aqe-readiness.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | `tests/live/ruflo-memory-routing.test.mjs`, `tests/kit/aqe-readiness.test.mjs`; disposable macOS and temporary Linux/Windows CI busy-rule evidence; remove temporary job before merge; #240 action follows result | +| C2 | `docs/host-support.md`; `src/lib/hook-audit/agentic-dependency-constraints.json` | `tests/kit/ruflo-support-window.test.mjs` plus link check; verify AQE 3.14.4 #528/#532/#535 and Ruflo #2356/#420 first | +| C3 | `.github/workflows/nightly.yml`; vidaunited's `trace-ort.mjs` hook (obtain and verify its exact script path before adding) | `tests/kit/upstream-watch-workflow.test.mjs` plus macOS artifact receipt; exact upstream #2885 post text requires user approval | +| C4 | `scripts/upstream-watch/classify.mjs`, `fetch.mjs`, `ledger.mjs`, `dispatch.mjs`, `render.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | `tests/kit/upstream-watch-script.test.mjs`, `upstream-watch-record.test.mjs`, `upstream-watch-dispatch.test.mjs`, `upstream-watch-registry.test.mjs`; use ignored `reports/n5-253-deferred-minors.md` §2 for M7/M8/minors 1–12; M10 declined | +| C5 | `src/lib/aqe-guidance.mjs`; `src/commands/setup.mjs`; ignored `.superpowers/sdd/2026-09-28-follow-ups-v2/c5-issue-draft.md` | D-6 **A approved**: controller's isolated AQE init reproduction is evidence handoff; draft issue with command/version/expected/actual, then obtain approval of exact posting text. B0-16 draft only if D-16 B | +| C6 | `docs/host-support.md`; `src/lib/ruflo-support-window.mjs`, `aqe-readiness.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | `tests/kit/ruflo-support-window.test.mjs`, `aqe-readiness.test.mjs`; pre-PR live checks against newest supported Ruflo `security secrets --path`, Codex read-only app-server flags, and AQE 3.14.x | + +B1 used disposable homes, guarded focused tests, and the ignored B1 report at `.superpowers/sdd/2026-09-28-follow-ups-v2/b1-report.md`. No shared manifests, lockfiles, ADR index, or decision log change belongs to this plan update. The controller owns integration and the whole-branch gate. diff --git a/src/commands/uninstall.mjs b/src/commands/uninstall.mjs index c27a1c2a..1a9fe198 100644 --- a/src/commands/uninstall.mjs +++ b/src/commands/uninstall.mjs @@ -101,9 +101,9 @@ function hasDejaVuOwnership(cfg) { function protectedDejaVuRoots(homeDir, env) { const absolute = (value) => typeof value === 'string' && path.isAbsolute(value); - const configBases = [path.join(homeDir, '.config'), env.XDG_CONFIG_HOME, env.APPDATA] + const configBases = [path.join(homeDir, '.config'), paths.xdgBase('XDG_CONFIG_HOME', null, { env }), env.APPDATA] .filter(absolute); - const dataBases = [path.join(homeDir, '.local', 'share'), env.XDG_DATA_HOME] + const dataBases = [path.join(homeDir, '.local', 'share'), paths.xdgBase('XDG_DATA_HOME', null, { env })] .filter(absolute); return { sourceRoots: [ diff --git a/src/lib/footprint/consumers.mjs b/src/lib/footprint/consumers.mjs index 41345a2a..8b5c0081 100644 --- a/src/lib/footprint/consumers.mjs +++ b/src/lib/footprint/consumers.mjs @@ -55,7 +55,7 @@ import fs from 'node:fs'; import path from 'node:path'; import { - claudeDir, codexDir, configDir, globalRoot, home, isWindows, npxCacheDir, + claudeDir, codexDir, configDir, globalRoot, home, isWindows, npxCacheDir, xdgBase, } from '../paths.mjs'; import { hasValue, measured, rootMeasurements, sumMeasurements, unknown, walkTree, @@ -167,8 +167,8 @@ export const CONSUMER_WALK_LIMITS = Object.freeze({ // to audit for the sake of a read-only ranking. Kit and host paths still come // from paths.mjs — nothing home-relative that the kit itself owns is spelled out // below. -const xdgCache = (env) => env.XDG_CACHE_HOME || path.join(home, '.cache'); -const xdgData = (env) => env.XDG_DATA_HOME || path.join(home, '.local', 'share'); +const xdgCache = (env) => xdgBase('XDG_CACHE_HOME', path.join(home, '.cache'), { env }); +const xdgData = (env) => xdgBase('XDG_DATA_HOME', path.join(home, '.local', 'share'), { env }); const macCache = () => path.join(home, 'Library', 'Caches'); const winLocalAppData = (env) => env.LOCALAPPDATA || path.join(home, 'AppData', 'Local'); diff --git a/src/lib/footprint/index.mjs b/src/lib/footprint/index.mjs index 863a2a1d..35daee7e 100644 --- a/src/lib/footprint/index.mjs +++ b/src/lib/footprint/index.mjs @@ -28,7 +28,7 @@ import fs from 'node:fs'; import path from 'node:path'; import { - claudeDir, claudeSettingsPath, claudeUserMcpPath, codexConfigPath, codexDir, configDir, home, + claudeDir, claudeSettingsPath, claudeUserMcpPath, codexConfigPath, codexDir, configDir, stateBase, } from '../paths.mjs'; import { loadKitConfig } from '../config.mjs'; import { defaultOpencodeDbPath } from '../usage-opencode.mjs'; @@ -65,8 +65,8 @@ export const INCLUDE_PROJECT_TREES_DEFAULT = false; * point: these are the files that grow fastest between deep scans (ledgers, * tee files, index caches), and a user watching one grow should not have to * run a deep scan to see it move. */ -function knownFileSpecs() { - const stateRoot = process.env.XDG_STATE_HOME || path.join(home, '.local', 'state'); +export function knownFileSpecs() { + const stateRoot = stateBase(); const kit = (name) => path.join(configDir(), name); // [id, host, category, label, path] — the categories are STORAGE_CATEGORIES' // vocabulary so a known file and its deep-tier node land in the same bucket. diff --git a/src/lib/footprint/install.mjs b/src/lib/footprint/install.mjs index a83414c1..f4c96e03 100644 --- a/src/lib/footprint/install.mjs +++ b/src/lib/footprint/install.mjs @@ -26,7 +26,7 @@ import path from 'node:path'; import { MANAGED_COMPANION_REGISTRY } from '../adapters/companion-registry.mjs'; import { HOST_REGISTRY } from '../adapters/registries.mjs'; import { - home, isWindows, globalRoot, npxCacheDir, claudeDir, codexPluginCacheDir, + home, isWindows, globalRoot, npxCacheDir, claudeDir, codexPluginCacheDir, xdgBase, } from '../paths.mjs'; import { installedVersion, KIT_PKG } from '../versions.mjs'; import { kbDir, present as brainPresent, installedVersion as brainVersion } from '../ruvnet-brain.mjs'; @@ -580,7 +580,7 @@ function vibiumCachePath({ env, platform }) { if (platform === 'win32') { return path.join(env.LOCALAPPDATA || path.join(home, 'AppData', 'Local'), 'vibium'); } - return path.join(env.XDG_CACHE_HOME || path.join(home, '.cache'), 'vibium'); + return path.join(xdgBase('XDG_CACHE_HOME', path.join(home, '.cache'), { env }), 'vibium'); } const presentFile = (file, fsImpl) => { diff --git a/src/lib/footprint/storage-reclaim-detectors.mjs b/src/lib/footprint/storage-reclaim-detectors.mjs index e162425a..97da3b89 100644 --- a/src/lib/footprint/storage-reclaim-detectors.mjs +++ b/src/lib/footprint/storage-reclaim-detectors.mjs @@ -16,15 +16,15 @@ // process.platform, so the wrong-platform root simply reads absent and a machine // carrying both (a tool that moved its cache) reports both. import path from 'node:path'; -import { home, isWindows } from '../paths.mjs'; +import { home, isWindows, xdgBase } from '../paths.mjs'; import { decodeClaudeProjectDir } from './project-sources.mjs'; import { rootMeasurements, measured, unknown, statNode, sumMeasurements, hasValue, } from './walk.mjs'; import { candidate } from './storage-reclaim.mjs'; -const xdgCache = (env) => env.XDG_CACHE_HOME || path.join(home, '.cache'); -const xdgData = (env) => env.XDG_DATA_HOME || path.join(home, '.local', 'share'); +const xdgCache = (env) => xdgBase('XDG_CACHE_HOME', path.join(home, '.cache'), { env }); +const xdgData = (env) => xdgBase('XDG_DATA_HOME', path.join(home, '.local', 'share'), { env }); const macCache = () => path.join(home, 'Library', 'Caches'); const winLocalAppData = (env) => env.LOCALAPPDATA || path.join(home, 'AppData', 'Local'); diff --git a/src/lib/footprint/storage.mjs b/src/lib/footprint/storage.mjs index 3b62e2c3..db519ecd 100644 --- a/src/lib/footprint/storage.mjs +++ b/src/lib/footprint/storage.mjs @@ -52,7 +52,7 @@ // `detectWorktrees` is false. import fs from 'node:fs'; import path from 'node:path'; -import { home, claudeDir, codexDir, configDir } from '../paths.mjs'; +import { home, claudeDir, codexDir, configDir, stateBase } from '../paths.mjs'; import { defaultOpencodeDbPath } from '../usage-opencode.mjs'; import { decodeClaudeProjectDir, transcriptMetadata } from './project-sources.mjs'; import { classifyWorkingContext } from './working-context.mjs'; @@ -118,8 +118,9 @@ const flatDir = () => true; * * @returns {StorageRoot[]} */ -export function defaultStorageRoots({ env = process.env, projects = null } = {}) { - const stateRoot = env.XDG_STATE_HOME || path.join(home, '.local', 'state'); +export function defaultStorageRoots({ env = process.env, projects = null, + home: h = home, platform = process.platform, p = path } = {}) { + const stateRoot = stateBase({ env, home: h, platform, p }); const opencodeData = path.dirname(defaultOpencodeDbPath()); const claude = (name) => path.join(claudeDir(), name); const codex = (name) => path.join(codexDir(), name); @@ -183,7 +184,7 @@ export function defaultStorageRoots({ env = process.env, projects = null } = {}) { id: 'ak-runtime-debug', category: 'ledgers-and-logs', host: 'agentic-kit', label: 'runtime-debug.log', - path: path.join(stateRoot, 'agentic-kit', 'runtime-debug.log'), layout: 'tree', + path: p.join(stateRoot, 'agentic-kit', 'runtime-debug.log'), layout: 'tree', }, { id: 'ak-config', category: 'kit-caches', host: 'agentic-kit', diff --git a/src/lib/hook-audit/providers/opencode.mjs b/src/lib/hook-audit/providers/opencode.mjs index 25cbc560..a7738968 100644 --- a/src/lib/hook-audit/providers/opencode.mjs +++ b/src/lib/hook-audit/providers/opencode.mjs @@ -1,6 +1,7 @@ import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; +import { xdgBase } from '../../paths.mjs'; import { normalizedOccurrence, publicSource, readBoundedFile, readJsonSource, @@ -90,7 +91,7 @@ function moduleRecords(source) { } export function auditOpenCodeHooks({ - opencodeRoot = path.join(process.env.XDG_CONFIG_HOME || path.join(os.homedir(), '.config'), 'opencode'), + opencodeRoot = path.join(xdgBase('XDG_CONFIG_HOME', path.join(os.homedir(), '.config')), 'opencode'), projectRoots = [process.cwd()], opencodeVersion = 'unknown', ownership = null, diff --git a/src/lib/host-readiness-local.mjs b/src/lib/host-readiness-local.mjs index ebd19eaa..2ea1bb7d 100644 --- a/src/lib/host-readiness-local.mjs +++ b/src/lib/host-readiness-local.mjs @@ -9,6 +9,7 @@ import path from 'node:path'; import { createHash } from 'node:crypto'; import { readContextConfig } from './codex-context-config.mjs'; import { withDb } from './sqlite.mjs'; +import { xdgBase } from './paths.mjs'; const LIMIT = 1024 * 1024; const plain = x => x !== null && typeof x === 'object' && !Array.isArray(x); @@ -199,8 +200,8 @@ function selectedOpenCodeAgent(config, agentDirs) { } function loadOpenCode({ cwd, home, env }, evidence) { - const global = path.join(env.XDG_CONFIG_HOME || path.join(home, '.config'), 'opencode'); - const data = path.join(env.XDG_DATA_HOME || path.join(home, '.local/share'), 'opencode'); + const global = path.join(xdgBase('XDG_CONFIG_HOME', path.join(home, '.config'), { env }), 'opencode'); + const data = path.join(xdgBase('XDG_DATA_HOME', path.join(home, '.local/share'), { env }), 'opencode'); const auth = document(path.join(data, 'auth.json'), evidence, env); if (Object.values(auth).some(value => value?.type === 'wellknown') || openCodeRemote(data, evidence)) throw new Error('unsupported'); if (fs.existsSync(path.join(global, 'config'))) throw new Error('unsupported'); // legacy TOML migration is native-owned @@ -289,7 +290,7 @@ function defaultOpenCode({ config, auth, home, env, allowed, credentialed }, evi // Native default tries recent available selections, then a configured // provider. Do not borrow credentials from an unrelated provider. - const recent = document(path.join(env.XDG_STATE_HOME || path.join(home, '.local/state'), 'opencode/model.json'), evidence, env).recent; + const recent = document(path.join(xdgBase('XDG_STATE_HOME', path.join(home, '.local/state'), { env }), 'opencode/model.json'), evidence, env).recent; if (Array.isArray(recent) && recent.length) throw new Error('unsupported'); // availability requires native provider catalog const configured = Object.keys(config.provider ?? {}).filter(allowed); if (configured.length === 1) provider = configured[0]; diff --git a/src/lib/live/process-sessions.mjs b/src/lib/live/process-sessions.mjs index acf2eb80..ce105b15 100644 --- a/src/lib/live/process-sessions.mjs +++ b/src/lib/live/process-sessions.mjs @@ -1,10 +1,10 @@ import { execFile } from 'node:child_process'; import fs from 'node:fs'; -import os from 'node:os'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; import { promisify } from 'node:util'; import { inspectGitWorkspace } from './git-workspace.mjs'; +import { stateBase } from '../paths.mjs'; const execFileAsync = promisify(execFile); @@ -31,7 +31,7 @@ const WIN32_SURVEY_SCRIPT = fileURLToPath( function runtimeDebug(stage, fields = {}) { if (!process?.env || process.env.AK_RUNTIME_DEBUG !== '1') return; try { - const root = process.env.XDG_STATE_HOME || path.join(os.homedir(), '.local', 'state'); + const root = stateBase(); const file = process.env.AK_RUNTIME_DEBUG_FILE || path.join(root, 'agentic-kit', 'runtime-debug.log'); const safeStage = String(stage || 'unknown').replace(/[^a-z0-9._-]/gi, '_').slice(0, 64); const kv = Object.entries(fields) diff --git a/src/lib/paths.mjs b/src/lib/paths.mjs index 393fb420..134cd45b 100644 --- a/src/lib/paths.mjs +++ b/src/lib/paths.mjs @@ -10,14 +10,20 @@ import { writePrivateFileAtomic } from './file-write.mjs'; const home = os.homedir(); const isWindows = process.platform === 'win32'; +/** The XDG Base Directory spec ignores relative environment overrides. */ +export function xdgBase(name, fallback, { env = process.env, p = path } = {}) { + const value = env[name]; + return value && p.isAbsolute(value) ? value : fallback; +} + /** Kit config dir: XDG on POSIX, %APPDATA% on Windows. */ function configBase() { if (isWindows) return process.env.APPDATA || path.join(home, 'AppData', 'Roaming'); - return process.env.XDG_CONFIG_HOME || path.join(home, '.config'); + return xdgBase('XDG_CONFIG_HOME', path.join(home, '.config')); } -function stateBase() { - if (isWindows) return process.env.LOCALAPPDATA || path.join(home, 'AppData', 'Local'); - return process.env.XDG_STATE_HOME || path.join(home, '.local', 'state'); +export function stateBase({ env = process.env, home: h = home, platform = process.platform, p = path } = {}) { + if (platform === 'win32') return env.LOCALAPPDATA || p.join(h, 'AppData', 'Local'); + return xdgBase('XDG_STATE_HOME', p.join(h, '.local', 'state'), { env, p }); } export const configDir = () => path.join(configBase(), 'agentic-kit'); export const telemetryDir = () => path.join(configDir(), 'telemetry'); @@ -98,8 +104,10 @@ export function toolInternalDirs({ home: h = home, env = process.env, platform = p.join(h, '.claude'), env.CLAUDE_CONFIG_DIR, p.join(h, '.codex'), env.CODEX_HOME, p.join(h, '.claude-flow'), p.join(h, '.ruflo'), - p.join(h, '.config'), env.XDG_CONFIG_HOME, p.join(h, '.local'), env.XDG_DATA_HOME, - env.XDG_STATE_HOME, p.join(h, '.cache'), env.XDG_CACHE_HOME, + p.join(h, '.config'), xdgBase('XDG_CONFIG_HOME', null, { env, p }), + p.join(h, '.local'), xdgBase('XDG_DATA_HOME', null, { env, p }), + xdgBase('XDG_STATE_HOME', null, { env, p }), p.join(h, '.cache'), + xdgBase('XDG_CACHE_HOME', null, { env, p }), ]; if (platform === 'win32') dirs.push(p.join(h, 'AppData'), env.APPDATA, env.LOCALAPPDATA); if (platform === 'darwin') dirs.push(p.join(h, 'Library', 'Application Support'), p.join(h, 'Library', 'Caches')); @@ -349,8 +357,8 @@ export function hostHealthInputPaths(cwd, env = process.env) { path.join(codex, 'requirements.toml'), '/etc/codex/config.toml', '/etc/codex/requirements.toml', ...(process.platform === 'win32' ? [path.join(env.ProgramData || 'C:\\ProgramData', 'OpenAI', 'Codex', 'config.toml')] : []), path.join(opencode, 'config.json'), path.join(opencode, 'opencode.json'), path.join(opencode, 'opencode.jsonc'), - path.join(env.XDG_STATE_HOME || path.join(home, '.local', 'state'), 'opencode', 'model.json'), - path.join(env.XDG_DATA_HOME || path.join(home, '.local', 'share'), 'opencode', 'auth.json'), + path.join(xdgBase('XDG_STATE_HOME', path.join(home, '.local', 'state'), { env }), 'opencode', 'model.json'), + path.join(xdgBase('XDG_DATA_HOME', path.join(home, '.local', 'share'), { env }), 'opencode', 'auth.json'), env.OPENCODE_CONFIG, ].filter(Boolean); let root = path.resolve(cwd); diff --git a/src/lib/usage-opencode.mjs b/src/lib/usage-opencode.mjs index 47c2cb5d..abaacd05 100644 --- a/src/lib/usage-opencode.mjs +++ b/src/lib/usage-opencode.mjs @@ -34,11 +34,12 @@ import { addUsage, blankSession, noteContextSample, noteLatencySample, notePromptFingerprint, } from './usage-parsers.mjs'; import { normalizeMode } from './usage-modes.mjs'; +import { xdgBase } from './paths.mjs'; import { observeUsageProject } from './usage-project-evidence.mjs'; /** The live opencode store. Overridable via roots in tests. */ export function defaultOpencodeDbPath() { - const home = process.env.XDG_DATA_HOME ?? null; + const home = xdgBase('XDG_DATA_HOME', null); return home ? `${home}/opencode/opencode.db` : `${process.env.HOME ?? process.env.USERPROFILE}/.local/share/opencode/opencode.db`; diff --git a/tests/kit/xdg-relative.test.mjs b/tests/kit/xdg-relative.test.mjs new file mode 100644 index 00000000..b0ad4e42 --- /dev/null +++ b/tests/kit/xdg-relative.test.mjs @@ -0,0 +1,111 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { spawnSync } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; +import * as pathModule from '../../src/lib/paths.mjs'; +import { defaultStorageRoots } from '../../src/lib/footprint/storage.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; + +const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..'); + +test('xdgBase keeps only absolute XDG values for both path flavors', () => { + assert.equal(typeof pathModule.xdgBase, 'function'); + for (const [p, absolute] of [[path.posix, '/opt/cfg'], [path.win32, 'C:\\cfg']]) { + const fallback = p.join(absolute, 'fallback'); + for (const value of [undefined, '', 'rel/cfg', './cfg']) { + assert.equal(pathModule.xdgBase('XDG_CONFIG_HOME', fallback, { env: { XDG_CONFIG_HOME: value }, p }), fallback); + } + assert.equal(pathModule.xdgBase('XDG_CONFIG_HOME', fallback, { env: { XDG_CONFIG_HOME: absolute }, p }), absolute); + } +}); + +test('Windows deep runtime-log root follows LOCALAPPDATA with a distinct XDG state base', () => { + const home = 'C:\\Users\\Ada'; + const env = { + LOCALAPPDATA: 'C:\\Users\\Ada\\AppData\\Local', + XDG_STATE_HOME: 'D:\\xdg-state', + }; + const expected = 'C:\\Users\\Ada\\AppData\\Local\\agentic-kit\\runtime-debug.log'; + const options = { env, home, platform: 'win32', p: path.win32 }; + assert.equal(pathModule.stateBase(options), env.LOCALAPPDATA); + assert.equal(defaultStorageRoots(options).find((row) => row.id === 'ak-runtime-debug')?.path, expected); +}); + +test('runtime-log writer, known-file reader, and deep root agree with distinct native state bases', (t) => { + const home = tempDir('ak-xdg-state-agreement', t); + const local = path.join(home, 'native-local'); + const xdg = path.join(home, 'xdg-state'); + const pathsUrl = new URL('../../src/lib/paths.mjs', import.meta.url).href; + const footprintUrl = new URL('../../src/lib/footprint/index.mjs', import.meta.url).href; + const storageUrl = new URL('../../src/lib/footprint/storage.mjs', import.meta.url).href; + const script = `import path from 'node:path'; +import { stateBase } from ${JSON.stringify(pathsUrl)}; +import { knownFileSpecs } from ${JSON.stringify(footprintUrl)}; +import { defaultStorageRoots } from ${JSON.stringify(storageUrl)}; +console.log(JSON.stringify({ writer: path.join(stateBase(), 'agentic-kit', 'runtime-debug.log'), + known: knownFileSpecs().find((row) => row.id === 'ak-runtime-debug').path, + deep: defaultStorageRoots().find((row) => row.id === 'ak-runtime-debug').path }));`; + const child = spawnSync(process.execPath, ['--input-type=module', '-e', script], { + cwd: home, + env: spawnEnv(home, { LOCALAPPDATA: local, XDG_STATE_HOME: xdg }), + encoding: 'utf8', + }); + assert.equal(child.status, 0, child.stderr); + const paths = JSON.parse(child.stdout); + const base = process.platform === 'win32' ? local : xdg; + const expected = path.join(base, 'agentic-kit', 'runtime-debug.log'); + assert.deepEqual(paths, { writer: expected, known: expected, deep: expected }); +}); + +test('relative XDG values cannot redirect live paths or tool root discovery into cwd', { skip: process.platform === 'win32' }, (t) => { + const home = tempDir('ak-xdg-relative-home', t); + const cwd = path.join(home, 'work'); + fs.mkdirSync(cwd); + const pathsUrl = new URL('../../src/lib/paths.mjs', import.meta.url).href; + const footprintUrl = new URL('../../src/lib/footprint/index.mjs', import.meta.url).href; + const script = `import * as paths from ${JSON.stringify(pathsUrl)}; +import { knownFileSpecs } from ${JSON.stringify(footprintUrl)}; +console.log(JSON.stringify([paths.configDir(), paths.evidenceDir(), + ...knownFileSpecs().map((row) => row.path), ...paths.toolInternalDirs()]));`; + const child = spawnSync(process.execPath, ['--input-type=module', '-e', script], { + cwd, + env: spawnEnv(home, { + XDG_CONFIG_HOME: 'rel/cfg', XDG_STATE_HOME: 'rel/state', + XDG_DATA_HOME: 'rel/data', XDG_CACHE_HOME: 'rel/cache', + }), + encoding: 'utf8', + }); + assert.equal(child.status, 0, child.stderr); + const paths = JSON.parse(child.stdout); + assert.ok(paths.length > 10); + for (const candidate of paths) { + assert.ok(path.isAbsolute(candidate), candidate); + assert.ok(candidate.startsWith(`${home}${path.sep}`), candidate); + assert.doesNotMatch(candidate, /(?:^|[/\\])rel(?:[/\\]|$)/); + } + assert.ok(paths.includes(path.join(home, '.local', 'state', 'agentic-kit', 'runtime-debug.log'))); +}); + +test('new source readers use the shared XDG base validator', () => { + const allowed = new Set(['src/lib/paths.mjs', 'src/templates/statusline-footer.cjs', + 'src/lib/adapters/manifest.mjs']); + const matches = []; + const visit = (dir) => { + for (const entry of fs.readdirSync(dir, { withFileTypes: true })) { + const at = path.join(dir, entry.name); + if (entry.isDirectory()) { visit(at); continue; } + if (!entry.isFile()) continue; + const relative = path.relative(root, at).split(path.sep).join('/'); + if (allowed.has(relative)) continue; + const source = fs.readFileSync(at, 'utf8') + .replace(/\/\*[\s\S]*?\*\//g, '') + .replace(/\/\/[^\n]*/g, ''); + if (/\b(?:process\.)?env\.XDG_[A-Z_]+|env\[['"]XDG_/.test(source)) matches.push(relative); + } + }; + visit(path.join(root, 'src')); + assert.deepEqual(matches, []); +}); From af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b Mon Sep 17 00:00:00 2001 From: Chris Phillipson Date: Tue, 29 Sep 2026 03:58:31 -0700 Subject: [PATCH 03/10] fix(test-runner): finish guarded execution and environment hygiene (#274) * docs(plan): map runner hygiene tasks and evidence gates * docs(research): how test temp folders leak and how a run root is proven abandoned * docs(research): complete test creator lifecycle census * docs(research): account for parent cleanup in credential census * test(about): render the Ruflo install-edit line on the About card * test(tripwire): list the live Ruflo session's .claude-flow folder and proven-config files as concurrent writers * fix(test-runner): guard owner roots and keep sibling cleanup list-only * fix(test-runner): fail hygiene on own-root cleanup errors * feat(test-runner): guard focused runs and prove interrupted retention * test(test-runner): observe orphan exit before retention check * test(runner): preserve tool selector propagation * fix(ui): isolate Chrome launch environment * test(status): await owned spawn guard child before cleanup * test(runner): verify owned fork exits before sandbox cleanup * test(runner): retain own root for unresolved child holds * test(runner): await owned process cleanup on cancellation * docs(archive): record completed runner hygiene work * test(runner): keep cancellation fixture alive on Node 22 * test(runner): use owned native termination for signal retention * test(runner): retry fixture removal after confirmed child exit --- AGENTS.md | 20 +- .../archive/2026-09-28-plan-runner-hygiene.md | 57 +++ ...09-28-research-test-temp-folder-cleanup.md | 449 ++++++++++++++++++ docs/archive/README.md | 2 + scripts/real-state-tripwire.mjs | 3 +- scripts/run-roots.mjs | 228 +++++++++ scripts/run-tests.mjs | 76 ++- tests/kit/about-install-edit-render.test.mjs | 47 ++ tests/kit/helpers/interruption-scope.mjs | 92 ++++ tests/kit/helpers/temp-dir.mjs | 10 +- tests/kit/real-state-tripwire.test.mjs | 37 ++ tests/kit/run-roots.test.mjs | 266 +++++++++++ tests/kit/run-tests-cancellation.test.mjs | 120 +++++ tests/kit/run-tests-interruption.test.mjs | 112 +++++ tests/kit/run-tests-runner.test.mjs | 358 +++++++++++++- tests/kit/status-zero-spawn.test.mjs | 187 +++++++- tests/kit/ui-chrome-launch.test.mjs | 41 ++ tests/ui/helpers/launch-chrome.mjs | 32 +- 18 files changed, 2093 insertions(+), 44 deletions(-) create mode 100644 docs/archive/2026-09-28-plan-runner-hygiene.md create mode 100644 docs/archive/2026-09-28-research-test-temp-folder-cleanup.md create mode 100644 scripts/run-roots.mjs create mode 100644 tests/kit/about-install-edit-render.test.mjs create mode 100644 tests/kit/helpers/interruption-scope.mjs create mode 100644 tests/kit/run-roots.test.mjs create mode 100644 tests/kit/run-tests-cancellation.test.mjs create mode 100644 tests/kit/run-tests-interruption.test.mjs diff --git a/AGENTS.md b/AGENTS.md index c0d3f693..8f613e30 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -307,7 +307,7 @@ compression, or neural-routing targets are not measured agentic-kit guarantees. pnpm test # One focused suite -node --test tests/kit/dispatch-surface.test.mjs +node scripts/run-tests.mjs focus tests/kit/dispatch-surface.test.mjs # Browser verification pnpm run test:ui @@ -320,6 +320,8 @@ pnpm run lint:md pnpm run build ``` +A plain `node --test` run lacks the wrapper's real-state tripwire and temp-root checks. + `pnpm test` and `pnpm run test:ui` run through `scripts/run-tests.mjs`, which fingerprints `~/.config/agentic-kit`, `~/.local/state/agentic-kit` (or `%APPDATA%`/`%LOCALAPPDATA%` on Windows), `~/.claude/CLAUDE.md`, `~/.claude/settings.json`, `~/.claude.json`, @@ -332,8 +334,20 @@ them, and `sandboxHome()` and `redirectToolState()` do the same for in-process c Code's own `~/.claude.json`) are listed as "concurrent writers" and do not fail a local run; CI (or `AK_TRIPWIRE_STRICT=1`) fails on them too. Every command also runs with `TMPDIR`/`TEMP`/`TMP` pointed at a fresh `ak-suite-*` folder: anything left in it afterwards fails the run and is -listed, and the runner refuses to start when that folder sits inside a git repository (point -`TMPDIR` elsewhere). The runner also drops `FORCE_COLOR` (Claude Code shells set it), because +listed (excluding its private atomic `.ak-suite-owner.json`, child-hold directory and Node compile cache). The runner +refuses home/filesystem-root temp bases before allocation and refuses roots inside a git +repository (point `TMPDIR` elsewhere). A completed run removes only its own validated direct, +canonical, nonsymlink, current-owner root. Tests with known child lifetime uncertainty acquire +`acquireRunRootHold()` before launching those children and release only after proving their exits. +An unresolved or unreadable hold retains the own root; it is not a general descendant-exit proof. +The runner then lists sibling suite roots: missing, invalid, +foreign or uncertain owner metadata means keep. Sibling handling is list-only on macOS, Linux +and Windows because no installed probe proves all descendants have exited; even a dead owner +is insufficient. Interrupted runs remove and collect nothing. Sibling listing/collection errors +do not change the suite's exit code. Own-root inspection failure retains the root; inspection, +removal or safety-refusal failure returns hygiene exit 4 unless a command or tripwire failure +already takes precedence. Removal errors may leave a partially removed own root. The runner also +drops `FORCE_COLOR` (Claude Code shells set it), because tests read plain text from pipes. Tests make temporary folders with `tempDir()` from `tests/kit/helpers/temp-dir.mjs`, and spawned children get their environment from `spawnEnv()` in `tests/kit/helpers/home-sandbox.mjs`. UI tests launch Chrome with `launchChrome()` from diff --git a/docs/archive/2026-09-28-plan-runner-hygiene.md b/docs/archive/2026-09-28-plan-runner-hygiene.md new file mode 100644 index 00000000..1385f29b --- /dev/null +++ b/docs/archive/2026-09-28-plan-runner-hygiene.md @@ -0,0 +1,57 @@ +# Runner hygiene execution plan + +## Status + +**Implemented and independently reviewed**, captured before PR integration on 2026-09-29. +Code head `e0fcc2eb` passed all local unit, UI, typecheck, lint, complexity, Markdown, +build and internal-link gates. Unit results: 5,947 passed, zero failed, six native +Windows-only skips; coverage 94.33% lines, 83.27% branches and 93.47% functions. +System Chrome passed 495 dashboard checks and 15 Node UI tests. Native Windows +and Linux Chrome proof remain the feature PR's CI gate at this capture point. + +Delivered: the About renderer regression, exact concurrent-writer exceptions, +owner records and guarded focused runs, list-only sibling handling on every platform, +Chrome environment isolation, tool-selector propagation regression, and owned-process +exit/cancellation checks. A participating test acquires a run-bound hold before launching +its children; uncertainty retains its own fixture and the enclosing run root. Holds do +not discover unregistered descendants or authorize sibling removal. + +Task 13's immutable inventory and literal-path list were independently reviewed against +snapshot `c4aa2015ccd66c47f99f9204d0444faac007e83e9a8b709ddcee04791819e6bb`: +33,978 entries before the runner cutoff, four afterward and three legacy ownerless roots; +27 unattributed, 21 recent and one lsof-matched entry were retained separately. The private +handoff contains machine paths and stays outside git. Age, prefix attribution and lsof +absence do not establish deletion safety. No backlog removal was performed. + +The source-bound research census recorded 647 sites in 279 files at its stated baseline; +it does not count later edits or prove historical leak causes. LQ-1's propagation-defect +premise was refuted, and a mutation-tested regression preserves the existing behavior. +Final review found one cancellation-order defect; `e0fcc2eb` fixed it and passed scoped +rereview. Releases, global installation and the aggregate main merge remain separately gated. + +## Contract and dependencies + +The [v2 scope](../plans/2026-09-28-remediation-program-v2.md#v5-testrunner-hygiene-suites-that-clean-up-after-themselves) inherits [archived Branch 9](2026-09-28-superpowers-plan-branch-9-follow-ups.md) Tasks 5, 7 and 10–13, under B9-R1–R8. V1 is integrated at the baseline. V4/V6 must coordinate before changing environment-helper consumers. Worktree ownership is limited to this branch; shared manifests remain the integration owner's responsibility. + +The [cleanup design](2026-09-28-research-test-temp-folder-cleanup.md) selects B9-R5's explicit list-only fallback. Task 11 must not interpret a dead PID, empty process group, empty registry or empty handle scan as proof of abandonment. Task 12 must keep the interrupted root even after its known child exits on list-only platforms. This conditions the archived example's removal assertion; it does not relax B9-R5. + +## File, test and dependency map + +| Unit | Exact owned files or proposed files | Validation and acceptance | Depends on | +|---|---|---|---| +| Task 10 research | This plan; `docs/plans/2026-09-28-test-temp-folder-cleanup-design.md`; ignored scratch report/probes | Real macOS orphan observation, official docs, explicit unmeasured platforms; Markdown, links, docs citations/layout | Baseline and brief | +| Task 5 About render | New `tests/kit/about-install-edit-render.test.mjs`; read `src/lib/dashboard/client/about.mjs`, `src/lib/install-edits.mjs`, `src/commands/status/sections/natives.mjs` | Real About renderer, escaped single Ruflo pin line, no AgentDB line/no-edit line; wording mutation fails; existing `about-install-edits.test.mjs`, `about-agentdb-join.test.mjs` | Research handoff; fresh file claims | +| Task 7 concurrent writers | `scripts/real-state-tripwire.mjs`, `tests/kit/real-state-tripwire.test.mjs` | Absent-to-present `.claude-flow`, two proven-config files are concurrent locally, fail in strict mode; config.json still fails; `run-tests-runner.test.mjs`, `home-sandbox-tripwire.test.mjs` | Reverify installed supported Ruflo sources; no fabricated version | +| Task 11 owner record and safe paths | New `scripts/run-roots.mjs`, `tests/kit/run-roots.test.mjs`; `scripts/run-tests.mjs`, `tests/kit/run-tests-runner.test.mjs`, `AGENTS.md` | Red then green path/owner/schema/symlink/UID/host/invalid data tests; listing does not change exit; interrupted roots never removed by defaults; `real-state-tripwire.test.mjs`, `temp-dir-helper.test.mjs` | Task 10 decisions accepted; preserve synchronous spawn/signal behavior | +| Task 12 focus and exit proof | `scripts/run-tests.mjs`, `tests/kit/run-tests-runner.test.mjs`, `AGENTS.md` | Focus pass=0, leftover=4, missing args=2; killed runner's idle child and concurrently live runner retained; root still retained after child death on list-only platforms; exact-PID cleanup finally | Task 11; controller updates external brief template | +| Windows smoke lifetime diagnosis | Proposed `tests/kit/status-zero-spawn.test.mjs`; new disposable fixture only if needed | Handshake shows whether fork outlives parent; compare explicit close wait; preserve assertion and cleanup errors; Windows Node 24 regression required | Controller authorizes implementation; native Windows CI, not fixture simulation | +| LQ-1 environment premise | Read `scripts/run-tests.mjs`, `tests/kit/helpers/home-sandbox.mjs`; if proven defect, those files plus `tests/kit/run-tests-runner.test.mjs`, `tests/kit/spawn-env-guard.test.mjs` | Current runner deletes FORCE_COLOR only; demonstrate sentinel AQE variables across each actual child boundary before any patch; retain state isolation | Task 12; reconcile scope text with source | +| LQ-4 Chrome environment | `tests/ui/helpers/launch-chrome.mjs`; new `tests/kit/launch-chrome-env.test.mjs`; environment helper only with exact claim | Preserve required display/path/platform variables and private Chrome temp; exclude user state; mocked launch failure and close cleanup; native UI smoke | Task 12; coordinate helper users; installed Playwright available | +| Task 13 reviewed inventory | Ignored source/list/report in controller-approved report folder, no tracked cleanup program | Literal absolute paths, prefix attribution, exclusion counts, independent review of same snapshot; no removal | Last focused run; controller provides report destination/reviewer | +| Branch handoff | Update this plan and design status; archive via `scripts/docs-relocate.mjs` in completion PR with index rows | Focused gates, type/lint/build and hermetic full gate appropriate to eventual implementation; exact commit/evidence receipt | All implementation units complete; integration approval | + +## Execution boundaries + +Use `node scripts/run-tests.mjs focus ` with disposable home/state roots for focused tests. Do not use pnpm in a worktree with symlinked dependencies. Run focused failure-path checks before wider gates; do not repeat green gates without a new concern. Each code unit needs its own failing/passing evidence and conventional commit after authorization. Before editing/staging `AGENTS.md`, verify there is no injected drift. + +Task 13 excludes recently modified entries, live-handle matches, unattributed prefixes, product-created `ak-sync-preview-npm-*`, and valid owner roots. Age is a manual-review filter only. Task 13 does not implement deletion. No unit claims interrupted-run backlog reclamation until a platform can prove every descendant gone. diff --git a/docs/archive/2026-09-28-research-test-temp-folder-cleanup.md b/docs/archive/2026-09-28-research-test-temp-folder-cleanup.md new file mode 100644 index 00000000..a4a58dc0 --- /dev/null +++ b/docs/archive/2026-09-28-research-test-temp-folder-cleanup.md @@ -0,0 +1,449 @@ +# Test temp folder cleanup design + +## Status + +**Research complete.** The selected list-only policy was implemented and independently +reviewed in the [execution plan](2026-09-28-plan-runner-hygiene.md), through `e0fcc2eb`. +The remainder of this document preserves the initial research snapshot and its limits. + +Research snapshot: 2026-09-28, `e2f9dcae0554ff63921df618a819fd5e6afe80d2`, macOS Darwin 27.0.0, Node 26.4.0; Node 22.22.3 used for CLI checks. Selected behavior is **list-only for abandoned sibling roots on macOS, Linux and Windows**. No candidate establishes complete descendant liveness. This uses B9-R5's explicit fallback, preserves B9-R1–R8, and introduces no native sweeper or deletion authority. + +The [execution plan](2026-09-28-plan-runner-hygiene.md) maps subsequent work. Raw commands, JSON results, full lexical census, counts and literal experiment paths are retained in ignored `.superpowers/sdd/2026-09-28-runner-hygiene/`. No production/test code changed during this research unit. Creator lifecycles at that baseline are fully classified below; native Windows/Linux behavior remains explicitly unmeasured. + +## Verified runner and creator behavior + +At `scripts/run-tests.mjs:57-85`, the runner creates a unique suite root, redirects all three temp variables, runs synchronous commands, reports leftovers and removes its root after commands finish, including failure. SIGKILL cannot reach this cleanup. The source has no owner record or sibling collector. It strips FORCE_COLOR only (`scripts/run-tests.mjs:68`); LQ-1's premise that this layer strips AQE_EMBEDDER variables is refuted at this revision. Downstream boundaries still need sentinel tests. + +The completed AST and source census below classifies 647 creator sites across 279 files. The original 634-hit lexical inventory is superseded; two comment hits were excluded, three inline allocations and seven aliased calls were recovered, five sandboxConfigBase calls were added, and two executable child templates were retained separately. + +| Lifecycle class | Source evidence | Failure boundary | +|---|---|---| +| Cleanup registered before caller assertions | `tests/kit/helpers/temp-dir.mjs:16-20` creates then registers t.after or file after; `tests/kit/run-tests-runner.test.mjs:15-16` registers immediately | Assertion failure is covered after registration; process kill, allocation-to-registration failure and removal error remain | +| Manual cleanup after successful operations | `tests/kit/status-zero-spawn.test.mjs:86-90` removes its ledger after exec/read | Failed exec or parsing skips ledger removal; the parent run still diagnoses it | +| Module-level creator with file hook | `tests/kit/evidence.test.mjs:9` and `:231`; `tests/kit/refresh.test.mjs:17-19` | Abrupt exit misses hooks; in evidence, an early import/assertion may occur before hook registration | +| Exit hook in plain scripts | `tests/kit/helpers/private-tmpdir.cjs:11-16`; dashboard/statusline callers | Normal exit runs synchronous removal; SIGKILL never does; registration is after allocation | +| Caller-owned tools state | `tests/kit/helpers/home-sandbox.mjs:108-130` | redirectToolState returns restore; caller must arrange finally/hook before assertions; helper itself registers none | +| Caller-owned home/project | `tests/kit/helpers/home-sandbox.mjs:141-157` and `:303-306` | Creation does not register cleanup; consumers determine lifetime | +| Child temp base | `tests/kit/helpers/home-sandbox.mjs:83-97` | spawnEnv uses home/tmp and permits extra overrides; containment requires caller's home and final TMP values to be inside run root | +| Non-Node child | `tests/ui/helpers/launch-chrome.mjs:20-33` | Launch failure and browser.close remove private root; killed test or omitted close bypasses cleanup; Chrome is outside a Node-only registry | +| Hardcoded /tmp strings | `tests/kit/dashboard-live-source.test.mjs:7-21` | Path parsing and assertions only; these lines create no directory or file | + +Every site now has a current cleanup mechanism and evidence reference. This does not establish historical leak causality, successful native cleanup, or freedom from setup-before-registration gaps. The success-path classes identify assertion/error leak exposure; the parent-hook classes distinguish allocations already covered by enclosing cleanup. + +## Child temp routing and paths outside the run root + +The retained `temp-routing-sites.txt` records explicit temp-variable sites. `spawnEnv` places child temp under the supplied home; both it and redirectToolState remain beneath the outer suite when the home/base was created there. Nested runner refusal uses a temp base deliberately inside a disposable project (`tests/kit/run-tests-runner.test.mjs:119`); harvest redirects all three variables into its disposable repository (`tests/kit/agentdb-retirement.test.mjs:221-225`). Neither is a real repository escape. Chrome pins all four platform variables at `tests/ui/helpers/launch-chrome.mjs:22`. + +The source does contain child environments that lose outer-root containment: the live AQE version probe at `tests/live/aqe-stop-hook-conformance.test.mjs:40` passes only PATH and NO_COLOR, so it does not inherit the runner's temp variables. Shell command-discovery probes at `tests/live/aqe-stop-hook-conformance.test.mjs:24` and `tests/live/aqe-codex-guidance-conformance.test.mjs:43` likewise use PATH-only environments. These are source-confirmed routing gaps, not measured folder leaks. Later live fixtures explicitly set TMPDIR but do not establish Windows TEMP/TMP containment. The proof-key guard test sets only TMPDIR (`tests/kit/aqe-live-proof-key-guard.test.mjs:24`); its expected early refusal does not make that a portable temp-isolation contract. + +Literal Windows temp paths in ruflo-memory-location tests are injected path-classification inputs, not spawned child environments. dashboard-live-source's /tmp values are also parsing inputs. No cleanup design may assume every subprocess preserves the root simply because the top runner sets it. + +## The fourteen reported post-runner leaks + +The [archived premise table](2026-09-28-superpowers-plan-branch-9-follow-ups.md#premise-verification-done-by-the-planner-tasks-carry-the-evidence-forward) records fourteen newer folders by prefix. These are historical observations, not fourteen reproduced failures today. + +| Historical entries | Attributed creator and cleanup | What can be concluded | +|---|---|---| +| ak-evidence-home ×2 | `tests/kit/evidence.test.mjs:9`, file after at `:231` | Creation precedes imports/assertions and late hook. Early failure or killed process can leak; exact historical cause unknown | +| ak-ruflo-components-evidence-location-home ×2 | `tests/kit/ruflo-components-evidence-location.test.mjs:6`, after at `:18` | Early import/assertion before hook or interruption possible; exact cause unknown | +| ak-refresh-home ×3; ak-refresh-proj ×3 | `tests/kit/refresh.test.mjs:17-19` | Hook is early, so ordinary later assertion failure should clean. Kill, failure before registration or failed removal remain hypotheses | +| ak-status-live-home ×3 | `tests/kit/status-live.test.mjs:17`, after at `:267` | Late registration leaves early initialization failure window; interruption/removal failure possible | +| ak-live-checks-home ×1 | `tests/kit/live-checks.test.mjs:21`, after at `:688` | Late registration has the same exposure; no historical exit trace identifies cause | + +A basename and mtime cannot tell whether an assertion, import, kill or cleanup error caused a specific leak. Reconstructing that requires corresponding process/test logs. Do not rewrite this attribution as proof that every ordinary failed test leaks. + +## Concurrent runs and race windows + +Each mkdtemp root is unique, but sibling worktrees share the temp parent. An owner file is proposed to be written atomically via a temporary file and rename immediately after root creation. Until a valid record exists, keep the root (B9-R3). Old runners also have no owner and remain untouched. Malformed records, wrong host/user/platform, unreadable entries and unknown schema all mean keep. + +An owner may finish during inspection, a PID may be reused, a descendant may start after a snapshot, and a root could change between validation and deletion. Owner records and start identities do not close these races. Current selection performs no sibling removal, so racing observations cannot authorize it. A future collector needs complete process containment plus a stable filesystem identity/revalidation protocol; a path name and PID are insufficient. + +## Abandonment candidates and proof limits + +| Candidate | macOS | Linux | Windows | Signal and cost assessment | +|---|---|---|---|---| +| A: process group / parent descent | Reject: detached descendants escape recorded group | Same source counterexample; no native execution here | Reject: snapshots can lose exited intermediate parents; PID reuse complicates descent | detached changes session/group and requires signal forwarding; no equivalence proof | +| B: parent identity plus handle scan / descent | Reject: measured live orphan has no root handle | Same logical gap; native /proc and lsof not measured | ParentProcessId/CreationDate cannot recover every missing intermediate ancestor | Avoiding signal changes is possible; fast probes still cannot prove absence | +| C: every Node process imports PID registrar | Reject: non-Node children; env replacement; startup registration race | Same coverage gap; no native execution | Same coverage gap; no native execution | NODE_OPTIONS affects children and can be removed; no transparent behavior proof | +| Selected: list-only | Never returns abandoned | Never returns abandoned | Never returns abandoned | No process launch/signal change; no expensive liveness probe required for removal | + +`tests/kit/process-tree.test.mjs:52`, `tests/kit/exec-kill-tree.test.mjs:53` and `tests/kit/mcp-tool-call.test.mjs:21` deliberately use detached children. POSIX detached children create a new group/session; unref and stdio determine parent waiting behavior. Thus even an empty original group is insufficient. [Node child process documentation](https://nodejs.org/api/child_process.html#optionsdetached). + +Windows ParentProcessId may refer to a dead or reused parent. CreationDate helps disambiguate identity, but a snapshot cannot reconstruct an already vanished chain of intermediate processes. This is a design inference from the documented fields, not a native Windows experiment. [Microsoft Win32_Process](https://learn.microsoft.com/en-us/windows/win32/cimwin32prov/win32-process). + +B9-R6's startedAt milliseconds and 2-second reuse tolerance can distinguish a later process, but never establishes that the original process's children exited. Clock precision, permission errors and unparseable identity must resolve to unknown/keep. A newly started owner record should bind observed process start identity, not simply assume its file-write timestamp is the process start. + +## Real macOS orphan experiment and timings + +Scratch `probe.mjs` created an isolated temp parent, home, project and tmp. It launched the existing runner using `exec --repo -- `. The child wrote PID/cwd/TMPDIR to a handshake file then idled for 60 seconds. The controller killed only the exact runner PID, waited for its exit, inspected the known child and finally sent SIGTERM to that child. No process search was used to select kill targets. + +Trimmed evidence: + +```text +runner PID 57521: SIGKILL +child PID 57522: alive=true + PID PPID PGID COMMAND +57522 1 57490 node /sleep.mjs +child TMPDIR: /tmp/ak-suite-tHEZEe +child cwd: /project +root exists=true +lsof -nP +D : exit 1, stdout empty, stderr empty +cleanup ps -p 57522: header only; child gone +``` + +The child cwd was the disposable project, outside its suite root. It had no open root file. The root was retained for review. This proves candidate B unsound for this case and disproves parent-death-only collection. The script's finally targets only its exact owned PIDs; its bounded timer is secondary protection. + +Five sequential samples on this host (no cross-platform claim): + +| Probe/workload | Measured range | Median | +|---|---|---| +| process.kill(childPid, 0) | 0.00075–0.008 ms | 0.00108 ms | +| ps -o lstart= -p childPid | 1.625–2.054 ms | 1.742 ms | +| lsof -nP +D empty suite root | 144.769–150.222 ms | 145.634 ms | +| Plain node --test inert.test.mjs | 64.153–66.150 ms | 64.621 ms | +| Guarded exec of same one-file test | 91.135–100.622 ms | 96.753 ms | + +Observed median guarded overhead was 32.132 ms with 22 empty sandbox tripwire roots. This is not a measurement of a populated real home, a large backlog or the future collector. `/proc` and PowerShell CIM cost are **unmeasured** because there is no native Linux/Windows runtime in this task. A list-only inventory still needs bounded I/O and later performance validation against many roots; no production latency claim is made. + +## Windows locks and CI failure + +The controller supplied CI run **36520154869**, Windows Node 24: `status-zero-spawn.test.mjs` failed in inSandbox rmSync(project), line 55, EPERM; the leftover was ak-spawn-guard-smoke-proj. This failure is attributed evidence from the brief, not a freshly downloaded CI log. + +Verified source: the smoke script forks at `tests/kit/status-zero-spawn.test.mjs:82` then calls process.exit(0) at `:83`, explicitly avoiding waiting for the grandchild. execFileSync waits for that immediate child; it does not establish that the grandchild released its project cwd. A surviving grandchild causing the observed EPERM is a plausible hypothesis, not proven root cause. The fork target itself exits immediately, making scheduling relevant. + +Proposed regression: in a copied disposable fixture, hold the fork on a bounded handshake; record exact child/grandchild PID and spawn/exit/close timestamps; attempt cleanup while held; compare an explicit wait-for-close variant. Capture Windows error code/path and remaining entries. Preserve both assertion and cleanup errors rather than letting finally mask the first. The ignored `windows-probe-plan.md` specifies native Windows Node 24/26 diagnostics, one timed CIM snapshot and exact-PID cleanup. No workflow was edited or run. + +Recursive rm is not atomic; an error can leave a partially removed tree. On a removal error, report retained/partially removed and never claim an intact preserved root. Windows cwd/open-handle behavior depends on handle sharing; not every open handle universally blocks deletion. Native locking behavior remains unmeasured here. The proof must precede any future removal attempt. + +Node documents recursive rm retries for EBUSY, EMFILE, ENFILE, ENOTEMPTY and EPERM with linear backoff; maxRetries defaults to 0 and retryDelay to 100 ms. These options are ignored without recursive mode. Retries cannot establish ownership or abandonment. [Node fs.rmSync](https://nodejs.org/api/fs.html#fsrmsyncpath-options). + +## Focused runs on Node 22 and 26 + +Both installed binaries were executed in the sandbox with an invalid node.config.json and one inert test. Plain `node --test inert.test.mjs` passed on 22.22.3 and 26.4.0, demonstrating that neither loaded the config by default. Adding `--experimental-default-config-file` failed with exit 9 and invalid-content diagnostics on both. Explicit `--experimental-config-file` and `--import` are also opt-in; external NODE_OPTIONS can inject imports, but a repository file cannot silently establish that environment. + +`--test-global-setup=./missing.mjs` is rejected as a bad option (exit 9) by installed 22.22.3; 26.4.0 recognizes it and fails resolving the intentionally missing module (exit 7). Thus the inherited wording must not imply global setup exists on Node 22.22.3. B9-R7's conclusion stands: plain node --test is unguarded; use the explicit wrapper. [Node 22.22.3 CLI](https://nodejs.org/download/release/v22.22.3/docs/api/cli.html), [Node 26.4.0 CLI](https://nodejs.org/download/release/v26.4.0/docs/api/cli.html). + +## Backlog and classification boundary + +A read-only direct-child listing of the real temp parent observed **34,013** current-user, non-symlink entries beginning ak-, grouped into 127 suffix-normalized prefixes. These raw counts include this research's own new root and concurrent activity; they are not removal candidates. The full per-prefix snapshot is `backlog-counts.json`, generated by retained `census.py`. Leading counts: ak-usage 2,520; ak-adapter-conformance-cli 2,301; ak-quota 2,152; ak-live-service 2,150; ak-adapter-consent 2,124; ak-intel-history 2,106; ak-adapter-grants 1,836; ak-stamp 1,525; ak-conformance-tiers-grants 1,512; ak-usage-solo 1,400; ak-host-cli 855; ak-host-project 855. + +Task 13 must take a fresh snapshot after the last focused run. Direct children only, real absolute parent, current owner, no symlinks; exclude valid owner roots, product ak-sync-preview-npm roots, all paths found by one lsof snapshot, anything changed in 24 hours, and unattributed prefixes. A: before 2026-09-27 12:01 local runner landing; B: later attributable leaks; C: legacy ownerless suite roots. Keep all exclusion counts and script source, independently rederive counts, and submit literal paths for maintainer judgment. A missing lsof result due to error is incomplete evidence, not proof that nothing is in use. Idle orphans can evade lsof, so this remains a manual review list with no deletion authority. + +## Completed creator census + +The final census classifies **647 creator sites in 279 files, with zero unclassified current lifecycle sites**: 645 AST calls plus two executable child-template sites. It adds seven `makeTempDir` aliases, five sandboxConfigBase calls and three inline allocator definitions, and removes two comment-only lexical hits. Local factory invocations are represented by their allocator definition and caller contract, rather than counted as additional allocations. The retained AST/parser and manual override scripts reproduce the census without executing tests or tools. + +This is a source lifecycle classification, not a guarantee that cleanup runs after SIGKILL or that recursive removal succeeds. Each returned fixture is traced to caller cleanup; mixed callers stay mixed. Assertions before hook registration were checked in the allocating scope, excluding callbacks that run later. `sandboxHome` and `sandboxProject` register **no** cleanup themselves; they must not inherit tempDir's safe-return contract. Allocation, setup and multiple cleanup operations can still throw before protection or skip a later cleanup. + +| Code | Sites | Current lifecycle | +|---|---:|---| +| H | 130 | Shared helper registers cleanup before return; see temp-dir:19 or home-sandbox:176. | +| R | 168 | Local test/file hook registration; cleanup line shown. No preceding direct assert/assertSandboxed call in the allocating scope was found. | +| M | 87 | Module allocation with file hook; early imports/assertions can precede registration. | +| F | 113 | Removal in finally; setup before entering try remains exposed. | +| S | 67 | Direct cleanup reached only on normal execution, often after assertions. | +| CF | 20 | Factory returns root; callers remove in finally; pre-return setup remains exposed. | +| CH | 1 | Factory returns root; caller registers a hook; pre-registration setup remains exposed. | +| CM | 5 | Factory callers have mixed success-only/finally/hook lifecycles; exact examples in JSON. | +| O | 2 | Helper transfers ownership without registering cleanup; callers classified separately. | +| CS | 15 | Factory returns root; callers remove only on normal execution. | +| E | 6 | Process exit handler; normal exit only. | +| XS | 1 | Executable child template uses success-path cleanup. | +| XL | 1 | Executable child template deliberately leaks to exercise runner detection. | +| PE | 10 | Allocation covered transitively by enclosing private temp exit handler. | +| P | 16 | Allocation covered transitively by parent folder cleanup hook. | +| G | 2 | Root added to collection consumed by already registered/file cleanup hook. | +| C | 3 | Returns a cleanup method; construction failure handling and caller obligations described in JSON. | + +Source index below uses `allocation line:code→cleanup/caller evidence line` within the named file. H and E also use the shared helper references in the legend. Full cleanup expressions, all evidence locations, caller notes and scope boundaries are in ignored `creator-lifecycle-final.json`; reproducible scripts are `ast-census.mjs`, `lifecycle.mjs`, `classify.py` and `render-census.py`. No site is deemed safe simply because its file contains an unrelated hook. + +| Source file | Every creator site and lifecycle evidence | +|---|---| +| `tests/dashboard.test.cjs` | 19:E→helper; 70:PE→19 | +| `tests/kit/about-install-edits.test.mjs` | 11:M→12 | +| `tests/kit/about-security.test.mjs` | 10:M→11 | +| `tests/kit/adapter-admission.test.mjs` | 253:H→helper; 274:H→helper; 303:H→helper; 506:F→540 | +| `tests/kit/adapter-aqe-provider.test.mjs` | 89:CM→115,206,252; 247:H→helper; 350:H→helper; 498:F→514 | +| `tests/kit/adapter-conformance.test.mjs` | 127:S→240; 333:F→345; 356:F→369 | +| `tests/kit/adapter-execution.test.mjs` | 481:F→499 | +| `tests/kit/adapter-grants.test.mjs` | 16:H→helper; 169:H→helper | +| `tests/kit/adapter-hook-runner.test.mjs` | 197:F→208; 304:CS→305,313,320 | +| `tests/kit/adapter-integrity.test.mjs` | 48:F→61; 66:F→76; 81:F→96; 109:F→129; 134:F→152; 157:F→173; 158:F→174; 179:F→187 | +| `tests/kit/adapter-registries.test.mjs` | 74:H→helper | +| `tests/kit/adapter-sources.test.mjs` | 150:F→159; 171:F→182; 187:F→194; 199:F→208; 349:F→361; 367:F→376; 382:F→389; 395:F→402; 439:F→453; 458:F→470; 475:F→487; 492:F→505; 512:F→528 | +| `tests/kit/agent-browser-runtime.test.mjs` | 23:CS→23,88,126 | +| `tests/kit/agentdb-retirement.test.mjs` | 26:M→27; 40:M→41; 218:R→225 | +| `tests/kit/ak-launcher-evidence.test.mjs` | 17:H→helper | +| `tests/kit/aqe-embedding-probe.test.mjs` | 10:R→11 | +| `tests/kit/aqe-embedding-projection.test.mjs` | 14:R→15 | +| `tests/kit/aqe-embedding-transport.test.mjs` | 93:R→94; 109:R→110 | +| `tests/kit/aqe-guidance.test.mjs` | 38:R→39; 50:R→51; 71:H→helper | +| `tests/kit/aqe-lifecycle-migration.test.mjs` | 18:R→19 | +| `tests/kit/aqe-live-proof-key-guard.test.mjs` | 15:H→helper | +| `tests/kit/aqe-project-pin.test.mjs` | 24:H→helper; 283:H→helper; 349:H→helper | +| `tests/kit/aqe-readiness.test.mjs` | 11:F→17 | +| `tests/kit/aqe-store-holders.test.mjs` | 28:H→helper; 39:H→helper; 47:H→helper; 57:H→helper; 78:H→helper; 93:H→helper; 131:H→helper; 149:H→helper; 157:H→helper; 167:H→helper; 183:H→helper; 196:H→helper | +| `tests/kit/aqe-store-merge-fixture.test.mjs` | 30:H→helper; 49:H→helper; 71:H→helper | +| `tests/kit/aqe-store-merge-preview.test.mjs` | 44:H→helper | +| `tests/kit/aqe-store-merge.test.mjs` | 174:H→helper | +| `tests/kit/blocks-drift-parity.test.mjs` | 19:M→20; 30:M→31,36 | +| `tests/kit/blocks-dual-mode.test.mjs` | 33:S→53; 58:S→74; 79:S→96 | +| `tests/kit/blocks.test.mjs` | 99:S→106; 110:S→135; 139:S→146; 178:S→185; 189:S→199 | +| `tests/kit/brain-held-refresh-sync.test.mjs` | 21:M→36; 31:M→36 | +| `tests/kit/brain-held-refresh.test.mjs` | 13:M→14 | +| `tests/kit/claude-env-projection.test.mjs` | 13:R→15; 14:R→15 | +| `tests/kit/claude-window-ledger.test.mjs` | 14:H→helper | +| `tests/kit/clean-machine-setup.test.mjs` | 14:S→46; 50:S→76 | +| `tests/kit/cli-help.test.mjs` | 13:M→14 | +| `tests/kit/cli-json-honesty.test.mjs` | 16:M→18; 17:M→18 | +| `tests/kit/codex-context-command.test.mjs` | 7:M→67 | +| `tests/kit/codex-context.test.mjs` | 11:R→12 | +| `tests/kit/codex-mcp-convergence.test.mjs` | 11:M→12; 23:M→24 | +| `tests/kit/codex-mcp.test.mjs` | 14:CF→18,38,46; 94:F→126; 131:F→145; 150:F→164; 169:F→178; 183:F→196; 201:F→219; 224:F→256; 261:F→297; 302:F→327; 332:F→354; 360:F→398 | +| `tests/kit/codex-plugins.test.mjs` | 10:M→269; 214:F→231 | +| `tests/kit/codex-state.test.mjs` | 13:H→helper | +| `tests/kit/codex-statusline.test.mjs` | 13:G→14,103 | +| `tests/kit/codex-usage-diagnostic.test.mjs` | 13:R→14 | +| `tests/kit/conformance-tiers.test.mjs` | 24:H→helper; 260:H→helper; 298:H→helper; 317:H→helper; 355:H→helper; 389:H→helper; 673:H→helper | +| `tests/kit/context-audit.test.mjs` | 151:F→178 | +| `tests/kit/daemon-sweep-evidence.test.mjs` | 33:H→helper; 49:F→54 | +| `tests/kit/daemons-status.test.mjs` | 11:R→12 | +| `tests/kit/dashboard-context-hooks.test.mjs` | 203:F→262; 267:F→304 | +| `tests/kit/dashboard-hermetic-defaults.test.mjs` | 15:M→16; 56:P→14,16,56 | +| `tests/kit/dashboard-intel-integration.test.mjs` | 129:H→helper | +| `tests/kit/dashboard-project-identity.test.mjs` | 16:R→17 | +| `tests/kit/dashboard-status-cost.test.mjs` | 27:F→75; 29:F→76 | +| `tests/kit/dashboard-status-inprocess.test.mjs` | 36:CF→48,101,104; 38:CF→48,101,104 | +| `tests/kit/deja-vu-lifecycle.test.mjs` | 17:H→helper | +| `tests/kit/deja-vu-teardown-verify.test.mjs` | 15:M→503 | +| `tests/kit/deja-vu.test.mjs` | 160:F→197; 202:F→237; 242:F→277; 282:F→339; 344:F→359 | +| `tests/kit/dispatch-surface.test.mjs` | 79:H→helper | +| `tests/kit/disposable-memory-project.test.mjs` | 40:R→41; 130:R→131 | +| `tests/kit/drift-freshness.test.mjs` | 16:M→17; 26:M→27 | +| `tests/kit/dry-run-nudge.test.mjs` | 26:CF→43,64,65; 28:CF→43,64,65; 37:CF→43,64,65 | +| `tests/kit/evidence.test.mjs` | 9:M→231 | +| `tests/kit/exec-kill-tree.test.mjs` | 23:R→24; 81:R→82 | +| `tests/kit/exec.test.mjs` | 93:F→111; 147:F→185; 190:F→215; 222:F→234 | +| `tests/kit/execution-runner.test.mjs` | 470:H→helper | +| `tests/kit/external-lifecycle.test.mjs` | 31:M→445; 172:F→188; 193:F→209; 214:F→230; 235:F→251; 258:F→271; 276:F→288; 293:F→306; 313:F→328; 335:F→356; 369:F→440; 370:F→441 | +| `tests/kit/file-identity-bigint.test.mjs` | 151:H→helper | +| `tests/kit/footprint-collectors.test.mjs` | 50:R→51 | +| `tests/kit/footprint-executable-paths.test.mjs` | 9:R→10 | +| `tests/kit/footprint-known-files.test.mjs` | 12:F→19 | +| `tests/kit/footprint-observation-forest.test.mjs` | 11:R→12 | +| `tests/kit/footprint-performance.test.mjs` | 20:R→21 | +| `tests/kit/footprint-projects.test.mjs` | 44:R→45 | +| `tests/kit/footprint-snapshot-v2.test.mjs` | 12:R→13 | +| `tests/kit/footprint-stack.test.mjs` | 47:R→48 | +| `tests/kit/guidance-targets.test.mjs` | 34:S→40; 74:S→82; 88:S→95; 100:S→109; 101:S→110; 239:S→259; 265:S→279; 283:S→296 | +| `tests/kit/heal-natives.test.mjs` | 30:H→helper; 34:H→helper; 52:H→helper; 114:H→helper; 139:H→helper; 150:H→helper; 166:H→helper; 214:H→helper; 254:H→helper; 277:H→helper; 297:H→helper | +| `tests/kit/helper-stamp.test.mjs` | 85:H→helper | +| `tests/kit/helpers/aqe-store-merge-fixture.mjs` | 14:H→helper | +| `tests/kit/helpers/aqe-store-merge-harness.mjs` | 119:H→helper | +| `tests/kit/helpers/codex-rollout.mjs` | 130:H→helper | +| `tests/kit/helpers/home-sandbox.mjs` | 110:C→124,125,129; 140:O→139,158,303; 169:R→181; 304:O→139,158,303 | +| `tests/kit/helpers/private-tmpdir.cjs` | 12:E→16 | +| `tests/kit/helpers/project-isolation.mjs` | 117:R→122 | +| `tests/kit/helpers/temp-dir.mjs` | 17:H→18,19,20 | +| `tests/kit/home-sandbox-tripwire.test.mjs` | 19:R→20; 27:XS→49,50,63 | +| `tests/kit/hook-audit-hosts.test.mjs` | 21:CF→138,139,164 | +| `tests/kit/hook-audit.test.mjs` | 20:CF→61,62,90 | +| `tests/kit/hook-auto-memory-retirement.test.mjs` | 123:CF→176,177,192 | +| `tests/kit/hook-legacy-retirement.test.mjs` | 84:CF→125,126,157 | +| `tests/kit/hook-remediation-cli.test.mjs` | 27:F→83; 88:F→141 | +| `tests/kit/hook-remediation.test.mjs` | 30:CF→68,69,98; 104:F→164; 169:F→204; 240:F→274; 279:F→305; 432:F→456; 481:F→506; 531:F→537; 542:F→553 | +| `tests/kit/hook-upstream.test.mjs` | 17:F→25 | +| `tests/kit/host-adapters-cli.test.mjs` | 77:H→helper; 754:H→helper | +| `tests/kit/host-alignment.test.mjs` | 7:M→8; 9:M→10; 11:M→12 | +| `tests/kit/host-cli-migration.test.mjs` | 13:H→helper; 14:H→helper | +| `tests/kit/host-dry-run.test.mjs` | 38:R→39; 41:R→42 | +| `tests/kit/host-executable.test.mjs` | 9:H→helper | +| `tests/kit/host-health-connected.test.mjs` | 202:R→203 | +| `tests/kit/host-health-evidence.test.mjs` | 9:R→10; 29:R→30; 39:R→40 | +| `tests/kit/host-readiness-local.test.mjs` | 9:R→12 | +| `tests/kit/host-setup-evidence.test.mjs` | 28:H→helper; 44:F→52 | +| `tests/kit/hosts.test.mjs` | 152:H→helper | +| `tests/kit/install-edits.test.mjs` | 21:R→22 | +| `tests/kit/integration-command-facts.test.mjs` | 9:M→10 | +| `tests/kit/intel-history.test.mjs` | 21:H→helper | +| `tests/kit/intelligence-picker-groups.test.mjs` | 51:R→52; 67:R→68 | +| `tests/kit/intelligence-table-groups.test.mjs` | 10:R→11 | +| `tests/kit/intelligence-watch.test.mjs` | 8:H→helper | +| `tests/kit/language-coverage.test.mjs` | 10:R→10 | +| `tests/kit/live-check-evidence.test.mjs` | 15:M→260; 25:M→260 | +| `tests/kit/live-checks.test.mjs` | 21:M→688; 30:M→688; 401:F→415; 421:R→422; 440:F→449; 468:R→469; 701:R→702; 733:R→734 | +| `tests/kit/live-core.test.mjs` | 149:H→helper; 178:H→helper; 202:H→helper; 226:H→helper; 248:H→helper; 263:H→helper; 274:H→helper; 275:H→helper; 288:H→helper; 299:H→helper; 309:H→helper | +| `tests/kit/live-folder-correlator.test.mjs` | 14:R→15 | +| `tests/kit/live-process-sessions.test.mjs` | 250:F→293; 299:F→314 | +| `tests/kit/live-qe-contract.test.mjs` | 40:H→helper | +| `tests/kit/live-service.test.mjs` | 9:H→helper | +| `tests/kit/live-tailer.test.mjs` | 9:H→helper; 150:H→helper | +| `tests/kit/live-transcript.test.mjs` | 15:H→helper | +| `tests/kit/maintenance-action-service.test.mjs` | 25:R→26 | +| `tests/kit/maintenance-cli.test.mjs` | 15:H→helper | +| `tests/kit/maintenance-dashboard-api.test.mjs` | 600:R→601 | +| `tests/kit/maintenance-dashboard-e2e.test.mjs` | 177:R→179 | +| `tests/kit/maintenance-discovery-checkpoint.test.mjs` | 14:R→15 | +| `tests/kit/maintenance-discovery-configuration.test.mjs` | 21:R→22 | +| `tests/kit/maintenance-discovery-coverage.test.mjs` | 15:R→16 | +| `tests/kit/maintenance-discovery-orchestrator.test.mjs` | 23:R→24; 649:R→650 | +| `tests/kit/maintenance-discovery-preview.test.mjs` | 17:R→18 | +| `tests/kit/maintenance-git-project-patch.test.mjs` | 26:R→27; 41:R→42; 178:R→179 | +| `tests/kit/maintenance-host-alignment.test.mjs` | 6:M→7; 8:M→9 | +| `tests/kit/maintenance-interruption-audit.test.mjs` | 21:R→22 | +| `tests/kit/maintenance-management-activity.test.mjs` | 15:H→helper; 149:H→helper | +| `tests/kit/maintenance-management-procedures.test.mjs` | 21:R→22 | +| `tests/kit/maintenance-management-service.test.mjs` | 42:R→43 | +| `tests/kit/maintenance-native-findings.test.mjs` | 24:R→25 | +| `tests/kit/maintenance-one-action.test.mjs` | 20:R→21 | +| `tests/kit/maintenance-owned-providers.test.mjs` | 25:R→26 | +| `tests/kit/maintenance-persistence-support.test.mjs` | 22:R→23 | +| `tests/kit/maintenance-project-kind.test.mjs` | 13:R→14 | +| `tests/kit/maintenance-read-model.test.mjs` | 87:R→88; 104:R→105; 144:R→145; 187:R→188; 205:R→206; 236:R→237; 251:R→252; 269:R→270 | +| `tests/kit/maintenance-recovery.test.mjs` | 20:R→21 | +| `tests/kit/maintenance-transaction.test.mjs` | 19:R→20 | +| `tests/kit/mcp-scopes.test.mjs` | 21:R→26 | +| `tests/kit/mcp-tool-call.test.mjs` | 64:R→65 | +| `tests/kit/memory-maintenance.test.mjs` | 17:R→18 | +| `tests/kit/memory-probe-cleanup.test.mjs` | 16:M→17,186; 59:R→60 | +| `tests/kit/model-dashboard-read-model.test.mjs` | 553:F→604; 609:F→645 | +| `tests/kit/model-inventory-store.test.mjs` | 27:CS→38,60,73 | +| `tests/kit/natives-probe.test.mjs` | 16:CF→31,40,52 | +| `tests/kit/natives-runtime.test.mjs` | 23:H→helper; 26:CM→39,47,57; 74:S→82 | +| `tests/kit/natives.test.mjs` | 13:CS→29,41,47; 33:S→35; 53:S→60; 68:S→79; 88:S→95; 99:S→106; 110:S→113; 117:CF→128,130,143 | +| `tests/kit/node-runtime.test.mjs` | 23:H→helper | +| `tests/kit/npx.test.mjs` | 33:CS→40,47,54; 109:S→116 | +| `tests/kit/nudge.test.mjs` | 15:H→helper | +| `tests/kit/opencode-agents-stale-reason.test.mjs` | 25:M→39; 35:M→39 | +| `tests/kit/opencode-aqe-embedding.test.mjs` | 9:R→10 | +| `tests/kit/opencode-ruflo-gateway.test.mjs` | 8:CF→9,237,238 | +| `tests/kit/opencode-state-hermeticity.test.mjs` | 27:F→32; 43:R→44 | +| `tests/kit/opencode-stock-ruflo-gateway.test.mjs` | 33:CH→280,289,297 | +| `tests/kit/opencode-version-drift.test.mjs` | 14:M→56 | +| `tests/kit/opencode.test.mjs` | 20:M→21; 23:P→20,21,23 | +| `tests/kit/output-progress.test.mjs` | 119:H→helper | +| `tests/kit/owned-env-backup-prune.test.mjs` | 13:M→14 | +| `tests/kit/owned-env-projection.test.mjs` | 11:R→12; 131:R→132 | +| `tests/kit/paths-global-root-evidence.test.mjs` | 37:H→helper; 156:H→helper; 159:F→198 | +| `tests/kit/paths-global-root.test.mjs` | 94:F→106; 111:F→117 | +| `tests/kit/project-census.test.mjs` | 27:H→helper | +| `tests/kit/project-guidance.test.mjs` | 20:R→21 | +| `tests/kit/project-isolation.test.mjs` | 27:R→28 | +| `tests/kit/project-memory-status.test.mjs` | 11:R→12; 37:R→38; 83:R→92; 118:R→119; 139:R→140; 156:R→157; 170:R→171; 195:R→196; 211:R→212; 235:R→236; 328:R→329 | +| `tests/kit/project-memory.test.mjs` | 11:M→51,65,79; 82:R→83; 98:R→99; 107:R→108; 120:R→121; 133:R→134; 143:R→144; 160:R→168; 190:R→191; 201:R→202; 222:R→223; 269:R→270; 283:R→284 | +| `tests/kit/project-sources-imports.test.mjs` | 21:R→22 | +| `tests/kit/prompts-mainline-boundary.test.mjs` | 14:F→22 | +| `tests/kit/provider-cli.test.mjs` | 51:CS→86,103,109; 62:CS→86,103,109; 193:CF→297,335,336; 215:CF→297,335,336 | +| `tests/kit/provider-credentials.test.mjs` | 24:M→25; 33:P→24,25,33; 40:P→24,25,42 (local S→42) | +| `tests/kit/provider-refresh-cli.test.mjs` | 28:CS→48,84,89; 43:CS→48,84,89 | +| `tests/kit/provider-teardown-preservation.test.mjs` | 9:R→11 | +| `tests/kit/providers-drift-parity.test.mjs` | 22:M→113; 62:R→63; 96:R→97 | +| `tests/kit/providers-external.test.mjs` | 22:G→16,18,23 | +| `tests/kit/providers.test.mjs` | 170:CS→175,193,210; 178:S→180; 371:S→378; 437:F→466; 475:S→483 | +| `tests/kit/qeCourt.test.mjs` | 214:S→216; 220:S→226; 230:F→253; 257:F→279; 283:R→284 | +| `tests/kit/quota-codex-presence.test.mjs` | 18:M→25 | +| `tests/kit/quota.test.mjs` | 17:H→helper | +| `tests/kit/real-state-tripwire.test.mjs` | 16:R→17 | +| `tests/kit/reference-command.test.mjs` | 11:M→64; 19:F→60 | +| `tests/kit/refresh.test.mjs` | 17:M→19; 18:M→19 | +| `tests/kit/reverse-bridge.test.mjs` | 10:H→helper; 67:H→helper | +| `tests/kit/routing-config.test.mjs` | 147:S→176; 180:S→189; 193:S→211 | +| `tests/kit/routing-projection.test.mjs` | 13:CS→17,49,50; 21:CS→17,49,50 | +| `tests/kit/routing-retirement-convergence.test.mjs` | 29:M→30; 109:R→110,124; 128:R→129,154; 165:R→166,185 | +| `tests/kit/ruflo-components-apply.test.mjs` | 132:R→133; 164:R→165; 189:R→190; 211:R→212; 234:R→235; 245:R→246; 262:R→263; 276:R→277 | +| `tests/kit/ruflo-components-catalogue.test.mjs` | 81:R→82 | +| `tests/kit/ruflo-components-convergence.test.mjs` | 12:M→13; 42:R→43; 197:R→198; 213:R→214 | +| `tests/kit/ruflo-components-env.test.mjs` | 15:R→16 | +| `tests/kit/ruflo-components-evidence-location.test.mjs` | 6:M→18 | +| `tests/kit/ruflo-components-evidence.test.mjs` | 67:R→68; 188:R→189 | +| `tests/kit/ruflo-components-git-exclude.test.mjs` | 19:H→helper; 33:R→34; 100:R→101 | +| `tests/kit/ruflo-components-hosts.test.mjs` | 19:R→20; 45:R→46; 142:R→143; 161:F→170 | +| `tests/kit/ruflo-components-snapshot.test.mjs` | 194:F→210 | +| `tests/kit/ruflo-daemon-config.test.mjs` | 19:R→20 | +| `tests/kit/ruflo-mcp-launcher.test.mjs` | 21:R→22; 83:R→84 | +| `tests/kit/ruflo-memory-location.test.mjs` | 19:R→20; 50:R→51 | +| `tests/kit/ruflo-memory-root-pin.test.mjs` | 24:R→25 | +| `tests/kit/ruflo-memory.test.mjs` | 11:F→31; 39:R→40 | +| `tests/kit/run-tests-runner.test.mjs` | 15:R→16; 108:XL→16,110 | +| `tests/kit/ruvector.test.mjs` | 18:M→214; 25:M→214 | +| `tests/kit/ruvnet-brain-plugin.test.mjs` | 10:R→11 | +| `tests/kit/ruvnet-brain.test.mjs` | 87:H→helper; 118:H→helper; 181:H→helper; 196:H→helper; 213:H→helper; 222:H→helper; 238:H→helper; 257:H→helper | +| `tests/kit/rvf.test.mjs` | 22:CS→49,56,64 | +| `tests/kit/scaffold.test.mjs` | 19:CF→34,42,52; 25:CF→34,42,52; 85:F→91; 143:F→157 | +| `tests/kit/security-status.test.mjs` | 12:M→13 | +| `tests/kit/settings-config.test.mjs` | 12:S→19; 29:S→34; 38:S→45; 49:S→57; 61:S→67; 71:S→77; 81:S→99; 103:S→119; 130:S→141; 146:S→153; 178:S→188; 192:S→197; 201:S→209; 213:S→224; 228:S→234 | +| `tests/kit/setup-command.test.mjs` | 19:M→104,775; 122:S→136; 191:S→197; 208:S→220; 225:S→239; 474:F→484; 489:F→506; 512:F→541; 546:F→588; 593:S→601; 606:S→622; 607:P→775; 627:P→775; 783:S→798; 803:S→811 | +| `tests/kit/setup-host-flags.test.mjs` | 103:S→107 | +| `tests/kit/setup-memory-probe.test.mjs` | 26:R→27; 83:R→84 | +| `tests/kit/spawn-env-guard.test.mjs` | 176:R→177 | +| `tests/kit/sqlite.test.mjs` | 18:S→27; 31:S→42; 46:S→60 | +| `tests/kit/status-agent-browser.test.mjs` | 13:R→17 | +| `tests/kit/status-aqe-drift.test.mjs` | 17:M→18; 58:R→59; 71:R→72; 83:R→84; 103:H→helper; 127:R→128; 145:R→146; 160:R→161; 171:R→172 | +| `tests/kit/status-command.test.mjs` | 19:M→1451; 29:M→386,536,671; 340:F→387 | +| `tests/kit/status-golden.test.mjs` | 21:M→22; 29:M→30 | +| `tests/kit/status-live.test.mjs` | 17:M→267; 31:M→144,267; 125:F→144 | +| `tests/kit/status-manual-fixes.test.mjs` | 17:M→28; 27:M→28 | +| `tests/kit/status-repair-contract.test.mjs` | 17:M→18; 30:M→31,146 | +| `tests/kit/status-setup-hints.test.mjs` | 14:M→38,62; 22:M→62; 23:P→14,23,62 | +| `tests/kit/status-version-drift-refresh.test.mjs` | 27:M→40; 37:M→40; 47:P→40,47 | +| `tests/kit/status-viability.test.mjs` | 24:M→347; 35:M→347 | +| `tests/kit/status-zero-spawn.test.mjs` | 41:F→54; 43:F→55 | +| `tests/kit/statusline-config-dir-parity.test.mjs` | 42:H→helper | +| `tests/kit/statusline-version.test.mjs` | 20:M→212; 62:P→62,212 | +| `tests/kit/statusline.test.mjs` | 22:M→23; 44:H→helper | +| `tests/kit/sync-command.test.mjs` | 20:M→35,1320; 31:M→1320 | +| `tests/kit/sync-daemon-repair.test.mjs` | 16:M→17; 27:CF→46,61,76 | +| `tests/kit/sync-dry-run-preview.test.mjs` | 20:M→42; 33:M→42; 183:P→40,42,339; 184:P→40,42,339; 254:P→40,42,339; 325:P→40,42,339; 339:P→40,42,339 | +| `tests/kit/sync-host-repair.test.mjs` | 5:M→6 | +| `tests/kit/sync-needs-your-action.test.mjs` | 25:M→26; 35:M→36 | +| `tests/kit/sync-self-freshness.test.mjs` | 10:M→11 | +| `tests/kit/sync-skip-versions.test.mjs` | 22:M→43; 34:M→43; 107:P→41,43,107 | +| `tests/kit/system-command.test.mjs` | 21:H→helper; 22:H→helper | +| `tests/kit/system-summary.test.mjs` | 644:H→helper; 666:H→helper | +| `tests/kit/telemetry-cli.test.mjs` | 15:R→16; 19:H→helper | +| `tests/kit/telemetry-source-bounds.test.mjs` | 19:R→20 | +| `tests/kit/temp-dir-helper.test.mjs` | 10:H→helper | +| `tests/kit/uninstall-command.test.mjs` | 18:M→493; 170:S→187; 192:S→203; 449:CS→448,452,474; 458:S→474; 501:S→519; 507:S→519 | +| `tests/kit/upstream-watch-fixtures.mjs` | 31:F→48 | +| `tests/kit/upstream-watch-ledger-branch.test.mjs` | 128:F→167 | +| `tests/kit/upstream-watch-registry.test.mjs` | 30:F→38 | +| `tests/kit/upstream-watch-script.test.mjs` | 878:F→904 | +| `tests/kit/usage-audit-211.test.mjs` | 15:R→16 | +| `tests/kit/usage-claude-dedup.test.mjs` | 193:H→helper | +| `tests/kit/usage-cli.test.mjs` | 16:H→helper | +| `tests/kit/usage-codex-large-rollout.test.mjs` | 22:H→helper | +| `tests/kit/usage-deps-contract.test.mjs` | 42:H→helper | +| `tests/kit/usage-git-projects.test.mjs` | 37:R→38 | +| `tests/kit/usage-index-claude-window.test.mjs` | 31:CM→70,81,138 | +| `tests/kit/usage-index-opencode.test.mjs` | 15:CM→118,130,268 | +| `tests/kit/usage-index-v6.test.mjs` | 87:H→helper | +| `tests/kit/usage-index.test.mjs` | 33:H→helper; 66:H→helper; 929:H→helper; 970:H→helper | +| `tests/kit/usage-local-pricing.test.mjs` | 20:CF→126,133,173 | +| `tests/kit/usage-opencode.test.mjs` | 14:CM→95,137,325 | +| `tests/kit/usage-openrouter.test.mjs` | 13:CS→137,192,211 | +| `tests/kit/usage-project-groups.test.mjs` | 81:R→82 | +| `tests/kit/usage-truncation.test.mjs` | 43:H→helper | +| `tests/kit/verify-memory-routes.test.mjs` | 91:M→92; 96:M→97; 106:H→helper; 157:H→helper; 217:R→219; 218:P→217,218,219 | +| `tests/kit/version-lookup-record.test.mjs` | 22:M→41 | +| `tests/kit/versions.test.mjs` | 112:F→126 | +| `tests/kit/working-context.test.mjs` | 9:R→10 | +| `tests/live/aqe-codex-guidance-conformance.test.mjs` | 84:R→85 | +| `tests/live/aqe-external-provider-transport.test.mjs` | 258:R→280 | +| `tests/live/aqe-stop-hook-conformance.test.mjs` | 44:R→45 | +| `tests/live/codex-context-contract.test.mjs` | 16:R→17 | +| `tests/live/disposable-memory-project.mjs` | 49:C→41,43,57 | +| `tests/live/ruflo-memory-routing.test.mjs` | 26:M→28 | +| `tests/statusline-brain.test.cjs` | 18:E→helper; 41:PE→18 | +| `tests/statusline-segments.test.cjs` | 19:E→helper; 64:PE→19; 346:PE→19; 380:PE→19 | +| `tests/statusline-window-ledger.test.cjs` | 21:E→helper; 45:F→54; 176:PE→21; 186:PE→21 | +| `tests/ui/dashboard-ui.mjs` | 68:E→helper; 162:PE→68; 1102:PE→68; 1103:PE→68 | +| `tests/ui/helpers/launch-chrome.mjs` | 20:C→21,29,31 | + +## Decisions for Tasks 11–13 + +1. Implement one proveAbandoned interface with **abandoned=false, reason=cannot prove complete descendant exit** by default on macOS/Linux/Windows. Unit fixtures may exercise collector plumbing but cannot enable a production deletion path or constitute native platform proof. +2. Proposed owner fields: schema=1, random runId, absolute canonical root/temp parent, pid, startedAt milliseconds, hostname, platform, uid (null only when unavailable), proofMode=list-only. Validate schema, finite positive PID/timestamp, host/user identity, path and file type. Atomic record writing improves attribution only. +3. Retain B9-R4 path rules for own-root removal; refuse real home/filesystem-root temp bases before allocation, check absolute canonical direct parent, exact suite basename, lstat/non-symlink, POSIX owner. Unknown/error means refuse. List sibling roots without deleting them; do not change command/tripwire/leftover exit precedence (B9-R2). +4. Preserve synchronous runner and Ctrl-C/tool-kill behavior. Owner files are ignored in own leftovers. Interrupted runners remove nothing; completed runs retain current own-root semantics under B9-R1. Own-root cleanup is not a descendant-exit proof; the Windows lifetime regression must address unfinished children explicitly. +5. Task 12's list-only exit proof keeps a killed runner's root while its child lives **and after that child exits**. Do not execute the archived example's unconditional run-3 removal on a list-only platform. This follows B9-R5; controller was notified before dependent implementation. +6. B9-R1–R8 need no safety-rule relaxation. R7 has a factual clarification: global setup is unavailable in installed 22.22.3; config/import still require explicit opt-in. Task 13 remains report-only. Native Windows diagnostics and historical-cause limitations stay visible; no claim of exhaustive leak causes or automatic backlog cleanup is justified. diff --git a/docs/archive/README.md b/docs/archive/README.md index d3a40b56..e4c03435 100644 --- a/docs/archive/README.md +++ b/docs/archive/README.md @@ -154,6 +154,8 @@ reconfirmed by this metadata audit. The per-file inventory and limitations are r | [2026-09-27-superpowers-plan-branch-4b-upstream-watch-actions.md](2026-09-27-superpowers-plan-branch-4b-upstream-watch-actions.md) | `docs/superpowers/plans/2026-09-27-branch-4b-upstream-watch-actions.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | | [2026-09-27-superpowers-plan-branch-5-aqe-store-integrity.md](2026-09-27-superpowers-plan-branch-5-aqe-store-integrity.md) | `docs/superpowers/plans/2026-09-27-branch-5-aqe-store-integrity.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | | [2026-09-27-superpowers-plan-branch-6a-evidence-store.md](2026-09-27-superpowers-plan-branch-6a-evidence-store.md) | `docs/superpowers/plans/2026-09-27-branch-6a-evidence-store.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | +| [2026-09-28-plan-runner-hygiene.md](2026-09-28-plan-runner-hygiene.md) | `docs/plans/2026-09-28-runner-hygiene.md` | Completed V5 execution plan | Local implementation and independent review through `e0fcc2eb`; feature CI and integration tracked by the PR. Private backlog list is a manual review aid, not deletion authority. | +| [2026-09-28-research-test-temp-folder-cleanup.md](2026-09-28-research-test-temp-folder-cleanup.md) | `docs/plans/2026-09-28-test-temp-folder-cleanup-design.md` | Runner cleanup research | Source-bound creator census and experiments supporting list-only sibling handling; no complete native descendant proof or cleanup authority. | | [2026-09-28-superpowers-plan-branch-6b-one-refresh-flag.md](2026-09-28-superpowers-plan-branch-6b-one-refresh-flag.md) | `docs/superpowers/plans/2026-09-28-branch-6b-one-refresh-flag.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | | [2026-09-28-superpowers-plan-branch-9-follow-ups.md](2026-09-28-superpowers-plan-branch-9-follow-ups.md) | `docs/superpowers/plans/2026-09-28-branch-9-follow-ups.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | | [2026-09-28-superpowers-plan-upstream-watch-ledger-branch.md](2026-09-28-superpowers-plan-upstream-watch-ledger-branch.md) | `docs/superpowers/plans/2026-09-28-upstream-watch-ledger-branch.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | diff --git a/scripts/real-state-tripwire.mjs b/scripts/real-state-tripwire.mjs index e74198d2..09b23794 100644 --- a/scripts/real-state-tripwire.mjs +++ b/scripts/real-state-tripwire.mjs @@ -34,7 +34,8 @@ export const CONCURRENT_WRITERS = [ { kind: 'state', pattern: /^statusline-debug\.log$/, writer: 'statusline debug log (src/templates/statusline-footer.cjs:2-15)' }, { kind: 'repo', pattern: /^\.swarm(\/|$)/, writer: 'Ruflo hooks and daemon of a live session' }, { kind: 'repo', pattern: /^\.agentic-qe\/(?!llm-config\.json)/, writer: 'AQE hooks of a live session' }, - { kind: 'repo', pattern: /^\.claude-flow\/(?!config\.json$)/, writer: 'Ruflo hooks and statusline caches of a live session' }, + { kind: 'repo', pattern: /^\.claude-flow(?:\/(?!config\.json$)|$)/, writer: 'Ruflo hooks and statusline caches of a live session' }, + { kind: 'repo', pattern: /^\.claude\/(?:proven-config\.json|\.proven-config-version)$/, writer: 'Ruflo proven-config adoption on CLI startup (@claude-flow/cli 3.48.0 dist/src/config/proven-config-refresh.js:24,30,94,120-125; dist/src/index.js:173-175)' }, { kind: 'user-file', pattern: /^\.claude\.json$/, writer: 'Claude Code session state in ~/.claude.json (https://code.claude.com/docs/en/settings)' }, ]; diff --git a/scripts/run-roots.mjs b/scripts/run-roots.mjs new file mode 100644 index 00000000..a077cec5 --- /dev/null +++ b/scripts/run-roots.mjs @@ -0,0 +1,228 @@ +// Builtin-only attribution and conservative run-root handling. Native probes +// deliberately cannot authorize sibling deletion on any supported platform. +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { randomUUID } from 'node:crypto'; + +export const RUN_ROOT_NAME = /^ak-suite-[A-Za-z0-9]{6}$/; +export const OWNER_FILE = '.ak-suite-owner.json'; +export const HOLD_DIR = '.ak-suite-holds'; +export const IGNORED_IN_ROOT = new Set(['node-compile-cache', OWNER_FILE, HOLD_DIR]); +const MAX_OWNER_BYTES = 8192; +const currentUid = () => process.getuid?.() ?? null; + +/** Attribution timestamp only: startedAt is NOT an observed OS process start. + * @param {{pid?:number, now?:number, hostname?:string, uid?:number|null, platform?:string}} [options] + */ +export function ownerRecord({ pid = process.pid, now = Date.now(), hostname = os.hostname(), + uid = currentUid(), platform = process.platform } = {}) { + return { schema: 1, runId: randomUUID(), pid, startedAt: now, hostname, uid, platform, proofMode: 'list-only' }; +} + +/** Private, exclusive staging file followed by atomic publication. */ +export function writeOwner(root, record) { + const canonical = fs.realpathSync(root); + if (canonical !== root || !fs.lstatSync(root).isDirectory()) throw Error('noncanonical run root'); + const bound = { ...record, root, tmpdir: path.dirname(root) }; + if (!validRecord(bound, root)) throw Error('invalid owner record'); + const staging = path.join(root, `${OWNER_FILE}.tmp`); + fs.writeFileSync(staging, JSON.stringify(bound), { flag: 'wx', mode: 0o600 }); + fs.renameSync(staging, path.join(root, OWNER_FILE)); +} + +function validRecord(r, root) { + return r !== null && typeof r === 'object' && !Array.isArray(r) + && r.schema === 1 && typeof r.runId === 'string' && /^[a-f0-9-]{36}$/.test(r.runId) + && Number.isSafeInteger(r.pid) && r.pid > 0 + && Number.isSafeInteger(r.startedAt) && r.startedAt > 0 + && r.hostname === os.hostname() && r.uid === currentUid() && r.platform === process.platform + && r.proofMode === 'list-only' && r.root === root && r.tmpdir === path.dirname(root); +} + +/** Bounded, no-follow metadata read; missing, changed or invalid means unknown. */ +export function readOwner(root) { + let fd; + let record = null; + try { + const file = path.join(root, OWNER_FILE); + const before = fs.lstatSync(file); + if (!before.isFile() || before.isSymbolicLink() || before.nlink !== 1 + || before.size > MAX_OWNER_BYTES || (currentUid() !== null && before.uid !== currentUid())) { + throw Error('unsafe owner file'); + } + // Bitwise flags treat an unavailable platform constant as zero. + fd = fs.openSync(file, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | fs.constants.O_NONBLOCK); + const opened = fs.fstatSync(fd); + if (!opened.isFile() || !sameIdentity(before, opened) || opened.size > MAX_OWNER_BYTES) { + throw Error('owner file changed at open'); + } + const bytes = Buffer.alloc(MAX_OWNER_BYTES + 1); + const count = fs.readSync(fd, bytes, 0, bytes.length, 0); + if (count > MAX_OWNER_BYTES || !sameIdentity(opened, fs.lstatSync(file))) throw Error('owner file changed at read'); + const parsed = JSON.parse(bytes.subarray(0, count).toString('utf8')); + if (validRecord(parsed, root)) record = parsed; + } catch { /* Unknown metadata never grants ownership. */ } + finally { if (fd !== undefined) fs.closeSync(fd); } + return record; +} + +/** Create the private hold directory before launching any suite command. */ +export function prepareRunRootHolds(root, runId) { + if (readOwner(root)?.runId !== runId || fs.realpathSync(root) !== root) throw Error('run root owner mismatch'); + fs.mkdirSync(path.join(root, HOLD_DIR), { mode: 0o700 }); +} + +/** A command acquires this hold before launching children that use its run root. + * Outside a guarded run it returns null; local sandbox retention still applies. + * @param {{env?:NodeJS.ProcessEnv}} [options] + */ +export function acquireRunRootHold({ env = process.env } = {}) { + const root = env.AK_SUITE_ROOT; + const runId = env.AK_SUITE_RUN_ID; + if (root === undefined && runId === undefined) return null; + if (!root || !runId || readOwner(root)?.runId !== runId || fs.realpathSync(root) !== root) { + throw Error('cannot establish run-root hold: owner mismatch'); + } + const dir = path.join(root, HOLD_DIR); + const stat = fs.lstatSync(dir); + if (!stat.isDirectory() || stat.isSymbolicLink() || fs.realpathSync(dir) !== dir) { + throw Error('cannot establish run-root hold: unsafe hold directory'); + } + const id = randomUUID(); + const token = randomUUID(); + const file = path.join(dir, id); + fs.writeFileSync(file, token, { flag: 'wx', mode: 0o600 }); + return { root, runId, file, token, pid: process.pid }; +} + +/** Remove only the marker returned to this process by acquireRunRootHold. */ +export function releaseRunRootHold(hold) { + if (hold === null) return; + if (!hold || hold.pid !== process.pid || readOwner(hold.root)?.runId !== hold.runId + || fs.realpathSync(hold.root) !== hold.root + || path.dirname(hold.file) !== path.join(hold.root, HOLD_DIR)) throw Error('run-root hold owner mismatch'); + const dir = path.join(hold.root, HOLD_DIR); + const parent = fs.lstatSync(dir); + if (!parent.isDirectory() || parent.isSymbolicLink() || fs.realpathSync(dir) !== dir) { + throw Error('run-root hold directory changed'); + } + const stat = fs.lstatSync(hold.file); + if (!stat.isFile() || stat.isSymbolicLink() || stat.nlink !== 1 + || fs.readFileSync(hold.file, 'utf8') !== hold.token) throw Error('run-root hold changed'); + fs.unlinkSync(hold.file); +} + +/** Missing or unreadable hold state is uncertainty, never permission to remove. */ +export function inspectRunRootHolds(root, runId) { + try { + if (readOwner(root)?.runId !== runId) throw Error('owner changed'); + const dir = path.join(root, HOLD_DIR); + const stat = fs.lstatSync(dir); + if (!stat.isDirectory() || stat.isSymbolicLink() || fs.realpathSync(dir) !== dir) throw Error('unsafe hold directory'); + return { unresolved: fs.readdirSync(dir).length > 0, reason: 'unresolved child hold' }; + } catch { return { unresolved: true, reason: 'hold inspection uncertain' }; } +} + +/** Pure path check also accepts Windows paths in cross-platform unit fixtures. */ +export function unsafeTempBase(tmpdir, homedir) { + const windows = path.win32.isAbsolute(tmpdir) && !path.posix.isAbsolute(tmpdir); + const flavor = windows ? path.win32 : path.posix; + const normalize = (p) => windows ? flavor.resolve(p).toLowerCase() : flavor.resolve(p); + const tmp = normalize(tmpdir); + if (tmp === normalize(flavor.parse(tmp).root)) return 'filesystem root'; + if (tmp === normalize(homedir)) return 'home directory'; + return null; +} + +/** @param {string} dir + * @param {{tmpdir:string, homedir:string, uid?:number|null, requireOwner?:boolean}} options + * @returns {{ok:boolean, reason?:string}} + */ +export function removableRunRoot(dir, { tmpdir, homedir, uid = currentUid(), requireOwner = true }) { + try { + if (!path.isAbsolute(dir) || path.resolve(dir) !== dir || !path.isAbsolute(tmpdir) + || fs.realpathSync(tmpdir) !== tmpdir) return { ok: false, reason: 'noncanonical absolute path required' }; + const unsafe = unsafeTempBase(tmpdir, fs.realpathSync(homedir)); + if (unsafe) return { ok: false, reason: `unsafe temp base: ${unsafe}` }; + if (!RUN_ROOT_NAME.test(path.basename(dir)) || path.dirname(dir) !== tmpdir) { + return { ok: false, reason: 'not an exact direct run-root child' }; + } + const stat = fs.lstatSync(dir); + if (!stat.isDirectory() || stat.isSymbolicLink() || fs.realpathSync(dir) !== dir) { + return { ok: false, reason: 'not a canonical nonsymlink directory' }; + } + if (uid !== currentUid() || (uid !== null && stat.uid !== uid)) return { ok: false, reason: 'foreign filesystem owner' }; + if (requireOwner && !readOwner(dir)) return { ok: false, reason: 'no valid owner record (missing, malformed or foreign)' }; + return { ok: true }; + } catch { return { ok: false, reason: 'path inspection failed' }; } +} + +/** Probes are injected code, never derived from metadata or CLI input. + * completeExit must prove ALL users/descendants gone, not a snapshot/handle scan. + * @typedef {{alive:(pid:number)=>boolean|null, startedAfter:(pid:number,ms:number)=>boolean|null, + * completeExit:(root:string,owner:object)=>boolean|null, listOnly?:boolean}} Probes + * @param {string} root + * @param {ReturnType} owner + * @param {Probes} probes + */ +export function proveAbandoned(root, owner, probes) { + try { + if (!validRecord(owner, root)) return { abandoned: false, reason: 'invalid owner metadata' }; + if (probes.listOnly) return { abandoned: false, reason: 'cannot prove complete descendant exit (list-only)' }; + const alive = probes.alive(owner.pid); + if (alive !== false && (alive !== true || probes.startedAfter(owner.pid, owner.startedAt) !== true)) { + return { abandoned: false, reason: 'owner alive or identity uncertain' }; + } + if (probes.completeExit(root, owner) !== true) return { abandoned: false, reason: 'descendant exit uncertain or live user' }; + return { abandoned: true, reason: 'injected complete exit proof' }; + } catch { return { abandoned: false, reason: 'probe failed; exit uncertain' }; } +} + +/** @returns {Probes} No process scans: none could authorize removal. */ +export function defaultProbes(_platform = process.platform) { + return { listOnly: true, alive: () => null, startedAfter: () => null, completeExit: () => null }; +} + +function sameIdentity(a, b) { return a.dev === b.dev && a.ino === b.ino && a.ctimeMs === b.ctimeMs; } + +/** Revalidate after injected probes; recursive rm can still fail partway through. + * The fixture seam is not an installed platform containment implementation. + * @param {{tmpdir:string, selfRoot?:string, homedir:string, uid?:number|null, probes?:Probes, + * log?:(s:string)=>void, remove?:(root:string)=>void}} options + */ +export function collectAbandonedRoots({ tmpdir, selfRoot, homedir, uid = currentUid(), + probes = defaultProbes(), log = console.error, remove = (root) => fs.rmSync(root, { recursive: true }) }) { + /** @type {{removed:string[], kept:Array<{path:string,reason:string}>}} */ + const result = { removed: [], kept: [] }; + const report = (message) => { try { log(message); } catch { /* Reporting cannot alter cleanup outcomes. */ } }; + const keep = (root, reason) => { result.kept.push({ path: root, reason }); report(`kept run root ${root}: ${reason}`); }; + let names; + try { names = fs.readdirSync(tmpdir); } + catch { keep(tmpdir, 'could not list run roots'); return result; } + for (const name of names) { + if (!RUN_ROOT_NAME.test(name)) continue; + const root = path.join(tmpdir, name); + if (root === selfRoot) continue; + const options = { tmpdir, homedir, uid, requireOwner: true }; + const safe = removableRunRoot(root, options); + if (!safe.ok) { keep(root, safe.reason); continue; } + try { + const identity = fs.lstatSync(root); + const owner = readOwner(root); + if (!owner) { keep(root, 'owner changed during inspection'); continue; } + const proof = proveAbandoned(root, owner, probes); + if (!proof.abandoned) { keep(root, proof.reason); continue; } + const boundary = removableRunRoot(root, options); + if (!boundary.ok || !sameIdentity(identity, fs.lstatSync(root)) + || JSON.stringify(owner) !== JSON.stringify(readOwner(root))) { + keep(root, 'root or owner changed before removal'); continue; + } + try { remove(root); } + catch (error) { keep(root, `removal failed; root may be partially removed: ${error.message}`); continue; } + result.removed.push(root); + report(`removed abandoned run root ${root}`); + } catch { keep(root, 'inspection changed or failed; removal not attempted'); } + } + return result; +} diff --git a/scripts/run-tests.mjs b/scripts/run-tests.mjs index d579b881..166a5324 100644 --- a/scripts/run-tests.mjs +++ b/scripts/run-tests.mjs @@ -9,6 +9,8 @@ import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; +import { ownerRecord, writeOwner, prepareRunRootHolds, inspectRunRootHolds, + unsafeTempBase, removableRunRoot, collectAbandonedRoots, IGNORED_IN_ROOT } from './run-roots.mjs'; import { realStateRoots, snapshotRoots, compareSnapshots, isStrict, formatReport } from './real-state-tripwire.mjs'; const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); @@ -47,23 +49,50 @@ export function commandsFor(mode, env = process.env) { * @param {{ env?: NodeJS.ProcessEnv, repoRoot?: string, platform?: string, homedir?: string, log?: (s: string) => void }} [o] * @returns {number} exit code: 2 when the suite temp root sits inside a git repository, * else the first failing command's, else 3 on a real-state change, else 4 on leftover - * temp folders, else 0 + * temp folders or failed own-root inspection/cleanup, else 0 */ export function runGuarded(commands, { env = process.env, repoRoot = REPO, platform = process.platform, homedir = os.homedir(), log = console.error, } = {}) { // Every command runs with this run's own templated temp root: leftovers are then // attributable to the run, and they fail it. - const tempRoot = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'ak-suite-'))); + const tmpdir = fs.realpathSync(os.tmpdir()); + const unsafe = unsafeTempBase(tmpdir, fs.realpathSync(homedir)); + if (unsafe) { log(`unsafe temp base ${tmpdir}: ${unsafe}`); return 2; } + const tempRoot = fs.mkdtempSync(path.join(tmpdir, 'ak-suite-')); + const owner = ownerRecord(); + try { writeOwner(tempRoot, owner); } + catch (error) { log(`could not record run owner; kept run root ${tempRoot}: ${error.message}`); return 2; } + const identity = fs.lstatSync(tempRoot); + const removeOwnRoot = () => { + const safe = removableRunRoot(tempRoot, { tmpdir, homedir, requireOwner: false }); + if (!safe.ok) { log(`kept own run root ${tempRoot}: ${safe.reason}`); return false; } + try { + const current = fs.lstatSync(tempRoot); + if (current.dev !== identity.dev || current.ino !== identity.ino || current.birthtimeMs !== identity.birthtimeMs) { + log(`kept own run root ${tempRoot}: directory identity changed`); return false; + } + fs.rmSync(tempRoot, { recursive: true, force: true, maxRetries: 3 }); + try { log(`removed own run root ${tempRoot}`); } + catch { /* Reporting cannot change a completed removal into a failure. */ } + return true; + } catch (error) { + log(`own run root removal failed; may be partially removed ${tempRoot}: ${error.message}`); + return false; + } + }; const enclosing = enclosingRepository(tempRoot); if (enclosing) { - fs.rmSync(tempRoot, { recursive: true, force: true }); + removeOwnRoot(); log(`the suite temp root ${tempRoot} is inside the git repository ${enclosing}; tests that probe "outside a ` + 'git repository" would write into it. Point TMPDIR outside any repository.'); return 2; } + try { prepareRunRootHolds(tempRoot, owner.runId); } + catch (error) { log(`could not prepare run-root holds; kept ${tempRoot}: ${error.message}`); return 2; } /** @type {NodeJS.ProcessEnv} */ - const childEnv = { ...env, TMPDIR: tempRoot, TEMP: tempRoot, TMP: tempRoot }; + const childEnv = { ...env, TMPDIR: tempRoot, TEMP: tempRoot, TMP: tempRoot, + AK_SUITE_ROOT: tempRoot, AK_SUITE_RUN_ID: owner.runId }; // Tests assert on plain text; a shell's FORCE_COLOR (Claude Code sets 3) // colours console.log into pipes and, beside NO_COLOR, adds a Node warning. delete childEnv.FORCE_COLOR; @@ -74,16 +103,33 @@ export function runGuarded(commands, { let code = 0; for (const args of commands) { const r = spawnSync(process.execPath, args, { cwd: repoRoot, env: childEnv, stdio: 'inherit' }); + if (r.signal) { log(`interrupted run; kept run root ${tempRoot}: ${r.signal}`); return 1; } if (r.error) { log(`could not run node ${args.join(' ')}: ${r.error.message}`); code = 1; break; } if (r.status !== 0) { code = r.status ?? 1; break; } } - const leftovers = fs.readdirSync(tempRoot).filter((name) => name !== 'node-compile-cache'); - fs.rmSync(tempRoot, { recursive: true, force: true, maxRetries: 3 }); + let leftovers = []; + let ownHygieneFailed = false; + try { leftovers = fs.readdirSync(tempRoot).filter((name) => !IGNORED_IN_ROOT.has(name)); } + catch (error) { + log(`could not list own run root; kept ${tempRoot}: ${error.message}`); + ownHygieneFailed = true; + } + // A child may have explicitly declared unresolved ownership before launch. + // Ordinary test failures still remove their own roots when all holds clear. + if (!ownHygieneFailed) { + const holds = inspectRunRootHolds(tempRoot, owner.runId); + if (holds.unresolved) { + log(`kept own run root ${tempRoot}: ${holds.reason}`); + ownHygieneFailed = true; + } else if (!removeOwnRoot()) ownHygieneFailed = true; + } + try { collectAbandonedRoots({ tmpdir, selfRoot: tempRoot, homedir, log }); } + catch (error) { log(`could not list sibling run roots: ${error.message}`); } if (leftovers.length) log(`temp folders left behind by the run (${leftovers.length}):\n ${leftovers.join('\n ')}`); const result = compareSnapshots(before, snapshotRoots(roots), { strict: isStrict(env) }); const report = formatReport(result); if (report) log(report); - return code || (result.failing.length ? 3 : 0) || (leftovers.length ? 4 : 0); + return code || (result.failing.length ? 3 : 0) || (leftovers.length || ownHygieneFailed ? 4 : 0); } /** The nearest folder at or above `dir` that holds a `.git` entry, or null. */ @@ -100,6 +146,15 @@ function enclosingRepository(dir) { function main(argv) { const [mode] = argv; if (mode === 'unit' || mode === 'ui') return runGuarded(commandsFor(mode)); + if (mode === 'focus') { + const files = argv.slice(1); + // Reject Node options disguised as filenames before constructing an argv vector. + if (!files.length || files.some((file) => !file || file.startsWith('-') || !isTestFile(file))) { + console.error('usage: run-tests.mjs focus (each file must exist)'); + return 2; + } + return runGuarded([['--test', ...files]]); + } if (mode === 'exec') { const sep = argv.indexOf('--'); const repoAt = argv.indexOf('--repo'); @@ -107,10 +162,15 @@ function main(argv) { const repoRoot = repoAt >= 0 && repoAt < sep ? path.resolve(argv[repoAt + 1]) : REPO; return runGuarded([argv.slice(sep + 1)], { repoRoot }); } - console.error('usage: run-tests.mjs unit|ui|exec'); + console.error('usage: run-tests.mjs unit|ui|exec|focus'); return 2; } +function isTestFile(file) { + try { return fs.statSync(path.resolve(REPO, file)).isFile(); } + catch { return false; } +} + // Compare real paths (drive-letter case differs on Windows): a missed match would // make `pnpm test` exit 0 having run nothing. const isMain = () => { diff --git a/tests/kit/about-install-edit-render.test.mjs b/tests/kit/about-install-edit-render.test.mjs new file mode 100644 index 00000000..3af9b5c6 --- /dev/null +++ b/tests/kit/about-install-edit-render.test.mjs @@ -0,0 +1,47 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import vm from 'node:vm'; +import { RANK, esc } from '../../src/lib/dashboard/groups.mjs'; +import { RUFLO_PIN_NOTE } from '../../src/lib/install-edits.mjs'; +import { installEditRows } from '../../src/commands/status/sections/natives.mjs'; + +const ABOUT_SOURCE = fs.readFileSync(new URL('../../src/lib/dashboard/client/about.mjs', import.meta.url), 'utf8'); +const context = vm.createContext({ RANK, esc, sourceHostIcon: () => '', aboutHostChip: () => null }); +vm.runInContext(ABOUT_SOURCE.replace(/^import .*;$/gm, '').replace(/\bexport /g, ''), context); + +function renderCard(id, rows) { + context.__entry = { id, name: id, category: 'engine-memory', tagline: '', paragraph: '', icon: { ref: 'R' } }; + context.__rows = rows; + return vm.runInContext('aboutCard(__entry, { rows: __rows })', context); +} + +test('About renders one escaped Ruflo install-edit line from the real natives row', () => { + const rufloRoot = '/test/ruflo'; + const rows = installEditRows([{ + state: 'applied', + file: `${rufloRoot}/node_modules/@claude-flow/cli/package.json`, + section: 'optionalDependencies', + name: 'better-sqlite3', + from: '', + to: '^12.10.0', + }], { rufloRoot }); + assert.equal(rows.length, 1); + assert.match(rows[0].message, /^ak applied Ruflo's native SQLite pin \(ruvnet\/ruflo#2219\)/); + + const rufloCard = renderCard('ruflo', rows); + const editLines = rufloCard.match(/
[^<]*<\/div>/g) || []; + assert.deepEqual(editLines, [`
${esc(rows[0].message)}
`]); + assert.ok(editLines[0].startsWith('
ak applied Ruflo's native SQLite pin (ruvnet/ruflo#2219)')); + assert.doesNotMatch(rufloCard, //); + assert.doesNotMatch(renderCard('agentdb', rows), /
/); + assert.doesNotMatch(renderCard('ruflo', []), /
/); +}); + +test('About edit-line matcher accepts the status row wording contract', () => { + const editLineSource = ABOUT_SOURCE.split('function aboutEditLine(')[1]?.split('function aboutCard(')[0]; + assert.ok(editLineSource, 'the shipped About edit-line function exists'); + const pattern = editLineSource.match(/\/\^([^/]+)\/\.test\(String\(er\.message/); + assert.ok(pattern, 'the shipped About edit-line matcher exists'); + assert.match(RUFLO_PIN_NOTE, new RegExp(`^${pattern[1]}`)); +}); diff --git a/tests/kit/helpers/interruption-scope.mjs b/tests/kit/helpers/interruption-scope.mjs new file mode 100644 index 00000000..512fa0c6 --- /dev/null +++ b/tests/kit/helpers/interruption-scope.mjs @@ -0,0 +1,92 @@ +import fs from 'node:fs'; +import { tempDir } from './temp-dir.mjs'; +import { acquireRunRootHold, releaseRunRootHold } from '../../../scripts/run-roots.mjs'; + +export function childGone(pid) { + try { process.kill(pid, 0); return false; } + catch (error) { return error.code === 'ESRCH'; } +} + +export async function until(check, description, timeout = 5000, signal) { + const deadline = Date.now() + timeout; + while (Date.now() < deadline) { + signal?.throwIfAborted(); + const result = check(); + if (result) return result; + await new Promise((resolve) => setTimeout(resolve, 25)); + } + throw Error(`timed out waiting for ${description}`); +} + +async function stopPid(pid) { + if (childGone(pid)) return; + try { await until(() => childGone(pid), 'private child exit', 2000); return; } catch { /* Escalate exact PID only. */ } + for (const signal of ['SIGTERM', 'SIGKILL']) { + if (childGone(pid)) return; + try { process.kill(pid, signal); } catch (error) { if (error.code !== 'ESRCH') throw error; } + try { await until(() => childGone(pid), 'signaled child exit', 2000); return; } catch { /* Retain on uncertainty. */ } + } + throw Error(`cannot prove owned child ${pid} exited`); +} + +/** One hook owns process shutdown and conditional directory deletion. */ +export function interruptionScope(t, { beforeRemove = () => {} } = {}) { + const hold = acquireRunRootHold(); + const home = tempDir('ak-interrupt', undefined, { manual: true }); + const entries = []; + let cleaning; + let stopping = false; + const active = () => { + t.signal.throwIfAborted(); + if (stopping) throw Error('fixture cleanup has started'); + }; + const cleanup = () => { + stopping = true; + cleaning ??= (async () => { + const results = await Promise.allSettled(entries.map(async (entry) => { + if (entry.stop) fs.writeFileSync(entry.stop, 'exit'); + if (entry.handshake && !(entry.launchError && !entry.child.pid)) { + const data = await until(() => fs.existsSync(entry.handshake) + && JSON.parse(fs.readFileSync(entry.handshake, 'utf8')), 'owned child identity', 2000); + if (!Number.isSafeInteger(data.pid) || data.pid <= 0) throw Error('invalid owned child identity'); + await stopPid(data.pid); + } + if (!entry.closed) { + try { await until(() => entry.closed, 'runner close', 2000); } catch { + entry.child.kill('SIGTERM'); + try { await until(() => entry.closed, 'runner termination', 2000); } catch { + entry.child.kill('SIGKILL'); + await until(() => entry.closed, 'runner forced close', 2000); + } + } + } + if (entry.child.pid && !childGone(entry.child.pid)) throw Error('owned runner PID remains'); + })); + const errors = results.filter(r => r.status === 'rejected').map(r => r.reason); + if (errors.length) throw new AggregateError(errors, 'owned process exit uncertain; retaining fixture and run root'); + releaseRunRootHold(hold); + })(); + return cleaning; + }; + const abort = () => { void cleanup().catch(() => {}); }; + t.after(async () => { + t.signal.removeEventListener('abort', abort); + await cleanup(); + beforeRemove(home); + fs.rmSync(home, { recursive: true, force: true, maxRetries: 3 }); + }); + t.signal.addEventListener('abort', abort, { once: true }); + return { + home, cleanup, active, + wait: (check, description, timeout) => until(check, description, timeout, t.signal), + launch(start, { handshake, stop } = {}) { + active(); + const child = start(); + const entry = { child, handshake, stop, closed: false, launchError: null }; + child.on('error', error => { entry.launchError = error; }); + child.once('close', () => { entry.closed = true; }); + entries.push(entry); + return child; + }, + }; +} diff --git a/tests/kit/helpers/temp-dir.mjs b/tests/kit/helpers/temp-dir.mjs index 817b4a67..70b2df28 100644 --- a/tests/kit/helpers/temp-dir.mjs +++ b/tests/kit/helpers/temp-dir.mjs @@ -8,14 +8,18 @@ import path from 'node:path'; /** * Create `/-XXXXXX` and remove it when the test (or, without - * `t`, the file) finishes. A caller that chdir-ed into it must chdir out first. + * `t`, the file) finishes, unless manual cleanup is requested. A caller that + * chdir-ed into it must chdir out first. * @param {string} prefix * @param {import('node:test').TestContext} [t] + * @param {{manual?:boolean}} [options] Caller owns removal when manual is true. * @returns {string} the real path of the new folder */ -export function tempDir(prefix, t) { +export function tempDir(prefix, t, { manual = false } = {}) { const dir = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), `${prefix}-`))); const remove = () => fs.rmSync(dir, { recursive: true, force: true, maxRetries: 3 }); - if (t) t.after(remove); else after(remove); + if (!manual) { + if (t) t.after(remove); else after(remove); + } return dir; } diff --git a/tests/kit/real-state-tripwire.test.mjs b/tests/kit/real-state-tripwire.test.mjs index ef911ec8..2a90383c 100644 --- a/tests/kit/real-state-tripwire.test.mjs +++ b/tests/kit/real-state-tripwire.test.mjs @@ -144,6 +144,43 @@ test('developer mode moves live-session writers to "concurrent"; strict mode fai assert.match(formatReport(dev), /concurrent writers \(not failing\)/); }); +test('Ruflo-created .claude-flow root and proven-config files are concurrent only in developer mode', (t) => { + const home = tmp(t, 'ak-trip-ruflo'); + const repo = path.join(home, 'repo'); + fs.mkdirSync(path.join(repo, '.claude'), { recursive: true }); + const roots = realStateRoots({ platform: process.platform, homedir: home, repoRoot: repo, env: {} }); + const before = snapshotRoots(roots); + fs.mkdirSync(path.join(repo, '.claude-flow')); + fs.writeFileSync(path.join(repo, '.claude-flow', 'session.json'), '{}'); + fs.writeFileSync(path.join(repo, '.claude', 'proven-config.json'), '{}'); + fs.writeFileSync(path.join(repo, '.claude', '.proven-config-version'), 'v1'); + fs.writeFileSync(path.join(repo, '.claude-flow', 'config.json'), '{}'); + const after = snapshotRoots(roots); + const dev = compareSnapshots(before, after, { strict: false }); + assert.deepEqual(dev.concurrent.map((c) => c.rel).sort(), [ + '.claude-flow/', '.claude-flow/session.json', '.claude/.proven-config-version', '.claude/proven-config.json', + ].sort()); + assert.deepEqual(dev.failing.map((c) => c.rel), ['.claude-flow/config.json']); + const strict = compareSnapshots(before, after, { strict: true }); + assert.deepEqual(strict.concurrent, []); + assert.deepEqual(strict.failing.map((c) => c.rel).sort(), [ + '.claude-flow/', '.claude-flow/config.json', '.claude-flow/session.json', + '.claude/.proven-config-version', '.claude/proven-config.json', + ].sort()); +}); + +test('removing the .claude-flow root is concurrent only in developer mode', (t) => { + const home = tmp(t, 'ak-trip-ruflo-remove'); + const repo = path.join(home, 'repo'); + fs.mkdirSync(path.join(repo, '.claude-flow'), { recursive: true }); + const roots = realStateRoots({ platform: process.platform, homedir: home, repoRoot: repo, env: {} }); + const before = snapshotRoots(roots); + fs.rmSync(path.join(repo, '.claude-flow'), { recursive: true }); + const after = snapshotRoots(roots); + assert.deepEqual(compareSnapshots(before, after, { strict: false }).concurrent.map((c) => c.rel), ['.claude-flow/']); + assert.deepEqual(compareSnapshots(before, after, { strict: true }).failing.map((c) => c.rel), ['.claude-flow/']); +}); + test('CI and AK_TRIPWIRE_STRICT make the comparison strict', () => { assert.equal(isStrict({ CI: 'true' }), true); assert.equal(isStrict({ CI: '1' }), true); diff --git a/tests/kit/run-roots.test.mjs b/tests/kit/run-roots.test.mjs new file mode 100644 index 00000000..ae0b1154 --- /dev/null +++ b/tests/kit/run-roots.test.mjs @@ -0,0 +1,266 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { ownerRecord, writeOwner, readOwner, unsafeTempBase, removableRunRoot, + proveAbandoned, collectAbandonedRoots, defaultProbes, OWNER_FILE, + prepareRunRootHolds, acquireRunRootHold, releaseRunRootHold, inspectRunRootHolds } from '../../scripts/run-roots.mjs'; + +const uid = process.getuid?.() ?? null; +const complete = { alive: () => false, startedAfter: () => false, completeExit: () => true }; +function fixture(t) { + const tmpdir = tempDir('ak-root-fixture', t); + const root = fs.mkdtempSync(path.join(tmpdir, 'ak-suite-')); + const homedir = path.join(tmpdir, 'home'); + fs.mkdirSync(homedir); + writeOwner(root, ownerRecord()); + return { root, tmpdir, homedir, uid }; +} +function collect(f, probes = complete, extra = {}) { + return collectAbandonedRoots({ ...f, probes, log: () => {}, ...extra }); +} + +test('private atomic owner metadata round trips and does not leave staging data', (t) => { + const f = fixture(t); + const record = readOwner(f.root); + assert.equal(record.pid, process.pid); + assert.equal(record.root, f.root); + assert.equal(record.tmpdir, f.tmpdir); + assert.equal(record.proofMode, 'list-only'); + assert.deepEqual(fs.readdirSync(f.root), [OWNER_FILE]); + if (process.platform !== 'win32') assert.equal(fs.statSync(path.join(f.root, OWNER_FILE)).mode & 0o777, 0o600); +}); + +test('parallel prelaunch holds release only their own marker; uncertain inspection retains', (t) => { + const f = fixture(t); + const owner = readOwner(f.root); + prepareRunRootHolds(f.root, owner.runId); + const env = { AK_SUITE_ROOT: f.root, AK_SUITE_RUN_ID: owner.runId }; + const first = acquireRunRootHold({ env }); + const second = acquireRunRootHold({ env }); + assert.equal(inspectRunRootHolds(f.root, owner.runId).unresolved, true); + assert.throws(() => releaseRunRootHold({ ...first, pid: 0 })); + const dir = path.join(f.root, '.ak-suite-holds'); + const saved = path.join(f.root, 'saved-holds'); + fs.renameSync(dir, saved); + try { + fs.symlinkSync(saved, dir, 'junction'); + assert.throws(() => releaseRunRootHold(first)); + } finally { + if (fs.existsSync(dir)) fs.unlinkSync(dir); + fs.renameSync(saved, dir); + } + releaseRunRootHold(first); + assert.equal(inspectRunRootHolds(f.root, owner.runId).unresolved, true); + releaseRunRootHold(second); + assert.equal(inspectRunRootHolds(f.root, owner.runId).unresolved, false); + assert.throws(() => acquireRunRootHold({ env: { ...env, AK_SUITE_RUN_ID: 'foreign' } })); + assert.equal(acquireRunRootHold({ env: {} }), null); + fs.rmSync(path.join(f.root, '.ak-suite-holds'), { recursive: true }); + assert.equal(inspectRunRootHolds(f.root, owner.runId).unresolved, true); + assert.throws(() => acquireRunRootHold({ env })); +}); + +test('native defaults never prove abandonment, including dead owners and reused PIDs', (t) => { + const f = fixture(t); + for (const platform of ['darwin', 'linux', 'win32', 'other']) { + assert.equal(proveAbandoned(f.root, readOwner(f.root), defaultProbes(platform)).abandoned, false); + assert.deepEqual(collect(f, defaultProbes(platform)).removed, []); + } + for (const probes of [ { ...complete, alive: () => true }, { ...complete, alive: () => null }, + { ...complete, completeExit: () => null }, { ...complete, completeExit: () => false }, + { ...complete, alive: () => true, startedAfter: () => null }, + { ...complete, alive: () => { throw Error('uncertain'); } } ]) { + assert.equal(collect(f, probes).kept.length, 1); + assert.ok(fs.existsSync(f.root)); + } +}); + +test('only explicit complete fixture proof permits deletion; reused PID needs independent descendant proof', (t) => { + const f = fixture(t); + fs.writeFileSync(path.join(f.root, 'payload'), 'fixture'); + assert.deepEqual(collect(f, { ...complete, alive: () => true, startedAfter: () => true }).removed, [f.root]); + assert.equal(fs.existsSync(f.root), false); +}); + +test('missing, malformed, oversized, foreign and path-mismatched owners stay listed', (t) => { + const f = fixture(t); + const record = readOwner(f.root); + const file = path.join(f.root, OWNER_FILE); + const values = ['{', 'x'.repeat(8193), 'null', '[]', JSON.stringify({ ...record, schema: 2 }), + ...[{ pid: 0 }, { startedAt: -1 }, { hostname: 'foreign' }, { uid: 123456789 }, + { platform: 'foreign' }, { root: f.tmpdir }, { tmpdir: f.root }, { proofMode: 'delete' }, + { runId: '' }].map((patch) => JSON.stringify({ ...record, ...patch }))]; + fs.unlinkSync(file); + assert.equal(collect(f).kept.length, 1); + for (const value of values) { + fs.writeFileSync(file, value); + assert.equal(readOwner(f.root), null, value.slice(0, 100)); + assert.equal(collect(f).kept.length, 1); + assert.ok(fs.existsSync(f.root)); + } +}); + +test('symlink owner files and directory owners are never read', (t) => { + const f = fixture(t); + const file = path.join(f.root, OWNER_FILE); + fs.renameSync(file, path.join(f.tmpdir, 'record')); + fs.symlinkSync(path.join(f.tmpdir, 'record'), file); + assert.equal(readOwner(f.root), null); + assert.equal(collect(f).kept.length, 1); + fs.unlinkSync(file); fs.mkdirSync(file); + assert.equal(readOwner(f.root), null); +}); + +test('unsafe home and filesystem-root temp parents are refused for POSIX and Windows', () => { + for (const [tmp, home] of [['/', '/home/me'], ['/home/me', '/home/me'], ['C:\\', 'C:\\Users\\me'], + ['C:\\Users\\ME', 'c:\\users\\me'], ['\\\\server\\share\\', 'C:\\Users\\me']]) { + assert.ok(unsafeTempBase(tmp, home)); + } + assert.equal(unsafeTempBase('/tmp', '/home/me'), null); + assert.equal(unsafeTempBase('C:\\Temp', 'C:\\Users\\me'), null); +}); + +test('root guards reject path escapes, aliases, non-direct children, wrong owners and symlinks', (t) => { + const f = fixture(t); + assert.equal(removableRunRoot(f.root, { ...f, requireOwner: true }).ok, true); + for (const dir of ['relative', f.tmpdir, path.join(f.root, 'ak-suite-AAAAAA'), + `${f.tmpdir}/x/../${path.basename(f.root)}`]) { + assert.equal(removableRunRoot(dir, f).ok, false); + } + assert.equal(removableRunRoot(f.root, { ...f, tmpdir: f.homedir }).ok, false); + assert.equal(removableRunRoot(f.root, { ...f, homedir: f.tmpdir }).ok, false); + if (uid !== null) assert.equal(removableRunRoot(f.root, { ...f, uid: uid + 1 }).ok, false); + const alias = path.join(f.tmpdir, 'ak-suite-AAAAAA'); + fs.symlinkSync(f.root, alias, 'junction'); + fs.mkdirSync(path.join(f.tmpdir, 'ak-suite-not-exact')); + assert.equal(removableRunRoot(alias, f).ok, false); + const r = collect(f, complete, { selfRoot: f.root }); + assert.deepEqual(r.removed, []); + assert.equal(r.kept.length, 1); + assert.ok(fs.existsSync(f.root)); +}); + +test('revalidation refuses root replacement during the proof', (t) => { + const f = fixture(t); + const r = collect(f, { ...complete, completeExit: () => { + fs.renameSync(f.root, `${f.root}-saved`); + fs.mkdirSync(f.root); writeOwner(f.root, ownerRecord()); + return true; + } }); + assert.deepEqual(r.removed, []); + assert.match(r.kept[0].reason, /changed/); + assert.ok(fs.existsSync(f.root)); +}); + +test('removal errors disclose possible partial deletion and listing errors are reported', (t) => { + const f = fixture(t); + fs.writeFileSync(path.join(f.root, 'payload'), 'fixture'); + const r = collect(f, complete, { remove: (root) => { + fs.unlinkSync(path.join(root, 'payload')); throw Error('EBUSY'); + } }); + assert.deepEqual(r.removed, []); + assert.match(r.kept[0].reason, /partially removed.*EBUSY/); + assert.equal(fs.existsSync(path.join(f.root, 'payload')), false); + assert.equal(collect({ ...f, tmpdir: path.join(f.tmpdir, 'absent') }).kept.length, 1); +}); + +test('default probe functions explicitly report unknown without supplying authority', () => { + const probes = defaultProbes(); + assert.equal(probes.alive(process.pid), null); + assert.equal(probes.startedAfter(process.pid, Date.now()), null); + assert.equal(probes.completeExit('/unused', {}), null); +}); + +test('owner publication rejects invalid attribution and refuses existing staging links', (t) => { + const f = fixture(t); + assert.throws(() => writeOwner(`${f.root}/.`, ownerRecord()), /noncanonical/); + assert.throws(() => writeOwner(f.root, ownerRecord({ pid: 0 })), /invalid/); + const target = path.join(f.tmpdir, 'target'); + fs.writeFileSync(target, 'untouched'); + fs.symlinkSync(target, path.join(f.root, `${OWNER_FILE}.tmp`)); + assert.throws(() => writeOwner(f.root, ownerRecord()), /EEXIST/); + assert.equal(fs.readFileSync(target, 'utf8'), 'untouched'); +}); + +test('owner reader refuses oversized or replaced files at its descriptor boundary', (t) => { + const f = fixture(t); + const realFstat = fs.fstatSync; + const realRead = fs.readSync; + for (const patch of [{ size: 8193 }, { ino: -1 }, { isFile: () => false }]) { + const mock = t.mock.method(fs, 'fstatSync', (...args) => Object.assign(realFstat(...args), patch)); + assert.equal(readOwner(f.root), null); + mock.mock.restore(); + } + const mock = t.mock.method(fs, 'readSync', (...args) => { realRead(...args); return 8193; }); + assert.equal(readOwner(f.root), null); + mock.mock.restore(); + const file = path.join(f.root, OWNER_FILE); + const hardlink = path.join(f.tmpdir, 'hardlink'); + fs.linkSync(file, hardlink); + assert.equal(readOwner(f.root), null); +}); + +test('inspection races retain roots before any removal attempt', (t) => { + const f = fixture(t); + const original = fs.lstatSync; + let calls = 0; + const mock = t.mock.method(fs, 'lstatSync', (...args) => { + if (args[0] === f.root && ++calls === 2) throw Error('inspection denied'); + return original(...args); + }); + const r = collect(f); + assert.deepEqual(r.removed, []); + assert.match(r.kept[0].reason, /inspection.*failed/); + mock.mock.restore(); + const ownerFile = path.join(f.root, OWNER_FILE); + let reads = 0; + const mock2 = t.mock.method(fs, 'lstatSync', (...args) => { + if (args[0] === ownerFile && ++reads === 3) throw Error('owner vanished'); + return original(...args); + }); + assert.match(collect(f).kept[0].reason, /owner changed/); + mock2.mock.restore(); + assert.ok(fs.existsSync(f.root)); +}); + +test('default collection keeps valid roots without injected probes', (t) => { + const f = fixture(t); + const messages = []; + const r = collectAbandonedRoots({ ...f, log: (s) => messages.push(s) }); + assert.deepEqual(r.removed, []); + assert.match(messages[0], /list-only/); +}); + +test('missing roots and noncanonical parents fail closed', (t) => { + const f = fixture(t); + fs.rmSync(f.root, { recursive: true }); + assert.equal(removableRunRoot(f.root, f).ok, false); + assert.equal(removableRunRoot(f.root, { ...f, tmpdir: 'relative' }).ok, false); +}); + +test('platforms without numeric uid retain attributable roots by default', (t) => { + const original = Object.getOwnPropertyDescriptor(process, 'getuid'); + Object.defineProperty(process, 'getuid', { value: undefined, configurable: true }); + t.after(() => { if (original) Object.defineProperty(process, 'getuid', original); }); + const f = fixture(t); + assert.equal(readOwner(f.root).uid, null); + assert.deepEqual(collectAbandonedRoots({ ...f, uid: null, log: () => {} }).removed, []); +}); + +test('logging failure cannot turn a completed removal into an intact-preservation claim', (t) => { + const f = fixture(t); + assert.doesNotThrow(() => { + const result = collectAbandonedRoots({ ...f, probes: complete, log: () => { throw Error('closed pipe'); } }); + assert.deepEqual(result.removed, [f.root]); + assert.deepEqual(result.kept, []); + }); +}); + +test('direct abandonment proof refuses malformed metadata even with complete injected probes', (t) => { + const f = fixture(t); + for (const record of [null, {}, { ...readOwner(f.root), pid: -1 }]) { + assert.equal(proveAbandoned(f.root, record, complete).abandoned, false); + } +}); diff --git a/tests/kit/run-tests-cancellation.test.mjs b/tests/kit/run-tests-cancellation.test.mjs new file mode 100644 index 00000000..881edc46 --- /dev/null +++ b/tests/kit/run-tests-cancellation.test.mjs @@ -0,0 +1,120 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { spawnSync } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; + +const runner = fileURLToPath(new URL('../../scripts/run-tests.mjs', import.meta.url)); +const sandbox = new URL('./helpers/home-sandbox.mjs', import.meta.url).href; +const helper = new URL('./helpers/interruption-scope.mjs', import.meta.url).href; + +test('real test timeout closes owned runner and orphan before cwd removal and releases the outer hold', t => { + const home = tempDir('ak-cancel-check', t); + const evidence = path.join(home, 'evidence.json'); + const fixture = path.join(home, 'cancel.test.mjs'); + fs.writeFileSync(fixture, ` + import { test, after } from 'node:test'; + import assert from 'node:assert/strict'; + import fs from 'node:fs'; + import { spawn } from 'node:child_process'; + import { spawnEnv } from ${JSON.stringify(sandbox)}; + import { interruptionScope, childGone, until } from ${JSON.stringify(helper)}; + let scope, child, closed = false, removed = false; + test('intentional cancellation', { timeout: 500 }, async t => { + // Node 22 unrefs its test timeout; keep this fixture alive until it fires. + // This owned handle is bounded even if cancellation cleanup fails. + const keepAlive = setTimeout(() => {}, 10000); + t.after(() => clearTimeout(keepAlive)); + scope = interruptionScope(t, { beforeRemove(home) { + assert.equal(closed, true, 'close must precede cwd removal'); + assert.equal(childGone(child.pid), true, 'OS must report ESRCH before removal'); + assert.equal(fs.existsSync(home), true); + const orphan = JSON.parse(fs.readFileSync(home + '/ready', 'utf8')); + assert.equal(childGone(orphan.pid), true, 'orphan must also have native ESRCH before removal'); + fs.writeFileSync(${JSON.stringify(evidence)}, JSON.stringify({ pid: child.pid, orphanPid: orphan.pid, closed, gone: childGone(child.pid), home })); + removed = true; + }}); + fs.writeFileSync(scope.home + '/hold.mjs', \` + import fs from 'node:fs'; + fs.writeFileSync(process.argv[2] + '.tmp', JSON.stringify({ pid: process.pid })); + fs.renameSync(process.argv[2] + '.tmp', process.argv[2]); + setInterval(() => { if (fs.existsSync(process.argv[3])) process.exit(0); }, 25); + setTimeout(() => process.exit(8), 10000); + \`); + const env = spawnEnv(scope.home); + delete env.NODE_TEST_CONTEXT; + child = scope.launch(() => spawn(process.execPath, [${JSON.stringify(runner)}, 'exec', '--repo', scope.home, '--', scope.home + '/hold.mjs', scope.home + '/ready', scope.home + '/stop'], { env, stdio: 'ignore' }), + { handshake: scope.home + '/ready', stop: scope.home + '/stop' }); + child.once('close', () => { closed = true; }); + try { + await until(() => fs.existsSync(scope.home + '/ready'), 'orphan readiness', 5000, t.signal); + child.kill('SIGTERM'); + await until(() => closed, 'runner close', 5000, t.signal); + await new Promise(resolve => t.signal.addEventListener('abort', resolve, { once: true })); + } + finally { + await scope.cleanup(); + let attempted = false; + assert.throws(() => scope.launch(() => { attempted = true; return child; })); + assert.equal(attempted, false, 'cancellation must block the launch callback'); + } + }); + after(() => { + assert.equal(removed, true); + assert.equal(fs.existsSync(scope.home), false); + }); + `); + const env = spawnEnv(home, { CI: 'true' }); + delete env.NODE_TEST_CONTEXT; + const result = spawnSync(process.execPath, [runner, 'exec', '--repo', home, '--', '--test', fixture], { + env, encoding: 'utf8', + }); + assert.equal(result.error, undefined); + assert.equal(result.status, 1, result.stdout + result.stderr); + assert.match(result.stdout + result.stderr, /test timed out after 500ms/); + const proof = JSON.parse(fs.readFileSync(evidence, 'utf8')); + assert.equal(proof.closed, true); + assert.equal(proof.gone, true); + assert.equal(fs.existsSync(proof.home), false); + assert.throws(() => process.kill(proof.pid, 0), { code: 'ESRCH' }); + assert.throws(() => process.kill(proof.orphanPid, 0), { code: 'ESRCH' }); + assert.deepEqual(fs.readdirSync(env.TMPDIR), [], 'guarded timeout has no unresolved hold or fixture leftovers'); +}); + +test('uncertain fixture identity retains its local directory and enclosing guarded root', t => { + const home = tempDir('ak-cancel-retain-check', t); + const evidence = path.join(home, 'evidence.json'); + const fixture = path.join(home, 'uncertain.test.mjs'); + fs.writeFileSync(fixture, ` + import { test } from 'node:test'; + import fs from 'node:fs'; + import { spawn } from 'node:child_process'; + import { interruptionScope } from ${JSON.stringify(helper)}; + test('intentional missing child identity', async t => { + const scope = interruptionScope(t); + const child = scope.launch(() => spawn(process.execPath, ['-e', 'process.exit(0)'], { cwd: scope.home, stdio: 'ignore' }), + { handshake: scope.home + '/missing-handshake', stop: scope.home + '/stop' }); + await new Promise(resolve => child.once('close', resolve)); + fs.writeFileSync(${JSON.stringify(evidence)}, JSON.stringify({ pid: child.pid, home: scope.home, root: process.env.AK_SUITE_ROOT })); + await scope.cleanup(); + }); + `); + const env = spawnEnv(home, { CI: 'true' }); + delete env.NODE_TEST_CONTEXT; + const result = spawnSync(process.execPath, [runner, 'exec', '--repo', home, '--', '--test', fixture], { env, encoding: 'utf8' }); + assert.equal(result.error, undefined); + assert.equal(result.status, 1, result.stdout + result.stderr); + assert.match(result.stdout + result.stderr, /owned process exit uncertain/); + assert.match(result.stderr, /unresolved child hold/); + const proof = JSON.parse(fs.readFileSync(evidence, 'utf8')); + assert.throws(() => process.kill(proof.pid, 0), { code: 'ESRCH' }); + assert.equal(fs.existsSync(proof.home), true); + assert.equal(fs.existsSync(proof.root), true); + assert.equal(fs.readdirSync(path.join(proof.root, '.ak-suite-holds')).length, 1); + // This enclosing test knows its exact fixture ran only process.exit(0), and + // independently established ESRCH above; remove only that disposable root. + fs.rmSync(proof.root, { recursive: true }); +}); diff --git a/tests/kit/run-tests-interruption.test.mjs b/tests/kit/run-tests-interruption.test.mjs new file mode 100644 index 00000000..2f36f5f8 --- /dev/null +++ b/tests/kit/run-tests-interruption.test.mjs @@ -0,0 +1,112 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { spawn, spawnSync } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; +import { interruptionScope, childGone } from './helpers/interruption-scope.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..'); +const RUNNER = path.join(ROOT, 'scripts', 'run-tests.mjs'); +const ownerFile = '.ak-suite-owner.json'; + +function roots(parent) { + return fs.readdirSync(parent).filter((name) => /^ak-suite-[A-Za-z0-9]{6}$/.test(name)).sort(); +} + +function runnerFor(repo, script, handshake, stop, done, env, log) { + const fd = fs.openSync(log, 'w'); + try { + return spawn(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', script, handshake, stop, done], { + env, stdio: ['ignore', fd, fd], + }); + } finally { fs.closeSync(fd); } +} + +test('interrupted and live sibling roots stay listed, including after the orphan exits', { + skip: process.platform === 'win32' && 'POSIX runner termination probe; Windows retains siblings by list-only policy', + timeout: 20000, +}, async (t) => { + const scope = interruptionScope(t); + const { home, wait: until } = scope; + const repo = path.join(home, 'repo'); + fs.mkdirSync(path.join(repo, '.git'), { recursive: true }); + const env = spawnEnv(home, { APPDATA: path.join(home, 'AppData', 'Roaming'), CI: 'true' }); + delete env.NODE_TEST_CONTEXT; + const parent = env.TMPDIR; + const baseline = fs.readdirSync(parent); + const script = path.join(home, 'hold.mjs'); + fs.writeFileSync(script, `import fs from 'node:fs'; + const [handshake, stop, done] = process.argv.slice(2); + fs.writeFileSync(handshake + '.tmp', JSON.stringify({ pid: process.pid, cwd: process.cwd(), tmpdir: process.env.TMPDIR })); + fs.renameSync(handshake + '.tmp', handshake); + const timer = setInterval(() => { + if (!fs.existsSync(stop)) return; + const request = fs.readFileSync(stop, 'utf8'); + if (!fs.existsSync(done)) fs.writeFileSync(done, 'stopped'); + if (request === 'exit') { clearInterval(timer); process.exit(0); } + }, 25); + setTimeout(() => process.exit(8), 12000);`); + const clean = path.join(home, 'clean.test.mjs'); + fs.writeFileSync(clean, "import { test } from 'node:test'; test('clean', () => {});"); + const first = ['first-handshake', 'first-stop', 'first-done'].map((name) => path.join(home, name)); + const live = ['live-handshake', 'live-stop', 'live-done'].map((name) => path.join(home, name)); + let runner1; + let runner4; + let interruptedRoot; + try { + runner1 = scope.launch(() => runnerFor(repo, script, ...first, env, path.join(home, 'run1.log')), + { handshake: first[0], stop: first[1] }); + const firstData = await until(() => fs.existsSync(first[0]) && JSON.parse(fs.readFileSync(first[0], 'utf8')), 'first child handshake'); + const root1 = await until(() => roots(parent).map((name) => path.join(parent, name)).find((root) => { + try { return JSON.parse(fs.readFileSync(path.join(root, ownerFile), 'utf8')).pid === runner1.pid; } + catch { return false; } + }), 'first owner record'); + interruptedRoot = root1; + assert.equal(firstData.cwd, repo); + assert.equal(firstData.tmpdir, root1); + assert.ok(firstData.pid > 0); + assert.equal(runner1.kill('SIGTERM'), true); + await until(() => runner1.exitCode !== null || runner1.signalCode !== null, 'owned runner termination'); + assert.ok(fs.existsSync(root1)); + assert.equal(fs.existsSync(first[2]), false, 'orphan has not stopped'); + + const runFocus = () => { + scope.active(); + return spawnSync(process.execPath, [RUNNER, 'focus', clean], { env, encoding: 'utf8' }); + }; + const second = runFocus(); + assert.equal(second.status, 0, second.stdout + second.stderr); + assert.match(second.stderr, /kept run root.*cannot prove complete descendant exit \(list-only\)/); + assert.deepEqual(roots(parent), [path.basename(root1)]); + + runner4 = scope.launch(() => runnerFor(repo, script, ...live, env, path.join(home, 'run4.log')), + { handshake: live[0], stop: live[1] }); + const liveData = await until(() => fs.existsSync(live[0]) && JSON.parse(fs.readFileSync(live[0], 'utf8')), 'live child handshake'); + const root4 = await until(() => roots(parent).map((name) => path.join(parent, name)).find((root) => { + try { return JSON.parse(fs.readFileSync(path.join(root, ownerFile), 'utf8')).pid === runner4.pid; } + catch { return false; } + }), 'live owner record'); + assert.equal(liveData.tmpdir, root4); + fs.writeFileSync(first[1], 'stop'); + await until(() => fs.existsSync(first[2]), 'orphan private stop acknowledgement'); + assert.equal(childGone(firstData.pid), false, 'acknowledgement precedes child exit'); + fs.writeFileSync(first[1], 'exit'); + await until(() => childGone(firstData.pid), 'orphan PID absent after private exit request'); + assert.equal(childGone(liveData.pid), false, 'concurrent child remains alive'); + const third = runFocus(); + assert.equal(third.status, 0, third.stdout + third.stderr); + assert.match(third.stderr, /kept run root.*cannot prove complete descendant exit \(list-only\)/); + assert.deepEqual(roots(parent), [path.basename(root1), path.basename(root4)].sort()); + assert.ok(fs.existsSync(root1), 'known stopped child does not authorize sibling removal'); + assert.ok(fs.existsSync(root4), 'concurrently live sibling remains'); + } finally { + await scope.cleanup(); + } + if (t.signal.aborted) return; + // Only this test's disposable fixture root is removed after its child stops. + assert.deepEqual(roots(parent), [path.basename(interruptedRoot)]); + fs.rmSync(interruptedRoot, { recursive: true }); + assert.deepEqual(fs.readdirSync(parent), baseline); +}); diff --git a/tests/kit/run-tests-runner.test.mjs b/tests/kit/run-tests-runner.test.mjs index fb74f826..6f7b9a6d 100644 --- a/tests/kit/run-tests-runner.test.mjs +++ b/tests/kit/run-tests-runner.test.mjs @@ -2,27 +2,203 @@ import { test } from 'node:test'; import assert from 'node:assert/strict'; import fs from 'node:fs'; -import os from 'node:os'; import path from 'node:path'; import { spawnSync } from 'node:child_process'; import { fileURLToPath } from 'node:url'; +import { tempDir } from './helpers/temp-dir.mjs'; import { spawnEnv } from './helpers/home-sandbox.mjs'; const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..'); const RUNNER = path.join(ROOT, 'scripts', 'run-tests.mjs'); function sandbox(t) { - const home = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'ak-runner-'))); - t.after(() => fs.rmSync(home, { recursive: true, force: true })); + const home = tempDir('ak-runner', t); const repo = path.join(home, 'repo'); fs.mkdirSync(path.join(repo, '.git'), { recursive: true }); const env = spawnEnv(home, { APPDATA: path.join(home, 'AppData', 'Roaming'), CI: 'true' }); delete env.AK_TRIPWIRE_STRICT; + // Nested CLI fixtures must run as independent test processes. + delete env.NODE_TEST_CONTEXT; return { home, repo, env }; } const stub = (dir, name, body) => { const f = path.join(dir, name); fs.writeFileSync(f, body); return f; }; +async function pidGone(pid, timeoutMs = 5_000) { + const end = Date.now() + timeoutMs; + while (Date.now() < end) { + try { process.kill(pid, 0); } catch (error) { if (error.code === 'ESRCH') return true; } + await new Promise((resolve) => setTimeout(resolve, 20)); + } + return false; +} + +test('focus runs a literal clean test file through the guarded root', (t) => { + const { home, env } = sandbox(t); + const file = stub(home, 'clean.test.mjs', "import { test } from 'node:test'; test('clean', () => {});"); + const before = fs.readdirSync(env.TMPDIR); + const r = spawnSync(process.execPath, [RUNNER, 'focus', file], { env, encoding: 'utf8' }); + assert.equal(r.status, 0, r.stdout + r.stderr); + assert.match(r.stderr, /real-state tripwire: watching/); + assert.match(r.stderr, /removed own run root /); + assert.deepEqual(fs.readdirSync(env.TMPDIR), before); +}); + +test('focus reports a leaked temp folder with hygiene exit code', (t) => { + const { home, env } = sandbox(t); + const file = stub(home, 'leaky.test.mjs', `import { test } from 'node:test'; + import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; + test('leak', () => { fs.mkdtempSync(path.join(os.tmpdir(), 'focus-leak-')); });`); + const r = spawnSync(process.execPath, [RUNNER, 'focus', file], { env, encoding: 'utf8' }); + assert.equal(r.status, 4, r.stdout + r.stderr); + assert.match(r.stderr, /temp folders left behind.*focus-leak-/s); +}); + +async function removeExitedFixture(root) { + for (let attempt = 0; ; attempt++) { + try { fs.rmSync(root, { recursive: true, force: true }); return; } + catch (error) { + // Match the fixture helpers' three-retry policy without depending on + // Node's JS/C++ rm implementation. Exit has already been established. + if (!['EBUSY', 'EPERM', 'ENOTEMPTY', 'EEXIST'].includes(error.code) || attempt === 3) throw error; + await new Promise(resolve => setTimeout(resolve, (attempt + 1) * 100)); + } + } +} + +async function heldRootFixture(t, { check = () => {}, beforeRemove = () => () => {} } = {}) { + const { home, repo, env } = sandbox(t); + const pidFile = path.join(home, 'held-child-pid'); + const child = stub(home, 'held-child.cjs', `const fs = require('node:fs'); + const release = process.argv[2]; + fs.writeFileSync(require('node:path').join(process.cwd(), 'held-child-ready'), 'ready'); + const timer = setInterval(() => { if (fs.existsSync(release)) { clearInterval(timer); process.exit(0); } }, 20); + setTimeout(() => process.exit(2), 10000);`); + const script = stub(home, 'unresolved-hold.mjs', `import fs from 'node:fs'; + import path from 'node:path'; import { spawn } from 'node:child_process'; + import { acquireRunRootHold } from ${JSON.stringify(new URL('../../scripts/run-roots.mjs', import.meta.url).href)}; + acquireRunRootHold(); + const root = process.env.AK_SUITE_ROOT; + fs.writeFileSync(path.join(root, 'held-sentinel'), 'keep'); + const child = spawn(process.execPath, [${JSON.stringify(child)}, path.join(root, 'release-child')], + { cwd: root, stdio: 'ignore', detached: true }); + child.once('error', (error) => { throw error; }); + fs.writeFileSync(${JSON.stringify(pidFile)}, String(child.pid)); + child.unref(); + const ready = path.join(root, 'held-child-ready'); + const deadline = Date.now() + 2000; + while (!fs.existsSync(ready) && Date.now() < deadline) await new Promise((resolve) => setTimeout(resolve, 10)); + if (!fs.existsSync(ready)) throw Error('fixture child did not become ready'); + process.exit(7);`); + let root; + let pid; + let failure; + try { + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', script], { env, encoding: 'utf8' }); + const roots = fs.readdirSync(env.TMPDIR).filter((name) => /^ak-suite-[A-Za-z0-9]{6}$/.test(name)); + if (roots.length === 1) root = path.join(env.TMPDIR, roots[0]); + if (fs.existsSync(pidFile)) pid = Number(fs.readFileSync(pidFile, 'utf8')); + assert.equal(r.status, 7, r.stdout + r.stderr); + assert.match(r.stderr, /kept own run root .*unresolved child hold/); + assert.equal(roots.length, 1, r.stderr); + assert.equal(fs.readFileSync(path.join(root, 'held-sentinel'), 'utf8'), 'keep'); + assert.equal(fs.readFileSync(path.join(root, 'held-child-ready'), 'utf8'), 'ready'); + assert.doesNotThrow(() => process.kill(pid, 0), 'the held child should still own the retained cwd'); + check(); + } catch (error) { failure = error; } + try { + if (root) fs.writeFileSync(path.join(root, 'release-child'), 'release'); + if (Number.isInteger(pid) && pid > 0) { + if (!root) { try { process.kill(pid); } catch { /* already exited */ } } + if (!await pidGone(pid)) { try { process.kill(pid); } catch { /* already exited */ } } + assert.ok(await pidGone(pid), `fixture child ${pid} did not exit`); + } + if (root) { + const restore = beforeRemove(root, pid); + // Exit proof above is required; retries only address post-exit filesystem refusal. + try { await removeExitedFixture(root); } + finally { restore(); } + } + } catch (error) { + failure = failure ? new AggregateError([failure, error], 'fixture assertion and cleanup failed') : error; + } + if (failure) throw failure; +} + +test('an unresolved prelaunch hold retains the guarded root and sentinel after test failure', heldRootFixture); + +function injectBusyRemoval(limit) { + let calls = 0; + return { + calls: () => calls, + beforeRemove(root, pid) { + const original = fs.rmSync; + fs.rmSync = (dir, ...args) => { + if (String(dir) === root) { + assert.throws(() => process.kill(pid, 0), { code: 'ESRCH' }); + if (++calls <= limit) throw Object.assign(new Error('injected post-exit EBUSY'), { code: 'EBUSY' }); + } + return original(dir, ...args); + }; + return () => { fs.rmSync = original; }; + }, + }; +} + +test('post-exit fixture removal retries transient EBUSY', async t => { + const busy = injectBusyRemoval(2); + await heldRootFixture(t, busy); + assert.equal(busy.calls(), 3); +}); + +test('exhausted post-exit retries preserve both assertion and cleanup failures', async t => { + const busy = injectBusyRemoval(Infinity); + const original = new assert.AssertionError({ message: 'original fixture assertion' }); + await assert.rejects(heldRootFixture(t, { ...busy, check: () => { throw original; } }), error => { + assert.ok(error instanceof AggregateError); + assert.equal(error.errors[0], original); + assert.equal(error.errors[1].code, 'EBUSY'); + return true; + }); + assert.equal(busy.calls(), 4, 'one attempt plus three bounded retries'); +}); + +test('unresolved holds preserve command, tripwire, then hygiene exit priority', (t) => { + for (const [mode, expected] of [['command', 7], ['tripwire', 3], ['hygiene', 4]]) { + const { home, repo, env } = sandbox(t); + const script = stub(home, `${mode}-hold.mjs`, `import fs from 'node:fs'; import path from 'node:path'; + import { acquireRunRootHold } from ${JSON.stringify(new URL('../../scripts/run-roots.mjs', import.meta.url).href)}; + acquireRunRootHold(); + if (process.env.HOLD_MODE === 'tripwire') { + const file = path.join(process.env.XDG_CONFIG_HOME, 'agentic-kit', 'kit.json'); + fs.mkdirSync(path.dirname(file), { recursive: true }); fs.writeFileSync(file, '{}'); + } + if (process.env.HOLD_MODE === 'command') process.exit(7);`); + let root; + try { + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', script], { + env: { ...env, HOLD_MODE: mode }, encoding: 'utf8', + }); + const roots = fs.readdirSync(env.TMPDIR).filter((name) => /^ak-suite-[A-Za-z0-9]{6}$/.test(name)); + if (roots.length === 1) root = path.join(env.TMPDIR, roots[0]); + assert.equal(r.status, expected, r.stdout + r.stderr); + assert.match(r.stderr, /kept own run root .*unresolved child hold/); + assert.equal(roots.length, 1); + } finally { if (root) fs.rmSync(root, { recursive: true, force: true }); } + } +}); + +test('focus rejects missing files and option-shaped filenames before creating roots', (t) => { + const { env } = sandbox(t); + const before = fs.readdirSync(env.TMPDIR); + for (const args of [[], ['--test-reporter=dot'], ['does-not-exist.test.mjs']]) { + const r = spawnSync(process.execPath, [RUNNER, 'focus', ...args], { env, encoding: 'utf8' }); + assert.equal(r.status, 2, r.stdout + r.stderr); + assert.match(r.stderr, /usage: run-tests\.mjs focus/); + assert.deepEqual(fs.readdirSync(env.TMPDIR), before); + } +}); + test('a command that writes real state fails the run and the path is named', (t) => { const { home, repo, env } = sandbox(t); const leak = stub(home, 'leak.mjs', `import fs from 'node:fs'; import path from 'node:path'; @@ -130,3 +306,179 @@ test('the suite runs without the shell FORCE_COLOR', (t) => { const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', probe], { env: { ...env, FORCE_COLOR: '3' }, encoding: 'utf8' }); assert.equal(r.status, 0, r.stdout + r.stderr); }); + +test('guarded runner preserves tool selectors while scrubbing FORCE_COLOR', (t) => { + const { home, repo, env } = sandbox(t); + const probe = stub(home, 'selectors.mjs', ` + import assert from 'node:assert/strict'; + assert.equal(process.env.AQE_EMBEDDER_PROVIDER, 'sentinel-provider'); + assert.equal(process.env.AQE_EMBEDDER_MODEL, 'sentinel-model'); + assert.equal(process.env.FORCE_COLOR, undefined); + `); + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', probe], { + env: { ...env, AQE_EMBEDDER_PROVIDER: 'sentinel-provider', AQE_EMBEDDER_MODEL: 'sentinel-model', FORCE_COLOR: '3' }, + encoding: 'utf8', + }); + assert.equal(r.status, 0, r.stdout + r.stderr); +}); + +test('home temp base is refused before creating a run root', (t) => { + const { home, repo, env } = sandbox(t); + const before = fs.readdirSync(home); + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', '-e', ''], { + env: { ...env, TMPDIR: home, TMP: home, TEMP: home }, encoding: 'utf8', + }); + assert.equal(r.status, 2, r.stderr); + assert.match(r.stderr, /unsafe temp base/); + assert.deepEqual(fs.readdirSync(home), before); +}); + +test('completed runners retain sibling roots and ignore their own owner metadata', async (t) => { + const { ownerRecord, writeOwner } = await import('../../scripts/run-roots.mjs'); + const { home, repo, env } = sandbox(t); + const parent = env.TMPDIR; + const sibling = fs.mkdtempSync(path.join(parent, 'ak-suite-')); + writeOwner(sibling, ownerRecord({ pid: process.pid })); + fs.writeFileSync(path.join(sibling, 'sentinel'), 'preserve'); + for (const exit of [0, 7]) { + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', '-e', `process.exit(${exit})`], { env, encoding: 'utf8' }); + assert.equal(r.status, exit, r.stderr); + assert.match(r.stderr, /kept run root.*cannot prove complete descendant exit/); + assert.doesNotMatch(r.stderr, /temp folders left behind/); + assert.equal(fs.readFileSync(path.join(sibling, 'sentinel'), 'utf8'), 'preserve'); + assert.deepEqual(fs.readdirSync(parent), [path.basename(sibling)]); + } + assert.ok(home); +}); + +test('a reaped owner does not authorize sibling deletion after a completed run', async (t) => { + const { ownerRecord, writeOwner } = await import('../../scripts/run-roots.mjs'); + const { repo, env } = sandbox(t); + const startedAt = Date.now(); + const child = spawnSync(process.execPath, ['-e', ''], { env }); + assert.equal(child.status, 0); + const sibling = fs.mkdtempSync(path.join(env.TMPDIR, 'ak-suite-')); + writeOwner(sibling, ownerRecord({ pid: child.pid, now: startedAt })); + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', '-e', ''], { env, encoding: 'utf8' }); + assert.equal(r.status, 0, r.stderr); + assert.match(r.stderr, /cannot prove complete descendant exit/); + assert.ok(fs.existsSync(sibling)); +}); + +test('native self-termination reports the platform-specific child result', (t) => { + const { env } = sandbox(t); + const result = spawnSync(process.execPath, ['-e', "process.kill(process.pid, 'SIGTERM')"], { env }); + assert.equal(result.error, undefined); + // Windows uv_kill uses TerminateProcess(1); only uv_process_kill records + // exit_signal on the parent's owned handle. A child's self-kill cannot do so. + const expected = process.platform === 'win32' + ? { status: 1, signal: null } : { status: null, signal: 'SIGTERM' }; + assert.deepEqual({ status: result.status, signal: result.signal }, expected); + assert.throws(() => process.kill(result.pid, 0), { code: 'ESRCH' }); + t.diagnostic(`native self-termination: ${JSON.stringify(expected)}`); +}); + +test('a native timeout signal retains its run root and performs no sibling collection', (t) => { + const { home, repo, env } = sandbox(t); + const sibling = fs.mkdtempSync(path.join(env.TMPDIR, 'ak-suite-')); + fs.writeFileSync(path.join(sibling, 'sentinel'), 'preserve'); + const evidence = path.join(home, 'signal-result.json'); + const command = stub(home, 'wait-for-signal.mjs', ` + import fs from 'node:fs'; import path from 'node:path'; + fs.writeFileSync(path.join(process.env.AK_SUITE_ROOT, 'sentinel'), 'preserve'); + setTimeout(() => {}, 10000); + `); + // Alter only the native spawn options in this isolated driver. The actual + // OS result is passed through unchanged; no signal/result is fabricated. + const driver = stub(home, 'owned-timeout.mjs', ` + import cp from 'node:child_process'; import fs from 'node:fs'; + import { syncBuiltinESMExports } from 'node:module'; + import { runGuarded } from ${JSON.stringify(new URL('../../scripts/run-tests.mjs', import.meta.url).href)}; + const nativeSpawn = cp.spawnSync; + cp.spawnSync = (file, args, options) => { + const result = nativeSpawn(file, args, { ...options, timeout: 1000, killSignal: 'SIGTERM' }); + fs.writeFileSync(${JSON.stringify(evidence)}, JSON.stringify({ + pid: result.pid, status: result.status, signal: result.signal, error: result.error?.code, + })); + return result; + }; + syncBuiltinESMExports(); + try { process.exitCode = runGuarded([[${JSON.stringify(command)}]], { repoRoot: ${JSON.stringify(repo)} }); } + finally { cp.spawnSync = nativeSpawn; syncBuiltinESMExports(); } + `); + const r = spawnSync(process.execPath, [driver], { env, encoding: 'utf8' }); + assert.equal(r.error, undefined); + assert.equal(r.status, 1, r.stderr); + const native = JSON.parse(fs.readFileSync(evidence, 'utf8')); + assert.deepEqual({ status: native.status, signal: native.signal, error: native.error }, + { status: null, signal: 'SIGTERM', error: 'ETIMEDOUT' }); + assert.throws(() => process.kill(native.pid, 0), { code: 'ESRCH' }); + assert.match(r.stderr, /interrupted run/); + assert.doesNotMatch(r.stderr, /^kept run root/m); + assert.equal(fs.readdirSync(env.TMPDIR).length, 2); + const own = fs.readdirSync(env.TMPDIR).map(name => path.join(env.TMPDIR, name)).find(root => root !== sibling); + assert.equal(fs.readFileSync(path.join(own, 'sentinel'), 'utf8'), 'preserve'); + assert.equal(fs.readFileSync(path.join(sibling, 'sentinel'), 'utf8'), 'preserve'); + t.diagnostic(`native timeout: ${JSON.stringify(native)}`); +}); + +// Inject failures only at this subprocess's disposable temp boundary. +function cleanupProbe(home) { + return stub(home, 'cleanup-errors.mjs', `import fs from 'node:fs'; + import os from 'node:os'; import path from 'node:path'; + import { runGuarded } from ${JSON.stringify(new URL('../../scripts/run-tests.mjs', import.meta.url).href)}; + const [repo, mode, code, tripwire] = process.argv.slice(2); + const parent = fs.realpathSync(os.tmpdir()); + const own = (p) => path.dirname(String(p)) === parent && /^ak-suite-[A-Za-z0-9]{6}$/.test(path.basename(String(p))); + const readdir = fs.readdirSync, remove = fs.rmSync, lstat = fs.lstatSync; + fs.readdirSync = (p, ...args) => { + if ((mode === 'inspection' && own(p)) || (mode === 'sibling' && p === parent)) throw Error('EACCES'); + return readdir(p, ...args); + }; + fs.rmSync = (p, ...args) => { + if (mode === 'removal' && own(p)) throw Error('EBUSY'); + return remove(p, ...args); + }; + let ownStats = 0; + fs.lstatSync = (p, ...args) => { + const stat = lstat(p, ...args); + if (own(p)) { + ownStats++; + if (mode === 'refusal' && ownStats >= 3) stat.isDirectory = () => false; + if (mode === 'identity' && ownStats >= 4) stat.birthtimeMs += 1; + } + return stat; + }; + const child = "const fs=require('fs'),path=require('path'),os=require('os');" + + (mode === 'inspection' ? "fs.writeFileSync(path.join(os.tmpdir(),'leftover'),'retain me');" : '') + + (tripwire === 'yes' ? "fs.mkdirSync(path.join(process.env.XDG_CONFIG_HOME,'agentic-kit'),{recursive:true});fs.writeFileSync(path.join(process.env.XDG_CONFIG_HOME,'agentic-kit','kit.json'),'{}');" : '') + + 'process.exit(' + code + ')'; + process.exitCode = runGuarded([['-e', child]], { repoRoot: repo });`); +} + +for (const mode of ['inspection', 'removal', 'refusal', 'identity']) { + test(`own-root ${mode} failure fails hygiene and retains command/tripwire precedence`, (t) => { + for (const [commandCode, tripwire, expected] of [[0, 'no', 4], [7, 'no', 7], [0, 'yes', 3]]) { + const { home, repo, env } = sandbox(t); + const r = spawnSync(process.execPath, [cleanupProbe(home), repo, mode, String(commandCode), tripwire], { env, encoding: 'utf8' }); + assert.equal(r.status, expected, r.stderr); + const roots = fs.readdirSync(env.TMPDIR).filter((name) => /^ak-suite-[A-Za-z0-9]{6}$/.test(name)); + assert.equal(roots.length, 1, r.stderr); + if (mode === 'inspection') { + assert.match(r.stderr, /could not list own run root/); + assert.equal(fs.readFileSync(path.join(env.TMPDIR, roots[0], 'leftover'), 'utf8'), 'retain me'); + } else if (mode === 'removal') assert.match(r.stderr, /may be partially removed/); + else assert.match(r.stderr, /kept own run root/); + } + }); +} + +test('sibling listing failure stays nonfatal and does not mask command failure', (t) => { + for (const code of [0, 7]) { + const { home, repo, env } = sandbox(t); + const r = spawnSync(process.execPath, [cleanupProbe(home), repo, 'sibling', String(code), 'no'], { env, encoding: 'utf8' }); + assert.equal(r.status, code, r.stderr); + assert.match(r.stderr, /could not list run roots/); + assert.deepEqual(fs.readdirSync(env.TMPDIR), []); + } +}); diff --git a/tests/kit/status-zero-spawn.test.mjs b/tests/kit/status-zero-spawn.test.mjs index a14ea7e2..e7a68d86 100644 --- a/tests/kit/status-zero-spawn.test.mjs +++ b/tests/kit/status-zero-spawn.test.mjs @@ -17,12 +17,13 @@ // `refresh: true` (always probes again, even warm). import { test } from 'node:test'; import assert from 'node:assert/strict'; -import { execFileSync } from 'node:child_process'; +import { execFileSync, spawn } from 'node:child_process'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { fileURLToPath, pathToFileURL } from 'node:url'; import { spawnEnv, sandboxProject } from './helpers/home-sandbox.mjs'; +import { acquireRunRootHold, releaseRunRootHold } from '../../scripts/run-roots.mjs'; const HERE = path.dirname(fileURLToPath(import.meta.url)); const PKG_ROOT = path.resolve(HERE, '..', '..'); @@ -35,9 +36,30 @@ function readLedger(file) { return fs.readFileSync(file, 'utf8').split('\n').filter(Boolean).map((line) => JSON.parse(line)); } +async function waitUntil(check, timeoutMs = 5_000) { + const deadline = Date.now() + timeoutMs; + while (!check() && Date.now() < deadline) await new Promise((resolve) => setTimeout(resolve, 10)); + return check(); +} + +async function waitFor(promise, timeoutMs = 5_000) { + let timer; + try { + return await Promise.race([promise, new Promise((_, reject) => { + timer = setTimeout(() => reject(new Error('owned process did not close')), timeoutMs); + })]); + } finally { clearTimeout(timer); } +} + +function processGone(pid) { + if (!Number.isInteger(pid) || pid <= 0) return false; + try { process.kill(pid, 0); return false; } + catch (error) { return error.code === 'ESRCH'; } +} + /** Runs `fn` inside a disposable sandboxed HOME/project, with `extraEnv` - * merged into the child's environment. Cleans up unconditionally. */ -function inSandbox(prefix, extraEnv, fn) { + * merged into the child's environment. Retains both roots when a child may still use them. */ +async function inSandbox(prefix, extraEnv, fn) { const home = fs.mkdtempSync(path.join(os.tmpdir(), `${prefix}-home-`)); fs.mkdirSync(path.join(home, '.config'), { recursive: true }); const project = sandboxProject(prefix); @@ -48,16 +70,22 @@ function inSandbox(prefix, extraEnv, fn) { PATH: path.join(home, 'no-such-bin'), ...extraEnv, }); - try { - return fn({ home, project, env }); - } finally { - fs.rmSync(home, { recursive: true, force: true }); - fs.rmSync(project, { recursive: true, force: true }); + let result; + let failure; + let retained = false; + try { result = await fn({ home, project, env, retain: () => { retained = true; } }); } + catch (error) { failure = error; } + if (retained) throw failure ?? new Error(`retained uncertain sandbox: ${home}, ${project}`); + for (const root of [home, project]) { + try { fs.rmSync(root, { recursive: true, force: true }); } + catch (error) { failure = failure ? new AggregateError([failure, error], 'sandbox assertion and cleanup failed') : error; } } + if (failure) throw failure; + return result; } -test('spawn-guard is a no-op when AK_SPAWN_LEDGER_FILE is unset', () => { - inSandbox('ak-spawn-guard-noop', {}, ({ project, env }) => { +test('spawn-guard is a no-op when AK_SPAWN_LEDGER_FILE is unset', async () => { + await inSandbox('ak-spawn-guard-noop', {}, ({ project, env }) => { // A vacuous "no ledger file exists" check would pass even if the guard // patched child_process regardless of the env var — assert the function // ITSELF is untouched (native name `spawn`, not our wrapper's `patched`). @@ -68,25 +96,121 @@ test('spawn-guard is a no-op when AK_SPAWN_LEDGER_FILE is unset', () => { }); }); -test('spawn-guard records a spawn made inside the guarded child, by every wrapped form', () => { - inSandbox('ak-spawn-guard-smoke', {}, ({ project, env: baseEnv }) => { - const ledgerFile = path.join(os.tmpdir(), `ak-spawn-guard-smoke-${process.pid}.ndjson`); +async function runGuardedSmoke({ childExitCode = 0, childStallsOnRelease = false, + closeTimeoutMs = 5_000, launchFailure = false } = {}) { + // The guarded runner must know about uncertainty before any process uses + // its temp root. Standalone node --test has no runner hold to acquire. + const hold = acquireRunRootHold(); + await inSandbox('ak-spawn-guard-smoke', {}, async ({ project, env: baseEnv, retain }) => { + const ledgerFile = path.join(project, 'spawn-ledger.ndjson'); + const readyFile = path.join(project, 'fork-ready'); + const releaseFile = path.join(project, 'fork-release'); + const doneFile = path.join(project, 'fork-done'); + const pidFile = path.join(project, 'fork-pid'); + const attemptFile = path.join(project, 'fork-attempt'); + const waitingFile = path.join(project, 'parent-waiting-for-fork'); + const parentAckFile = path.join(project, 'fork-closed-by-parent'); const env = { ...baseEnv, AK_SPAWN_LEDGER_FILE: ledgerFile }; - const forkTarget = path.join(project, 'fork-target.mjs'); - fs.writeFileSync(forkTarget, 'process.exit(0);\n'); + const forkTarget = path.join(project, 'fork-target.cjs'); + fs.writeFileSync(forkTarget, [ + "const fs = require('node:fs');", + `fs.writeFileSync(${JSON.stringify(readyFile)}, 'ready');`, + `const release = ${JSON.stringify(releaseFile)};`, + 'const timer = setInterval(() => {', + ' if (!fs.existsSync(release)) return;', + ` fs.writeFileSync(${JSON.stringify(doneFile)}, 'done');`, + ...(childStallsOnRelease ? [' return;'] : []), + ' clearInterval(timer);', + ` process.exit(${childExitCode});`, + '}, 10);', + ].join('\n')); const script = [ + 'async function main() {', "const { spawnSync, execFileSync: ef, execSync: es, fork } = require('node:child_process');", "spawnSync(process.execPath, ['-e', '0']);", "ef(process.execPath, ['-e', '0']);", 'try { es(\'true\'); } catch {}', // shell builtin: exercised even with PATH broken - `fork(${JSON.stringify(forkTarget)}, [], { stdio: 'ignore' });`, - 'process.exit(0);', // do not wait on the forked grandchild's IPC channel + `require('node:fs').writeFileSync(${JSON.stringify(attemptFile)}, 'attempt');`, + `const child = fork(${JSON.stringify(forkTarget)}, [], { stdio: 'ignore' });`, + `require('node:fs').writeFileSync(${JSON.stringify(pidFile)}, String(child.pid));`, + `require('node:fs').writeFileSync(${JSON.stringify(waitingFile)}, 'waiting');`, + 'await new Promise((resolve, reject) => {', + ' child.once("error", reject);', + ' child.once("close", (code, signal) => code === 0 && !signal', + ' ? resolve() : reject(new Error(`fork failed: code=${code}, signal=${signal}`)));', + '});', + 'if (child.exitCode !== 0 || child.signalCode) throw new Error("fork has not exited cleanly");', + `require('node:fs').writeFileSync(${JSON.stringify(parentAckFile)}, 'fork closed');`, + '}', + 'main().catch((error) => { console.error(error); process.exitCode = 1; });', ].join(' '); - execFileSync(process.execPath, [ + const guarded = spawn(launchFailure ? path.join(project, 'missing-node') : process.execPath, [ `--import=${SPAWN_GUARD_URL}`, '-e', script, - ], { cwd: project, env, encoding: 'utf8' }); + ], { cwd: project, env, stdio: ['ignore', 'ignore', 'pipe'] }); + let stderr = ''; + let launchError; + guarded.once('error', (error) => { launchError = error; }); + guarded.stderr.on('data', (chunk) => { stderr += chunk; }); + let closed = false; + const parentClose = new Promise((resolve) => guarded.once('close', (code, signal) => { + closed = true; + resolve({ code, signal }); + })); + let failure; + try { + await waitUntil(() => (fs.existsSync(waitingFile) && fs.existsSync(readyFile)) || closed || launchError); + if (launchError) throw launchError; + assert.ok(fs.existsSync(waitingFile), `guarded parent did not reach fork wait: ${stderr}`); + assert.ok(fs.existsSync(readyFile), `fork did not become ready: ${stderr}`); + // A premature parent may write its acknowledgment or close on a later + // event-loop turn. Give those events time to surface before release. + await waitUntil(() => fs.existsSync(parentAckFile) || closed, 300); + assert.ok(!fs.existsSync(parentAckFile), 'parent acknowledged fork close before child release'); + assert.ok(!closed, 'guarded parent exited while its owned fork was still running'); + } catch (error) { failure = error; } + const cleanupErrors = []; + let parentExited = closed; + let pid = NaN; + try { fs.writeFileSync(releaseFile, 'release'); } catch (error) { cleanupErrors.push(error); } + try { pid = fs.existsSync(pidFile) ? Number(fs.readFileSync(pidFile, 'utf8')) : NaN; } + catch (error) { cleanupErrors.push(error); } + try { await waitFor(parentClose, closeTimeoutMs); parentExited = true; } + catch { + // Keep the guarded parent alive to receive its own fork's close event. + if (Number.isInteger(pid) && pid > 0) { + try { process.kill(pid); } catch (error) { if (error.code !== 'ESRCH') cleanupErrors.push(error); } + } + try { await waitFor(parentClose, closeTimeoutMs); parentExited = true; } + catch { + if (!closed) guarded.kill(); + try { await waitFor(parentClose, closeTimeoutMs); parentExited = true; } + catch (error) { cleanupErrors.push(error); } + } + } + parentExited ||= closed; + // The acknowledgment is written only after the guarded parent receives + // its fork's close event. Otherwise require independent OS exit evidence. + let forkExited = fs.existsSync(parentAckFile) || (!fs.existsSync(attemptFile) && parentExited) + || await waitUntil(() => processGone(pid)); + if (!forkExited && Number.isInteger(pid) && pid > 0) { + try { process.kill(pid); } catch (error) { if (error.code !== 'ESRCH') cleanupErrors.push(error); } + forkExited = await waitUntil(() => processGone(pid)); + } + if (!forkExited) cleanupErrors.push(new Error(`cannot establish owned fork exit: pid=${pid}`)); + if (!parentExited) cleanupErrors.push(new Error('cannot establish guarded parent exit')); + if (!parentExited || !forkExited) retain(); + else { + try { releaseRunRootHold(hold); } catch (error) { cleanupErrors.push(error); } + } + if (cleanupErrors.length) failure = failure + ? new AggregateError([failure, ...cleanupErrors], 'fork assertion and cleanup failed') + : new AggregateError(cleanupErrors, 'fork cleanup failed'); + if (failure) throw failure; + assert.ok(fs.existsSync(doneFile), 'owned fork completed before sandbox cleanup'); + const { code, signal } = await parentClose; + assert.strictEqual(code, 0, `guarded parent failed (${signal}): ${stderr}`); + assert.ok(fs.existsSync(parentAckFile), 'guarded parent did not acknowledge the fork close event'); const lines = readLedger(ledgerFile); - fs.rmSync(ledgerFile, { force: true }); // Each wrapped form by name, not an exact total: execSync goes through a // platform shell, and only its own record is what this test is about. const got = JSON.stringify(lines); @@ -96,6 +220,22 @@ test('spawn-guard records a spawn made inside the guarded child, by every wrappe assert.ok(lines.some((l) => l.cmd === forkTarget), `fork() records the module path as cmd; got ${got}`); assert.ok(lines.every((l) => typeof l.at === 'string' && !Number.isNaN(Date.parse(l.at))), 'every line has an ISO timestamp'); }); +} + +test('spawn-guard records every wrapped form and waits for its owned fork', async () => { + await runGuardedSmoke(); +}); + +test('nonzero owned fork exit fails after the guarded parent reaps it', async () => { + await assert.rejects(runGuardedSmoke({ childExitCode: 7 }), /guarded parent failed/); +}); + +test('stalled owned fork is signaled before its parent is reaped', async () => { + await assert.rejects(runGuardedSmoke({ childStallsOnRelease: true, closeTimeoutMs: 200 }), /guarded parent failed/); +}); + +test('guarded launch failure is reported and its sandbox is cleaned', async () => { + await assert.rejects(runGuardedSmoke({ launchFailure: true }), { code: 'ENOENT' }); }); // A ledger line from the fixture's own npm registry lookups @@ -130,15 +270,14 @@ function sliceByCallBoundary(lines, labels) { return slices; } -test('plain ak status spawns nothing on a warm cache; --refresh always re-probes', () => { - inSandbox('ak-status-zero-spawn', {}, ({ project, env: baseEnv }) => { - const ledgerFile = path.join(os.tmpdir(), `ak-status-zero-spawn-${process.pid}.ndjson`); +test('plain ak status spawns nothing on a warm cache; --refresh always re-probes', async () => { + await inSandbox('ak-status-zero-spawn', {}, ({ project, env: baseEnv }) => { + const ledgerFile = path.join(project, 'status-ledger.ndjson'); const env = { ...baseEnv, AK_SPAWN_LEDGER_FILE: ledgerFile }; execFileSync(process.execPath, [ `--import=${SPAWN_GUARD_URL}`, FIXTURE, PKG_ROOT, project, ], { cwd: project, env, encoding: 'utf8', timeout: 30_000 }); const lines = readLedger(ledgerFile); - fs.rmSync(ledgerFile, { force: true }); const { first, second, third } = sliceByCallBoundary(lines, ['first', 'second', 'third']); diff --git a/tests/kit/ui-chrome-launch.test.mjs b/tests/kit/ui-chrome-launch.test.mjs index e279710a..e0e43aa7 100644 --- a/tests/kit/ui-chrome-launch.test.mjs +++ b/tests/kit/ui-chrome-launch.test.mjs @@ -61,3 +61,44 @@ test('launchChrome removes its temp folder when Chrome fails to start', async (t await assert.rejects(launchChrome(), /no chrome/); assert.equal(fs.existsSync(dir), false); }); + +test('launchChrome excludes credentials and caller env while retaining launch options', async (t) => { + const { chromium } = await import('playwright'); + const { launchChrome } = await import('../ui/helpers/launch-chrome.mjs'); + const injected = ['AK_CHROME_SECRET_SENTINEL', 'AQE_EMBEDDER_PROVIDER', 'CODEX_HOME', 'CLAUDE_CONFIG_DIR']; + const original = Object.fromEntries(injected.map((key) => [key, process.env[key]])); + for (const key of injected) process.env[key] = `sentinel-${key}`; + t.after(() => { + for (const key of injected) { + if (original[key] === undefined) delete process.env[key]; + else process.env[key] = original[key]; + } + }); + let seen; + t.mock.method(chromium, 'launch', async (options) => { seen = options; return { close: async () => {} }; }); + const browser = await launchChrome({ headless: false, args: ['--disable-gpu'], env: { AK_CHROME_SECRET_SENTINEL: 'override' } }); + try { + assert.equal(seen.headless, false); + assert.deepEqual(seen.args, ['--disable-gpu']); + assert.equal(seen.env.AK_CHROME_SECRET_SENTINEL, undefined); + for (const key of injected) { + assert.equal(seen.env[key], undefined, `${key} should not reach Chrome`); + } + for (const key of ['HOME', 'USERPROFILE', 'XDG_CONFIG_HOME', 'XDG_CACHE_HOME', 'APPDATA', 'LOCALAPPDATA']) { + assert.ok(seen.env[key].startsWith(seen.env.TMPDIR), `${key} should be private`); + } + } finally { await browser.close(); } +}); + +test('Chrome environment selects Windows names without duplicate case variants', async () => { + const { chromeEnv } = await import('../ui/helpers/launch-chrome.mjs'); + const env = chromeEnv({ Path: 'first', PATH: 'second', display: ':8', SystemRoot: 'C:\\Windows', + temp: 'real-temp', TOKEN: 'secret' }, 'private-temp', 'win32'); + assert.equal(env.Path, 'second'); + assert.equal(env.SystemRoot, 'C:\\Windows'); + assert.equal(env.TEMP, 'private-temp'); + assert.equal(env.TMP, 'private-temp'); + assert.equal(Object.keys(env).filter((key) => key.toUpperCase() === 'PATH').length, 1); + assert.equal(Object.keys(env).filter((key) => key.toUpperCase() === 'TEMP').length, 1); + assert.equal(env.TOKEN, undefined); +}); diff --git a/tests/ui/helpers/launch-chrome.mjs b/tests/ui/helpers/launch-chrome.mjs index bb38d9bb..25f485f8 100644 --- a/tests/ui/helpers/launch-chrome.mjs +++ b/tests/ui/helpers/launch-chrome.mjs @@ -12,6 +12,35 @@ import os from 'node:os'; import path from 'node:path'; import { chromium } from 'playwright'; +// Chrome needs the executable search path, display connection and a few Windows +// process basics. Its home and temp state belong to this launch, not the caller. +const CHROME_KEYS = ['PATH', 'DISPLAY', 'WAYLAND_DISPLAY', 'XAUTHORITY', 'XDG_RUNTIME_DIR', + 'DBUS_SESSION_BUS_ADDRESS', 'SYSTEMROOT', 'WINDIR', 'COMSPEC', 'PATHEXT']; +const WINDOWS_NAMES = { PATH: 'Path', SYSTEMROOT: 'SystemRoot', WINDIR: 'windir', COMSPEC: 'ComSpec', PATHEXT: 'PATHEXT' }; + +/** @param {NodeJS.ProcessEnv} source @param {string} dir @param {string} [platform] */ +export function chromeEnv(source, dir, platform = process.platform) { + const windows = platform === 'win32'; + const env = {}; + for (const key of CHROME_KEYS) { + let value = source[key]; + if (windows) { + const matches = Object.keys(source).filter((name) => name.toUpperCase() === key); + const chosen = matches.includes(key) ? key : matches.sort()[0]; + value = chosen === undefined ? undefined : source[chosen]; + } + if (value !== undefined) env[windows ? (WINDOWS_NAMES[key] ?? key) : key] = value; + } + return { + ...env, + HOME: dir, USERPROFILE: dir, + XDG_CONFIG_HOME: path.join(dir, 'config'), XDG_CACHE_HOME: path.join(dir, 'cache'), + XDG_DATA_HOME: path.join(dir, 'data'), APPDATA: path.join(dir, 'appdata'), + LOCALAPPDATA: path.join(dir, 'localappdata'), + TMPDIR: dir, TEMP: dir, TMP: dir, MAC_CHROMIUM_TMPDIR: dir, + }; +} + /** * @param {import('playwright').LaunchOptions} [options] merged over { channel: 'chrome', headless: true } * @returns {Promise} a browser whose close() also removes Chrome's temp folder @@ -19,12 +48,11 @@ import { chromium } from 'playwright'; export async function launchChrome(options = {}) { const dir = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'ak-ui-chrome-'))); const remove = () => fs.rmSync(dir, { recursive: true, force: true, maxRetries: 3 }); - const temp = { TMPDIR: dir, TEMP: dir, TMP: dir, MAC_CHROMIUM_TMPDIR: dir }; let browser; try { browser = await chromium.launch({ channel: 'chrome', headless: true, ...options, - env: { ...process.env, ...temp }, // spawn-env: inherits (the browser needs the display and PATH; only its temp dir moves) + env: chromeEnv(process.env, dir), }); } catch (error) { remove(); throw error; } const close = browser.close.bind(browser); From eb963f503f806f80df7a8173d7c71314adbd7cfa Mon Sep 17 00:00:00 2001 From: Chris Phillipson Date: Tue, 29 Sep 2026 04:42:41 -0700 Subject: [PATCH 04/10] feat(dashboard): deliver explicit refresh and live-session corrections (#276) * docs(plan): map dashboard refresh delivery * feat(dashboard): one server refresh operation behind POST /api/refresh * docs(plan): complete dashboard refresh carry-ins * docs: repair dashboard usage citation * feat(dashboard): one Refresh control with the CLI three strengths; Reload re-reads the view * fix(dashboard): show Unknown for unassessed Claude Code configuration * fix(dashboard): meet icon contrast in the host header * fix(dashboard): honor changed Maintenance URL state * test(dashboard): prove refresh request and write boundaries * docs(dashboard): record refresh client gate status * test(dashboard): restore Maintenance write safety journeys * fix(dashboard): reread active System views on Reload * fix(dashboard): retain write guard until refresh status reconciles * fix(dashboard): reconcile superseded refresh operations * fix(dashboard): start scans with POST requests * docs: align dashboard docs with the one Refresh control * docs: correct refresh labels and machine stage order * docs(adr): reconcile dashboard refresh supersessions * test(dashboard): cover ruflo component cwd forwarding * fix(live): resume displaced native transcript readers * docs(live): state re-entry and structured-source limits * docs(dashboard): reconcile final V3 evidence * fix(activity): show paused scan record time * fix(activity): reject impossible scan dates * docs(archive): record dashboard refresh implementation * fix(dashboard): honor current project-tree refresh selection * test(maintenance): align evidence method with Refresh vocabulary --- .../paused-time-report.md | 35 ++ README.md | 2 +- docs/adr/0012-observability.md | 12 + docs/adr/0025-machine-footprint-metrics.md | 21 +- ...receipt-aware-maintenance-control-plane.md | 10 +- ...bindings-and-explicit-maintenance-scans.md | 22 +- ...ory-led-maintenance-resource-management.md | 15 +- ...st-setup-evidence-and-usage-diagnostics.md | 14 +- ...3-evidence-store-and-refresh-vocabulary.md | 93 +-- docs/adr/README.md | 4 +- .../2026-09-28-plan-dashboard-refresh.md | 28 + docs/archive/README.md | 1 + docs/dashboard.md | 84 +-- docs/ddd/machine-footprint.md | 63 +- docs/ddd/observability.md | 2 + docs/ddd/ubiquitous-language.md | 2 +- docs/maintenance.md | 60 +- docs/models.md | 2 +- docs/observability.md | 16 +- ...-237-238-239-verification-and-decisions.md | 17 + docs/upgrading.md | 13 +- docs/usage-scorecard-metrics.md | 2 +- src/lib/dashboard-server.mjs | 97 ++-- src/lib/dashboard/client.mjs | 3 +- src/lib/dashboard/client/boot.mjs | 2 + src/lib/dashboard/client/host-readiness.mjs | 9 +- .../dashboard/client/maintenance-activity.mjs | 14 +- .../client/maintenance-discovery.mjs | 4 +- .../dashboard/client/maintenance-guidance.mjs | 4 +- .../client/maintenance-inspector.mjs | 4 +- .../client/maintenance-operation.mjs | 44 +- .../client/maintenance-workspace.mjs | 18 +- src/lib/dashboard/client/poll.mjs | 35 +- src/lib/dashboard/client/refresh-control.mjs | 110 ++++ .../client/system-maintenance-actions.mjs | 2 +- .../dashboard/client/system-maintenance.mjs | 2 +- src/lib/dashboard/client/system-projects.mjs | 30 +- src/lib/dashboard/client/system-readout.mjs | 10 +- src/lib/dashboard/host-health-api.mjs | 7 +- src/lib/dashboard/maintenance-api.mjs | 18 +- src/lib/dashboard/page.mjs | 20 +- src/lib/dashboard/refresh-api.mjs | 97 ++++ src/lib/dashboard/styles/base.mjs | 12 +- src/lib/dashboard/styles/maintenance.mjs | 1 - src/lib/dashboard/styles/usage.mjs | 1 + src/lib/live/live-sessions-service.mjs | 19 +- src/lib/maintenance/management/activity.mjs | 16 +- .../management/service-discovery.mjs | 2 +- src/lib/maintenance/service.mjs | 2 +- tests/dashboard.test.cjs | 57 +- tests/fixtures/dashboard-status-child.mjs | 7 +- tests/kit/dashboard-get-is-read-only.test.mjs | 92 +++ .../kit/dashboard-hermetic-defaults.test.mjs | 5 +- tests/kit/dashboard-refresh-api.test.mjs | 188 ++++++ tests/kit/dashboard-status-cost.test.mjs | 45 +- tests/kit/dashboard-status-inprocess.test.mjs | 48 +- tests/kit/host-health-api.test.mjs | 4 +- tests/kit/live-service.test.mjs | 45 ++ tests/kit/maintenance-dashboard-api.test.mjs | 123 ++-- ...intenance-dashboard-client-labels.test.mjs | 14 +- tests/kit/maintenance-dashboard-e2e.test.mjs | 16 +- .../kit/maintenance-dashboard-v2-api.test.mjs | 41 +- .../maintenance-management-activity.test.mjs | 45 ++ tests/kit/maintenance-presentation.test.mjs | 80 +-- tests/kit/refresh-vocabulary-guard.test.mjs | 31 +- tests/kit/refresh.test.mjs | 2 +- tests/kit/system-summary.test.mjs | 13 +- tests/ui/dashboard-ui.mjs | 542 +++++++++++++----- tests/ui/host-readiness.mjs | 35 +- tests/ui/maintenance-host-alignment.mjs | 1 + 70 files changed, 1750 insertions(+), 785 deletions(-) create mode 100644 .superpowers/sdd/2026-09-28-dashboard-refresh/paused-time-report.md create mode 100644 docs/archive/2026-09-28-plan-dashboard-refresh.md create mode 100644 src/lib/dashboard/client/refresh-control.mjs create mode 100644 src/lib/dashboard/refresh-api.mjs create mode 100644 tests/kit/dashboard-get-is-read-only.test.mjs create mode 100644 tests/kit/dashboard-refresh-api.test.mjs diff --git a/.superpowers/sdd/2026-09-28-dashboard-refresh/paused-time-report.md b/.superpowers/sdd/2026-09-28-dashboard-refresh/paused-time-report.md new file mode 100644 index 00000000..8682c06f --- /dev/null +++ b/.superpowers/sdd/2026-09-28-dashboard-refresh/paused-time-report.md @@ -0,0 +1,35 @@ +# V3 Activity paused-time consumer report + +## Source and contract + +- Base: `66982f22` in `feat/dashboard-refresh`. +- Read the V4 B8 producer in `agentic-kit-v4-rest` without changing it. A paused history row has `recordedAt` and `completedAt: null`; completed rows retain `completedAt`. + +## Change + +- Activity projection keeps a bounded, valid ISO `recordedAt` separately from `completedAt`, uses it to choose the latest state per source and environment, and falls back to a valid completion time. Invalid scan timestamps become `null`; unrelated metadata is not projected. +- The v2 Activity API allowlists a bounded ISO pause timestamp and omits invalid stamps and private metadata. +- The Activity history table sorts, groups, and displays by the valid recorded time or completion time. Its labels and empty copy describe scan records without claiming every scan completed. + +## Evidence + +- Red phase: guarded focused suites had 3 expected failures for missing pause time and stale latest-state selection. +- Green phase: `env -u FORCE_COLOR node scripts/run-tests.mjs exec -- --test tests/kit/maintenance-management-activity.test.mjs tests/kit/maintenance-dashboard-v2-api.test.mjs`: 58 passed, 0 failed. +- Guarded actual dashboard browser run, `env -u FORCE_COLOR node scripts/run-tests.mjs exec -- tests/ui/dashboard-ui.mjs`: 514 passed, 0 failed. The new assertion inspects actual rendered table rows for a newer paused record and older completed records. +- `./node_modules/.bin/tsc -p tsconfig.json --noEmit`: passed. +- Focused ESLint: 0 errors, one pre-existing `max-lines` warning in `maintenance-api.mjs` (file has 1021 lines, threshold 1000). +- `git diff --check`: passed. + +## Limits + +- V4 B8 producer remains on its separate branch. This change is the consumer contract only, pending integration and independent review. +- Full unit and UI suites were not repeated after the final timestamp-validation refinement; focused unit tests, lint, and typecheck passed after it. The guarded browser run covered the Activity renderer before that refinement. + +## Scoped review fix: impossible calendar dates + +- Review found that the prior ISO shape plus `Date.parse` accepted `2026-09-31T12:00:00Z`, letting a paused row outrank a real September 30 completion and render on October 1. +- The Activity projection now checks the calendar day against its month and leap year, plus clock component bounds, before accepting a scan timestamp. The v2 Activity API uses the same validator for `recordedAt`. +- New cases reject September 31 and a non-leap February 29 at both projection boundaries, retain a valid completion timestamp when the recorded time is rejected, and retain a valid leap day with an offset and fractional seconds. +- Guarded focused tests: `env -u FORCE_COLOR node scripts/run-tests.mjs exec -- --test tests/kit/maintenance-management-activity.test.mjs tests/kit/maintenance-dashboard-v2-api.test.mjs` passed 61 tests after the validator change. The final fallback assertions were added afterward and rerun before this fix commit. +- `./node_modules/.bin/tsc -p tsconfig.json --noEmit` passed. Focused ESLint had 0 errors and the existing file-length warning in `maintenance-api.mjs`. +- Browser and full suites were not repeated for this scoped validation fix; the prior guarded browser run remains the renderer evidence, with full gates assigned to integration. diff --git a/README.md b/README.md index bc502fb3..cefb605d 100644 --- a/README.md +++ b/README.md @@ -147,7 +147,7 @@ and current platform limits. | **setup** | Installs/updates ruflo + agentic-qe globally (handling npm ≥11.17's `allow-scripts` so natives build; AgentDB ships inside ruflo, so ak installs no separate copy), installs and verifies the exact Ruflo-compatible **agent-browser** native executor without adding its plugin/skills (`--no-agent-browser` disables it), installs the **RuvNet Brain** (an offline knowledge base over the rUv stack, powering the `search_ruvnet` MCP — a ~2 GB one-time download, prompted; skip with `--no-ruvnet-brain`), deploys the token-audit skill, merges the managed guidance blocks into the machine-wide guidance files (`~/.claude/CLAUDE.md`, plus `~/.codex/AGENTS.md` on codex machines), offers one-time MCP registration (user scope, with a tool-family picker), and — inside a repo — initializes the project: sanitized `ruflo init`, absolute memory-path pin, a **verified** store→disk write, statusline footer, and a background daemon with **local-only ($0) workers** (token-spending AI workers stay opt-in behind upstream's machine-wide budget). Project scope triggers on a `.git` entry in the current folder; without one it's skipped with a note. `--project` forces the same project setup in the current directory (e.g. a not-yet-`git init`-ed folder); it does not locate an ancestor repository. Project initialization runs `ruflo init --full --force` and can replace existing agent configuration, so read the [setup scope and project mutation contract](docs/setup.md) before using it on an existing project. `--minimal` skips it, `--yes` accepts all prompts (non-interactive), `--no-aqe` / `--no-agent-browser` / `--no-ruvnet-brain` / `--no-security` disable those subsystems, and `--reconfigure` re-offers MCP registration. `--codex` enables + installs the Codex host during setup (ambidextrous dual-host mode; both hosts become available for routing), and `--primary-host claude\|codex` picks which host leads (codex implies `--codex`). | | **status** | Per-subsystem ✓/⚠/✗ (versions, the kit's own version, **ruvnet-brain**, natives, **memory-pin**, security, learning, aqe/RVF, the managed **agent-browser** package/native/config/browser readiness, MCP, **hosts**, **providers**, **routing**, daemons, guidance blocks, statusline), each drift row naming what `sync` would do about it, or marking a step you must take yourself as `→ manual:` (sync never plans those) — plus a **health-history** line that flags regressions since the last sync. Browser status is filesystem-only: it never runs doctor or launches Chrome. | | **sync** | The one convergence verb: upgrades first when a new release exists, then re-heals everything an upgrade wipes, then re-checks and reports. Included in that heal: it **installs any enabled frontier host** (claude/codex/opencode) that's entirely absent — never touching an external (mise/brew/native) install — and **re-applies provider wiring** (the `ENABLE_*` host env, OpenCode's native configuration, the AQE default/fallback/agent overrides, admitted Agentic-QE 3.13.12+ `externalProviders`, and ruflo API providers) whenever it has drifted. External-provider reconciliation preserves foreign entries, refuses same-id conflicts, and prunes only entries whose exact value still matches an agentic-kit ownership receipt. On a dual-host project, sync also **seeds/heals the Claude/Codex default routing policy**. It appends a health-history snapshot, refreshes RuvNet Brain when enabled, and self-updates the kit last. A planned fix whose status row is still there afterwards is reported `unresolved:` and sync exits 1. A row whose fix you do by hand (`→ manual:`) never changes sync's exit code; a failing or warning one is listed under "needs your action". `--no-upgrade` skips self-update and package upgrades. `--skip ` (repeatable) leaves one subsystem out of this run only, including the step it owns; it is reported "skipped by request" and never counts as a failure. `--json` prints one JSON result on stdout (`plan`, `steps`, `unresolved`, `skipped`, `needsYourAction`, `converged`, `exitCode`) and sends the human lines to stderr. Model refresh/diff/plan findings remain advisory. | -| **dashboard** | Opens the local web dashboard (`127.0.0.1:7431`, localhost-only, never detaches) with five primary areas: **About · Overview · Usage · Observability · System**. Ordinary views remain observation-only. System's **Full scan** remeasures local inventory and then chains one provider check. **System → Maintenance** is the sole action surface, with four destinations: **Inventory** (scope → repository where applicable → type → resource → exact installation), **Guidance** (only outcomes the kit can ground), **Discovery** (where it looks), and **Activity** (receipts, undo, interruption audits). Every write is one exact placement and one action, with a server-derived short-lived plan, explicit confirmation, and a one-use capability. Advisory remains a measurement; the former Catalog tab redirects to Inventory and its cards now sit in System Summary. The page is self-contained, offline-first, protected by a per-session token, and never executes a browser-supplied command. Action targets resolve server-side; Discovery accepts validated source-root configuration. Full navigation and security semantics: [Dashboard guide](docs/dashboard.md); provider and recovery limits: [Maintenance runbook](docs/maintenance.md). **Auto-opens your browser** (`--no-open` for headless/SSH); `--port N` changes the port. Stop with Ctrl-C. (Also available as `ak x dashboard`.) | +| **dashboard** | Opens the local web dashboard (`127.0.0.1:7431`, localhost-only, never detaches) with five primary areas: **About · Overview · Usage · Observability · System**. Ordinary views remain observation-only. Choose **Refresh machine** in the header and press **Refresh** to remeasure the machine, refresh Maintenance evidence, and rebuild the inventory. **System → Maintenance** is the sole action surface, with four destinations: **Inventory** (scope → repository where applicable → type → resource → exact installation), **Guidance** (only outcomes the kit can ground), **Discovery** (where it looks), and **Activity** (receipts, undo, interruption audits). Every write is one exact placement and one action, with a server-derived short-lived plan, explicit confirmation, and a one-use capability. Advisory remains a measurement; the former Catalog tab redirects to Inventory and its cards now sit in System Summary. The page is self-contained, offline-first, protected by a per-session token, and never executes a browser-supplied command. Action targets resolve server-side; Discovery accepts validated source-root configuration. Full navigation and security semantics: [Dashboard guide](docs/dashboard.md); provider and recovery limits: [Maintenance runbook](docs/maintenance.md). **Auto-opens your browser** (`--no-open` for headless/SSH); `--port N` changes the port. Stop with Ctrl-C. (Also available as `ak x dashboard`.) | | **usage** | `score` and `prompts` summarize retained local transcript evidence. `status` reads provider-account analytics from cache; `refresh openrouter` explicitly contacts the OpenRouter management API using `OPENROUTER_MANAGEMENT_KEY`, then writes a credential-free mode-`0600` cache. Cache reads make no OpenRouter request. Account rows have no grounded host/session/project correlation and are never merged into transcript totals. | | **models** | Builds a private, host-scoped model inventory from Claude, Codex, OpenCode, Ollama, bounded local usage evidence, and a dated bundled record of Anthropic's public model/lifecycle facts. `status`, `diff`, `explain`, and `plan` are cache-only and read-only; `refresh --online` is the sole online-catalogue boundary. Public facts never imply account or OpenRouter routability. Swap plans enumerate routes plus Agentic QE/Ruflo consumers and print a copyable canonical action without executing it. The CLI exposes exact local evidence deliberately; the Dashboard exposes source-proven public catalogue identity and uses the owner-visible model read contract; secret-shaped values remain masked. See [Model lifecycle intelligence](docs/models.md). | | **admin** | Opens the **maintainer admin** (`127.0.0.1:7432`, localhost-only, foreground) — the project-telemetry sibling of `dashboard`, with the same dark/light visual theme and persisted theme preference: unique repo visitors and cloners (GitHub traffic API, needs a push-access token via `GITHUB_TOKEN`/`GH_TOKEN`/`gh auth token` — panels degrade honestly without one), contributors and watchers, npm download momentum (last 7d vs prior 7d, sparklines — shown as trend only, never an absolute reach number, since mirrors/CI inflate the raw count), latest CI run status and open Dependabot alerts, a **"since you last looked"** delta strip over a local baseline, open issues/PRs from others (oldest first), and external humans ranked by recency (bots excluded). Access is gated by a **per-session token** carried in the URL fragment and sent header-only; the page makes **zero external fetches** (the server proxies GitHub/npm; your credential never reaches the page or the payload). Where `dashboard` is offline-first, `admin` does deliberate GitHub/npm egress — that contract split is why they're siblings, not tabs. `--port N`, `--no-open`; Ctrl-C stops. (Also available as `ak x admin`.) | diff --git a/docs/adr/0012-observability.md b/docs/adr/0012-observability.md index 90adc04f..c095b68b 100644 --- a/docs/adr/0012-observability.md +++ b/docs/adr/0012-observability.md @@ -670,3 +670,15 @@ turns the dashboard into a fleet service nor makes telemetry collection continuo - **Exact-folder leases (#238 item 2).** A runtime lease no longer requires a Git repository when an exact-folder match joins the process to its transcript; see the 2026-08-03 runtime identity amendment above. + +### 2026-09-29 follow-up: bounded re-entry and structured-source evidence + +An idle stop retains active tailer offsets. A native transcript displaced from the newest-file +window also retains its reader state in memory, up to twice the configured file bound (at least +two dormant readers). Re-entry within that bound resumes without replaying accepted records. +After dormant eviction, re-entry reads from the start; a new dashboard process has no persisted +native offset and bootstraps existing-file metadata before following new appends. This does not +create durable exactly-once delivery. + +The optional Ruflo and agentic-qe structured live-events input is experimental. Parser fixtures +exist, but no real producer has been verified. An explicit source path does not prove activity. diff --git a/docs/adr/0025-machine-footprint-metrics.md b/docs/adr/0025-machine-footprint-metrics.md index fa112028..5dacb494 100644 --- a/docs/adr/0025-machine-footprint-metrics.md +++ b/docs/adr/0025-machine-footprint-metrics.md @@ -1,7 +1,11 @@ # ADR-0025 — Machine footprint: infrastructure metrics for install, runtime, storage, and catalog - **Status:** Implemented -- **Updated:** 2026-09-28 — CLI parity (§5) is now `ak system [--refresh[=live|machine]] +- **Updated:** 2026-09-29 — §5 deep measurement now starts with `POST /api/refresh` at + `machine` strength; GET routes are passive. The old GET-started scan rationale is withdrawn + because a read request must not start measurement work. See + [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md). +- **Earlier update:** 2026-09-28 — CLI parity (§5) is now `ak system [--refresh[=live|machine]] [--project-trees] [--json]` (`src/lib/refresh.mjs`'s shared strengths, replacing the retired `--deep`); §5's `GET /api/system?refresh=deep` rationale is untouched by this branch and belongs to a later remediation program's dashboard work (remediation program, branch 6b; see @@ -322,14 +326,14 @@ silent "Other" slice into a to-do list a release can close. ([ADR-0014](0014-dashboard-auth-and-remediation.md)), and zero egress ([ADR-0007](0007-maintainer-admin-local-telemetry.md)'s offline side of the line) as every other dashboard route. -- `GET /api/system?refresh=deep` — starts or attaches to the single-flight deep scan. The - dashboard server is deliberately GET-only; a refresh is a re-*measurement* of local state, not - a mutation of user data, so it stays within that contract. `&trees=1|0` sets whether that scan - walks project working trees; it is a **measurement** parameter, not a view filter, because - trees that were never walked cannot be un-hidden client-side. +- `POST /api/refresh` with `{"strength":"machine","projectTrees":true}` starts + the staged single-flight refresh; `projectTrees` is an optional boolean measurement choice. + The earlier `GET /api/system?refresh=deep&trees=1|0` trigger and its GET-only rationale + are withdrawn: a GET must never start scan work, even when the work only measures local + state. Trees that were never walked cannot be un-hidden client-side. - `ak system [--refresh[=live|machine]] [--project-trees] [--json]` — CLI parity sharing the same collector, following the usage-scorecard precedent of one collector behind both surfaces - (`--refresh=machine` is the CLI equivalent of the `?refresh=deep` route below). + (`--refresh=machine` selects the same strength as the dashboard POST). - `GET /api/system/summary` (amendment, 2026-09-26; extended 2026-09-28) — the page's read: the same payload and parameters with `catalog`, `storage`, `install`, `projects` and `consumers` each projected to an allow-list of keys and items cut to what the page draws (including @@ -568,7 +572,8 @@ The draft left four points open. All four are decided; this section is the recor large corpus — the surprise cost is worse than a stale figure that says how stale it is. The snapshot's `asOf` is always rendered, and beyond `SNAPSHOT_STALE_AFTER_MS` (7 days) the freshness label turns amber and reads "stale, rescan". Opening the System tab issues a plain - `GET /api/system/summary`; only the Rescan control adds `?refresh=deep`. + `GET /api/system/summary`; only explicit Refresh machine starts a measurement with + `POST /api/refresh`. 4. **Windows ships a current-user census plus a best-effort true `cwd`, degrading honestly, with no dependency added.** The draft's "unsupported on win32" answer would have blanked the whole Runtime view on a supported platform. Instead `src/lib/live/win-process-survey.ps1` — a plain text diff --git a/docs/adr/0044-receipt-aware-maintenance-control-plane.md b/docs/adr/0044-receipt-aware-maintenance-control-plane.md index 63b02baa..9280e4aa 100644 --- a/docs/adr/0044-receipt-aware-maintenance-control-plane.md +++ b/docs/adr/0044-receipt-aware-maintenance-control-plane.md @@ -2,7 +2,9 @@ - **Status:** Implemented - **Date:** 2026-09-03 -- **Updated:** 2026-09-28 — the CLI verb for the explicit scan this ADR's v1 `ak maintain scan` +- **Updated:** 2026-09-29 — the dashboard explicit scan starts with `POST /api/refresh`; + the retired `GET /api/maintenance?refresh=scan` is rejected. See ADR-0063. +- **Earlier update:** 2026-09-28 — the CLI verb for the explicit scan this ADR's v1 `ak maintain scan` contract describes is now `ak maintain --refresh[=machine]` (the exact `?refresh=scan` dashboard route below is unaffected — that half of the vocabulary belongs to a later remediation program's dashboard work; see [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md)) @@ -232,9 +234,9 @@ a body no larger than 64 KiB. The SSE query-token exception does not apply. The read-only interruption audit and separately confirmed single-receipt reconciliation. Plain GET /api/maintenance reads the latest persisted scan report and never polls a -provider. The exact ?refresh=scan query performs and atomically persists a provider scan; -other or duplicate query parameters are rejected. The global browser poll remains passive. A -successful persisted System deep rescan chains exactly one Maintenance scan. See ADR-0045. +provider. An explicit `POST /api/refresh` runs the provider scan as its Maintenance stage and persists +the report; `GET /api/maintenance?refresh=scan` is rejected. The global browser poll remains +passive. A successful machine measurement precedes one Maintenance scan. See ADR-0045. The view groups **Updates ready**, **Safe cleanup**, **Needs review**, **Unsupported or blocked**, and **Recent changes / Undo**. Every row exposes a direct imperative; the selected finding adds its diff --git a/docs/adr/0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md b/docs/adr/0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md index e7959521..5c1cef1b 100644 --- a/docs/adr/0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md +++ b/docs/adr/0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md @@ -2,7 +2,10 @@ - **Status:** Implemented - **Date:** 2026-09-03 -- **Updated:** 2026-09-09 — reconciled against repository source and tests for issue #211 +- **Updated:** 2026-09-29 — the explicit provider scan now starts through + `POST /api/refresh`; `GET /api/maintenance` only reads, and the retired + `?refresh=scan` GET trigger is superseded by ADR-0063. +- **Earlier update:** 2026-09-09 — reconciled against repository source and tests for issue #211 - **Earlier update:** 2026-09-04 — proposed ADR-0048 retains physical artifact and consumer-binding identity, adds exact management placements, and plans configurable/resumable discovery; current explicit provider-scan behavior remains authoritative until implementation @@ -24,9 +27,9 @@ Artifact/consumer identity and the explicit provider-scan contract remain current. The **Browser refresh / Scan now** wording below describes the v1 view; current -Maintenance uses the consolidated measurement toolbar and v2 scan routes under -ADR-0048. Neither passive report/query reads nor filesystem discovery grant -provider mutation authority. FootprintSnapshot is now v7; v6 below records the +Maintenance uses the single Refresh control and explicit POST operation under +ADR-0063, alongside ADR-0048's separate v2 scan routes. Neither passive report/query +reads nor filesystem discovery grant provider mutation authority. FootprintSnapshot is now v7; v6 below records the Catalog v4 migration. The independent configurable Discovery scan does not replace the Footprint deep worker or the explicit provider scan. @@ -93,15 +96,16 @@ a session. Those would require host-native runtime receipts. ### Make scanning explicit and browser refresh passive -Maintenance has two read paths: +Maintenance has one report read and one explicit refresh start: - GET /api/maintenance reads the latest private persisted report. It does not call a host CLI, provider, registry, network source, or version detector. -- GET /api/maintenance?refresh=scan performs one explicit provider scan, persists the - resulting report atomically, and returns it. Unknown or duplicate query parameters fail closed. +- `POST /api/refresh` explicitly runs the Maintenance evidence stage, persists its provider + scan report, and then rebuilds inventory. A GET with the retired `?refresh=scan` + query is rejected. -The dashboard labels these controls **Browser refresh** and **Scan now**. The global poll clock uses -the first path. **Scan now** uses the second. A successful persisted System deep rescan chains one +The dashboard uses **Reload** to re-read a view and **Refresh** to start the staged POST +operation. The global poll clock only reads. A successful persisted System deep rescan chains one Maintenance provider scan so inventory and provider evidence converge without double-scanning concurrent callers. diff --git a/docs/adr/0048-inventory-led-maintenance-resource-management.md b/docs/adr/0048-inventory-led-maintenance-resource-management.md index a3f43266..06e67562 100644 --- a/docs/adr/0048-inventory-led-maintenance-resource-management.md +++ b/docs/adr/0048-inventory-led-maintenance-resource-management.md @@ -1,8 +1,13 @@ # ADR-0048 — Inventory-led Maintenance resource management - **Status:** Accepted — implementation delivered 2026-09-05; Implemented withheld pending - human-evaluation and cross-platform gates -- **Updated:** 2026-09-28 — Branch 6b gives this ADR's two scan controls CLI equivalents: + human usability, screen-reader, and cross-platform evaluation gates deferred to v5; + the former Refresh evidence and Re-measure machine controls are superseded by ADR-0063 +- **Updated:** 2026-09-29 — [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md) replaces the + two dashboard scan controls with one Refresh control; its three visible choices start the + shared staged POST operation. D-15 moves the human usability, screen-reader, and + cross-platform evaluation gates to v5; automated checks do not satisfy them. +- **Earlier update:** 2026-09-28 — Branch 6b gives this ADR's two scan controls CLI equivalents: `ak maintain --refresh` and `ak maintain --refresh=machine` (`src/lib/refresh.mjs`'s `--refresh[=live|machine]`, `ak status --help` for the shared stages). The `scan` verb, the `scans start`/`plan --deep`/`--refresh-inventory` re-measure flags, and `ak maintain recipes @@ -113,7 +118,8 @@ The Focus browser amendment approved on 2026-09-08 is implemented and passes foc approved prototype establishes interaction intent, not production or adapter completeness. It is not yet Implemented in this record's own sense, because the [live acceptance criteria](../maintenance-acceptance.md)'s human-evaluation and -cross-platform acceptance gates have not run on this machine. The dashboard's Maintenance panel now +cross-platform acceptance gates have not run on this machine and are deferred to v5 under +D-15. The dashboard's Maintenance panel now renders this ADR's Inventory/Guidance/Discovery/Activity workspace; ADR-0044's v1 HTTP routes and CLI verbs remain available as a documented compatibility surface until those gates pass and this record is updated again. ADR-0044 is not marked Superseded; see "Implementation status" for why. @@ -121,7 +127,8 @@ record is updated again. ADR-0044 is not marked Superseded; see "Implementation ## Current implementation boundary (2026-09-09) This stays **Accepted with implementation delivered**, not a completed human or -cross-platform release certification. ADR-0050 adds evidence-backed repository +cross-platform release certification. D-15 defers the usability, screen-reader, and +cross-platform evaluation gates to v5. ADR-0050 adds evidence-backed repository groups, independent Desktop-origin filtering, and wrapping language icons. Current cards have no three-icon disclosure or repeated uncertainty labels. Installation counts and exact action identities are unchanged. diff --git a/docs/adr/0053-host-setup-evidence-and-usage-diagnostics.md b/docs/adr/0053-host-setup-evidence-and-usage-diagnostics.md index 3890587c..685042ea 100644 --- a/docs/adr/0053-host-setup-evidence-and-usage-diagnostics.md +++ b/docs/adr/0053-host-setup-evidence-and-usage-diagnostics.md @@ -12,7 +12,10 @@ 15-minute-capped, consent-gated, untouched by this branch (its `host-health-evidence.mjs` input-fingerprint helper, used only to invalidate that in-memory cache, is also untouched). See [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md) (remediation program, branch 6a tasks 5 and 7) -- **Updated:** 2026-09-28 — `ak host check-connection ` is the CLI twin of +- **Updated:** 2026-09-29 — the dashboard **Check again** local re-check is folded into + header Refresh; `POST /api/host-health/local` was removed. The separate consent-gated + connection check remains. See [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md). +- **Earlier update:** 2026-09-28 — `ak host check-connection ` is the CLI twin of this connection check: it reuses `createHostReadinessReader`, so it refuses for exactly the same hosts and reasons the dashboard dialog would (managed-only, `canCheckConnection`), and applies the same consent rule (`--yes` or an interactive y/N; a non-TTY without `--yes` is refused; @@ -112,8 +115,9 @@ the requesting client cancels the owned connected subprocess. ### HTTP and presentation boundaries -`GET /api/host-health` reads local/cached evidence. The separate POST allowlist is -`/api/host-health/local` and `/api/host-health/connection`. Both require the session +`GET /api/host-health` reads local/cached evidence. The local re-check now runs within +`POST /api/refresh`, while the separate consent-gated connection check uses +`POST /api/host-health/connection`. Both require the session token header and exact same-origin fetch metadata; query tokens cannot authorize POST. Requests are size-bounded and accept fixed fields, never arbitrary commands, paths, prompts, environment or client-selected models. Connection checks additionally @@ -168,8 +172,8 @@ from one module, `src/lib/host-management.mjs`. - **The hint is the complete host list.** `ak host pick --host` replaces the enabled set, so the hint names every currently enabled host (including admitted external hosts) plus the one to add, for example `ak host pick --host claude,codex`. -- **The paid connection check stays managed-only.** The local re-check button reads - **Check again** and runs for every host. +- **The paid connection check stays managed-only.** The local re-check is part of header + **Refresh** and runs for every host. **Consequences.** Automatic local checks now spawn at most the same bounded, read-only commands for up to three hosts per minute while the dashboard is open. The previous diff --git a/docs/adr/0063-evidence-store-and-refresh-vocabulary.md b/docs/adr/0063-evidence-store-and-refresh-vocabulary.md index c3a15207..f7001912 100644 --- a/docs/adr/0063-evidence-store-and-refresh-vocabulary.md +++ b/docs/adr/0063-evidence-store-and-refresh-vocabulary.md @@ -1,7 +1,8 @@ # ADR-0063 — One evidence store and the refresh vocabulary - **Status:** Accepted -- **Updated:** 2026-09-28 — Branch 6b delivered the CLI refresh vocabulary +- **Updated:** 2026-09-29 — Branch 6c delivered the dashboard refresh operation and retired GET-started scans +- **Earlier update:** 2026-09-28 — Branch 6b delivered the CLI refresh vocabulary - **Date:** 2026-09-28 - **Deciders:** agentic-kit maintainers - **Related:** [ADR-0025](0025-machine-footprint-metrics.md) (`/api/system/summary` projection), @@ -15,10 +16,11 @@ already recorded its own `Updated:` line), [ADR-0055](0055-aqe-embedding-lifecycle.md) (live-check evidence storage relocation, mechanical), [ADR-0053](0053-host-setup-evidence-and-usage-diagnostics.md) (host setup checks are now persisted evidence with an age rule) -- **Supersedes:** [ADR-0048](0048-inventory-led-maintenance-resource-management.md)'s use of the - word "evidence" for its own, separate **Refresh evidence** / **Re-measure machine** scan - controls — terminology only; see "Relationship to ADR-0048" below. Their UI, backing code - (`scan-store.mjs`), and evidence semantics are unchanged by this branch. +- **Supersedes:** [ADR-0048](0048-inventory-led-maintenance-resource-management.md)'s separate + **Refresh evidence** / **Re-measure machine** dashboard controls; + [ADR-0025](0025-machine-footprint-metrics.md) §5's GET-started deep refresh rationale; and + [ADR-0045](0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md)'s + `GET /api/maintenance?refresh=scan` trigger. Their underlying measurements and stores remain. ## Context @@ -404,7 +406,7 @@ the [issues 237–239 audit](../plans/2026-09-26-issues-237-238-239-verification Item 4 named. It closed CLI-only: the dashboard's own controls are untouched, and the dashboard half of this work moved to the next remediation program — see [the branch 6b plan](../archive/2026-09-28-superpowers-plan-branch-6b-one-refresh-flag.md)'s "Closing -this branch" section, and "Ahead: the dashboard half" below. +this branch" section and "Delivered in 6c" below. - **One flag, three strengths, one ordered stage table** — `--refresh[=live|machine]` across `ak status`, `ak system` and `ak maintain [report]`; see "The `--refresh` flag's three strengths" @@ -485,42 +487,55 @@ this branch" section, and "Ahead: the dashboard half" below. and stopping before any write; `ak host adapters` refuses `--dry-run` outright (exit 2) because its verbs have no preview. -## Ahead: the dashboard half - -6b closed CLI-only — see -[the branch 6b plan](../archive/2026-09-28-superpowers-plan-branch-6b-one-refresh-flag.md)'s "Closing -this branch" section. The dashboard's own controls are unchanged by this branch and remain future -work for the next remediation program: - -- One dashboard **Refresh** control offering the same three strengths the CLI now has, and a - **Reload** control that only re-reads the current view. -- A `POST /api/refresh` route driving that control through the same `runRefresh`/stage machinery - this branch built for the CLI, and the dashboard's existing read-only `GET` routes. -- The eventual supersession of ADR-0048's **Refresh evidence** / **Re-measure machine** controls - and of [ADR-0025](0025-machine-footprint-metrics.md) §5's `GET ?refresh=deep` rationale, once - that dashboard work lands — neither is superseded by this branch, and both remain exactly as - their own ADRs describe them today. - -What stays out of scope regardless: this store's storage does not unify with Maintenance's own -`scan-store.mjs`/deep-snapshot system (R1 — one flag and, eventually, one dashboard control drive -the existing chain; the footprint snapshot and its storage stay where they are), and -`src/lib/host-health-evidence.mjs` (ADR-0053's setup-proof input-fingerprint helper) is still not -folded into this store — see "What is deliberately not folded into this store" above, unchanged -by 6b. +## Delivered in 6c + +Branch 6c adds one dashboard **Refresh** control. Its visible choices are **Refresh**, +**Refresh live**, and **Refresh machine**; the operation request calls their conceptual strengths +`local`, `live`, and `machine`. The separate header **Reload** re-reads the active view and +starts no checks. The former **Check again** local host-health button is folded into Refresh. + +`POST /api/refresh` starts one explicit operation with a bounded JSON body: `strength` is +required; `projectTrees` is an optional boolean for `machine` only. The response is 202 with +`started: true` and the operation state, or 409 with the current state when work is already +running. The server requires the per-session `x-dash-token` header and same-origin mutation +metadata; a query token cannot authorize this POST. `GET /api/refresh` reads the latest state +without starting work. That state has an `operationId`, strength, timestamps, running/completion +fields, and sanitized stage progress, but no stage results. `GET /api/system`, +`GET /api/system/summary`, `GET /api/maintenance`, and the host-health read remain reads: +legacy refresh/scan query arguments are rejected, and `/api/host-health/local` was removed. +The separate consent-gated `POST /api/host-health/connection` remains. + +The dashboard runs the shared `runRefresh` stage order from `src/lib/refresh.mjs`: + +| Strength | Ordered stages | +|---|---| +| `local` (Refresh) | Maintenance evidence → inventory → local evidence and versions | +| `live` (Refresh live) | Maintenance evidence → inventory → live checks → local evidence and versions | +| `machine` (Refresh machine) | Machine measurement → Maintenance evidence → inventory → local evidence and versions | + +A failed machine measurement skips its dependent Maintenance and inventory stages; the local +stage still runs. Other stage failures do not stop later stages. Machine measurement can include +project trees when explicitly selected. The Maintenance stage scans provider evidence; inventory +rebuilds after it; local collects status and forces the host-readiness check. These stages use +the existing Maintenance scan and footprint stores, not a merged evidence store. + +`createRefreshOperation` holds one current operation in one dashboard server process. +Two tabs connected to that server share its single-flight state, so a concurrent POST gets 409. +This is volatile process state, with neither durable operation history nor distributed mutual +exclusion across servers. The client retains its accepted `operationId` and reports a result +only for that identity. If its POST response is lost or another operation supersedes the +server's latest state, the page may say its outcome is unavailable; it does not claim the +other operation's result. ## Relationship to ADR-0048 -ADR-0048's **Refresh evidence** and **Re-measure machine** dashboard controls predate this branch -and use the word "evidence" in the Maintenance-inventory sense (provider probes feeding the -Inventory/Guidance/Discovery/Activity workspace), which is conceptually adjacent to — but a -genuinely separate system from — the evidence store this ADR describes. This ADR's evidence store -is the eventual "live" tier's technical precursor and this branch's own interim `--refresh` boolean -is its no-suffix groundwork; ADR-0048's own controls, their backing code (`scan-store.mjs`), and -their UI are unchanged by this branch. No file under Maintenance's own scan system was touched by -Tasks 1–11. A reader should not infer that ADR-0048's controls now share code, storage, or an age -rule with this ADR's evidence store — they do not, yet. Branch 6b gives those two controls CLI -equivalents — `ak maintain --refresh` and `ak maintain --refresh=machine` — without changing the -controls, their backing code, or their UI themselves; see "Delivered in 6b" above. +ADR-0048's separate **Refresh evidence** and **Re-measure machine** dashboard controls are +superseded by the single Refresh control above. Its Inventory/Guidance/Discovery/Activity +workspace, scan storage (`scan-store.mjs`), evidence semantics, and guarded management +actions remain. The CLI equivalents delivered in 6b are `ak maintain --refresh` and +`ak maintain --refresh=machine`. The control unifies the user's start path, not the +underlying storage or age rules. `src/lib/host-health-evidence.mjs` also remains outside the +shared evidence envelope. ## Known limitations (recorded, not fixed, by this branch) diff --git a/docs/adr/README.md b/docs/adr/README.md index 24982403..7b12aab7 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -58,7 +58,7 @@ Consequences**, and cites the grounded source it rests on where relevant. | [0045](0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md) | Physical artifacts, host consumers, and explicit Maintenance scans | Implemented | | [0046](0046-scan-local-observation-reuse-and-nonblocking-deep-scans.md) | Scan-local observation reuse and nonblocking deep scans | Implemented | | [0047](0047-streaming-observation-forest.md) | Streaming observation forest for deep scans | Accepted; Projects pilot and separate Discovery continuation implemented | -| [0048](0048-inventory-led-maintenance-resource-management.md) | Inventory-led Maintenance resource management | Accepted; Focus browser implemented and focused checks pass; human/cross-platform gates pending | +| [0048](0048-inventory-led-maintenance-resource-management.md) | Inventory-led Maintenance resource management | Accepted; two dashboard controls superseded by 0063; human usability, screen-reader and cross-platform gates deferred to v5 | | [0050](0050-dashboard-project-identity-and-context-reporting.md) | Dashboard project identity and context reporting | Implemented | | [0051](0051-supported-peer-delegation-and-host-realignment.md) | Supported peer delegation and scoped host realignment | Accepted; implemented locally | | [0052](0052-codex-usage-attribution.md) | Codex usage attribution: own usage, imports, segments, streaming | Accepted | @@ -69,7 +69,7 @@ Consequences**, and cites the grounded source it rests on where relevant. | [0060](0060-session-surface-initiator-and-product-names.md) | Session surface, initiator and official product names | Proposed; §3 implemented for project discovery (2026-09-27), the rest staged follow-on | | [0061](0061-brain-reclaim-stuck-remediation.md) | RuvNet Brain "unresolved rollback state" remediation | Accepted | | [0062](0062-aqe-project-store-integrity.md) | AQE project store integrity | Accepted | -| [0063](0063-evidence-store-and-refresh-vocabulary.md) | One evidence store and the refresh vocabulary | Accepted | +| [0063](0063-evidence-store-and-refresh-vocabulary.md) | One evidence store and the refresh vocabulary | Accepted; CLI and dashboard refresh operation delivered | Theme: ADRs **0001–0006** define **dual-host LLM routing and leadership** — how `ak` lets ruflo route each development activity (architecture, implementation, testing, review, …) to the right host (Claude diff --git a/docs/archive/2026-09-28-plan-dashboard-refresh.md b/docs/archive/2026-09-28-plan-dashboard-refresh.md new file mode 100644 index 00000000..de0a8107 --- /dev/null +++ b/docs/archive/2026-09-28-plan-dashboard-refresh.md @@ -0,0 +1,28 @@ +# Dashboard Refresh delivery plan + +## Status + +**Implemented; final integration gates pending** — Captured 2026-09-29 after `d2c1833b`. Tasks 6c-1 through 6c-5, the live-view follow-up, native plain-folder proof, and the paused Activity timestamp handoff passed scoped independent reviews. The branch incorporates green `develop@af825c9f`. Full branch gates, whole-branch review and feature PR CI remain at this archival capture; this status does not claim a merge or release. + +Implement the deferred 6c work in order, with one reviewed task and unit commit at a time. The [remediation program](../plans/2026-09-28-remediation-program-v2.md) and [6b handoff](2026-09-28-superpowers-plan-branch-6b-one-refresh-flag.md) define the contracts. A passing exact-head develop CI gate precedes production edits. Tests use sandbox state and injected services; final branch gates and integration belong to the controller. + +| Task | Dependency | Code and proof | +| --- | --- | --- | +| 6c-1: additive server operation | 6b Tasks 2 and 4 | Add `dashboard/refresh-api.mjs` with shared stage runner, single-flight state, validated POST and GET; wire it into `dashboard-server.mjs` without changing the client or poll cost. Add API, security, stage and cost tests. This dispatch only. | +| 6c-2: one Refresh control | 6c-1, 6b Tasks 2, 3 and 5 | Add `client/refresh-control.mjs`; wire page, boot and client views to POST once, poll only the page's operation, and reload the active view. Remove retired controls. Prove request counts, stage progress, blocked Maintenance writes, host consent and narrow-screen UI. | +| 6c-3: read-only GET routes | 6c-2 | Reject scan-starting GET query parameters, remove GET scan paths and `/api/host-health/local`; keep the 6c-1 stage's Maintenance and inventory refresh dependencies. Prove all GET routes have no scan side effects. | +| 6c-4: current vocabulary | 6c-1 through 6c-3, 6b Task 13 | Extend the vocabulary guard for retired dashboard strings and legacy refresh URLs; update dashboard, Maintenance, upgrading, DDD and installed guidance. Include the remaining README/dashboard noun and `maintenance-discovery.mjs` string. Check links and Markdown. | +| 6c-5: decisions and supersessions | 6c-1 through 6c-4, 6b Task 14 | Reconcile ADR-0063 with ADR-0048, ADR-0025, ADR-0045, ADR-0044 and ADR-0053, including index/status rows and the decision log; record actual POST behavior and read-only GETs. Run docs and final branch gates. | + +V3 carry-ins from the program: + +- **6c-2 client fixes:** B0-22 shows the Claude Code badge as "Unknown" when Configuration was not assessed. B0-23 measures the Codex header icon against WCAG's 3:1 non-text contrast minimum and changes it only if the measurement fails. B6a-12 makes `mntSyncHash` and `mntApplyHashState` re-derive state from `location.hash` if 6c-2 touches that code; otherwise open a small issue. +- **6c-5 decision record:** B6a-9 requires ADR-0063 to state that two dashboard tabs share one server process's module state. If 6c-1's tests build the `ruflo-components` path, cover its cwd case; otherwise record the ruling that drops that test case. ADR-0048's status line moves its human-evaluation gates to v5 under D-15. +- **Live view (#256):** Confirm the session re-read changes no total before documenting D-18; if it does, apply D-18's offset alternative. Label the structured live-events input experimental under D-19. Observe a plain-folder, non-Git bind on a real machine once and fix what the observation shows. +- **Live-view follow-up (2026-09-29):** A nonzero regression showed that bounded-window re-entry increased the accepted-record total, so D-18 now retains displaced native readers and offsets in a bounded in-memory map. A genuine Codex transcript kept accepted, session, and project totals stable across idle restart and window re-entry in an isolated service probe. Eviction can still replay old records, and a new service process has no persisted offset; this does not establish exactly-once delivery or unchanged historical token accounting. D-19's structured live-events input is experimental; no real producer was verified. See the V3 live-view report. +- **Plain-folder native proof (2026-09-29):** An independently reviewed, genuine Codex `0.159.0` native `thread/fork` created a new transcript in an isolated plain folder. The real process survey and `LiveSessionsService` joined the actual host PID/cwd to that new transcript, with observed presence. The copied parent retained its exact upstream bytes and remained presence-unknown. The five-session service snapshot included unrelated controllers; it does not show five plain-folder joins. This was an idle fork with no `turn/start`, inference, billing measurement, detailed activity proof, or browser journey. The fork's empty RPC `turns` field does not mean its copied history was empty. See the V3 plain-native report. +- **Evidence-gated issue #254:** Diagnose the "CONNECTING" stall only if the browser network trace specified by the program arrives; otherwise leave it to V7. +- **Paused Activity handoff:** V4 B8 records pauses with a distinct `recordedAt` and no completion time. This branch carries that validated timestamp through Activity projection and the API, then sorts and displays pause records without claiming completion. Impossible calendar dates are rejected; valid completion timestamps remain a fallback. The V4 producer is integrated separately. +- **Pre-PR flags check (2026-09-29):** The latest Codex remained `0.159.0` at the 11:16 UTC registry recheck. Its native artifact passed integrity verification, help checks and `-s read-only -a never app-server` initialization in an isolated home. This proves flag acceptance and startup, not inference or billing. The installed global CLI was not changed. + +Extend 6c-4's guard exclusions to `docs/plans/` and the current archive/proposal/ADR layout. Move this finished plan to `docs/archive/` in the completing pull request with an archive index row. diff --git a/docs/archive/README.md b/docs/archive/README.md index e4c03435..157b2176 100644 --- a/docs/archive/README.md +++ b/docs/archive/README.md @@ -162,6 +162,7 @@ reconfirmed by this metadata audit. The per-file inventory and limitations are r | [2026-09-28-superpowers-spec-upstream-watch-ledger-branch-design.md](2026-09-28-superpowers-spec-upstream-watch-ledger-branch-design.md) | `docs/superpowers/specs/2026-09-28-upstream-watch-ledger-branch-design.md` | Finished Superpowers specification | Implemented work record preserved after its implementing PR merged. | | [2026-09-28-plan-docs-taxonomy-and-archive.md](2026-09-28-plan-docs-taxonomy-and-archive.md) | `docs/plans/2026-09-28-docs-taxonomy-and-archive.md` | Finished documentation taxonomy and archive plan | Implemented in [PR #264](https://github.com/pacphi/agentic-kit/pull/264) and [PR #266](https://github.com/pacphi/agentic-kit/pull/266). Current layout rules live in the [docs index](../README.md) and [layout guard](../../scripts/docs-layout.mjs); this plan records the build steps. | | [2026-09-28-plan-sonnet-5-5-routing-refresh.md](2026-09-28-plan-sonnet-5-5-routing-refresh.md) | `docs/plans/2026-09-28-sonnet-5-5-routing-refresh.md` | Routing and pricing research with the implemented model-catalog decision | Implemented in [PR #268](https://github.com/pacphi/agentic-kit/pull/268). The operative tier decision is in [ADR-0006](../adr/0006-primary-host-and-ambidextrous-mirroring.md); prices and benchmarks here are dated evidence. | +| [2026-09-28-plan-dashboard-refresh.md](2026-09-28-plan-dashboard-refresh.md) | `docs/plans/2026-09-28-dashboard-refresh.md` | Completed V3 implementation plan | Scoped implementation and independent review through `d2c1833b`, including native plain-folder evidence and paused Activity timestamps. Full branch gates, PR CI and develop integration remain separate gates at capture. | ## Naming convention diff --git a/docs/dashboard.md b/docs/dashboard.md index 9c6ccc16..ad48fdea 100644 --- a/docs/dashboard.md +++ b/docs/dashboard.md @@ -53,7 +53,7 @@ permanent. | System | Sessions | `#system/sessions` | Sessions | The largest retained sessions, with a localized two-line native identity, working context, and share of that host's retained bytes | | System | Storage | `#system/storage` | Storage | Where the retained bytes are, by category and host — learning stores counted separately because they dwarf everything else — plus per-series growth | | System | Runtime | `#system/runtime` | Runtime | Live host processes, their CPU and memory, background daemons, and machine denominators — refreshed on the header's poll clock while open | -| System | Catalog | `#system/catalog` | (redirect) | Retired as a visible destination. The link redirects to Maintenance › Inventory. Full scan still collects the catalog measurement, and its cards now sit in Summary | +| System | Catalog | `#system/catalog` | (redirect) | Retired as a visible destination. The link redirects to Maintenance › Inventory. Refresh machine still collects the catalog measurement, and its cards now sit in Summary | | System | Projects | `#system/projects` | Projects | Every repository with a remote that a host has recorded a session in — its approximate lines of code, language mix, total disk size and last activity. Worktrees, sub-folders and remote-less repositories are counted below the table, not listed | | System | Maintenance | `#system/maintenance` | Maintenance | Four destinations: **Inventory** (`#system/maintenance/inventory`, Focus navigation from scope through resource family to exact installation details), **Guidance** (`/guidance`, only outcomes the kit can ground, in five lanes), **Discovery** (`/discovery`, automatic sources, exact projects, collection roots, exclusions, scan coverage), and **Activity** (`/activity`, receipts, undo, interruption audits, dispositions, recipe changes, scan records). Inventory links carry scope, view, sort, `facet.` values, and the selected placement as opaque state | @@ -557,6 +557,8 @@ updates. See [Observability](https://github.com/pacphi/agentic-kit/blob/main/docs/observability.md) for the map legend, workspace facts, host capability coverage, History/Review semantics, privacy limits, and troubleshooting. +The optional `--live-source` structured input is experimental: its Ruflo and +agentic-qe examples have fixture coverage, with no verified real producer. ## System @@ -573,13 +575,11 @@ Opening System costs almost nothing. The cheap tier — the live process census, file sizes, and the figures carried forward from the last full scan — is served on every read and cached briefly. -Everything else comes from the **Full scan**—the dashboard name for the deep tier—which walks -install trees, retained-data roots, host catalog surfaces, and the eligible hosted-repository -population. That is real I/O and can take minutes on a large machine, so it runs **only when you -press Full scan** (or run `ak system --refresh=machine`). Opening the tab never triggers it. Production runs -the synchronous collectors in one worker thread so the page can report phases and remain usable -while they run. Its status names the current phase, bounded count when available, and elapsed time. -Worker containment does not claim that the filesystem work itself completes faster. +Everything else comes from choosing **Refresh machine** in the header and pressing **Refresh**. +It walks install trees, retained-data roots, host catalog surfaces, and eligible hosted +repositories. That is real I/O and can take minutes, so opening System never starts it. +Production runs the synchronous collectors in one worker thread so the page can report phases +and remain usable. Worker containment does not make the filesystem work itself faster. One scan may reuse a complete physical observation when another section asks the same bounded question. Catalog reads one physical surface once per compatible reader contract even when several @@ -591,17 +591,19 @@ the same scan. Reuse is confined to that scan. Incomplete, older, differently rooted, or differently scoped evidence falls back to a fresh bounded walk rather than being treated as equivalent. -Measurement views fetch once, then again only while a scan you started is running (Runtime also -refreshes on the header's poll clock). They read `GET /api/system/summary`, which carries only what -the page draws; `GET /api/system` and `ak system --json` keep the complete payload. Maintenance -loads when you open it, and also reloads on the shared status poll while it stays the open view -(and no measurement or provider check is already running), reading the last complete inventory each -time; opening it checks no host provider and executes nothing. **Refresh evidence** on the -Maintenance workspace is the explicit control that runs provider probes, and it rebuilds the -Inventory afterwards; **Re-measure machine** beside it runs the System Full scan, walks every -discovery source to completion, then refreshes evidence. Full scan from the System rail chains the -same provider check after the snapshot is persisted. `ak maintain --refresh` and -`ak maintain --refresh=machine` are the CLI equivalents. +Measurement views read `GET /api/system/summary`, which carries only what the page draws; +`GET /api/system` and `ak system --json` keep the complete payload. Opening Maintenance reads +the last complete inventory and starts no provider or machine check. The header has one +**Refresh** button. Its selector shows **Refresh** (local strength), **Refresh live** (live +strength), and **Refresh machine** (machine strength). The local strength refreshes Maintenance +evidence, rebuilds the inventory, and re-checks local evidence and versions. The live strength +adds bounded live checks. The machine strength first measures the machine, then refreshes +Maintenance evidence; its inventory stage walks the discovery sources and rebuilds from the new +measurement. A failed machine measurement skips those two dependent stages. +The control starts an operation with `POST /api/refresh` and reads its progress with +`GET /api/refresh`. The System and Maintenance GET routes remain read-only. +`ak maintain --refresh` and `ak maintain --refresh=machine` offer the CLI equivalents. +**Reload** re-reads the active view; it starts no machine or provider check. ### Session identity and local time @@ -632,7 +634,7 @@ while provider-backed actions live only in Maintenance. ### Catalog cards in Summary -The cross-host capability catalog is still measured by Full scan, and its three cards now live at +The cross-host capability catalog is still measured by Refresh machine, and its three cards now live at the bottom of Summary: **Host inventory profile**, **Unique across hosts** (the presence matrix, filtered by what to show, which host carries it, and which source scope), and **Project skill pressure** (a project-by-host table with per-project host disclosure). Project, user, and @@ -669,15 +671,17 @@ Guidance and Activity tabs carry a count only when something is admitted or need A fresh installation shows an empty Inventory and every installed automatic source as **Not scanned yet**; a host that is not installed reads **Not installed**. -Two actions sit side by side above the tabs, each with its helper text: **Refresh evidence** runs -provider probes on the saved measurement and rebuilds the inventory in seconds, and **Re-measure -machine** walks the filesystem, then every discovery source, then refreshes evidence, which takes -minutes. Choose Refresh evidence to build the inventory; `ak maintain --refresh` -does the same from a terminal, together with the local status checks. While either runs, both -buttons are disabled, the status line names what is running ("Refreshing evidence…"; during -Re-measure machine, each phase in turn, from "Preparing measurement…" through "Machine measured · -refreshing evidence…"), and apply, undo, and record are refused; if the work does not finish, the -previous evidence is kept. +Select **Refresh** in the header selector and press the **Refresh** button to build the inventory +from saved measurement. **Refresh live** adds bounded live checks. **Refresh machine** measures +the machine, then refreshes Maintenance evidence; the inventory stage walks the discovery sources +and rebuilds the inventory. `ak maintain --refresh` runs the local stages from a terminal. The +ordered progress labels are **Measuring the machine** (machine strength only), +**Refreshing Maintenance evidence**, **Rebuilding the inventory** (including the discovery walk +for machine strength), **Running live checks** (live strength only), and +**Re-checking local evidence and versions**. While an operation runs, another cannot start, +and Maintenance apply, undo, and record are refused. If work does not finish, the previous +complete evidence is kept. + After the probes settle the inventory builds in the background: the empty state reads **Building the inventory…** until rows appear, or names the reason if the build did not complete. @@ -743,8 +747,8 @@ A saved root starts scanning at once. Scan progress reads as visited work, never added offer **Pause** and **Stop** while running, **Resume** and **Stop** while paused, **Retry scan** after a failure, and **Scan this root** if never run; stopping shows what would be affected and asks **Stop this source?**. Automatic sources carry no per-source control: each reads Not -scanned yet with "measured by Re-measure machine", or Complete with "covered by the last -measurement". A host source whose folder is not on this machine reads **Not installed** and is +scanned yet, or Complete after the last measurement. A host source whose folder is not on this +machine reads **Not installed** and is not counted in the progress sentence or the Inventory banner. A started source keeps running until it completes, pauses, stops, or fails. Host configuration sources skip transcript, session, log, and cache trees by name so they can complete. @@ -783,12 +787,11 @@ the totals. Every parent with breakdowns also gets an "everything else" row, so adds up to its parent. Roots that do not exist on this machine are listed as absent rather than ranked at 0 B, and roots that could not be read say so with their reason. -**Project trees** are excluded by default, and the chip that includes them is a *scan* control, -not a filter. One large repository can outweigh every shared cache combined, and a chart -containing it is a chart of one repository — so the ranking says, in the panel, that they were -left out. Turning the chip on starts a new Full scan that walks them (and turning it off starts -one that does not); it is disabled while a scan is running. `ak system --refresh=machine` scans -without project trees; add `--project-trees` to include them. +**Project trees** are excluded by default. The **Include project trees** option is +available only when **Refresh machine** is selected; it changes the measurement scope. +One large repository can outweigh every shared cache combined, so the panel says when +project trees were left out. Select that option and press Refresh to measure them. +`ak system --refresh=machine` omits them unless `--project-trees` is added. ### Two reclaimable tiers, never one total @@ -810,7 +813,7 @@ removes anything; where a CLI already owns the cleanup, the row names it. ### Reading the numbers honestly -- **A section that has never been scanned says so.** It reads "not measured yet — run Full scan", +- **A section that has never been scanned says so.** It reports an unmeasured state, never `0`. A zero here means a real, measured zero. - **A total whose inputs were incomplete renders as `≥ N`.** If one subtree could not be read or a walk hit its cap, the sum is a floor, not a total, and is labeled that way. @@ -954,8 +957,9 @@ provider, model/default agent, and applicable credentials or local endpoint. Automatic checks do not invoke its config-debug command, which can install dependencies. Unresolved remote configuration and native overrides stay Unknown. -**Check again** refreshes the local evidence for any host. **Check connection** -runs only for hosts managed by ak, and requires +Select **Refresh** in the header selector and press the **Refresh** button to re-check local +evidence for any host. +**Check connection** runs only for hosts managed by ak, and requires checking a confirmation box first: it sends one small provider request, using normal billing and native context. Native startup may initialize dependencies and update local cache/session files. Agent tools are restricted, and no repair diff --git a/docs/ddd/machine-footprint.md b/docs/ddd/machine-footprint.md index 28effd21..a27b8295 100644 --- a/docs/ddd/machine-footprint.md +++ b/docs/ddd/machine-footprint.md @@ -193,7 +193,8 @@ FootprintSnapshot { asOf, completeness, install, runtime, storage, catalog, pro v Delivery GET /api/system → cheap tier + persisted snapshot (token auth, loopback, no egress) - GET /api/system?refresh=deep → start-or-attach the single-flight deep scan + POST /api/refresh → start the single-flight staged refresh + GET /api/refresh → read that operation's progress GET /api/system/summary → the same read, catalog/storage/install/projects/consumers each projected to what the System page draws ak system [--refresh[=live|machine]] [--project-trees] [--json] → the same collector, CLI-rendered @@ -521,7 +522,7 @@ That identity also governs acquisition cost inside one Catalog collection. Compa keyed by normalized physical path and reader contract, so Claude and OpenCode bindings to the same skill surface share one bounded observation while retaining two ConsumerBindings. Markdown and file-stem entrypoints likewise compute one digest per file. The observation map is created and -discarded inside the collection; a later Full scan always observes the filesystem again. A path +discarded inside the collection; a later Refresh machine always observes the filesystem again. A path match under a different reader contract is not reusable evidence. Every occurrence retains host, surface, source scope (`user`, `project`, or `plugin`), project @@ -704,38 +705,28 @@ The link is user-initiated browser navigation; the kit itself never fetches the token auth ([ADR-0014](../adr/0014-dashboard-auth-and-remediation.md)), `no-store`, zero egress. The response is the cheap tier computed fresh (TTL ~60s, shared-cache pattern like the project-snapshot cache) merged with the persisted deep snapshot and its `asOf`. -`?refresh=deep` starts the dashboard's **Full scan** or attaches to the one in flight -(single-flight, like the usage index's coalesced builds). In production, `index.mjs` retains the -single-flight promise and public activity state while `deep-scan-worker.mjs` runs the synchronous -runner in one worker thread. Phase and Projects progress messages return to the main thread, so -ordinary reads remain responsive while the worker is busy. Injected collectors and filesystem -implementations run the same `deep-scan-runner.mjs` inline rather than attempting to serialize test -functions. This containment is not evidence that total scan duration decreased. - -The System measurement routes stay GET-only: a Full scan re-measures local state and writes only this domain's own -snapshot file — it mutates no user data. - -`GET /api/system` is the complete read model, the same shape as `ak system --json`. -`GET /api/system/summary` is the page's read: the same payload (and the same `?refresh=deep` and -`&trees=` parameters) with `catalog`, `storage`, `install`, `projects` and `consumers` each -projected by `dashboard/system-summary.mjs` to an allow-list of keys — the catalog's items cut to -key, kind, name, hosts, source scopes, digest coverage, and distinct plugin providers -(`presence[].provider` with `ref` and `version`); storage's category/host/project/session tree cut -to key, label, bytes and children; a measured project row's per-tool native-addon lists and -per-project framework/dependency stack detection dropped entirely (never rendered). The catalog's -repeated presence copies (`item.presence` details, `consumerBindings`, `artifacts`) grow with -items × projects × hosts and are not drawn, so the page and its 30-second Runtime poll never -download them; the other four sections carried the same shape of excess and were the majority of -the endpoint's real-machine bytes once the catalog alone was slimmed. The projection is a -Dashboard-delivery view; this domain's collector output is unchanged. - -**A deep scan never runs on its own.** Opening the System area issues a plain `GET /api/system/summary`; -only **Full scan** adds `?refresh=deep`. A deep scan can cost minutes of I/O on a large corpus, and -making the act of *looking* cost that is a worse trade than a stale figure that states -how stale it is. Staleness is therefore surfaced rather than pre-empted: every deep-tier figure -renders with its snapshot's `asOf`, and past `SNAPSHOT_STALE_AFTER_MS` (7 days) the freshness -label turns amber and reads "stale, scan again". The client polls only while a user-started scan is -running, and stops when it finishes. +The header's **Refresh** control starts an operation through `POST /api/refresh`. +Choosing **Refresh machine** runs the deep collector first; `GET /api/refresh` reads the operation's +stage progress. In production, `index.mjs` retains the single-flight promise and public +activity state while `deep-scan-worker.mjs` runs the synchronous runner in one worker thread. +Phase and Projects progress messages return to the main thread so ordinary reads remain +responsive. Injected collectors run the same `deep-scan-runner.mjs` inline in tests. +Worker containment does not show that total scan duration decreased. + +The System measurement routes are read-only GETs. `GET /api/system` returns the complete +read model, the same shape as `ak system --json`. `GET /api/system/summary` projects +`catalog`, `storage`, `install`, `projects`, and `consumers` to the allow-listed keys +drawn by the page. The catalog omits repeated presence details, consumer bindings and +artifacts; Projects omits native-addon and stack details. This projection reduces the +payload downloaded by the page and its Runtime poll while the collector output stays +unchanged. + +Opening System reads the saved snapshot and starts no measurement. A deep scan can cost +minutes of I/O, so the user explicitly chooses **Refresh machine** and presses **Refresh**. +Every deep-tier figure renders with its snapshot's `asOf`; after `SNAPSHOT_STALE_AFTER_MS` (7 days), +the freshness label turns amber. The client reads progress for the user-started operation +and stops polling when it finishes. **Reload** re-reads the active view without starting +machine or provider checks. One deliberate divergence from Observability's delivery: absolute paths are **part of this payload**. `publicLivePayload`'s leaf-only rule exists to keep incidental provenance out of @@ -746,7 +737,7 @@ nothing to leak. The CLI twin (`ak system`) renders the same collector output, `--json` emitting the collector's payload verbatim, following the one-collector-two-surfaces precedent of the usage scorecard. -`ak system --refresh=machine` is the terminal spelling of **Full scan** and writes the same snapshot. +`ak system --refresh=machine` is the terminal spelling of **Refresh machine** and writes the same snapshot. ## Invariants @@ -856,7 +847,7 @@ normative and this table restates it for readers of this document. | Definition digest | SHA-256 over one complete bounded observed capability definition; equality proves those files match, not host selection, ownership, usage, or removal safety | | ProjectCapabilityPressure | Project/user/plugin contributions and exact overlap per project and host; context inclusion remains unknown | | ProjectFootprint | One eligible hosted repository's size facts: approximate LOC by language, tree/`.git`/`node_modules` bytes, last activity, and a proven HTTPS web link | -| Deep scan | The explicit, user-triggered, single-flight measurement pass called **Full scan** in the dashboard; it produces a FootprintSnapshot over the stated bounded populations | +| Deep scan | The explicit, user-triggered, single-flight measurement pass selected by **Refresh machine** in the dashboard; it produces a FootprintSnapshot over the stated bounded populations | | Cheap tier | The per-request census + known-file stats + snapshot carry-forward served on every read | ## References diff --git a/docs/ddd/observability.md b/docs/ddd/observability.md index 513ff96e..33e7aaeb 100644 --- a/docs/ddd/observability.md +++ b/docs/ddd/observability.md @@ -901,6 +901,8 @@ agentic-qe source adapters require explicit, repeatable `--live-source 'surface= registration, where `surface` is `ruflo` or `aqe`. Explicit sources consume tailer capacity before Claude/Codex discovery. Paths resolve against the startup working directory and are not confined to the project, so registration is an operator authorization to read that file. +The structured input remains experimental: parser fixtures exist, but no real producer has been +verified. Registration alone supplies no activity or runtime coverage. Independent plugin, skill, MCP, and gate registries are not implemented. This is an explicit source coverage limitation; ADR-0012 is Implemented because the supported adapter contract does not claim automatic upstream discovery. diff --git a/docs/ddd/ubiquitous-language.md b/docs/ddd/ubiquitous-language.md index d84bbe64..c62b9a92 100644 --- a/docs/ddd/ubiquitous-language.md +++ b/docs/ddd/ubiquitous-language.md @@ -88,7 +88,7 @@ token estimates never become observed token evidence. An absent or incompatible | Quick live checks | The bounded, no-cost subset of checks that a plain `ak status --refresh=live` runs in parallel: AQE embedding request, Codex MCP handshake, provider wiring, security packages, deja-vu structure, and the `memory` check's temp-dir CLI store/retrieve/purge round trip (no MCP tool calls; that is the separate `memory-routes` slow proof); a timeout is `inconclusive`. `--only CHECK[,CHECK...]` names exactly which checks to run, including a slow proof | | Slow proof | A live check that runs only when named with `--only`, up to six minutes each: `learning` (trains a temporary fixture, asserts patterns persist), `harvest` (records an outcome and distills through Ruflo in an isolated store), `aqe` (storage, embedding configuration/provenance, and the browser payload), `memory-routes` (the memory round trip plus whether CLI and MCP see each other's writes, remembered as the `memory` check). A named check runs even when it would not otherwise apply; `learning` and `harvest` are never remembered | | Connection check | `ak host check-connection [--yes] [--json] [--dry-run]`: the consent-gated, paid probe of a managed host's live connection. It is never a strength of `--refresh` and runs only for a managed host whose local checks already pass, after printing the shared disclosure and asking `[y/N]` on a TTY (refused without `--yes` off a TTY) | -| Machine measurement | The full re-measure `--refresh=machine` runs: walking install trees, storage, the cross-host catalog, and (with `--project-trees`) your projects' working trees, then persisting the result and rebuilding the inventory. The dashboard's Full scan and Re-measure machine controls run the same chain | +| Machine measurement | The full re-measure `--refresh=machine` runs: walking install trees, storage, the cross-host catalog, and (with `--project-trees`) your projects' working trees, then persisting the result and rebuilding the inventory. Choosing **Refresh machine** in the dashboard and pressing **Refresh** runs the same chain; **Reload** only re-reads the active view | | Memory route observation | What `ak status --refresh=live --only memory-routes` reports after its CLI proof, in its throwaway project only: whether a key written through Ruflo's CLI is readable through MCP and the reverse, where each landed, and the MCP backend seen. A split is a warning and an unusable interface is "not observed"; it never fails the suite, is not part of the quick live checks, and says nothing about an existing corpus | | Observed routing pair | An exact `@claude-flow/cli` release and platform on which the memory route observation was recorded. Only for such a pair does status say which Ruflo interface reads which project-memory store; a neighbouring, prerelease or build-tagged version, or another platform, stays unverified | | Canonical memory store | `/.swarm` for the root every ak memory launch contract pins (the repository root, else the folder, unless that is an unsuitable memory folder): `memory.db` and, with the native bridge, `agentdb-memory.db`. Status reports it from any subfolder, with each file's size, live WAL, largest namespace and how much of it is set to expire | diff --git a/docs/maintenance.md b/docs/maintenance.md index fa9bafb4..441b072c 100644 --- a/docs/maintenance.md +++ b/docs/maintenance.md @@ -23,7 +23,7 @@ environment; native Windows mutation support remains an integration gate. ## Host alignment in User and Project views Select **Host alignment** under **More views**, then choose **User** or **Projects** -and an optional project filter. Use **Refresh evidence** to inspect current host +and an optional project filter. Use **Refresh** to inspect current host configuration. The rows identify retired peer transports and other host-alignment anomalies without exposing configuration contents or local paths in the inventory. @@ -58,25 +58,22 @@ The dashboard workspace has four tabs. Each answers a different question. | **Activity** | What changed? Receipts, undo, interruption audits, dispositions, recipe changes, and scan records. | Opening Maintenance reads the last complete inventory and opens **Inventory** across all scopes. -Nothing scans on open. A fresh installation has no inventory yet; the empty state reads **No -inventory has been built yet. Use Refresh evidence, above, to build it.** and every installed -automatic source reads **Not scanned yet** (a host that is not installed reads **Not installed**). Two actions sit side by side above the tabs, each with its own helper -text: - -- **Refresh evidence** runs provider probes on the saved measurement and rebuilds the inventory. - It takes seconds. The CLI equivalent is `ak maintain --refresh`. -- **Re-measure machine** walks the filesystem to re-measure installs, storage, projects, and every - discovery source, then refreshes evidence. It takes minutes. The CLI equivalent is - `ak maintain --refresh=machine`. - -Both controls run provider probes: Re-measure machine includes the Refresh evidence stage. While either action -runs, both buttons are disabled, the status line names what is running ("Refreshing evidence…"; -during Re-measure machine, each phase in turn, from "Preparing measurement…" through "Machine -measured · refreshing evidence…"), and apply, undo, and record are refused. If -the work does not finish, the previous evidence is kept. The inventory build runs after the probes -settle and can take a few seconds on a large footprint; the empty state reads **Building the -inventory…** until the rows appear, and **The last inventory build did not complete** with a short -reason if it fails. The retired Catalog link (`#system/catalog`) redirects to Inventory. +Nothing scans on open. A fresh installation has no inventory yet; the empty state asks you to +use **Refresh** in the header, and every installed automatic source reads **Not scanned yet** +(a host that is not installed reads **Not installed**). + +The header has one **Refresh** button. Its selector shows **Refresh** (local strength), +**Refresh live** (live strength), and **Refresh machine** (machine strength). The local strength +runs provider probes on the saved measurement, rebuilds the inventory, and re-checks local +evidence and versions. The live strength adds bounded live checks. The machine strength first +re-measures installs, storage, and projects, then refreshes Maintenance evidence. Its inventory +stage walks the discovery sources before rebuilding the inventory from the new measurement. +The CLI equivalents are `ak maintain --refresh`, `ak maintain --refresh=live`, and +`ak maintain --refresh=machine`. **Include project trees** applies only to **Refresh machine**. +While an operation runs, another refresh cannot start; apply, undo, and record are refused. +The prior complete evidence is retained if work does not finish. The empty state reads +**Building the inventory…** while the inventory builds, or names the failure reason. +The retired Catalog link (`#system/catalog`) redirects to Inventory. ## Inventory @@ -299,7 +296,7 @@ Discovery is where you tell Agentic Kit where to look. Configuration is user int configuration, Codex user configuration, OpenCode user configuration, Hermes user configuration, Projects (every project a recorded host session has visited), Runtimes, Package managers, Ollama (over loopback only), and Providers. Each states what it inspects, never a path. Automatic - sources have no per-source scan control: their coverage comes from **Re-measure machine**, and + sources have no per-source scan control: their coverage comes from **Refresh machine**, and the non-filesystem ones (Runtimes, Package managers, Ollama, Providers) are covered by the provider check. Asking `ak maintain scans start` to walk one of those is refused with `SOURCE_NOT_SCANNABLE`. A host source whose folder is not on this machine (for example Hermes @@ -337,7 +334,7 @@ Scans are resumable and completion-oriented. run. A started root keeps running through its work slices until it completes, pauses, stops, or fails; you never have to resume it yourself. `ak maintain scans start --source ID` does the same from the CLI and waits for the final state. Automatic sources show instead whether they are - measured by Re-measure machine or covered by the last measurement. + measured by choosing **Refresh machine** or covered by the last measurement. Progress is factual: "Scanned N entries. X of Y sources are complete. N sources have not been scanned yet." Counts are visited work, never totals. @@ -460,7 +457,7 @@ ak maintain undo --receipt RECEIPT_ID --yes Undo needs the recorded provider and version, a reversible or compensating operation, and an exact current postimage. If anything changed after apply, undo refuses instead of overwriting the new state. If no inventory has been built yet, plan and apply refuse with `SCAN_REQUIRED`; choose -**Refresh evidence** or run `ak maintain --refresh` first. If a placement cannot be +**Refresh** or run `ak maintain --refresh` first. If a placement cannot be bound to an exact executable finding, they refuse with `PLACEMENT_FINDING_UNRESOLVED`. Rollback classes are separate from safety: **reversible** (the provider restores and verifies the @@ -494,7 +491,8 @@ server registration and are not separate MCP installations. Measured user instruction files retain their resolved configuration location through **Reveal exact path**. Their paths remain private in ordinary inventory responses. -Refresh evidence after upgrading to populate locations missing from an older snapshot. +Select **Refresh** in the header selector and press the **Refresh** button after upgrading to +populate locations missing from an older snapshot. Inventory can relate a standalone skill and a plugin-contributed skill by exact name, bounded entrypoint digest, or bounded full-definition digest. Full-definition equality includes the @@ -542,7 +540,7 @@ transaction applies; follow the verb-specific options below. | Verb | What it does | |------|--------------| -| `[report] [--refresh[=live\|machine]] [--project-trees]` | `report` (the default verb) reads the last measurement. A bare `--refresh` refreshes Maintenance evidence and rebuilds the Inventory first (the dashboard's **Refresh evidence**); `--refresh=machine` re-measures System first and walks every discovery source to completion before rebuilding (the dashboard's **Re-measure machine**); `--project-trees` with `--refresh=machine` also measures your projects' working trees. | +| `[report] [--refresh[=live\|machine]] [--project-trees]` | `report` (the default verb) reads the last measurement. A bare `--refresh` refreshes Maintenance evidence and rebuilds the Inventory first (the dashboard's **Refresh**); `--refresh=machine` re-measures System first and walks every discovery source to completion before rebuilding (the dashboard's **Refresh machine**); `--project-trees` with `--refresh=machine` also measures your projects' working trees. | | `inventory [--scope S] [--view V] [--facet name=value ...] [--search TEXT] [--sort ORDER] [--cursor TOKEN] [--limit N]` | Queries placements. | | `show --placement ID [--reveal]` | Prints the inspector; `--reveal` prints the exact, owner-only path. | | `guidance [--lane LANE]` | Lists admitted Guidance entries and per-lane counts. | @@ -627,10 +625,13 @@ or an export contains a local path unless you reveal or export it deliberately. ## Dashboard security boundary -Maintenance is the only dashboard mutation surface. The v1 routes remain as compatibility: +Maintenance actions are the dashboard's exact-placement mutation surface. The v1 routes remain +for reads and explicit actions; refresh starts through the shared operation route: ```text -GET /api/maintenance (?refresh=scan runs the provider check, then rebuilds the Inventory) +GET /api/maintenance (reads the saved report) +POST /api/refresh (starts the selected refresh strength) +GET /api/refresh (reads operation progress) POST /api/maintenance/plans POST /api/maintenance/apply POST /api/maintenance/undo @@ -788,5 +789,6 @@ rows. Column headers stay pinned while dates and records scroll. Executable installations expose their measured launcher through **Reveal exact path**. Detection checks PATH (including Windows PATHEXT), resolves symlinks, and reads bounded npm `bin` metadata when a launcher is not on PATH. When only the installation root was measured, -that root remains revealable. Paths stay out of the public inventory. Re-measure machine -to collect new launcher evidence; discovery covers the environment running the scan. +that root remains revealable. Paths stay out of the public inventory. Select **Refresh machine** +in the header selector and press **Refresh** to collect new launcher evidence; discovery covers +the environment running the scan. diff --git a/docs/models.md b/docs/models.md index a24aa393..8c9c1906 100644 --- a/docs/models.md +++ b/docs/models.md @@ -236,7 +236,7 @@ browser never derives a provider or publisher from a name and never invents an e An `unknown` cell explains which evidence is absent. Model details separately name published or discovered status, account access, local routability, and the operator's next evidence step. The table labels the discovery dimension **Catalogued**, not **Available**, so provider publication or -local discovery cannot be mistaken for account entitlement. A local refresh now resolves OpenCode's +local discovery cannot be mistaken for account entitlement. A local model refresh resolves OpenCode's effective configuration, removing unknowns caused only by ignored global, JSONC, agent, or command layers. Discovery still does not establish entitlement; configuration does not establish successful use; and a model id never establishes the serving provider. Catalog Explorer model details show an diff --git a/docs/observability.md b/docs/observability.md index ee049371..d8a1ff98 100644 --- a/docs/observability.md +++ b/docs/observability.md @@ -23,8 +23,9 @@ ak dashboard \ > OpenCode process presence is observed. What you don't get: ruflo and agentic-qe activity, which are never > auto-discovered and only appear once you register their event file > explicitly (see [Evidence and limitations](#evidence-and-limitations)). -> `--live-source` reads a file that something else already writes. It does not make Ruflo or -> agentic-qe produce events. +> `--live-source` is **experimental**. Its schema and parser have fixture coverage, but no +> verified Ruflo or agentic-qe producer currently writes a structured live-events file. +> Registering a path reads that file; it does not start a producer or establish runtime coverage. Open `#observability/live` or `#observability/history`, for example `http://127.0.0.1:7431/#observability/live` — once the dashboard's per-session token is already in @@ -37,6 +38,14 @@ leaves, collectors stop after 30 seconds by default. The next request resumes ea stopped, so work written while nobody was watching still appears. Stopping the dashboard closes the live service and all clients. +When a native transcript leaves the newest-file window, the service keeps a bounded in-memory +reader state. A file that returns while that state is retained resumes at its prior byte offset; +its accepted-record count does not rise from replay alone. The retained state is limited to twice +the configured file bound, with a minimum of two readers. Once evicted, a returning file is read +from its start. A new dashboard process has no saved native offset; its first scan bootstraps +metadata and begins following existing files at their ends. The live view does not provide a +durable, exactly-once event archive. + Model lifecycle is a separate read model under **Usage → Models**. It may consume bounded model ids already derived by the historical usage index, but it never consumes live transcript content or changes Observability state. Public catalogue enrichment cannot rename, add, remove, or alter @@ -358,7 +367,8 @@ paths remain absolute. The parser rejects an unsupported/missing surface or an empty path, but registration does not prove that the file exists, is a regular file, is inside the current project, or is produced by the named subsystem. Registration also does not turn on event output: `--live-source` observes a file an -existing producer writes. Unreadable/malformed sources degrade their adapter rather than +external producer may write. No real Ruflo or agentic-qe producer has been verified for this +structured format. Unreadable/malformed sources degrade their adapter rather than crashing the dashboard. Only register a local file you trust the dashboard process to read. The structured adapter still constructs allowlisted events, so arbitrary JSON fields do not pass through to the browser. diff --git a/docs/plans/2026-09-26-issues-237-238-239-verification-and-decisions.md b/docs/plans/2026-09-26-issues-237-238-239-verification-and-decisions.md index bf641d53..07114b43 100644 --- a/docs/plans/2026-09-26-issues-237-238-239-verification-and-decisions.md +++ b/docs/plans/2026-09-26-issues-237-238-239-verification-and-decisions.md @@ -2540,3 +2540,20 @@ in a throwaway repository (run 36451224053) showed that the job token pushes the email). Each firing is recorded as a `fired` line, so a routine that fails is fired again at most once; the Actions API's last successful run replaces a daily heartbeat commit; #243 closes. Design: `docs/archive/2026-09-28-superpowers-spec-upstream-watch-ledger-branch-design.md`. + +### V3 dashboard refresh implementation status (2026-09-29) + +The V3 branch at `19953b6d` delivers Addendum 3 Item 4's dashboard half: one header +Refresh control with Refresh, Refresh live, and Refresh machine choices; Reload only +re-reads the active view. `POST /api/refresh` starts the ordered operation; +`GET /api/refresh` reads the latest process-local state. GET system and Maintenance +routes reject retired scan-starting query arguments. ADR-0063 records the operation +identity and volatile single-flight boundary, and supersedes ADR-0048's two old +controls, ADR-0025 §5's GET-started scan rationale, and ADR-0045's GET scan trigger. +ADR-0044 and ADR-0053 now point to the explicit POST and header Refresh paths. + +D-15 keeps ADR-0048 Accepted with partial delivery: the human usability, screen-reader, +and cross-platform evaluation gates move to v5. The V4 offline retry change is on a +separate unmerged branch; this V3 source retains ADR-0063 Known limitations item 1. +This entry records V3 task 6c-5's source-bound documentation integration, not final +branch acceptance or completion of the separately dispatched live-view work. diff --git a/docs/upgrading.md b/docs/upgrading.md index c6e61c49..0b93b93b 100644 --- a/docs/upgrading.md +++ b/docs/upgrading.md @@ -150,7 +150,7 @@ When the ChatGPT desktop app imports a Claude Code transcript, it saves a copy a Project discovery no longer counts these copies: they give a folder no Codex host and no Desktop origin, and System says how many it set aside. A snapshot taken before this change still holds the old hosts and origins, so the Footprint snapshot schema advances to v8. This build reports a v7 -snapshot as unreadable until you run **Full scan** in System or `ak system --refresh=machine`. It +snapshot as unreadable until you run **Refresh machine** in System or `ak system --refresh=machine`. It is never shown under the new rule. See [ADR-0060](adr/0060-session-surface-initiator-and-product-names.md) §3. ## 2026-09-27: Ruflo support window @@ -389,7 +389,7 @@ working context for the already-ranked top-N rows; it does not read prompts, tit System > Sessions renders the identity as one two-line transcript link: localized date/time first, then a shortened opaque native ID. Focus or hover discloses the original filename, full native ID, and detailed localized time with timezone. If an older snapshot or host has no declared opening -instant, the measured mtime is explicitly labeled **Last active**. Run **Full scan** or +instant, the measured mtime is explicitly labeled **Last active**. Run **Refresh machine** or `ak system --refresh=machine` to populate native identity for an existing snapshot; no configuration or payload migration is required. @@ -403,7 +403,7 @@ such as the user home from triggering several hundred thousand unrelated filesys Because that population is narrower than the v6 measurement contract, the Footprint snapshot schema advances to v7. A v6 snapshot is reported as unreadable by this build until the next explicit -**Full scan** or `ak system --refresh=machine`; it is never silently reinterpreted. +**Refresh machine** or `ak system --refresh=machine`; it is never silently reinterpreted. ## 2026-09-03: System Catalog snapshot v6 @@ -452,9 +452,10 @@ user's agentic-kit state directory. Existing System snapshot files remain read-o Catalog schema v4 is still refreshed with `ak system --refresh=machine`. `ak sync` neither selects nor executes Maintenance findings. -Browser refresh now reads the saved Maintenance report without polling providers. Use **Refresh -evidence** or `ak maintain --refresh` for current provider/version evidence. A successful System -deep rescan also chains one Maintenance scan after the snapshot is persisted. +Browser **Reload** reads the saved Maintenance report without polling providers. Use the header's +**Refresh** control with **Refresh** selected, or `ak maintain --refresh`, for current provider/version +evidence. A successful **Refresh machine** operation persists a new System snapshot before +updating Maintenance. The first provider set is intentionally narrower than the inventory. Claude plugin disable, update, and remove; exact Codex plugin/MCP removal; exact receipt-owned skill archive; one bounded diff --git a/docs/usage-scorecard-metrics.md b/docs/usage-scorecard-metrics.md index 4221efdd..ed402269 100644 --- a/docs/usage-scorecard-metrics.md +++ b/docs/usage-scorecard-metrics.md @@ -1951,7 +1951,7 @@ aggregated here, and deriving the baseline from a widened bound would silently stretch it to whatever lookback the caller happened to pass. The dashboard route widens it to the depth the personal tap-share baseline needs rather than to the previous window alone: `days + BASELINE_TRAILING_DAYS` (`lookbackDays`, -`src/lib/dashboard-server.mjs:1860`); `ak usage score` applies the same rule +`src/lib/dashboard-server.mjs:1872`); `ak usage score` applies the same rule (`src/commands/usage.mjs:316`). One extra window would be a strict subset — too shallow for `promptBaselines`, which needs BASELINE_MIN_ACTIVE_DAYS of history BEFORE the displayed window and returns diff --git a/src/lib/dashboard-server.mjs b/src/lib/dashboard-server.mjs index 5b86f267..cef33622 100644 --- a/src/lib/dashboard-server.mjs +++ b/src/lib/dashboard-server.mjs @@ -1,4 +1,6 @@ import { HOST_HEALTH_POST_ROUTES, handleHostHealthPost } from './dashboard/host-health-api.mjs'; +import { REFRESH_POST_ROUTES, REFRESH_LOCAL_TIMEOUT_MS, createRefreshOperation, + dashboardRefreshStages, handleRefreshPost, handleRefreshGet } from './dashboard/refresh-api.mjs'; import { createHostReadinessReader } from './host-readiness.mjs'; // dashboard-server.mjs — a read-only, localhost-only web dashboard for the kit. // @@ -37,14 +39,13 @@ import { createHostReadinessReader } from './host-readiness.mjs'; // GET /api/system → the machine-footprint payload (ADR-0025): the cheap // tier (runtime census + known-file stats, TTL-cached // ~60s) merged with the last persisted deep snapshot, -// carried forward with ITS asOf. `?refresh=deep` starts -// or attaches to the single-flight deep scan and returns -// immediately with progress state; `&trees=1|0` sets -// whether that scan walks project working trees. -// GET /api/system/summary → the same read (same `?refresh=deep&trees=`) +// carried forward with ITS asOf. This GET is read-only. +// GET /api/system/summary → the same read // with the catalog projected to what the System page // draws (dashboard/system-summary.mjs). The page and its // Runtime poll read this; /api/system stays complete. +// POST /api/refresh → starts one explicit staged refresh operation. +// GET /api/refresh → reads progress for that operation. // // The status rows are gathered by calling status.mjs's own collect() IN // PROCESS — safe only because the evidence store (ADR-0063) made a warm-cache @@ -148,16 +149,16 @@ const STATUS_TIMEOUT_MS = 30_000; * settles within STATUS_TIMEOUT_MS, resolves to an honest empty payload rather than * rejecting or hanging the server, so /api/status always answers with valid * JSON. */ -function inProcessStatus(cwd) { +function inProcessStatus(cwd, { refresh = false, timeoutMs = STATUS_TIMEOUT_MS } = {}) { return () => new Promise((resolve) => { let settled = false; const timer = setTimeout(() => { if (settled) return; settled = true; resolve({ overall: 'unknown', rows: [], error: 'status collection timed out' }); - }, STATUS_TIMEOUT_MS); + }, timeoutMs); timer.unref?.(); - statusCollect({ pkgRoot: PKG_ROOT, cwd, refresh: false }) + statusCollect({ pkgRoot: PKG_ROOT, cwd, refresh }) .then((rows) => ({ overall: worstLevel(rows), rows })) .catch((e) => ({ overall: 'unknown', rows: [], error: String(e?.message ?? e) })) .then((result) => { @@ -1086,7 +1087,7 @@ function lazyLive(liveOptions = {}) { * machineWideIntel?: (projects: Array) => any, * models?: any, modelScopeKey?: string, system?: any, systemOptions?: any, * maintenance?: any, maintenanceOptions?: any, hostReadiness?: any, - * management?: any, managementOptions?: any }} [opts] + * management?: any, managementOptions?: any, refreshStages?: Record Promise> }} [opts] * @returns {Promise<{ url: string, urlWithToken: string, port: number, token: string, close: () => Promise }>} */ export function startDashboard({ @@ -1097,7 +1098,7 @@ export function startDashboard({ transcriptClientBuffer = 64, transcriptMaxClients = 16, intelWatch, intelClientBuffer = 256, intelMaxClients = 32, discoverProjects, machineWideIntel, models, modelScopeKey, system, systemOptions = {}, - maintenance, maintenanceOptions = {}, management, managementOptions = {}, hostReadiness, + maintenance, maintenanceOptions = {}, management, managementOptions = {}, hostReadiness, refreshStages, } = {}) { const refused = refusesDefaultState({ fetchStatus, usage, limits, hooks, live, transcripts, intelWatch, discoverProjects, machineWideIntel, @@ -1166,7 +1167,7 @@ export function startDashboard({ const provideMaintenance = typeof maintenance === 'function' ? maintenance : maintenance ? async () => maintenance : async () => { if (refused) { - // Logged here because refreshMaintenanceAfterSystem swallows errors. + // Log once when the default Maintenance service is refused. if (!refusalLogged) { refusalLogged = true; console.error(HERMETIC_REFUSAL); } throw new TypeError(HERMETIC_REFUSAL); } @@ -1216,24 +1217,11 @@ export function startDashboard({ .finally(() => { inventoryRefreshPromise = null; }); return inventoryRefreshPromise; } - let maintenanceRefreshSource = null; - let maintenanceRefreshPromise = null; - function refreshMaintenanceAfterSystem(deepScan) { - if (maintenanceRefreshSource === deepScan) return maintenanceRefreshPromise; - maintenanceRefreshSource = deepScan; - maintenanceRefreshPromise = Promise.resolve(deepScan).then(async (result) => { - if (result?.ok !== true || result?.persisted?.ok === false) return null; - const model = await (await getMaintenance()).scan({ deep: false }); - refreshInventoryAfterProviderScan({ measured: true }); - return model; - }).catch(() => null).finally(() => { - if (maintenanceRefreshSource === deepScan) { - maintenanceRefreshSource = null; - maintenanceRefreshPromise = null; - } - }); - return maintenanceRefreshPromise; - } + const refreshOperation = createRefreshOperation({ stages: refreshStages ?? dashboardRefreshStages({ + cwd, pkgRoot: PKG_ROOT, getSystem, getMaintenance, refreshInventoryAfterProviderScan, + getHostReadiness, statusCollect: inProcessStatus(cwd, { refresh: true, timeoutMs: REFRESH_LOCAL_TIMEOUT_MS }), + loadConfig: loadKitConfig, + }) }); let transcriptServicePromise; const provideTranscripts = typeof transcripts === 'function' ? transcripts : transcripts ? async () => transcripts : async () => { @@ -1403,7 +1391,7 @@ export function startDashboard({ let maintenanceApiPromise; const getMaintenanceApi = async () => (maintenanceApiPromise ||= getMaintenance() .then((service) => createMaintenanceDashboardApi({ - service, management: getManagement, sessionToken: token, afterScan: refreshInventoryAfterProviderScan, + service, management: getManagement, sessionToken: token, }))); const server = http.createServer(async (req, res) => { @@ -1417,7 +1405,8 @@ export function startDashboard({ const maintenanceMutation = req.method === 'POST' && (MAINTENANCE_MUTATION_ROUTES.has(url) || MAINTENANCE_V2_MUTATION_ROUTES.has(url)); const healthMutation = req.method === 'POST' && HOST_HEALTH_POST_ROUTES.has(url); - if (req.method !== 'GET' && !maintenanceMutation && !healthMutation) { + const refreshMutation = req.method === 'POST' && REFRESH_POST_ROUTES.has(url); + if (req.method !== 'GET' && !maintenanceMutation && !healthMutation && !refreshMutation) { res.writeHead(405).end('method not allowed'); return; } @@ -1445,20 +1434,21 @@ export function startDashboard({ // Query tokens remain an SSE compatibility exception for GET. Mutation // capability can only be reached with the explicit header; it never rides // in a URL, browser history, referrer or server log. - const authorized = (maintenanceMutation || healthMutation) + const authorized = (maintenanceMutation || healthMutation || refreshMutation) ? tokenMatches(req.headers['x-dash-token'], token) : checkToken(req, query); if (url.startsWith('/api/') && !authorized) { sendUnauthorized(res, 'Wrong or missing dashboard token.'); return; } - if (maintenanceMutation || healthMutation) { + if (maintenanceMutation || healthMutation || refreshMutation) { const mutationRejection = maintenanceMutationRejection(req.headers); if (mutationRejection) { res.writeHead(403, { 'content-type': 'text/plain; charset=utf-8' }); res.end(mutationRejection); return; } - if (healthMutation) { await handleHostHealthPost(url, req, res, getHostReadiness); return; } + if (healthMutation) { await handleHostHealthPost(req, res, getHostReadiness); return; } + if (refreshMutation) { await handleRefreshPost(req, res, refreshOperation); return; } try { await (await getMaintenanceApi()).mutate(url, req, res); } catch { sendJson(res, 503, { error: 'maintenance operation unavailable' }); } return; @@ -1987,35 +1977,13 @@ export function startDashboard({ // `project` shapes the answer for a route: identity for /api/system (the // documented `ak system --json` shape), systemSummaryPayload for the page. async function handleSystem(req, res, query, project = (payload) => payload) { + if (query.has('refresh') || query.has('trees')) { + sendJson(res, 400, { error: 'start a refresh with POST /api/refresh' }); + return; + } try { const collector = await getSystem(); - // ORDER IS LOAD-BEARING: assemble the payload BEFORE starting a scan. - // The deep collectors are synchronous, so the first phase occupies the - // event loop the moment it gets a turn — and `read()` awaits, which - // hands it that turn. Starting first therefore made the *initiating* - // request wait out the phase it had just kicked off (measured: 9s), - // which is precisely the hang the progress state exists to avoid. const payload = await collector.read(); - if (query.get('refresh') === 'deep') { - // Start-or-attach and answer NOW. The collector's single flight means - // a second refresh joins the running scan rather than racing it, and - // it never rejects — the catch guards an injected collector that does - // not honour that contract, so a bad one cannot take the process down - // with an unhandled rejection. - // `trees` is a MEASUREMENT parameter, not a view filter: project - // working trees are only walked when it is set, and one large - // repository outweighs every shared cache combined — so the ranking - // has to be re-measured, not re-sorted. Absent means "keep whatever - // the collector already defaults to". - const trees = query.get('trees'); - const deepScan = Promise.resolve(collector.refreshDeep( - trees == null ? undefined : { includeProjectTrees: trees === '1' }, - )); - refreshMaintenanceAfterSystem(deepScan); - // The payload predates the start by microseconds; re-stamp the live - // scan block so this response reads "running", not "idle". - if (typeof collector.scanState === 'function') payload.scan = collector.scanState(); - } sendJson(res, 200, project(payload)); } catch (e) { sendJson(res, 503, { error: 'system footprint unavailable', reason: String(e && e.message || e) }); @@ -2024,13 +1992,11 @@ export function startDashboard({ } async function handleMaintenance(req, res, query) { - const refresh = query.getAll('refresh'); - if ([...query.keys()].some((key) => key !== 'refresh') - || refresh.length > 1 || (refresh.length === 1 && refresh[0] !== 'scan')) { - sendJson(res, 400, { error: 'invalid maintenance scan request' }); + if (query.size > 0) { + sendJson(res, 400, { error: 'start a refresh with POST /api/refresh' }); return; } - try { await (await getMaintenanceApi()).report(req, res, { refresh: query.get('refresh') === 'scan' }); } + try { await (await getMaintenanceApi()).report(req, res); } catch { sendJson(res, 503, { error: 'maintenance evidence unavailable' }); } return; } @@ -2113,6 +2079,7 @@ export function startDashboard({ // out separately into sse.mjs's sseRoute(). const ROUTES = { '/api/status': handleStatus, + '/api/refresh': (_req, res) => handleRefreshGet(res, refreshOperation), '/api/host-health': async (_req, res) => { try { sendJson(res, 200, await getHostReadiness()); } catch { sendJson(res, 503, { error: 'Host health checks unavailable.' }); } diff --git a/src/lib/dashboard/client.mjs b/src/lib/dashboard/client.mjs index 90d166ba..47c217fb 100644 --- a/src/lib/dashboard/client.mjs +++ b/src/lib/dashboard/client.mjs @@ -94,6 +94,7 @@ const datetimeSrc = readSplit('datetime.mjs'); const hostReadinessSrc = readSplit('host-readiness.mjs'); const intelligenceSrc = readSplit('intelligence.mjs'); const pollSrc = readSplit('poll.mjs'); +const refreshControlSrc = readSplit('refresh-control.mjs'); // usage-rhythm.mjs declares its OWN `esc` on disk, and its comment says why: // the tests import it as real ESM, where bootstrap.mjs's `esc` is still the // build-time stub. In the concatenated bundle every file shares ONE scope, so @@ -164,5 +165,5 @@ const bootSrc = readSplit('boot.mjs'); export const JS = ` (function(){ ${bootstrapSrc}${contextCard.toString()}${contextHostCard.toString()}${repositoryTree.toString()}${overviewSrc}${datetimeSrc}${hostReadinessSrc} -${intelligenceSrc}${pollSrc}${usageRhythmSrc}${usagePromptsSrc}${usageContextHooksSrc}${usageSrc}${modelLifecycleSrc}${usageOrchestratorsSrc}${rufloComponentsSrc}${aboutSrc}${systemReadoutSrc}${systemProjectsSrc}${maintenanceWorkspaceSrc}${maintenanceFiltersSrc}${maintenanceCardsSrc}${maintenanceOperationSrc}${maintenanceLanguageLogosSrc}${maintenanceFocusSrc}${maintenanceInventorySrc}${maintenanceRelationshipsSrc}${maintenanceInspectorSrc}${maintenanceGuidanceSrc}${maintenanceDiscoverySrc}${maintenanceActivitySrc}${systemMaintenanceActionsSrc}${systemMaintenanceSrc}${bootSrc}})(); +${intelligenceSrc}${pollSrc}${refreshControlSrc}${usageRhythmSrc}${usagePromptsSrc}${usageContextHooksSrc}${usageSrc}${modelLifecycleSrc}${usageOrchestratorsSrc}${rufloComponentsSrc}${aboutSrc}${systemReadoutSrc}${systemProjectsSrc}${maintenanceWorkspaceSrc}${maintenanceFiltersSrc}${maintenanceCardsSrc}${maintenanceOperationSrc}${maintenanceLanguageLogosSrc}${maintenanceFocusSrc}${maintenanceInventorySrc}${maintenanceRelationshipsSrc}${maintenanceInspectorSrc}${maintenanceGuidanceSrc}${maintenanceDiscoverySrc}${maintenanceActivitySrc}${systemMaintenanceActionsSrc}${systemMaintenanceSrc}${bootSrc}})(); `; diff --git a/src/lib/dashboard/client/boot.mjs b/src/lib/dashboard/client/boot.mjs index a825ccaa..7b9f6cda 100644 --- a/src/lib/dashboard/client/boot.mjs +++ b/src/lib/dashboard/client/boot.mjs @@ -6,6 +6,7 @@ import { renderAbout, wireAboutNudge } from './about.mjs'; import { activeTab, initialLiveScope, setSystemView, setTab, syncHash, systemView } from './bootstrap.mjs'; import { tickClock, wireIntelPicker } from './intelligence.mjs'; import { pollStatus, schedulePoll, wirePoll, wireStripCollapse } from './poll.mjs'; +import { wireRefresh } from './refresh-control.mjs'; import { renderSystemFreshness, wireCatalogFilters, wireSystem } from './system-projects.mjs'; import { wireMaintenance } from './system-maintenance.mjs'; import { wireUsage } from './usage-orchestrators.mjs'; @@ -22,6 +23,7 @@ import { loadUsage, setUsageView } from './usage.mjs'; renderSystemFreshness(); wireHostHealth(); wirePoll(); + wireRefresh(); wireUsage(); wireIntelPicker(); wireAboutNudge(); diff --git a/src/lib/dashboard/client/host-readiness.mjs b/src/lib/dashboard/client/host-readiness.mjs index c05aef0b..8fd5dc47 100644 --- a/src/lib/dashboard/client/host-readiness.mjs +++ b/src/lib/dashboard/client/host-readiness.mjs @@ -1,6 +1,7 @@ // @ts-nocheck — browser bundle source; assembled by ../client.mjs. import { esc, authHeaders } from './bootstrap.mjs'; import { sourceHostIcon } from './usage.mjs'; +import { startRefresh, refreshRunning } from './refresh-control.mjs'; var HEALTH_REPORT=null, HEALTH_HOST=null, HEALTH_BUSY=false, HEALTH_BUSY_HOST=null, HEALTH_ACK=null; var HEALTH_NAMES={claude:'Claude Code',codex:'Codex',opencode:'OpenCode'}; @@ -30,7 +31,9 @@ export function renderHostReadiness(report,checking){ el.hidden=false; el.innerHTML=['claude','codex','opencode'].map(function(host){ var row=report&&report.hosts&&report.hosts[host]; - var state=checking||(HEALTH_BUSY&&HEALTH_BUSY_HOST===host)?'checking':row&&HEALTH_LABELS[row.status]?row.status:'unknown'; + var configuration=row&&row.checks&&row.checks.configuration; + var unassessed=host==='claude'&&configuration&&['unknown','not-run','not-checked'].includes(configuration.state); + var state=checking||(HEALTH_BUSY&&HEALTH_BUSY_HOST===host)?'checking':unassessed?'unknown':row&&HEALTH_LABELS[row.status]?row.status:'unknown'; var managed=!row||hostIsManaged(row); // A managed host's badge is its health; any other host's badge is its // management word, in a neutral colour: its problems are information. @@ -123,7 +126,7 @@ function renderHealthDialog(){ if(!row||HEALTH_ACK!==row.evidenceKey){consent.checked=false;HEALTH_ACK=null;} consent.disabled=HEALTH_BUSY||!row||!row.canCheckConnection; document.getElementById('host-health-connect').disabled=HEALTH_BUSY||!row||!row.canCheckConnection||!consent.checked; - document.getElementById('host-health-refresh').disabled=HEALTH_BUSY; + document.getElementById('host-health-run-refresh').disabled=HEALTH_BUSY||refreshRunning(); } async function runHealthCheck(connected){ @@ -175,7 +178,7 @@ export function wireHostHealth(){ var button=region.querySelector('[data-health-host="'+HEALTH_HOST+'"]');if(button)button.focus(); }); document.getElementById('host-health-consent').addEventListener('change',function(event){HEALTH_ACK=event.target.checked&&healthRow()?healthRow().evidenceKey:null;renderHealthDialog();}); - document.getElementById('host-health-refresh').addEventListener('click',function(){runHealthCheck(false);}); + document.getElementById('host-health-run-refresh').addEventListener('click',function(){startRefresh('local');}); document.getElementById('host-health-connect').addEventListener('click',function(){runHealthCheck(true);}); dialog.addEventListener('click',function(event){ var button=event.target.closest('[data-copy]'); diff --git a/src/lib/dashboard/client/maintenance-activity.mjs b/src/lib/dashboard/client/maintenance-activity.mjs index d075d107..611351e7 100644 --- a/src/lib/dashboard/client/maintenance-activity.mjs +++ b/src/lib/dashboard/client/maintenance-activity.mjs @@ -49,19 +49,23 @@ import { beginMaintUndo } from './system-maintenance-actions.mjs'; +(entry.at?" — "+esc(mntAge(entry.at)):"")+""; } var mntExpandedHistoryDays=new Set(); + function mntScanTime(entry){ + var recorded=entry&&entry.recordedAt,completed=entry&&entry.completedAt; + return recorded&&Number.isFinite(Date.parse(recorded))?recorded:(completed&&Number.isFinite(Date.parse(completed))?completed:null); + } function renderMntScanHistory(){ var el=document.getElementById("mnt-scan-history");if(!el)return; var history=(MNT.activity&&(MNT.activity.scanHistory||MNT.activity.scans))||[]; - if(!history.length){el.innerHTML="

Scan history

No scans have completed yet.

";return;} + if(!history.length){el.innerHTML="

Scan history

No scan records yet.

";return;} var groups=new Map(); - history.slice().sort(function(a,b){return (Date.parse(b.completedAt)||0)-(Date.parse(a.completedAt)||0);}).forEach(function(entry){ - var day=formatLocalDay(entry.completedAt)||'Date not recorded'; + history.slice().sort(function(a,b){return (Date.parse(mntScanTime(b))||0)-(Date.parse(mntScanTime(a))||0);}).forEach(function(entry){ + var day=formatLocalDay(mntScanTime(entry))||'Date not recorded'; if(!groups.has(day))groups.set(day,[]); groups.get(day).push(entry); }); - el.innerHTML='

Scan history

' + el.innerHTML='

Scan history

Scanned sources grouped by local date, newest first
Date / timeSource scannedStatusEntries
' +Array.from(groups,function(group,index){var expanded=mntExpandedHistoryDays.has(group[0]);return '' - +group[1].map(function(entry){return '';}).join('')+'';}).join('')+'
Scan records grouped by local date, newest first
Date / timeSourceStatusEntries
'+esc(formatLocalTime(entry.completedAt)||'Time not recorded')+''+esc(entry.label||'Source no longer configured')+''+esc(MNT_SOURCE_COVERAGE_LABELS[entry.state]||(entry.state==='published'?'Complete':entry.state)) + +group[1].map(function(entry){return ''+esc(formatLocalTime(mntScanTime(entry))||'Time not recorded')+''+esc(entry.label||'Source no longer configured')+''+esc(MNT_SOURCE_COVERAGE_LABELS[entry.state]||(entry.state==='published'?'Complete':entry.state)) +(entry.limitingReason?''+esc(entry.limitingReason)+'':'')+''+(Number.isFinite(entry.visited)?esc(entry.visited.toLocaleString()):'—')+'
'; el.onclick=function(event){ var button=event.target.closest&&event.target.closest('[data-mnt-history-day]'); diff --git a/src/lib/dashboard/client/maintenance-discovery.mjs b/src/lib/dashboard/client/maintenance-discovery.mjs index cf6effe7..8774f6f4 100644 --- a/src/lib/dashboard/client/maintenance-discovery.mjs +++ b/src/lib/dashboard/client/maintenance-discovery.mjs @@ -144,7 +144,7 @@ import { MNT, mntAge, MNT_SOURCE_COVERAGE_LABELS, mntGet, mntPost, mntRegisterDe // optional structured per-source detail, not text — this workspace has // no use for it beyond what `coverage` already gives, so it stays unread. var narrative=(MNT.discovery&&(MNT.discovery.narrative||MNT.discovery.progress))||""; - // Automatic sources are covered by Re-measure machine (System Full scan) + // Automatic sources are covered by Refresh machine (machine measurement) // and carry no per-source controls; only roots the user added expose // Pause/Stop while running, Resume/Stop while paused, and Retry after a // failure. A user root starts scanning when it is saved (no "scan now"). @@ -170,7 +170,7 @@ import { MNT, mntAge, MNT_SOURCE_COVERAGE_LABELS, mntGet, mntPost, mntRegisterDe +(entry.limitingReason?''+esc(entry.limitingReason)+'':'') +(entry.lastCompletedAt?''+esc(mntAge(entry.lastCompletedAt))+'':'')+' '+controls+''; }).join('')+'' - +'

Evidence checks

Runtimes, package managers, Ollama, and providers are checked by Refresh evidence, separately from filesystem coverage. The toolbar reports the latest operation outcome.

' + +'

Evidence checks

Runtimes, package managers, Ollama, and providers are checked by Refresh, separately from filesystem coverage. The toolbar reports the latest operation outcome.

' +'

Project coverage

'+esc(MNT.discovery&&MNT.discovery.projectCoverageNote||'Configured project roots contribute to filesystem coverage. Machine-discovered projects appear in Inventory.')+'

'; } export function renderMntDiscovery(){ diff --git a/src/lib/dashboard/client/maintenance-guidance.mjs b/src/lib/dashboard/client/maintenance-guidance.mjs index eef6576a..5ece2538 100644 --- a/src/lib/dashboard/client/maintenance-guidance.mjs +++ b/src/lib/dashboard/client/maintenance-guidance.mjs @@ -205,7 +205,7 @@ import { beginMaintPreview, beginMaintReconcile } from './system-maintenance-act var el=document.getElementById('mnt-guidance-coverage');if(!el)return; var coverage=MNT.guidance&&MNT.guidance.coverage||[]; el.innerHTML='

Guidance shows evidence-backed issues, updates, and recovery work. Optional management actions are available in Inventory.

' - +(coverage.length?'
Host coverage

Counts reflect saved evidence, not a health assessment. Action checks cover the listed resource types in host-specific adapters. Refresh evidence to update these checks.

    '+coverage.map(function(row){ + +(coverage.length?'
    Host coverage

    Counts reflect saved evidence, not a health assessment. Action checks cover the listed resource types in host-specific adapters. Refresh to update these checks.

      '+coverage.map(function(row){ return '
    • '+esc(row.label)+' · '+esc(row.placements)+' installations · '+esc(row.recommendations)+' guidance items · '+esc(row.optionalActions)+' optional actions
      '+esc(row.actionStatusLabel) +((row.actionKinds||[]).length?' ('+esc(row.actionKinds.map(mntKindLabel).join(', '))+')':'')+'
    • '; }).join('')+'
    ':''); @@ -282,7 +282,7 @@ import { beginMaintPreview, beginMaintReconcile } from './system-maintenance-act el.querySelector("#mnt-procedure-close").focus(); }).catch(function(){ if(seq!==mntProcedureSeq||!el.open)return; - el.innerHTML='

    This procedure could not be loaded. Refresh evidence and try again.

    '; + el.innerHTML='

    This procedure could not be loaded. Refresh and try again.

    '; el.querySelector("#mnt-procedure-close").focus(); }); } diff --git a/src/lib/dashboard/client/maintenance-inspector.mjs b/src/lib/dashboard/client/maintenance-inspector.mjs index a8eac097..bfd3797f 100644 --- a/src/lib/dashboard/client/maintenance-inspector.mjs +++ b/src/lib/dashboard/client/maintenance-inspector.mjs @@ -134,7 +134,7 @@ import { mntRenderGuidanceEntry, mntWireGuidanceActions } from './maintenance-gu if(MNT.inspector&&MNT.inspector.scanRequired){ el.hidden=false; el.innerHTML=mntInspectorCloseButton()+'

    No inventory has been built yet. ' - +"Use Refresh evidence, above, to build it.

    "; + +"Use Refresh, above, to build it.

    "; return; } if(mntInspectorError||!MNT.inspector){ @@ -194,7 +194,7 @@ import { mntRenderGuidanceEntry, mntWireGuidanceActions } from './maintenance-gu mntRevealed=result;mntRevealError=null;renderMntInspector(); }).catch(function(){ if(seq!==mntInspectorSeq||placementId!==MNT.plc)return; - mntRevealError="The exact path could not be revealed. Refresh evidence and try again.";renderMntInspector(); + mntRevealError="The exact path could not be revealed. Refresh and try again.";renderMntInspector(); }); } diff --git a/src/lib/dashboard/client/maintenance-operation.mjs b/src/lib/dashboard/client/maintenance-operation.mjs index 0c9d6cb2..1ca3c0ba 100644 --- a/src/lib/dashboard/client/maintenance-operation.mjs +++ b/src/lib/dashboard/client/maintenance-operation.mjs @@ -1,11 +1,11 @@ // @ts-nocheck — dashboard browser bundle. import { MNT, mntGet, mntRefreshActiveDestination } from './maintenance-workspace.mjs'; -import { loadSystem } from './system-projects.mjs'; -import { SYSTEM, systemBusy } from './system-readout.mjs'; +import { SYSTEM } from './system-readout.mjs'; +import { refreshRunning } from './refresh-control.mjs'; var mntOperationTimer=null,mntOperationWired=false; var MNT_SCAN_PHASES={install:'Reading installed tools',storage:'Measuring retained data',catalog:'Comparing skills, plugins, and MCP servers',projects:'Measuring projects',consumers:'Ranking disk use',persist:'Saving report'}; -export function mntWritesBlocked(){return !!(MNT.providersBusy||MNT.remeasureBusy||MNT.externalScanBusy);} +export function mntWritesBlocked(){return !!(refreshRunning()||MNT.externalScanBusy);} export function mntOperationText(){return MNT.operation&&MNT.operation.message||'No measurement in progress.';} function mntElapsed(started){ var seconds=Math.max(0,Math.floor((Date.now()-started)/1000)); @@ -13,7 +13,6 @@ function mntElapsed(started){ } export function renderMntOperation(){ var op=MNT.operation,busy=mntWritesBlocked(); - ['mnt-check-providers','mnt-remeasure'].forEach(function(id){var button=document.getElementById(id);if(button)button.disabled=busy;}); var el=document.getElementById('mnt-check-providers-status'); if(el){ var message=op?op.message:''; @@ -31,7 +30,7 @@ function mntSetOperation(message){ MNT.operation.message=message;renderMntOperation(); } function mntBeginOperation(kind){ - MNT.operation={kind:kind,startedAt:Date.now(),message:kind==='measure'?'Preparing measurement…':'Refreshing evidence…',failed:false}; + MNT.operation={kind:kind,startedAt:Date.now(),message:kind==='measure'?'Preparing measurement…':'Refreshing…',failed:false}; if(mntOperationTimer)clearInterval(mntOperationTimer); mntOperationTimer=setInterval(renderMntOperation,1000); renderMntOperation();mntRefreshActiveDestination(); @@ -41,16 +40,6 @@ function mntSetMeasurement(scan){ mntSetOperation(phase+(scan.total?' · '+scan.scanned+' of '+scan.total:'')); } function mntDelay(ms){return new Promise(function(resolve){setTimeout(resolve,ms);});} -function mntPollSystemMeasurement(){ - // Publish completion through System so its scheduled poll cannot replay stale running state. - return loadSystem().then(function(){ - var data=SYSTEM; - if(!data||data.error||!data.scan)throw new Error('The measurement status could not be read.'); - if(data.scan.running){mntSetMeasurement(data.scan);return mntDelay(3000).then(mntPollSystemMeasurement);} - if(data.scan.error)throw new Error('The measurement reported a problem.'); - mntSetOperation('Machine measured · refreshing evidence…'); - }); -} export function mntBuildStatusOf(page){ var refresh=page&&page.lastRefresh; if(refresh&&(refresh.status==='running'||refresh.status==='failed'))return refresh.status; @@ -81,7 +70,7 @@ function mntPollProviders(previousCheck){ if(activity&&activity.status==='failed')throw new Error('The evidence check failed.'); var fresh=previousCheck===undefined||(scan&&scan.checkedAt!==previousCheck); if(activity&&activity.status==='running'||!fresh){ - mntSetOperation('Refreshing evidence…'); + mntSetOperation('Refreshing…'); if(Date.now()-started>300000)throw new Error('The evidence check is still pending.'); return mntDelay(2000).then(tick); } @@ -92,27 +81,6 @@ function mntPollProviders(previousCheck){ });} return tick(); } -function mntRunOperation(measure){ - if(mntWritesBlocked()||(measure&&systemBusy))return Promise.resolve(); - MNT.remeasureBusy=measure;MNT.providersBusy=!measure;mntBeginOperation(measure?'measure':'evidence'); - var previousAt,previousCheck; - return Promise.all([mntGet('/api/maintenance/v2/inventory?limit=1'),mntGet('/api/maintenance')]).then(function(before){ - previousAt=before[0].lastRefresh&&before[0].lastRefresh.at; - previousCheck=before[1].scan&&before[1].scan.checkedAt; - if(measure){if(systemBusy)throw new Error('Another system request is in progress. Please retry.');return loadSystem(true).then(function(){if(!SYSTEM||SYSTEM.error)throw new Error('The measurement could not be started.');return mntPollSystemMeasurement();});} - return mntGet('/api/maintenance?refresh=scan'); - }).then(function(){return mntPollProviders(previousCheck);}) - .then(function(){return mntAwaitInventoryBuild(previousAt);}) - .then(function(){mntSetOperation(MNT.operation.staleMeasurement?'Evidence refreshed · machine measurement is stale. Re-measure machine.':MNT.operation.gaps?'Finished with coverage gaps · see Discovery.':'Inventory updated · checks complete.');}) - .catch(function(error){MNT.operation.failed=true;mntSetOperation(error.message+' Previous inventory remains available.');}) - .then(function(){ - MNT.remeasureBusy=false;MNT.providersBusy=false; - if(mntOperationTimer){clearInterval(mntOperationTimer);mntOperationTimer=null;} - renderMntOperation();mntRefreshActiveDestination(); - }); -} -export function mntRemeasureMachine(){return mntRunOperation(true);} -export function mntCheckProviders(){return mntRunOperation(false);} function mntObserveSystemScan(scan){ if(MNT.remeasureBusy||MNT.providersBusy){ MNT.externalScanBusy=!!(scan&&scan.running); @@ -133,7 +101,7 @@ function mntObserveSystemScan(scan){ Promise.resolve(op.baseline).then(function(before){ if(scan&&scan.error)throw new Error('The measurement reported a problem.'); return mntPollProviders(before&&before.check).then(function(){return mntAwaitInventoryBuild(before&&before.at);}); - }).then(function(){mntSetOperation(op.staleMeasurement?'Evidence refreshed · machine measurement is stale. Re-measure machine.':op.gaps?'Finished with coverage gaps · see Discovery.':'Inventory updated · checks complete.');}) + }).then(function(){mntSetOperation(op.staleMeasurement?'Evidence refreshed · machine measurement is stale. Refresh machine.':op.gaps?'Finished with coverage gaps · see Discovery.':'Inventory updated · checks complete.');}) .catch(function(error){op.failed=true;mntSetOperation(error.message+' Previous inventory remains available.');}) .then(function(){MNT.externalScanBusy=false;clearInterval(mntOperationTimer);mntOperationTimer=null;renderMntOperation();mntRefreshActiveDestination();}); } diff --git a/src/lib/dashboard/client/maintenance-workspace.mjs b/src/lib/dashboard/client/maintenance-workspace.mjs index 1ccfa42b..830f7905 100644 --- a/src/lib/dashboard/client/maintenance-workspace.mjs +++ b/src/lib/dashboard/client/maintenance-workspace.mjs @@ -10,7 +10,7 @@ // a node module — tests/kit/maintenance-dashboard-client-labels.test.mjs // asserts these literal maps stay byte-identical to that contract). import { authHeaders, esc } from './bootstrap.mjs'; -import { mntCheckProviders, mntRemeasureMachine, mntWireOperation } from './maintenance-operation.mjs'; +import { mntWireOperation } from './maintenance-operation.mjs'; import { ago } from './intelligence.mjs'; // ── Label vocabulary (copied from src/lib/maintenance/management/model.mjs) ─ @@ -186,7 +186,7 @@ import { ago } from './intelligence.mjs'; if(lastRefresh&&lastRefresh.status==="running"){ return '
    Building the inventory…
    '; } - return '
    No inventory has been built yet. Use Refresh evidence, above, to build it.
    '; + return '
    No inventory has been built yet. Use Refresh, above, to build it.
    '; } export function mntScanRequiredAnnouncement(lastRefresh){ if(lastRefresh&&lastRefresh.status==="failed"){ @@ -241,8 +241,16 @@ import { ago } from './intelligence.mjs'; }; } + var mntLastSyncedHash=null; export function mntSyncHash(){ - try{if(history.replaceState)history.replaceState(null,"",mntHash());}catch(e){} + var current=String(location.hash||""); + if(current&&!/^#system\/(?:maintenance|catalog)(?:\/|$)/.test(current))return; + if(mntLastSyncedHash!==null&¤t!==mntLastSyncedHash){ + var state=mntApplyHashState(); + if(state&&state.hasState)mntApplyState(state); + } + var next=mntHash(); + try{if(history.replaceState)history.replaceState(null,"",next);mntLastSyncedHash=next;}catch(e){} } // ── Preferences (owner-private; URL state overrides it on load, MNT-PRV-006) ─ @@ -434,8 +442,4 @@ import { ago } from './intelligence.mjs'; MNT.wired=true; mntWireTabs();mntWireOperation(); document.addEventListener("keydown",mntHandleEscape); - var checkProviders=document.getElementById("mnt-check-providers"); - if(checkProviders)checkProviders.addEventListener("click",mntCheckProviders); - var remeasure=document.getElementById("mnt-remeasure"); - if(remeasure)remeasure.addEventListener("click",mntRemeasureMachine); } diff --git a/src/lib/dashboard/client/poll.mjs b/src/lib/dashboard/client/poll.mjs index 9aed4904..1c161dec 100644 --- a/src/lib/dashboard/client/poll.mjs +++ b/src/lib/dashboard/client/poll.mjs @@ -13,7 +13,7 @@ import { loadModelLifecycle, loadUsage } from './usage.mjs'; // Governs EVERY tab, not just Usage (ADR-0009 §7). The old hardcoded 5 s poll // predated any expensive view; 30 s is the default now, and the whole range is // user-chosen and persisted. Every refresh path — automatic or manual — funnels - // through refreshAll(), so the single-flight guard and the cooldown are + // through reloadView(), so the single-flight guard and the cooldown are // impossible to route around. var LS_POLL="ak-dash-poll"; var POLL_DEFAULT_MS=30000; @@ -21,6 +21,7 @@ import { loadModelLifecycle, loadUsage } from './usage.mjs'; var POLL_LABEL={15000:"15s",30000:"30s",60000:"1m",300000:"5m",900000:"15m", 1800000:"30m",3600000:"1h",21600000:"6h",43200000:"12h",86400000:"24h"}; export var pollOn=true, pollMs=POLL_DEFAULT_MS, pollTimer=null, inflight=false, lastAttempt=0; + var lastManualAttempt=0; try{ var savedPoll=JSON.parse(localStorage.getItem(LS_POLL)||"null"); @@ -128,11 +129,17 @@ import { loadModelLifecycle, loadUsage } from './usage.mjs'; }); } - function refreshAll(){ + export function reloadView(force){ + // A deliberate Reload should not be lost to a recent background tick. + // Still coalesce a double-click before joining an in-flight read. + if(force==="manual"){ + if(Date.now()-lastManualAttempt0 + &&Number.isFinite(Date.parse(state.startedAt)) + &&typeof state.running==='boolean'&&Array.isArray(state.stages) + &&(state.running||typeof state.ok==='boolean'); +} +function refreshText(state){ + var stages=state&&state.stages||[]; + var active=stages.find(function(stage){return stage.state==='running';}); + var latest=active||stages[stages.length-1]; + var elapsed=Math.max(0,Math.floor((Date.now()-refreshStartedAt)/1000)); + if(state&&state.running)return (latest&&latest.label||'Preparing refresh')+' · '+elapsed+'s'; + return state&&state.ok?'Refresh complete.':state?'Refresh did not complete.':''; +} +function renderRefresh(state,message){ + var button=document.getElementById('refresh-run'),status=document.getElementById('refresh-status'); + if(button)button.disabled=refreshBusy; + if(status)status.textContent=message||refreshText(state); + document.querySelectorAll('[data-mnt-plan-plc], [data-mnt-undo-receipt], [data-mnt-reconcile-receipt]').forEach(function(action){ + if(refreshBusy){ + if(action.dataset.refreshWasDisabled===undefined)action.dataset.refreshWasDisabled=action.disabled?'1':'0'; + action.disabled=true; + }else if(action.dataset.refreshWasDisabled!==undefined){ + action.disabled=action.dataset.refreshWasDisabled==='1'; + delete action.dataset.refreshWasDisabled; + } + }); + if(refreshRenderedBusy!==refreshBusy){refreshRenderedBusy=refreshBusy;if(refreshBusy)mntRefreshActiveDestination();} +} +function refreshPoll(){ + if(!refreshBusy)return; + fetch('/api/refresh',{cache:'no-store',headers:authHeaders()}).then(function(response){ + if(!response.ok)throw new Error('Refresh status unavailable.'); + return response.json(); + }).then(function(state){ + if(!validRefreshState(state))throw new Error('Refresh status was incomplete.'); + if(refreshOperationId===null&&refreshRequireNewer&&Date.parse(state.startedAt) ({ kind: 'dict', value, max, keyPrefix: prefix }), either: (...options) => ({ kind: 'either', options }), /** ISO timestamp, bounded machine token, and a user-facing label that refuses PROHIBITED_LABELS. */ - iso: Object.freeze({ kind: 'iso' }), token: (max = 64) => ({ kind: 'token', max }), label: (max = 200) => ({ kind: 'label', max }), + iso: Object.freeze({ kind: 'iso' }), scanStamp: Object.freeze({ kind: 'scanStamp' }), token: (max = 64) => ({ kind: 'token', max }), label: (max = 200) => ({ kind: 'label', max }), }); const [DICT_KEY, MAX_PAGE_ROWS, TOKEN] = [/^[A-Za-z0-9._:-]{1,120}$/, 200, /^[A-Za-z0-9._:-]+$/]; @@ -430,6 +431,7 @@ const SCALARS = Object.freeze({ label: (node, value) => { const safe = evidenceText(value, node.max); return safe && !isProhibitedLabel(safe) ? safe : undefined; }, token: (node, value) => (typeof value === 'string' && value.length <= node.max && TOKEN.test(value) ? value : undefined), iso: (_node, value) => (typeof value === 'string' && value.length <= 40 && Number.isFinite(Date.parse(value)) ? value : undefined), + scanStamp: (_node, value) => validScanTime(value) ?? undefined, owner: (node, value) => text(value, node.max) ?? undefined, bool: (_node, value) => (typeof value === 'boolean' ? value : undefined), int: (_node, value) => (Number.isInteger(value) ? value : undefined), @@ -585,7 +587,7 @@ const CONFIGURED_SOURCE = T.obj({ sourceId: ID, kind: T.oneOf(SOURCE_TYPES), root: T.owner(1024), label: LABEL, maxDepth: T.int, includeNetwork: T.bool, present: T.bool, }); const SCAN_SUMMARY = T.obj({ - scanId: ID, sourceId: ID, environmentId: ID, state: T.oneOf(SCAN_STATES), startedAt: STAMP, completedAt: STAMP, visited: T.int, + scanId: ID, sourceId: ID, environmentId: ID, state: T.oneOf(SCAN_STATES), startedAt: STAMP, recordedAt: T.scanStamp, completedAt: STAMP, visited: T.int, limitingReason: T.oneOf(LIMITING_REASONS), ceiling: T.oneOf(SAFETY_CEILINGS), label: LABEL, }); const DISCOVERY = T.obj({ @@ -837,12 +839,10 @@ function reconcileConfirmation({ receiptId, outcome, audit, now }) { * now?: () => number, * capabilities?: ReturnType, * scanAckMs?: number, - * afterScan?: () => any, - * }} options `afterScan` runs (never awaited) after a successful `?refresh=scan` - * provider scan so the server can chain the inventory rebuild. + * }} options */ export function createMaintenanceDashboardApi({ - service, management = null, sessionToken, now = Date.now, capabilities, scanAckMs = 250, afterScan = null, + service, management = null, sessionToken, now = Date.now, capabilities, scanAckMs = 250, } = {}) { if (!service || typeof service.report !== 'function' || typeof service.scan !== 'function' || typeof service.plan !== 'function') { @@ -860,11 +860,9 @@ export function createMaintenanceDashboardApi({ const withActivity = (model) => ({ ...model, activity: typeof service.scanState === 'function' ? service.scanState() : null }); - async function report(_req, res, { refresh = false } = {}) { + async function report(_req, res) { try { - const model = await (refresh ? service.scan() : service.report()); - // The chained inventory rebuild is fire-and-forget: the scan response stands on its own. - if (refresh && typeof afterScan === 'function') { try { Promise.resolve(afterScan()).catch(() => {}); } catch { /* ignored */ } } + const model = await service.report(); sendJson(res, 200, publicMaintenanceModel(withActivity(model))); } catch (error) { diff --git a/src/lib/dashboard/page.mjs b/src/lib/dashboard/page.mjs index f0abcaa2..8ad12e06 100644 --- a/src/lib/dashboard/page.mjs +++ b/src/lib/dashboard/page.mjs @@ -114,7 +114,14 @@ export function renderPage({ name, version }) {
    - + +
    +
    + + + + +
+

@@ -219,8 +226,7 @@ export function renderPage({ name, version }) {
- full scan — not run yet - + machine measurement — not run yet
@@ -916,10 +922,6 @@ ${LIVE_HTML}

- - Runs provider probes on the saved measurement and rebuilds the inventory. Seconds. - - Walks the filesystem, then refreshes evidence. Minutes.
diff --git a/src/lib/dashboard/refresh-api.mjs b/src/lib/dashboard/refresh-api.mjs new file mode 100644 index 00000000..f588cf74 --- /dev/null +++ b/src/lib/dashboard/refresh-api.mjs @@ -0,0 +1,97 @@ +import { randomUUID } from 'node:crypto'; +import { REFRESH_STRENGTHS, runRefresh as sharedRunRefresh } from '../refresh.mjs'; +import { sendJson } from '../loopback-server.mjs'; +import { readMaintenanceJson } from './maintenance-security.mjs'; + +export const REFRESH_POST_ROUTES = new Set(['/api/refresh']); +export const REFRESH_LOCAL_TIMEOUT_MS = 5 * 60_000; + +/** One operation per dashboard server. Its public state never exposes stage results. */ +export function createRefreshOperation({ stages, runRefresh = sharedRunRefresh }) { + let current = null; + const state = () => current ? { ...current, stages: current.stages.map(stage => ({ ...stage })) } + : { running: false, lastRun: null }; + function start({ strength, projectTrees = false }) { + if (current?.running) return null; + current = { operationId: randomUUID(), running: true, strength, projectTrees, startedAt: new Date().toISOString(), + finishedAt: null, ok: null, stages: [] }; + Promise.resolve().then(() => runRefresh({ strength, projectTrees, stages, + onStage(event) { + const index = current.stages.findIndex(stage => stage.id === event.id); + if (index < 0) current.stages.push(event); + else current.stages[index] = event; + }, + })).then(outcome => { + current.ok = outcome.ok; + current.stages = outcome.stages.map(({ id, label, state: stageState, detail, elapsedMs }) => + ({ id, label, state: stageState, detail, elapsedMs })); + }).catch(error => { + current.ok = false; + current.stages.push({ id: 'refresh', label: 'Refresh', state: 'failed', + detail: error?.message ?? String(error), elapsedMs: 0 }); + }).finally(() => { current.running = false; current.finishedAt = new Date().toISOString(); }); + return state(); + } + return { start, state }; +} + +/** Dashboard collaborators are already memoized by startDashboard. */ +/** @param {{ cwd: string, pkgRoot?: string, getSystem: Function, getMaintenance: Function, + * refreshInventoryAfterProviderScan: Function, getHostReadiness: Function, + * statusCollect: Function, loadConfig: Function }} options */ +export function dashboardRefreshStages({ cwd, getSystem, getMaintenance, refreshInventoryAfterProviderScan, + getHostReadiness, statusCollect, loadConfig }) { + return { + async machine({ projectTrees }) { + const result = await (await getSystem()).refreshDeep({ includeProjectTrees: projectTrees === true }); + const ok = result?.ok === true && result.persisted?.ok !== false; + return { ok, detail: ok ? null : result?.error ?? 'the measurement did not finish' }; + }, + async maintenance() { + const model = await (await getMaintenance()).scan({ deep: false }); + const { providersChecked, providersTotal } = model?.scan ?? {}; + const detail = Number.isInteger(providersChecked) && Number.isInteger(providersTotal) + ? `checked ${providersChecked} of ${providersTotal} providers` : null; + return { ok: true, detail }; + }, + async inventory({ strength }) { + const result = await refreshInventoryAfterProviderScan({ measured: strength === 'machine' }); + return { ok: result != null, detail: result == null ? 'the inventory rebuild did not finish' : null }; + }, + async live() { + const { runLiveChecks } = await import('../live-checks.mjs'); + const results = await runLiveChecks({ cfg: loadConfig(), cwd }); + const counts = new Map(); + for (const { status } of results) counts.set(status, (counts.get(status) ?? 0) + 1); + return { ok: true, detail: results.length ? [...counts].map(([status, n]) => `${n} ${status}`).join(', ') + : 'no live check applies' }; + }, + async local() { + const status = await statusCollect(); + await getHostReadiness({ force: true }); + return { ok: !status?.error, detail: status?.error ?? null }; + }, + }; +} + +/** Authentication and same-origin policy are enforced by the server gate. */ +export async function handleRefreshPost(req, res, operation) { + let body; + try { body = await readMaintenanceJson(req, { maxBytes: 4096 }); } + catch (error) { + const code = error.status ?? error.statusCode; + sendJson(res, [400, 413, 415].includes(code) ? code : 400, { error: 'invalid refresh request' }); + return; + } + if (!body || Object.keys(body).some(key => !['strength', 'projectTrees'].includes(key)) + || !REFRESH_STRENGTHS.includes(body.strength) + || (body.projectTrees !== undefined && typeof body.projectTrees !== 'boolean') + || (body.projectTrees !== undefined && body.strength !== 'machine')) { + sendJson(res, 400, { error: 'invalid refresh request' }); return; + } + const state = operation.start(body); + if (!state) sendJson(res, 409, { error: 'a refresh is already running', state: operation.state() }); + else sendJson(res, 202, { started: true, state }); +} + +export function handleRefreshGet(res, operation) { sendJson(res, 200, operation.state()); } diff --git a/src/lib/dashboard/styles/base.mjs b/src/lib/dashboard/styles/base.mjs index b9127c46..71bd47b3 100644 --- a/src/lib/dashboard/styles/base.mjs +++ b/src/lib/dashboard/styles/base.mjs @@ -115,7 +115,7 @@ body.gated .band,body.gated .tabbar,body.gated main{display:none} background:var(--panel); } .verdict-text{font-size:13px; font-weight:500; letter-spacing:-.006em} -.band-tools{display:flex; align-items:center; gap:10px} +.band-tools{display:flex; align-items:center; gap:10px;flex-wrap:wrap;max-width:100%} .pulse{ width:8px; height:8px; border-radius:50%; background:var(--accent); flex:none; animation:pulse 2.4s ease-out infinite; @@ -148,7 +148,15 @@ body.gated .band,body.gated .tabbar,body.gated main{display:none} .poll .play.on{color:var(--accent)} .poll .ivl{min-width:56px; justify-content:space-between} .poll .caret{opacity:.5} -.poll .refresh{width:28px; padding:0; font-size:14px} +.poll .refresh{padding:0 8px; font-size:12px} +.refresh-control{display:flex;align-items:center;gap:5px;min-width:0;flex-wrap:wrap} +.refresh-control button,.refresh-control select{border:1px solid var(--line);border-radius:8px;background:var(--panel);color:var(--ink);font:inherit;font-size:11px;padding:5px 8px} +.refresh-control button{cursor:pointer;font-weight:700} +.refresh-control button:disabled{opacity:.5;cursor:wait} +.refresh-control :focus-visible{outline:2px solid var(--accent);outline-offset:2px} +.refresh-trees{display:flex;align-items:center;gap:3px;font-size:10px;white-space:nowrap} +#refresh-status{font-size:10px;color:var(--ink-2);min-width:0} +@media(max-width:560px){.band-tools{width:100%;gap:6px}.refresh-control{width:100%}.refresh-trees{white-space:normal}} .poll .refresh.spin{animation:spin .6s linear} @keyframes spin{to{transform:rotate(360deg)}} .menu{ diff --git a/src/lib/dashboard/styles/maintenance.mjs b/src/lib/dashboard/styles/maintenance.mjs index 4c5e1c69..8976c3dd 100644 --- a/src/lib/dashboard/styles/maintenance.mjs +++ b/src/lib/dashboard/styles/maintenance.mjs @@ -349,7 +349,6 @@ export const MAINTENANCE_CSS = ` } /* The Maintenance toolbar owns measurement actions in this destination. */ -body:has(#panel-sys-maintenance:not([hidden])) #sys-rescan{display:none} .mnt-project-section{list-style:none;margin:0 0 24px} .mnt-project-section>h4{display:flex;align-items:center;gap:8px;font-size:14px;font-weight:600;margin:8px 0 12px;color:var(--ink)} .mnt-inspector:not([hidden]){animation:mnt-inspector-enter .16s ease-out} diff --git a/src/lib/dashboard/styles/usage.mjs b/src/lib/dashboard/styles/usage.mjs index f0b96990..47e0d22d 100644 --- a/src/lib/dashboard/styles/usage.mjs +++ b/src/lib/dashboard/styles/usage.mjs @@ -173,6 +173,7 @@ export const USAGE_CSS = ` width:32px; height:32px; display:grid; place-items:center; border-radius:50%; background:var(--bg); border:1px solid var(--line); } +.tabbar .source-pill .live-host[data-host=codex]{color:var(--ink)} .tabbar .source-pill .live-host-icon{width:20px; height:20px; stroke-width:1.8} .source-pill .sp-status{ display:flex; align-items:center; padding:5px 14px 5px 10px; font-size:13px; diff --git a/src/lib/live/live-sessions-service.mjs b/src/lib/live/live-sessions-service.mjs index 0e22fc24..470141d0 100644 --- a/src/lib/live/live-sessions-service.mjs +++ b/src/lib/live/live-sessions-service.mjs @@ -57,6 +57,10 @@ export class LiveSessionsService { #projection = emptyLiveProjection(); #tailers = new Map(); #contexts = new Map(); + // Keep a bounded set of displaced native readers with their byte offsets + // and partial-line state. Re-entry within this retention window must not + // replay already counted records. + #dormantTailers = new Map(); #timer = null; #started = false; #edgeKeys = new Set(); @@ -140,6 +144,7 @@ export class LiveSessionsService { if (this.#timer != null) this.#options.clearInterval(this.#timer); this.#timer = null; for (const tailer of this.#tailers.values()) tailer.close(); + for (const { tailer } of this.#dormantTailers.values()) tailer.close(); this.#runtimeBindings.clear(); this.#historyPages.clear(); this.#started = false; @@ -253,9 +258,14 @@ export class LiveSessionsService { const desiredNative = new Set([...claude, ...codex]); for (const [file, context] of this.#contexts) { if (!['claude', 'codex'].includes(context.adapter) || desiredNative.has(file)) continue; - this.#tailers.get(file)?.close(); + const tailer = this.#tailers.get(file); + tailer?.close(); + if (tailer) this.#dormantTailers.set(file, { tailer, context }); this.#tailers.delete(file); this.#contexts.delete(file); + while (this.#dormantTailers.size > Math.max(2, this.#options.maxFiles * 2)) { + this.#dormantTailers.delete(this.#dormantTailers.keys().next().value); + } } for (const file of claude) { this.#add(file, { @@ -272,6 +282,13 @@ export class LiveSessionsService { #add(file, context, initial) { if (this.#tailers.has(file) || this.#tailers.size >= this.#options.maxFiles) return; + const dormant = this.#dormantTailers.get(file); + if (dormant) { + this.#dormantTailers.delete(file); + this.#tailers.set(file, dormant.tailer); + this.#contexts.set(file, dormant.context); + return; + } if (initial && ['claude', 'codex'].includes(context.adapter)) { // One shared bootstrap context so metadata learned early (codex // session_meta id/meta, project, model, provider) persists across the diff --git a/src/lib/maintenance/management/activity.mjs b/src/lib/maintenance/management/activity.mjs index 7c4093e8..531a1287 100644 --- a/src/lib/maintenance/management/activity.mjs +++ b/src/lib/maintenance/management/activity.mjs @@ -71,14 +71,26 @@ function recipeEventSummary(event) { return { kind: event.kind, recipeId: event.recipeId ?? null, recipeVersion: event.recipeVersion ?? null, at: event.at, pendingIds: event.pendingIds ?? undefined }; } +export function validScanTime(value) { + if (typeof value !== 'string' || value.length > 40) return null; + const parts = /^(\d{4})-(\d{2})-(\d{2})T(\d{2}):(\d{2}):(\d{2})(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})$/.exec(value); + if (!parts) return null; + const [, yearText, monthText, dayText, hourText, minuteText, secondText] = parts; + const [year, month, day, hour, minute, second] = [yearText, monthText, dayText, hourText, minuteText, secondText].map(Number); + const leap = year % 4 === 0 && (year % 100 !== 0 || year % 400 === 0); + const monthDays = [31, leap ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31]; + return day >= 1 && day <= monthDays[month - 1] && hour <= 23 && minute <= 59 && second <= 59 + && Number.isFinite(Date.parse(value)) ? value : null; +} + function scanSummary(entry) { if (!SCAN_STATES.includes(entry.state)) throw new TypeError(`unknown scan state: ${entry.state}`); - return { sourceId: entry.sourceId, environmentId: entry.environmentId, state: entry.state, label: entry.label, visited: entry.visited, limitingReason: entry.limitingReason ?? null, completedAt: entry.completedAt ?? null }; + return { sourceId: entry.sourceId, environmentId: entry.environmentId, state: entry.state, label: entry.label, visited: entry.visited, limitingReason: entry.limitingReason ?? null, recordedAt: validScanTime(entry.recordedAt), completedAt: validScanTime(entry.completedAt) }; } function latestScans(scanHistory) { const latest = new Map(); - const timestamp = (entry) => Date.parse(entry.completedAt) || 0; + const timestamp = (entry) => Date.parse(entry.recordedAt ?? entry.completedAt) || 0; for (const entry of scanHistory.map(scanSummary)) { const key = JSON.stringify([entry.environmentId, entry.sourceId]); const previous = latest.get(key); diff --git a/src/lib/maintenance/management/service-discovery.mjs b/src/lib/maintenance/management/service-discovery.mjs index 4339e33b..78e94538 100644 --- a/src/lib/maintenance/management/service-discovery.mjs +++ b/src/lib/maintenance/management/service-discovery.mjs @@ -391,7 +391,7 @@ export function scanProgress(ctx) { coverage, progress: ctx.orchestrator().progress(), narrative: progressNarrative(filesystemCoverage, { totalSources: filesystemCoverage.length }), - evidenceChecks: coverage.filter((entry) => entry.filesystem === false).map((entry) => ({ sourceId: entry.sourceId, label: entry.label, method: 'Refresh evidence' })), + evidenceChecks: coverage.filter((entry) => entry.filesystem === false).map((entry) => ({ sourceId: entry.sourceId, label: entry.label, method: 'Refresh' })), forbiddenClaims: claimsAllowed(filesystemCoverage), }; }; diff --git a/src/lib/maintenance/service.mjs b/src/lib/maintenance/service.mjs index b0228f86..1c00ba59 100644 --- a/src/lib/maintenance/service.mjs +++ b/src/lib/maintenance/service.mjs @@ -58,7 +58,7 @@ function scanRequiredModel({ status = 'not-scanned', now = Date.now } = {}) { observedUsage: { status: 'not-measured', statement: 'Usage evidence is unavailable.' }, consumerHosts: { basis: 'not-measured', hosts: [], count: 0, truncated: false }, impact: { summary: 'Scanning changes no installed resource.', bytes: null, files: null, dependencies: 'unknown', preserved: ['All installed resources'] }, - nextAction: { operation: 'scan', label, providerId: 'system.deep-scan', providerVersion: '1', safetyClass: 'never-automatic', rollback: 'reversible', restart: 'not-required', executable: false, recommendation: label, steps: ['Run `ak maintain --refresh=machine` to measure the machine.', 'Return to Maintenance when the scan completes.'], preserved: ['All installed resources'], blockedReason: 'A current saved scan is required before Maintenance can recommend changes.' }, + nextAction: { operation: 'scan', label, providerId: 'system.deep-scan', providerVersion: '1', safetyClass: 'never-automatic', rollback: 'reversible', restart: 'not-required', executable: false, recommendation: label, steps: ['Run `ak maintain --refresh=machine` or Refresh › Machine in the dashboard to measure the machine.', 'Return to Maintenance when the scan completes.'], preserved: ['All installed resources'], blockedReason: 'A current saved scan is required before Maintenance can recommend changes.' }, }; return deepFreeze({ schemaVersion: 1, mode: 'control-plane', capabilities: NO_CONTROL_CAPABILITIES, diff --git a/tests/dashboard.test.cjs b/tests/dashboard.test.cjs index 016c3d18..5eab2a07 100644 --- a/tests/dashboard.test.cjs +++ b/tests/dashboard.test.cjs @@ -1127,44 +1127,20 @@ async function main() { await sysSrv.close(); } - await test('?refresh=deep is single-flight — two concurrent refreshes share one scan', async () => { - let release; - const gate = new Promise((resolve) => { release = resolve; }); - // Gate the CHEAP tier, not the deep one. Both requests then resume from the - // same promise in one microtask drain, and runDeep's first act is a - // setImmediate — so the second request PROVABLY reaches refreshDeep() while - // the first still holds the slot. Racing two bare HTTP requests would be - // testing the scheduler, not the single-flight rule. - const fx = systemFixture({ collectors: { runtime: async () => { await gate; return runtimeCensus(); } } }); - let maintenanceScans = 0; - const maintenance = { - async report() { return {}; }, - async scan() { maintenanceScans += 1; return {}; }, - async plan() { return {}; }, - }; + await test('GET /api/system rejects scan queries before reading the collector', async () => { + const fx = systemFixture(); const srv = await startDashboard({ port: 0, cwd: fixture, fetchStatus: async () => STUB_STATUS, usage: spyUsage().api, - system: fx.collector, maintenance, + system: fx.collector, }); try { - const both = Promise.all([ - get(srv.url + 'api/system?refresh=deep', srv.token), - get(srv.url + 'api/system?refresh=deep', srv.token), - ]); - await eventually(() => fx.calls.runtime === 2, 'both refreshes must reach the collector'); - release(); - const [a, b] = await both; - assert(a.status === 200 && b.status === 200, 'both refreshes must answer 200'); - const scanA = JSON.parse(a.body).scan; - assert(scanA.running === true && scanA.phase !== 'idle', - 'a refresh must report the scan it started, got ' + JSON.stringify(scanA)); - await eventually(() => fx.calls.persist === 1, 'the shared scan must run to completion'); - await eventually(() => maintenanceScans === 1, 'the completed scan must refresh Maintenance evidence once'); - assert(fx.calls.install === 1 && fx.calls.storage === 1 - && fx.calls.catalog === 1 && fx.calls.projects === 1, - 'the deep collectors ran twice — the single-flight slot did not hold: ' + JSON.stringify(fx.calls)); - assert(fx.calls.persist === 1, 'a shared scan must write exactly one snapshot'); - assert(maintenanceScans === 1, 'two attached System requests must not double-run Maintenance providers'); + for (const query of ['refresh=deep', 'refresh=deep&refresh=scan', 'trees=1']) { + const response = await get(srv.url + 'api/system?' + query, srv.token); + assert(response.status === 400, 'scan query must be rejected: ' + query); + contains(response.body, 'start a refresh with POST /api/refresh'); + } + assert(fx.calls.runtime === 0 && fx.calls.persist === 0, + 'rejected GET queries must not run collectors or persist measurements'); } finally { await srv.close(); } @@ -1411,13 +1387,14 @@ async function main() { contains(r.body, 'POLL_COOLDOWN_MS=3000'); }); - await test('every refresh path is single-flight + cooldown guarded', async () => { + await test('Refresh control guards duplicate POSTs and status polling retains its cooldown', async () => { const r = await get(uiSrv.url); - const fn = r.body.slice(r.body.indexOf('function refreshAll(')); - const body = fn.slice(0, fn.indexOf('\n function ')); - assert(/inflight/.test(body), 'refreshAll must consult the single-flight flag'); - assert(/POLL_COOLDOWN_MS/.test(body), 'refreshAll must consult the cooldown'); - assert(/setInterval\(refreshAll/.test(r.body), 'the automatic poll must go through the SAME guarded path'); + const fn = r.body.slice(r.body.indexOf('function startRefresh(')); + const body = fn.slice(0, fn.indexOf('function refreshProjectTrees')); + contains(body, 'if(refreshBusy)return Promise.resolve(false)'); + contains(body, "fetch('/api/refresh',{method:'POST'"); + contains(r.body, 'POLL_COOLDOWN_MS=3000'); + assert(!/setInterval\(refreshAll/.test(r.body), 'retired automatic refresh path remains'); }); await test('the Usage tab is lazy — the shared status poll never fetches /api/usage', async () => { diff --git a/tests/fixtures/dashboard-status-child.mjs b/tests/fixtures/dashboard-status-child.mjs index 435ff3ce..9d4a4b51 100644 --- a/tests/fixtures/dashboard-status-child.mjs +++ b/tests/fixtures/dashboard-status-child.mjs @@ -10,9 +10,12 @@ // real HTTP — this script only starts the server and reports where it is // listening; the parent marks ledger call boundaries itself by appending // directly to the same ndjson file between requests. -import { startDashboard } from '../../src/lib/dashboard-server.mjs'; +// An optional test-only global root makes the Ruflo component path deterministic. +import { _setGlobalRootForTest } from '../../src/lib/paths.mjs'; -const [, , cwd] = process.argv; +const [, , cwd, fakeGlobalRoot] = process.argv; +if (fakeGlobalRoot) _setGlobalRootForTest(fakeGlobalRoot); +const { startDashboard } = await import('../../src/lib/dashboard-server.mjs'); const { port, token } = await startDashboard({ port: 0, cwd }); // Unbuffered, single line, parsed by the parent — printed only once the diff --git a/tests/kit/dashboard-get-is-read-only.test.mjs b/tests/kit/dashboard-get-is-read-only.test.mjs new file mode 100644 index 00000000..fa0f7769 --- /dev/null +++ b/tests/kit/dashboard-get-is-read-only.test.mjs @@ -0,0 +1,92 @@ +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import http from 'node:http'; +import { sandboxHome, sandboxProject, rmrf } from './helpers/home-sandbox.mjs'; + +const home = sandboxHome('ak-dashboard-read-only'); +const project = sandboxProject('ak-dashboard-read-only'); +after(() => rmrf(home, project)); +const { startDashboard } = await import('../../src/lib/dashboard-server.mjs'); + +const EXACT = [ + '/api/status', '/api/refresh', '/api/host-health', '/api/live', '/api/live/history', + '/api/live/events', '/api/live/intelligence', '/api/models', '/api/ruflo-components', + '/api/usage', '/api/hooks', '/api/limits', '/api/system', '/api/system/summary', + '/api/maintenance', '/api/sessions', +]; +const PARAMETERIZED = [ + '/api/maintenance/v2/inventory', '/api/hooks/source/bad', + '/api/live/playback/bad/bad', '/api/live/transcripts/bad/bad/events', '/api/session/bad', +]; +const ERROR = { error: 'start a refresh with POST /api/refresh' }; + +function request(server, route, method = 'GET') { + return new Promise((resolve, reject) => { + const req = http.request(new URL(route, server.url), { method, headers: { + 'x-dash-token': server.token, origin: server.url.replace(/\/$/, ''), + 'sec-fetch-site': 'same-origin', 'content-type': 'application/json', + } }, res => { + // SSE endpoints intentionally stay open. Their response headers are enough + // to prove that route dispatch has happened; close the reader promptly. + if (res.headers['content-type']?.includes('text/event-stream')) { + res.destroy(); resolve({ status: res.statusCode, body: '' }); return; + } + let body = ''; + res.on('data', chunk => { body += chunk; }); + res.on('end', () => resolve({ status: res.statusCode, body })); + }); + req.on('error', reject); + req.setTimeout(2000, () => req.destroy(new Error(`timed out: ${route}`))); + req.end(method === 'POST' ? '{}' : undefined); + }); +} + +test('GET route inventory stays aligned with the server dispatch table', () => { + const source = fs.readFileSync(new URL('../../src/lib/dashboard-server.mjs', import.meta.url), 'utf8'); + const exact = source.match(/const ROUTES = \{([\s\S]*?)\n {4}\};/)?.[1] ?? ''; + const parameterized = source.match(/const PARAM_ROUTES = \[([\s\S]*?)\n {4}\];/)?.[1] ?? ''; + assert.deepEqual([...exact.matchAll(/^ {6}'([^']+)':/gm)].map(match => match[1]), EXACT); + assert.deepEqual([...parameterized.matchAll(/^ {6}\[(\/.+?\/),/gm)].map(match => match[1]), [ + String.raw`/^\/api\/maintenance\/v2\/.*$/`, + String.raw`/^\/api\/hooks\/source\/([^/]+)$/`, + String.raw`/^\/api\/live\/playback\/([^/]+)\/([^/]+)$/`, + String.raw`/^\/api\/live\/transcripts\/([^/]+)\/([^/]+)\/events$/`, + String.raw`/^\/api\/session\/(.*)$/`, + ]); +}); + +test('all GET routes, including root, cannot start measurement or provider work through scan queries', async t => { + const calls = { system: 0, maintenance: 0, inventory: 0, rebuild: 0, provider: 0 }; + const hostReadiness = async () => ({ hosts: {} }); + hostReadiness.checkConnection = async () => { calls.provider++; return {}; }; + const server = await startDashboard({ port: 0, cwd: project, + fetchStatus: async () => ({ overall: 'ok', rows: [], drift: [] }), + hostReadiness, discoverProjects: () => [], + system: { read: async () => ({ runtime: {}, knownFiles: [], storage: {}, projects: [], scan: {} }), + refreshDeep: async () => { calls.system++; return { ok: true }; } }, + maintenance: { report: async () => ({}), scan: async () => { calls.maintenance++; return {}; }, + plan: async () => ({}) }, + management: { refreshInventory: async () => { calls.inventory++; }, + rebuildAfterMeasurement: async () => { calls.rebuild++; } }, + usage: { readIndex: async () => ({ sessions: [] }), readSession: async () => null, + masker: async () => value => value }, live: { snapshot: async () => ({}), replay: async () => ({ events: [] }), + subscribe: () => () => {} }, models: async () => ({ status: 'empty' }), + limits: async () => ({}), hooks: {}, transcripts: {}, + }); + t.after(() => server.close()); + for (const route of ['/', '/index.html', ...EXACT, ...PARAMETERIZED]) { + for (const suffix of ['?refresh=deep', '?refresh=scan&refresh=deep&trees=1', '?trees=1']) { + await request(server, route + suffix); + assert.deepEqual(calls, { system: 0, maintenance: 0, inventory: 0, rebuild: 0, provider: 0 }, route + suffix); + } + } + for (const route of ['/api/system', '/api/system/summary', '/api/maintenance']) { + for (const suffix of ['?refresh=deep', '?refresh=scan&refresh=deep&trees=1', '?trees=1']) { + const response = await request(server, route + suffix); + assert.equal(response.status, 400, route + suffix); + assert.deepEqual(JSON.parse(response.body), ERROR); + } + } + assert.equal((await request(server, '/api/host-health/local', 'POST')).status, 405); +}); diff --git a/tests/kit/dashboard-hermetic-defaults.test.mjs b/tests/kit/dashboard-hermetic-defaults.test.mjs index 20d8bd12..47b92d1d 100644 --- a/tests/kit/dashboard-hermetic-defaults.test.mjs +++ b/tests/kit/dashboard-hermetic-defaults.test.mjs @@ -10,7 +10,6 @@ import fs from 'node:fs'; import http from 'node:http'; import path from 'node:path'; import { sandboxHome, assertSandboxed, rmrf } from './helpers/home-sandbox.mjs'; -import { waitUntil } from './helpers/wait-until.mjs'; const home = sandboxHome('ak-dash-hermetic'); after(() => rmrf(home)); @@ -42,9 +41,7 @@ test('an injected System collector alone never builds the default maintenance se t.after(() => server.close()); const deep = await get(server, 'api/system?refresh=deep'); - assert.equal(deep.status, 200); - await waitUntil(() => errors.some((e) => /maintenanceOptions\.controlRoot/.test(e)), - 'the refused default maintenance service must be logged', { timeout: 5000 }); + assert.equal(deep.status, 400); const maintenance = await get(server, 'api/maintenance'); assert.equal(maintenance.status, 503); assert.equal(fs.existsSync(path.join(paths.maintenanceControlDir())), false, diff --git a/tests/kit/dashboard-refresh-api.test.mjs b/tests/kit/dashboard-refresh-api.test.mjs new file mode 100644 index 00000000..3a45b2e9 --- /dev/null +++ b/tests/kit/dashboard-refresh-api.test.mjs @@ -0,0 +1,188 @@ +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import http from 'node:http'; +import fs from 'node:fs'; +import path from 'node:path'; +import { sandboxHome, sandboxProject, rmrf } from './helpers/home-sandbox.mjs'; + +const home = sandboxHome('ak-dashboard-refresh'); +const project = sandboxProject('ak-dashboard-refresh'); +const controlRoot = fs.mkdtempSync(path.join(home, 'control-')); +after(() => rmrf(home, project)); +const { startDashboard } = await import('../../src/lib/dashboard-server.mjs'); + +function request(server, method, route, body, headers = {}) { + const url = new URL(route, server.url); + return new Promise((resolve, reject) => { + const req = http.request(url, { method, headers: { + 'content-type': 'application/json', 'x-dash-token': server.token, + origin: url.origin, 'sec-fetch-site': 'same-origin', ...headers, + } }, res => { + let data = ''; res.on('data', chunk => { data += chunk; }); + res.on('end', () => { + let json; + try { json = JSON.parse(data); } catch { json = null; } + resolve({ status: res.statusCode, json, body: data }); + }); + }); + req.on('error', reject); + req.end(body === undefined ? undefined : JSON.stringify(body)); + }); +} + +function stages(overrides = {}) { + return Object.fromEntries(['machine', 'maintenance', 'inventory', 'live', 'local'].map(id => + [id, overrides[id] ?? (async () => ({ ok: true }))])); +} + +async function serverWith(stagesMap) { + return startDashboard({ port: 0, cwd: project, + fetchStatus: async () => ({ overall: 'ok', rows: [] }), + hostReadiness: async () => ({ hosts: {} }), + system: { read: async () => ({}), refreshDeep: async () => ({ ok: true }) }, + maintenance: { report: async () => ({}), scan: async () => ({}), plan: async () => ({}) }, + management: { refreshInventory: async () => ({ ok: true }) }, + maintenanceOptions: { controlRoot }, usage: {}, discoverProjects: () => [], + refreshStages: stagesMap, + }); +} + +async function finished(server) { + const deadline = Date.now() + 3000; + while (Date.now() < deadline) { + const latest = await request(server, 'GET', '/api/refresh'); + if (latest.json.running === false && latest.json.lastRun !== null) return latest.json; + await new Promise(resolve => setTimeout(resolve, 5)); + } + throw new Error('refresh did not finish'); +} + +test('refresh POST needs the header capability and same origin, and validates its bounded body', async t => { + const server = await serverWith(stages()); + t.after(() => server.close()); + assert.deepEqual((await request(server, 'GET', '/api/refresh')).json, { running: false, lastRun: null }); + const valid = { strength: 'local' }; + assert.equal((await request(server, 'POST', '/api/refresh', valid, { 'x-dash-token': '' })).status, 401); + assert.equal((await request(server, 'POST', `/api/refresh?token=${server.token}`, valid, { 'x-dash-token': '' })).status, 401); + assert.equal((await request(server, 'POST', '/api/refresh', valid, { origin: 'http://evil.test' })).status, 403); + for (const body of [{ strength: 'other' }, { strength: 'local', extra: true }, + { strength: 'local', projectTrees: true }, { strength: 'machine', projectTrees: 'yes' }, {}]) { + assert.equal((await request(server, 'POST', '/api/refresh', body)).status, 400, JSON.stringify(body)); + } +}); + +test('refresh POST starts ordered local work once and GET exposes progress and completion', async t => { + const calls = []; + const server = await serverWith(stages(Object.fromEntries(['maintenance', 'inventory', 'local'].map(id => + [id, async () => { calls.push(id); return { ok: true }; }])))); + t.after(() => server.close()); + const post = await request(server, 'POST', '/api/refresh', { strength: 'local' }); + assert.equal(post.status, 202); + assert.equal(post.json.started, true); + const state = await finished(server); + assert.equal(state.ok, true); + assert.equal(state.strength, 'local'); + assert.deepEqual(state.stages.map(({ id, state: stageState }) => [id, stageState]), + [['maintenance', 'done'], ['inventory', 'done'], ['local', 'done']]); + assert.deepEqual(calls, ['maintenance', 'inventory', 'local']); + assert.ok(state.startedAt && state.finishedAt); +}); + +test('successive refreshes expose distinct stable operation identities', async t => { + const server = await serverWith(stages()); + t.after(() => server.close()); + const firstPost = await request(server, 'POST', '/api/refresh', { strength: 'local' }); + const first = await finished(server); + const secondPost = await request(server, 'POST', '/api/refresh', { strength: 'local' }); + const second = await finished(server); + assert.equal(firstPost.status, 202); + assert.equal(secondPost.status, 202); + assert.match(first.operationId, /^[0-9a-f-]{36}$/); + assert.equal(firstPost.json.state.operationId, first.operationId); + assert.equal(secondPost.json.state.operationId, second.operationId); + assert.notEqual(first.operationId, second.operationId); +}); + +test('a second POST cannot start work while the operation is in flight', async t => { + let release; + const held = new Promise(resolve => { release = resolve; }); + const server = await serverWith(stages({ maintenance: async () => { await held; return { ok: true }; } })); + t.after(() => { release(); return server.close(); }); + assert.equal((await request(server, 'POST', '/api/refresh', { strength: 'local' })).status, 202); + const conflict = await request(server, 'POST', '/api/refresh', { strength: 'machine' }); + assert.equal(conflict.status, 409); + assert.equal(conflict.json.error, 'a refresh is already running'); + assert.equal(conflict.json.state.running, true); + release(); + await finished(server); +}); + +test('a failed machine measurement skips dependent stages and still performs local refresh', async t => { + const called = []; + const server = await serverWith(stages({ + machine: async () => ({ ok: false, detail: 'measurement failed' }), + maintenance: async () => { called.push('maintenance'); return { ok: true }; }, + inventory: async () => { called.push('inventory'); return { ok: true }; }, + local: async () => { called.push('local'); return { ok: true }; }, + })); + t.after(() => server.close()); + assert.equal((await request(server, 'POST', '/api/refresh', { strength: 'machine', projectTrees: true })).status, 202); + const state = await finished(server); + assert.deepEqual(state.stages.map(({ id, state: stageState }) => [id, stageState]), + [['machine', 'failed'], ['maintenance', 'skipped'], ['inventory', 'skipped'], ['local', 'done']]); + assert.deepEqual(called, ['local']); + assert.equal(state.ok, false); +}); + +test('dashboard local stage forces host readiness after collecting status', async () => { + const { dashboardRefreshStages } = await import('../../src/lib/dashboard/refresh-api.mjs'); + const calls = []; + const actual = dashboardRefreshStages({ cwd: project, pkgRoot: project, + getSystem: async () => ({ refreshDeep: async () => ({ ok: true }) }), + getMaintenance: async () => ({ scan: async () => ({}) }), + refreshInventoryAfterProviderScan: async () => ({}), + statusCollect: async () => { calls.push('status'); return { rows: [] }; }, + getHostReadiness: async options => { calls.push(options); return { hosts: {} }; }, + loadConfig: () => ({}), + }); + assert.equal((await actual.local()).ok, true); + assert.deepEqual(calls, ['status', { force: true }]); +}); + +test('an inventory rebuild that the server reports unavailable fails its stage', async () => { + const { dashboardRefreshStages } = await import('../../src/lib/dashboard/refresh-api.mjs'); + const actual = dashboardRefreshStages({ cwd: project, + getSystem: async () => ({}), getMaintenance: async () => ({}), + refreshInventoryAfterProviderScan: async () => null, + statusCollect: async () => ({}), getHostReadiness: async () => ({}), loadConfig: () => ({}), + }); + assert.equal((await actual.inventory({ strength: 'local' })).ok, false); +}); + +test('machine refresh honors each current project-tree choice through a shared persistent collector', async t => { + const { dashboardRefreshStages } = await import('../../src/lib/dashboard/refresh-api.mjs'); + const { createSystemCollector } = await import('../../src/lib/footprint/index.mjs'); + const measured = []; + const collector = createSystemCollector({ cwd: project, + snapshotFile: path.join(home, 'refresh-snapshot.json'), + runWorkerImpl: async ({ includeProjectTrees, startedAt }) => { + measured.push(includeProjectTrees); + return { ok: true, asOf: startedAt, sections: {}, completeness: { complete: true }, + persisted: { ok: true }, error: null, + terminal: { running: false, phase: 'done', finishedAt: startedAt, durationMs: 0, error: null } }; + }, + }); + const actual = dashboardRefreshStages({ cwd: project, getSystem: async () => collector, + getMaintenance: async () => ({}), refreshInventoryAfterProviderScan: async () => ({}), + statusCollect: async () => ({}), getHostReadiness: async () => ({}), loadConfig: () => ({}), + }); + const server = await serverWith(stages({ machine: actual.machine })); + t.after(() => server.close()); + // Separate HTTP callers (including another tab) share this server's collector. + // An omitted choice must also override a prior checked selection. + for (const projectTrees of [false, true, false, true, undefined]) { + assert.equal((await request(server, 'POST', '/api/refresh', { strength: 'machine', projectTrees })).status, 202); + assert.equal((await finished(server)).ok, true); + } + assert.deepEqual(measured, [false, true, false, true, false]); +}); diff --git a/tests/kit/dashboard-status-cost.test.mjs b/tests/kit/dashboard-status-cost.test.mjs index f731e30a..50fcdb85 100644 --- a/tests/kit/dashboard-status-cost.test.mjs +++ b/tests/kit/dashboard-status-cost.test.mjs @@ -16,13 +16,28 @@ import assert from 'node:assert/strict'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; +import http from 'node:http'; import { spawnEnv, sandboxProject, writeKitConfig, offlineKitConfig } from './helpers/home-sandbox.mjs'; import { startGuardedDashboard, stopGuardedDashboard, getJson, markLedgerBoundary, readLedger, sliceByCallBoundary, isVersionDriftLookup, } from './helpers/dashboard-child-server.mjs'; -test('two 30s-poll-tick /api/status requests: the second starts no processes and transfers no more data than the first', async () => { +function postRefresh(port, token) { + return new Promise((resolve, reject) => { + const req = http.request({ host: '127.0.0.1', port, path: '/api/refresh', method: 'POST', headers: { + 'x-dash-token': token, 'content-type': 'application/json', + origin: `http://127.0.0.1:${port}`, 'sec-fetch-site': 'same-origin', + } }, res => { + let body = ''; res.on('data', chunk => { body += chunk; }); + res.on('end', () => resolve({ status: res.statusCode, body })); + }); + req.on('error', reject); + req.end(JSON.stringify({ strength: 'local' })); + }); +} + +test('two 30s-poll-tick /api/status requests: the second starts no processes and transfers no more data than the first', async t => { const ledgerFile = path.join(os.tmpdir(), `ak-dash-cost-${process.pid}.ndjson`); const home = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-dash-cost-home-')); fs.mkdirSync(path.join(home, '.config'), { recursive: true }); @@ -69,6 +84,29 @@ test('two 30s-poll-tick /api/status requests: the second starts no processes and assert.ok(second.bytes <= first.bytes + budget, `second /api/status response (${second.bytes} bytes) exceeds the first (${first.bytes} bytes) ` + `by more than the ${budget}-byte budget — possible in-process cache growth/leak across polls`); + + const startedAt = Date.now(); + const refreshStart = await postRefresh(port, token); + assert.equal(refreshStart.status, 202, refreshStart.body); + let refresh; + do { + refresh = await getJson(port, '/api/refresh', token); + if (refresh.json.running) await new Promise(resolve => setTimeout(resolve, 50)); + } while (refresh.json.running && Date.now() - startedAt < 60_000); + assert.equal(refresh.json.running, false, 'local refresh must finish within 60 s in the sandbox'); + assert.deepEqual(refresh.json.stages.map(({ id, state }) => [id, state]), + [['maintenance', 'done'], ['inventory', 'done'], ['local', 'done']]); + markLedgerBoundary(ledgerFile, 'refresh'); + const third = await getJson(port, '/api/status', token); + assert.equal(third.status, 200); + markLedgerBoundary(ledgerFile, 'third'); + const { third: thirdSlice } = sliceByCallBoundary(readLedger(ledgerFile), ['first', 'second', 'refresh', 'third']); + const thirdUnexplained = thirdSlice.filter((l) => !isVersionDriftLookup(l)); + assert.equal(thirdUnexplained.length, 0, JSON.stringify(thirdUnexplained.map((l) => [l.cmd, l.args]))); + assert.ok(third.bytes <= second.bytes + budget, `third /api/status: ${third.bytes} bytes; second: ${second.bytes}`); + assert.deepEqual(Object.keys(third.json).sort(), Object.keys(second.json).sort()); + t.diagnostic(`status bytes cold/warm/post-refresh=${first.bytes}/${second.bytes}/${third.bytes}; ` + + `local refresh=${Date.now() - startedAt}ms; third unexplained spawns=${thirdUnexplained.length}`); } finally { await stopGuardedDashboard(child); fs.rmSync(ledgerFile, { force: true }); @@ -76,3 +114,8 @@ test('two 30s-poll-tick /api/status requests: the second starts no processes and fs.rmSync(project, { recursive: true, force: true }); } }); + +test('the idle polling client never requests the refresh route', () => { + const poll = fs.readFileSync(new URL('../../src/lib/dashboard/client/poll.mjs', import.meta.url), 'utf8'); + assert.doesNotMatch(poll, /\/api\/refresh/); +}); diff --git a/tests/kit/dashboard-status-inprocess.test.mjs b/tests/kit/dashboard-status-inprocess.test.mjs index 5dcf8e9d..72c901f1 100644 --- a/tests/kit/dashboard-status-inprocess.test.mjs +++ b/tests/kit/dashboard-status-inprocess.test.mjs @@ -14,12 +14,12 @@ // return, which the in-process path would otherwise silently drop. import { test } from 'node:test'; import assert from 'node:assert/strict'; -import { spawnSync } from 'node:child_process'; +import { spawn, spawnSync } from 'node:child_process'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; -import { spawnEnv, sandboxProject, writeKitConfig, offlineKitConfig } from './helpers/home-sandbox.mjs'; +import { spawnEnv, sandboxProject, writeKitConfig, offlineKitConfig, fakeGlobalRoot } from './helpers/home-sandbox.mjs'; import { startGuardedDashboard, stopGuardedDashboard, getJson, markLedgerBoundary, readLedger, sliceByCallBoundary, isVersionDriftLookup, @@ -139,3 +139,47 @@ test('GET /api/status (in-process) carries the same {overall, rows} `ak status - cleanup(home, project); } }); + +test('GET /api/status passes the server cwd through ruflo-components project-root discovery', async () => { + const { home, project, env } = sandbox('ak-dash-ruflo-cwd'); + const decoy = sandboxProject('ak-dash-ruflo-decoy'); + const fakeRoot = fakeGlobalRoot(home, { ruflo: '9.9.9' }); + fs.mkdirSync(path.join(project, '.claude-flow')); + fs.mkdirSync(path.join(project, '.harness')); + fs.writeFileSync(path.join(project, '.harness', 'mcp-policy.json'), '{invalid'); + fs.mkdirSync(path.join(decoy, '.claude-flow')); + writeKitConfig(home, offlineKitConfig({ rufloComponents: { mcpGovernance: { maxCallsPerMinute: 60 } } })); + + let child; + try { + const ready = await new Promise((resolve, reject) => { + child = spawn(process.execPath, [path.join(PKG_ROOT, 'tests/fixtures/dashboard-status-child.mjs'), project, fakeRoot], { + cwd: decoy, env, stdio: ['ignore', 'pipe', 'pipe'], + }); + let out = '', err = ''; + child.stdout.on('data', chunk => { + out += chunk; + const match = out.match(/READY (\d+) (\S+)\n/); + if (match) resolve({ port: Number(match[1]), token: match[2] }); + }); + child.stderr.on('data', chunk => { err += chunk; }); + child.once('error', reject); + child.once('exit', code => reject(new Error(`dashboard child exited ${code}: ${err || out}`))); + }); + const response = await getJson(ready.port, '/api/status', ready.token); + assert.equal(response.status, 200); + assert.ok(response.json.rows.some(row => row.subsystem === 'ruflo-components' + && /ruflo components:.*ruflo 9\.9\.9/.test(row.message)), + 'the status request must use the disposable fake Ruflo package'); + const governance = response.json.rows.find(row => row.subsystem === 'ruflo-components' + && /MCP tool governance/.test(row.message)); + assert.ok(governance, 'fake installed Ruflo must reach its component projection'); + assert.equal(governance.state, 'blocked'); + assert.match(governance.message, /policy file is invalid/, + 'the server-supplied project cwd, not process.cwd(), must drive rufloProjectRoot'); + } finally { + await stopGuardedDashboard(child); + cleanup(home, project); + fs.rmSync(decoy, { recursive: true, force: true }); + } +}); diff --git a/tests/kit/host-health-api.test.mjs b/tests/kit/host-health-api.test.mjs index 4b43311e..ceae90cb 100644 --- a/tests/kit/host-health-api.test.mjs +++ b/tests/kit/host-health-api.test.mjs @@ -37,9 +37,9 @@ test('connection route requires same-origin header auth, explicit consent, and b assert.equal((await request(server, 'POST', '/api/host-health/connection', [])).status, 400); assert.equal((await request(server, 'POST', '/api/host-health/connection', body)).status, 200); assert.equal(connections, 1); - assert.equal((await request(server, 'POST', '/api/host-health/local', { host: 'codex' })).status, 200); + assert.equal((await request(server, 'POST', '/api/host-health/local', { host: 'codex' })).status, 405); assert.equal(connections, 1); - assert.ok(reads >= 3); + assert.equal(reads, 2, 'retired local POST does not run another host check'); await server.close(); assert.equal(closed, true); }); diff --git a/tests/kit/live-service.test.mjs b/tests/kit/live-service.test.mjs index b5cb6528..85491242 100644 --- a/tests/kit/live-service.test.mjs +++ b/tests/kit/live-service.test.mjs @@ -185,6 +185,51 @@ test('bounded native discovery rotates to a newer transcript created after start service.close(); }); +test('a native transcript re-entering the bounded window resumes without replaying accepted records', () => { + const sb = sandbox(); + const old = path.join(sb.claude, 'old.jsonl'); + const recent = path.join(sb.claude, 'recent.jsonl'); + fs.writeFileSync(old, line({ + type: 'user', sessionId: 'old', cwd: '/work/old-project', + timestamp: '2026-07-27T11:00:00Z', message: { content: 'fixture only' }, + })); + fs.utimesSync(old, new Date(1_000), new Date(1_000)); + let tick; + const service = new LiveSessionsService({ + roots: sb.roots, maxFiles: 1, readCodexState: () => null, + setInterval: (fn) => { tick = fn; return { unref() {} }; }, clearInterval: () => {}, + now: () => '2026-07-27T12:00:00Z', + }); + try { + service.start(); + const before = service.snapshot().health.claude.accepted; + assert.ok(before > 0); + fs.writeFileSync(recent, line({ + type: 'user', sessionId: 'recent', cwd: '/work/recent-project', + timestamp: '2026-07-27T11:01:00Z', message: { content: 'fixture only' }, + })); + fs.utimesSync(recent, new Date(2_000), new Date(2_000)); + tick(); + const rotated = service.snapshot(); + const afterRotation = rotated.health.claude.accepted; + assert.ok(afterRotation > before); + service.close(); + service.start(); + fs.utimesSync(old, new Date(3_000), new Date(3_000)); + tick(); + const reentered = service.snapshot(); + assert.equal(reentered.health.claude.accepted, afterRotation); + assert.equal(reentered.sessions.length, rotated.sessions.length); + assert.equal(reentered.projects.length, rotated.projects.length); + fs.appendFileSync(old, line({ + type: 'assistant', sessionId: 'old', cwd: '/work/old-project', + timestamp: '2026-07-27T12:00:01Z', message: { content: 'fixture only' }, + })); + tick(); + assert.equal(service.snapshot().health.claude.accepted, afterRotation + 1); + } finally { service.close(); } +}); + test('metadata bootstrap is adversarially privacy bounded', () => { const sb = sandbox(); fs.writeFileSync(path.join(sb.codex, 'rollout-2026-07-27T12-00-00-x1.jsonl'), [ diff --git a/tests/kit/maintenance-dashboard-api.test.mjs b/tests/kit/maintenance-dashboard-api.test.mjs index 9e3c1efe..56c56e77 100644 --- a/tests/kit/maintenance-dashboard-api.test.mjs +++ b/tests/kit/maintenance-dashboard-api.test.mjs @@ -303,14 +303,14 @@ test('dashboard Maintenance API keeps GET lazy and mutation paths exact', async assert.equal(unknownMutation.status, 405); const rescanned = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(rescanned.status, 200); - assert.equal(service.calls.scan, 1, 'only the explicit scan query invokes maintenance scanning'); + assert.equal(rescanned.status, 400); + assert.equal(service.calls.scan, 0, 'GET cannot invoke maintenance scanning'); const unknownRefresh = await request(server, '/api/maintenance?refresh=deep', { origin: false, fetchSite: null }); const duplicateRefresh = await request(server, '/api/maintenance?refresh=scan&refresh=scan', { origin: false, fetchSite: null }); const unknownQuery = await request(server, '/api/maintenance?extra=scan', { origin: false, fetchSite: null }); assert.deepEqual([unknownRefresh.status, duplicateRefresh.status, unknownQuery.status], [400, 400, 400]); - assert.equal(service.calls.scan, 1, 'ambiguous scan queries never invoke maintenance scanning'); + assert.equal(service.calls.scan, 0, 'scan queries never invoke maintenance scanning'); }); test('dashboard Maintenance reports provider activity and refuses action requests during a scan', async (t) => { @@ -465,18 +465,6 @@ test('dashboard Maintenance API distinguishes pre-mutation refusal from a receip // ── ADR-0048: provider scans chain the inventory rebuild (QE defect D5) ───── -function eventually(predicate, message, timeout = 1500) { - const started = Date.now(); - return new Promise((resolve, reject) => { - const check = () => { - if (predicate()) return resolve(undefined); - if (Date.now() - started >= timeout) return reject(new Error(message)); - setTimeout(check, 5); - }; - check(); - }); -} - function recordingManagement({ refreshInventory, measured = true } = {}) { const calls = []; const rebuilt = { inventoryId: 'inv_refreshed', capturedAt: '2026-09-05T12:00:00.000Z' }; @@ -493,84 +481,39 @@ function recordingManagement({ refreshInventory, measured = true } = {}) { return { calls, facade }; } -test('GET /api/maintenance?refresh=scan chains exactly one management.refreshInventory({ deep:false }) without awaiting it', async (t) => { - const service = fixtureService(); - let release; - const gate = new Promise((resolve) => { release = resolve; }); - const management = recordingManagement({ refreshInventory: () => gate }); - const server = await startDashboard({ port: 0, maintenance: service, management: management.facade, usage: {} }); - t.after(() => server.close()); - - const plain = await request(server, '/api/maintenance', { origin: false, fetchSite: null }); - assert.equal(plain.status, 200); - assert.deepEqual(management.calls, [], 'a plain read never rebuilds the inventory'); - - const first = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(first.status, 200, 'the provider-scan response never waits for the inventory rebuild'); - await eventually(() => management.calls.length === 1, 'the provider scan must chain one inventory rebuild'); - assert.deepEqual(management.calls, [{ deep: false }], 'a cheap provider check rebuilds only; it never walks discovery sources'); - - const second = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(second.status, 200); - assert.equal(service.calls.scan, 2); - assert.equal(management.calls.length, 1, 'a rebuild still in flight is joined, not duplicated'); - release(); - await eventually(() => JSON.parse(JSON.stringify(management.calls)).length === 1, 'settled'); - const third = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(third.status, 200); - await eventually(() => management.calls.length === 2, 'a later provider scan rebuilds again once the flight settled'); -}); - -test('an inventory rebuild failure is logged and never turns the provider-scan response into an error', async (t) => { - const service = fixtureService(); - const management = recordingManagement({ refreshInventory: async () => { throw new Error('inventory store unavailable'); } }); - const logged = []; - const original = console.error; - console.error = (...args) => { logged.push(args.map(String).join(' ')); }; - t.after(() => { console.error = original; }); - const server = await startDashboard({ port: 0, maintenance: service, management: management.facade, usage: {} }); - t.after(() => server.close()); - - const scanned = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(scanned.status, 200); - await eventually(() => logged.some((line) => /maintenance inventory refresh failed/.test(line)), 'the failure is logged'); - assert.deepEqual(management.calls, [{ deep: false }]); - const again = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(again.status, 200); - await eventually(() => management.calls.length === 2, 'a failed rebuild does not wedge the single-flight slot'); -}); - -test('a completed System deep scan chains one provider scan and then one inventory rebuild', async (t) => { +test('GET Maintenance scan queries never invoke provider checks or inventory rebuilds', async (t) => { const service = fixtureService(); const management = recordingManagement(); - const collector = { - async read() { return { scan: { running: false, phase: 'idle' }, generatedAt: '2026-09-05T12:00:00.000Z' }; }, - async refreshDeep() { return { ok: true, persisted: { ok: true } }; }, - scanState() { return { running: true, phase: 'system' }; }, - }; - const server = await startDashboard({ port: 0, system: collector, maintenance: service, management: management.facade, usage: {} }); + const server = await startDashboard({ port: 0, maintenance: service, management: management.facade, usage: {} }); t.after(() => server.close()); - - const started = await request(server, '/api/system?refresh=deep', { origin: false, fetchSite: null }); - assert.equal(started.status, 200); - await eventually(() => service.calls.scan === 1, 'the completed System scan refreshes Maintenance evidence once'); - await eventually(() => management.calls.length === 1, 'the provider scan then rebuilds the inventory once'); - assert.deepEqual(management.calls, ['rebuildAfterMeasurement'], - 'a machine measurement walks discovery sources and rebuilds, and never also runs the cheap rebuild'); + assert.equal((await request(server, '/api/maintenance', { origin: false, fetchSite: null })).status, 200); + for (const query of ['refresh=scan', 'refresh=deep', 'refresh=scan&refresh=scan', 'trees=1', 'extra=scan']) { + const response = await request(server, '/api/maintenance?' + query, { origin: false, fetchSite: null }); + assert.equal(response.status, 400, query); + assert.deepEqual(JSON.parse(response.body), { error: 'start a refresh with POST /api/refresh' }); + } + assert.equal(service.calls.scan, 0); + assert.deepEqual(management.calls, []); }); -test('a completed System deep scan falls back to the cheap rebuild when the facade has no measured rebuild', async (t) => { - const service = fixtureService(); - const management = recordingManagement({ measured: false }); - const collector = { - async read() { return { scan: { running: false, phase: 'idle' } }; }, - async refreshDeep() { return { ok: true, persisted: { ok: true } }; }, - }; - const server = await startDashboard({ port: 0, system: collector, maintenance: service, management: management.facade, usage: {} }); - t.after(() => server.close()); - assert.equal((await request(server, '/api/system?refresh=deep', { origin: false, fetchSite: null })).status, 200); - await eventually(() => management.calls.length === 1, 'the chain still rebuilds'); - assert.deepEqual(management.calls, [{ deep: false }]); +test('POST refresh stages retain machine measurement, provider scan and inventory rebuild order', async () => { + const { dashboardRefreshStages } = await import('../../src/lib/dashboard/refresh-api.mjs'); + const calls = []; + const stage = dashboardRefreshStages({ cwd: process.cwd(), + getSystem: async () => ({ refreshDeep: async options => { calls.push(['machine', options]); return { ok: true, persisted: { ok: true } }; } }), + getMaintenance: async () => ({ scan: async options => { calls.push(['maintenance', options]); return { scan: {} }; } }), + refreshInventoryAfterProviderScan: async options => { calls.push(['inventory', options]); return {}; }, + getHostReadiness: async () => ({}), statusCollect: async () => ({}), loadConfig: () => ({}), + }); + assert.equal((await stage.machine({ projectTrees: true })).ok, true); + assert.equal((await stage.maintenance()).ok, true); + assert.equal((await stage.inventory({ strength: 'machine' })).ok, true); + assert.deepEqual(calls, [['machine', { includeProjectTrees: true }], ['maintenance', { deep: false }], + ['inventory', { measured: true }]]); + calls.length = 0; + assert.equal((await stage.maintenance()).ok, true); + assert.equal((await stage.inventory({ strength: 'local' })).ok, true); + assert.deepEqual(calls, [['maintenance', { deep: false }], ['inventory', { measured: false }]]); }); test('an injected maintenance service without an injected facade never composes the default facade (hermetic)', async (t) => { @@ -578,8 +521,8 @@ test('an injected maintenance service without an injected facade never composes const server = await startDashboard({ port: 0, maintenance: service, usage: {} }); t.after(() => server.close()); const scanned = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(scanned.status, 200); - assert.equal(service.calls.scan, 1); + assert.equal(scanned.status, 400); + assert.equal(service.calls.scan, 0); const inventory = await request(server, '/api/maintenance/v2/inventory', { origin: false, fetchSite: null }); assert.equal(inventory.status, 503); assert.deepEqual(JSON.parse(inventory.body), { error: 'maintenance management unavailable' }); diff --git a/tests/kit/maintenance-dashboard-client-labels.test.mjs b/tests/kit/maintenance-dashboard-client-labels.test.mjs index 1d1e7272..48a92331 100644 --- a/tests/kit/maintenance-dashboard-client-labels.test.mjs +++ b/tests/kit/maintenance-dashboard-client-labels.test.mjs @@ -100,17 +100,9 @@ test('MNT-EVD-006: extractor is not vacuous (finds real, allowed literals)', () assert.ok(literals.includes('Clear all')); }); -// The Refresh-evidence rename introduced two labels the extractor structurally -// cannot see: "Refresh evidence"/"Refreshing evidence…" are assigned via a -// plain JS ternary (`button.textContent=busy?...:...`), not a `label:` -// property or a static HTML text node, and "Re-measure machine" lives only in -// page.mjs's server-rendered markup, which this file does not scan at all. -// Assert them directly so a future rename cannot silently reintroduce a -// prohibited word here without any test noticing. -test('MNT-EVD-006: the Refresh evidence / Re-measure machine controls carry no prohibited label', () => { - for (const label of ['Refresh evidence', 'Refreshing evidence…', 'Re-measure machine']) { - assert.equal(isProhibitedLabel(label), false, label); - } +test('the retired Maintenance refresh controls are absent from the page', () => { + const page = fs.readFileSync(new URL('../../src/lib/dashboard/page.mjs', import.meta.url), 'utf8'); + for (const id of ['mnt-check-providers', 'mnt-remeasure']) assert.doesNotMatch(page, new RegExp('id="' + id + '"')); }); // (b) The copied label maps in maintenance-workspace.mjs must equal diff --git a/tests/kit/maintenance-dashboard-e2e.test.mjs b/tests/kit/maintenance-dashboard-e2e.test.mjs index 73e5e65a..a41d249d 100644 --- a/tests/kit/maintenance-dashboard-e2e.test.mjs +++ b/tests/kit/maintenance-dashboard-e2e.test.mjs @@ -186,10 +186,24 @@ test('dashboard HTTP executes and undoes one maintenance finding through the rea const server = await startDashboard({ port: 0, maintenance: service, usage: {}, maintenanceOptions: HERMETIC_MAINTENANCE, fetchStatus: async () => ({ overall: 'ok', rows: [] }), + refreshStages: { + maintenance: async () => { await service.scan({ deep: false }); return { ok: true }; }, + inventory: async () => ({ ok: true }), local: async () => ({ ok: true }), + }, }); t.after(() => server.close()); - const report = await request(server, '/api/maintenance?refresh=scan'); + const started = await request(server, '/api/refresh', { method: 'POST', body: { strength: 'local' } }); + assert.equal(started.status, 202); + let state; + for (let attempt = 0; attempt < 100; attempt++) { + state = (await request(server, '/api/refresh')).body; + if (!state.running) break; + await new Promise(resolve => setTimeout(resolve, 5)); + } + assert.equal(state?.running, false, 'the explicit refresh finishes'); + assert.equal(state?.ok, true); + const report = await request(server, '/api/maintenance'); assert.equal(report.status, 200); assert.equal(report.headers['cache-control'], 'no-store'); assertNoPrivateTransport(report.body); diff --git a/tests/kit/maintenance-dashboard-v2-api.test.mjs b/tests/kit/maintenance-dashboard-v2-api.test.mjs index 9f9ddede..a8edec94 100644 --- a/tests/kit/maintenance-dashboard-v2-api.test.mjs +++ b/tests/kit/maintenance-dashboard-v2-api.test.mjs @@ -748,26 +748,18 @@ test('v2 inventory projection carries row kind and opaque-id facet labels over e assert.doesNotMatch(JSON.stringify(hostile), /Users\/alice|leak/); }); -test('report({ refresh:true }) fires afterScan once after a successful provider scan and never lets it fail the response', async () => { +test('report reads persisted evidence and ignores retired refresh arguments', async () => { const events = []; - const service = { async report() { return {}; }, async scan() { events.push('scan'); return {}; }, async plan() { return {}; } }; + const service = { async report() { events.push('report'); return {}; }, async scan() { events.push('scan'); return {}; }, async plan() { return {}; } }; const api = createMaintenanceDashboardApi({ service, sessionToken: SESSION, afterScan: () => { events.push('afterScan'); throw new Error('rebuild failed'); }, }); const plain = fakeRes(); await api.report({}, plain, { refresh: false }); - assert.deepEqual([plain.out.status, events], [200, []]); + assert.deepEqual([plain.out.status, events], [200, ['report']]); const refreshed = fakeRes(); await api.report({}, refreshed, { refresh: true }); - assert.deepEqual([refreshed.out.status, events], [200, ['scan', 'afterScan']]); - const failing = createMaintenanceDashboardApi({ - service: { ...service, async scan() { throw new Error('provider check failed'); } }, sessionToken: SESSION, - afterScan: () => { events.push('never'); }, - }); - const failed = fakeRes(); - await failing.report({}, failed, { refresh: true }); - assert.equal(failed.out.status, 503); - assert.equal(events.includes('never'), false, 'afterScan only follows a successful scan'); + assert.deepEqual([refreshed.out.status, events], [200, ['report', 'report']]); }); test('lastRefresh is allowlisted on the report, inventory, and guidance envelopes with a guarded, label-safe message (QE D6b)', async () => { @@ -895,6 +887,31 @@ test('public activity retains historical scans as well as latest source summarie assert.equal(payload.scanHistory.length, 2); }); +test('public activity allows a bounded pause time while omitting invalid timestamps and private metadata', () => { + const scan = { sourceId: SOURCE, state: 'paused', recordedAt: '2026-09-09T12:00:00.000Z', completedAt: null, privatePath: PRIVATE_PATH }; + const payload = publicActivity({ scans: [scan], scanHistory: [scan, { ...scan, recordedAt: PRIVATE_PATH }, { ...scan, recordedAt: '1' }] }); + assert.equal(payload.scans[0].recordedAt, scan.recordedAt); + assert.equal(payload.scans[0].completedAt, null); + assert.equal(payload.scanHistory[1].recordedAt, undefined); + assert.equal(payload.scanHistory[2].recordedAt, undefined); + assert.equal(JSON.stringify(payload).includes(PRIVATE_PATH), false); +}); + +test('public activity rejects impossible pause dates and retains valid leap-day offset stamps', () => { + const base = { sourceId: SOURCE, state: 'paused', completedAt: null }; + const valid = '2024-02-29T23:59:59.125+05:30'; + const payload = publicActivity({ scans: [{ ...base, recordedAt: valid }], scanHistory: [ + { ...base, recordedAt: '2026-09-31T12:00:00Z', completedAt: '2026-09-30T12:00:00Z' }, + { ...base, recordedAt: '2025-02-29T12:00:00Z' }, + { ...base, recordedAt: valid }, + ] }); + assert.equal(payload.scans[0].recordedAt, valid); + assert.equal(payload.scanHistory[0].recordedAt, undefined); + assert.equal(payload.scanHistory[0].completedAt, '2026-09-30T12:00:00Z'); + assert.equal(payload.scanHistory[1].recordedAt, undefined); + assert.equal(payload.scanHistory[2].recordedAt, valid); +}); + test('v2 reports native persistence refusal without suggesting an action started', async () => { const refusal = Object.assign(new Error('private adapter unavailable'), { code: 'MAINTENANCE_PERSISTENCE_UNAVAILABLE' }); const { post } = harness({ management: stubManagement({ planAction: async () => { throw refusal; } }).facade }); diff --git a/tests/kit/maintenance-management-activity.test.mjs b/tests/kit/maintenance-management-activity.test.mjs index e0a6de0f..b6fddc27 100644 --- a/tests/kit/maintenance-management-activity.test.mjs +++ b/tests/kit/maintenance-management-activity.test.mjs @@ -77,6 +77,51 @@ test('activity shows the latest scan per source and environment without changing assert.deepEqual(scanHistory, original); }); +test('a newer paused record is the latest state and retains its distinct recorded time', () => { + const base = { sourceId: 'claude', environmentId: 'local', label: 'Claude', visited: 12 }; + const complete = { ...base, state: 'complete', completedAt: '2026-09-08T12:00:00.000Z' }; + const paused = { ...base, state: 'paused', recordedAt: '2026-09-09T12:00:00.000Z', completedAt: null }; + const result = buildActivity({ scanHistory: [paused, complete] }); + assert.equal(result.scans[0].state, 'paused'); + assert.equal(result.scans[0].recordedAt, paused.recordedAt); + assert.equal(result.scans[0].completedAt, null); + assert.equal(result.scanHistory[1].completedAt, complete.completedAt); +}); + +test('invalid recorded time falls back to completion without passing metadata through', () => { + const base = { sourceId: 'claude', environmentId: 'local', state: 'complete' }; + const result = buildActivity({ scanHistory: [ + { ...base, completedAt: '2026-09-09T12:00:00.000Z', recordedAt: '/private/path', privatePath: '/private/path' }, + { ...base, completedAt: '2026-09-08T12:00:00.000Z' }, + ] }); + assert.equal(result.scans[0].completedAt, '2026-09-09T12:00:00.000Z'); + assert.equal(result.scans[0].recordedAt, null); + assert.equal(buildActivity({ scanHistory: [{ ...base, completedAt: '/private/path' }] }).scanHistory[0].completedAt, null); + assert.equal(JSON.stringify(result).includes('/private/path'), false); +}); + +test('an impossible pause date cannot supersede a real completion or invent a calendar day', () => { + const base = { sourceId: 'claude', environmentId: 'local' }; + const completed = { ...base, state: 'complete', completedAt: '2026-09-30T12:00:00.000Z' }; + const paused = { ...base, state: 'paused', recordedAt: '2026-09-31T12:00:00Z', completedAt: null }; + const result = buildActivity({ scanHistory: [paused, completed] }); + assert.equal(result.scans[0].state, 'complete'); + assert.equal(result.scanHistory[0].recordedAt, null); + assert.equal(result.scanHistory[0].completedAt, null); + const fallback = buildActivity({ scanHistory: [{ ...completed, recordedAt: paused.recordedAt }] }); + assert.equal(fallback.scans[0].recordedAt, null); + assert.equal(fallback.scans[0].completedAt, completed.completedAt); +}); + +test('scan timestamps accept leap days, offsets, and fractional seconds', () => { + const recordedAt = '2024-02-29T23:59:59.125+05:30'; + const completedAt = '2024-02-29T08:00:00Z'; + const result = buildActivity({ scanHistory: [{ sourceId: 'a', environmentId: 'local', state: 'complete', recordedAt, completedAt }] }); + assert.equal(result.scans[0].recordedAt, recordedAt); + assert.equal(result.scans[0].completedAt, completedAt); + assert.equal(buildActivity({ scanHistory: [{ sourceId: 'a', environmentId: 'local', state: 'complete', completedAt: '2025-02-29T08:00:00Z' }] }).scans[0].completedAt, null); +}); + test('no label anywhere in buildActivity output is prohibited', () => { const activity = buildActivity({ receipts: [INTERRUPTED_RECEIPT], diff --git a/tests/kit/maintenance-presentation.test.mjs b/tests/kit/maintenance-presentation.test.mjs index adc9ff12..9ef6fe36 100644 --- a/tests/kit/maintenance-presentation.test.mjs +++ b/tests/kit/maintenance-presentation.test.mjs @@ -63,59 +63,37 @@ test('a project filter never strips context from a user placement', () => { const html = cards(state).renderMntGroups([group('r1', [row('p1')])], { value: 0 }); assert.match(html, /User · Codex › Skills/); }); -function operation(get) { - const state = {}; - const nodes = Object.fromEntries(['mnt-check-providers', 'mnt-remeasure', 'mnt-check-providers-status', 'mnt-operation-elapsed'].map((id) => [id, { textContent: '', dataset: {} }])); - const api = client('maintenance-operation', { MNT: state, mntGet: get, mntRefreshActiveDestination() {}, loadSystem: async () => {}, SYSTEM: {}, systemBusy: false, document: { getElementById: (id) => nodes[id], addEventListener() {} }, setInterval: () => 1, clearInterval() {}, setTimeout: (fn) => queueMicrotask(fn) }, ['mntCheckProviders', 'mntBuildStatusOf', 'mntAwaitInventoryBuild']); - return { state, nodes, ...api }; -} +test('Maintenance writes are blocked during the shared Refresh operation', () => { + const api = client('maintenance-operation', { + MNT: { externalScanBusy: false }, refreshRunning: () => true, + }, ['mntWritesBlocked']); + assert.equal(api.mntWritesBlocked(), true); +}); +test('Maintenance hash synchronization adopts an externally changed destination', () => { + const location = { hash: '#system/maintenance/inventory?scope=across' }; + const history = { replaceState(_state, _title, hash) { location.hash = hash; } }; + const api = client('maintenance-workspace', { location, history, localStorage: { setItem() {} } }, ['MNT', 'mntSyncHash']); + api.mntSyncHash(); + location.hash = '#system/maintenance/guidance?scope=project'; + api.mntSyncHash(); + assert.equal(api.MNT.destination, 'guidance'); + assert.equal(api.MNT.scope, 'project'); + assert.match(location.hash, /^#system\/maintenance\/guidance\?scope=project/); + location.hash = '#usage/score'; + api.mntSyncHash(); + assert.equal(location.hash, '#usage/score'); +}); test('an existing inventory does not mask a running or failed refresh', () => { - const api = operation(() => {}); + const api = client('maintenance-operation', {}, ['mntBuildStatusOf']); assert.equal(api.mntBuildStatusOf({ scanRequired: false, lastRefresh: { status: 'running' } }), 'running'); assert.equal(api.mntBuildStatusOf({ scanRequired: false, lastRefresh: { status: 'failed' } }), 'failed'); }); -test('refresh keeps both buttons disabled until a fresh inventory is published', async () => { - let inventoryCalls = 0, providerCalls = 0, release; - const publication = new Promise((resolve) => { release = resolve; }); - let signalWaiting; - const waiting = new Promise((resolve) => { signalWaiting = resolve; }); - const api = operation(async (url) => { - if (url.includes('/v2/inventory')) { - inventoryCalls++; - if (inventoryCalls === 1) return { scanRequired: false, lastRefresh: { at: 'old', status: 'ok' } }; - signalWaiting();return publication; - } - if (url.includes('?refresh=scan')) return {}; - providerCalls++; - return { activity: { status: 'idle' }, scan: { status: 'complete', checkedAt: providerCalls === 1 ? 'old' : 'new', coverage: 'complete' } }; - }); - const run = api.mntCheckProviders(); - await waiting; - assert.equal(api.nodes['mnt-check-providers'].disabled, true); - assert.equal(api.nodes['mnt-remeasure'].disabled, true); - assert.match(api.state.operation.message, /Updating inventory/); - release({ scanRequired: false, lastRefresh: { at: 'new', status: 'ok' }, partialSources: { total: 1 } }); - await run; - assert.equal(api.nodes['mnt-remeasure'].disabled, false); - assert.match(api.state.operation.message, /coverage gaps/); -}); -test('provider failure is visible and releases the busy state', async () => { - const api = operation(async (url) => { - if (url.includes('/v2/inventory')) return { lastRefresh: { at: 'old' } }; - if (url.includes('?refresh=scan')) throw new Error('Evidence request failed.'); - return { scan: { checkedAt: 'old' } }; - }); - await api.mntCheckProviders(); - assert.equal(api.state.operation.failed, true); - assert.match(api.state.operation.message, /Evidence request failed/); - assert.equal(api.nodes['mnt-remeasure'].disabled, false); -}); test('filesystem completion excludes provider checks without inventing their success', () => { const ctx = { state: {}, orchestrator: () => ({ coverage: () => [{ sourceId: 'files', state: 'complete', visited: 4 }], progress: () => [] }), lastGoodDiscoveryStore: { current: () => [] }, listSources: () => [{ sourceId: 'files', filesystem: true }, { sourceId: 'providers', filesystem: false, label: 'Providers' }] }; const result = scanProgress(ctx)(); assert.match(result.narrative, /1 of 1/); assert.equal(result.coverage.find((c) => c.sourceId === 'providers').state, 'not-scanned'); - assert.equal(result.evidenceChecks[0].method, 'Refresh evidence'); + assert.equal(result.evidenceChecks[0].method, 'Refresh'); }); test('Hermes path honors HERMES_HOME and otherwise resolves the user configuration', () => { const previous = process.env.HERMES_HOME; @@ -306,20 +284,6 @@ test('public inventory uses the measured project location instead of guessing fr assert.equal(page.facetLabels.project[skill.projectId], 'ampel'); assert.doesNotMatch(JSON.stringify(page), /\/private\/repo/); }); -test('a completed refresh over stale machine evidence waits for publication and asks for remeasurement', async () => { - let inventoryCalls=0, providerCalls=0; - const api=operation(async url=>{ - if(url.includes('/v2/inventory'))return {scanRequired:false,lastRefresh:{at:++inventoryCalls===1?'old':'new',status:'ok'}}; - if(url.includes('?refresh=scan'))return {scan:{status:'complete',checkedAt:'new'}}; - return {activity:{status:'complete'},scan:{status:'stale',checkedAt:++providerCalls===1?'old':'new',coverage:'partial'}}; - }); - await api.mntCheckProviders(); - assert.equal(api.state.operation.failed,false); - assert.equal(inventoryCalls,2); - assert.match(api.state.operation.message,/Evidence refreshed.*Re-measure machine/); - assert.equal(api.nodes['mnt-check-providers'].disabled,false); -}); - test('version inspector localizes measured and checked instants without interpreting version identifiers', () => { const api = client('maintenance-inspector', { esc, mntKindLabel: (value) => value, diff --git a/tests/kit/refresh-vocabulary-guard.test.mjs b/tests/kit/refresh-vocabulary-guard.test.mjs index 7523a763..e36aeb38 100644 --- a/tests/kit/refresh-vocabulary-guard.test.mjs +++ b/tests/kit/refresh-vocabulary-guard.test.mjs @@ -1,21 +1,16 @@ // refresh-vocabulary-guard.test.mjs — ADR-0063 (the refresh vocabulary): no -// retired CLI spelling from before the one-refresh-flag vocabulary may +// retired CLI or dashboard spelling from before the one-refresh vocabulary may // reappear in help text, README, docs, or the installed `claude/` guidance. // -// Scope: src/** (comments included — they must describe current CLI -// behaviour), bin/**, claude/**, README.md, and living top-level docs/*.md. +// Scope: src/** (comments included — they must describe current behaviour), +// bin/**, claude/**, README.md, and current docs/**/*.md. // docs/adr/, docs/archive/, docs/plans/, and docs/proposals/ are records that // may preserve retired spellings. In // src/lib/hook-audit/agentic-dependency-constraints.json only the dated // watch[].history[].note strings are skipped — every other string, including // `adjustment`, is scanned like any other source text. // -// CLI patterns only: the dashboard's own retired spellings ("Full -// scan", "Refresh evidence", "Re-measure machine", "Check again", "refresh -// now") are out of this guard's scope until the dashboard half of this work -// lands in a later branch (see docs/superpowers/plans/2026-09-28-branch-6b- -// one-refresh-flag.md, "Closing this branch"). This guard never asserts an -// UPGRADING section or an old -> new table exists (no legacy, no hints). +// This guard never asserts an UPGRADING section or an old -> new table exists. import { test } from 'node:test'; import assert from 'node:assert/strict'; import fs from 'node:fs'; @@ -60,13 +55,25 @@ const RETIRED_CLI_PATTERNS = [ { label: 'ak usage prompts …--deep (flag anywhere on the same line)', pattern: /\bprompts\b[^\n]*\[?--deep\b/g }, ]; +const RETIRED_DASHBOARD_PATTERNS = [ + { label: 'retired dashboard scan control', pattern: /\bFull scan\b/g }, + { label: 'retired dashboard evidence control', pattern: /\bRefresh evidence\b/g }, + { label: 'retired dashboard measurement control', pattern: /\bRe-measure machine\b/g }, + { label: 'retired dashboard local-check control', pattern: /\bCheck again\b/g }, + { label: 'retired dashboard refresh prompt', pattern: /\brefresh now\b/g }, + { label: 'retired dashboard GET refresh query', pattern: /refresh=(?:deep|scan)/g }, + { label: 'retired dashboard host-health route', pattern: /\/api\/host-health\/local/g }, +]; + +const RETIRED_PATTERNS = [...RETIRED_CLI_PATTERNS, ...RETIRED_DASHBOARD_PATTERNS]; + function lineOf(text, offset) { return text.slice(0, offset).split('\n').length; } function violations(relPath, text) { const found = []; - for (const { label, pattern } of RETIRED_CLI_PATTERNS) { + for (const { label, pattern } of RETIRED_PATTERNS) { pattern.lastIndex = 0; for (const match of text.matchAll(pattern)) { found.push(`${relPath}:${lineOf(text, match.index)} ${label}: ${JSON.stringify(match[0])}`); @@ -157,14 +164,14 @@ function scopeFiles() { return files.filter((file) => file !== REGISTRY); } -test('no retired CLI spelling remains in help, README, docs, or installed guidance', () => { +test('no retired CLI or dashboard spelling remains in source, README, current docs, or installed guidance', () => { const found = []; for (const file of scopeFiles()) { const text = fs.readFileSync(file, 'utf8'); found.push(...violations(path.relative(ROOT, file), text)); } found.push(...registryViolations()); - assert.deepEqual(found, [], `retired CLI spellings remain:\n${found.join('\n')}`); + assert.deepEqual(found, [], `retired CLI or dashboard spellings remain:\n${found.join('\n')}`); }); test('the registry skip is narrow: dated watch[].history[].note strings still contain the old spellings they document', () => { diff --git a/tests/kit/refresh.test.mjs b/tests/kit/refresh.test.mjs index 8dcf0fae..0d5fed7e 100644 --- a/tests/kit/refresh.test.mjs +++ b/tests/kit/refresh.test.mjs @@ -163,7 +163,7 @@ test('runRefresh refuses an unknown strength or a missing stage', async () => { test('the refresh operation never references the paid connection check', () => { // Every module that runs refresh stages belongs in this list, including any // future server-side refresh module. - for (const file of [REFRESH_SOURCE]) { + for (const file of [REFRESH_SOURCE, path.join(PKG_ROOT, 'src/lib/dashboard/refresh-api.mjs')]) { const source = fs.readFileSync(file, 'utf8'); assert.doesNotMatch(source, /checkConnection|host-health-connected/, file); } diff --git a/tests/kit/system-summary.test.mjs b/tests/kit/system-summary.test.mjs index dbb9e4f2..a8d6553e 100644 --- a/tests/kit/system-summary.test.mjs +++ b/tests/kit/system-summary.test.mjs @@ -592,13 +592,13 @@ test('the projects note says how many imported copies discovery set aside, and n // ── The page reads the slim endpoint ──────────────────────────────────────── -test('loadSystem fetches /api/system/summary, deep refresh parameters included', async () => { +test('loadSystem only re-reads /api/system/summary', async () => { const urls = []; const fetchImpl = (url) => { urls.push(url); return Promise.resolve({ json: () => Promise.resolve(systemSummaryPayload(fullPayload(1))) }); }; const { projects } = systemClient({ fetchImpl }); await projects.loadSystem(); await projects.loadSystem(true, false); - assert.deepEqual(urls, ['/api/system/summary', '/api/system/summary?refresh=deep&trees=0']); + assert.deepEqual(urls, ['/api/system/summary', '/api/system/summary']); }); // ── The routes ────────────────────────────────────────────────────────────── @@ -661,15 +661,14 @@ test('GET /api/system/summary serves the projection; GET /api/system stays compl assert.ok(complete.catalog.items[0].presence[0].itemPath); }); -test('GET /api/system/summary?refresh=deep starts the scan and answers with its running state', async (t) => { +test('GET /api/system/summary rejects measurement queries before reading the collector', async (t) => { const collector = fakeCollector(); const cwd = tempDir('ak-system-summary'); const server = await startDashboard({ port: 0, cwd, system: collector, usage: {}, ...hermeticMaintenance() }); t.after(() => server.close()); const r = await request(server, '/api/system/summary?refresh=deep&trees=0'); - assert.equal(r.status, 200); + assert.equal(r.status, 400); const body = JSON.parse(r.body); - assert.deepEqual(collector.calls.refreshDeep, [{ includeProjectTrees: false }]); - assert.deepEqual(body.scan, { running: true, phase: 'catalog' }); - assert.equal('artifacts' in body.catalog, false); + assert.deepEqual(body, { error: 'start a refresh with POST /api/refresh' }); + assert.deepEqual(collector.calls.refreshDeep, []); }); diff --git a/tests/ui/dashboard-ui.mjs b/tests/ui/dashboard-ui.mjs index 5fff42c7..8ed62e36 100644 --- a/tests/ui/dashboard-ui.mjs +++ b/tests/ui/dashboard-ui.mjs @@ -789,10 +789,9 @@ const MAINTENANCE_PAYLOAD = { receipts: [], }; -let chainedMaintenanceScans = 0; const MAINTENANCE_STUB = { async report() { return MAINTENANCE_PAYLOAD; }, - async scan() { chainedMaintenanceScans += 1; return MAINTENANCE_PAYLOAD; }, + async scan() { return MAINTENANCE_PAYLOAD; }, async plan() { return {}; }, }; @@ -1133,6 +1132,12 @@ async function main() { console.log(`\ncorpus: ${REAL ? 'REAL (~/.claude, ~/.codex)' : 'fixtures (deterministic)'}`); console.log(`cache : ${cachePath} (temp — your real index is untouched)\n`); + const refreshStageGates = new Map(); + const refreshStages = Object.fromEntries(['machine', 'maintenance', 'inventory', 'live', 'local'].map((id) => [id, async () => { + const gate = refreshStageGates.get(id); + if (gate) await gate; + return { ok: true }; + }])); const srv = await startDashboard({ port: 0, fetchStatus: STATUS_STUB, @@ -1145,6 +1150,7 @@ async function main() { modelScopeKey: 'ab'.repeat(32), system: SYSTEM_STUB, maintenance: MAINTENANCE_STUB, + refreshStages, }); const ORIGIN = new URL(srv.url).origin; const modelHeaders = { 'x-dash-token': srv.token }; @@ -1184,11 +1190,9 @@ async function main() { const group = document.getElementById('secondary-system')?.getBoundingClientRect(); const tabs = document.getElementById('system-seg')?.getBoundingClientRect(); const status = document.getElementById('system-freshness')?.getBoundingClientRect(); - const button = document.getElementById('sys-rescan'); return { statusText: document.getElementById('sys-asof')?.innerText, running: document.getElementById('system-freshness')?.getAttribute('data-running'), - buttonHidden: button?.hidden, statusBesideTabs: !!status && !!tabs && status.top < tabs.bottom && status.bottom > tabs.top && status.left >= tabs.right + 12, trailingSegmentSpace: Math.abs((tabs?.right ?? 0) @@ -1199,10 +1203,9 @@ async function main() { documentFits: document.documentElement.scrollWidth <= globalThis.innerWidth, }; }); - check('running full-scan progress sits beside a content-width System menu on wide screens', + check('running machine measurement progress sits beside a content-width System menu on wide screens', runningScanLayout.running === '1' - && /Full scan running.*Ranking disk use.*15 of 15/.test(runningScanLayout.statusText ?? '') - && runningScanLayout.buttonHidden === true + && /Machine measurement running.*Ranking disk use.*15 of 15/.test(runningScanLayout.statusText ?? '') && runningScanLayout.statusBesideTabs && runningScanLayout.trailingSegmentSpace < 6 && runningScanLayout.statusInsideGroup @@ -1226,7 +1229,7 @@ async function main() { ? getComputedStyle(document.getElementById('sys-asof')).whiteSpace : null, }; }); - check('narrow System navigation scrolls internally while scan status stays in its own rail', + check('narrow System navigation scrolls internally while measurement status stays in its own rail', narrowScanLayout.documentFits && narrowScanLayout.tabsScrollInternally && narrowScanLayout.statusBelowTabs @@ -1264,9 +1267,37 @@ async function main() { `blocked-storage startup was ${JSON.stringify(blockedStorageStartup)} with status ${blockedStatus.join(',')}`); await storageBlockedPage.close(); + const idlePage = await browser.newPage(); + const idleRequests = []; + idlePage.on('request', request => { + const route = new URL(request.url()).pathname; + if (route === '/api/status' || route === '/api/refresh') idleRequests.push({ route, method: request.method() }); + }); + await idlePage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }); + await idlePage.click('#poll-ivl'); + await idlePage.click('#poll-menu [data-ms="15000"]'); + const idleDeadline = Date.now() + 35_000; + while (idleRequests.filter(request => request.route === '/api/status').length < 3 && Date.now() < idleDeadline) { + await idlePage.waitForTimeout(250); + } + check('two idle poll ticks issue no /api/refresh request or POST', + idleRequests.filter(request => request.route === '/api/status').length >= 3 + && idleRequests.every(request => request.route !== '/api/refresh'), + JSON.stringify(idleRequests)); + await idlePage.close(); + const page = await browser.newPage({ viewport: { width: 1440, height: 900 }, locale: 'en-US', timezoneId: 'America/Los_Angeles', }); + const refreshRequests = []; + const statusRequests = []; + const systemSummaryRequests = []; + page.on('request', (request) => { + const pathname = new URL(request.url()).pathname; + if (pathname === '/api/refresh') refreshRequests.push({ method: request.method(), body: request.method() === 'POST' ? request.postDataJSON() : null }); + if (pathname === '/api/status') statusRequests.push(request.url()); + if (pathname === '/api/system/summary') systemSummaryRequests.push(request.url()); + }); // Anything the page logs as an error, or any request it fails, is a defect — // collected globally so a failure in one view is not silently swallowed. @@ -1297,7 +1328,8 @@ async function main() { page.on('console', (m) => { if (m.type() !== 'error') return; const loc = m.location(); - if (/status of 409 \(Conflict\)/.test(m.text()) && loc?.url && expectedHttpConsoleErrors.delete(loc.url)) return; + if (/status of (?:409 \(Conflict\)|503 \(Service Unavailable\))/.test(m.text()) + && loc?.url && expectedHttpConsoleErrors.delete(loc.url)) return; const where = loc?.url ? ` @ ${loc.url}` : ''; consoleErrors.push(`${m.text()}${where}`); }); @@ -1424,7 +1456,6 @@ async function main() { // check: the inventory stub answers `running` that many times, then flips // scanRequired off so the workspace's bounded polling sees the built page. let maintenanceBuildPollsRemaining = 0; - let maintenanceRunningPollsServed = 0; // Per-label override for a coverage entry's `filesystem` flag, applied // on top of whatever the sentinel fixture's coverage() helper produced // (which never sets `filesystem` at all, exercising the "flag absent -> @@ -1472,7 +1503,8 @@ async function main() { for (const entry of entries) { counts[entry.lane] += 1; lanes[entry.lane].push(entry); } return { lanes, counts, entries }; } - await page.route(/\/api\/maintenance\/v2\//, async (route) => { + const maintenanceV2Route = /\/api\/maintenance\/v2\//; + const maintenanceV2Stub = async (route) => { const request = route.request(); const url = new URL(request.url()); const pathname = url.pathname; @@ -1496,7 +1528,6 @@ async function main() { maintenanceInventoryRequests.push(params); if (maintenanceBuildPollsRemaining > 0) { maintenanceBuildPollsRemaining -= 1; - maintenanceRunningPollsServed += 1; if (maintenanceBuildPollsRemaining === 0) maintenanceScanRequired = false; return reply(200, { scanRequired: true, total: 0, groups: [], facetCounts: {}, sortGroups: [], partialSources: [], @@ -1630,7 +1661,12 @@ async function main() { } if (request.method() === 'GET' && pathname === '/api/maintenance/v2/activity') { return reply(200, buildActivity({ - receipts: [INTERRUPTED_RECEIPT], dispositions: [], recipeEvents: [], scanHistory: [], inProgress: [], + receipts: [INTERRUPTED_RECEIPT], dispositions: [], recipeEvents: [], inProgress: [], + scanHistory: [ + { sourceId: 'src-claude', environmentId: 'env-local', label: 'Claude user configuration', state: 'complete', completedAt: '2026-09-07T12:00:00.000Z', visited: 12 }, + { sourceId: 'src-claude', environmentId: 'env-local', label: 'Claude user configuration', state: 'paused', recordedAt: '2026-09-09T12:00:00.000Z', completedAt: null, visited: 18 }, + { sourceId: 'src-other', environmentId: 'env-local', label: 'Codex configuration', state: 'complete', completedAt: '2026-09-08T12:00:00.000Z', visited: 8 }, + ], })); } const receiptMatch = pathname.match(/^\/api\/maintenance\/v2\/receipts\/([^/]+)$/); @@ -1762,7 +1798,8 @@ async function main() { }); } return reply(404, { code: 'NOT_FOUND' }); - }); + }; + await page.route(maintenanceV2Route, maintenanceV2Stub); // ── Refresh evidence (MNT-DSC-010): the workspace's explicit control reuses // the v1 endpoint verbatim, `?refresh=scan` then polling until settled. ── let maintenanceProviderPollCount = 0; @@ -1816,6 +1853,11 @@ async function main() { // connections. DOM readiness plus the application shell is the stable // navigation contract; network-idle can never be guaranteed by a Live UI. await page.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }); + check('Refresh and Reload controls replace the retired scan buttons', + await page.locator('#refresh-run').count() === 1 + && await page.locator('#refresh-strength').count() === 1 + && await page.locator('#poll-now').getAttribute('aria-label') === 'Reload — re-read this view; runs no checks' + && await page.locator('#sys-rescan, #mnt-check-providers, #mnt-remeasure, #host-health-refresh').count() === 0); await page.waitForSelector('#panel-overview', { state: 'attached' }); // ── ADR-0026 · About leads the bar but must NOT hijack the landing view ── @@ -2588,7 +2630,7 @@ async function main() { check('a fresh install names coverage gaps and keeps measurement in the toolbar only', /^4 sources have not been scanned yet\./.test((freshInstallBanner || '').trim()) && await page.locator('#mnt-partial button').count() === 0 - && await page.isVisible('#mnt-remeasure') + && await page.locator('#mnt-remeasure').count() === 0 && !/Fresh source/.test(freshInstallBanner || ''), `fresh-install banner read ${JSON.stringify(freshInstallBanner)}`); @@ -2602,8 +2644,7 @@ async function main() { await page.fill('#mnt-search', ''); await page.waitForFunction(() => document.querySelectorAll('#mnt-results .mnt-row').length > 3); - // ── Refresh evidence (MNT-DSC-010): explicit, labeled; disables Apply/ - // Undo while it runs; never fires on its own ── + // The shared Refresh control owns provider checks; opening Maintenance does not start one. await page.click('[data-mnt-dest="guidance"]'); await page.waitForSelector('#mnt-tab-guidance[aria-selected="true"]'); await page.waitForSelector('[data-mnt-plan-plc]'); @@ -2611,29 +2652,8 @@ async function main() { check('MNT-GUD-001: Guidance has exactly the visible lanes Can apply here, Steps available, Decisions to make, Updates available, and Recovery to finish', JSON.stringify(guidanceLaneLabels) === JSON.stringify([ 'Can apply here', 'Steps available', 'Decisions to make', 'Updates available', 'Recovery to finish', - ]), - `guidance lane labels read ${JSON.stringify(guidanceLaneLabels)}`); - check('a provider check never starts on its own', - maintenanceCheckProvidersReads === 0, 'the workspace probed providers without an explicit click'); - const providerRefreshResponse = page.waitForResponse((response) => new URL(response.url()).searchParams.get('refresh') === 'scan'); - await page.click('#mnt-check-providers'); - await providerRefreshResponse; - // The button disables synchronously; Apply's disabled attribute follows - // once the guidance re-render that mntCheckProviders() triggers resolves. - await page.waitForFunction(() => document.querySelector('[data-mnt-plan-plc]')?.disabled === true, null, { timeout: 5000 }); - const providersRunning = await page.evaluate(() => ({ - buttonDisabled: document.getElementById('mnt-check-providers')?.disabled, - applyDisabled: document.querySelector('[data-mnt-plan-plc]')?.disabled, - })); - check('Refresh evidence is explicit, labeled, and disables Apply while it runs', - maintenanceCheckProvidersReads === 1 && providersRunning.buttonDisabled === true - && providersRunning.applyDisabled === true, - `providers-running state was ${JSON.stringify(providersRunning)}`); - await page.waitForFunction(() => document.getElementById('mnt-check-providers')?.disabled === false, null, { timeout: 8000 }); - check('the provider check settles and re-enables Apply', - await page.$eval('[data-mnt-plan-plc]', (button) => button.disabled === false), - 'Apply stayed disabled after the provider check settled'); - + ])); + check('a provider check never starts on its own', maintenanceCheckProvidersReads === 0); // ── Dispositions (MNT-GUD-009/011): explained before confirmation, one // exact guidanceId per write ── await page.waitForSelector('.mnt-dispositions [data-mnt-disposition-open="acknowledged"]'); @@ -2716,6 +2736,18 @@ async function main() { // typed-confirmation dialog ── await page.click('[data-mnt-dest="activity"]'); await page.waitForSelector('#mnt-tab-activity[aria-selected="true"]'); + await page.waitForSelector('#mnt-scan-history .mnt-history-day'); + const scanRows = await page.$$eval('#mnt-scan-history tbody', (groups) => groups.map((group) => ({ + heading: group.querySelector('.mnt-history-day')?.textContent?.trim(), + rows: [...group.querySelectorAll('tr:not(.mnt-history-day)')].map((row) => row.textContent?.trim()), + }))); + check('paused scan renders at its recorded time ahead of older completed scans', + scanRows.length === 3 && /Paused/.test(scanRows[0].rows[0]) + && /Claude user configuration/.test(scanRows[0].rows[0]) + && !/Time not recorded/.test(scanRows[0].rows[0]) + && /Codex configuration/.test(scanRows[1].rows[0]) + && /Complete/.test(scanRows[2].rows[0]), + `scan rows read ${JSON.stringify(scanRows)}`); await page.waitForSelector('[data-mnt-audit-receipt]'); const auditTriggerLabel = await page.textContent('[data-mnt-audit-receipt]'); check('MNT-RCV-001: an interrupted receipt offers Audit interruption, not generic Verify again', @@ -2776,12 +2808,6 @@ async function main() { `export requests were ${JSON.stringify(maintenanceExportRequests)}`); await page.click('#mnt-receipt-close'); - // ── Discovery: a light smoke check of the fourth destination. Waits for - // the CONTENT of #mnt-scan-progress specifically (not just an
  • in - // #mnt-automatic-sources, whose two rows are static and already present - // from the earlier group-collapse fixture's own stale Discovery visits) - // — a fresh fetch against base()'s real coverage can otherwise still be - // in flight when a weaker wait resolves on leftover data. ── await page.click('[data-mnt-dest="discovery"]'); await page.waitForSelector('#mnt-tab-discovery[aria-selected="true"]'); await page.waitForFunction(() => /Claude user configuration/.test( @@ -2893,37 +2919,22 @@ async function main() { )); const failedRefreshEmpty = await visibleText(page, '#mnt-results'); const failedRefreshStatus = await page.textContent('#mnt-status'); - check('a failed lastRefresh names "did not complete" plus the sanitized message, never the raw code, and keeps Refresh evidence available', + check('a failed lastRefresh names "did not complete" plus the sanitized message, never the raw code, and keeps Refresh available', /did not complete/.test(failedRefreshEmpty) && /did not respond before the timeout/.test(failedRefreshEmpty) && !/PROVIDER_TIMEOUT/.test(failedRefreshEmpty) && !/PROVIDER_TIMEOUT/.test(failedRefreshStatus || '') && /did not respond before the timeout/.test(failedRefreshStatus || '') - && await page.isEnabled('#mnt-check-providers'), + && await page.isEnabled('#refresh-run'), `empty state read ${JSON.stringify(failedRefreshEmpty)}, status read ${JSON.stringify(failedRefreshStatus)}`); maintenanceInventoryLastRefresh = null; - // ── D5: scanRequired points at Refresh evidence, and settling it - // refreshes Inventory in place, with no page reload. Hops off Inventory - // and back so the destination switch re-fetches against the reset - // (no-lastRefresh) fixture state above, without duplicating route - // handlers on a fresh page. ── + // A missing inventory points to the shared Refresh control. await page.click('[data-mnt-dest="discovery"]'); await page.click('[data-mnt-dest="inventory"]'); await page.waitForFunction(() => /No inventory has been built yet/.test( document.getElementById('mnt-results')?.innerText || '', )); const scanRequiredEmpty = await visibleText(page, '#mnt-results'); - check('the scanRequired empty state points at Refresh evidence on this workspace, not a System full scan', - /Refresh evidence/.test(scanRequiredEmpty) && !/full scan from System/i.test(scanRequiredEmpty), - `scanRequired empty state read ${JSON.stringify(scanRequiredEmpty)}`); - await page.click('#mnt-check-providers'); - await page.waitForFunction(() => document.getElementById('mnt-check-providers')?.disabled === false, null, { timeout: 8000 }); - await page.waitForSelector('#mnt-results .mnt-row', { timeout: 8000 }); - check('settling Refresh evidence clears scanRequired and shows Inventory results in place, with no reload', - await page.$$eval('#mnt-results .mnt-row', (els) => els.length) > 0, - 'Inventory did not refresh in place once the provider check settled'); - check('D8: the workspace polled the inventory through the running build (bounded, 1.5 s apart) instead of re-fetching once', - maintenanceRunningPollsServed >= 2 && maintenanceBuildPollsRemaining === 0, - `running polls served: ${maintenanceRunningPollsServed}, remaining ${maintenanceBuildPollsRemaining}`); + check('the scanRequired empty state points at Refresh', /Refresh/.test(scanRequiredEmpty)); maintenanceScanRequired = false; // ── D8: while the server reports a running build, the empty state says @@ -2937,68 +2948,13 @@ async function main() { document.getElementById('mnt-results')?.innerText || '', )); const runningEmpty = await visibleText(page, '#mnt-results'); - check('a running lastRefresh renders "Building the inventory…" and keeps Refresh evidence available', + check('a running lastRefresh renders "Building the inventory…" and keeps Refresh available', /Building the inventory/.test(runningEmpty) && !/not been built yet/.test(runningEmpty) - && await page.isEnabled('#mnt-check-providers'), + && await page.isEnabled('#refresh-run'), `running empty state read ${JSON.stringify(runningEmpty)}`); maintenanceInventoryLastRefresh = null; maintenanceScanRequired = false; - // ── Re-measure machine (System Full scan, exposed beside Refresh - // evidence): delegates to System's own #sys-rescan, then blocks writes - // (mntWritesBlocked) for the whole measurement + provider-check + - // inventory-rebuild chain. Uses its own temporary /api/system route - // rather than the real SYSTEM_STUB — that stub's systemDeepScans counter - // is asserted against a clean slate by the dedicated System-tab Rescan - // tests later in this run, and this block must neither depend on nor - // perturb that count. ── - await page.click('[data-mnt-dest="guidance"]'); - await page.waitForSelector('[data-mnt-plan-plc]'); - await page.waitForFunction(() => document.querySelector('[data-mnt-plan-plc]')?.disabled === false); - let remeasureSystemReadCount = 0; - let remeasureDeepScanRequests = 0; - // Deterministic on REQUEST COUNT, not wall-clock: the deep-scan kickoff - // itself is always the first read (reports running), every read after is - // settled. This cannot race system-projects.mjs's own poll cadence and - // mntPollSystemMeasurement's independent one against a Node-side timer. - await page.route(/\/api\/system(\/summary)?(\?|$)/, (route) => { - const reqUrl = new URL(route.request().url()); - if (reqUrl.searchParams.get('refresh') === 'deep') remeasureDeepScanRequests += 1; - remeasureSystemReadCount += 1; - if (remeasureSystemReadCount === 2) { - // Deep measurement also completes a fresh provider check and inventory build. - maintenanceCheckProvidersReads += 1; - maintenanceProviderPollCount = 2; - } - return route.fulfill({ - status: 200, contentType: 'application/json', - body: JSON.stringify({ ...SYSTEM_PAYLOAD, scan: { ...SYSTEM_PAYLOAD.scan, running: remeasureSystemReadCount <= 1 } }), - }); - }); - await page.click('#mnt-remeasure'); - // mntRemeasureMachine() re-renders the active destination the instant it - // sets MNT.remeasureBusy — before it even clicks #sys-rescan — so both - // of these are observable synchronously, exactly like Refresh evidence. - const remeasureStarted = await page.evaluate(() => ({ - remeasureDisabled: document.getElementById('mnt-remeasure')?.disabled, - applyDisabled: document.querySelector('[data-mnt-plan-plc]')?.disabled, - })); - check('Re-measure machine disables itself and blocks writes (mntWritesBlocked) the instant it starts', - remeasureStarted.remeasureDisabled === true && remeasureStarted.applyDisabled === true, - `remeasure-start state was ${JSON.stringify(remeasureStarted)}`); - await page.waitForFunction(() => document.getElementById('mnt-remeasure')?.disabled === false, null, { timeout: 15_000 }); - check('Re-measure machine delegates to #sys-rescan (a real deep scan) and settles, re-enabling itself and Apply', - remeasureDeepScanRequests === 1 && await page.$eval('[data-mnt-plan-plc]', (b) => b.disabled === false), - `deep scan requests: ${remeasureDeepScanRequests}, Apply stayed disabled after Re-measure machine settled: ${await page.$eval('[data-mnt-plan-plc]', (b) => b.disabled)}`); - // A stale scheduled System poll used to restart a completed owned scan as - // an external operation, waiting forever for a second evidence generation. - await page.waitForTimeout(3200); - check('a completed remeasurement stays settled after the System poll interval', - await page.isEnabled('#mnt-remeasure') && await page.isEnabled('[data-mnt-plan-plc]') - && remeasureSystemReadCount === 2, - `system reads: ${remeasureSystemReadCount}; operation: ${await page.textContent('#mnt-check-providers-status')}`); - await page.unroute(/\/api\/system(\/summary)?(\?|$)/); - // ── #system/catalog redirects to Maintenance Inventory (ADR-0048) ── await page.evaluate(() => { location.hash = '#system/catalog'; }); await page.reload({ waitUntil: 'domcontentloaded' }); @@ -3518,8 +3474,6 @@ async function main() { text: el?.textContent.trim(), stale: el?.getAttribute('data-stale'), title: el?.getAttribute('title'), - rescanDisabled: document.getElementById('sys-rescan')?.disabled, - fullScanLabel: document.getElementById('sys-rescan')?.innerText, live: el?.getAttribute('aria-live'), }; }); @@ -3528,28 +3482,12 @@ async function main() { `the freshness label read ${JSON.stringify(freshness)} — the snapshot is nine days old`); check('past the staleness horizon the label nudges without scanning', freshness.stale === '1' && /stale/i.test(String(freshness.text)) - && freshness.rescanDisabled === false && /Full scan/.test(String(freshness.fullScanLabel)) && freshness.live === 'polite', `staleness presentation was ${JSON.stringify(freshness)}`); check('opening System never starts a deep scan', systemDeepScans === 0, `${systemDeepScans} deep scan(s) had already run after opening the area and all five views`); - const deepResponse = page.waitForResponse( - (r) => r.url().includes('/api/system') && r.url().includes('refresh=deep'), - { timeout: 8000 }, - ).catch(() => null); - await page.click('#sys-rescan'); - await deepResponse; - await page.waitForTimeout(200); - check('Rescan is the only thing that starts a deep scan, and it starts exactly one', - systemDeepScans === 1, - `the collector saw ${systemDeepScans} deep scan(s) after one Rescan click`); - await page.waitForTimeout(50); - check('a successful deep System rescan refreshes Maintenance provider evidence once', - chainedMaintenanceScans === 1, - `the Maintenance service saw ${chainedMaintenanceScans} scan(s)`); - // ── Observability: execution workspace + synchronized evidence ── await page.click('[data-tab="observability"]'); await page.waitForSelector('#live-nodes .live-node', { timeout: 8000 }); @@ -5771,6 +5709,330 @@ async function main() { check('and survives a reload rather than snapping back to the default', c2.expanded === 'true' && c2.hidden === false, JSON.stringify(c2)); + // Refresh is the only route that starts checks. Its stage state is server authored. + check('idle dashboard requested no refresh state or operation', refreshRequests.length === 0, + JSON.stringify(refreshRequests)); + await page.click('#tab-system'); + await page.click('[data-system-view="maintenance"]'); + await page.click('[data-mnt-dest="guidance"]'); + await page.waitForSelector('[data-mnt-plan-plc]'); + // The audit and undo actions are conditional on retained receipts. Keep + // their actual selectors in the fixture while this run has no such receipt. + await page.evaluate(() => { + const fixture = globalThis.document.createElement('div'); + fixture.id = 'refresh-write-fixture'; + fixture.hidden = true; + fixture.innerHTML = ''; + globalThis.document.body.appendChild(fixture); + }); + const stageRelease = new Map(); + for (const id of ['maintenance', 'inventory', 'local']) { + refreshStageGates.set(id, new Promise((resolve) => stageRelease.set(id, resolve))); + } + const statusReadsBefore = statusRequests.length; + await page.click('#refresh-run'); + await page.waitForFunction(() => document.getElementById('refresh-status')?.textContent.includes('Refreshing Maintenance evidence')); + check('Refresh status names the server stage and elapsed time', + /Refreshing Maintenance evidence · \d+s/.test(await page.locator('#refresh-status').innerText())); + check('Refresh starts exactly one local POST', refreshRequests.filter((r) => r.method === 'POST').length === 1 + && JSON.stringify(refreshRequests.find((r) => r.method === 'POST')?.body) === JSON.stringify({ strength: 'local' })); + await page.waitForFunction(() => document.querySelector('[data-mnt-plan-plc]')?.disabled === true); + check('Maintenance Apply, Undo and Record are disabled during Refresh', + await page.locator('[data-mnt-plan-plc]').first().isDisabled() + && await page.locator('#refresh-write-fixture [data-mnt-undo-receipt]').isDisabled() + && await page.locator('#refresh-write-fixture [data-mnt-reconcile-receipt]').isDisabled()); + await page.click('[data-mnt-dest="inventory"]'); + await page.waitForSelector('#mnt-tab-inventory[aria-selected="true"]'); + const activeMaintenanceReadsBefore = maintenanceInventoryRequests.length; + for (const [id, label] of [['maintenance', 'Rebuilding the inventory'], ['inventory', 'Re-checking local evidence and versions']]) { + stageRelease.get(id)(); + await page.waitForFunction((text) => document.getElementById('refresh-status')?.textContent.includes(text), label); + } + stageRelease.get('local')(); + await page.waitForFunction(() => document.getElementById('refresh-status')?.textContent === 'Refresh complete.'); + for (let attempt = 0; attempt < 30 && maintenanceInventoryRequests.length <= activeMaintenanceReadsBefore; attempt++) { + await page.waitForTimeout(100); + } + await page.click('[data-mnt-dest="guidance"]'); + await page.waitForFunction(() => document.querySelector('[data-mnt-plan-plc]')?.disabled === false); + check('Maintenance write controls are restored after Refresh', + await page.locator('[data-mnt-plan-plc]').first().isEnabled() + && await page.locator('#refresh-write-fixture [data-mnt-undo-receipt]').isEnabled() + && await page.locator('#refresh-write-fixture [data-mnt-reconcile-receipt]').isEnabled()); + await page.locator('#refresh-write-fixture').evaluate(element => element.remove()); + await page.waitForTimeout(200); + check('Refresh re-reads the active Maintenance view and host readiness after completion', + statusRequests.length > statusReadsBefore && maintenanceInventoryRequests.length > activeMaintenanceReadsBefore); + const localRefreshReads = refreshRequests.filter(request => request.method === 'GET').length; + await page.waitForTimeout(1700); + check('completed Refresh stops status polling', refreshRequests.filter(request => request.method === 'GET').length === localRefreshReads); + await page.selectOption('#refresh-strength', 'machine'); + await page.check('#refresh-project-trees'); + await page.click('#refresh-run'); + await page.waitForFunction(() => document.getElementById('refresh-status')?.textContent === 'Refresh complete.'); + check('machine Refresh carries the project tree scope', refreshRequests.filter((r) => r.method === 'POST').length === 2 + && JSON.stringify(refreshRequests.filter((r) => r.method === 'POST')[1].body) === JSON.stringify({ strength: 'machine', projectTrees: true })); + const postCount = refreshRequests.filter((r) => r.method === 'POST').length; + const reloadRequests = []; + const captureReload = request => reloadRequests.push(request.method()); + await page.click('#poll-play'); + page.on('request', captureReload); + await page.waitForTimeout(3100); + const reloadResponse = page.waitForResponse(response => new URL(response.url()).pathname === '/api/status'); + await page.click('#poll-now'); + await reloadResponse; + page.off('request', captureReload); + check('Reload issues only GETs and starts no checks', reloadRequests.length > 0 + && reloadRequests.every(method => method === 'GET') + && refreshRequests.filter((r) => r.method === 'POST').length === postCount); + for (const view of ['summary', 'storage', 'projects']) { + await page.click(`[data-system-view="${view}"]`); + await page.waitForTimeout(3100); + const before = systemSummaryRequests.length; + await page.click('#poll-now'); + await page.waitForTimeout(300); + check(`Reload re-reads active System ${view} without measuring`, + systemSummaryRequests.length > before + && systemSummaryRequests.slice(before).every(url => !new URL(url).searchParams.has('refresh')), + `System reads before/after: ${before}/${systemSummaryRequests.length}`); + } + await page.click('[data-system-view="storage"]'); + await page.selectOption('#refresh-strength', 'local'); + const beforeCompletion = systemSummaryRequests.length; + await page.click('#refresh-run'); + await page.waitForFunction(() => document.getElementById('refresh-status')?.textContent === 'Refresh complete.'); + await page.waitForTimeout(300); + check('completed Refresh re-reads active System Storage without measuring', + systemSummaryRequests.length > beforeCompletion + && systemSummaryRequests.slice(beforeCompletion).every(url => !new URL(url).searchParams.has('refresh')), + `System reads before/after: ${beforeCompletion}/${systemSummaryRequests.length}`); + await page.click('#poll-ivl'); + await page.click('#poll-menu [data-ms="15000"]'); + const backgroundBefore = systemSummaryRequests.length; + await page.click('#poll-play'); + await page.waitForTimeout(16_000); + await page.click('#poll-play'); + check('background poll keeps System Storage on the existing cheap-read policy', + systemSummaryRequests.length === backgroundBefore, + `System reads before/after background tick: ${backgroundBefore}/${systemSummaryRequests.length}`); + console.log(`refresh requests: idle 0; local POST 1, GET ${localRefreshReads}; machine POST 1, total GET ${refreshRequests.filter(request => request.method === 'GET').length}; Reload ${reloadRequests.length} GET`); + await page.setViewportSize({ width: 390, height: 844 }); + await page.evaluate(() => { localStorage.setItem('ak-dash-theme', 'light'); location.reload(); }); + await page.waitForFunction(() => document.documentElement.getAttribute('data-theme') === 'light'); + await page.screenshot({ path: path.join(SHOTS, 'refresh-header-light-390.png'), animations: 'disabled' }); + await page.evaluate(() => { localStorage.setItem('ak-dash-theme', 'dark'); location.reload(); }); + await page.waitForFunction(() => document.documentElement.getAttribute('data-theme') === 'dark'); + await page.screenshot({ path: path.join(SHOTS, 'refresh-header-dark-390.png'), animations: 'disabled' }); + check('Refresh header fits at 390px in light and dark themes', + await page.evaluate(() => { + const header = document.querySelector('.band'); + const refresh = document.querySelector('.refresh-control'); + return header.getBoundingClientRect().right <= globalThis.innerWidth + && refresh.getBoundingClientRect().right <= globalThis.innerWidth; + })); + await page.setViewportSize({ width: 1440, height: 900 }); + check('wide dark screenshot has the dark theme at capture time', + await page.evaluate(() => document.documentElement.getAttribute('data-theme') === 'dark' + && getComputedStyle(document.documentElement).getPropertyValue('--bg').trim() === '#000000')); + await page.screenshot({ path: path.join(SHOTS, 'refresh-header-dark-1440.png'), animations: 'disabled' }); + + // A rejected status read cannot release the write guard for an operation + // this page already started. The next valid state reconciles it. + await page.click('[data-system-view="maintenance"]'); + await page.click('[data-mnt-dest="guidance"]'); + await page.waitForSelector('[data-mnt-plan-plc]'); + let releaseStatusStage; + refreshStageGates.set('maintenance', new Promise(resolve => { releaseStatusStage = resolve; })); + let statusFailures = 2; + const rejectStatus = route => { + if (route.request().method() === 'GET' && statusFailures > 0) { + statusFailures--; + if (statusFailures === 1) expectedHttpConsoleErrors.add(route.request().url()); + return route.fulfill(statusFailures === 1 + ? { status: 503, contentType: 'application/json', body: JSON.stringify({ error: 'temporary status failure' }) } + : { status: 200, contentType: 'application/json', body: JSON.stringify({ running: 'unknown', stages: [] }) }); + } + return route.continue(); + }; + await page.route('**/api/refresh', rejectStatus); + const rejectedStatus = page.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' && response.status() === 503); + await page.click('#refresh-run'); + await rejectedStatus; + await page.waitForFunction(() => /retry/i.test(document.getElementById('refresh-status')?.textContent || '')); + check('rejected refresh-status GET keeps the owned operation and Maintenance writes blocked', + await page.locator('#refresh-run').isDisabled() + && await page.locator('[data-mnt-plan-plc]').first().isDisabled() + && /retry/i.test(await page.locator('#refresh-status').innerText())); + const readsAfterError = refreshRequests.filter(request => request.method === 'GET').length; + await page.waitForTimeout(1700); + check('refresh-status read retries after non-2xx JSON and keeps invalid success state blocked', + refreshRequests.filter(request => request.method === 'GET').length > readsAfterError + && await page.locator('#refresh-run').isDisabled() + && await page.locator('[data-mnt-plan-plc]').first().isDisabled()); + releaseStatusStage(); + await page.waitForFunction(() => document.getElementById('refresh-status')?.textContent === 'Refresh complete.', null, { timeout: 8000 }); + await page.waitForFunction(() => document.querySelector('[data-mnt-plan-plc]')?.disabled === false); + check('owned refresh completes after status recovery and then unblocks writes', + await page.locator('#refresh-run').isEnabled() + && await page.locator('[data-mnt-plan-plc]').first().isEnabled()); + await page.unroute('**/api/refresh', rejectStatus); + + // Two real dashboard pages can supersede a completed operation before its + // first polling GET. Test both a newer terminal state and a newer running + // state against the actual server operation, not a route-shaped fake. + activeMaintenanceInventory = structuredClone(SENTINEL_FIXTURES.base()); + const supersessionUpdate = activeMaintenanceInventory.guidanceEntries.find(entry => entry.lane === 'apply'); + Object.assign(supersessionUpdate, { verb: 'update', outcome: 'Update Claude plugin', + providerCapabilityId: 'claude-plugin:v1:update:user', + verifiedPremises: ['placement', 'installedVersion', 'published-update', 'consumers', 'impact'], + impact: { summary: 'Installs the verified newer frontend-design version.' } }); + async function latestRefresh() { + const response = await fetch(`${ORIGIN}/api/refresh`, { headers: modelHeaders }); + return response.json(); + } + async function waitServerRefresh(running, differentFrom) { + const deadline = Date.now() + 5000; + while (Date.now() < deadline) { + const state = await latestRefresh(); + const identity = state.operationId; + if (state.running === running && identity !== differentFrom && identity) return state; + await new Promise(resolve => setTimeout(resolve, 20)); + } + throw new Error('the fixture refresh did not reach the expected server state'); + } + async function exerciseSupersession(holdPeer) { + const firstPage = await browser.newPage(); + const peerPage = await browser.newPage(); + let releaseFirstPoll; + const firstPollGate = new Promise(resolve => { releaseFirstPoll = resolve; }); + let interceptFirstPoll = true; + let releasePeerStage; + const holdPeerStage = new Promise(resolve => { releasePeerStage = resolve; }); + try { + await firstPage.route(maintenanceV2Route, maintenanceV2Stub); + await Promise.all([ + firstPage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }), + peerPage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }), + ]); + await firstPage.route('**/api/refresh', async route => { + if (route.request().method() === 'GET' && interceptFirstPoll) { + interceptFirstPoll = false; + await firstPollGate; + } + await route.continue(); + }); + await firstPage.click('#tab-system'); + await firstPage.click('[data-system-view="maintenance"]'); + await firstPage.click('[data-mnt-dest="guidance"]'); + await firstPage.waitForSelector('[data-mnt-plan-plc]'); + refreshStageGates.set('maintenance', Promise.resolve()); + const firstPost = firstPage.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' + && response.request().method() === 'POST'); + await firstPage.click('#refresh-run'); + await firstPost; + const firstDone = await waitServerRefresh(false, null); + const firstIdentity = firstDone.operationId; + if (holdPeer) refreshStageGates.set('maintenance', holdPeerStage); + const peerPost = peerPage.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' + && response.request().method() === 'POST'); + await peerPage.click('#refresh-run'); + await peerPost; + const newer = await waitServerRefresh(holdPeer, firstIdentity); + const firstStatus = firstPage.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' + && response.request().method() === 'GET'); + releaseFirstPoll(); + await firstStatus; + await firstPage.waitForTimeout(100); + if (holdPeer) { + check('superseding running refresh keeps the first page and real Apply blocked', + !!newer.operationId && newer.operationId !== firstDone.operationId + && await firstPage.locator('#refresh-run').isDisabled() + && await firstPage.locator('[data-mnt-plan-plc]').first().isDisabled() + && /another refresh|supersed/i.test(await firstPage.locator('#refresh-status').innerText())); + releasePeerStage(); + await waitServerRefresh(false, firstIdentity); + await firstPage.waitForFunction(() => !document.getElementById('refresh-run')?.disabled, + null, { timeout: 8000 }).catch(() => {}); + } + check(`first page recovers from newer ${holdPeer ? 'running' : 'completed'} refresh without claiming its outcome`, + await firstPage.locator('#refresh-run').isEnabled() + && await firstPage.locator('[data-mnt-plan-plc]').first().isEnabled() + && /outcome unavailable|supersed/i.test(await firstPage.locator('#refresh-status').innerText()) + && !/Refresh complete\.|Refresh did not complete\./.test(await firstPage.locator('#refresh-status').innerText())); + } finally { + releaseFirstPoll(); + releasePeerStage(); + await Promise.all([firstPage.close(), peerPage.close()]); + } + } + await exerciseSupersession(false); + await exerciseSupersession(true); + + // A page that loses the single-flight POST race must respect the running + // operation reported in the 409 response before accepting Maintenance writes. + { + const ownerPage = await browser.newPage(); + const losingPage = await browser.newPage(); + let releaseOwner; + const ownerStage = new Promise(resolve => { releaseOwner = resolve; }); + try { + await losingPage.route(maintenanceV2Route, maintenanceV2Stub); + await Promise.all([ + ownerPage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }), + losingPage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }), + ]); + await losingPage.click('#tab-system'); + await losingPage.click('[data-system-view="maintenance"]'); + await losingPage.click('[data-mnt-dest="guidance"]'); + await losingPage.waitForSelector('[data-mnt-plan-plc]'); + refreshStageGates.set('maintenance', ownerStage); + await ownerPage.click('#refresh-run'); + await waitServerRefresh(true, null); + const conflict = losingPage.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' + && response.request().method() === 'POST' && response.status() === 409); + await losingPage.click('#refresh-run'); + await conflict; + check('409 refresh conflict blocks the losing page and real Apply while the winner runs', + await losingPage.locator('#refresh-run').isDisabled() + && await losingPage.locator('[data-mnt-plan-plc]').first().isDisabled()); + releaseOwner(); + await losingPage.waitForFunction(() => !document.getElementById('refresh-run')?.disabled, + null, { timeout: 8000 }).catch(() => {}); + check('409 losing page restores Apply with an outcome-unavailable message after the winner ends', + await losingPage.locator('[data-mnt-plan-plc]').first().isEnabled() + && /outcome unavailable/i.test(await losingPage.locator('#refresh-status').innerText())); + } finally { + releaseOwner(); + await Promise.all([ownerPage.close(), losingPage.close()]); + } + } + + // An unconfirmed POST cannot adopt an older terminal operation as proof + // that its own request ended. The page must retain the write guard. + { + const uncertainPage = await browser.newPage(); + try { + await uncertainPage.route(maintenanceV2Route, maintenanceV2Stub); + await uncertainPage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }); + await uncertainPage.click('#tab-system'); + await uncertainPage.click('[data-system-view="maintenance"]'); + await uncertainPage.click('[data-mnt-dest="guidance"]'); + await uncertainPage.waitForSelector('[data-mnt-plan-plc]'); + await uncertainPage.route('**/api/refresh', route => route.request().method() === 'POST' + ? route.abort() : route.continue()); + const staleRead = uncertainPage.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' + && response.request().method() === 'GET'); + await uncertainPage.click('#refresh-run'); + await staleRead; + check('uncertain POST does not adopt an older completed operation or release real Apply', + await uncertainPage.locator('#refresh-run').isDisabled() + && await uncertainPage.locator('[data-mnt-plan-plc]').first().isDisabled() + && /retry|unavailable/i.test(await uncertainPage.locator('#refresh-status').innerText())); + } finally { + await uncertainPage.close(); + } + } + // ── nothing errored anywhere along the way ── // A 404 from /api/session/ is CORRECT behaviour for a session that does // not exist — the route was changed to stop answering 200-with-a-null-body. diff --git a/tests/ui/host-readiness.mjs b/tests/ui/host-readiness.mjs index d8d8f491..247a47c9 100644 --- a/tests/ui/host-readiness.mjs +++ b/tests/ui/host-readiness.mjs @@ -19,7 +19,12 @@ const report = () => ({ checkedAt: '2026-09-20T10:00:00Z', scope: 'Dashboard lau host, status: 'ok', level: 'local', checks, evidenceKey: 'a'.repeat(64), canCheckConnection: true, checkedAt: '2026-09-20T10:00:00Z', connection: { state: 'not-run' }, target: { nativeDefault: true }, }])) }); - +const luminance = color => { + const rgb = color.match(/[\d.]+/g).slice(0,3).map(Number).map(value => value/255) + .map(value => value <= .04045 ? value/12.92 : ((value+.055)/1.055)**2.4); + return rgb[0]*.2126 + rgb[1]*.7152 + rgb[2]*.0722; +}; +const contrast = (a,b) => { const values=[luminance(a),luminance(b)].sort((x,y)=>y-x);return (values[0]+.05)/(values[1]+.05); }; test('all hosts have qualified OK, accessible details and explicitly confirmed connection checks', async t => { const browser = await launchChrome(); t.after(() => browser.close()); @@ -42,7 +47,7 @@ test('all hosts have qualified OK, accessible details and explicitly confirmed c return route.fulfill({contentType:'text/html',body:renderPage({name:'Health fixture',version:'test'}).replace(/]*>[\s\S]*?<\/script>/gi,'')}); }); await page.goto('http://health.test/'); - await page.addScriptTag({content:`${esc.toString()}\nfunction authHeaders(){return {'x-dash-token':'fixture'};}\n${source('usage')}\n${source('host-readiness')}\nwireHostHealth();`}); + await page.addScriptTag({content:`${esc.toString()}\nfunction authHeaders(){return {'x-dash-token':'fixture'};}\n${source('usage')}\nfunction refreshRunning(){return false;}\nfunction startRefresh(strength){(window.__refreshCalls ||= []).push(strength);return Promise.resolve(true); }\n${source('host-readiness')}\nwireHostHealth();`}); assert.equal(await page.locator('[data-health-host="codex"] .sp-status').innerText(),'Checking'); await page.evaluate(data=>globalThis.renderHostReadiness(data),report()); for(const [host,name] of [['claude','Claude Code'],['codex','Codex'],['opencode','OpenCode']]){ @@ -83,6 +88,20 @@ test('all hosts have qualified OK, accessible details and explicitly confirmed c await page.locator('.usage-source-details summary').click(); assert.match(await page.locator('.source-diagnostics').innerText(),/parse-yield-partial/); assert.equal(await page.locator('[data-health-host="codex"] .sp-status').innerText(),'OK'); + const unassessed=report(); + unassessed.hosts.claude={...unassessed.hosts.claude,checks:{...checks,configuration:{state:'unknown',reason:'Configuration was not assessed.'}}}; + await page.evaluate(data=>globalThis.renderHostReadiness(data),unassessed); + assert.equal(await page.locator('[data-health-host="claude"] .sp-status').innerText(),'Unknown'); + const iconColors=await page.evaluate(() => ['light','dark'].map(theme=>{ + globalThis.document.documentElement.setAttribute('data-theme',theme); + const chip=globalThis.document.querySelector('[data-health-host="codex"] .live-host'); + return {theme,fill:globalThis.getComputedStyle(chip.querySelector('path')).fill,background:globalThis.getComputedStyle(chip).backgroundColor}; + })); + for(const row of iconColors) { + console.log(`Codex icon ${row.theme}: ${row.fill} on ${row.background}, ${contrast(row.fill,row.background).toFixed(2)}:1`); + assert.ok(contrast(row.fill,row.background)>=3, + `Codex icon ${row.theme}: ${row.fill} on ${row.background}, ratio ${contrast(row.fill,row.background).toFixed(2)}:1`); + } await page.evaluate(()=>globalThis.renderHostReadiness(null)); assert.equal(await page.locator('[data-health-host="codex"] .sp-status').innerText(),'Unknown'); assert.equal(errors.length,0,errors.join('\n')); @@ -121,7 +140,7 @@ test('unmanaged hosts read their management state everywhere, with information-o return route.fulfill({contentType:'text/html',body:renderPage({name:'Health fixture',version:'test'}).replace(/]*>[\s\S]*?<\/script>/gi,'')}); }); await page.goto('http://health.test/'); - await page.addScriptTag({content:`${esc.toString()}\nfunction authHeaders(){return {'x-dash-token':'fixture'};}\n${source('usage')}\n${source('host-readiness')}\nwireHostHealth();`}); + await page.addScriptTag({content:`${esc.toString()}\nfunction authHeaders(){return {'x-dash-token':'fixture'};}\n${source('usage')}\nfunction refreshRunning(){return false;}\nfunction startRefresh(strength){(window.__refreshCalls ||= []).push(strength);return Promise.resolve(true); }\n${source('host-readiness')}\nwireHostHealth();`}); await page.evaluate(data=>globalThis.renderHostReadiness(data),unmanagedReport()); // Header pills: health for the managed host, the management words otherwise, never amber. @@ -144,10 +163,10 @@ test('unmanaged hosts read their management state everywhere, with information-o assert.match(await page.locator('#host-health-participation').innerText(),/not participating/i); assert.equal(await page.locator('#host-health-participation code').innerText(),'ak host pick --host claude,codex'); assert.equal(await page.locator('#host-health-participation [data-copy]').getAttribute('data-copy'),'ak host pick --host claude,codex'); - assert.equal(await page.locator('#host-health-refresh').innerText(),'Check again'); - await page.locator('#host-health-refresh').click(); - await page.waitForFunction(()=>globalThis.document.getElementById('host-health-message').textContent==='Check completed.'); - assert.deepEqual(requests,['/api/host-health/local']); + assert.equal(await page.locator('#host-health-run-refresh').innerText(),'Refresh'); + await page.locator('#host-health-run-refresh').click(); + assert.deepEqual(await page.evaluate(() => globalThis.__refreshCalls), ['local']); + assert.deepEqual(requests, []); await page.keyboard.press('Escape'); // Participation view (Overview → Hosts & Routing): one row per host, the hint copyable text. @@ -200,7 +219,7 @@ test('a dialog close that lands after the user moved on does not steal focus bac await page.route('http://health.test/**', route => route.fulfill({ contentType: 'text/html', body: renderPage({ name: 'Health fixture', version: 'test' }).replace(/]*>[\s\S]*?<\/script>/gi, '') })); await page.goto('http://health.test/'); - await page.addScriptTag({ content: `${esc.toString()}\nfunction authHeaders(){return {};}\n${source('usage')}\n${source('host-readiness')}\nwireHostHealth();` }); + await page.addScriptTag({ content: `${esc.toString()}\nfunction authHeaders(){return {};}\n${source('usage')}\nfunction refreshRunning(){return false;}\nfunction startRefresh(strength){(window.__refreshCalls ||= []).push(strength);return Promise.resolve(true); }\nfunction refreshRunning(){return false;}\nfunction startRefresh(strength){(window.__refreshCalls ||= []).push(strength);return Promise.resolve(true); }\n${source('host-readiness')}\nwireHostHealth();` }); await page.evaluate(data => globalThis.renderHostReadiness(data), report()); await page.locator('[data-health-host="claude"]').click(); // The race, made deterministic: close the dialog and move focus in the SAME diff --git a/tests/ui/maintenance-host-alignment.mjs b/tests/ui/maintenance-host-alignment.mjs index e64d7002..9f39385a 100644 --- a/tests/ui/maintenance-host-alignment.mjs +++ b/tests/ui/maintenance-host-alignment.mjs @@ -50,6 +50,7 @@ test('Host alignment view filters User and Project rows and offers exact registr function authHeaders(){return {};} function esc(value){return String(value).replace(/[&<>"']/g,function(c){return {'&':'&','<':'<','>':'>','"':'"',"'":'''}[c];});} function ago(){return '';} + function refreshRunning(){return false;} function beginMaintPreview(button, request){window.selectedPreview=request;} ${['maintenance-workspace','maintenance-operation','maintenance-cards','maintenance-filters','maintenance-guidance','maintenance-relationships','maintenance-inspector','maintenance-language-logos','maintenance-focus','maintenance-inventory'].map(clientSource).join('\n')} MNT.scope='user';MNT.view='host-alignment';wireMntInventory();wireMntInspector();wireMntGuidance();loadMntInventory(); From 8430757488d1c154611d0f857d6c32db73e6fcca Mon Sep 17 00:00:00 2001 From: Chris Phillipson Date: Tue, 29 Sep 2026 06:11:23 -0700 Subject: [PATCH 05/10] test: trace native learning dependencies on macOS (#277) * docs(trace): plan upstream native resolution evidence * feat(trace): record resolved Transformers and ORT package roots * test(nightly): retain macOS learning resolution traces * test(trace): preload hook with portable file URL * docs(archive): record native learning trace evidence --- .github/workflows/nightly.yml | 54 ++++++++- .../2026-09-29-native-learning-trace.md | 32 ++++++ .../2026-09-29-plan-upstream-native-trace.md | 19 ++++ docs/archive/README.md | 2 + scripts/trace-ort.mjs | 77 +++++++++++++ tests/kit/trace-ort.test.mjs | 105 ++++++++++++++++++ 6 files changed, 288 insertions(+), 1 deletion(-) create mode 100644 docs/archive/2026-09-29-native-learning-trace.md create mode 100644 docs/archive/2026-09-29-plan-upstream-native-trace.md create mode 100644 scripts/trace-ort.mjs create mode 100644 tests/kit/trace-ort.test.mjs diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index 92ffece7..be9bd50d 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -88,7 +88,12 @@ jobs: # same teardown abort on store-touching commands (`memory search` → correct # output, rc 134), so an `--only memory-routes` step added here would need the same # guard. Remove continue-on-error once that issue closes. + - name: Require Node 22.15+ for the macOS resolution hook + if: matrix.os == 'macos-latest' + run: node -e "const [major, minor] = process.versions.node.split('.').map(Number); if (major < 22 || (major === 22 && minor < 15)) { console.error('trace-ort requires Node 22.15+'); process.exit(1); }" + - name: Deep proof against the live packages (learning) + if: matrix.os != 'macos-latest' continue-on-error: true env: HOME: ${{ runner.temp }}/kit-home @@ -97,6 +102,52 @@ jobs: APPDATA: ${{ runner.temp }}/kit-home/AppData/Roaming run: node bin/agentic-kit.mjs status --refresh=live --only learning + - name: Deep proof against the live packages (learning, traced macOS) + if: matrix.os == 'macos-latest' + continue-on-error: true + shell: bash + env: + HOME: ${{ runner.temp }}/kit-home + USERPROFILE: ${{ runner.temp }}/kit-home + XDG_CONFIG_HOME: ${{ runner.temp }}/kit-home/.config + APPDATA: ${{ runner.temp }}/kit-home/AppData/Roaming + NODE_OPTIONS: --import=${{ github.workspace }}/scripts/trace-ort.mjs + TRACE_ORT_LOG: ${{ runner.temp }}/trace-ort.jsonl + run: | + set +e + node bin/agentic-kit.mjs status --refresh=live --only learning + learning_rc=$? + LEARNING_RC="$learning_rc" NODE_OPTIONS='' node --input-type=module -e ' + import fs from "node:fs"; + import crypto from "node:crypto"; + const source = fs.readFileSync("scripts/trace-ort.mjs"); + fs.writeFileSync(process.env.RUNNER_TEMP + "/trace-ort-receipt.json", JSON.stringify({ + sourceSha: process.env.GITHUB_SHA, + hookSha256: crypto.createHash("sha256").update(source).digest("hex"), + node: process.version, platform: process.platform, arch: process.arch, + learningExitCode: Number(process.env.LEARNING_RC), + tracePresent: fs.existsSync(process.env.TRACE_ORT_LOG), + }) + "\n"); + ' + exit "$learning_rc" + + - name: Check macOS trace artifact + if: always() && matrix.os == 'macos-latest' + shell: bash + run: | + test -s "$RUNNER_TEMP/trace-ort.jsonl" || { echo '::error::trace-ort artifact absent or empty'; exit 1; } + test -s "$RUNNER_TEMP/trace-ort-receipt.json" || { echo '::error::trace-ort receipt absent or empty'; exit 1; } + + - name: Upload macOS learning resolution trace + if: always() && matrix.os == 'macos-latest' + uses: actions/upload-artifact@v7 + with: + name: macos-learning-ort-trace + if-no-files-found: error + path: | + ${{ runner.temp }}/trace-ort.jsonl + ${{ runner.temp }}/trace-ort-receipt.json + clean-mac-setup: name: clean macOS setup (packed artifact) runs-on: macos-latest @@ -134,7 +185,7 @@ jobs: ollama serve > "$RUNNER_TEMP/ollama.log" 2>&1 & AK_OLLAMA_PID=$! trap 'kill "$AK_OLLAMA_PID" 2>/dev/null || true' EXIT - for attempt in {1..30}; do + for ((ak_readiness_attempt=0; ak_readiness_attempt<30; ak_readiness_attempt++)); do curl --fail --silent http://127.0.0.1:11434/api/version >/dev/null && break sleep 1 done @@ -142,6 +193,7 @@ jobs: git -C "$AK_PROJECT" init (cd "$AK_PROJECT" && ak setup --yes --no-ruvnet-brain --aqe-embedding-mode local) | tee "$RUNNER_TEMP/setup.log" (cd "$AK_PROJECT" && ak x aqe-embedding verify --json) | tee "$RUNNER_TEMP/embedding-proof.json" + # shellcheck disable=SC2016 # JavaScript template literals are evaluated by Node. node --input-type=module -e ' import fs from "node:fs"; import path from "node:path"; diff --git a/docs/archive/2026-09-29-native-learning-trace.md b/docs/archive/2026-09-29-native-learning-trace.md new file mode 100644 index 00000000..00aef9c2 --- /dev/null +++ b/docs/archive/2026-09-29-native-learning-trace.md @@ -0,0 +1,32 @@ +# Hosted macOS learning resolution trace + +## Status and inputs + +Captured 2026-09-29. This is an observation, not a native-readiness or causal verdict. + +- [Workflow run](https://github.com/pacphi/agentic-kit/actions/runs/36567908852), [macOS job](https://github.com/pacphi/agentic-kit/actions/runs/36567908852/job/109404326296), [artifact](https://github.com/pacphi/agentic-kit/actions/runs/36567908852/artifacts/11031869828). +- Source `ed8f4cb96836e1911aae9a89bbb3f015f033e115`; hook SHA-256 `98d9cc2c54e570eb18e24af83a7c401779c6eec12d58fe740e8c1a7622276dae`. +- macOS arm64, Node 22.23.2; Ruflo 3.48.0 and Agentic QE 3.14.5 installed in a disposable CI prefix. +- The kit's `sync --no-upgrade` healing step ran before `node bin/agentic-kit.mjs status --refresh=live --only learning`. This is a post-heal observation, not a pristine npm-tree measurement. +- The hook follows [vidaunited's resolution-hook proposal](https://github.com/ruvnet/ruflo/issues/2885#issuecomment-5867331087), with metadata-only JSONL and explicit artifact paths. + +## Observations + +The artifact contains 22 records: 16 preload starts and six package-resolution records across three processes. The two distinct package roots, with the runner-specific prefix omitted, are: + +```text +@huggingface/transformers@3.8.1 + npm-prefix/lib/node_modules/ruflo/node_modules/@claude-flow/cli/node_modules/@huggingface/transformers +onnxruntime-node@1.21.0 + npm-prefix/lib/node_modules/ruflo/node_modules/@claude-flow/cli/node_modules/@huggingface/transformers/node_modules/onnxruntime-node +``` + +The learning step invokes `ruflo neural train -p coordination -e 50`. Its output contains the default `fp32` dtype warning and `libc++abi` / `mutex lock failed: Invalid argument`. The outer kit command exited 1. The trace and receipt were retained despite that failure; the existing `continue-on-error` boundary explains the green macOS job. + +The separate clean-setup job failed at an AQE embedding process probe, and external link checking failed. Those results are not evidence that the trace worked or that setup is healthy. The trace was enabled only in the macOS learning step. + +## What this establishes + +The old package roots were resolved during the failing traced step. It remains unproved which installation/healing/dependency action introduced them and whether they caused the mutex failure. Resolution observations are not an exhaustive native-module census. A start-only artifact would establish only that preload reached a registration attempt. `learningExitCode` records the outer kit command, not a separately captured native Ruflo exit. + +Synthetic ESM/CommonJS, nested-copy, child-inheritance, unknown-version and failed-log-target tests passed. A later fixture-only correction uses file URLs for Windows preload paths and adds a real space-path test; it does not change the hook bytes captured here. The exact sanitized upstream comment was awaiting maintainer approval at this capture. No upstream message or user-global installation is implied by this record. diff --git a/docs/archive/2026-09-29-plan-upstream-native-trace.md b/docs/archive/2026-09-29-plan-upstream-native-trace.md new file mode 100644 index 00000000..b2391e68 --- /dev/null +++ b/docs/archive/2026-09-29-plan-upstream-native-trace.md @@ -0,0 +1,19 @@ +# Upstream native trace plan + +## Status + +**Implemented; final integration pending** — Captured 2026-09-29. The hook and nightly workflow passed independent review and all eight local gates. Hosted macOS run 36567908852 captured package resolutions and the learning failure. The Windows preload fixture was corrected to use a file URL and independently reviewed. Final documentation/PR CI and the exact upstream comment approval remain separate gates. + +## Scope + +Add a passive Node resolution hook for the nightly macOS learning probe. The hook records each resolved `@huggingface/transformers`, `@xenova/transformers`, and `onnxruntime-node` package root once per process, including its on-disk version when readable. It records no command arguments, environment values, or prompt content. The nightly workflow edit and native CI run belong to the integration owner. + +## Steps + +1. Add focused child-process tests with synthetic packages for ESM and CommonJS resolution, nested copies, missing/malformed versions, and an unusable log path. +2. Implement `scripts/trace-ort.mjs` from vidaunited's hook in [ruflo issue 2885](https://github.com/ruvnet/ruflo/issues/2885#issuecomment-5867331087). Require `TRACE_ORT_LOG` to be an absolute, explicit target. Emit a hook-start receipt and one JSONL record per package root per process. Observation failures must not change the observed command's exit result. +3. Run the focused guarded test and static checks, then provide exact nightly workflow hunks to the integration owner. Retain the trace, source/version, and exit receipt on failure. + +## Acceptance and limits + +The synthetic tests must prove observed resolution behavior without downloads or native ORT. The hosted macOS learning trace is captured in the accompanying dated evidence record. An empty trace is not proof that no relevant package loaded; a start receipt proves that preload reached the registration attempt, not that registration succeeded. Resolution hooks do not prove native addon teardown or capture packages loaded before registration. The receipt records the outer kit command exit, not necessarily Ruflo's raw exit. diff --git a/docs/archive/README.md b/docs/archive/README.md index 157b2176..614a56e4 100644 --- a/docs/archive/README.md +++ b/docs/archive/README.md @@ -163,6 +163,8 @@ reconfirmed by this metadata audit. The per-file inventory and limitations are r | [2026-09-28-plan-docs-taxonomy-and-archive.md](2026-09-28-plan-docs-taxonomy-and-archive.md) | `docs/plans/2026-09-28-docs-taxonomy-and-archive.md` | Finished documentation taxonomy and archive plan | Implemented in [PR #264](https://github.com/pacphi/agentic-kit/pull/264) and [PR #266](https://github.com/pacphi/agentic-kit/pull/266). Current layout rules live in the [docs index](../README.md) and [layout guard](../../scripts/docs-layout.mjs); this plan records the build steps. | | [2026-09-28-plan-sonnet-5-5-routing-refresh.md](2026-09-28-plan-sonnet-5-5-routing-refresh.md) | `docs/plans/2026-09-28-sonnet-5-5-routing-refresh.md` | Routing and pricing research with the implemented model-catalog decision | Implemented in [PR #268](https://github.com/pacphi/agentic-kit/pull/268). The operative tier decision is in [ADR-0006](../adr/0006-primary-host-and-ambidextrous-mirroring.md); prices and benchmarks here are dated evidence. | | [2026-09-28-plan-dashboard-refresh.md](2026-09-28-plan-dashboard-refresh.md) | `docs/plans/2026-09-28-dashboard-refresh.md` | Completed V3 implementation plan | Scoped implementation and independent review through `d2c1833b`, including native plain-folder evidence and paused Activity timestamps. Full branch gates, PR CI and develop integration remain separate gates at capture. | +| [2026-09-29-plan-upstream-native-trace.md](2026-09-29-plan-upstream-native-trace.md) | `docs/plans/2026-09-29-upstream-native-trace.md` | Completed C3 instrumentation plan | Hook, workflow and native trace captured; final PR integration and exact upstream-post approval remain separate gates at capture. | +| [2026-09-29-native-learning-trace.md](2026-09-29-native-learning-trace.md) | — (new) | Hosted macOS package-resolution evidence | Binds the observed 3.8.1/1.21.0 roots and learning failure to one source/run; introduction and causality remain unproved. | ## Naming convention diff --git a/scripts/trace-ort.mjs b/scripts/trace-ort.mjs new file mode 100644 index 00000000..7c5afcac --- /dev/null +++ b/scripts/trace-ort.mjs @@ -0,0 +1,77 @@ +// Passive adaptation of vidaunited's Node resolution hook: +// https://github.com/ruvnet/ruflo/issues/2885#issuecomment-5867331087 +// Enable only with an explicit absolute TRACE_ORT_LOG and NODE_OPTIONS=--import=. +import * as moduleApi from 'node:module'; +import fs from 'node:fs'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const SOURCE = 'ruvnet/ruflo#2885:issuecomment-5867331087'; +const log = process.env.TRACE_ORT_LOG; +const seen = new Set(); +let warned = false; + +function warn(message) { + if (warned) return; + warned = true; + try { process.stderr.write(`[trace-ort] ${message}\n`); } catch { /* observation is best effort */ } +} + +function append(record) { + try { + fs.appendFileSync(log, `${JSON.stringify({ schema: 1, pid: process.pid, ...record })}\n`, { flag: 'a' }); + } catch { + warn('log unavailable; trace artifact is incomplete'); + } +} + +function packageAt(file) { + const parts = path.normalize(file).split(path.sep); + for (let i = parts.length - 2; i >= 0; i--) { + if (parts[i] !== 'node_modules') continue; + const first = parts[i + 1]; + const scoped = first === '@huggingface' || first === '@xenova'; + const name = scoped ? `${first}/${parts[i + 2]}` : first; + if (!['@huggingface/transformers', '@xenova/transformers', 'onnxruntime-node'].includes(name)) continue; + const end = i + (scoped ? 3 : 2); + if (parts.length <= end) continue; + return { name, root: parts.slice(0, end).join(path.sep) || path.sep }; + } + return null; +} + +function versionAt(root) { + try { + const metadata = JSON.parse(fs.readFileSync(path.join(root, 'package.json'), 'utf8')); + return typeof metadata.version === 'string' && metadata.version.length <= 128 && metadata.version.length > 0 + ? metadata.version : 'unknown'; + } catch { return 'unknown'; } +} + +function note(url) { + if (!url?.startsWith('file:')) return; + const found = packageAt(fileURLToPath(url)); + if (!found || seen.has(found.root)) return; + seen.add(found.root); + const rootTruncated = found.root.length > 4096; + append({ type: 'package', name: found.name, version: versionAt(found.root), + root: found.root.slice(0, 4096), ...(rootTruncated ? { rootTruncated: true } : {}) }); +} + +if (typeof log !== 'string' || !path.isAbsolute(log)) { + warn('log target missing or relative; set absolute TRACE_ORT_LOG'); +} else if (typeof moduleApi.registerHooks !== 'function') { + warn('Node module.registerHooks unavailable; requires Node 22.15 or newer'); +} else { + append({ type: 'start', hook: 'trace-ort/1', source: SOURCE, node: process.version, + platform: process.platform, arch: process.arch }); + try { + moduleApi.registerHooks({ + resolve(specifier, context, nextResolve) { + const result = nextResolve(specifier, context); + try { note(result.url); } catch { warn('resolution observation failed; trace artifact is incomplete'); } + return result; + }, + }); + } catch { warn('hook registration failed; trace artifact is incomplete'); } +} diff --git a/tests/kit/trace-ort.test.mjs b/tests/kit/trace-ort.test.mjs new file mode 100644 index 00000000..58e62773 --- /dev/null +++ b/tests/kit/trace-ort.test.mjs @@ -0,0 +1,105 @@ +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { spawnSync } from 'node:child_process'; +import { fileURLToPath, pathToFileURL } from 'node:url'; +import { test } from 'node:test'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; + +const hook = new URL('../../scripts/trace-ort.mjs', import.meta.url).href; + +function pkg(root, name, version, { commonjs = false, malformed = false } = {}) { + const dir = path.join(root, 'node_modules', ...name.split('/')); + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, 'package.json'), malformed ? '{bad' : JSON.stringify({ name, version, main: 'index.js', type: commonjs ? 'commonjs' : 'module' })); + fs.writeFileSync(path.join(dir, 'index.js'), commonjs ? 'module.exports = 1;\n' : 'export default 1;\n'); + return dir; +} + +function run(root, source, log = path.join(root, 'trace.jsonl'), hookUrl = hook) { + const home = path.join(root, 'home'); + const entry = path.join(root, 'entry.mjs'); + fs.writeFileSync(entry, source); + const env = spawnEnv(home, { NODE_OPTIONS: `--import=${hookUrl}`, TRACE_ORT_LOG: log }); + const child = spawnSync(process.execPath, [entry], { cwd: root, env, encoding: 'utf8' }); + const records = fs.existsSync(log) ? fs.readFileSync(log, 'utf8').trim().split('\n').filter(Boolean).map(JSON.parse) : []; + return { child, records }; +} + +test('records CJS and ESM resolution once per distinct package root', (t) => { + const root = tempDir('trace-ort', t); + const huggingface = pkg(root, '@huggingface/transformers', '4.3.0'); + const ort = pkg(root, 'onnxruntime-node', '1.30.0', { commonjs: true }); + const nested = path.join(root, 'nested'); + const old = pkg(nested, '@huggingface/transformers', '3.8.1'); + const { child, records } = run(root, `import '@huggingface/transformers'; +import '@huggingface/transformers'; +import { createRequire } from 'node:module'; +const require = createRequire(import.meta.url); +require('onnxruntime-node'); +require('onnxruntime-node'); +const nestedRequire = createRequire(${JSON.stringify(path.join(nested, 'entry.cjs'))}); +nestedRequire('@huggingface/transformers'); +`); + assert.equal(child.status, 0, child.stderr); + assert.equal(records.filter((r) => r.type === 'start').length, 1); + assert.deepEqual(records.filter((r) => r.type === 'package').map((r) => [r.name, r.version, r.root]).sort(), [ + ['@huggingface/transformers', '3.8.1', old], + ['@huggingface/transformers', '4.3.0', huggingface], + ['onnxruntime-node', '1.30.0', ort], + ].sort()); + assert.ok(records.every((r) => r.pid === records[0].pid)); +}); + +test('reports unknown version without executing package metadata', (t) => { + const root = tempDir('trace-ort-version', t); + pkg(root, '@xenova/transformers', undefined); + const { child, records } = run(root, "import '@xenova/transformers';\n"); + assert.equal(child.status, 0, child.stderr); + assert.equal(records.find((r) => r.type === 'package')?.version, 'unknown'); +}); + +test('an unusable or absent log target does not change the command result', (t) => { + const root = tempDir('trace-ort-failure', t); + pkg(root, 'onnxruntime-node', '1.30.0', { commonjs: true }); + const source = "import { createRequire } from 'node:module'; createRequire(import.meta.url)('onnxruntime-node');\n"; + const invalid = run(root, source, path.join(root, 'missing', 'trace.jsonl')); + assert.equal(invalid.child.status, 0, invalid.child.stderr); + assert.match(invalid.child.stderr, /trace-ort.*log/i); + const absent = run(root, source, ''); + assert.equal(absent.child.status, 0, absent.child.stderr); + assert.match(absent.child.stderr, /trace-ort.*log/i); +}); + +test('NODE_OPTIONS traces inherited child processes with separate start receipts', (t) => { + const root = tempDir('trace-ort-child', t); + pkg(root, 'onnxruntime-node', '1.21.0', { commonjs: true }); + const childFile = path.join(root, 'child.cjs'); + fs.writeFileSync(childFile, "require('onnxruntime-node');\n"); + const { child, records } = run(root, `import { spawnSync } from 'node:child_process'; +const result = spawnSync(process.execPath, [${JSON.stringify(childFile)}], { encoding: 'utf8' }); +if (result.status !== 0) process.exit(result.status || 1); +`); + assert.equal(child.status, 0, child.stderr); + const starts = records.filter((r) => r.type === 'start'); + assert.equal(starts.length, 2); + assert.equal(new Set(starts.map((r) => r.pid)).size, 2); + assert.equal(records.filter((r) => r.type === 'package').length, 1); + assert.equal(records.find((r) => r.type === 'package')?.version, '1.21.0'); +}); + +test('preloads a hook from a path containing spaces', (t) => { + const root = tempDir('trace ort space', t); + const hookDir = path.join(root, 'hook with spaces'); + fs.mkdirSync(hookDir); + const copiedHook = path.join(hookDir, 'trace ort.mjs'); + fs.copyFileSync(fileURLToPath(hook), copiedHook); + pkg(root, 'onnxruntime-node', '1.30.0', { commonjs: true }); + const { child, records } = run(root, + "import { createRequire } from 'node:module'; createRequire(import.meta.url)('onnxruntime-node');\n", + path.join(root, 'trace output.jsonl'), pathToFileURL(copiedHook).href); + assert.equal(child.status, 0, child.stderr); + assert.equal(records.filter((r) => r.type === 'start').length, 1); + assert.equal(records.find((r) => r.type === 'package')?.version, '1.30.0'); +}); From 989c5e563c8aa0527b1497fb64643f8277664c0a Mon Sep 17 00:00:00 2001 From: Chris Phillipson Date: Tue, 29 Sep 2026 06:54:18 -0700 Subject: [PATCH 06/10] fix(watch): complete bounded polling and watcher follow-ups (#278) * fix(upstream-watch): skip retries for deterministic fetch failures * fix(upstream-watch): bound dispatch pull request observation * test(upstream-watch): cover ledger git and spawn failures * test(upstream-watch): reject invalid ledger branch before git * perf(upstream-watch): measure notice fit with prefix lengths * test(upstream-watch): cover singular notice and latest firing * fix(upstream-watch): report invalid record registry once * test(upstream-watch): pin invalid registry precedence over future since * refactor(upstream-watch): share ledger event vocabulary * fix(upstream-watch): fail ledger on invalid registry * docs(upstream-watch): remove stale Codex reprobe instruction * fix(upstream-watch): format exhausted firing sessions as a list * fix(watch): require a commit before sending a notice * fix(watch): explain blind workflow summaries * docs(watch): archive completed follow-up plan --- .github/workflows/upstream-watch.yml | 7 +++-- ...026-09-29-plan-upstream-watch-followups.md | 30 +++++++++++++++++++ docs/archive/README.md | 1 + docs/upstream-watch.md | 11 ++++--- scripts/upstream-watch.mjs | 16 ++++++---- scripts/upstream-watch/dispatch.mjs | 13 ++++++-- scripts/upstream-watch/fetch.mjs | 28 +++++++++-------- scripts/upstream-watch/ledger.mjs | 10 +++++-- .../agentic-dependency-constraints.json | 2 +- tests/kit/upstream-watch-dispatch.test.mjs | 26 ++++++++++++++-- tests/kit/upstream-watch-fixtures.mjs | 3 +- .../kit/upstream-watch-ledger-branch.test.mjs | 27 +++++++++++++++++ tests/kit/upstream-watch-notice.test.mjs | 23 ++++++++++++++ tests/kit/upstream-watch-query.test.mjs | 11 +++++++ tests/kit/upstream-watch-record.test.mjs | 26 +++++++++++++--- 15 files changed, 197 insertions(+), 37 deletions(-) create mode 100644 docs/archive/2026-09-29-plan-upstream-watch-followups.md diff --git a/.github/workflows/upstream-watch.yml b/.github/workflows/upstream-watch.yml index 656d67c1..907a0cda 100644 --- a/.github/workflows/upstream-watch.yml +++ b/.github/workflows/upstream-watch.yml @@ -58,7 +58,7 @@ jobs: set -e { echo "## Upstream watch preview (exit $code)" - jq -r '"since \(.since) (\(.sinceSource)), new records \(.records | length), could not check \(.fetchErrors | length), blind \(.blind), notice \(.notice.post), would fire \(.wouldFire | length)"' watch.json + jq -r 'if .blind then "blind \(.blind): \(.error // "unknown error"), could not check \(.fetchErrors | length)" else "since \(.since) (\(.sinceSource)), new records \(.records | length), could not check \(.fetchErrors | length), blind \(.blind), notice \(.notice.post), would fire \(.wouldFire | length)" end' watch.json jq -r '(.wouldFire // [])[] | "- would fire \(.id) \(.version) \(.branch)"' watch.json echo; echo '```text'; cat errors.txt; echo '```' } >> "$GITHUB_STEP_SUMMARY" @@ -102,7 +102,7 @@ jobs: set -e { echo "## Upstream watch (exit $code)" - jq -r '"since \(.since) (\(.sinceSource)), new records \(.records | length), could not check \(.fetchErrors | length), dispatch errors \(.dispatchErrors | length), blind \(.blind), commit \(.commit // "none"), would fire \(.wouldFire | length)"' watch.json + jq -r 'if .blind then "blind \(.blind): \(.error // "unknown error"), could not check \(.fetchErrors | length), dispatch errors \(.dispatchErrors | length)" else "since \(.since) (\(.sinceSource)), new records \(.records | length), could not check \(.fetchErrors | length), dispatch errors \(.dispatchErrors | length), blind \(.blind), commit \(.commit // "none"), would fire \(.wouldFire | length)" end' watch.json jq -r '(.wouldFire // [])[] | "- would fire \(.id) \(.version) \(.branch)"' watch.json echo; echo '```text'; cat errors.txt; echo '```' jq -r '.notice.body' watch.json @@ -120,7 +120,8 @@ jobs: if: env.RECORD == 'true' run: | [ "$(jq -r '.notice.post' watch.json)" = true ] || { echo 'Nothing needs the maintainer.'; exit 0; } - commit=$(jq -r '.commit' watch.json) + commit=$(jq -r '.commit // empty' watch.json) + [ -n "$commit" ] || { echo 'Notice requested without a ledger commit.' >&2; exit 1; } jq -r '.notice.body' watch.json > notice.md test -s notice.md gh api "repos/$GITHUB_REPOSITORY/commits/$commit/comments" -F body=@notice.md --jq .html_url diff --git a/docs/archive/2026-09-29-plan-upstream-watch-followups.md b/docs/archive/2026-09-29-plan-upstream-watch-followups.md new file mode 100644 index 00000000..ae6d6e52 --- /dev/null +++ b/docs/archive/2026-09-29-plan-upstream-watch-followups.md @@ -0,0 +1,30 @@ +# Upstream watch followups + +## Status at archival + +Implemented and independently reviewed on `fix/upstream-watch-followups` at +`b75c1e3fec4f645700f0f6d18220956867b7e5d0`. All eight local gates passed at that +source: unit/legacy tests, browser UI, typecheck, lint, complexity, Markdown, +build, and offline links. Controller workflow commits `aebe2ae6` and `b75c1e3f` +complete items 11/12; their shell and jq behavior was independently exercised. +Green `develop@84307574` was subsequently merged at `ff1eff27` without conflicts. +Final-head PR CI and merge remain pending at capture; no real dispatch or +notification was performed for this lane. + +M7 distinguishes deterministic validation failures from transient retries. +M8 bounds eligible fired-PR polling to seven days after the latest firing, +retaining the ledger; a later PR requires manual reconciliation. All twelve +minors have a fix, regression proof of existing behavior, or explicit no-change +disposition. M10 remains declined. Current behavior is documented in +[Upstream watch](../upstream-watch.md). + +## Execution and acceptance + +Scope: M7, M8, and the twelve numbered deferred minors in the recovered PR #253 report. M10 is declined because the numeric PR field matches emitted records. This lane owns watcher scripts, focused tests, one Codex registry entry, and current watcher documentation. The integration owner owns workflow edits. + +1. Add focused failing tests for deterministic retry failures, bounded dispatch polling, ledger errors, and the numbered edge cases. Keep synthetic fetch, dispatch, and git boundaries. +2. Implement the smallest watcher changes that make those tests pass. Check already-correct behavior and record no-change findings without empty commits. +3. Commit independently verifiable items separately. Run focused Node tests and static checks without a full suite, dispatch call, or GitHub write. +4. Put item 11/12 workflow hunks, commands, per-item dispositions, and residual limits in the ignored C4 report. Stop for independent review. + +Acceptance: deterministic errors do not sleep; transient errors retry at most twice; fired PR polling stops after seven days or when registry state is ineligible; invalid ledger registry exits nonzero; notices retain their body and maximum length semantics; all twelve minors are either fixed, proved already handled, or handed off as exact workflow hunks. diff --git a/docs/archive/README.md b/docs/archive/README.md index 614a56e4..559d1038 100644 --- a/docs/archive/README.md +++ b/docs/archive/README.md @@ -68,6 +68,7 @@ reconfirmed by this metadata audit. The per-file inventory and limitations are r | File | Original location | What it was | Why it's historical | |---|---|---|---| +| [2026-09-29-plan-upstream-watch-followups.md](2026-09-29-plan-upstream-watch-followups.md) | `docs/plans/2026-09-29-upstream-watch-followups.md` | V4 C4 execution plan for bounded PR polling, deterministic retry failures and twelve deferred watcher minors. | Implementation and independent review complete; all eight local gates passed at `b75c1e3f`. Final PR CI and merge were pending at archival. Current contract: [Upstream watch](../upstream-watch.md). | | [2026-06-upstream-findings-f1-f6.md](2026-06-upstream-findings-f1-f6.md) | `docs/upstream/ruflo-self-improvement-findings.md` | The F1–F6 findings series: proofs/refutations of ruflo's self-improvement claims (Q-learning persistence, state-encoder collapse, SONA learn→inference wiring, native-training misreporting), with filed upstream issues. | Every finding is now fixed upstream: F2 in 3.10.6 ([#2222](https://github.com/ruvnet/ruflo/issues/2222)), F2b in 3.10.7, F3 in 3.10.11 ([#2239](https://github.com/ruvnet/ruflo/issues/2239)), F4 in `@ruvector/ruvllm` 2.5.6 ([RuVector#519](https://github.com/ruvnet/RuVector/issues/519)), F6 in 3.18.1/3.19.0 + ruvllm 2.5.7 ([#2549](https://github.com/ruvnet/ruflo/issues/2549), closed 2026-07-03). | | [2026-06-token-consumption-incident.md](2026-06-token-consumption-incident.md) | `docs/usage/token-consumption-findings-and-mitigation-2026-06.md` | Root-cause report for the June 2026 token-burn incident: six immortal auto-started daemons consumed ~8.1B tokens over 7 days via headless worker sessions. Produced the opt-in daemon policy, TTL reaper, ⚙ statusline alarm, and `ruflo-token-audit`. | The root cause was fixed upstream in ruflo 3.27/3.28 ([#2661](https://github.com/ruvnet/ruflo/issues/2661)): AI workers are opt-in, launches are governed by a machine-wide budget with telemetry, one supervisor daemon per repo, native daemon TTL. The kit's daemon policy flipped back to default-on (local-only workers) on that baseline; the reapers and token-audit remain as an independent check. | | [2026-06-11-token-consumption-recurrence.md](2026-06-11-token-consumption-recurrence.md) | `docs/usage/token-consumption-recurrence-and-cleanup-2026-06-11.md` | Follow-up audit 10 days later: 17 daemons had accumulated but the TTL auto-reaper had already contained them; cleanup of daemon-state files, logs, and two plugin MCP servers. | Same incident class as above — governed upstream since 3.27/3.28. Kept as evidence the TTL-reaper safety net worked. | diff --git a/docs/upstream-watch.md b/docs/upstream-watch.md index 5857fc88..32be1c39 100644 --- a/docs/upstream-watch.md +++ b/docs/upstream-watch.md @@ -107,8 +107,8 @@ Every command also takes `--concurrency <1-16>` (default 4) and `--registry ] [--registry ] @@ -28,7 +28,6 @@ const USAGE = `usage: node scripts/upstream-watch.mjs report [--json] [--concurr const PENDING = new Set(['watching', 'fixed-unreleased']); // record exits BLIND when gh, the registry, the ledger branch or every upstream thread is unreadable. const BLIND = 3; -const LEDGER_EVENTS = ['reply', 'acknowledged', 'closed', 'merged', 'released', 'reopened', 'stale', 'retire-proposed', 'retest-due', 'idle', 'fired', 'dispatch-pr']; class UsageError extends Error {} @@ -248,8 +247,8 @@ async function ledgerQuery(registry, options, { stdout, stderr, now, ledgerStore const WEEK = 7 * 86_400_000; -function blindRecord(error, { stdout, stderr, json }, extra = {}) { - stderr.write(`${error}\n`); +function blindRecord(error, { stdout, stderr, json }, extra = {}, { writeError = true } = {}) { + if (writeError) stderr.write(`${error}\n`); const result = { blind: true, error, records: [], fetchErrors: [], dispatchErrors: [], wouldFire: [], fired: [], parent: null, commit: null, notice: { post: false, body: '' }, ...extra, }; @@ -286,7 +285,8 @@ async function record(registry, fetcher, options, { stdout, stderr, now, ledgerS const recorded = ledger.records.map((item) => item.line).join('\n'); const all = ledgerEvents(report, registry, { since }); const released = all.filter((event) => event.event === 'released' && event.fields.branch); - const fired = await dispatch({ released, records: ledger.records, dispatcher, repo, sentinel, now, recordedAt: runAt, dryRun: options.dryRun }); + const eligibleIds = new Set(registry.watch.filter((entry) => PENDING.has(entry.status)).map((entry) => entry.id)); + const fired = await dispatch({ released, records: ledger.records, dispatcher, repo, sentinel, now, recordedAt: runAt, dryRun: options.dryRun, eligibleIds }); const records = [...withoutRecorded(all, recorded).map((event) => toRecord(event, runAt)), ...fired.records]; const checkedAt = fetchErrors.length ? (ledger.checkedAt ?? since) : runAt; let commit = null; @@ -333,7 +333,11 @@ export async function main(argv, { stderr.write(`upstream registry is ${registry.registryStatus ?? registry.status}:\n${registry.errors.map((error) => ` ${error}`).join('\n')}\n`); const status = { status: registry.registryStatus ?? registry.status, errors: registry.errors }; if (options.command === 'record') { - return blindRecord(`upstream registry is ${status.status}`, { stdout, stderr, json: options.json }, { registry: status }); + return blindRecord(`upstream registry is ${status.status}`, { stdout, stderr, json: options.json }, { registry: status }, { writeError: false }); + } + if (options.command === 'ledger') { + if (options.json) stdout.write(`${JSON.stringify({ registry: status }, null, 2)}\n`); + return BLIND; } stdout.write(options.json ? `${JSON.stringify({ registry: status }, null, 2)}\n` : 'No report: the upstream registry is not valid.\n'); return 0; diff --git a/scripts/upstream-watch/dispatch.mjs b/scripts/upstream-watch/dispatch.mjs index 005e17d8..f3bebe1e 100644 --- a/scripts/upstream-watch/dispatch.mjs +++ b/scripts/upstream-watch/dispatch.mjs @@ -10,6 +10,7 @@ import { toRecord } from './ledger-branch.mjs'; export const FIRE_URL = (routine) => `https://api.anthropic.com/v1/claude_code/routines/${routine}/fire`; export const FIRE_HEADERS = { 'anthropic-beta': 'experimental-cc-routine-2026-04-01', 'anthropic-version': '2023-06-01', 'content-type': 'application/json' }; export const REFIRE_AFTER_DAYS = 3; +export const PR_OBSERVE_DAYS = 7; export const MAX_FIRES = 2; export const FIRE_TIMEOUT_MS = 30_000; // `gh pr list --head` matches the branch name in any fork; only a pull request @@ -19,6 +20,11 @@ const ROUTINE = /^trig_[A-Za-z0-9]+$/; const DISPATCH_BRANCH = /^upstream\/[a-z0-9._-]+$/; const DAY = 86_400_000; +export function sessionList(sessions) { + if (sessions.length < 3) return sessions.join(' and '); + return `${sessions.slice(0, -1).join(', ')}, and ${sessions.at(-1)}`; +} + export function createDispatcher({ exec = run, fetchImpl = globalThis.fetch, env = process.env } = {}) { return { async branchExists(branch) { @@ -56,7 +62,7 @@ export function createDispatcher({ exec = run, fetchImpl = globalThis.fetch, env } /** Fire (or, in a dry run, list in `wouldFire`) each released fix; the branch and pull request lookups only read. */ -export async function dispatch({ released, records, dispatcher, repo, sentinel, now, recordedAt, dryRun = false }) { +export async function dispatch({ released, records, dispatcher, repo, sentinel, now, recordedAt, dryRun = false, eligibleIds = null }) { const out = []; const errors = []; const wouldFire = []; @@ -68,7 +74,7 @@ export async function dispatch({ released, records, dispatcher, repo, sentinel, if (await dispatcher.branchExists(branch)) continue; const firings = recordsOf(event.id, 'fired'); if (firings.length >= MAX_FIRES) { - errors.push({ id: event.id, error: `dispatch did not complete after ${firings.length} firings; see ${firings.map((item) => item.fields.session).join(' and ')}` }); + errors.push({ id: event.id, error: `dispatch did not complete after ${firings.length} firings; see ${sessionList(firings.map((item) => item.fields.session))}` }); continue; } const newest = Math.max(...firings.map((item) => Date.parse(item.recordedAt)), 0); @@ -85,6 +91,9 @@ export async function dispatch({ released, records, dispatcher, repo, sentinel, } for (const id of new Set(records.filter((item) => item.event === 'fired').map((item) => item.id))) { if (recordsOf(id, 'dispatch-pr').length) continue; + if (eligibleIds && !eligibleIds.has(id)) continue; + const latestFiring = Math.max(...recordsOf(id, 'fired').map((item) => Date.parse(item.recordedAt))); + if (!Number.isFinite(latestFiring) || now.getTime() - latestFiring >= PR_OBSERVE_DAYS * DAY) continue; try { const branch = recordsOf(id, 'fired').at(-1).fields.branch; const pr = await dispatcher.openPullRequest(repo, branch); diff --git a/scripts/upstream-watch/fetch.mjs b/scripts/upstream-watch/fetch.mjs index 198fad9a..5077d683 100644 --- a/scripts/upstream-watch/fetch.mjs +++ b/scripts/upstream-watch/fetch.mjs @@ -14,6 +14,10 @@ const VERSION = /^\d+\.\d+\.\d+(?:-[\w.]+)?$/; const NOT_FOUND = /HTTP 404|Not Found/i; const NO_MATCH = /No match found for version/; +/** An invalid local argument or fixture cannot recover by waiting for the network. */ +export class PermanentFetchError extends Error {} +const invalid = (message) => new PermanentFetchError(message); + // What closed a thread: its closing pull requests, else the ClosedEvent's closer. export const FIXING_CHANGES_QUERY = `query($owner:String!,$name:String!,$number:Int!){repository(owner:$owner,name:$name){defaultBranchRef{name} issueOrPullRequest(number:$number){__typename ... on Issue{closedByPullRequestsReferences(first:10,includeClosedPrs:true){nodes{number merged mergedAt baseRefName mergeCommit{oid} repository{nameWithOwner}}} timelineItems(last:1,itemTypes:[CLOSED_EVENT]){nodes{... on ClosedEvent{closer{__typename ... on Commit{oid} ... on PullRequest{number merged mergedAt baseRefName mergeCommit{oid} repository{nameWithOwner}}}}}}} ... on PullRequest{number merged mergedAt baseRefName mergeCommit{oid} repository{nameWithOwner}}}}}`; @@ -80,7 +84,7 @@ export function createFetcher({ exec = run } = {}) { }, async thread(id) { const [, repo, number] = ID.exec(id) ?? []; - if (!repo) throw new Error(`not an owner/repo#number id: ${id}`); + if (!repo) throw invalid(`not an owner/repo#number id: ${id}`); const issue = await json('gh', ['api', `repos/${repo}/issues/${number}`]); return { issue, comments: await this.comments(repo, number) }; }, @@ -89,7 +93,7 @@ export function createFetcher({ exec = run } = {}) { * per line; `--slurp` would need gh 2.48, newer than apt's gh on Ubuntu 24.04. */ async comments(repo, number) { - if (!OWNER_REPO.test(repo ?? '') || !/^[1-9]\d*$/.test(String(number))) throw new Error(`not an issue: ${repo}#${number}`); + if (!OWNER_REPO.test(repo ?? '') || !/^[1-9]\d*$/.test(String(number))) throw invalid(`not an issue: ${repo}#${number}`); const args = ['api', '--paginate', '--jq', '.[]', `repos/${repo}/issues/${number}/comments?per_page=100`]; const result = await exec('gh', args); if (result.status !== 0) throw new Error(`gh ${args.join(' ')} failed: ${(result.stderr || result.error?.message || 'no output').trim()}`); @@ -98,7 +102,7 @@ export function createFetcher({ exec = run } = {}) { /** Merged pull requests (or the closing commit) that fixed a thread; empty when none qualifies. */ async fixingChanges(id) { const [, repo, number] = ID.exec(id) ?? []; - if (!repo) throw new Error(`not an owner/repo#number id: ${id}`); + if (!repo) throw invalid(`not an owner/repo#number id: ${id}`); const [owner, name] = repo.split('/'); const answer = await json('gh', ['api', 'graphql', '-f', `query=${FIXING_CHANGES_QUERY}`, '-F', `owner=${owner}`, '-F', `name=${name}`, '-F', `number=${number}`]); return changesOf(repo, answer?.data?.repository); @@ -109,10 +113,10 @@ export function createFetcher({ exec = run } = {}) { * a rate limit never reads as "not contained". */ async contains(repo, refs, sha) { - if (!SHA.test(sha ?? '')) throw new Error(`not a commit: ${sha}`); - if (!OWNER_REPO.test(repo ?? '')) throw new Error(`not an owner/repo: ${repo}`); + if (!SHA.test(sha ?? '')) throw invalid(`not a commit: ${sha}`); + if (!OWNER_REPO.test(repo ?? '')) throw invalid(`not an owner/repo: ${repo}`); for (const ref of refs) { - if (!REF.test(ref ?? '')) throw new Error(`not a tag name: ${ref}`); + if (!REF.test(ref ?? '')) throw invalid(`not a tag name: ${ref}`); const args = ['api', `repos/${repo}/compare/${ref}...${sha}`, '--jq', '{status:.status}']; const result = await exec('gh', args); if (result.status !== 0) { @@ -120,7 +124,7 @@ export function createFetcher({ exec = run } = {}) { throw new Error(`gh ${args.join(' ')} failed: ${(result.stderr || result.error?.message || 'no output').trim()}`); } const { status } = JSON.parse(result.stdout); - if (!['behind', 'identical', 'ahead', 'diverged'].includes(status)) throw new Error(`gh ${args.join(' ')} returned status ${status}`); + if (!['behind', 'identical', 'ahead', 'diverged'].includes(status)) throw invalid(`gh ${args.join(' ')} returned status ${status}`); return { ref, contained: status === 'behind' || status === 'identical' }; } return { ref: null, contained: null }; @@ -132,8 +136,8 @@ export function createFetcher({ exec = run } = {}) { * every range to its highest published match with npm. */ async bundled(chain, name, at = null) { - for (const pkg of [...chain, name]) if (!PACKAGE_NAME.test(pkg ?? '')) throw new Error(`not a package name: ${pkg}`); - if (at !== null && !VERSION.test(at)) throw new Error(`not a version: ${at}`); + for (const pkg of [...chain, name]) if (!PACKAGE_NAME.test(pkg ?? '')) throw invalid(`not a package name: ${pkg}`); + if (at !== null && !VERSION.test(at)) throw invalid(`not a version: ${at}`); const carrierVersion = at ?? await json('npm', ['view', chain[0], 'version', '--json']); let [pkg, version] = [chain[0], carrierVersion]; const trail = [`${pkg} ${version}`]; @@ -159,13 +163,13 @@ export function createFetcher({ exec = run } = {}) { * only scheduled runs show that the watch is alive. */ async lastRun(repo) { - if (!OWNER_REPO.test(repo ?? '')) throw new Error(`not an owner/repo: ${repo}`); + if (!OWNER_REPO.test(repo ?? '')) throw invalid(`not an owner/repo: ${repo}`); const answer = await json('gh', ['api', `repos/${repo}/actions/workflows/upstream-watch.yml/runs?status=success&event=schedule&per_page=1`]); const run = answer?.workflow_runs?.[0]; return run ? { at: run.run_started_at, url: run.html_url } : null; }, async release({ channel, name }) { - if (!PACKAGE_NAME.test(name)) throw new Error(`not a package or repository name: ${name}`); + if (!PACKAGE_NAME.test(name)) throw invalid(`not a package or repository name: ${name}`); if (channel === 'npm') return releaseFacts('npm', await json('npm', ['view', name, 'time', 'dist-tags', '--json'])); return releaseFacts('github-release', await json('gh', ['api', `repos/${name}/releases?per_page=100`])); }, @@ -186,7 +190,7 @@ export function retrying(fetcher, { delays = [2000, 10_000], sleep = (ms) => new try { return await method.apply(fetcher, args); } catch (error) { - if (attempt >= delays.length) throw error; + if (error instanceof PermanentFetchError || attempt >= delays.length) throw error; await sleep(delays[attempt]); } } diff --git a/scripts/upstream-watch/ledger.mjs b/scripts/upstream-watch/ledger.mjs index 88a27467..12e75be5 100644 --- a/scripts/upstream-watch/ledger.mjs +++ b/scripts/upstream-watch/ledger.mjs @@ -5,6 +5,9 @@ // autolinks to this repository and `owner/repo#n` mentions the upstream thread. const code = (value) => `\`${value}\``; +/** Event names accepted by the ledger query and rendered below. */ +export const LEDGER_EVENTS = ['reply', 'acknowledged', 'closed', 'merged', 'released', 'reopened', 'stale', 'retire-proposed', 'retest-due', 'idle', 'fired', 'dispatch-pr']; + export const isoSeconds = (date) => new Date(date).toISOString().replace(/\.\d{3}Z$/, 'Z'); /** @@ -80,10 +83,13 @@ export function renderNotice({ records, mention, date, recordedAt }) { const bullets = items.map((item) => `- ${sentence(item)}${item.event === 'released' && sessions.has(item.id) ? ` Routine session: ${sessions.get(item.id)}` : ''}`); const head = `@${mention} upstream watch: ${items.length} ${items.length === 1 ? 'item needs' : 'items need'} you (${date}).`; const foot = `The full record: \`node scripts/upstream-watch.mjs ledger --recorded-since ${recordedAt}\``; + const lengths = [0]; + for (const bullet of bullets) lengths.push(lengths.at(-1) + bullet.length); for (let count = bullets.length; count > 0; count--) { const more = bullets.length - count; - const body = `${[head, '', ...bullets.slice(0, count), ...(more ? ['', `${more} more; see the ledger.`] : []), '', foot].join('\n')}\n`; - if (body.length <= NOTICE_MAX) return body; + const suffix = more ? `${more} more; see the ledger.` : ''; + const length = head.length + foot.length + lengths[count] + suffix.length + count + 4 + (more ? 2 : 0); + if (length <= NOTICE_MAX) return `${[head, '', ...bullets.slice(0, count), ...(more ? ['', suffix] : []), '', foot].join('\n')}\n`; } return `${[head, '', `${bullets.length} items; see the ledger.`, '', foot].join('\n')}\n`; } diff --git a/src/lib/hook-audit/agentic-dependency-constraints.json b/src/lib/hook-audit/agentic-dependency-constraints.json index 8f51ef94..ff0af7e5 100644 --- a/src/lib/hook-audit/agentic-dependency-constraints.json +++ b/src/lib/hook-audit/agentic-dependency-constraints.json @@ -287,7 +287,7 @@ "refs": ["closed-upstream review 2026-09-26"], "files": ["docs/adr/0034-schema-native-handoffs-and-hermetic-seats.md", "src/lib/execution/codex.mjs"] }, - "adjustment": "None: closed 2026-04-03 as model behaviour without a fix; parseHandoffText is the validation wrapper the maintainer recommended. Re-probe --output-schema with MCP on Codex upgrades (last probed on 0.149.1).", + "adjustment": "None: closed 2026-04-03 as model behaviour without a fix; parseHandoffText is the validation wrapper the maintainer recommended.", "status": "retired", "constraintIds": [], "history": [ diff --git a/tests/kit/upstream-watch-dispatch.test.mjs b/tests/kit/upstream-watch-dispatch.test.mjs index d6c6446a..11ee674c 100644 --- a/tests/kit/upstream-watch-dispatch.test.mjs +++ b/tests/kit/upstream-watch-dispatch.test.mjs @@ -5,7 +5,7 @@ import test from 'node:test'; import assert from 'node:assert/strict'; import { eventLine } from '../../scripts/upstream-watch/classify.mjs'; -import { FIRE_HEADERS, FIRE_URL, SAME_REPO_PR, createDispatcher, dispatch } from '../../scripts/upstream-watch/dispatch.mjs'; +import { FIRE_HEADERS, FIRE_URL, PR_OBSERVE_DAYS, SAME_REPO_PR, createDispatcher, dispatch, sessionList } from '../../scripts/upstream-watch/dispatch.mjs'; const NOW = new Date('2026-10-02T14:17:00Z'); const RECORDED_AT = '2026-10-02T14:17:00Z'; @@ -26,9 +26,15 @@ function fakeDispatcher({ exists = false, pr = null, session = 'https://claude.a fire: async (text) => { calls.fire.push(text); if (fireError) throw new Error(fireError); return session; }, }; } -const run = (dispatcher, records = [], { dryRun, list = [released] } = {}) => dispatch({ released: list, records, dispatcher, repo: 'pacphi/agentic-kit', sentinel: 'UPSTREAM-WATCH', now: NOW, recordedAt: RECORDED_AT, dryRun }); +const run = (dispatcher, records = [], { dryRun, list = [released], eligibleIds = new Set([ID]), now = NOW } = {}) => dispatch({ released: list, records, dispatcher, repo: 'pacphi/agentic-kit', sentinel: 'UPSTREAM-WATCH', now, recordedAt: RECORDED_AT, dryRun, eligibleIds }); const NOTHING = { records: [], errors: [], wouldFire: [] }; +test('exhausted firing links read naturally at any count', () => { + assert.equal(sessionList(['one']), 'one'); + assert.equal(sessionList(['one', 'two']), 'one and two'); + assert.equal(sessionList(['one', 'two', 'three']), 'one, two, and three'); +}); + test('a released fix without a branch fires once and is recorded with its session', async () => { const dispatcher = fakeDispatcher(); const result = await run(dispatcher); @@ -108,6 +114,22 @@ test('a fired branch with an open pull request records dispatch-pr once', async assert.deepEqual(again.calls.pr, []); }); +test('PR observation stops on ineligible status or seven days after latest firing', async () => { + assert.equal(PR_OBSERVE_DAYS, 7); + const first = fired('2026-09-20T14:17:00Z'); + const latest = fired('2026-09-26T14:17:00Z'); + const inactive = fakeDispatcher({ exists: true, pr: 261 }); + assert.deepEqual(await run(inactive, [latest], { list: [], eligibleIds: new Set() }), NOTHING); + assert.deepEqual(inactive.calls.pr, []); + const before = fakeDispatcher({ exists: true, pr: 261 }); + const inside = await run(before, [first, latest], { list: [], now: new Date('2026-10-03T14:16:59Z') }); + assert.equal(inside.records[0].event, 'dispatch-pr'); + assert.deepEqual(before.calls.pr, [['pacphi/agentic-kit', BRANCH]]); + const boundary = fakeDispatcher({ exists: true, pr: 261 }); + assert.deepEqual(await run(boundary, [first, latest], { list: [], now: new Date('2026-10-03T14:17:00Z'), dryRun: true }), NOTHING); + assert.deepEqual(boundary.calls.pr, []); +}); + test('the trigger call sends the payload with the documented headers and never prints the token', async () => { const requests = []; const fetchImpl = async (url, init) => { requests.push({ url, init }); return { status: 200, json: async () => ({ claude_code_session_url: 'https://claude.ai/code/session_x' }) }; }; diff --git a/tests/kit/upstream-watch-fixtures.mjs b/tests/kit/upstream-watch-fixtures.mjs index bb2dee0c..7d440260 100644 --- a/tests/kit/upstream-watch-fixtures.mjs +++ b/tests/kit/upstream-watch-fixtures.mjs @@ -7,6 +7,7 @@ import path from 'node:path'; import { UPSTREAM_REGISTRY_FILE } from '../../src/lib/hook-audit/upstream.mjs'; import { releaseFacts } from '../../scripts/upstream-watch/classify.mjs'; +import { PermanentFetchError } from '../../scripts/upstream-watch/fetch.mjs'; const FIXTURES = path.resolve('tests/fixtures/upstream-watch'); const threads = JSON.parse(fs.readFileSync(path.join(FIXTURES, 'threads.json'), 'utf8')).threads; @@ -55,7 +56,7 @@ export function fixtureFetcher({ authenticated = true, failing = new Set(), flak thread: async (id) => { if (failing.has(id)) throw new Error('HTTP 502'); if (flaky.get(id) > 0) { flaky.set(id, flaky.get(id) - 1); throw new Error('HTTP 502'); } - if (!threads[id]) throw new Error(`no fixture for ${id}`); + if (!threads[id]) throw new PermanentFetchError(`no fixture for ${id}`); return clone(threads[id]); }, release: async ({ name }) => releaseFacts('npm', npm[name]), diff --git a/tests/kit/upstream-watch-ledger-branch.test.mjs b/tests/kit/upstream-watch-ledger-branch.test.mjs index e7447970..231f2eaa 100644 --- a/tests/kit/upstream-watch-ledger-branch.test.mjs +++ b/tests/kit/upstream-watch-ledger-branch.test.mjs @@ -61,6 +61,33 @@ test('an absent ledger branch reads as an empty ledger without a fetch', async ( assert.deepEqual(calls.map((call) => call.args), [['ls-remote', '--exit-code', '--heads', 'origin', 'upstream-watch-ledger']]); }); +test('runWithInput reports a spawn failure without rejecting', async () => { + const result = await runWithInput('ak-missing-command-for-test', []); + assert.equal(result.status, null); + assert.equal(result.error?.code, 'ENOENT'); +}); + +test('read reports failures from rev-parse, show and log', async () => { + for (const failedCommand of ['rev-parse', 'show', 'log']) { + const sha = 'a'.repeat(40); + const responses = [ + { stdout: `${sha}\trefs/heads/upstream-watch-ledger\n` }, {}, + { stdout: `${sha}\n` }, { stdout: '' }, { stdout: '' }, + ]; + const index = { 'rev-parse': 2, show: 3, log: 4 }[failedCommand]; + responses[index] = { status: 128, stderr: `${failedCommand} failed` }; + const { exec, calls } = fakeExec(responses); + await assert.rejects(createLedgerStore({ exec }).read('upstream-watch-ledger', { now: NOW }), new RegExp(`git ${failedCommand} failed: ${failedCommand} failed`)); + assert.equal(calls.at(-1).args[0], failedCommand); + } +}); + +test('read rejects an invalid branch before any git call', async () => { + const { exec, calls } = fakeExec([]); + await assert.rejects(createLedgerStore({ exec }).read('../ledger'), /not a branch name/); + assert.equal(calls.length, 0); +}); + test('a failed ls-remote or fetch throws', async () => { const lookup = fakeExec([{ status: 128, stderr: 'fatal: unable to access: HTTP 403\n' }]); await assert.rejects(createLedgerStore({ exec: lookup.exec }).read('upstream-watch-ledger', { now: NOW }), /git ls-remote origin upstream-watch-ledger failed: fatal: unable to access/); diff --git a/tests/kit/upstream-watch-notice.test.mjs b/tests/kit/upstream-watch-notice.test.mjs index 7735d3ef..78c17bcf 100644 --- a/tests/kit/upstream-watch-notice.test.mjs +++ b/tests/kit/upstream-watch-notice.test.mjs @@ -33,6 +33,18 @@ test('a quiet run has no notice; an action run mentions the maintainer first', ( assert.match(body, /\nThe full record: `node scripts\/upstream-watch\.mjs ledger --recorded-since 2026-10-02T14:17:00Z`\n$/); }); +test('one action uses singular wording and the latest fired session for an id', () => { + const records = [ + rec('released', { version: '1.0.0', branch: 'upstream/ruvnet-ruflo-1' }), + rec('fired', { branch: 'upstream/ruvnet-ruflo-1', session: 'https://claude.ai/code/session_old' }), + rec('fired', { branch: 'upstream/ruvnet-ruflo-1', session: 'https://claude.ai/code/session_new' }), + ]; + const body = renderNotice({ records, mention: 'pacphi', date: '2026-10-02', recordedAt: '2026-10-02T14:17:00Z' }); + assert.match(body, /^@pacphi upstream watch: 1 item needs you/); + assert.match(body, /Routine session: https:\/\/claude\.ai\/code\/session_new/); + assert.doesNotMatch(body, /session_old/); +}); + test('thread ids never autolink; only a dispatch pull request number does', () => { const body = renderNotice({ records: [rec('reply', { by: 'x', at: '10:00:00Z' }, 'a/b#5'), rec('dispatch-pr', { branch: 'upstream/a-b-5', pr: 261 }, 'a/b#5')], mention: 'pacphi', date: '2026-10-02', recordedAt: '2026-10-02T14:17:00Z' }); const prose = body.replace(/`[^`]*`/g, ''); @@ -46,6 +58,17 @@ test('a notice too long for GitHub lists what fits and says how many more', () = assert.ok(body.length <= NOTICE_MAX, String(body.length)); assert.match(body, /\n\d+ more; see the ledger\.\n/); assert.ok(body.includes('`owner/repo#1`') && !body.includes('`owner/repo#3000`')); + const items = records.filter(isActionRecord); + const bullets = items.map((item) => `- ${sentence(item)}`); + const head = `@pacphi upstream watch: ${items.length} items need you (2026-10-02).`; + const foot = 'The full record: `node scripts/upstream-watch.mjs ledger --recorded-since 2026-10-02T14:17:00Z`'; + let expected; + for (let count = bullets.length; count > 0; count--) { + const more = bullets.length - count; + const candidate = `${[head, '', ...bullets.slice(0, count), ...(more ? ['', `${more} more; see the ledger.`] : []), '', foot].join('\n')}\n`; + if (candidate.length <= NOTICE_MAX) { expected = candidate; break; } + } + assert.equal(body, expected, 'the optimized truncation keeps the exact previous body'); }); // A commit message is plain text: GitHub turns `owner/repo#n` or `#n` there into diff --git a/tests/kit/upstream-watch-query.test.mjs b/tests/kit/upstream-watch-query.test.mjs index c13b3506..b064d4b6 100644 --- a/tests/kit/upstream-watch-query.test.mjs +++ b/tests/kit/upstream-watch-query.test.mjs @@ -60,6 +60,17 @@ test('--recorded-since belongs to ledger only', async () => { }); }); +test('ledger fails visibly on an invalid registry without reading the ledger', async () => { + await withRegistryFile([entry('ruvnet/ruflo#3153', { status: 'done' })], async (file) => { + let reads = 0; + const result = await query(file, [], { read: async () => { reads++; throw new Error('unexpected read'); } }); + assert.equal(result.code, 3); + assert.equal(reads, 0); + assert.match(result.err, /upstream registry is invalid/); + assert.doesNotMatch(result.out, /No report/); + }); +}); + test('the fetcher reads the last successful scheduled watch run from the Actions API', async () => { const calls = []; const exec = async (command, args) => { diff --git a/tests/kit/upstream-watch-record.test.mjs b/tests/kit/upstream-watch-record.test.mjs index c0f18243..6332946d 100644 --- a/tests/kit/upstream-watch-record.test.mjs +++ b/tests/kit/upstream-watch-record.test.mjs @@ -3,7 +3,7 @@ import test from 'node:test'; import assert from 'node:assert/strict'; -import { retrying } from '../../scripts/upstream-watch/fetch.mjs'; +import { PermanentFetchError, retrying } from '../../scripts/upstream-watch/fetch.mjs'; import { isActionRecord } from '../../scripts/upstream-watch/ledger.mjs'; import { main } from '../../scripts/upstream-watch.mjs'; import { @@ -17,15 +17,22 @@ async function record(file, argv, { fetcher = fixtureFetcher(), ledgerStore = me return { code, result: out.text() ? JSON.parse(out.text()) : null, err: err.text(), ledgerStore, dispatcher }; } -test('retrying retries every method but auth, then gives up', async () => { +test('retrying handles transient failures within the budget, but never retries auth or deterministic failures', async () => { let calls = 0; const waits = []; const fetcher = retrying({ auth: async () => { throw new Error('auth is not retried'); }, thread: async () => { calls += 1; if (calls < 3) throw new Error('HTTP 502'); return 'ok'; } }, { sleep: async (ms) => { waits.push(ms); } }); assert.equal(await fetcher.thread('a/b#1'), 'ok'); assert.deepEqual(waits, [2000, 10000]); await assert.rejects(fetcher.auth(), /auth is not retried/); - const always = retrying({ thread: async () => { throw new Error('HTTP 404'); } }, { sleep: noSleep }); - await assert.rejects(always.thread('a/b#1'), /HTTP 404/); + let exhausted = 0; + const always = retrying({ thread: async () => { exhausted++; throw new Error('HTTP 502'); } }, { delays: [1, 2], sleep: async (ms) => { waits.push(ms); } }); + await assert.rejects(always.thread('a/b#1'), /HTTP 502/); + assert.equal(exhausted, 3); + assert.deepEqual(waits, [2000, 10000, 1, 2]); + let deterministic = 0; + const invalid = retrying({ thread: async () => { deterministic++; throw new PermanentFetchError('no fixture'); } }, { sleep: async () => { assert.fail('deterministic failure slept'); } }); + await assert.rejects(invalid.thread('a/b#1'), /no fixture/); + assert.equal(deterministic, 1); }); test('record on an absent ledger starts from --since, commits every record and prints the notice', async () => { @@ -159,6 +166,7 @@ test('record is blind (exit 3) when gh, the ledger, the registry or every upstre assert.equal(invalid.code, 3); assert.deepEqual([invalid.result.blind, invalid.result.records, invalid.result.commit], [true, [], null]); assert.match(invalid.result.error, /registry/); + assert.equal(invalid.err.match(/upstream registry is/g)?.length, 1); }); }); @@ -181,3 +189,13 @@ test('future --since is a usage error', async () => { assert.match(err, /--since is in the future/); }); }); + +test('invalid registry takes precedence over a future --since', async () => { + await withRegistryFile([entry('ruvnet/ruflo#3153', { status: 'done' })], async (file) => { + const { code, result, err } = await record(file, ['--since', '2026-10-01T00:00:00Z']); + assert.equal(code, 3); + assert.equal(result.blind, true); + assert.match(err, /upstream registry is invalid/); + assert.doesNotMatch(err, /--since is in the future/); + }); +}); From bb2e7efe88abd3cb2d73baf26530386549e3dc2e Mon Sep 17 00:00:00 2001 From: Chris Phillipson Date: Tue, 29 Sep 2026 07:21:01 -0700 Subject: [PATCH 07/10] fix: remediate CLI, memory and upstream integration follow-ups (#275) * fix(cli): keep command usage failures machine-readable * fix(cli): validate usage and adapter revocation arguments * fix(host): keep dry-run JSON previews structured * fix(host): avoid status evidence writes in dry runs * fix(versions): throttle offline self retries by channel * fix(versions): scope self cache freshness to checked tags * test(maintain): guard injected refresh construction * fix(setup): isolate memory probe native mirror * test(evidence): prove repair commands re-record fresh facts * test(evidence): exercise setup machine host lifecycle wiring * test(aqe): verify live-lock fallback on installed artifact * test(aqe): guard live-lock probe cancellation * fix(memory): explain unsuitable locations and strict temp nesting * fix(memory): bind coexistence status to routing evidence * test(status): align offline self cache with checked tags * fix(memory): preserve cli proof across route checks * fix(memory): discover stray stores in ordinary dot directories * fix(status): report AQE home store separately * fix(status): replace broken AQE Codex setup hint * fix(memory): keep dependency markers and metadata errors out of complete scans * fix(status): give one ruflo component restart instruction * fix(daemon): preserve YAML configuration precedence * fix(sync): preview version-triggered daemon convergence * fix(daemon): respect explicit config and active restart source * fix(discovery): restore paused coverage from durable summary * fix(exec): abort owned process trees * fix(exec): bound uncertain Windows cleanup and byte caps * test(setup): isolate host rerecord project fixture * fix(test): hold C1 run root until owned children close * fix(test): accept Windows bootstrap env casing * test(ci): seed both requested self-drift tags * test(ci): probe pinned upstream conformance in isolated homes * docs(host-support): align upstream risks with verified releases * docs(upstream): register aqe init settings churn report * docs(host-support): clarify Ruflo source caveat * test(exec): retain a ready descendant after Windows parent exit * test(exec): launch a script through the native PowerShell fixture * test(ruflo): diagnose native Windows MCP transport boundaries * test(ruflo): isolate diagnostic launches and retain uncertain cleanup * test(ci): capture native Windows MCP transport diagnostics * fix(aqe): retire FsyncFailed live-lock exception * test(exec): hardcode the PowerShell fixture entry point * fix(exec): launch recognized npm Windows shims through their public bins * fix(identity): preserve exact persisted file IDs * test(ruflo): await natural closure for EOF diagnostics * test(exec): preserve native extensions in PowerShell ownership fixture * test(exec): keep reported parent out of cleanup authority * fix(live-checks): report skipped deja-vu and clean proof temp dirs * test(identity): correct Windows persisted identity fixtures * test(ci): retire completed native integration proof job * fix(test-runner): compare exact file identities before cleanup * docs(adr): scope self-version retry evidence to checked channels * docs(v4): record bounded C6 checks and accepted work * docs(remediation): archive verified V4 follow-ups plan --- .github/workflows/ci.yml | 2 +- bin/agentic-kit.mjs | 4 +- docs/adr/0055-aqe-embedding-lifecycle.md | 7 +- docs/adr/0062-aqe-project-store-integrity.md | 1 + ...3-evidence-store-and-refresh-vocabulary.md | 28 +- docs/archive/2026-09-28-plan-follow-ups-v2.md | 55 ++++ docs/archive/README.md | 1 + docs/host-support.md | 53 +++- docs/maintenance.md | 7 +- docs/plans/2026-09-28-follow-ups-v2.md | 35 --- .../2026-09-28-remediation-program-v2.md | 7 +- ...-09-28-remediation-v2-develop-execution.md | 17 +- docs/troubleshooting.md | 11 +- scripts/run-roots.mjs | 19 +- scripts/run-tests.mjs | 6 +- src/commands/audit.mjs | 14 +- src/commands/heal.mjs | 28 +- src/commands/models.mjs | 36 ++- src/commands/setup.mjs | 76 ++++-- src/commands/status.mjs | 4 +- src/commands/status/sections/codex-mcp.mjs | 2 +- src/commands/status/sections/daemons.mjs | 21 +- .../status/sections/project-memory.mjs | 16 +- .../status/sections/ruflo-components.mjs | 5 +- src/commands/status/sections/user-memory.mjs | 21 ++ src/commands/sync.mjs | 26 +- src/commands/telemetry.mjs | 4 +- src/commands/usage.mjs | 17 +- src/commands/x/aqe-embedding.mjs | 13 +- src/commands/x/aqe-store.mjs | 5 +- src/commands/x/codex-context.mjs | 8 +- src/commands/x/daemon-gc.mjs | 23 +- src/commands/x/harvest.mjs | 9 +- src/commands/x/host-adapters-grants.mjs | 10 +- src/commands/x/host-adapters.mjs | 20 +- src/commands/x/host.mjs | 127 ++++++--- src/commands/x/reference.mjs | 7 +- src/commands/x/skills.mjs | 5 +- src/commands/x/statusline.mjs | 15 +- src/lib/aqe-readiness.mjs | 22 +- src/lib/exec.mjs | 181 +++++++++---- src/lib/file-identity.mjs | 18 ++ .../agentic-dependency-constraints.json | 54 ++-- src/lib/hook-remediation/engine.mjs | 5 +- src/lib/hook-remediation/fs-port.mjs | 28 +- src/lib/hook-remediation/store.mjs | 3 +- src/lib/host-alignment.mjs | 7 +- src/lib/host-health-evidence.mjs | 5 +- src/lib/live-check-evidence.mjs | 20 +- src/lib/live-checks.mjs | 74 +++-- src/lib/live/jsonl-tailer.mjs | 20 +- src/lib/live/transcript-streams.mjs | 6 +- src/lib/maintenance/discovery/history.mjs | 15 +- .../maintenance/discovery/orchestrator.mjs | 33 ++- src/lib/maintenance/discovery/partitions.mjs | 7 +- src/lib/mcp-probe.mjs | 3 +- src/lib/mcp-tool-call.mjs | 3 +- src/lib/project-memory.mjs | 45 ++-- src/lib/ruflo-daemon-config.mjs | 92 +++++-- src/lib/ruflo-memory.mjs | 26 +- src/lib/versions.mjs | 50 ++-- src/lib/windows-npm-shim.mjs | 83 ++++++ tests/fixtures/npm-windows-shim/license.txt | 15 ++ tests/fixtures/npm-windows-shim/ruflo.cmd | 17 ++ tests/fixtures/npm-windows-shim/ruflo.ps1 | 28 ++ tests/kit/aqe-live-lock-process.test.mjs | 138 ++++++++++ tests/kit/aqe-verification.test.mjs | 39 +-- tests/kit/cli-json-honesty.test.mjs | 70 ++++- tests/kit/daemon-gc-rerecord.test.mjs | 62 +++++ tests/kit/deja-vu-teardown-verify.test.mjs | 2 +- tests/kit/exec.test.mjs | 254 +++++++++++++++++- tests/kit/helpers/home-sandbox.mjs | 4 +- tests/kit/host-dry-run.test.mjs | 85 ++++++ tests/kit/host-pick-rerecord.test.mjs | 57 ++++ tests/kit/live-check-evidence.test.mjs | 69 ++++- tests/kit/live-checks.test.mjs | 199 +++++++++++++- ...aintenance-discovery-orchestrator.test.mjs | 108 ++++++++ .../maintenance-management-service.test.mjs | 29 ++ .../maintenance-refresh-injection.test.mjs | 78 ++++++ tests/kit/models-command.test.mjs | 9 + tests/kit/persisted-file-identity.test.mjs | 229 ++++++++++++++++ tests/kit/project-memory.test.mjs | 62 ++++- tests/kit/ruflo-components-status.test.mjs | 40 +++ tests/kit/ruflo-daemon-config.test.mjs | 168 ++++++++++++ tests/kit/ruflo-memory-location.test.mjs | 64 +++++ .../ruflo-windows-diagnostic-process.test.mjs | 191 +++++++++++++ tests/kit/run-root-identities.test.mjs | 139 ++++++++++ tests/kit/run-tests-runner.test.mjs | 5 +- tests/kit/setup-host-rerecord.test.mjs | 106 ++++++++ tests/kit/setup-memory-probe.test.mjs | 112 ++++++++ tests/kit/status-command.test.mjs | 15 +- .../kit/status-version-drift-refresh.test.mjs | 2 +- tests/kit/sync-daemon-repair.test.mjs | 47 ++++ tests/kit/sync-dry-run-preview.test.mjs | 38 +++ tests/kit/sync-self-freshness.test.mjs | 159 +++++++++++ tests/kit/telemetry-cli.test.mjs | 9 + tests/kit/upstream-watch-registry.test.mjs | 13 +- tests/kit/verify-memory-routes.test.mjs | 22 +- tests/kit/windows-npm-shim.test.mjs | 151 +++++++++++ tests/live/aqe-live-lock-conformance.test.mjs | 150 +++++++++++ tests/live/aqe-live-lock-process.mjs | 107 ++++++++ .../live/ruflo-windows-diagnostic-process.mjs | 103 +++++++ tests/live/ruflo-windows-transport.test.mjs | 95 +++++++ 103 files changed, 4333 insertions(+), 522 deletions(-) create mode 100644 docs/archive/2026-09-28-plan-follow-ups-v2.md delete mode 100644 docs/plans/2026-09-28-follow-ups-v2.md create mode 100644 src/lib/file-identity.mjs create mode 100644 src/lib/windows-npm-shim.mjs create mode 100644 tests/fixtures/npm-windows-shim/license.txt create mode 100644 tests/fixtures/npm-windows-shim/ruflo.cmd create mode 100644 tests/fixtures/npm-windows-shim/ruflo.ps1 create mode 100644 tests/kit/aqe-live-lock-process.test.mjs create mode 100644 tests/kit/daemon-gc-rerecord.test.mjs create mode 100644 tests/kit/host-pick-rerecord.test.mjs create mode 100644 tests/kit/maintenance-refresh-injection.test.mjs create mode 100644 tests/kit/persisted-file-identity.test.mjs create mode 100644 tests/kit/ruflo-windows-diagnostic-process.test.mjs create mode 100644 tests/kit/run-root-identities.test.mjs create mode 100644 tests/kit/setup-host-rerecord.test.mjs create mode 100644 tests/kit/windows-npm-shim.test.mjs create mode 100644 tests/live/aqe-live-lock-conformance.test.mjs create mode 100644 tests/live/aqe-live-lock-process.mjs create mode 100644 tests/live/ruflo-windows-diagnostic-process.mjs create mode 100644 tests/live/ruflo-windows-transport.test.mjs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b5dfc6d8..fb59a044 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -60,7 +60,7 @@ jobs: # The smoke test proves local CLI behavior, not npm reachability. # Seed a fresh empty drift cache so an offline Windows runner does # not pay four sequential 20s `npm view` timeouts inside status. - node -e 'const fs=require("node:fs"),p=require("node:path"),base=process.platform==="win32"?process.env.APPDATA:process.env.XDG_CONFIG_HOME,dir=p.join(base,"agentic-kit"),last=Date.now();fs.mkdirSync(dir,{recursive:true});fs.writeFileSync(p.join(dir,"kit.json"),JSON.stringify({versionCheck:{last,seen:{},self:{last,best:null}}}))' + node -e 'const fs=require("node:fs"),p=require("node:path"),base=process.platform==="win32"?process.env.APPDATA:process.env.XDG_CONFIG_HOME,dir=p.join(base,"agentic-kit"),last=Date.now();fs.mkdirSync(dir,{recursive:true});fs.writeFileSync(p.join(dir,"kit.json"),JSON.stringify({versionCheck:{last,seen:{},self:{last,best:null,lastTags:["latest","next"]}}}))' node bin/agentic-kit.mjs --version node bin/agentic-kit.mjs --help --all > /dev/null # status must emit valid JSON and exit deterministically even on a diff --git a/bin/agentic-kit.mjs b/bin/agentic-kit.mjs index d18c1ba6..b73b87c9 100755 --- a/bin/agentic-kit.mjs +++ b/bin/agentic-kit.mjs @@ -194,7 +194,9 @@ async function main() { } catch (err) { if (!String(err?.code ?? '').startsWith('ERR_PARSE_ARGS_')) throw err; if (cmd === 'telemetry') { - console.error('Telemetry failed: invalid command options.'); + const error = 'Telemetry failed: invalid command options.'; + console.error(error); + console.log(JSON.stringify({ error, exitCode: 2 })); return 2; } // Under --json a rejected option still answers with one JSON object (the diff --git a/docs/adr/0055-aqe-embedding-lifecycle.md b/docs/adr/0055-aqe-embedding-lifecycle.md index 0f5bacd4..2475c1df 100644 --- a/docs/adr/0055-aqe-embedding-lifecycle.md +++ b/docs/adr/0055-aqe-embedding-lifecycle.md @@ -14,6 +14,7 @@ - **Updated:** 2026-09-27 — the recognizer accepts every plain npx spelling of AQE's server (optional `-y`/`--yes`; unversioned, `@latest` or an exact version), audit item 5 choice A - **Updated:** 2026-09-27 — a passing embedding check reads "embedder verified"; status, `ak x verify aqe` and setup say AQE's pattern index binding stays unverified (agentic-qe#754) and corpus compatibility stays separate - **Updated:** 2026-09-27 — the busy rule's removal condition is agentic-qe#574 fixed in a released agentic-qe that is the kit floor; agentic-qe#719 (carried by 3.14.4) is only a partial fix +- **Updated:** 2026-09-29 — the later approved N-1 criterion supersedes that floor condition: released AQE 3.14.4 passed native macOS and Linux live-owner probes without `FsyncFailed`, so ak removed the exact 3.14.3 `FsyncFailed`-as-busy exception. Any `FsyncFailed`/`0x0303` now fails, including the old sequence. Ordinary `LockHeld` SQLite fallback remains busy with owner health and RVF integrity unknown. AQE 3.14.4 is the verified baseline for this decision, not a universal minimum; native Windows AQE conformance was not run - **Updated:** 2026-09-27 — beside these projections, `ak sync` and `ak setup` pin AQE to the project root (absolute `AQE_PROJECT_ROOT`, `AQE_MEMORY_PATH`, `AQE_STORAGE_PATH`) in `.claude/settings.local.json`, the recognized `.mcp.json` entry and both AQE tables of the project `.codex/config.toml`, under receipts from the same owned-env engine; a file git tracks is not pinned, and AQE's own relative `AQE_MEMORY_PATH` is taken back even after an AQE re-init. Stray AQE stores are merged and archived by `ak x aqe-store merge`. See [ADR-0062](0062-aqe-project-store-integrity.md) (remediation Branch 5, B5-D1 to B5-D5, B5-M5) - **Updated:** 2026-09-28 — live-check evidence storage relocated from `/agentic-kit/live-checks/.json` to the shared `/agentic-kit/evidence/live-check/.json` layout; this ADR's own live-check BEHAVIOR (TTL, statuses, remembered-check display) is unchanged, only where the evidence file lives. `ak x aqe-embedding verify` (distinct from `ak x verify`'s `aqe-embedding` row) does not persist evidence either way — it is a one-shot, unpersisted synthetic-backend proof. See [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md) (remediation program, branch 6a task 2) - **Updated:** 2026-09-28 — `ak x verify` is retired; its live checks (this ADR's own quick @@ -21,9 +22,9 @@ full `aqe` proof runs with `--only aqe`. A live-check evidence row now carries the source id `status-refresh-live`, labelled "ak status --refresh=live"; a row recorded before this rename under the retired `verify`/`status-live` source ids still reads back, labelled "an earlier live - check" — the label never names a retired command. Evidence ids are unchanged: `memory-routes` - still records under the `memory` id, and the full `aqe` proof still records only its embedding - request under `aqe-embedding` (remediation program, branch 6b; see + check" — the label never names a retired command. `memory-routes` records its CLI round trip + under `memory` and its routing observation under `memory-routes`; the full `aqe` proof still + records only its embedding request under `aqe-embedding` (remediation program, branch 6b; see [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md)) - **Related:** [ADR-0023](0023-fail-closed-operations-and-explicit-degradation.md), [September repair](https://github.com/pacphi/agentic-kit/blob/main/docs/archive/2026-09-09-audit-aqe-integration-repair.md) diff --git a/docs/adr/0062-aqe-project-store-integrity.md b/docs/adr/0062-aqe-project-store-integrity.md index 2790861d..dcd5b6d5 100644 --- a/docs/adr/0062-aqe-project-store-integrity.md +++ b/docs/adr/0062-aqe-project-store-integrity.md @@ -6,6 +6,7 @@ re-init value taken back, clean release; holder checks that time out refuse; nested repositories are not strays; stores fingerprinted at copy time; root checked before backup; applying receipt; starter patterns the root holds keep their usage +- **Updated:** 2026-09-29 — released AQE 3.14.4 passed native macOS and Linux live-owner conformance with `LockHeld` and no `FsyncFailed`; ak retired only the old exact `FsyncFailed`-as-busy exception in AQE startup classification. This is a verified baseline for that rule, not a universal AQE minimum. The 3.14.4 minimum below applies only to store merge. Native Windows AQE conformance remains unverified - **Deciders:** agentic-kit maintainers - **Related:** [ADR-0016](0016-capability-driven-integration-adapters.md) (project memory status and stray stores), [ADR-0055](0055-aqe-embedding-lifecycle.md) (the AQE embedding projections this diff --git a/docs/adr/0063-evidence-store-and-refresh-vocabulary.md b/docs/adr/0063-evidence-store-and-refresh-vocabulary.md index f7001912..8a0e1752 100644 --- a/docs/adr/0063-evidence-store-and-refresh-vocabulary.md +++ b/docs/adr/0063-evidence-store-and-refresh-vocabulary.md @@ -1,7 +1,7 @@ # ADR-0063 — One evidence store and the refresh vocabulary - **Status:** Accepted -- **Updated:** 2026-09-29 — Branch 6c delivered the dashboard refresh operation and retired GET-started scans +- **Updated:** 2026-09-29 — Branch 6c delivered the dashboard refresh operation and retired GET-started scans; 2026-09-29 V4 A3: self-version retry attempts and last freshness are scoped to checked channels - **Earlier update:** 2026-09-28 — Branch 6b delivered the CLI refresh vocabulary - **Date:** 2026-09-28 - **Deciders:** agentic-kit maintainers @@ -430,9 +430,9 @@ this branch" section and "Delivered in 6c" below. `learning`, `harvest`, `aqe`, `memory-routes` — slow, `--only`-only) now live in `src/lib/live-checks.mjs` and run as `--refresh=live`'s `live` stage. `memory` is the quick store/retrieve/purge round trip; `memory-routes` additionally observes whether the CLI and MCP - see each other's writes, and its result is remembered under the `memory` evidence id, not its - own. A live-check evidence row now carries the source id `status-refresh-live`, labelled "ak - status --refresh=live"; a row recorded before this branch under the retired `verify` or + see each other's writes. Its CLI result is remembered under `memory` and its routing result + under `memory-routes`. A live-check evidence row carries the source id + `status-refresh-live`, labelled "ak status --refresh=live"; a row recorded before this branch under the retired `verify` or `status-live` source ids still reads back, labelled "an earlier live check" — the label never names a retired command (`live-check-evidence.mjs`'s `SOURCE_LABEL`). - **The Codex quota presence gate.** `/api/limits` asks `codex app-server` for its @@ -539,25 +539,7 @@ shared evidence envelope. ## Known limitations (recorded, not fixed, by this branch) -1. **Resolved in Branch 6b: the failed-lookup rule is now the same in all four version-drift - functions.** This item originally recorded that `ruvector.mjs`/`ruvnet-brain.mjs`'s `drift()` - could silently drop a known update on a failed forced fetch, unlike `versions.mjs`'s - `driftReport()`/`selfDrift()`. Branch 6b fixed both (`fix(versions): a failed lookup keeps the - cached version and waits one TTL window before retrying`, and its follow-ups), so all four now - share one rule: on a total lookup failure, the cached `latest`/`best`/`installedRelease` value - is kept — never overwritten with `null` — and the TTL stamp (`last`) is restamped, so the next - unforced call waits one more TTL window before retrying (`force` bypasses this and retries - immediately). `observedAt` records the real time a value was last actually observed, not the - time of a failed retry: `ruvector.mjs`'s `drift()` keeps `observedAt: cached.observedAt ?? - cached.last` on failure (`:70`); `ruvnet-brain.mjs`'s `drift()` does the same - (`:288`, `recordedRelease()`); `versions.mjs`'s `driftReport()`'s `lookUpLatest()` restamps - `observedAt` from the prior `last` only for packages that were never individually observed - (`:82`); its `selfDrift()`'s `selfRecord()` restamps on a *total* failure, including one with no - cached candidate at all — only a partial answer (something answered live but did not win), or a - total failure whose cached candidate is unusable (a `next` candidate on a stable install), saves - nothing (`:172-176`). None of the four applies this rule under `record: false` (`ak sync - --dry-run`, ADR-0063's own `record` parameter) or a cache-only read (`cacheOnly: true`, `ak - sync --skip `): both skip the network and the write entirely, by design. +1. **Resolved in Branch 6b, refined in V4 A3: failed version lookups retain recorded evidence without claiming a new observation.** A total failed lookup of managed packages, Brain, or ruvector keeps its cached candidate and restamps `last` for one retry per configured TTL; `observedAt` remains the time that candidate was actually seen. The kit follows that rule when its cached candidate is usable. A partial answer that leaves the kit's cached candidate winning, or a stable install whose cached `next` candidate is unusable, keeps `last`, `observedAt`, and `best` unchanged and separately records `versionCheck.self.attempt` with its time and exact channel tags. Successful and total-failure kit lookups record `lastTags`, the channel scope of `last`; changing from stable to prerelease therefore probes an untried `next` channel even within the prior TTL. Legacy records without `lastTags` are reused for the single `latest` channel or where a `next` winner proves it was checked; a legacy `latest` winner cannot suppress an untried `next`. Malformed or future attempt metadata cannot suppress retries. `force` bypasses freshness. `record: false` permits a lookup without saving its result or attempt, while `cacheOnly: true` performs neither a lookup nor a write. 2. **`globalRoot()`'s `record`-persistence structural fragility** — see "The `npm-global-root` exception" above. 3. **Task 9's dashboard timeout bounds async hangs only** — see "The dashboard poll's two cost diff --git a/docs/archive/2026-09-28-plan-follow-ups-v2.md b/docs/archive/2026-09-28-plan-follow-ups-v2.md new file mode 100644 index 00000000..fc1c8a5e --- /dev/null +++ b/docs/archive/2026-09-28-plan-follow-ups-v2.md @@ -0,0 +1,55 @@ +# Follow ups v2: V4 branch plan + +## Status at archival + +**Implementation and local verification complete (2026-09-29).** All eight local +gates passed at `2915265402b758ddcd73d0dd663db9308637f3b2`: 6,159 unit tests passed +with seven skips and no failures, 514 browser assertions plus 15 UI tests passed, +and typecheck, lint, complexity, Markdown, build and offline links passed. +Measured coverage was 94.00% lines, 83.36% branches and 93.27% functions. +Independent whole-branch review found no actionable findings; its additional +focused run passed 131 tests with two Windows-only skips. + +B1 merged in PR #273, V3 dashboard changes in #276, C3 trace in #277 and C4 watch +in #278. This branch includes green `develop@989c5e56`, the reviewed exact runner +identity follow-up `f80bc55e`, temporary C1 job removal `257e6940`, A3 ADR amendment +`44dc4e9` and C6 evidence alignment `29152654`. B13 required no product fix after +the approved conditional check. B6's extra Codex hook fix line remains deferred +pending a Ruflo-supported answer to #3419. + +Native macOS/Linux AQE live-lock conformance passed on the named released +artifacts; native Windows AQE was not run. Native Windows Ruflo 3.48.0 memory +visibility was observed with its native bridge disabled. Final-head feature PR +CI, including the corrected Windows identity fixtures, and squash integration +remain pending at capture. This archive does not claim main merge, release, +installation or operational cleanup. + +The [remediation program V4](../plans/2026-09-28-remediation-program-v2.md#v4-fixfollow-ups-v2-every-small-product-cli-and-upstream-item) defines scope. The [archived Branch 9 plan](2026-09-28-superpowers-plan-branch-9-follow-ups.md) supplies task details. Paths below name current source seams and focused test targets. After an explicit directory prefix, subsequent bare filenames in the same cell use that directory. A new test named below is a proposed file. Later implementers must verify dependencies before editing. + +| Row | Source or artifact mapping | Focused proof and prerequisite | +| --- | --- | --- | +| A1 | `bin/agentic-kit.mjs`; `src/commands/usage.mjs`, `models.mjs`, `audit.mjs`, `heal.mjs`, `telemetry.mjs`, `x/host.mjs` | `tests/kit/cli-json-honesty.test.mjs`, `usage-cli.test.mjs`, `models-command.test.mjs`, `telemetry-cli.test.mjs`, `status-command.test.mjs`; include unknown models verb and status positional | +| A2 | `src/commands/x/host.mjs`; `bin/agentic-kit.mjs` | `tests/kit/host-dry-run.test.mjs`, `host-cli-migration.test.mjs`; pick refusal, off, reset-routes under `--dry-run --json` | +| A3 | `src/lib/versions.mjs`; `docs/adr/0063-evidence-store-and-refresh-vocabulary.md` | Accepted in `175677a6`; `versionCheck.self.attempt` and `lastTags` scope offline retries. ADR-0063 item 1 is amended; local gates passed at `29152654`, with final PR CI pending | +| A4 | `src/commands/status.mjs`; `src/lib/refresh.mjs` | `tests/kit/refresh.test.mjs`, `status-version-drift-refresh.test.mjs`; injected `refreshStages` plus `service` builds no collector | +| B1 | `src/lib/paths.mjs`; `src/lib/footprint/index.mjs`, `storage.mjs`, `consumers.mjs`, `storage-reclaim-detectors.mjs`, `install.mjs`; `src/lib/host-readiness-local.mjs`, `live/process-sessions.mjs`, `hook-audit/providers/opencode.mjs`, `usage-opencode.mjs`; `src/commands/uninstall.mjs` | `tests/kit/xdg-relative.test.mjs` and specified regressions; exact-head CI gate passed before edit; preserve nullable OpenCode fallback | +| B2 | `src/commands/x/daemon-gc.mjs`, `src/commands/x/host.mjs`, `src/commands/setup.mjs` | New `tests/kit/daemon-gc-rerecord.test.mjs`, `setup-host-rerecord.test.mjs`, `host-pick-rerecord.test.mjs`; Branch 9 Task 8 plus deferred host pick; compare `sync-host-repair.test.mjs` | +| B3 | `src/lib/ruflo-memory.mjs`, `paths.mjs` | `tests/kit/ruflo-memory-location.test.mjs`, `project-memory-status.test.mjs`; compose both unsuitable reasons and make `inside()` exclude equality | +| B4 | `src/commands/status/sections/project-memory.mjs`; `src/lib/live-check-evidence.mjs`, `live-checks.mjs` | **Accepted:** distinct `memory-routes` evidence binds installed CLI version and platform; generic `memory` cannot lower the row. Focused evidence, runner, status, and routing tests cover pass, upgrade, failure, timeout, and read-only render. | +| B5 | `src/lib/project-memory.mjs`; `src/commands/status/sections/user-memory.mjs`, `codex-mcp.mjs`; #757 registry entry | **Accepted:** bounded ordinary dot-folder discovery, read-only AQE home data row, and an AQE-owned init hint. `tests/kit/project-memory.test.mjs`, `project-memory-status.test.mjs`, `ruflo-memory-location.test.mjs`, `status-command.test.mjs` cover the three units. No real store was merged or moved. | +| B6 | `src/commands/status/sections/ruflo-components.mjs` | **Accepted:** applied-but-unverified keeps its state and meaning in the message and gives one restart/recheck instruction in its manual fix. Rendered-row and neighboring-state tests cover the contract. The Codex-hooks fix line remains conditional on a Ruflo-supported answer to #3419 and a pre-PR recheck. | +| B7 | `src/lib/ruflo-daemon-config.mjs`; `src/commands/sync.mjs`, `sync/plan-versions.mjs` | `tests/kit/sync-daemon-repair.test.mjs`, `sync-dry-run-preview.test.mjs`, `sync-skip-versions.test.mjs`; F6 hidden YAML keys and F7 versions-only preview parity | +| B8 | `src/lib/maintenance/discovery/orchestrator.mjs`, `history.mjs` | `tests/kit/maintenance-discovery-orchestrator.test.mjs`, `maintenance-recovery.test.mjs`; restart after pause shows paused history | +| B9 | `src/lib/exec.mjs`, `execution/process-tree.mjs` | `tests/kit/process-tree.test.mjs`; abort kills descendants; Windows CI required | +| B10 | `src/lib/maintenance/discovery/partitions.mjs`; inventory `src/lib/live/jsonl-tailer.mjs`, `live/transcript-streams.mjs`, `telemetry/store.mjs`, `maintenance/management/service-store.mjs` for additional persisted IDs | `tests/kit/file-identity-bigint.test.mjs`; distinguish IDs above `2^53`; enumerate the exact sites before edit | +| B11 | `src/lib/live-checks.mjs` | `tests/kit/live-checks.test.mjs`; skipped deja-vu check says skipped and check-created temp folders are cleaned | +| B12 | `src/commands/setup.mjs`; `src/lib/memory-probe-cleanup.mjs` unchanged | **Fixed:** `tests/kit/setup-memory-probe.test.mjs`; Ruflo 3.48.0 seeded reproduction created an unused native side file, and a disposable candidate run confirmed a private mirror leaves no canonical side file or probe row | +| B13 | `src/commands/sync.mjs`; `src/lib/aqe-project-pin.mjs` | Conditional check found the AQE pin converged across all four targets; no B13 product fix was made | +| C1 | `.github/workflows/ci.yml`; `src/lib/aqe-readiness.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | **Accepted; temporary CI job removed:** native macOS and Linux live-owner probes on released AQE 3.14.4 omitted `FsyncFailed`; the exact exception is retired. Ordinary `LockHeld` remains busy, while any `FsyncFailed` fails. The temporary CI job was removed in `257e6940` after its evidence was reviewed. #240 closure waits for the final main PR. No native Windows AQE conformance is claimed. | +| C2 | `docs/host-support.md`; `src/lib/hook-audit/agentic-dependency-constraints.json` | `tests/kit/ruflo-support-window.test.mjs` plus link check; verify AQE 3.14.4 #528/#532/#535 and Ruflo #2356/#420 first | +| C3 | `.github/workflows/nightly.yml`; `scripts/trace-ort.mjs` | Merged #277; native macOS trace run 36567908852. Exact approved comment posted and body verified at [Ruflo #2885](https://github.com/ruvnet/ruflo/issues/2885#issuecomment-5891510508). The post-heal trace is noncausal; the learning step remains nonblocking even though its outer command exited 1. | +| C4 | `scripts/upstream-watch/classify.mjs`, `fetch.mjs`, `ledger.mjs`, `dispatch.mjs`, `render.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | Merged #278 as `989c5e56`; develop CI 36578593054 passed all 13 jobs. M10 remained declined. | +| C5 | `src/lib/aqe-guidance.mjs`; `src/commands/setup.mjs`; ignored `.superpowers/sdd/2026-09-28-follow-ups-v2/c5-issue-draft.md` | Approved exact AQE repeated-init issue posted as [#778](https://github.com/proffesor-for-testing/agentic-qe/issues/778); the 3.14.4 disposable repro does not establish 3.14.5 behavior | +| C6 | `docs/host-support.md`; `src/lib/ruflo-support-window.mjs`, `aqe-readiness.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | 2026-09-29 13:42 UTC registry: Ruflo 3.48.0, AQE 3.14.5, Codex 0.159.0. Narrow disposable Ruflo one-file scan and integrity-verified native Codex read-only App Server initialize passed; AQE 3.14.4/3.14.5 live-lock proof passed on macOS/Linux. No provider turn or native Windows AQE proof. Local V4 gates passed at `29152654`; final PR CI pending | + +B1 used disposable homes, guarded focused tests, and the ignored B1 report at `.superpowers/sdd/2026-09-28-follow-ups-v2/b1-report.md`. No shared manifests, lockfiles, ADR index, or decision log change belongs to this plan update. The controller owns integration and the whole-branch gate. diff --git a/docs/archive/README.md b/docs/archive/README.md index 559d1038..6e65c9bb 100644 --- a/docs/archive/README.md +++ b/docs/archive/README.md @@ -68,6 +68,7 @@ reconfirmed by this metadata audit. The per-file inventory and limitations are r | File | Original location | What it was | Why it's historical | |---|---|---|---| +| [2026-09-28-plan-follow-ups-v2.md](2026-09-28-plan-follow-ups-v2.md) | `docs/plans/2026-09-28-follow-ups-v2.md` | V4 product, CLI, memory, process-lifecycle and upstream integration follow-ups. | All eight local gates and independent whole-branch review passed at `29152654`; final-head PR CI and squash integration were pending at archival. Conditional Ruflo #3419 guidance remains deferred; native Windows AQE was not tested. | | [2026-09-29-plan-upstream-watch-followups.md](2026-09-29-plan-upstream-watch-followups.md) | `docs/plans/2026-09-29-upstream-watch-followups.md` | V4 C4 execution plan for bounded PR polling, deterministic retry failures and twelve deferred watcher minors. | Implementation and independent review complete; all eight local gates passed at `b75c1e3f`. Final PR CI and merge were pending at archival. Current contract: [Upstream watch](../upstream-watch.md). | | [2026-06-upstream-findings-f1-f6.md](2026-06-upstream-findings-f1-f6.md) | `docs/upstream/ruflo-self-improvement-findings.md` | The F1–F6 findings series: proofs/refutations of ruflo's self-improvement claims (Q-learning persistence, state-encoder collapse, SONA learn→inference wiring, native-training misreporting), with filed upstream issues. | Every finding is now fixed upstream: F2 in 3.10.6 ([#2222](https://github.com/ruvnet/ruflo/issues/2222)), F2b in 3.10.7, F3 in 3.10.11 ([#2239](https://github.com/ruvnet/ruflo/issues/2239)), F4 in `@ruvector/ruvllm` 2.5.6 ([RuVector#519](https://github.com/ruvnet/RuVector/issues/519)), F6 in 3.18.1/3.19.0 + ruvllm 2.5.7 ([#2549](https://github.com/ruvnet/ruflo/issues/2549), closed 2026-07-03). | | [2026-06-token-consumption-incident.md](2026-06-token-consumption-incident.md) | `docs/usage/token-consumption-findings-and-mitigation-2026-06.md` | Root-cause report for the June 2026 token-burn incident: six immortal auto-started daemons consumed ~8.1B tokens over 7 days via headless worker sessions. Produced the opt-in daemon policy, TTL reaper, ⚙ statusline alarm, and `ruflo-token-audit`. | The root cause was fixed upstream in ruflo 3.27/3.28 ([#2661](https://github.com/ruvnet/ruflo/issues/2661)): AI workers are opt-in, launches are governed by a machine-wide budget with telemetry, one supervisor daemon per repo, native daemon TTL. The kit's daemon policy flipped back to default-on (local-only workers) on that baseline; the reapers and token-audit remain as an independent check. | diff --git a/docs/host-support.md b/docs/host-support.md index 6bbfb259..fc445551 100644 --- a/docs/host-support.md +++ b/docs/host-support.md @@ -26,6 +26,24 @@ Claude Code `2.1.222`, Codex CLI `0.146.0`, and OpenCode `1.18.x`. Host and upstream behavior changes quickly; open issues below are a risk snapshot, not a promise that an issue remains open forever. +**2026-09-29 support-window addendum.** Registry metadata at 13:42 UTC listed +Ruflo 3.48.0, Agentic QE 3.14.5, and Codex CLI 0.159.0. In a network-denied, +disposable scan, the installed Ruflo 3.48.0 `security secrets --action scan +--path ` reported one synthetic file scanned and exited 0 without +changing the target. The npm-integrity-verified native Codex 0.159.0 binary +accepted `-s read-only -a never app-server`; an `initialize` request answered +successfully without a provider turn. These are narrow command checks, not +end-to-end host conformance. + +Released AQE 3.14.4 passed disposable live-owner lock checks on macOS and Linux: +the status command and shipped adapter reported `LockHeld` without +`FsyncFailed`, while the holder and storage bytes remained intact. The same +check passed on AQE 3.14.5 on macOS and Linux. Native Windows AQE was not run. +Ruflo 3.48.0 showed CLI-to-MCP and MCP-to-CLI memory visibility on native +Windows with one `memory.db`; the reported backend was sql.js + HNSW with its +native bridge disabled. The earlier Linux result was asymmetric, so these +observations do not establish one cross-platform native-backend guarantee. + The stock OpenCode gateway acceptance test currently covers the stable compatibility window **`>=1.18.18 <1.19.0`**. This is a tested release-line window, not a claim that every future OpenCode release is compatible and not a target for `ak sync` to @@ -118,16 +136,21 @@ Official extension references: [Claude hooks](https://code.claude.com/docs/en/ho | Upgrade convergence | `ak sync` heals managed assets | `ak sync` heals Ruflo/AQE access and retires owned legacy MCP | `ak sync` regenerates the embedded catalogue and repairs exact-receipted plugins/config | | Teardown | Managed blocks and registrations | Receipt-based managed teardown | Value- and hash-receipt teardown; user-owned values survive | -Ruflo MCP access and Ruflo-backed inference are different contracts. In -particular, Ruflo's [`agent_execute` provider-key behavior](https://github.com/ruvnet/ruflo/issues/2356) -can still require a separate provider credential even when invoked from Codex. +Ruflo MCP access and Ruflo-backed inference are different contracts. +In Ruflo 3.48.0's shipped `agent_execute` path, execution uses a separately +configured inference provider; its no-provider branch still returns an error +instead of delegating to the MCP host +([ruflo #2356](https://github.com/ruvnet/ruflo/issues/2356)). This is a source +check, not a credentialed runtime probe from Codex. `ak run` avoids that conflation by executing the selected host directly and using Ruflo for tools, memory, routing context, and orchestration assets. The dated upstream risk inventory includes: -- force initialization can overwrite unrelated `.mcp.json` content - ([ruflo #420](https://github.com/ruvnet/ruflo/issues/420)); +- force initialization still writes a generated `.mcp.json` over the existing + file in Ruflo 3.48.0's shipped source, without merging unrelated servers + ([ruflo #420](https://github.com/ruvnet/ruflo/issues/420)); this was not + exercised on a real project; - generated Claude and Codex instructions can diverge ([#2638](https://github.com/ruvnet/ruflo/issues/2638)); - init and plugin installation can duplicate assets or hooks @@ -139,6 +162,12 @@ The dated upstream risk inventory includes: - hierarchical AgentDB writes can report success without durable persistence ([#2887](https://github.com/ruvnet/ruflo/issues/2887)). +The additional Codex hook-environment fix line remains conditional. At the +2026-09-29 check, [Ruflo #3419](https://github.com/ruvnet/ruflo/issues/3419) +was open with only agentic-kit's Codex source-analysis comment, not a +maintainer-supported answer or a live hook observation. Version tags alone do +not close that evidence gap. + ## Agentic QE support | AQE capability | Claude Code | Codex | OpenCode | @@ -171,12 +200,18 @@ Current AQE includes a subscription-backed `codex` provider. Agentic-kit accepts `ak host pick --aqe-provider codex`, admits Codex fallback rungs, enables Codex providers referenced by `agentOverrides`, and projects Codex activity routes. -The dated AQE risk inventory includes its -[MCP entrypoint double-spawn](https://github.com/proffesor-for-testing/agentic-qe/issues/528), -[multi-platform initialization behavior](https://github.com/proffesor-for-testing/agentic-qe/issues/532), -[MCP tool correctness gaps](https://github.com/proffesor-for-testing/agentic-qe/issues/535), +AQE 3.14.4 adopted fixes for the +[MCP entrypoint double-spawn](https://github.com/proffesor-for-testing/agentic-qe/issues/528) +and [exclusive platform initialization](https://github.com/proffesor-for-testing/agentic-qe/issues/532) +(`aqe init --no-claude`). Both upstream issues remain open; versions below +3.14.4 retain those gaps. The remaining dated AQE risk inventory includes +[GOAP `maxSteps` and world-state, test-generation quality, and coherence recommendation-text gaps](https://github.com/proffesor-for-testing/agentic-qe/issues/535) +(the 3.14.4 recheck did not exercise `goap_execute`), [RVF recovery loop](https://github.com/proffesor-for-testing/agentic-qe/issues/574), and [local-embedding audit findings](https://github.com/proffesor-for-testing/agentic-qe/issues/615). +The newer 3.14.5 registry version does not establish that the repeated-init +settings rewrite reported in [AQE #778](https://github.com/proffesor-for-testing/agentic-qe/issues/778) +has been fixed; the disposable reproduction used 3.14.4. The Codex QE-Court investigation in [agentic-kit #108](https://github.com/pacphi/agentic-kit/issues/108) is a diff --git a/docs/maintenance.md b/docs/maintenance.md index 441b072c..f77d26d7 100644 --- a/docs/maintenance.md +++ b/docs/maintenance.md @@ -772,13 +772,14 @@ See the [dated top-50 coverage list](https://github.com/pacphi/agentic-kit/blob/ Activity presents scan history as a table grouped by the browser’s local calendar date, newest first. Each date has a chevron toggle to expand or collapse its source rows; groups -start collapsed and retain their state while the dashboard stays open. Source rows show completion time and timezone, source, status, and entry count. +start collapsed and retain their state while the dashboard stays open. Source rows show a completion time when available, plus source, status, and entry count. Discovery focuses on source configuration and current coverage; historical scans appear only in Activity. Groups represent dates, not inferred shared scan runs. Missing dates remain explicitly unknown. Version measurements, update checks, and snooze deadlines also use local date/time formatting. -Scan history retains the latest 10 completed records per source and environment, within the -90-day retention limit. Each new record replaces the oldest retained record for that source. +Scan history retains the latest 10 records per source and environment, within the +90-day retention limit. A pause records a continuation boundary with a recorded time, +not a scan completion time. Each new record replaces the oldest retained record for that source. Activity places recovery, work in progress, receipts, dispositions, and recipe changes in a responsive card grid above the full-width scan history. Narrow screens use a single column. diff --git a/docs/plans/2026-09-28-follow-ups-v2.md b/docs/plans/2026-09-28-follow-ups-v2.md deleted file mode 100644 index 7c1bc897..00000000 --- a/docs/plans/2026-09-28-follow-ups-v2.md +++ /dev/null @@ -1,35 +0,0 @@ -# Follow ups v2: V4 branch plan - -## Status - -**Active.** Branch `fix/follow-ups-v2`; exact base `e2f9dcae0554ff63921df618a819fd5e6afe80d2` (develop bootstrap #272). B1 is complete in `fb54f02b`; other rows remain unimplemented. The controller reviews and assigns later rows. One test-first unit commit per row. - -The [remediation program V4](2026-09-28-remediation-program-v2.md#v4-fixfollow-ups-v2-every-small-product-cli-and-upstream-item) defines scope. The [archived Branch 9 plan](../archive/2026-09-28-superpowers-plan-branch-9-follow-ups.md) supplies task details. Paths below name current source seams and focused test targets. After an explicit directory prefix, subsequent bare filenames in the same cell use that directory. A new test named below is a proposed file. Later implementers must verify dependencies before editing. - -| Row | Source or artifact mapping | Focused proof and prerequisite | -| --- | --- | --- | -| A1 | `bin/agentic-kit.mjs`; `src/commands/usage.mjs`, `models.mjs`, `audit.mjs`, `heal.mjs`, `telemetry.mjs`, `x/host.mjs` | `tests/kit/cli-json-honesty.test.mjs`, `usage-cli.test.mjs`, `models-command.test.mjs`, `telemetry-cli.test.mjs`, `status-command.test.mjs`; include unknown models verb and status positional | -| A2 | `src/commands/x/host.mjs`; `bin/agentic-kit.mjs` | `tests/kit/host-dry-run.test.mjs`, `host-cli-migration.test.mjs`; pick refusal, off, reset-routes under `--dry-run --json` | -| A3 | `src/lib/versions.mjs`; `docs/adr/0063-evidence-store-and-refresh-vocabulary.md` | `tests/kit/version-lookup-record.test.mjs`, `drift-freshness.test.mjs`; offline tried-at TTL and ADR wording | -| A4 | `src/commands/status.mjs`; `src/lib/refresh.mjs` | `tests/kit/refresh.test.mjs`, `status-version-drift-refresh.test.mjs`; injected `refreshStages` plus `service` builds no collector | -| B1 | `src/lib/paths.mjs`; `src/lib/footprint/index.mjs`, `storage.mjs`, `consumers.mjs`, `storage-reclaim-detectors.mjs`, `install.mjs`; `src/lib/host-readiness-local.mjs`, `live/process-sessions.mjs`, `hook-audit/providers/opencode.mjs`, `usage-opencode.mjs`; `src/commands/uninstall.mjs` | `tests/kit/xdg-relative.test.mjs` and specified regressions; exact-head CI gate passed before edit; preserve nullable OpenCode fallback | -| B2 | `src/commands/x/daemon-gc.mjs`, `src/commands/x/host.mjs`, `src/commands/setup.mjs` | New `tests/kit/daemon-gc-rerecord.test.mjs`, `setup-host-rerecord.test.mjs`, `host-pick-rerecord.test.mjs`; Branch 9 Task 8 plus deferred host pick; compare `sync-host-repair.test.mjs` | -| B3 | `src/lib/ruflo-memory.mjs`, `paths.mjs` | `tests/kit/ruflo-memory-location.test.mjs`, `project-memory-status.test.mjs`; compose both unsuitable reasons and make `inside()` exclude equality | -| B4 | `src/commands/status/sections/project-memory.mjs`; `src/lib/ruflo-memory-contract.mjs`, `live-check-evidence.mjs` | D-4 **B approved**: `tests/kit/project-memory-status.test.mjs`, `live-check-evidence.test.mjs`, `verify-memory-routes.test.mjs`; info only after successful installed-version `memory-routes` evidence, warn on upgrade or failure | -| B5 | `src/lib/project-memory.mjs`, `aqe-readiness.mjs`; `src/commands/status/sections/project-memory.mjs`, `aqe.mjs` | `tests/kit/project-memory.test.mjs`, `aqe-readiness.test.mjs`, `project-memory-status.test.mjs`; dot-folder scan, real `~/.agentic-qe`, and agentic-qe#757 hint | -| B6 | `src/commands/status/sections/ruflo-components.mjs`; `src/lib/ruflo-components/states.mjs` | `tests/kit/ruflo-components-status.test.mjs`; applied-but-unverified row; hooks fix line requires #3419 answer first | -| B7 | `src/lib/ruflo-daemon-config.mjs`; `src/commands/sync.mjs`, `sync/plan-versions.mjs` | `tests/kit/sync-daemon-repair.test.mjs`, `sync-dry-run-preview.test.mjs`, `sync-skip-versions.test.mjs`; F6 hidden YAML keys and F7 versions-only preview parity | -| B8 | `src/lib/maintenance/discovery/orchestrator.mjs`, `history.mjs` | `tests/kit/maintenance-discovery-orchestrator.test.mjs`, `maintenance-recovery.test.mjs`; restart after pause shows paused history | -| B9 | `src/lib/exec.mjs`, `execution/process-tree.mjs` | `tests/kit/process-tree.test.mjs`; abort kills descendants; Windows CI required | -| B10 | `src/lib/maintenance/discovery/partitions.mjs`; inventory `src/lib/live/jsonl-tailer.mjs`, `live/transcript-streams.mjs`, `telemetry/store.mjs`, `maintenance/management/service-store.mjs` for additional persisted IDs | `tests/kit/file-identity-bigint.test.mjs`; distinguish IDs above `2^53`; enumerate the exact sites before edit | -| B11 | `src/lib/live-checks.mjs` | `tests/kit/live-checks.test.mjs`; skipped deja-vu check says skipped and check-created temp folders are cleaned | -| B12 | `src/commands/setup.mjs`; `src/lib/memory-probe-cleanup.mjs` | `tests/kit/setup-memory-probe.test.mjs`; disposable real Ruflo reproduction first, fix only if unused `agentdb-memory.db` appears | -| B13 | `src/commands/sync.mjs`; `src/lib/aqe-project-pin.mjs` | `tests/kit/sync-command.test.mjs`, `aqe-project-pin.test.mjs`; only if program §2 step 2 shows sync omitted the AQE pin | -| C1 | `.github/workflows/ci.yml`; `src/lib/aqe-readiness.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | `tests/live/ruflo-memory-routing.test.mjs`, `tests/kit/aqe-readiness.test.mjs`; disposable macOS and temporary Linux/Windows CI busy-rule evidence; remove temporary job before merge; #240 action follows result | -| C2 | `docs/host-support.md`; `src/lib/hook-audit/agentic-dependency-constraints.json` | `tests/kit/ruflo-support-window.test.mjs` plus link check; verify AQE 3.14.4 #528/#532/#535 and Ruflo #2356/#420 first | -| C3 | `.github/workflows/nightly.yml`; vidaunited's `trace-ort.mjs` hook (obtain and verify its exact script path before adding) | `tests/kit/upstream-watch-workflow.test.mjs` plus macOS artifact receipt; exact upstream #2885 post text requires user approval | -| C4 | `scripts/upstream-watch/classify.mjs`, `fetch.mjs`, `ledger.mjs`, `dispatch.mjs`, `render.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | `tests/kit/upstream-watch-script.test.mjs`, `upstream-watch-record.test.mjs`, `upstream-watch-dispatch.test.mjs`, `upstream-watch-registry.test.mjs`; use ignored `reports/n5-253-deferred-minors.md` §2 for M7/M8/minors 1–12; M10 declined | -| C5 | `src/lib/aqe-guidance.mjs`; `src/commands/setup.mjs`; ignored `.superpowers/sdd/2026-09-28-follow-ups-v2/c5-issue-draft.md` | D-6 **A approved**: controller's isolated AQE init reproduction is evidence handoff; draft issue with command/version/expected/actual, then obtain approval of exact posting text. B0-16 draft only if D-16 B | -| C6 | `docs/host-support.md`; `src/lib/ruflo-support-window.mjs`, `aqe-readiness.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | `tests/kit/ruflo-support-window.test.mjs`, `aqe-readiness.test.mjs`; pre-PR live checks against newest supported Ruflo `security secrets --path`, Codex read-only app-server flags, and AQE 3.14.x | - -B1 used disposable homes, guarded focused tests, and the ignored B1 report at `.superpowers/sdd/2026-09-28-follow-ups-v2/b1-report.md`. No shared manifests, lockfiles, ADR index, or decision log change belongs to this plan update. The controller owns integration and the whole-branch gate. diff --git a/docs/plans/2026-09-28-remediation-program-v2.md b/docs/plans/2026-09-28-remediation-program-v2.md index 3da32f49..4a6b526d 100644 --- a/docs/plans/2026-09-28-remediation-program-v2.md +++ b/docs/plans/2026-09-28-remediation-program-v2.md @@ -10,6 +10,11 @@ Windows timing evidence. #251 is done, and the alpha.60 release commit is on `ma Publication and global installation were not verified in the execution-plan baseline. The historical decision batch and schedule below remain as scope/evidence references; the approved execution section governs wherever their authority or timing differs. +At the 2026-09-29 checkpoint, V3 #276, V5 #274, V4 B1 #273, V4 C3 #277, and +V4 C4 #278 had merged to `develop@989c5e56`. Remaining V4 implementation +units were accepted on an isolated branch; its final gates, whole-branch +review, PR CI and integration remained open. V6 and V7 still have work. The +final `develop` → `main` PR has not been opened. ### Approved execution and precedence @@ -308,7 +313,7 @@ One unit commit per line, test-first. The source text is the Branch 9 plan where 1. A relative `XDG_*` value is ignored (Branch 9 Task 6; B6a-6, B2-6). 2. Re-record seams for `x/daemon-gc.mjs` and `setup.mjs` (Task 8), plus the one for `x/host.mjs` `pick` (deferred 13a) (B6a-3). -3. `rufloMemoryLocation` names both reasons when the root and the folder are both unsuitable (deferred item 11). `inside()` stops treating equality as "inside", so a `TMPDIR` set to a tool folder is read correctly (B9-12, B0-15). +3. `rufloMemoryLocation` names both reasons when the root and the folder are both unsuitable (deferred item 11). `inside()` stops treating equality as "inside", so a `TMPDIR` set to a tool folder is read correctly (B9-12, B0-15). Implemented in V4 B3; isolated-branch review pending (`.superpowers/sdd/2026-09-28-follow-ups-v2/b3-report.md`). 4. N4, as D-4 decides (B9-3). 5. The stray scan walks dot folders below the root (B5-9, D-7). The real `~/.agentic-qe` home store is listed (B5-11). The `codex-mcp` hint stops suggesting AQE's broken Codex platform setup (agentic-qe#757) (B5-10). 6. `ruflo-components`: the applied-but-unverified row stops repeating the restart instruction (B0-21). The rows reading "partial — missing: Codex hooks" get a fix line once ruvnet/ruflo#3419 answers; if it is still unanswered at the pre-PR check, this part waits (LQ-2). diff --git a/docs/plans/2026-09-28-remediation-v2-develop-execution.md b/docs/plans/2026-09-28-remediation-v2-develop-execution.md index 567dec18..bed7dd78 100644 --- a/docs/plans/2026-09-28-remediation-v2-develop-execution.md +++ b/docs/plans/2026-09-28-remediation-v2-develop-execution.md @@ -12,6 +12,16 @@ unit commits, feature PRs into `develop`, and conditional squash integration; th `develop` → `main` PR remains open for human review. Releases, installation, real-data operations and cleanup remain separately gated. Baseline inspected: `main@94890a00`. +**2026-09-29 execution update:** Bootstrap, V3 #276, V5 #274, V4 B1 #273, +V4 C3 #277, and V4 C4 #278 have merged to `develop@989c5e56`; their relevant +CI receipts passed. Remaining V4 implementation units are accepted on the +isolated branch, and its temporary C1 CI job has been removed. V4's ADR/C6 +alignment is underway. The final V4 full gates, whole-branch review, feature +PR CI including Windows, and integration are still pending. V6 and V7 remain +separate work. The final `develop` → `main` PR and operational gates have not +occurred. Historical baseline rows below describe planning-time state, not +current completion. + **Goal:** Complete all remaining v2 remediation through feature PRs into `develop`, then open one aggregate `develop` → `main` PR for human review. @@ -68,9 +78,10 @@ Appendix A/B row and its referenced v1 plans, rulings and evidence. The original program's "not yet started" status and main-only flow are stale. Reconcile them in the bootstrap PR, retaining historical evidence rather than replaying completed work. -ADR-0063 is **Accepted**, updated 2026-09-28, with CLI delivery recorded; V3 completes its -dashboard changes and V4 its offline retry limitation. ADR-0048 is **Accepted**, updated -2026-09-28, with human evaluation gates outstanding; V3 records their approved v5 deferral. +ADR-0063 is **Accepted**, updated 2026-09-29, with CLI and V3 dashboard delivery +recorded; V4 A3 refines its offline retry limitation on the feature branch. +ADR-0048 is **Accepted**, updated 2026-09-28, with human evaluation gates +outstanding; V3 records their approved v5 deferral. ADR-0060 is **Proposed**, updated 2026-09-27, with discovery partly implemented; V6 implements its approved remaining scope and records acceptance and the actual delivered subset. diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index 576d9f40..58d06d53 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -151,15 +151,18 @@ up to six minutes each: ak status --refresh=live --only learning # trains a cycle in an isolated dir; asserts patterns persist to disk ak status --refresh=live --only aqe # agentic-qe genuinely on ruvector (no FsyncFailed) ak status --refresh=live --only harvest # Ruflo's learning-write path (post-task + distill) in an isolated store -ak status --refresh=live --only memory-routes # CLI/MCP routing observation, remembered as the memory check +ak status --refresh=live --only memory-routes # CLI round trip remembered as memory; routing observation remembered separately ak status --refresh=live --only learning,harvest,aqe,memory-routes,security,deja-vu,providers,mcp,aqe-embedding ``` If `ak status --refresh=live --only aqe` warns that RVF is held by another live process, another AQE process (usually the AQE MCP server in an open Claude Code session) owns the store. -That is contention, not a storage failure, even though agentic-qe 3.14.3 also prints -`FsyncFailed` in this case ([#240](https://github.com/pacphi/agentic-kit/issues/240)). -A `FsyncFailed` without the live-owner lines still fails verification. +Released agentic-qe 3.14.4 passed native macOS and Linux live-owner probes with `LockHeld` +and no `FsyncFailed` ([#240](https://github.com/pacphi/agentic-kit/issues/240)). +The old 3.14.3 sequence also printed `FsyncFailed`; ak now treats any `FsyncFailed` or +`0x0303` as an RVF failure, even alongside live-owner lines. An ordinary live lock stays +busy with SQLite fallback observed; owner health and RVF integrity remain unverified. +Native Windows AQE live-lock conformance has not been run. ## Known upstream gaps (not fixable by sync) diff --git a/scripts/run-roots.mjs b/scripts/run-roots.mjs index a077cec5..ff4621c2 100644 --- a/scripts/run-roots.mjs +++ b/scripts/run-roots.mjs @@ -46,20 +46,20 @@ export function readOwner(root) { let record = null; try { const file = path.join(root, OWNER_FILE); - const before = fs.lstatSync(file); - if (!before.isFile() || before.isSymbolicLink() || before.nlink !== 1 - || before.size > MAX_OWNER_BYTES || (currentUid() !== null && before.uid !== currentUid())) { + const before = fs.lstatSync(file, { bigint: true }); + if (!before.isFile() || before.isSymbolicLink() || before.nlink !== 1n + || before.size > BigInt(MAX_OWNER_BYTES) || (currentUid() !== null && before.uid !== BigInt(currentUid()))) { throw Error('unsafe owner file'); } // Bitwise flags treat an unavailable platform constant as zero. fd = fs.openSync(file, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | fs.constants.O_NONBLOCK); - const opened = fs.fstatSync(fd); - if (!opened.isFile() || !sameIdentity(before, opened) || opened.size > MAX_OWNER_BYTES) { + const opened = fs.fstatSync(fd, { bigint: true }); + if (!opened.isFile() || !sameIdentity(before, opened) || opened.size > BigInt(MAX_OWNER_BYTES)) { throw Error('owner file changed at open'); } const bytes = Buffer.alloc(MAX_OWNER_BYTES + 1); const count = fs.readSync(fd, bytes, 0, bytes.length, 0); - if (count > MAX_OWNER_BYTES || !sameIdentity(opened, fs.lstatSync(file))) throw Error('owner file changed at read'); + if (count > MAX_OWNER_BYTES || !sameIdentity(opened, fs.lstatSync(file, { bigint: true }))) throw Error('owner file changed at read'); const parsed = JSON.parse(bytes.subarray(0, count).toString('utf8')); if (validRecord(parsed, root)) record = parsed; } catch { /* Unknown metadata never grants ownership. */ } @@ -184,7 +184,8 @@ export function defaultProbes(_platform = process.platform) { return { listOnly: true, alive: () => null, startedAfter: () => null, completeExit: () => null }; } -function sameIdentity(a, b) { return a.dev === b.dev && a.ino === b.ino && a.ctimeMs === b.ctimeMs; } +// Keep inode/device IDs and change times exact; Number stats can alias distinct files. +function sameIdentity(a, b) { return a.dev === b.dev && a.ino === b.ino && a.ctimeNs === b.ctimeNs; } /** Revalidate after injected probes; recursive rm can still fail partway through. * The fixture seam is not an installed platform containment implementation. @@ -208,13 +209,13 @@ export function collectAbandonedRoots({ tmpdir, selfRoot, homedir, uid = current const safe = removableRunRoot(root, options); if (!safe.ok) { keep(root, safe.reason); continue; } try { - const identity = fs.lstatSync(root); + const identity = fs.lstatSync(root, { bigint: true }); const owner = readOwner(root); if (!owner) { keep(root, 'owner changed during inspection'); continue; } const proof = proveAbandoned(root, owner, probes); if (!proof.abandoned) { keep(root, proof.reason); continue; } const boundary = removableRunRoot(root, options); - if (!boundary.ok || !sameIdentity(identity, fs.lstatSync(root)) + if (!boundary.ok || !sameIdentity(identity, fs.lstatSync(root, { bigint: true })) || JSON.stringify(owner) !== JSON.stringify(readOwner(root))) { keep(root, 'root or owner changed before removal'); continue; } diff --git a/scripts/run-tests.mjs b/scripts/run-tests.mjs index 166a5324..d0587a8d 100644 --- a/scripts/run-tests.mjs +++ b/scripts/run-tests.mjs @@ -63,13 +63,13 @@ export function runGuarded(commands, { const owner = ownerRecord(); try { writeOwner(tempRoot, owner); } catch (error) { log(`could not record run owner; kept run root ${tempRoot}: ${error.message}`); return 2; } - const identity = fs.lstatSync(tempRoot); + const identity = fs.lstatSync(tempRoot, { bigint: true }); const removeOwnRoot = () => { const safe = removableRunRoot(tempRoot, { tmpdir, homedir, requireOwner: false }); if (!safe.ok) { log(`kept own run root ${tempRoot}: ${safe.reason}`); return false; } try { - const current = fs.lstatSync(tempRoot); - if (current.dev !== identity.dev || current.ino !== identity.ino || current.birthtimeMs !== identity.birthtimeMs) { + const current = fs.lstatSync(tempRoot, { bigint: true }); + if (current.dev !== identity.dev || current.ino !== identity.ino || current.birthtimeNs !== identity.birthtimeNs) { log(`kept own run root ${tempRoot}: directory identity changed`); return false; } fs.rmSync(tempRoot, { recursive: true, force: true, maxRetries: 3 }); diff --git a/src/commands/audit.mjs b/src/commands/audit.mjs index 90a1141f..69535fcb 100644 --- a/src/commands/audit.mjs +++ b/src/commands/audit.mjs @@ -9,6 +9,7 @@ import { collectContextEvidence } from '../lib/context-audit-sources.mjs'; import { loadKitConfig } from '../lib/config.mjs'; import { projectCensus, projectsInScope } from '../lib/project-census.mjs'; import { installedVersion } from '../lib/versions.mjs'; +import { reportFailure } from '../lib/output.mjs'; export const options = { json: { type: 'boolean', default: false }, @@ -152,8 +153,11 @@ export async function run({ contextCollectorFn = collectContextAudit, }) { if (positionals.length !== 1 || !['hooks', 'context'].includes(positionals[0])) { - console.error('ak audit requires the hooks or context subcommand'); - console.log(help); + const error = 'ak audit requires the hooks or context subcommand'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => { + console.error(error); + console.log(help); + } }); return 2; } if (positionals[0] === 'context') { @@ -161,7 +165,8 @@ export async function run({ try { report = await contextCollectorFn({ flags, pkgRoot, loadConfigFn }); } catch (error) { - console.error(`context audit failed: ${error.message}`); + const message = `context audit failed: ${error.message}`; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => console.error(message) }); return 2; } if (flags.json) console.log(JSON.stringify(report, null, 2)); @@ -172,7 +177,8 @@ export async function run({ try { report = collectHookAudit({ flags, detectVersionFn, loadConfigFn }); } catch (error) { - console.error(`hook audit failed: ${error.message}`); + const message = `hook audit failed: ${error.message}`; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => console.error(message) }); return 2; } if (flags.json) { diff --git a/src/commands/heal.mjs b/src/commands/heal.mjs index cd68085e..8b30d965 100644 --- a/src/commands/heal.mjs +++ b/src/commands/heal.mjs @@ -1,4 +1,5 @@ import path from 'node:path'; +import { reportFailure } from '../lib/output.mjs'; import { collectHookAudit } from './audit.mjs'; import { @@ -99,25 +100,28 @@ function validateMode(flags) { if (flags.apply && !flags.yes) throw new TypeError('--apply requires --yes'); } +function usageError(flags, message, showHelp = false) { + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => { + console.error(message); + if (showHelp) console.log(help); + } }); + return 2; +} + export async function run({ flags, positionals, detectVersionFn, loadConfigFn }) { if (positionals.length !== 1 || positionals[0] !== 'hooks') { - console.error('ak heal requires the hooks subcommand'); - console.log(help); - return 2; + return usageError(flags, 'ak heal requires the hooks subcommand', true); } try { validateMode(flags); } catch (error) { - console.error(`hook healing refused: ${error.message}`); - return 2; + return usageError(flags, `hook healing refused: ${error.message}`); } const transactionsRoot = transactionRoot(flags); if (flags.undo && flags.recover) { - console.error('hook healing refused: --undo and --recover are mutually exclusive'); - return 2; + return usageError(flags, 'hook healing refused: --undo and --recover are mutually exclusive'); } if (flags.undo || flags.recover) { if ((flags.action?.length ?? 0) || flags['plan-digest']) { - console.error('hook healing refused: rollback/recovery cannot be combined with plan action flags'); - return 2; + return usageError(flags, 'hook healing refused: rollback/recovery cannot be combined with plan action flags'); } const result = flags.recover ? (flags.apply @@ -145,8 +149,7 @@ export async function run({ flags, positionals, detectVersionFn, loadConfigFn }) return audit.summary.invalidSources || audit.summary.configurationIssues ? 1 : 0; } if (!flags.action?.length || !flags['plan-digest']) { - console.error('hook healing refused: apply requires --action and --plan-digest from a preview'); - return 2; + return usageError(flags, 'hook healing refused: apply requires --action and --plan-digest from a preview'); } if (unfinishedTransactions.length) { throw new Error(`unfinished hook transaction(s) require --recover first: ${unfinishedTransactions.map((item) => item.id).join(', ')}`); @@ -160,7 +163,6 @@ export async function run({ flags, positionals, detectVersionFn, loadConfigFn }) }); return printResult(result, flags.json); } catch (error) { - console.error(`hook healing failed: ${error.message}`); - return 2; + return usageError(flags, `hook healing failed: ${error.message}`); } } diff --git a/src/commands/models.mjs b/src/commands/models.mjs index 89026435..b9562567 100644 --- a/src/commands/models.mjs +++ b/src/commands/models.mjs @@ -1,4 +1,4 @@ -import { heading, info, ok, warn, dim } from '../lib/output.mjs'; +import { heading, info, ok, warn, dim, reportFailure } from '../lib/output.mjs'; import { loadKitConfig } from '../lib/config.mjs'; import { aqeRouterFile } from '../lib/providers.mjs'; import { readJson } from '../lib/settings.mjs'; @@ -129,10 +129,6 @@ async function runRefresh(ctx) { function runStatus(ctx) { const { flags, cacheFile, store, latest } = ctx; - if (flags.host && !ALL_OWNERS.includes(flags.host)) { - warn(`unsupported model host: ${flags.host}`); - return 2; - } const snapshot = visibleSnapshot(latest, flags.host); const since = flags.since ? Date.parse(flags.since) : null; const history = store.snapshots.filter((entry) => entry.scope.fingerprint === latest.scope.fingerprint @@ -174,7 +170,7 @@ function runDiff(ctx) { function runExplain(ctx) { const { positionals, flags, latest } = ctx; const selector = positionals[1] ?? flags.to; - if (!selector) { warn('usage: ak models explain HOST:MODEL'); return 2; } + if (!selector) { modelUsageError(flags, 'usage: ak models explain HOST:MODEL'); return 2; } const result = explainModel(latest, selector); if (flags.json) printJson(result); else if (!result.found) warn(`Model not found: ${selector}`); @@ -193,7 +189,7 @@ function runPlan(ctx) { const { flags, positionals, latest } = ctx; const activity = flags.activity; const to = flags.to ?? positionals[1]; - if (!activity || !to) { warn('usage: ak models plan --activity ACTIVITY [--from HOST:MODEL] --to HOST:MODEL'); return 2; } + if (!activity || !to) { modelUsageError(flags, 'usage: ak models plan --activity ACTIVITY [--from HOST:MODEL] --to HOST:MODEL'); return 2; } const result = planModelChange(latest, { activity, from: flags.from, to }); if (flags.json) printJson(result); else { @@ -211,9 +207,34 @@ function runPlan(ctx) { // dispatched by name once that store/latest snapshot is in hand (below). const READ_ACTIONS = { status: runStatus, diff: runDiff, explain: runExplain, plan: runPlan }; +function modelUsageError(flags, message) { + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => warn(message) }); +} + /** @param {{flags: Record, positionals: string[], deps?: Record}} input */ export async function run({ flags, positionals, deps = {} }) { const action = positionals[0] ?? 'status'; + if (action !== 'refresh' && !Object.hasOwn(READ_ACTIONS, action)) { + modelUsageError(flags, 'usage: ak models status|refresh|diff|explain|plan'); + return 2; + } + const maxPositionals = action === 'diff' ? 3 : action === 'explain' || action === 'plan' ? 2 : 1; + if (positionals.length > maxPositionals) { + modelUsageError(flags, `unexpected argument '${positionals[maxPositionals]}'`); + return 2; + } + if (action === 'explain' && !positionals[1] && !flags.to) { + modelUsageError(flags, 'usage: ak models explain HOST:MODEL'); + return 2; + } + if (action === 'plan' && (!flags.activity || !(flags.to ?? positionals[1]))) { + modelUsageError(flags, 'usage: ak models plan --activity ACTIVITY [--from HOST:MODEL] --to HOST:MODEL'); + return 2; + } + if (action === 'status' && flags.host && !ALL_OWNERS.includes(flags.host)) { + modelUsageError(flags, `unsupported model host: ${flags.host}`); + return 2; + } const cacheFile = deps.cacheFile ?? modelInventoryPath(); const readStore = deps.readStore ?? readModelStore; const append = deps.append ?? appendModelSnapshot; @@ -229,6 +250,5 @@ export async function run({ flags, positionals, deps = {} }) { if (!latest) return noSnapshot(flags, cacheFile); const handler = READ_ACTIONS[action]; - if (!handler) { warn('usage: ak models status|refresh|diff|explain|plan'); return 2; } return handler({ ...ctx, store, latest }); } diff --git a/src/commands/setup.mjs b/src/commands/setup.mjs index 772294d5..fc649db6 100644 --- a/src/commands/setup.mjs +++ b/src/commands/setup.mjs @@ -5,6 +5,7 @@ // Project scope (when run inside a git repo / --project): the port of // ruflo-setup-project — init, sanitize, pin, activate, verify, daemon. import fs from 'node:fs'; +import os from 'node:os'; import path from 'node:path'; import readline from 'node:readline/promises'; import { run as runCmd, have } from '../lib/exec.mjs'; @@ -381,21 +382,24 @@ function deployTokenAuditSkill(pkgRoot) { * left alone. Shares HOSTS/hostInstallState/installHost with `ak sync`'s * and `ak host pick`'s own host-install loops; the interactive confirmation * here (vs. their unconditional install) is this command's own UX. */ -async function installEnabledAbsentHosts(cfg, flags) { +export async function installEnabledAbsentHosts(cfg, flags, lifecycle = {}) { + const { installState, install, collectFacts } = { + installState: hostInstallState, install: installHost, collectFacts: collectIntegrationFacts, ...lifecycle, + }; let installed = false; for (const h of HOSTS) { if (!cfg.integrations?.hosts?.[h.id]) continue; - const st = await hostInstallState(h); + const st = await installState(h); if (st.method === 'absent') { if (await ask(`${h.id} CLI not found — install ${h.pkg} globally?`, true, flags.yes)) { - const r = await installHost(h.id); + const r = await install(h.id); (r.ok ? ok : warn)(`${h.id}: ${r.detail}`); if (r.ok) { installed = true; // hostInstallState() above already recorded the pre-install // 'absent' evidence; re-probe now so a subsequent `ak status` // doesn't read that stale row back. - await hostInstallState(h, { refresh: true, record: true, source: 'setup' }); + await installState(h, { refresh: true, record: true, source: 'setup' }); } } else warn(`${h.id} not installed — enable/install later with: ak host pick`); } else { @@ -404,7 +408,7 @@ async function installEnabledAbsentHosts(cfg, flags) { } // host-setup covers every host in one call; refresh it once after the // loop, not per host, once anything actually changed. - if (installed) await collectIntegrationFacts({ cfg, refresh: true, record: true, source: 'setup' }); + if (installed) await collectFacts({ cfg, refresh: true, record: true, source: 'setup' }); } /** Step 6b: host lifecycle wiring — connected MCPs, compact lazy gateway, @@ -467,7 +471,7 @@ async function printUndetectedHostHints(cfg) { } } -export async function run_machine({ flags, pkgRoot, cfg }) { +export async function run_machine({ flags, pkgRoot, cfg, deps = { hostLifecycle: undefined } }) { heading('machine setup'); if (flags['dry-run']) { info('dry-run: would ensure packages (incl. agent-browser and ruvnet-brain), deploy skill (blocks + MCP land in the final pass)'); return true; } @@ -479,7 +483,7 @@ export async function run_machine({ flags, pkgRoot, cfg }) { // key on `codex` being on PATH / dual-mode enablement). Running them here // warned + drifted on genuinely bare machines. deployTokenAuditSkill(pkgRoot); - await installEnabledAbsentHosts(cfg, flags); + await installEnabledAbsentHosts(cfg, flags, deps.hostLifecycle); if (!(await applyMachineHostLifecycles(cfg, pkgRoot))) return false; if (cfg.codexContext && cfg.integrations?.hosts?.codex) { try { await manageCodexContext(cfg, { persist: saveKitConfig }); } @@ -602,25 +606,52 @@ export async function startProjectDaemon(root, { } else warn('daemon failed to start — try: ruflo daemon start'); } -/** Step 7: write-verification (store → actual on-disk row, then clean up). - * The CLI mirrors the write into agentdb-memory.db under the memory root - * (CLAUDE_FLOW_MEMORY_PATH, else a config persistPath, else /.swarm). - * The probe pins that root beside the pinned memory.db, so both copies land - * in the stores cleanup checks even when the project's root is redirected; - * this is the user's real corpus, so nothing of the probe may be left behind. */ +/** Step 7: verify a real primary write. Ruflo also writes a native mirror + * under CLAUDE_FLOW_MEMORY_PATH; keep that disposable mirror in a private + * directory so setup cannot create a spare project store. */ export async function verifyProjectMemoryWrite(root, env, { runner = runCmd } = {}) { const probeKey = `_setup/verify-${process.pid}-${Date.now()}`; - const probeEnv = { ...env, CLAUDE_FLOW_MEMORY_PATH: path.dirname(env?.CLAUDE_FLOW_DB_PATH ?? paths.projectMemoryDb(root)) }; - const stored = (await runner('ruflo', ['memory', 'store', '-k', probeKey, '--value', 'setup-verify', '-n', '_setup'], { cwd: root, env: probeEnv })).code === 0; - const landed = stored ? findMemoryEntry(root, '_setup', probeKey) : null; - if (!landed) { + let mirrorDir; + let stored = false; + let landed = null; + let cleanup; + let runnerError = false; + let mirrorCleanupError = false; + try { + mirrorDir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-setup-memory-probe-')); + const probeEnv = { + ...env, + CLAUDE_FLOW_DB_PATH: env?.CLAUDE_FLOW_DB_PATH ?? paths.projectMemoryDb(root), + CLAUDE_FLOW_MEMORY_PATH: mirrorDir, + RUFLO_DAEMON_AUTOSTART: '0', + }; + stored = (await runner('ruflo', ['memory', 'store', '-k', probeKey, '--value', 'setup-verify', '-n', '_setup'], { cwd: root, env: probeEnv })).code === 0; + } catch { + runnerError = true; + } finally { + // A failed command may still have written its primary row. Keep the + // existing refusal behavior for unreadable or busy project stores. + if (mirrorDir) { + try { + landed = findMemoryEntry(root, '_setup', probeKey); + cleanup = removeMemoryProbe(root, '_setup', probeKey); + } catch { + runnerError = true; + } finally { + try { fs.rmSync(mirrorDir, { recursive: true, maxRetries: 3 }); } + catch { mirrorCleanupError = true; } + } + } + } + if (cleanup?.failed.length) { + warn(`memory probe cleanup failed in ${cleanup.failed.map((f) => `${path.basename(f.file)} (${f.kind})`).join(', ')} — remove ${probeKey} from _setup manually`); + } + if (mirrorCleanupError) warn(`memory probe temporary mirror cleanup failed at ${mirrorDir} — inspect it manually`); + if (!stored || !landed || runnerError || cleanup?.failed.length || mirrorCleanupError) { fail('memory write verification FAILED — run: ak status / ruflo doctor -c memory'); return; } - const cleanup = removeMemoryProbe(root, '_setup', probeKey); - if (cleanup.failed.length) { - warn(`memory write verified, but probe cleanup failed in ${cleanup.failed.map((f) => `${path.basename(f.file)} (${f.kind})`).join(', ')} — remove ${probeKey} from _setup manually`); - } else ok(`memory write VERIFIED (store → ${path.basename(landed.file)} row confirmed)`); + ok(`memory write VERIFIED (store → ${path.basename(landed.file)} row confirmed)`); } function reportProjectGuidance(result) { @@ -926,6 +957,7 @@ async function applySetupCodexRepairs(flags, repairPlan, cwd, repairTopology) { export async function run({ flags, pkgRoot, confirm = ask, dejaVuLifecycle = DEFAULT_DEJA_VU_LIFECYCLE, + deps = { hostLifecycle: undefined }, ...runtimeOverrides }) { const runtime = { ...DEFAULT_SETUP_RUNTIME, ...runtimeOverrides }; @@ -966,7 +998,7 @@ export async function run({ flags, cfg, hostFlags, companionPreflight, dejaVuFlagsResult, }); - if (!(await runtime.machineSetup({ flags, pkgRoot, cfg }))) return 1; + if (!(await runtime.machineSetup({ flags, pkgRoot, cfg, deps }))) return 1; if (!flags['dry-run'] && cfg.aqe !== false) { // Persist the selected choice even on failure, so retry/sync has an exact plan. saveKitConfig(cfg); diff --git a/src/commands/status.mjs b/src/commands/status.mjs index 0020ec5f..f7147d31 100644 --- a/src/commands/status.mjs +++ b/src/commands/status.mjs @@ -89,7 +89,7 @@ Slow proofs run only when named with --only, up to six minutes each: aqe storage, embedding configuration and provenance, and the browser payload memory-routes the memory round trip, plus whether CLI and MCP see each - other's writes (remembered as the memory check) + other's writes (CLI result remembered as memory; routing separately) A named check runs even when it would not apply; its result is remembered only when it applies. learning and harvest are never remembered. @@ -281,9 +281,9 @@ function strayArgumentError(positionals) { export async function run({ flags, positionals = [], pkgRoot, deps = {} }) { const request = refreshRequestFromFlags(flags); if ('error' in request) return usageError(flags, request.error); - if (!request.strength) return report(flags, { rows: await collect({ pkgRoot, refresh: false }) }); const stray = strayArgumentError(positionals); if (stray) return usageError(flags, stray); + if (!request.strength) return report(flags, { rows: await collect({ pkgRoot, refresh: false }) }); return runRefreshed({ flags, pkgRoot, request, deps }); } diff --git a/src/commands/status/sections/codex-mcp.mjs b/src/commands/status/sections/codex-mcp.mjs index 620db669..2e46c1c1 100644 --- a/src/commands/status/sections/codex-mcp.mjs +++ b/src/commands/status/sections/codex-mcp.mjs @@ -87,7 +87,7 @@ function topologyRows(cwd, cfg) { if (!topology.agenticQeRegistrations.length) { // Agentic-QE owns its Codex registration (ADR-0033); sync never writes it. rows.push(row('codex-mcp', 'warn', 'agentic-qe MCP is not concretely registered in Codex', - 'run: aqe platform setup codex --overwrite --with-ruflo', { repair: 'manual' })); + 'run: aqe init --auto --with-codex --codex-guidance compact in this project, then recheck; AQE 3.14.4 may still omit Codex assets (agentic-qe#755)', { repair: 'manual' })); } else { rows.push(row('codex-mcp', 'ok', 'agentic-qe MCP concretely registered in Codex')); } diff --git a/src/commands/status/sections/daemons.mjs b/src/commands/status/sections/daemons.mjs index 152eb9c5..e156cd14 100644 --- a/src/commands/status/sections/daemons.mjs +++ b/src/commands/status/sections/daemons.mjs @@ -87,6 +87,15 @@ const flatKeys = (entries, pick) => entries.map((e) => `"${e.key}": ${JSON.strin export function heldRow(held) { const file = DAEMON_CONFIG_RELATIVE.split(path.sep).join('/'); const want = flatKeys(held.entries, (e) => e.want); + if (held.reason === 'yaml-shadow') return row('daemons', 'warn', + `${file} is not ak-managed: creating it would hide existing .claude-flow/config.yaml or config.yml daemon values`, + `review the YAML daemon values and set ${want} in the active config yourself, ${RESTART}`, { repair: 'manual' }); + if (held.reason === 'higher-priority-json') return row('daemons', 'warn', + `${file} is not ak-managed: Ruflo reads claude-flow.config.json first`, + `review claude-flow.config.json and set ${want} there yourself, ${RESTART}`, { repair: 'manual' }); + if (held.reason === 'explicit-config') return row('daemons', 'warn', + `${file} is not ak-managed: Ruflo currently reads CLAUDE_FLOW_CONFIG before YAML`, + `review the CLAUDE_FLOW_CONFIG file and set ${want} there yourself, ${RESTART}`, { repair: 'manual' }); return held.invalid ? row('daemons', 'warn', `${file} is not ak-managed: it is unreadable or not a JSON object, so ak leaves it untouched`, `fix ${file} so it is a JSON object holding ${want} (flat keys), ${RESTART}`, { repair: 'manual' }) @@ -96,16 +105,16 @@ export function heldRow(held) { /** The Ruflo repository around `cwd`, kit.json, and the keys its config.json * keeps from ak (read once for the deferral and drift rows). */ -function rufloContext(cwd, { loadConfig, rufloVersion, platform }) { +function rufloContext(cwd, { loadConfig, rufloVersion, platform, env }) { const root = rufloDaemonProjectRoot(cwd); if (!root) return { root: null, cfg: null, held: null }; const cfg = loadConfig(); - return { root, cfg, held: daemonConfigHeld(root, { cfg, rufloVersion, platform }) }; + return { root, cfg, held: daemonConfigHeld(root, { cfg, rufloVersion, platform, env }) }; } -function driftRows({ root, cfg, held }, { rufloVersion, platform }) { +function driftRows({ root, cfg, held }, { rufloVersion, platform, env }) { if (!root) return []; - const parts = daemonDrift(root, { cfg, rufloVersion, platform }); + const parts = daemonDrift(root, { cfg, rufloVersion, platform, env }); return [held && heldRow(held), parts && row('daemons', 'warn', `ak-managed daemon settings differ from what Ruflo ${rufloVersion ?? '(version unknown)'} ` + `needs: ${parts.join('; ')}`, "sync applies ak's Ruflo daemon settings")].filter(Boolean); } @@ -134,10 +143,10 @@ export default { rows.push(row('daemons', 'ok', daemons.length ? `${daemons.length} running (one per active project is expected)` : 'none running')); } - const ruflo = rufloContext(cwd, { loadConfig, rufloVersion, platform }); + const ruflo = rufloContext(cwd, { loadConfig, rufloVersion, platform, env }); const deferral = deferralRow(root, { now, platform, ruflo }); if (deferral) rows.push(deferral); - rows.push(...driftRows(ruflo, { rufloVersion, platform })); + rows.push(...driftRows(ruflo, { rufloVersion, platform, env })); } catch (e) { rows.push(row('daemons', 'warn', `daemon check unavailable: ${e.message}`)); } diff --git a/src/commands/status/sections/project-memory.mjs b/src/commands/status/sections/project-memory.mjs index a76c2f57..e9055539 100644 --- a/src/commands/status/sections/project-memory.mjs +++ b/src/commands/status/sections/project-memory.mjs @@ -26,7 +26,7 @@ import fs from 'node:fs'; import path from 'node:path'; import { projectDaemonAlive } from '../../../lib/daemons.mjs'; -import { formatLiveCheckAge as ago } from '../../../lib/live-check-evidence.mjs'; +import { formatLiveCheckAge as ago, liveCheckInputsKey, readLiveCheck } from '../../../lib/live-check-evidence.mjs'; import { memoryMaintenanceStatus } from '../../../lib/memory-maintenance.mjs'; import { findStrayMemoryStores, projectMemoryStatus } from '../../../lib/project-memory.mjs'; import { findProbeRows } from '../../../lib/memory-probe-cleanup.mjs'; @@ -206,7 +206,19 @@ export default { ? storeMessage(store) : `${path.basename(store.file)} store is unreadable (${store.file}); existing-corpus access unverified`)); } - if (memory.secondary) rows.push(row('memory', 'warn', twoStoreMessage(rufloVersion, platform))); + if (memory.secondary) { + const routing = readLiveCheck('memory-routes', { + inputsKey: liveCheckInputsKey('memory-routes', { routingVersion: rufloVersion, platform }), now, + }); + const observed = typeof rufloVersion === 'string' && rufloVersion.length > 0 && + routing?.status === 'passed' && !routing.invalidated; + const message = observed + ? twoStoreMessage(rufloVersion, platform).replace('MCP routing needs separate verification.', 'Isolated CLI/MCP routing was observed.') + : twoStoreMessage(rufloVersion, platform); + rows.push(row('memory', observed ? 'info' : 'warn', observed + ? `${message}; isolated CLI/MCP routing observed (${ago(routing.ageMs)}) for this installed CLI version and platform; existing-corpus access unverified` + : message)); + } const orphaned = orphanedStoreRow(root, memory); if (orphaned) rows.push(orphaned); rows.push(...maintenanceRows(root, memory, now)); diff --git a/src/commands/status/sections/ruflo-components.mjs b/src/commands/status/sections/ruflo-components.mjs index 9bbdebad..ee5270eb 100644 --- a/src/commands/status/sections/ruflo-components.mjs +++ b/src/commands/status/sections/ruflo-components.mjs @@ -29,7 +29,10 @@ export function rufloComponentRows(snapshot) { const rows = [row('ruflo-components', snapshot.summary.active === snapshot.summary.total ? 'ok' : 'info', `ruflo components: ${snapshot.summary.active} of ${snapshot.summary.total} active (ruflo ${snapshot.rufloVersion ?? 'not installed'})`)]; for (const c of snapshot.components) { - const text = `${c.label} — ${c.state.label}: ${c.state.meaning}${c.state.action ? ` ${c.state.action}` : ''}`; + // The applied-unverified action is carried by its manual fix below; repeating it + // in the message renders the same host restart instruction twice. + const action = c.state.id === 'applied-unverified' ? '' : c.state.action; + const text = `${c.label} — ${c.state.label}: ${c.state.meaning}${action ? ` ${action}` : ''}`; rows.push({ ...(FIXABLE.has(c.state.id) ? row('ruflo-components', LEVEL(c.state.id), text, `sync applies ${c.label} (${c.state.action || 'reconcile'})`) diff --git a/src/commands/status/sections/user-memory.mjs b/src/commands/status/sections/user-memory.mjs index 2759b8e3..d2306d6c 100644 --- a/src/commands/status/sections/user-memory.mjs +++ b/src/commands/status/sections/user-memory.mjs @@ -5,6 +5,8 @@ // that store when it exists, and the stray stores such sessions left before: // `~/.swarm` and `~/.codex/.chatgpt-projects/*/.swarm`. Everything here is // information only: ak never moves, merges or deletes a store. +import fs from 'node:fs'; +import path from 'node:path'; import * as paths from '../../../lib/paths.mjs'; import { findUserStrayStores, memoryDirStatus } from '../../../lib/project-memory.mjs'; import { homeRelative } from '../../../lib/ruflo-memory.mjs'; @@ -30,6 +32,23 @@ function strayRow(found, home, userDir) { + (found.complete ? '' : '; only the first 500 Codex project folders were checked')); } +function homeAqeRow(home) { + const dir = path.join(home, '.agentic-qe'); + let folder; + try { folder = fs.lstatSync(dir); } catch (e) { if (e.code === 'ENOENT') return null; throw e; } + if (!folder.isDirectory()) return row('aqe', 'info', `AQE home path ${dir} is not a directory; contents unverified`); + const db = path.join(dir, 'memory.db'); + let file; + try { file = fs.lstatSync(db); } catch (e) { if (e.code !== 'ENOENT') throw e; } + if (!file?.isFile()) return row('aqe', 'info', `AQE home directory ${dir}: memory.db absent; contents and runtime health unverified`); + let walBytes = 0; + try { + const wal = fs.lstatSync(`${db}-wal`); + if (wal.isFile()) walBytes = wal.size; + } catch (e) { if (e.code !== 'ENOENT') throw e; } + return row('aqe', 'info', `AQE home directory ${dir}: memory.db present (${formatBytes(file.size + walBytes)} with WAL); contents and runtime health unverified`); +} + export default { id: 'user-memory', /** @param {{ home?: string, env?: NodeJS.ProcessEnv, cfg?: any }} [ctx] */ @@ -47,6 +66,8 @@ export default { const found = findUserStrayStores({ home, codexHome: env.CODEX_HOME || undefined }); const stray = strayRow(found, home, dir); if (stray) rows.push(stray); + const aqe = homeAqeRow(home); + if (aqe) rows.push(aqe); } catch (e) { rows.push(row('memory', 'warn', `user-level memory check unavailable: ${e.message}`)); } diff --git a/src/commands/sync.mjs b/src/commands/sync.mjs index 3e6ef56c..48ca68ba 100644 --- a/src/commands/sync.mjs +++ b/src/commands/sync.mjs @@ -22,7 +22,7 @@ import { hostsWithLifecycle, lifecycleAdapterFor, lifecycleExecutionEnabled, det import { companionLifecycleFor } from '../lib/adapters/companion-lifecycle-registry.mjs'; import { renderApplyReport } from '../lib/adapters/lifecycle-render.mjs'; import { listDaemons, staleDaemons, reap } from '../lib/daemons.mjs'; -import { applyRufloDaemon } from '../lib/ruflo-daemon-config.mjs'; +import { applyRufloDaemon, rufloDaemonProjectRoot } from '../lib/ruflo-daemon-config.mjs'; import { cleanupProbeRows } from '../lib/memory-probe-cleanup.mjs'; import { rufloMemoryLocation } from '../lib/ruflo-memory.mjs'; import { installedRoutingVersion } from '../lib/ruflo-memory-contract.mjs'; @@ -484,16 +484,20 @@ export const SYNC_STEPS = [ // that are actually still alive, not the ones just killed. if (reaped.some((r) => r.killed)) await list({ cwd: ctx.cwd, refresh: true, record: true, source: 'sync' }); // Read the version now: the versions step may have just upgraded Ruflo. - const applied = await applyRufloDaemon(ctx.cwd, { - cfg: ctx.cfg, rufloVersion: installedRoutingVersion() ?? installedVersion('ruflo'), + const applied = await (ctx.daemonApply ?? applyRufloDaemon)(ctx.cwd, { + cfg: ctx.cfg, rufloVersion: (ctx.daemonVersion ?? (() => installedRoutingVersion() ?? installedVersion('ruflo')))(), }); if (!applied) return; - saveKitConfig(ctx.cfg); + (ctx.saveConfig ?? saveKitConfig)(ctx.cfg); const { config, autostart } = applied.result; if (applied.result.changed) ok(`ruflo daemon settings: config ${config}, start-on-use ${autostart}`); const { held } = applied.result; if (held) { - warn(`.claude-flow/config.json is not ak-managed here (${held.invalid ? 'unreadable or not a JSON object' : 'a key holds your own value'}); ` + const reason = held.reason === 'yaml-shadow' ? 'creating JSON would hide existing YAML daemon values' + : held.reason === 'higher-priority-json' ? 'Ruflo reads root claude-flow.config.json first' + : held.reason === 'explicit-config' ? 'Ruflo reads CLAUDE_FLOW_CONFIG before YAML' + : held.invalid ? 'unreadable or not a JSON object' : 'a key holds your own value'; + warn(`.claude-flow/config.json is not ak-managed here (${reason}); ` + `left as is, and the daemon is not restarted for ${held.entries.map((e) => e.key).join(', ')}`); } if (applied.restarted) ok('ruflo daemon restarted so it reads its settings'); @@ -1123,6 +1127,18 @@ async function converge({ // (not-applied/drifted/blocked) don't need an upgrade and stay in the plan. .filter((r) => !(flags['no-upgrade'] && r.subsystem === 'ruflo-components' && r.state === 'needs-ruflo')); + // The daemon step also runs for a versions item, even when the collector + // reports no current daemon drift. Its settings are decided after upgrades + // from the installed Ruflo version, so the preview can promise only this + // recheck, not exact keys or a restart. + if (!skip.has('versions') && candidates.some((r) => r.subsystem === 'versions') + && !candidates.some((r) => r.subsystem === 'daemons') + && rufloDaemonProjectRoot(cwd)) { + candidates.push(row('daemons', 'info', + 'package changes may change the daemon settings needed by the installed Ruflo version', + 'recheck daemon settings against the installed Ruflo version; write receipted keys or restart a running daemon only if needed')); + } + const cfg = loadKitConfig(); if (cfg.aqe !== false && cfg.aqeEmbedding && cfg.aqeEmbedding.mode !== 'unmanaged') { candidates.push(row('aqe-embedding', 'info', 'selected semantic backend requires live verification', diff --git a/src/commands/telemetry.mjs b/src/commands/telemetry.mjs index 77d3c503..142eaa66 100644 --- a/src/commands/telemetry.mjs +++ b/src/commands/telemetry.mjs @@ -98,7 +98,9 @@ export async function run({ flags, positionals, pkgRoot, deps = {} }) { return 0; } catch { // Source failures and hostile input must not echo filenames or parser details. - console.error('Telemetry failed: check command options, schema/digest, compatible snapshots, private identity, and readable input/new output files.'); + const error = 'Telemetry failed: check command options, schema/digest, compatible snapshots, private identity, and readable input/new output files.'; + console.error(error); + console.log(JSON.stringify({ error, exitCode: 2 })); return 2; } } diff --git a/src/commands/usage.mjs b/src/commands/usage.mjs index 2d786207..e2c578f7 100644 --- a/src/commands/usage.mjs +++ b/src/commands/usage.mjs @@ -2,7 +2,7 @@ // offline text scorecard (`score`) rendered from the SAME local-transcript // aggregate `ak dashboard`'s Usage tab reads — no cost/token/percentile // arithmetic is redone here; see the score section below for the boundary. -import { heading, info, ok, warn, dim } from '../lib/output.mjs'; +import { heading, info, ok, warn, dim, reportFailure } from '../lib/output.mjs'; import { stripUnsafeChars } from '../lib/text-safety.mjs'; import { readIndex } from '../lib/usage-index.mjs'; import { @@ -321,7 +321,8 @@ function scoreProjection(agg, windowDays) { async function runScore({ flags, deps }) { const windowDays = parseScoreWindow(flags.window ?? '14'); if (windowDays == null) { - warn(`ak usage score: --window must be 7, 14, or 30 (got ${JSON.stringify(flags.window)})`); + const message = `ak usage score: --window must be 7, 14, or 30 (got ${JSON.stringify(flags.window)})`; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => warn(message) }); return 2; } const readAgg = deps.readIndex ?? readIndex; @@ -953,7 +954,8 @@ function printDeepPass(deep) { async function runPrompts({ flags, deps }) { const win = parsePromptWindow(flags.window); if (win == null) { - warn(`ak usage prompts: --window must be 7, 14, 30, or all (got ${JSON.stringify(flags.window)})`); + const message = `ak usage prompts: --window must be 7, 14, 30, or all (got ${JSON.stringify(flags.window)})`; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => warn(message) }); return 2; } const readAgg = deps.readIndex ?? readIndex; @@ -1087,6 +1089,12 @@ export async function run({ flags, positionals, deps = {} }) { const provider = positionals[1]; const cacheFile = deps.cacheFile ?? openRouterActivityFile(); + if (['score', 'prompts'].includes(action) && positionals.length > 1) { + const message = `unexpected argument '${positionals[1]}'`; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => warn(message) }); + return 2; + } + if (action === 'status' && provider === undefined) { return runOpenRouterStatus({ flags, cacheFile, read: deps.read ?? readOpenRouterActivity }); } @@ -1096,6 +1104,7 @@ export async function run({ flags, positionals, deps = {} }) { return runOpenRouterRefresh({ flags, cacheFile, refresh: deps.refresh ?? refreshOpenRouterActivity }); } - warn('usage: ak usage status | ak usage refresh openrouter | ak usage score | ak usage prompts'); + const message = 'usage: ak usage status | ak usage refresh openrouter | ak usage score | ak usage prompts'; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => warn(message) }); return 2; } diff --git a/src/commands/x/aqe-embedding.mjs b/src/commands/x/aqe-embedding.mjs index 25601712..fad9d06e 100644 --- a/src/commands/x/aqe-embedding.mjs +++ b/src/commands/x/aqe-embedding.mjs @@ -3,6 +3,7 @@ import { embeddingIntentFromFlags, embeddingSetupDisclosure } from '../../lib/aq import { prepareAqeEmbedding, AQE_EMBEDDING_COACHING } from '../../lib/aqe-embedding-lifecycle.mjs'; import { inspectAqeEmbeddingProjections, reconcileAqeEmbeddingProjections } from '../../lib/aqe-embedding-projection.mjs'; import { reconcileOpencodeAqeEmbedding } from '../../lib/opencode-core.mjs'; +import { reportFailure } from '../../lib/output.mjs'; export const options = { 'aqe-embedding-mode': { type: 'string' }, 'aqe-embedding-endpoint': { type: 'string' }, @@ -43,11 +44,19 @@ export async function run({ flags = {}, positionals = [], reconcileOpenCode = reconcileOpencodeAqeEmbedding, }) { const action = positionals[0] ?? 'status'; - if (!['status', 'configure', 'prepare', 'verify'].includes(action) || positionals.length > 1) return 2; + if (!['status', 'configure', 'prepare', 'verify'].includes(action) || positionals.length > 1) { + const error = 'usage: ak x aqe-embedding [status|configure|prepare|verify]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => console.error(error) }); + return 2; + } const cfg = load(); if (action === 'configure') { try { cfg.aqeEmbedding = embeddingIntentFromFlags(cfg, flags); } - catch { console.error('Invalid embedding selection; run ak x aqe-embedding --help.'); return 2; } + catch { + const error = 'Invalid embedding selection; run ak x aqe-embedding --help.'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => console.error(error) }); + return 2; + } } const emit = value => console.log(flags.json ? JSON.stringify(value) : value.detail); const openCodeReady = (dryRun) => { diff --git a/src/commands/x/aqe-store.mjs b/src/commands/x/aqe-store.mjs index fb26b5b7..9d4e5158 100644 --- a/src/commands/x/aqe-store.mjs +++ b/src/commands/x/aqe-store.mjs @@ -3,7 +3,7 @@ // while any AQE writer is open. The work is in src/lib/aqe-store-merge.mjs. import { mergeAqeStores, restoreSteps } from '../../lib/aqe-store-merge.mjs'; import { repoRoot } from '../../lib/paths.mjs'; -import { ok, warn, fail, info } from '../../lib/output.mjs'; +import { ok, warn, fail, info, reportFailure } from '../../lib/output.mjs'; export const options = { 'dry-run': { type: 'boolean', default: false }, @@ -139,7 +139,8 @@ function printInterrupted(result) { export async function run({ flags = {}, positionals = [], cwd = process.cwd(), merge = mergeAqeStores }) { const action = positionals[0] ?? 'status'; if (!['status', 'merge'].includes(action) || positionals.length > 1) { - fail('usage: ak x aqe-store [status|merge] [--yes] [--dry-run] [--json]'); + const error = 'usage: ak x aqe-store [status|merge] [--yes] [--dry-run] [--json]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); return 2; } const root = repoRoot(cwd); diff --git a/src/commands/x/codex-context.mjs b/src/commands/x/codex-context.mjs index 3502beca..16866127 100644 --- a/src/commands/x/codex-context.mjs +++ b/src/commands/x/codex-context.mjs @@ -1,6 +1,6 @@ import { loadKitConfig, saveKitConfig } from '../../lib/config.mjs'; import { inspectCodexContext, manageCodexContext, releaseCodexContext } from '../../lib/codex-context.mjs'; -import { info, ok, warn } from '../../lib/output.mjs'; +import { info, ok, warn, reportFailure } from '../../lib/output.mjs'; export const options = { 'dry-run': { type: 'boolean', default: false }, json: { type: 'boolean', default: false } }; export const help = `ak x codex-context — manage Codex's native per-model context capacities @@ -21,7 +21,11 @@ Examples: export async function run({ flags, positionals, contextOptions = {} }) { const choice = positionals[0] ?? 'status'; - if (!['status', 'max', 'off'].includes(choice) || positionals.length > 1) { warn(help); return 2; } + if (!['status', 'max', 'off'].includes(choice) || positionals.length > 1) { + const error = 'usage: ak x codex-context [status|max|off] [--dry-run] [--json]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(help) }); + return 2; + } const cfg = loadKitConfig(); const status = inspectCodexContext(cfg, contextOptions); if (choice === 'status' || flags['dry-run']) { diff --git a/src/commands/x/daemon-gc.mjs b/src/commands/x/daemon-gc.mjs index d6d6947d..8a1e691a 100644 --- a/src/commands/x/daemon-gc.mjs +++ b/src/commands/x/daemon-gc.mjs @@ -4,7 +4,7 @@ import { listDaemons, staleDaemons, reap, listMcpTransports, orphanedMcpTransports, reapMcpTransports, } from '../../lib/daemons.mjs'; -import { ok, warn, dim } from '../../lib/output.mjs'; +import { ok, warn, dim, reportFailure } from '../../lib/output.mjs'; export const options = { kill: { type: 'boolean', default: false }, @@ -32,10 +32,19 @@ Examples: ak x daemon-gc --kill reap stale background daemons ak x daemon-gc --mcp --kill also reap same-user PPID-1 MCP orphans`; -export async function run({ flags }) { - const daemons = await listDaemons(); +export async function run({ flags, positionals = [], deps = { daemonLifecycle: undefined, mcpLifecycle: undefined } }) { + if (positionals.length) { + const error = `unexpected argument '${positionals[0]}'`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); + return 2; + } + const { list, reap: reapFn } = { list: listDaemons, reap, ...deps.daemonLifecycle }; + const { list: listMcp, reap: reapMcp } = { + list: listMcpTransports, reap: reapMcpTransports, ...deps.mcpLifecycle, + }; + const daemons = await list(); const stale = staleDaemons(daemons); - const mcpTransports = await listMcpTransports(); + const mcpTransports = await listMcp(); const mcpOrphans = orphanedMcpTransports(mcpTransports); if (flags.json) { console.log(JSON.stringify({ @@ -47,14 +56,14 @@ export async function run({ flags }) { return 0; } if (stale.length && flags.kill) { - const reaped = reap(stale); + const reaped = reapFn(stale); for (const r of reaped) { if (r.killed) ok(`stopped stale daemon pid=${r.pid} ${dim(r.workspace ?? '')}`); else warn(`could not stop pid=${r.pid} (already exited?)`); } // Refresh daemon-sweep evidence so a later read sees the daemons that are // actually still alive, not the pre-reap list. - if (reaped.some((r) => r.killed)) await listDaemons({ refresh: true, record: true, source: 'daemon-gc' }); + if (reaped.some((r) => r.killed)) await list({ refresh: true, record: true, source: 'daemon-gc' }); } else if (stale.length) { for (const d of stale) { warn(`stale daemon pid=${d.pid} ${dim(d.workspace ?? '(unknown workspace)')} ${dim(d.workspaceExists ? `age ${d.ageSecs}s > TTL` : 'workspace gone')}`); @@ -63,7 +72,7 @@ export async function run({ flags }) { } if (flags.mcp && flags.kill) { - for (const result of reapMcpTransports(mcpOrphans)) { + for (const result of reapMcp(mcpOrphans)) { if (result.killed) ok(`stopped orphaned Ruflo MCP pid=${result.pid}`); else warn(`could not stop MCP pid=${result.pid} (identity or orphan proof changed)`); } diff --git a/src/commands/x/harvest.mjs b/src/commands/x/harvest.mjs index 86f2dcbc..7b690dc6 100644 --- a/src/commands/x/harvest.mjs +++ b/src/commands/x/harvest.mjs @@ -7,7 +7,7 @@ // memory root. It NEVER starts a daemon and NEVER backgrounds anything. import { loadKitConfig } from '../../lib/config.mjs'; import { planHarvest, runHarvest } from '../../lib/harvest.mjs'; -import { ok, fail, warn, info, dim, heading } from '../../lib/output.mjs'; +import { ok, fail, warn, info, dim, heading, reportFailure } from '../../lib/output.mjs'; export const options = { 'dry-run': { type: 'boolean', default: false }, @@ -45,7 +45,12 @@ Examples: ak x harvest record the outcome (only when opted in) ak x harvest --distill record, then distill the project store`; -export async function run({ flags }) { +export async function run({ flags, positionals = [] }) { + if (positionals.length) { + const error = `unexpected argument '${positionals[0]}'`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); + return 2; + } const cwd = process.cwd(); const cfg = loadKitConfig(); const enabled = cfg.harvest === true; diff --git a/src/commands/x/host-adapters-grants.mjs b/src/commands/x/host-adapters-grants.mjs index 8e3dbe8f..12c2ddd3 100644 --- a/src/commands/x/host-adapters-grants.mjs +++ b/src/commands/x/host-adapters-grants.mjs @@ -19,7 +19,7 @@ import { import { bootstrapHostAdapters } from '../../lib/adapters/admission.mjs'; import { loadKitConfig, saveKitConfig } from '../../lib/config.mjs'; import { applyAqeRouter } from '../../lib/providers.mjs'; -import { ok, warn, fail, info, bold } from '../../lib/output.mjs'; +import { ok, warn, fail, info, bold, reportFailure } from '../../lib/output.mjs'; import { findEntry, loadAndHash, stripControl, hookCommandsFor, } from './host-adapters.mjs'; @@ -396,12 +396,16 @@ async function reconcileRevokedAqeProvider({ } export async function revokeGrant({ - name, capability, grantsFile, cfg, env, cwd = process.cwd(), + name, capability, grantsFile, cfg, env, cwd = process.cwd(), flags = /** @type {{json?: boolean}} */ ({}), saveConfig = saveKitConfig, bootstrapAdapters = bootstrapHostAdapters, applyRouter = applyAqeRouter, }) { - if (typeof name !== 'string' || !name) { fail('usage: ak host adapters revoke-grant [capability]'); return 2; } + if (typeof name !== 'string' || !name) { + const error = 'usage: ak host adapters revoke-grant [capability]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); + return 2; + } const safeName = stripControl(name); if (typeof capability === 'string' && capability) { diff --git a/src/commands/x/host-adapters.mjs b/src/commands/x/host-adapters.mjs index 0264c2fe..9d2091db 100644 --- a/src/commands/x/host-adapters.mjs +++ b/src/commands/x/host-adapters.mjs @@ -25,7 +25,7 @@ import { HOST_REGISTRY } from '../../lib/adapters/registries.mjs'; import * as consentStore from '../../lib/adapters/consent.mjs'; import { runTieredConformance as defaultRunTieredConformance } from '../../lib/adapters/conformance.mjs'; import { loadKitConfig } from '../../lib/config.mjs'; -import { ok, warn, fail, info, dim, bold } from '../../lib/output.mjs'; +import { ok, warn, fail, info, dim, bold, reportFailure } from '../../lib/output.mjs'; import { grant as grantCap, gate as gateTier, status as statusReport, revokeGrant, } from './host-adapters-grants.mjs'; @@ -298,8 +298,12 @@ async function trust({ name, cfg, consent, reader, ask, isTTY, yes, expectHash } return 0; } -function revoke({ name, consent }) { - if (typeof name !== 'string' || !name) { fail('usage: ak host adapters revoke '); return 2; } +function revoke({ name, consent, flags = /** @type {{json?: boolean}} */ ({}) }) { + if (typeof name !== 'string' || !name) { + const error = 'usage: ak host adapters revoke '; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); + return 2; + } const existed = consent.revokeConsent(name); if (existed) { ok(`revoked consent for '${name}'`); return 0; } info(`no recorded consent for '${name}'`); @@ -437,9 +441,9 @@ async function conformance({ // standing consent or grant record, or it silently reactivates the next time // the flag is turned back on. const FAIL_SAFE_HANDLERS = { - revoke: (ctx) => revoke({ name: ctx.name, consent: ctx.consent }), + revoke: (ctx) => revoke({ name: ctx.name, consent: ctx.consent, flags: ctx.flags }), 'revoke-grant': (ctx) => revokeGrant({ - name: ctx.name, capability: ctx.positionals[2], grantsFile: ctx.grantsFile, cfg: ctx.cfg, env: ctx.env, cwd: ctx.cwd, + name: ctx.name, capability: ctx.positionals[2], grantsFile: ctx.grantsFile, cfg: ctx.cfg, env: ctx.env, cwd: ctx.cwd, flags: ctx.flags, ...(ctx.saveConfig ? { saveConfig: ctx.saveConfig } : {}), ...(ctx.bootstrapAdapters ? { bootstrapAdapters: ctx.bootstrapAdapters } : {}), ...(ctx.applyRouter ? { applyRouter: ctx.applyRouter } : {}), @@ -505,7 +509,8 @@ export async function run({ if (failSafe) return failSafe(ctx); if (!flagEnabled(env)) { - fail(`experimental host-adapter surface is disabled — set ${FLAG_ENV_VAR}=1`); + const error = `experimental host-adapter surface is disabled — set ${FLAG_ENV_VAR}=1`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); return 2; } @@ -514,6 +519,7 @@ export async function run({ const gated = GATED_HANDLERS[sub]; if (gated) return gated(ctx); - fail(`unknown host adapters subcommand: ${sub} (list|trust|revoke|conformance|grant|bless|gate|status|revoke-grant)`); + const error = `unknown host adapters subcommand: ${sub} (list|trust|revoke|conformance|grant|bless|gate|status|revoke-grant)`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); return 2; } diff --git a/src/commands/x/host.mjs b/src/commands/x/host.mjs index 73231e87..8f4ec772 100644 --- a/src/commands/x/host.mjs +++ b/src/commands/x/host.mjs @@ -32,7 +32,7 @@ import { hostManagement, hostEnableCommand, HOST_MANAGEMENT_LABELS, NOT_PARTICIPATING, } from '../../lib/host-management.mjs'; import { - ok, warn, fail, info, dim, bold, yellow, humanOutputToStderr, + ok, warn, fail, info, dim, bold, yellow, humanOutputToStderr, reportFailure, } from '../../lib/output.mjs'; import { repoRoot } from '../../lib/paths.mjs'; import { writeJsonWithBackup } from '../../lib/settings.mjs'; @@ -181,13 +181,19 @@ export const parseFallback = (str) => str.split(';').map((s) => s.trim()).filter return { provider: provider.trim().toLowerCase(), models: models.split(',').map((m) => m.trim()).filter(Boolean) }; }); -export async function run({ flags, positionals, pkgRoot }) { +export async function run({ flags, positionals, pkgRoot, deps = { hostLifecycle: undefined } }) { const sub = positionals[0] ?? 'status'; const cwd = process.cwd(); + if (['status', 'off', 'pick', 'reset-routes', 'align'].includes(sub) && positionals.length > 1) { + const error = `unexpected argument '${positionals[1]}'`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); + return 2; + } + if (sub === 'status') return status({ flags, cwd }); if (sub === 'off') return off({ cwd, pkgRoot, flags }); - if (sub === 'pick') return pick({ flags, cwd, pkgRoot }); + if (sub === 'pick') return pick({ flags, cwd, pkgRoot, deps }); if (sub === 'reset-routes') return resetRoutes({ flags, cwd }); if (sub === 'align') return (await import('./host-align.mjs')).run({ flags }); if (sub === 'adapters') { @@ -196,12 +202,17 @@ export async function run({ flags, positionals, pkgRoot }) { // read-only preview to give --dry-run, so it is refused outright // instead of silently behaving like a real run: a flag we declare is a // flag we honor, or refuse. - if (flags['dry-run']) { fail('ak host adapters has no preview; run it without --dry-run'); return 2; } + if (flags['dry-run']) { + const error = 'ak host adapters has no preview; run it without --dry-run'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); + return 2; + } return (await import('./host-adapters.mjs')).run({ flags, positionals: positionals.slice(1) }); } if (sub === 'check-connection') return (await import('./host-connection.mjs')).run({ flags, positionals: positionals.slice(1) }); - fail(`unknown host subcommand: ${sub} (status|pick|reset-routes|off|check-connection|adapters|align)`); + const error = `unknown host subcommand: ${sub} (status|pick|reset-routes|off|check-connection|adapters|align)`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); return 2; } @@ -322,7 +333,7 @@ const STATUS_SECTIONS = [ async function status({ flags, cwd }) { const cfg = loadKitConfig(); - const facts = await collectIntegrationFacts({ cwd, cfg }); + const facts = await collectIntegrationFacts({ cwd, cfg, record: !flags['dry-run'] }); const hosts = facts.hosts; const providers = facts.providers; const { scope } = settingsTarget(cwd); @@ -378,27 +389,38 @@ async function resetRoutes({ flags, cwd }) { const cfg = loadKitConfig(); const policy = cfg.routing?.routes ?? {}; const diverged = divergedRoutes(policy); - if (!diverged.length) { ok('no seeded routes diverge from the current defaults'); return 0; } - - console.log(bold('seeded routes that diverge from current defaults')); - for (const d of diverged) { - const head = d.modelDiverged ? `${d.model} → ${d.defaultModel}` : d.model; - console.log(` ${d.activity.padEnd(18)} ${d.host.padEnd(7)} ${head}`); - for (const e of d.escalation) console.log(` ${dim('escalation:')} ${e.model} → ${e.defaultModel}`); - // The trade, not just the ids: choosing on a price axis while paying on a - // turns axis is the misreading this whole surface exists to prevent. - if (d.modelDiverged && d.currentNote) console.log(dim(` now: ${d.model} — ${d.currentNote}`)); - if (d.modelDiverged && d.defaultNote) console.log(dim(` default: ${d.defaultModel} — ${d.defaultNote}`)); - for (const e of d.escalation) { - const note = modelNote(e.defaultModel); - if (note) console.log(dim(` default: ${e.defaultModel} — ${note}`)); + const jsonDryRun = flags['dry-run'] && flags.json; + const previewJson = (activities) => console.log(JSON.stringify({ dryRun: true, activities }, null, 2)); + if (!diverged.length) { + if (jsonDryRun) previewJson([]); + else ok('no seeded routes diverge from the current defaults'); + return 0; + } + + if (!jsonDryRun) { + console.log(bold('seeded routes that diverge from current defaults')); + for (const d of diverged) { + const head = d.modelDiverged ? `${d.model} → ${d.defaultModel}` : d.model; + console.log(` ${d.activity.padEnd(18)} ${d.host.padEnd(7)} ${head}`); + for (const e of d.escalation) console.log(` ${dim('escalation:')} ${e.model} → ${e.defaultModel}`); + // The trade, not just the ids: choosing on a price axis while paying on a + // turns axis is the misreading this whole surface exists to prevent. + if (d.modelDiverged && d.currentNote) console.log(dim(` now: ${d.model} — ${d.currentNote}`)); + if (d.modelDiverged && d.defaultNote) console.log(dim(` default: ${d.defaultModel} — ${d.defaultNote}`)); + for (const e of d.escalation) { + const note = modelNote(e.defaultModel); + if (note) console.log(dim(` default: ${e.defaultModel} — ${note}`)); + } } } let picked; if (flags.activity !== undefined) { const want = flags.activity.split(',').map((s) => s.trim()).filter(Boolean); - for (const a of want.filter((a) => !ACTIVITIES.includes(a))) warn(`unknown activity '${a}' — ignored`); + for (const a of want.filter((a) => !ACTIVITIES.includes(a))) { + if (jsonDryRun) await humanOutputToStderr(() => warn(`unknown activity '${a}' — ignored`)); + else warn(`unknown activity '${a}' — ignored`); + } picked = want.filter((a) => diverged.some((d) => d.activity === a)); } else if (flags.yes || flags['dry-run']) { // --dry-run never prompts: with nothing else naming a subset, preview the @@ -413,11 +435,18 @@ async function resetRoutes({ flags, cwd }) { ? diverged.map((d) => d.activity) : ans.split(',').map((s) => s.trim()).filter((a) => diverged.some((d) => d.activity === a)); } - if (!picked.length) { info('no routes reset — routes left as they are'); return 0; } + if (!picked.length) { + if (jsonDryRun) previewJson([]); + else info('no routes reset — routes left as they are'); + return 0; + } if (flags['dry-run']) { - info(`would reset ${picked.length} route(s) to the current defaults: ${picked.join(', ')}`); - info('dry run — nothing changed'); + if (jsonDryRun) previewJson(picked); + else { + info(`would reset ${picked.length} route(s) to the current defaults: ${picked.join(', ')}`); + info('dry run — nothing changed'); + } return 0; } @@ -457,11 +486,30 @@ function printOffDryRunSummary(cfg, opts) { if (codexOwned) console.log(' would remove ak-managed Codex MCP wiring'); } -async function off({ cwd, pkgRoot, flags = {} }) { +async function off({ cwd, pkgRoot, flags = /** @type {{ json?: boolean, 'dry-run'?: boolean }} */ ({}) }) { const cfg = loadKitConfig(); if (flags['dry-run']) { - printOffDryRunSummary(cfg, { pkgRoot }); - info('dry run — nothing changed'); + if (flags.json) { + await humanOutputToStderr(() => { + printOffDryRunSummary(cfg, { pkgRoot }); + info('dry run — nothing changed'); + }); + console.log(JSON.stringify({ + dryRun: true, + wouldDisable: HOSTS.filter((h) => cfg.integrations.hosts[h.id]).map((h) => h.id), + primaryHost: DEFAULT_PRIMARY_HOST, + wouldClear: ['aqe provider/fallback', 'ruflo providers', 'activity routing'], + wouldStripManagedProviderEnv: true, + wouldRestoreOrRemoveManagedAqeConfig: true, + wouldReconcileOpencodeGuidance: !!pkgRoot, + wouldTeardownOpencode: !!(cfg.integrations.hosts.opencode || cfg.integrations?.ownership?.opencode), + wouldRemoveManagedCodexMcp: cfg.integrations?.ownership?.codex?.mcp === 'ak' + || cfg.integrations?.ownership?.codex?.reverseMcp === 'ak', + }, null, 2)); + } else { + printOffDryRunSummary(cfg, { pkgRoot }); + info('dry run — nothing changed'); + } return 0; } const codexMcpManaged = cfg.integrations?.ownership?.codex?.mcp === 'ak'; @@ -715,8 +763,9 @@ async function resolvePickDecision(cfg, { const known = new Set([...MANAGED_HOSTS, ...EFFECTIVE_ROUTING]); const unknown = enabled.filter((h) => !known.has(h)); if (unknown.length) { - fail(`unknown host(s): ${unknown.join(', ')} (valid: ${[...known].join(', ')}) — nothing changed`); - return { code: 2 }; + const error = `unknown host(s): ${unknown.join(', ')} (valid: ${[...known].join(', ')}) — nothing changed`; + fail(error); + return { code: 2, error }; } // The routing set needs at least one primary-capable member; OpenCode remains // routable but cannot satisfy that primary-host invariant on its own. @@ -869,25 +918,28 @@ async function retireCodexOnDisable(cfg, cwd, { codexMcpManaged, rufloCodexManag /** Install any enabled host that is entirely absent (external installs * untouched). Unlike setup's install loop, pick never prompts first — the * user already confirmed the trust manifest for this exact enable. */ -async function installPickAbsentHosts(cfg, cwd) { +export async function installPickAbsentHosts(cfg, cwd, lifecycle = {}) { + const { installState, install, collectFacts } = { + installState: hostInstallState, install: installHost, collectFacts: collectIntegrationFacts, ...lifecycle, + }; let installed = false; for (const h of HOSTS) { if (!cfg.integrations.hosts[h.id]) continue; - if ((await hostInstallState(h)).method !== 'absent') continue; + if ((await installState(h)).method !== 'absent') continue; info(`${h.id} not installed — installing ${h.pkg}…`); - const r = await installHost(h.id); + const r = await install(h.id); (r.ok ? ok : warn)(`${h.id}: ${r.detail}`); if (r.ok) { installed = true; // hostInstallState() above already recorded the pre-install 'absent' // evidence; re-probe now so a subsequent `ak status` doesn't read that // stale row back. - await hostInstallState(h, { refresh: true, record: true, source: 'host-pick' }); + await installState(h, { refresh: true, record: true, source: 'host-pick' }); } } // host-setup covers every host in one call; refresh it once after the // loop, not per host, once anything actually changed. - if (installed) await collectIntegrationFacts({ cwd, cfg, refresh: true, record: true, source: 'host-pick' }); + if (installed) await collectFacts({ cwd, cfg, refresh: true, record: true, source: 'host-pick' }); } /** opencode enable half: apply the same owner-module stack setup/sync use — @@ -1058,7 +1110,7 @@ async function resolvePickIntentAndPreview({ aqeProviderTypes, aqeChainProviderTypes, }); - if (decision.code !== undefined) return { code: decision.code }; + if (decision.code !== undefined) return decision; const { enabled, routing, primaryHost, aqeProvider, seed, } = decision; @@ -1095,7 +1147,7 @@ async function resolvePickIntentAndPreview({ }; } -export async function pick({ flags, cwd, pkgRoot, migrateRoutes = migrateRetiredRoutesInConfig }) { +export async function pick({ flags, cwd, pkgRoot, deps = { hostLifecycle: undefined }, migrateRoutes = migrateRetiredRoutesInConfig }) { const aqeProviderTypes = aqeSelectableProviderTypes(); const aqeChainProviderTypes = aqeSelectableChainProviderTypes(); const cfg = loadKitConfig(); @@ -1133,6 +1185,7 @@ export async function pick({ flags, cwd, pkgRoot, migrateRoutes = migrateRetired const outcome = jsonDryRun ? await humanOutputToStderr(resolve) : await resolve(); if (outcome.code !== undefined) { if (outcome.json) console.log(JSON.stringify(outcome.json, null, 2)); + else if (jsonDryRun) console.log(JSON.stringify({ error: outcome.error ?? 'host pick refused', exitCode: outcome.code }, null, 2)); return outcome.code; } const { @@ -1145,7 +1198,7 @@ export async function pick({ flags, cwd, pkgRoot, migrateRoutes = migrateRetired } saveKitConfig(cfg); - await installPickAbsentHosts(cfg, cwd); + await installPickAbsentHosts(cfg, cwd, deps.hostLifecycle); const { incompleteTeardown } = await applyPickOpencodeLifecycle(cfg, { pkgRoot, cwd, prevOpencode }); const { router } = await applyPickProviderStack(cfg, cwd, { diff --git a/src/commands/x/reference.mjs b/src/commands/x/reference.mjs index 8c42cd83..d37b40ff 100644 --- a/src/commands/x/reference.mjs +++ b/src/commands/x/reference.mjs @@ -1,7 +1,7 @@ // x reference — inspect (diff) or reconcile (sync) every managed host-guidance target. import { reconcileGuidance } from '../../lib/blocks.mjs'; import { loadKitConfig } from '../../lib/config.mjs'; -import { ok, warn, dim } from '../../lib/output.mjs'; +import { ok, warn, dim, reportFailure } from '../../lib/output.mjs'; export const options = { json: { type: 'boolean', default: false } }; @@ -20,6 +20,11 @@ Examples: export async function run({ flags, positionals, pkgRoot }) { const sub = positionals[0] ?? 'diff'; + if (!['diff', 'sync'].includes(sub) || positionals.length > 1) { + const error = 'usage: ak x reference [diff|sync] [--json]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); + return 2; + } const cfg = loadKitConfig(); const dryRun = sub !== 'sync'; const res = await reconcileGuidance({ diff --git a/src/commands/x/skills.mjs b/src/commands/x/skills.mjs index 08b62044..5dc7f6ac 100644 --- a/src/commands/x/skills.mjs +++ b/src/commands/x/skills.mjs @@ -3,7 +3,7 @@ import path from 'node:path'; import { collectCatalog } from '../../lib/footprint/catalog.mjs'; import { buildSkillMaintenancePlan } from '../../lib/skill-maintenance-plan.mjs'; import { loadKitConfig } from '../../lib/config.mjs'; -import { heading, info, warn, dim } from '../../lib/output.mjs'; +import { heading, info, warn, dim, reportFailure } from '../../lib/output.mjs'; export const options = { project: { type: 'string' }, @@ -27,7 +27,8 @@ Examples: export async function run({ flags, positionals }) { if ((positionals[0] ?? 'plan') !== 'plan' || positionals.length > 1) { - warn('usage: ak x skills plan [--project PATH] [--json]'); + const error = 'usage: ak x skills plan [--project PATH] [--json]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); return 2; } const project = path.resolve(flags.project ?? process.cwd()); diff --git a/src/commands/x/statusline.mjs b/src/commands/x/statusline.mjs index 13465758..92567c16 100644 --- a/src/commands/x/statusline.mjs +++ b/src/commands/x/statusline.mjs @@ -3,7 +3,7 @@ import { PRESETS, applyCodexStatusline, inspectCodexStatusline, projectionFor, removeCodexStatusline, statuslineDrift, } from '../../lib/codex-statusline.mjs'; -import { ok, info, warn } from '../../lib/output.mjs'; +import { ok, info, warn, reportFailure } from '../../lib/output.mjs'; export const options = { 'dry-run': { type: 'boolean', default: false }, @@ -33,7 +33,7 @@ Examples: export async function run({ flags, positionals }) { const cfg = loadKitConfig(); const [target = 'status', choice] = positionals; - if (target === 'status') { + if (target === 'status' && positionals.length <= 1) { const current = inspectCodexStatusline(); const drift = statuslineDrift(cfg); const result = { ownership: cfg.statusline?.codex ?? null, current, drifted: drift.drifted }; @@ -45,8 +45,9 @@ export async function run({ flags, positionals }) { } return 0; } - if (target !== 'codex' || !choice) { - warn('usage: ak x statusline status | codex native | codex extended | codex off'); + if (target !== 'codex' || !choice || positionals.length !== 2) { + const error = 'usage: ak x statusline status | codex native | codex extended | codex off'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); return 2; } if (choice === 'off') { @@ -63,7 +64,11 @@ export async function run({ flags, positionals }) { ok(`codex status-line management disabled${result.changed ? '; unchanged managed keys removed' : '; user-modified keys preserved'}`); return 0; } - if (!PRESETS[choice]) { warn(`unknown preset '${choice}' (expected native, extended, or off)`); return 2; } + if (!PRESETS[choice]) { + const error = `unknown preset '${choice}' (expected native, extended, or off)`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); + return 2; + } if (flags['dry-run']) { info(`[dry-run] apply Codex ${choice} preset at user scope`); return 0; } // Persist ownership first. If the TOML merge then fails, sync retains enough // intent to report/retry it; the inverse ordering could mutate config.toml diff --git a/src/lib/aqe-readiness.mjs b/src/lib/aqe-readiness.mjs index 6c8c204f..91f819c4 100644 --- a/src/lib/aqe-readiness.mjs +++ b/src/lib/aqe-readiness.mjs @@ -24,33 +24,13 @@ export function aqeEmbeddingConfiguration({ packageRoot = aqeRoot(), env = proce } catch { return { status: 'missing-backend', backend: null }; } } -// TEMPORARY (remove once fixed upstream; tracked by pacphi/agentic-kit#240): on a live -// RVF lock, agentic-qe 3.14.3 logs the busy warning, then falls through to a create -// attempt that fails with FsyncFailed; store and lock are untouched (agentic-qe#574). -// 3.14.4 carries agentic-qe#719, a partial fix (it rethrows LockHeld); whether 3.14.4 -// still emits this sequence is unverified, so the rule stays for it. Only that exact -// sequence is contention. -// The middle line is emitted solely by AQE's live-owner quarantine refusal, so a bare -// FsyncFailed, or a lock warning plus FsyncFailed without it, still fails below. The rule -// has no version gate. Remove it and its test when a released agentic-qe fixes -// agentic-qe#574 and that release is the kit floor; #719 alone does not remove it. -const LIVE_OWNER_CONTENTION = [ - /is locked by a live process/, - /is unusable but its lock is held by a live process/, - /FsyncFailed|0x0303/, -]; - export function classifyAqeStartup(result) { if (result.code !== 0) return { status: 'failed', reason: 'AQE command failed' }; const output = `${result.stdout ?? ''}\n${result.stderr ?? ''}`; - const contention = LIVE_OWNER_CONTENTION.every(line => line.test(output)); - if (!contention && /FsyncFailed|0x0303/.test(output)) return { status: 'failed', reason: 'RVF backend failed' }; + if (/FsyncFailed|0x0303/.test(output)) return { status: 'failed', reason: 'RVF backend failed' }; if (/prewarm failed|Transformer initialization previously failed|Embedding model failed|semantic embedding unavailable/i.test(output)) { return { status: 'degraded', reason: 'Embedding initialization failed' }; } - if (contention) { - return { status: 'busy', reason: 'RVF is held by another live process; its FsyncFailed came from an AQE create attempt during contention (agentic-qe#574), not a storage error; SQLite fallback observed; owner health and RVF integrity unverified' }; - } if (/locked by a live process|lock is held by a live process/.test(output)) { return { status: 'busy', reason: 'RVF is held by another live process; SQLite fallback observed; owner health and RVF integrity unverified' }; } diff --git a/src/lib/exec.mjs b/src/lib/exec.mjs index 6726a0f5..9a30aa6a 100644 --- a/src/lib/exec.mjs +++ b/src/lib/exec.mjs @@ -1,5 +1,5 @@ // Subprocess helpers. Rule (binding, from the plan): NOTHING goes through a -// shell string — execFile with argv arrays only, shell ALWAYS false. +// shell string — spawn with argv arrays only, shell ALWAYS false. // // npm/npx/claude/deja/ruflo/aqe/claude-flow are .cmd shims on Windows, and // Windows' CreateProcess cannot launch a .cmd directly — that historically @@ -7,26 +7,23 @@ // command line to cmd.exe as ONE string (CVE-class: any arg with `&`/`|`/`^` // breaks out into a second command). The actual fix is resolving the shim to // its real file on PATH. Native .com/.exe files run directly. A .cmd shim is -// never passed to execFile: Node does not execute batch files without a shell. -// Instead, its sibling .ps1 shim runs through Windows PowerShell's `-File` -// interface, preserving every caller argument as a separate argv element. -import { execFile, spawn } from 'node:child_process'; +// never passed to spawn: Node does not execute batch files without a shell. +// An exact recognized npm wrapper pair launches its declared public bin with +// Node directly, preserving interactive stdio and literal argv. Other wrappers +// keep their sibling .ps1 through Windows PowerShell's `-File` interface. +import { spawn } from 'node:child_process'; import { AsyncLocalStorage } from 'node:async_hooks'; -import { promisify } from 'node:util'; import fs from 'node:fs'; import path from 'node:path'; import { isWindows } from './paths.mjs'; +import { npmShimInvocation, windowsEnvValue, mergeWindowsEnv } from './windows-npm-shim.mjs'; -const pexecFile = promisify(execFile); const MAX_EXEC_BUFFER = 16 * 1024 * 1024; // A caller that bounds work it does not own (a live check under `ak status // --refresh=live` running the full live-check suite) scopes an AbortSignal // here; every run() inside the scope that passes no signal of its own uses it, -// so a timed-out check's direct child processes stop and its own cleanup still -// runs. The abort signals only that direct child: a process it started keeps -// running (on Windows a .cmd shim's child is PowerShell, so the ruflo or aqe -// node process behind it survives the abort). +// so a timed-out check's entire owned process tree stops and cleanup still runs. const abortScope = new AsyncLocalStorage(); /** Run `fn` with `signal` as the default abort signal for run() calls in it. @@ -41,18 +38,20 @@ const CMD_SHIMS = new Set([ /** Build a shell-free invocation for `cmd`, trying Windows' shim extensions in * PATHEXT order. A native executable is launched directly; a .cmd shim is - * accepted only when its sibling .ps1 and system PowerShell both exist. + * mapped to its own public Node bin only for an exact recognized npm wrapper + * pair; other .cmd wrappers require a sibling .ps1 and system PowerShell. * Falls back to the bare name with resolved:false when no safe target exists. * Exported: the execution adapters spawn these CLIs directly (subprocess.mjs * for claude/codex, opencode.mjs for the serve child, x/ruflo-mcp.mjs for the * Ruflo MCP launcher) and must share the same * resolution `run()`/`have()` use, or readiness passes but launch ENOENTs on - * Windows (swarm review, #88). */ -export function resolveShim(cmd, args = [], { windows = isWindows, env = process.env } = {}) { + * Windows (swarm review, #88). `npmBin:false` retains the PowerShell boundary + * for native diagnostics. */ +export function resolveShim(cmd, args = [], { windows = isWindows, env = process.env, npmBin = true } = {}) { const direct = { command: cmd, args: [...args], resolved: !windows }; if (!windows) return direct; - const systemRoot = env.SystemRoot || env.WINDIR; + const systemRoot = windowsEnvValue(env, 'SystemRoot') || windowsEnvValue(env, 'WINDIR'); const powershell = systemRoot ? path.join(systemRoot, 'System32', 'WindowsPowerShell', 'v1.0', 'powershell.exe') : null; @@ -64,7 +63,10 @@ export function resolveShim(cmd, args = [], { windows = isWindows, env = process if (ext === '.com' || ext === '.exe' || !ext) { return { command: candidate, args: [...args], resolved: true }; } - if (ext !== '.cmd' || !powershell) return null; + if (ext !== '.cmd') return null; + const npm = npmBin ? npmShimInvocation(candidate, args, env) : null; + if (npm) return npm; + if (!powershell) return null; const script = `${candidate.slice(0, -ext.length)}.ps1`; try { if (!fs.statSync(script).isFile() || !fs.statSync(powershell).isFile()) return null; @@ -80,15 +82,21 @@ export function resolveShim(cmd, args = [], { windows = isWindows, env = process }; if (path.isAbsolute(cmd)) return invocationFor(cmd) ?? direct; - const exts = (env.PATHEXT || '.COM;.EXE;.BAT;.CMD') + const exts = (windowsEnvValue(env, 'PATHEXT') || '.COM;.EXE;.BAT;.CMD') .split(';') .map((ext) => ext.trim()) .filter(Boolean); - for (const dir of (env.PATH || env.Path || '').split(path.delimiter)) { + for (const dir of (windowsEnvValue(env, 'PATH') || '').split(path.delimiter)) { if (!dir) continue; for (const ext of exts) { - const invocation = invocationFor(path.join(dir, cmd + ext.toLowerCase())); + const candidate = path.join(dir, cmd + ext.toLowerCase()); + const invocation = invocationFor(candidate); if (invocation) return invocation; + // An earlier custom/unusable .cmd still selects this installation. Do + // not silently switch to a later package because its shim is recognized. + if (ext.toLowerCase() === '.cmd') { + try { if (fs.statSync(candidate).isFile()) return direct; } catch { /* absent */ } + } } } return direct; @@ -142,13 +150,18 @@ export function killProcessTree(child, { platform = process.platform, spawnFn = /** Accumulate one child stream, capped at `maxBuffer` — `execFile` applies its * own cap internally, so the spawn-based path below has to reimplement it. */ function captureStream(stream, encoding, maxBuffer, onOverflow) { - const state = { text: '', overflowed: false }; - stream?.setEncoding?.(encoding); + const chunks = []; + const state = { + bytes: 0, + overflowed: false, + get text() { return Buffer.concat(chunks).toString(encoding); }, + }; stream?.on('data', (chunk) => { if (state.overflowed) return; - state.text += chunk; - if (state.text.length > maxBuffer) { - state.text = state.text.slice(0, maxBuffer); + const remaining = maxBuffer - state.bytes; + chunks.push(chunk.subarray(0, remaining)); + state.bytes += chunk.length; + if (state.bytes > maxBuffer) { state.overflowed = true; onOverflow(); } @@ -156,63 +169,121 @@ function captureStream(stream, encoding, maxBuffer, onOverflow) { return state; } -/** The stdin-feeding, process-group variant of `run()`, used whenever a caller - * passes `opts.input`. Two reasons it exists, both from the security review: - * the payload stays out of argv (SEC-7 — `ps -ww` shows argv to the same user - * here, and `/proc//cmdline` shows it to ANY local user on Linux), and - * the child leads its own process group so the timeout reaps its subprocesses - * (SEC-8). - * - * WHY `spawn` AND NOT `execFile`: execFile forwards only a fixed whitelist of - * options through to spawn, and `detached` is not on it — passing it there is - * silently ignored, and the child stays in the PARENT's process group, where - * `process.kill(-pid)` fails ESRCH and the grandchild survives. Measured, not - * assumed. The timeout is likewise managed here rather than handed to the - * child process API, whose own `timeout` signals only the direct child. */ -function runWithInput(command, args, execOpts, { windows, input }) { +/** `spawn` is required for both paths: execFile silently drops `detached`, so + * its AbortSignal and timeout can only stop the direct child. Input stays on + * stdin and never enters argv. The Windows taskkill completion is awaited + * before returning, even if the direct child closes first. */ +function runOwned(command, args, execOpts, { windows, input }) { const { timeout, encoding, maxBuffer, cwd, env, signal } = execOpts; + if (signal?.aborted) return Promise.resolve({ code: 1, stdout: '', stderr: 'The operation was aborted' }); return new Promise((resolve) => { let child; try { - child = spawn(command, args, { cwd, env, signal, shell: false, detached: !windows }); + child = spawn(command, args, { cwd, env, shell: false, detached: !windows }); } catch (err) { resolve(failureResult(err)); return; } let failure = null; - const abort = (reason) => { failure ??= reason; killGroup(child); }; + let closed = false; + let stopping = Promise.resolve(); + const closePipes = () => { + child.stdout?.destroy(); + child.stderr?.destroy(); + child.stdin?.destroy(); + }; + const incomplete = (reason) => { + failure = `${reason}; incomplete process-tree cleanup`; + closePipes(); + }; + const abort = (reason) => { + if (closed) return; + if (failure) return; + failure = reason; + if (!windows) { killGroup(child); return; } + // A reaped root PID may already name a different process. The owned + // descendant may still hold our pipes, but cannot be safely found by PID. + if (child.exitCode !== null || child.signalCode !== null) { + incomplete(reason); + return; + } + if (!Number.isInteger(child.pid) || child.pid < 1) { + incomplete(reason); + return; + } + stopping = new Promise((done) => { + let killer; + const fallback = (detail) => { + incomplete(`${reason} (${detail})`); + if (child.exitCode === null && child.signalCode === null) { + try { child.kill('SIGKILL'); } catch { /* exited */ } + } + done(); + }; + try { + killer = spawn('taskkill.exe', ['/PID', String(child.pid), '/T', '/F'], { + stdio: 'ignore', shell: false, + }); + } catch { fallback('taskkill could not start'); return; } + const finish = (code, detail) => { + clearTimeout(deadline); + killer.removeListener('error', onError); + killer.removeListener('close', onClose); + if (code === 0) done(); + else fallback(detail); + }; + const onError = () => finish(null, 'taskkill could not start'); + const onClose = (code) => finish(code, `taskkill exited ${code}`); + const deadline = setTimeout(() => { + try { killer.kill('SIGKILL'); } catch { /* already exited */ } + killer.unref?.(); + finish(null, 'taskkill exceeded cleanup deadline'); + }, 1_000); + killer.once('error', onError); + killer.once('close', onClose); + }); + }; + const onSignal = () => abort('The operation was aborted'); + signal?.addEventListener('abort', onSignal, { once: true }); + if (signal?.aborted) onSignal(); const out = captureStream(child.stdout, encoding, maxBuffer, () => abort('stdout maxBuffer length exceeded')); const errOut = captureStream(child.stderr, encoding, maxBuffer, () => abort('stderr maxBuffer length exceeded')); - const timer = setTimeout(() => abort(`timed out after ${timeout}ms`), timeout); - timer.unref?.(); + const timer = timeout > 0 ? setTimeout(() => abort(`timed out after ${timeout}ms`), timeout) : null; + timer?.unref?.(); child.on('error', (err) => { clearTimeout(timer); - resolve(failureResult(err, out.text, errOut.text)); + signal?.removeEventListener('abort', onSignal); + stopping.then(() => resolve(failureResult(err, out.text, errOut.text))); }); - child.on('close', (code) => { + child.on('close', (code, exitSignal) => { + closed = true; clearTimeout(timer); + signal?.removeEventListener('abort', onSignal); const exitCode = typeof code === 'number' ? code : 1; - resolve({ + stopping.then(() => resolve({ code: failure ? exitCode || 1 : exitCode, stdout: out.text, - stderr: failure ? errOut.text || failure : errOut.text, - }); + stderr: failure ? [errOut.text, failure].filter(Boolean).join('\n') : errOut.text, + ...(exitSignal ? { signal: exitSignal } : {}), + })); }); // A child that exits without reading its input makes this write fail with // EPIPE. That is the child's own non-zero exit to report, not a crash here. child.stdin?.on('error', () => {}); - child.stdin?.end(input); + if (typeof input === 'string') child.stdin?.end(input); + else child.stdin?.end(); }); } /** Run a command; never throws. Returns {code, stdout, stderr}. * `opts.input` (a string) is delivered on the child's stdin instead of argv; - * see `runWithInput` for the two guarantees that carries. */ + * `runOwned` keeps that payload out of the process table. */ export async function run(cmd, args = [], opts = {}) { try { - const env = opts.env ? { ...process.env, ...opts.env } : process.env; const windows = opts.windows ?? isWindows; + const env = opts.env + ? (windows ? mergeWindowsEnv(process.env, opts.env) : { ...process.env, ...opts.env }) : process.env; const invocation = CMD_SHIMS.has(cmd) ? resolveShim(cmd, args, { windows, env }) : { command: cmd, args }; @@ -229,13 +300,9 @@ export async function run(cmd, args = [], opts = {}) { env, shell: false, }; - if (typeof opts.input === 'string') { - return await runWithInput(invocation.command, invocation.args, execOpts, { - windows, input: opts.input, - }); - } - const { stdout, stderr } = await pexecFile(invocation.command, invocation.args, execOpts); - return { code: 0, stdout, stderr }; + return await runOwned(invocation.command, invocation.args, execOpts, { + windows, input: opts.input, + }); } catch (err) { return failureResult(err); } diff --git a/src/lib/file-identity.mjs b/src/lib/file-identity.mjs new file mode 100644 index 00000000..b2f764c3 --- /dev/null +++ b/src/lib/file-identity.mjs @@ -0,0 +1,18 @@ +// Persisted file IDs are decimal strings. A legacy Number is comparable only +// when it was a safe integer; an unsafe Number has already lost information. +export function fileId(value) { + if (typeof value === 'bigint') return value >= 0n ? value.toString() : null; + if (typeof value === 'number') return Number.isSafeInteger(value) && value >= 0 ? String(value) : null; + if (typeof value === 'string' && /^(?:0|[1-9]\d*)$/.test(value)) return value; + return null; +} + +export function sameFileId(left, right) { + const a = fileId(left); + return a !== null && a === fileId(right); +} + +export function statMtimeMs(stat) { + if (typeof stat.mtimeNs !== 'bigint') return stat.mtimeMs; + return Number(stat.mtimeNs / 1_000_000n) + Number(stat.mtimeNs % 1_000_000n) / 1_000_000; +} diff --git a/src/lib/hook-audit/agentic-dependency-constraints.json b/src/lib/hook-audit/agentic-dependency-constraints.json index ff0af7e5..bac4266e 100644 --- a/src/lib/hook-audit/agentic-dependency-constraints.json +++ b/src/lib/hook-audit/agentic-dependency-constraints.json @@ -501,14 +501,15 @@ "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, "mapping": "mapped", "kitImpact": { "refs": ["lane C: live-lock busy rule", "pacphi/agentic-kit#240"], "files": ["docs/host-support.md"] }, - "adjustment": "agentic-qe 3.14.4 reports the real LockHeld instead of FsyncFailed after a live lock (Windows confirmation 2026-09-28). Re-run the live-owner fixture on macOS and Linux against 3.14.4; if no FsyncFailed follows the live-lock warning, remove the temporary busy rule in src/lib/aqe-readiness.mjs and close pacphi/agentic-kit#240.", - "status": "watching", + "adjustment": "Released agentic-qe 3.14.4 passed disposable live-owner conformance on macOS and Linux: status and the shipped adapter emitted LockHeld without FsyncFailed, and RVF/lock bytes were unchanged. ak retired the exact FsyncFailed-as-busy exception; old 3.14.3 output now fails closed. Ordinary live-lock SQLite fallback remains busy. This is a verified artifact baseline for this rule, not a universal AQE minimum or native Windows conformance. Upstream issue state remains separate from this kit retirement.", + "status": "retired", "constraintIds": [], "history": [ { "date": "2026-09-26", "event": "commented" }, { "date": "2026-09-26", "event": "registered" }, { "date": "2026-09-27", "event": "commented", "note": "#719 fixed FsyncFailed; a small fall-through remains (issuecomment-5863219357)" }, - { "date": "2026-09-28", "event": "commented", "note": "thanked fedor-drmanovic for the Windows 3.14.4 results; ak re-runs its live-owner fixture on macOS and Linux next (issuecomment-5878041570)" } + { "date": "2026-09-28", "event": "commented", "note": "thanked fedor-drmanovic for the Windows 3.14.4 results; ak re-runs its live-owner fixture on macOS and Linux next (issuecomment-5878041570)" }, + { "date": "2026-09-29", "event": "retired", "note": "ak retired its exact FsyncFailed-as-busy exception after released 3.14.4 passed native macOS and Linux live-owner probes; ordinary LockHeld fallback remains busy; no native Windows AQE proof or upstream closure claimed" } ] }, { @@ -749,7 +750,7 @@ { "date": "2026-09-26", "event": "registered" }, { "date": "2026-09-27", "event": "retired", "note": "merged and released in 3.14.4; context only: agentic-qe#574, tracked by pacphi/agentic-kit#240, drives removing the busy rule" } ], - "note": "Context only: a partial fix for agentic-qe#574 (rethrows LockHeld). Its release alone does not prove a live lock no longer surfaces FsyncFailed, so agentic-qe#574 drives removing the busy rule and closing pacphi/agentic-kit#240." + "note": "Context only: a partial fix for agentic-qe#574 (rethrows LockHeld), released in 3.14.4. Native macOS and Linux conformance against that artifact later showed no FsyncFailed under a live lock; ak retired the exact exception on 2026-09-29. Native Windows AQE conformance and pacphi/agentic-kit#240 closure remain separate." }, { "id": "proffesor-for-testing/agentic-qe#734", @@ -1128,7 +1129,8 @@ { "date": "2026-09-26", "event": "commented", "note": "tested the single-threaded ONNX Runtime mitigation on Apple Silicon: inconclusive, the abort does not reproduce locally (issuecomment-5850382245)" }, { "date": "2026-09-27", "event": "commented", "note": "hosted probe run 36333572972 on Ruflo 3.46.1: 10/10 aborts by default and 10/10 with single-threaded ONNX Runtime sessions; mitigation disproven (issuecomment-5857781254)" }, { "date": "2026-09-27", "event": "commented", "note": "3.47.0 status, transformers 3.x / ONNX Runtime 1.21 lead, fix approaches; 11/11 nightly aborts in the continue-on-error step (issuecomment-5863246672)" }, - { "date": "2026-09-28", "event": "commented", "note": "thanked vidaunited for the trace hook; 12 of 12 macOS nightly aborts on 3.47.0, with a new WASM backend line before the abort (issuecomment-5878044230)" } + { "date": "2026-09-28", "event": "commented", "note": "thanked vidaunited for the trace hook; 12 of 12 macOS nightly aborts on 3.47.0, with a new WASM backend line before the abort (issuecomment-5878044230)" }, + { "date": "2026-09-29", "event": "commented", "note": "posted the approved source-bound native trace on #2885 (issuecomment-5891510508); the failure occurred after heal, the outer command exited 1, and its learning step remained nonblocking for the job; the trace does not establish a cause or released fix" } ] }, { @@ -1524,7 +1526,8 @@ "history": [ { "date": "2026-09-24", "event": "filed" }, { "date": "2026-09-26", "event": "registered" }, - { "date": "2026-09-27", "event": "commented", "note": "answered from the Codex CLI source (issuecomment-5863289657)" } + { "date": "2026-09-27", "event": "commented", "note": "answered from the Codex CLI source (issuecomment-5863289657)" }, + { "date": "2026-09-29", "event": "reviewed", "note": "still open with only ak's source-analysis comment; no maintainer-supported guidance or live hook proof, so the conditional Codex fix line remains deferred" } ] }, { @@ -2024,15 +2027,16 @@ "dependency": null, "doneWhen": { "state": "closed-completed", "release": null }, "mapping": "mapped", - "kitImpact": { "refs": ["decision 7", "lane C: live-lock busy rule"], "files": ["src/lib/aqe-readiness.mjs"] }, - "adjustment": "Close #240 when a released agentic-qe no longer falls through to create after a live RVF lock (agentic-qe#574) and ak removes the temporary busy rule. agentic-qe#719 (partial fix: rethrows LockHeld) was released in 3.14.4 on 2026-09-27; whether that release alone clears FsyncFailed under a live lock is not yet verified.", + "kitImpact": { "refs": ["decision 7", "lane C: live-lock busy rule"], "files": ["docs/troubleshooting.md"] }, + "adjustment": "After the final main PR merges, the controller can close #240 with the released 3.14.4 native macOS/Linux live-owner evidence and ak's exact exception retirement for agentic-qe#574. The old 3.14.3 FsyncFailed sequence now fails closed; ordinary LockHeld fallback remains busy. No native Windows AQE conformance or upstream issue closure is claimed.", "status": "watching", "constraintIds": [], "tracks": ["proffesor-for-testing/agentic-qe#574", "proffesor-for-testing/agentic-qe#719"], "history": [ { "date": "2026-09-26", "event": "filed" }, { "date": "2026-09-26", "event": "registered" }, - { "date": "2026-09-27", "event": "commented", "note": "moved the upstream part of this tracker to the watch registry; agentic-qe 3.14.4 carries PR 719, not yet verified to stop FsyncFailed under a live lock (issuecomment-5858567140)" } + { "date": "2026-09-27", "event": "commented", "note": "moved the upstream part of this tracker to the watch registry; agentic-qe 3.14.4 carries PR 719, not yet verified to stop FsyncFailed under a live lock (issuecomment-5858567140)" }, + { "date": "2026-09-29", "event": "reviewed", "note": "released 3.14.4 passed native macOS and Linux live-owner conformance and ak retired the exact exception; issue closure waits for final main PR" } ] }, { @@ -2129,10 +2133,10 @@ "kind": "issue", "title": "`agentic-qe mcp` double-spawns the server (stdio:'inherit') and drops stdin → premature shutdown", "dependency": "agentic-qe", - "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, + "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": "3.14.4" } }, "mapping": "mapped", "kitImpact": { "refs": [], "files": ["docs/host-support.md"] }, - "adjustment": "Once a released agentic-qe mcp stops double-spawning the server, drop the host-support.md risk link.", + "adjustment": "The 3.14.4 MCP entry runs in-process; host-support.md records adoption from that version. Keep watching the still-open upstream issue until its closed-completed condition is met.", "status": "watching", "constraintIds": [], "history": [ @@ -2147,10 +2151,10 @@ "kind": "issue", "title": "init: platform installs are additive, not exclusive — always installs Claude Code surface; need per-platform/only mode", "dependency": "agentic-qe", - "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, + "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": "3.14.4" } }, "mapping": "mapped", "kitImpact": { "refs": [], "files": ["docs/host-support.md"] }, - "adjustment": "Once a released agentic-qe init offers an exclusive per-platform mode, drop the host-support.md risk link.", + "adjustment": "The 3.14.4 aqe init --no-claude option supports exclusive platform initialization; host-support.md records adoption from that version. Keep watching the still-open upstream issue until its closed-completed condition is met.", "status": "watching", "constraintIds": [], "history": [ @@ -2168,7 +2172,7 @@ "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, "mapping": "mapped", "kitImpact": { "refs": [], "files": ["docs/host-support.md"] }, - "adjustment": "Once a released agentic-qe fixes the listed MCP tool bugs, drop the host-support.md risk link.", + "adjustment": "Keep the host-support.md risk link limited to the remaining 3.14.4 GOAP maxSteps/world-state, test-generation quality, and coherence recommendation-text gaps. memory_delete and cross-phase stats were fixed; goap_execute was not re-exercised.", "status": "watching", "constraintIds": [], "history": [ @@ -2702,8 +2706,8 @@ "dependency": "agentic-qe", "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, "mapping": "mapped", - "kitImpact": {"refs": ["ak runs aqe init --with-codex, not platform setup", "the codex-mcp status row's manual fix names aqe platform setup codex (src/commands/status/sections/codex-mcp.mjs)"], "files": []}, - "adjustment": "none: ak does not run platform setup; once fixed, the codex-mcp status row's manual fix (aqe platform setup codex --overwrite --with-ruflo) works as written.", + "kitImpact": {"refs": ["ak runs aqe init --with-codex, not platform setup", "the codex-mcp status row formerly named the broken platform setup command (src/commands/status/sections/codex-mcp.mjs)"], "files": []}, + "adjustment": "The codex-mcp status row now points to AQE's supported aqe init --auto --with-codex --codex-guidance compact flow and notes the separate 3.14.4 Codex asset gap (agentic-qe#755). ak does not run platform setup or write AQE's Codex registration.", "status": "fixed-unreleased", "constraintIds": [], "history": [ @@ -2749,6 +2753,24 @@ { "date": "2026-09-27", "event": "registered" }, { "date": "2026-09-28", "event": "closed", "note": "completed by PR 766; no release contains it yet" } ] + }, + { + "id": "proffesor-for-testing/agentic-qe#778", + "url": "https://github.com/proffesor-for-testing/agentic-qe/issues/778", + "relation": "filed", + "kind": "issue", + "title": "fix: repeated aqe init changes generated settings on an unchanged project", + "dependency": "agentic-qe", + "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, + "mapping": "mapped", + "kitImpact": { "refs": ["ak setup invokes aqe init and can inherit its repeat-run settings churn"], "files": [] }, + "adjustment": "AQE 3.14.4 repeated init rewrites .claude/settings.json, changes domain and learning defaults on the second run, and creates a backup; CLAUDE.md and AGENTS.md hashes and mtimes stayed unchanged. Keep setup convergence claims bounded until a released fix is verified; 3.14.5 behavior has not been established.", + "status": "watching", + "constraintIds": [], + "history": [ + { "date": "2026-09-29", "event": "filed", "note": "approved exact draft posted; remote title and body match receipt" }, + { "date": "2026-09-29", "event": "registered", "note": "3.14.4 disposable project repro: settings changed on all three runs; guidance hashes and mtimes did not" } + ] } ] } diff --git a/src/lib/hook-remediation/engine.mjs b/src/lib/hook-remediation/engine.mjs index 767b449e..2bbfa475 100644 --- a/src/lib/hook-remediation/engine.mjs +++ b/src/lib/hook-remediation/engine.mjs @@ -1,5 +1,6 @@ import fs from 'node:fs'; import path from 'node:path'; +import { sameFileId } from '../file-identity.mjs'; import { assertHookHealingPlanIntegrity, buildHookHealingPlan, @@ -27,8 +28,8 @@ function sameOwnerAndParent(snapshot, expected) { && snapshot.gid === (expected.gid ?? snapshot.gid) && snapshot.specialMode === (expected.specialMode ?? 0) && snapshot.parent.realPath === (expected.parent?.realPath ?? snapshot.parent.realPath) - && snapshot.parent.dev === (expected.parent?.dev ?? snapshot.parent.dev) - && snapshot.parent.ino === (expected.parent?.ino ?? snapshot.parent.ino); + && sameFileId(snapshot.parent.dev, expected.parent?.dev ?? snapshot.parent.dev) + && sameFileId(snapshot.parent.ino, expected.parent?.ino ?? snapshot.parent.ino); } function preflight(plan, actionIds, expectedPlanDigest, options) { diff --git a/src/lib/hook-remediation/fs-port.mjs b/src/lib/hook-remediation/fs-port.mjs index 90a6ad9f..42a9ae5c 100644 --- a/src/lib/hook-remediation/fs-port.mjs +++ b/src/lib/hook-remediation/fs-port.mjs @@ -3,6 +3,7 @@ import path from 'node:path'; import { randomBytes } from 'node:crypto'; import { MAX_AUDIT_SOURCE_BYTES, sha256 } from '../hook-audit/common.mjs'; +import { fileId, sameFileId, statMtimeMs } from '../file-identity.mjs'; export const MAX_HOOK_TARGET_BYTES = MAX_AUDIT_SOURCE_BYTES; @@ -44,29 +45,27 @@ export function inspectHookTarget(file, containmentRoot, { const realFile = fsImpl.realpathSync(file); if (!contained(realRoot, realFile, platform)) throw new Error('target escapes its containment root'); descriptor = fsImpl.openSync(file, fs.constants.O_RDONLY | (fs.constants.O_NOFOLLOW ?? 0)); - const opened = fsImpl.fstatSync(descriptor); + const opened = fsImpl.fstatSync(descriptor, { bigint: true }); if (!opened.isFile() || opened.size > maxBytes) throw new Error('opened target is not a bounded regular file'); - // BigInt identity: Windows file IDs can exceed 2^53; `opened` stays Number for the image. - const openedId = fsImpl.fstatSync(descriptor, { bigint: true }); - if (openedId.dev !== stat.dev || openedId.ino !== stat.ino) { + if (opened.dev !== stat.dev || opened.ino !== stat.ino) { throw new Error('target identity changed between inspection and open'); } const reopened = fsImpl.realpathSync(file); if (reopened !== realFile || !contained(realRoot, reopened, platform)) { throw new Error('target path changed between inspection and open'); } - const bytes = readDescriptor(fsImpl, descriptor, opened.size); + const bytes = readDescriptor(fsImpl, descriptor, Number(opened.size)); const parent = path.dirname(realFile); - const parentStat = fsImpl.statSync(parent); + const parentStat = fsImpl.statSync(parent, { bigint: true }); return { file: path.resolve(file), containmentRoot: path.resolve(containmentRoot), realFile, realRoot, bytes, sha256: sha256(bytes), size: bytes.length, - mode: platform === 'win32' ? null : opened.mode & 0o777, - modeSupported: platform !== 'win32', mtimeMs: opened.mtimeMs, - uid: typeof opened.uid === 'number' ? opened.uid : null, - gid: typeof opened.gid === 'number' ? opened.gid : null, - specialMode: platform === 'win32' ? 0 : opened.mode & 0o7000, - parent: { realPath: parent, dev: parentStat.dev, ino: parentStat.ino }, + mode: platform === 'win32' ? null : Number(opened.mode & 0o777n), + modeSupported: platform !== 'win32', mtimeMs: statMtimeMs(opened), + uid: typeof opened.uid === 'bigint' ? Number(opened.uid) : null, + gid: typeof opened.gid === 'bigint' ? Number(opened.gid) : null, + specialMode: platform === 'win32' ? 0 : Number(opened.mode & 0o7000n), + parent: { realPath: parent, dev: fileId(parentStat.dev), ino: fileId(parentStat.ino) }, }; } finally { if (descriptor !== undefined) fsImpl.closeSync(descriptor); @@ -134,9 +133,10 @@ export function atomicReplaceHookTarget(snapshot, bytes, desiredMode = snapshot. || current.specialMode !== snapshot.specialMode) { throw new Error(`target changed immediately before replacement: ${snapshot.file}`); } - const currentParent = fsImpl.statSync(path.dirname(current.realFile)); + const currentParent = fsImpl.statSync(path.dirname(current.realFile), { bigint: true }); if (current.parent.realPath !== snapshot.parent.realPath - || currentParent.dev !== snapshot.parent.dev || currentParent.ino !== snapshot.parent.ino) { + || !sameFileId(currentParent.dev, snapshot.parent.dev) + || !sameFileId(currentParent.ino, snapshot.parent.ino)) { throw new Error(`target parent changed immediately before replacement: ${snapshot.file}`); } const suffix = randomBytes(12).toString('hex'); diff --git a/src/lib/hook-remediation/store.mjs b/src/lib/hook-remediation/store.mjs index 32a91cf7..d4214eb4 100644 --- a/src/lib/hook-remediation/store.mjs +++ b/src/lib/hook-remediation/store.mjs @@ -3,6 +3,7 @@ import path from 'node:path'; import { randomBytes } from 'node:crypto'; import { MAX_AUDIT_SOURCE_BYTES, sha256, stableJson } from '../hook-audit/common.mjs'; +import { fileId } from '../file-identity.mjs'; export const HOOK_HEAL_RECEIPT_SCHEMA = 'hook-heal-receipt/v1'; const RECEIPT_ID = /^tx-[0-9TZ.-]+-[a-f0-9]{16}$/; @@ -202,7 +203,7 @@ function validImage(image, { preimage = false } = {}) { && (image.gid === null || Number.isInteger(image.gid)) && Number.isInteger(image.specialMode) && image.specialMode >= 0 && image.specialMode <= 0o7000 && path.isAbsolute(image.parent?.realPath ?? '') - && Number.isInteger(image.parent?.dev) && Number.isInteger(image.parent?.ino) + && fileId(image.parent?.dev) !== null && fileId(image.parent?.ino) !== null )); } diff --git a/src/lib/host-alignment.mjs b/src/lib/host-alignment.mjs index ab9867c9..b1757f5c 100644 --- a/src/lib/host-alignment.mjs +++ b/src/lib/host-alignment.mjs @@ -4,6 +4,7 @@ import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { createHash, randomUUID } from 'node:crypto'; +import { fileId } from './file-identity.mjs'; import { writeFileWithBackup } from './file-write.mjs'; import { inspectCodexTomlStructure, isTomlTableLine } from './codex-toml-safety.mjs'; import { enabledPluginRefs } from './codex-plugins.mjs'; @@ -58,13 +59,13 @@ function uniqueJson(source) { } function readSource(file) { - const stat = fs.lstatSync(file); + const stat = fs.lstatSync(file, { bigint: true }); if (!stat.isFile() || stat.isSymbolicLink() || stat.size > 2 * 1024 * 1024) throw new Error('configuration is not a bounded regular file'); const bytes = fs.readFileSync(file); const source = bytes.toString('utf8'); if (!bytes.equals(Buffer.from(source))) throw new Error('configuration is not UTF-8'); - return { file, source, digest: hash(bytes), mode: stat.mode & 0o777, - identity: { real: fs.realpathSync(file), device: stat.dev, inode: stat.ino, mode: stat.mode } }; + return { file, source, digest: hash(bytes), mode: Number(stat.mode & 0o777n), + identity: { real: fs.realpathSync(file), device: fileId(stat.dev), inode: fileId(stat.ino), mode: Number(stat.mode) } }; } function safeJsonTransport(entry) { diff --git a/src/lib/host-health-evidence.mjs b/src/lib/host-health-evidence.mjs index 600b17ed..2d2cd983 100644 --- a/src/lib/host-health-evidence.mjs +++ b/src/lib/host-health-evidence.mjs @@ -4,6 +4,7 @@ import fs from 'node:fs'; import path from 'node:path'; import { createHmac, randomBytes } from 'node:crypto'; import { hostHealthInputPaths } from './paths.mjs'; +import { fileId, statMtimeMs } from './file-identity.mjs'; export function createHostHealthSnapshot({ secret = randomBytes(32), env = process.env, inputPaths = hostHealthInputPaths } = {}) { return ({ cwd, cfg }) => { @@ -35,9 +36,9 @@ export function createHostHealthSnapshot({ secret = randomBytes(32), env = proce for (const dir of (env.PATH ?? '').split(path.delimiter).slice(0, 256)) { const file = path.resolve(dir, host + (process.platform === 'win32' ? '.cmd' : '')); try { - const st = fs.statSync(file); + const st = fs.statSync(file, { bigint: true }); if (!st.isFile()) continue; - hash.update(JSON.stringify([host, fs.realpathSync(file), st.size, st.mtimeMs, st.ino])); + hash.update(JSON.stringify([host, fs.realpathSync(file), Number(st.size), statMtimeMs(st), fileId(st.dev), fileId(st.ino)])); break; } catch { /* next PATH entry */ } } diff --git a/src/lib/live-check-evidence.mjs b/src/lib/live-check-evidence.mjs index 3ec1448f..c3143f03 100644 --- a/src/lib/live-check-evidence.mjs +++ b/src/lib/live-check-evidence.mjs @@ -30,8 +30,10 @@ import { warn } from './output.mjs'; // The checks whose results are remembered: the quick, free live checks. One // list, owned by the refresh vocabulary (a constants-only import). import { LIVE_CHECK_IDS } from './refresh.mjs'; +import { installedRoutingVersion } from './ruflo-memory-contract.mjs'; export { LIVE_CHECK_IDS }; +export const RECORDED_CHECK_IDS = Object.freeze([...LIVE_CHECK_IDS, 'memory-routes']); const STATUSES = new Set(['passed', 'failed', 'inconclusive']); /** The sources a record is written with, and the ones it may still be read with. */ const WRITE_SOURCES = new Set(['sync', 'status-refresh-live']); @@ -45,7 +47,7 @@ const REASON_MAX = 200; export const liveCheckDir = () => path.join(paths.evidenceDir(), 'live-check'); function assertKnownId(id) { - if (!LIVE_CHECK_IDS.includes(id)) throw new TypeError(`unknown live check id: ${String(id).slice(0, 40)}`); + if (!RECORDED_CHECK_IDS.includes(id)) throw new TypeError(`unknown live check id: ${String(id).slice(0, 40)}`); } /** Printable, single-line, bounded. Stored reasons come from the kit's own @@ -141,18 +143,22 @@ const INPUTS = { security: () => ({ ruflo: rufloVersion() }), 'deja-vu': ({ cfg }) => ({ dejaVu: cfg.integrations?.tools?.dejaVu ?? null }), memory: () => ({ ruflo: rufloVersion() }), + 'memory-routes': ({ routingVersion, platform }) => ({ + routingVersion: routingVersion === undefined ? installedRoutingVersion() : routingVersion, + platform: platform ?? process.platform, + }), }; /** * The inputs key a check ran against. Writers and readers MUST both call this * so the same configuration yields the same key. Never throws. * @param {string} id - * @param {{cfg?:any,env?:NodeJS.ProcessEnv,cwd?:string}} [facts] + * @param {{cfg?:any,env?:NodeJS.ProcessEnv,cwd?:string,routingVersion?:string|null,platform?:string}} [facts] */ -export function liveCheckInputsKey(id, { cfg = {}, env = process.env, cwd = process.cwd() } = {}) { +export function liveCheckInputsKey(id, { cfg = {}, env = process.env, cwd = process.cwd(), routingVersion, platform } = {}) { assertKnownId(id); let parts; - try { parts = INPUTS[id]({ cfg: cfg ?? {}, env, cwd }); } catch { parts = { unavailable: true }; } + try { parts = INPUTS[id]({ cfg: cfg ?? {}, env, cwd, routingVersion, platform }); } catch { parts = { unavailable: true }; } return digest({ id, ...parts }); } @@ -181,12 +187,12 @@ export function embeddingProbeOutcome(live) { * when it could not be remembered. Returns whether it was recorded. * @param {string} id * @param {{status:string,reason?:string|null}|null} outcome - * @param {{source:string,cfg?:any,env?:NodeJS.ProcessEnv,cwd?:string,now?:number}} context + * @param {{source:string,cfg?:any,env?:NodeJS.ProcessEnv,cwd?:string,now?:number,inputsKey?:string}} context */ -export function rememberLiveCheck(id, outcome, { source, cfg, env, cwd, now } = /** @type {any} */ ({})) { +export function rememberLiveCheck(id, outcome, { source, cfg, env, cwd, now, inputsKey } = /** @type {any} */ ({})) { if (!outcome) return false; const recorded = recordLiveCheck({ id, status: outcome.status, reason: outcome.reason ?? null, source, - inputsKey: liveCheckInputsKey(id, { cfg, env, cwd }) }, { now }); + inputsKey: inputsKey ?? liveCheckInputsKey(id, { cfg, env, cwd }) }, { now }); if (!recorded) warn(`${id}: could not remember this live check result; ak status will not show it`); return recorded; } diff --git a/src/lib/live-checks.mjs b/src/lib/live-checks.mjs index 2aaab40c..9553008a 100644 --- a/src/lib/live-checks.mjs +++ b/src/lib/live-checks.mjs @@ -31,7 +31,7 @@ import { runHarvest } from './harvest.mjs'; import { runLifecycle } from './adapters/lifecycle.mjs'; import { companionLifecycleFor } from './adapters/companion-lifecycle-registry.mjs'; import { ok, warn, fail, info, heading, captureOutput } from './output.mjs'; -import { rememberLiveCheck, embeddingProbeOutcome } from './live-check-evidence.mjs'; +import { rememberLiveCheck, liveCheckInputsKey, embeddingProbeOutcome } from './live-check-evidence.mjs'; import { LIVE_CHECK_IDS, SLOW_PROOF_IDS } from './refresh.mjs'; export { LIVE_CHECK_IDS, SLOW_PROOF_IDS }; @@ -107,9 +107,11 @@ export async function observeProjectMemoryRoutes(tmp, env, namespace, deps) { try { const observation = await probeProjectMemoryRoutes(tmp, env, namespace, deps); for (const { level, message } of describeMemoryRoutes(observation)) (level === 'ok' ? ok : warn)(message); + return observation; } catch (e) { // An observation problem must never turn a working CLI proof into a failure. warn(`cross-interface routing not observed: ${e.message}`); + return null; } } @@ -132,11 +134,12 @@ async function purgeProofNamespace(tmp, env, namespace, key, runner = runCmd) { /** The memory round trip in an isolated folder under `tmpRoot`, removed * whatever happens. `observeRoutes: false` keeps the quick `memory` check to * the CLI round trip: the route observation (the `memory-routes` proof) - * starts a real MCP server and can only add warnings, which a live-check - * record does not carry. - * @param {{ observeRoutes?: boolean, tmpRoot?: string, runner?: typeof runCmd, haveCmd?: typeof have }} [options] */ + * starts a real MCP server. Its observation has separate evidence, while + * the CLI round-trip evidence remains under `memory`. + * @param {{ observeRoutes?: boolean, routeVerdict?: boolean, onCliOutcome?: (outcome:{status:string,reason:null})=>void, + * tmpRoot?: string, runner?: typeof runCmd, haveCmd?: typeof have }} [options] */ export async function verifyMemory({ - observeRoutes = true, tmpRoot = os.tmpdir(), runner = runCmd, haveCmd = have, + observeRoutes = true, routeVerdict = false, onCliOutcome, tmpRoot = os.tmpdir(), runner = runCmd, haveCmd = have, } = {}) { heading('memory — store, retrieve, locate the on-disk row, purge, and observe CLI/MCP routing in an isolated dir'); if (!(await haveCmd('ruflo'))) { fail('ruflo CLI not installed — cannot prove project memory'); return false; } @@ -182,17 +185,29 @@ export async function verifyMemory({ purged = await purgeProofNamespace(tmp, env, namespace, key, runner); if (!purged) { fail('isolated namespace purge did not remove the proof row'); return false; } ok('isolated proof namespace purged'); - if (observeRoutes) await observeProjectMemoryRoutes(tmp, env, namespace); + onCliOutcome?.({ status: 'passed', reason: null }); + if (observeRoutes) { + const route = /** @type {{status?:string,cliToMcp?:string,mcpToCli?:string}|null} */ + (await observeProjectMemoryRoutes(tmp, env, namespace)); + if (routeVerdict) return route?.status === 'observed' && + ['visible', 'not-visible'].includes(route.cliToMcp) && + ['visible', 'not-visible'].includes(route.mcpToCli) + ? { status: 'passed', reason: null } + : { status: 'inconclusive', reason: 'cross-interface routing not observed completely' }; + } return true; } catch (e) { fail(`memory proof error: ${e.message}`); return false; } finally { - if (stored && !purged) { - await runner('ruflo', ['memory', 'purge', '--namespace', namespace, '--force'], - { cwd: tmp, env, timeout: 120_000 }); + try { + if (stored && !purged) { + await runner('ruflo', ['memory', 'purge', '--namespace', namespace, '--force'], + { cwd: tmp, env, timeout: 120_000 }); + } + } finally { + fs.rmSync(tmp, { recursive: true, force: true }); } - fs.rmSync(tmp, { recursive: true, force: true }); } } @@ -451,10 +466,10 @@ async function verifyProjectProviders(root, cfg, { runner, haveCmd }) { * bridge derives agentdb-memory.db from (CLAUDE_FLOW_MEMORY_PATH), and * AGENTDB_PATH. An inherited value of any of them would otherwise receive the * proof rows. Nothing is seeded: the proof is Ruflo's own verbs succeeding. */ -export async function verifyHarvest({ runner = runCmd, haveCmd = have } = {}) { +export async function verifyHarvest({ tmpRoot = os.tmpdir(), runner = runCmd, haveCmd = have } = {}) { heading('harvest — record an outcome and distill, in an isolated store'); if (!(await haveCmd('ruflo'))) { fail('ruflo CLI not installed — cannot prove the harvest write path'); return false; } - const tmp = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'agentic-kit-harvest-'))); + const tmp = fs.realpathSync(fs.mkdtempSync(path.join(tmpRoot, 'agentic-kit-harvest-'))); const swarm = path.join(tmp, '.swarm'); // Pinned explicitly: a temporary folder inside a Git checkout would make the // derived project root the enclosing repository. @@ -577,7 +592,7 @@ export async function verifyDejaVu({ const enabled = cfg?.integrations?.tools?.dejaVu?.enabled === true; if (!dejaVuProofApplies(cfg)) { warn('deja-vu disabled and unowned — skipped'); - return true; + return { status: 'skipped', reason: 'disabled and unowned' }; } if (!adapter) { fail('deja-vu lifecycle adapter unavailable'); @@ -637,9 +652,10 @@ const CHECKS = Object.freeze([ // The full AQE proof remembers only its live embedding request, itself, and // only for a backend the kit manages. { ...slow('aqe'), run: ({ cfg, cwd, onEvidence }) => verifyAqe({ cfg, cwd, onEvidence }) }, - // The memory round trip plus the CLI/MCP route observation; its verdict is - // the memory check's. - { ...slow('memory-routes', 'memory'), run: () => verifyMemory({ observeRoutes: true }) }, + // The memory round trip plus the CLI/MCP route observation has distinct + // evidence; a CLI-only pass cannot establish routing. + { ...slow('memory-routes', 'memory-routes'), run: ({ onCliOutcome }) => + verifyMemory({ observeRoutes: true, routeVerdict: true, onCliOutcome }) }, ].map((check) => Object.freeze(check))); /** @@ -673,7 +689,9 @@ const duration = (ms) => (ms < 1000 ? `${ms} ms` : `${Math.round(ms / 1000)} s`) async function runOneLiveCheck(check, ctx, { timeoutMs, graceMs }) { const controller = new AbortController(); const started = Date.now(); - const work = captureOutput(() => withAbortSignal(controller.signal, () => check.run(ctx))) + let cliOutcome = null; + const work = captureOutput(() => withAbortSignal(controller.signal, + () => check.run({ ...ctx, onCliOutcome: (outcome) => { cliOutcome = outcome; } }))) .then(({ result, entries }) => ({ outcome: checkOutcome(result, entries, check.id), entries }), () => ({ outcome: { status: 'inconclusive', reason: 'the check could not run' }, entries: [] })); const deadline = sleep(timeoutMs); @@ -687,7 +705,8 @@ async function runOneLiveCheck(check, ctx, { timeoutMs, graceMs }) { settled = { outcome: { status: 'inconclusive', reason: `no result within ${duration(timeoutMs)}` }, entries: late?.entries ?? [] }; } const { outcome, entries } = settled; - return { id: check.id, status: outcome.status, reason: outcome.reason ?? null, elapsedMs: Date.now() - started, entries }; + return { id: check.id, status: outcome.status, reason: outcome.reason ?? null, + elapsedMs: Date.now() - started, entries, cliOutcome }; } /** @@ -706,15 +725,28 @@ export async function runLiveChecks({ cfg = loadKitConfig(), cwd = process.cwd(), only = [], checks = liveChecksFor(cfg, only), timeoutMs, graceMs = LIVE_CHECK_GRACE_MS, source = 'status-refresh-live', } = {}) { - const remember = (id, outcome) => rememberLiveCheck(id, outcome, { source, cfg, cwd }); + const remember = (id, outcome, inputsKey) => rememberLiveCheck(id, outcome, { source, cfg, cwd, inputsKey }); const ctx = { cfg, cwd, onEvidence: remember }; + // Capture the installed implementation before any proof starts. A package + // upgrade while the slow check runs cannot turn an old observation into a + // pass for the newly installed CLI. + const routeKeys = checks.map((check) => check.id === 'memory-routes' + ? liveCheckInputsKey('memory-routes', { cfg, cwd }) : null); const results = await Promise.all(checks.map(async (check) => ({ ...(await runOneLiveCheck(check, ctx, { timeoutMs: timeoutMs ?? check.timeoutMs ?? QUICK_TIMEOUT_MS, graceMs })), applies: check.applies?.(cfg ?? {}) ?? true, }))); checks.forEach((check, i) => { const evidenceId = check.evidenceId === undefined ? check.id : check.evidenceId; - if (evidenceId && results[i].applies) remember(evidenceId, results[i]); + if (!results[i].applies || results[i].status === 'skipped') return; + if (check.id === 'memory-routes') { + remember('memory', results[i].cliOutcome ?? results[i]); + if (routeKeys[i] !== liveCheckInputsKey('memory-routes', { cfg, cwd })) { + results[i].status = 'inconclusive'; + results[i].reason = 'installed routing implementation changed during the check'; + } + remember('memory-routes', results[i], routeKeys[i]); + } else if (evidenceId) remember(evidenceId, results[i]); }); - return results; + return results.map(({ cliOutcome: _cliOutcome, ...result }) => result); } diff --git a/src/lib/live/jsonl-tailer.mjs b/src/lib/live/jsonl-tailer.mjs index 8a3c3663..610abc44 100644 --- a/src/lib/live/jsonl-tailer.mjs +++ b/src/lib/live/jsonl-tailer.mjs @@ -1,5 +1,6 @@ import fs from 'node:fs'; import { StringDecoder } from 'node:string_decoder'; +import { fileId } from '../file-identity.mjs'; const limit = (value, ceiling) => Number.isSafeInteger(value) && value > 0 ? Math.min(value, ceiling) : ceiling; @@ -70,7 +71,7 @@ export class JsonlTailer { reconcile() { if (!this.canRead(this.#file)) return; let stat; - try { stat = fs.statSync(this.#file); } catch (error) { + try { stat = fs.statSync(this.#file, { bigint: true }); } catch (error) { if (error.code === 'ENOENT') this.#absent(); else this.#unreadable(error); return; @@ -82,31 +83,32 @@ export class JsonlTailer { })); return; } - const identity = `${stat.dev}:${stat.ino}`; + const identity = `${fileId(stat.dev)}:${fileId(stat.ino)}`; + const size = Number(stat.size); if (this.#identity == null) { this.#identity = identity; // A file that appeared after tailing began holds only new records, so // neither a resume offset nor startAtEnd may skip any of it. if (this.#absentSeen) this.#offset = 0; - else if (this.startOffset != null) this.#offset = Math.min(this.startOffset, stat.size); - else if (this.startAtEnd) this.#offset = stat.size; - } else if (this.#identity !== identity || stat.size < this.#offset) { + else if (this.startOffset != null) this.#offset = Math.min(this.startOffset, size); + else if (this.startAtEnd) this.#offset = size; + } else if (this.#identity !== identity || size < this.#offset) { this.#identity = identity; this.#resetPosition(); } - if (stat.size <= this.#offset) { + if (size <= this.#offset) { // Nothing new to read. Prove readability anyway when it is not yet known // (or was lost), so a mode-000 file cannot pass as healthy by staying // the same size. if (this.#presence !== 'readable') this.#probeReadable(); - this.#coverage(stat.size); + this.#coverage(size); return; } try { const flags = fs.constants.O_RDONLY | (fs.constants.O_NOFOLLOW ?? 0); const fd = fs.openSync(this.#file, flags); try { - let remaining = Math.min(stat.size - this.#offset, this.maxReadBytes); + let remaining = Math.min(size - this.#offset, this.maxReadBytes); const bytes = Buffer.alloc(Math.min(remaining, this.maxChunkBytes)); while (remaining > 0) { const count = fs.readSync(fd, bytes, 0, Math.min(bytes.length, remaining), this.#offset); @@ -118,7 +120,7 @@ export class JsonlTailer { } finally { fs.closeSync(fd); } this.#presence = 'readable'; } catch (error) { this.#unreadable(error); } - this.#coverage(stat.size); + this.#coverage(size); } #resetPosition() { diff --git a/src/lib/live/transcript-streams.mjs b/src/lib/live/transcript-streams.mjs index 32dee5a4..32a41201 100644 --- a/src/lib/live/transcript-streams.mjs +++ b/src/lib/live/transcript-streams.mjs @@ -1,5 +1,6 @@ import fs from 'node:fs'; import path from 'node:path'; +import { fileId } from '../file-identity.mjs'; import { JsonlTailer } from './jsonl-tailer.mjs'; import { LiveReplayStream } from './replay-stream.mjs'; import { @@ -155,12 +156,13 @@ class TranscriptStream { this.#host = host; this.#sessionId = sessionId; this.#options = options; - const epoch = fs.statSync(file).ino; + const stat = fs.statSync(file, { bigint: true }); + const epoch = fileId(stat.ino); this.#stream = new LiveReplayStream({ capacity: options.replayCapacity, prefix: `tx-${host}-${sessionId}-${epoch}`, }); - const offset = fs.statSync(file).size; + const offset = Number(stat.size); const history = tailLines(file, offset, options.maxHistoryBytes, options.maxHistoryRecords); const candidates = []; for (const raw of history.lines) { diff --git a/src/lib/maintenance/discovery/history.mjs b/src/lib/maintenance/discovery/history.mjs index 0d2c0691..0b6f5aff 100644 --- a/src/lib/maintenance/discovery/history.mjs +++ b/src/lib/maintenance/discovery/history.mjs @@ -1,9 +1,9 @@ // ADR-0048 scan history — bounded, retained scan summaries // (docs/maintenance.md "Retention"). -// This store holds ONLY terminal scan summaries. It structurally cannot reach +// This store holds terminal scan summaries and paused continuation boundaries. It cannot reach // receipts, dispositions, or recipe acceptance records — those live in other // agents' stores — so `clearHistory` cannot violate MNT-PRV-008 by scope -// alone; the `isProtected` guard below is defense in depth for a summary that +// alone; the `hasOpenContinuation` guard below is defense in depth for a summary that // still names an open continuation. import fs from 'node:fs'; import path from 'node:path'; @@ -30,14 +30,14 @@ function pruneSummaries(summaries, retention, now) { const cutoff = now - retention.maxAgeDays * 86_400_000; const bySource = new Map(); for (const summary of summaries - .filter((entry) => Date.parse(entry.completedAt) >= cutoff) - .sort((a, b) => Date.parse(a.completedAt) - Date.parse(b.completedAt))) { + .filter((entry) => Date.parse(entry.recordedAt ?? entry.completedAt) >= cutoff) + .sort((a, b) => Date.parse(a.recordedAt ?? a.completedAt) - Date.parse(b.recordedAt ?? b.completedAt))) { const key = JSON.stringify([summary.environmentId, summary.sourceId]); const list = bySource.get(key) ?? []; list.push(summary); bySource.set(key, list.slice(-retention.maxSummaries)); } - return [...bySource.values()].flat().sort((a, b) => Date.parse(a.completedAt) - Date.parse(b.completedAt)); + return [...bySource.values()].flat().sort((a, b) => Date.parse(a.recordedAt ?? a.completedAt) - Date.parse(b.recordedAt ?? b.completedAt)); } /** @@ -47,7 +47,8 @@ function pruneSummaries(summaries, retention, now) { */ export function createScanHistoryStore(dir, { fsImpl = fs, now = Date.now, retention = SCAN_HISTORY_RETENTION, - hasOpenContinuation = (/** @type {string} */ _scanId) => false, + hasOpenContinuation = (scanId) => typeof scanId === 'string' && /^[A-Za-z0-9_-]{1,80}$/u.test(scanId) + && fsImpl.existsSync(path.join(dir, 'checkpoints', `${scanId}.json`)), } = {}) { const file = path.join(dir, 'scan-history.json'); let effectiveRetention = clampRetention(retention); @@ -86,7 +87,7 @@ export function createScanHistoryStore(dir, { return environmentId ? summaries.filter((entry) => entry.environmentId === environmentId) : summaries; } - /** Remove only terminal, non-protected summaries; refuses (never throws) to + /** Remove only non-protected summaries; refuses (never throws) to * touch a summary whose scanId still names an open continuation. * @param {{environmentId?: string}} [options] */ function clearHistory({ environmentId } = {}) { diff --git a/src/lib/maintenance/discovery/orchestrator.mjs b/src/lib/maintenance/discovery/orchestrator.mjs index 38aa917d..4ad5ce95 100644 --- a/src/lib/maintenance/discovery/orchestrator.mjs +++ b/src/lib/maintenance/discovery/orchestrator.mjs @@ -118,11 +118,18 @@ function toCoverageRecord(record) { }; } -function toSummary(record, now) { +function toSummary(record, now, state = record.scanState) { + const recordedAt = new Date(now()).toISOString(); return { scanId: record.scanId, sourceId: record.sourceId, environmentId: record.environmentId, - state: record.scanState, startedAt: record.createdAt, completedAt: new Date(now()).toISOString(), + state, startedAt: record.createdAt, + completedAt: state === 'paused' ? null : recordedAt, + ...(state === 'paused' ? { recordedAt } : {}), visited: record.visited, limitingReason: record.limitingReason, ceiling: record.ceiling ?? null, + ...(state === 'paused' ? { + completedPartitions: record.completedPartitions.length, + pendingPartitions: record.pendingPartitions.length, + } : {}), }; } @@ -180,7 +187,8 @@ function remainingEntryBudget(ceilingEntries, visited) { * remove: (scanId: string) => void, list: () => string[] }} CheckpointStore * @typedef {{ recordSummary: (summary: object) => any, * list: (options?: {environmentId?: string}) => Array<{sourceId: string, environmentId: string, - * state: string, completedAt: string, visited?: number, limitingReason?: string|null, + * state: string, completedAt: string|null, recordedAt?: string, visited?: number, + * completedPartitions?: number, pendingPartitions?: number, limitingReason?: string|null, * ceiling?: string|null}> }} HistoryStore * @typedef {{ write: (entry: object) => boolean, * current: () => Array<{sourceId: string, environmentId: string}> }} LastGoodStore @@ -446,6 +454,11 @@ export function createScanOrchestrator({ error.code = 'SOURCE_NOT_PAUSABLE'; throw error; } + if (!historyStore) throw new Error('scan history unavailable; cannot persist pause'); + // Commit the continuation before claiming that pause is durable. If + // either store rejects the write, leave the live state unchanged. + persistCheckpoint(record); + historyStore.recordSummary(toSummary(record, now, 'paused')); transition(record, 'paused'); return coverageFor(toCoverageRecord(record)); } @@ -486,8 +499,8 @@ export function createScanOrchestrator({ /** A configured source with no live record — never started, paused, or * stopped in this process (e.g. right after a restart, when `records` is - * empty again) — first restores its latest terminal history summary when - * that summary is itself a failure, or a stop at a limit other than the + * empty again) — first restores its latest history summary when + * that summary is a pause, a failure, or a stop at a limit other than the * user's own request (audit finding M1b): a real failure must read the * same on the Discovery panel and the Inventory banner both before and * after a restart, since both read this same `coverage()`. A user stop is @@ -501,11 +514,13 @@ export function createScanOrchestrator({ const record = records.get(source.sourceId); if (record) return coverageFor(toCoverageRecord(record)); const summary = summaryBySource?.get(`${source.environmentId}:${source.sourceId}`); - if (summary && (summary.state === 'failed' || (summary.state === 'stopped' && summary.limitingReason !== 'stopped-by-user'))) { + if (summary && (summary.state === 'paused' || summary.state === 'failed' || (summary.state === 'stopped' && summary.limitingReason !== 'stopped-by-user'))) { return coverageFor({ ...toCoverageRecord(initRecord(source)), scanState: summary.state, visited: summary.visited ?? 0, + completedPartitions: summary.completedPartitions ?? 0, + pendingPartitions: summary.pendingPartitions ?? 0, limitingReason: summary.limitingReason ?? null, ceiling: summary.ceiling ?? null, }); @@ -516,10 +531,10 @@ export function createScanOrchestrator({ return coverageFor(toCoverageRecord(initRecord(source))); } - /** Index the newest terminal summary per `(environmentId, sourceId)` from + /** Index the newest summary per `(environmentId, sourceId)` from * the history store, read once per `coverage()` call rather than once per * source (M1b). `historyStore.list()` is already ascending by - * `completedAt` with ties in recording order, so `>=` (not `>`) keeps the + * record timestamp with ties in recording order, so `>=` (not `>`) keeps the * most-recently-recorded summary on a same-millisecond tie — a later * successful scan must win over an earlier failure even when both * complete within the same clock tick. */ @@ -528,7 +543,7 @@ export function createScanOrchestrator({ for (const summary of historyStore?.list() ?? []) { const key = `${summary.environmentId}:${summary.sourceId}`; const existing = bySource.get(key); - if (!existing || Date.parse(summary.completedAt) >= Date.parse(existing.completedAt)) bySource.set(key, summary); + if (!existing || Date.parse(summary.recordedAt ?? summary.completedAt) >= Date.parse(existing.recordedAt ?? existing.completedAt)) bySource.set(key, summary); } return bySource; } diff --git a/src/lib/maintenance/discovery/partitions.mjs b/src/lib/maintenance/discovery/partitions.mjs index d2b249a3..6066d605 100644 --- a/src/lib/maintenance/discovery/partitions.mjs +++ b/src/lib/maintenance/discovery/partitions.mjs @@ -5,14 +5,15 @@ // partition is executed as one or more `observeWalkForest` calls. import fs from 'node:fs'; import path from 'node:path'; +import { fileId, sameFileId, statMtimeMs } from '../../file-identity.mjs'; const DEFAULT_MAX_FANOUT = 64; function stampFor(target, fsImpl) { try { - const stat = fsImpl.lstatSync(target); + const stat = fsImpl.lstatSync(target, { bigint: true }); return { - target, mtimeMs: stat.mtimeMs, ino: Number(stat.ino) || null, size: stat.isFile() ? stat.size : null, + target, mtimeMs: statMtimeMs(stat), ino: fileId(stat.ino), size: stat.isFile() ? Number(stat.size) : null, }; } catch (error) { return { target, mtimeMs: null, ino: null, size: null, reason: error?.code ?? 'io' }; @@ -110,6 +111,6 @@ export function mergeWalkResults(results) { export function partitionDrifted(partition, { fsImpl = fs } = {}) { return partition.sourceStamps.some((recorded) => { const current = stampFor(recorded.target, fsImpl); - return current.mtimeMs !== recorded.mtimeMs || current.ino !== recorded.ino || current.size !== recorded.size; + return current.mtimeMs !== recorded.mtimeMs || !sameFileId(current.ino, recorded.ino) || current.size !== recorded.size; }); } diff --git a/src/lib/mcp-probe.mjs b/src/lib/mcp-probe.mjs index dcdc4416..d7302370 100644 --- a/src/lib/mcp-probe.mjs +++ b/src/lib/mcp-probe.mjs @@ -2,6 +2,7 @@ // repository content, environment values, or stderr are returned in receipts. import { spawn } from 'node:child_process'; import { resolveShim, killProcessTree } from './exec.mjs'; +import { mergeWindowsEnv } from './windows-npm-shim.mjs'; /** @param {{command:string,args?:string[],cwd?:string,env?:NodeJS.ProcessEnv,timeoutMs?:number}} options */ export function probeMcp({ command, args = [], cwd, env = {}, timeoutMs = 30_000 }) { @@ -12,7 +13,7 @@ export function probeMcp({ command, args = [], cwd, env = {}, timeoutMs = 30_000 } return new Promise((resolve) => { const started = performance.now(); - const mergedEnv = { ...process.env, ...env }; + const mergedEnv = process.platform === 'win32' ? mergeWindowsEnv(process.env, env) : { ...process.env, ...env }; const invocation = resolveShim(command, args, { env: mergedEnv }); const child = spawn(invocation.command, invocation.args, { cwd, env: mergedEnv, shell: false, detached: process.platform !== 'win32', diff --git a/src/lib/mcp-tool-call.mjs b/src/lib/mcp-tool-call.mjs index 7b83678d..26154fb9 100644 --- a/src/lib/mcp-tool-call.mjs +++ b/src/lib/mcp-tool-call.mjs @@ -14,6 +14,7 @@ // shim's node process). import { spawn } from 'node:child_process'; import { resolveShim, killProcessTree } from './exec.mjs'; +import { mergeWindowsEnv } from './windows-npm-shim.mjs'; const MAX_OUTPUT_BYTES = 2 * 1024 * 1024; /** How long the killed server may take to exit before the call reports @@ -44,7 +45,7 @@ export async function callMcpTools({ }) { validate({ command, args, calls, timeoutMs, exitGraceMs }); return new Promise((resolve) => { - const merged = { ...process.env, ...env }; + const merged = process.platform === 'win32' ? mergeWindowsEnv(process.env, env) : { ...process.env, ...env }; const invocation = resolveShim(command, args, { env: merged }); const child = spawn(invocation.command, invocation.args, { cwd, env: merged, shell: false, detached: process.platform !== 'win32', diff --git a/src/lib/project-memory.mjs b/src/lib/project-memory.mjs index e42dbe3b..07856e85 100644 --- a/src/lib/project-memory.mjs +++ b/src/lib/project-memory.mjs @@ -143,24 +143,34 @@ export function removeMemoryProbe(root, namespace, key) { // aqe a .agentic-qe/ folder below the project root (AQE resolves a // relative AQE_MEMORY_PATH against the folder it runs in) // Bounded: at most `maxDirs` folders listed and `maxDepth` levels deep; dot -// folders (other checkouts under .claude/worktrees, .git) and node_modules are -// never walked, only a root dot folder's own markers are checked. A folder that +// tool homes (including other checkouts under .claude/worktrees), .git and +// node_modules are never walked. Ordinary dot folders are walked. A folder that // holds `.git` (nested repository, submodule, worktree inside the checkout) is // another repository: neither it nor anything below it is searched; it is // listed in `nestedRepositories`. const RUFLO_STORE_FILES = Object.freeze(['memory.db', 'agentdb-memory.db']); const ROOT_STRAYS = Object.freeze([['agentdb.db', 'agentdb-cli'], ['agentdb.rvf', 'agentdb-rvf'], ['ruvector.db', 'ruvector']]); +const SCAN_EXCLUDED_DIRS = new Set(['.git', '.swarm', '.agentic-qe', '.claude', '.codex', '.claude-flow', '.agents', '.harness', 'node_modules']); -const isDirectory = (file) => { try { return fs.lstatSync(file).isDirectory(); } catch { return false; } }; const isFile = (file) => { try { return fs.lstatSync(file).isFile(); } catch { return false; } }; const storeBytes = (file) => (fileBytes(file) ?? 0) + (fileBytes(`${file}-wal`) ?? 0); export function findStrayMemoryStores(root, { maxDepth = 4, maxDirs = 2000 } = {}) { - if (!isDirectory(root)) return { strays: [], complete: true, visited: 0, nestedRepositories: [] }; + let rootStat; + try { rootStat = fs.lstatSync(root); } catch (e) { + return { strays: [], complete: e.code === 'ENOENT', visited: 0, nestedRepositories: [] }; + } + if (!rootStat.isDirectory()) return { strays: [], complete: true, visited: 0, nestedRepositories: [] }; const found = new Map(); const nestedRepositories = []; let visited = 0; let complete = true; + const stat = (file) => { + try { return fs.lstatSync(file); } catch (e) { + if (e.code !== 'ENOENT') complete = false; + return null; + } + }; const add = (kind, file, sizeBytes) => { const relative = path.relative(root, file).split(path.sep).join('/'); found.set(relative, { kind, path: relative, file, sizeBytes }); @@ -170,14 +180,14 @@ export function findStrayMemoryStores(root, { maxDepth = 4, maxDirs = 2000 } = { visited += 1; try { return fs.readdirSync(dir, { withFileTypes: true }).sort((a, b) => a.name.localeCompare(b.name)); - } catch { return []; } + } catch { complete = false; return []; } }; const checkMarkers = (dir, { ruflo = true } = {}) => { - if (isDirectory(path.join(dir, '.agentic-qe'))) add('aqe', path.join(dir, '.agentic-qe'), null); + if (stat(path.join(dir, '.agentic-qe'))?.isDirectory()) add('aqe', path.join(dir, '.agentic-qe'), null); if (!ruflo) return; for (const name of RUFLO_STORE_FILES) { const file = path.join(dir, '.swarm', name); - if (isFile(file)) add('ruflo', file, storeBytes(file)); + if (stat(file)?.isFile()) add('ruflo', file, storeBytes(file)); } }; const walkSwarm = (dir, depth) => { @@ -192,7 +202,7 @@ export function findStrayMemoryStores(root, { maxDepth = 4, maxDirs = 2000 } = { // a worktree inside the checkout) is another repository: its stores are its // own, and ak's pin makes its `.agentic-qe` that repository's store (review M3). const otherRepository = (full) => { - try { fs.lstatSync(path.join(full, '.git')); } catch { return false; } + if (!stat(path.join(full, '.git'))) return false; nestedRepositories.push(path.relative(root, full).split(path.sep).join('/')); return true; }; @@ -200,25 +210,26 @@ export function findStrayMemoryStores(root, { maxDepth = 4, maxDirs = 2000 } = { const entries = list(dir); for (const entry of entries ?? []) { if (!entry.isDirectory()) continue; + if (entry.name === 'node_modules') continue; const full = path.join(dir, entry.name); if (entry.name !== '.git' && otherRepository(full)) continue; - if (entry.name.startsWith('.')) { - // .swarm's own subtree is walked separately; only its AQE marker here. - if (depth === 0 && entry.name !== '.git') checkMarkers(full, { ruflo: entry.name !== '.swarm' }); - continue; - } - if (entry.name === 'node_modules') continue; - checkMarkers(full); + if (entry.name !== '.git') checkMarkers(full, { ruflo: entry.name !== '.swarm' }); + if (SCAN_EXCLUDED_DIRS.has(entry.name)) continue; if (depth + 1 < maxDepth) walk(full, depth + 1); + else { + // A marker at the depth boundary is visible, but descendants are not. + const children = list(full); + if (children?.some((child) => child.isDirectory() && !SCAN_EXCLUDED_DIRS.has(child.name))) complete = false; + } } }; for (const [name, kind] of ROOT_STRAYS) { const file = path.join(root, name); - if (isFile(file)) add(kind, file, storeBytes(file)); + if (stat(file)?.isFile()) add(kind, file, storeBytes(file)); } // .swarm first: it is small, and the most likely home of a stray Ruflo store. - if (isDirectory(path.join(root, '.swarm'))) walkSwarm(path.join(root, '.swarm'), 0); + if (stat(path.join(root, '.swarm'))?.isDirectory()) walkSwarm(path.join(root, '.swarm'), 0); walk(root, 0); const strays = [...found.values()].sort((a, b) => a.path.localeCompare(b.path)); return { strays, complete, visited, nestedRepositories: nestedRepositories.sort() }; diff --git a/src/lib/ruflo-daemon-config.mjs b/src/lib/ruflo-daemon-config.mjs index f517d557..81048959 100644 --- a/src/lib/ruflo-daemon-config.mjs +++ b/src/lib/ruflo-daemon-config.mjs @@ -100,10 +100,14 @@ function readDaemonConfig(file) { } } +const pathPresent = (candidate) => { + try { fs.lstatSync(candidate); return true; } catch (error) { return error?.code !== 'ENOENT'; } +}; + /** A desired key ak leaves to the user: the file is unreadable (`invalid`), * or it holds the user's own value `have` instead of `want`. * @typedef {{ key: string, want: number, have?: unknown }} HeldKey - * @typedef {{ invalid: boolean, entries: HeldKey[] } | null} HeldConfig */ + * @typedef {{ invalid: boolean, entries: HeldKey[], reason?: 'yaml-shadow'|'higher-priority-json'|'explicit-config' } | null} HeldConfig */ /** Plan the config.json edit: which owned keys to drop, which to set. */ function planConfig(current, owned, desired) { @@ -127,26 +131,64 @@ function planConfig(current, owned, desired) { return { next, nextOwned, removed, written, conflicts }; } -function reconcileConfig(root, receipt, desired, dryRun) { +/** @returns {{status: string, changed: boolean, activeChanged: boolean, held: HeldConfig}} */ +function reconcileConfig(root, receipt, desired, dryRun, env) { const file = path.join(root, DAEMON_CONFIG_RELATIVE); + // Never follow a user-controlled .claude-flow link (or special file) when + // reading, creating, replacing or removing config.json. + try { + const dir = fs.lstatSync(path.dirname(file)); + if (!dir.isDirectory() || dir.isSymbolicLink()) { + const entries = Object.entries(desired).map(([key, want]) => ({ key, want })); + return { status: 'user-managed', changed: false, activeChanged: false, held: entries.length ? { invalid: true, entries } : null }; + } + } catch (error) { + if (error?.code !== 'ENOENT') { + const entries = Object.entries(desired).map(([key, want]) => ({ key, want })); + return { status: 'user-managed', changed: false, activeChanged: false, held: entries.length ? { invalid: true, entries } : null }; + } + } + const higherPriority = pathPresent(path.join(root, 'claude-flow.config.json')); + // ConfigFileManager.findConfig checks the explicit path only after both + // JSON candidates. existsSync resolves a relative env path from process.cwd, + // exactly as Ruflo does; it does not resolve it against the project root. + const explicitConfig = !higherPriority && typeof env?.CLAUDE_FLOW_CONFIG === 'string' + && fs.existsSync(env.CLAUDE_FLOW_CONFIG); const owned = receipt.configKeys ?? {}; const current = readDaemonConfig(file); if (current.state === 'invalid') { const entries = Object.entries(desired).map(([key, want]) => ({ key, want })); - return { status: 'user-managed', changed: false, held: entries.length ? { invalid: true, entries } : null }; + return { status: 'user-managed', changed: false, activeChanged: false, held: entries.length ? { invalid: true, entries } : null }; } if (current.state === 'absent') { receipt.configKeys = {}; receipt.configCreated = false; - if (!Object.keys(desired).length) return { status: 'absent', changed: false, held: null }; + if (!Object.keys(desired).length) return { status: 'absent', changed: false, activeChanged: false, held: null }; + // Ruflo 3.48.0 chooses root JSON, then .claude-flow/config.json, then + // config.yaml/yml. Creating our JSON over YAML would hide all user YAML + // daemon values; under root JSON this file would have no effect at all. + const yaml = ['config.yaml', 'config.yml'].some((name) => pathPresent(path.join(root, '.claude-flow', name))); + if (higherPriority || explicitConfig || yaml) return { + status: 'user-managed', changed: false, activeChanged: false, + held: { + invalid: false, reason: higherPriority ? 'higher-priority-json' : explicitConfig ? 'explicit-config' : 'yaml-shadow', + entries: Object.entries(desired).map(([key, want]) => ({ key, want })), + }, + }; if (!dryRun) { writePrivateFileAtomic(file, `${JSON.stringify(desired, null, 2)}\n`); receipt.configKeys = { ...desired }; receipt.configCreated = true; } - return { status: 'written', changed: true, held: null }; + return { status: 'written', changed: true, activeChanged: true, held: null }; } - const plan = planConfig(current.value, owned, desired); + // A root JSON file wins even over existing .claude-flow/config.json. Keep + // receipted still-needed keys and clean obsolete ones, but add no new keys + // to a file the daemon does not read. + const effectiveDesired = higherPriority + ? Object.fromEntries(Object.entries(desired).filter(([key]) => Object.hasOwn(owned, key))) + : desired; + const plan = planConfig(current.value, owned, effectiveDesired); const changed = plan.written || plan.removed; const empty = Object.keys(plan.next).length === 0; if (!dryRun) { @@ -156,10 +198,13 @@ function reconcileConfig(root, receipt, desired, dryRun) { receipt.configCreated ??= false; if (changed && empty && receipt.configCreated === true) receipt.configCreated = false; } - const held = plan.conflicts.length ? { invalid: false, entries: plan.conflicts } : null; - if (plan.written) return { status: 'written', changed, held }; - if (plan.removed) return { status: 'removed', changed, held }; - return { status: held ? 'user-managed' : 'converged', changed: false, held }; + /** @type {HeldConfig} */ + const held = higherPriority && Object.keys(desired).length + ? { invalid: false, reason: 'higher-priority-json', entries: Object.entries(desired).map(([key, want]) => ({ key, want })) } + : plan.conflicts.length ? { invalid: false, entries: plan.conflicts } : null; + if (plan.written) return { status: 'written', changed, activeChanged: !higherPriority, held }; + if (plan.removed) return { status: 'removed', changed, activeChanged: !higherPriority, held }; + return { status: held ? 'user-managed' : 'converged', changed: false, activeChanged: false, held }; } function reconcileAutostart(root, receipt, wanted, dryRun) { @@ -196,25 +241,27 @@ const hasOwnership = (receipt) => Object.keys(receipt.configKeys ?? {}).length > * Converge one project's managed daemon settings. * @param {string} root the Ruflo project root * @param {{rufloVersion?: (string|null), platform?: string, receipts: Record, - * autoStart?: boolean, dryRun?: boolean, desired?: Record, runner?: unknown}} options + * autoStart?: boolean, dryRun?: boolean, desired?: Record, runner?: unknown, + * env?: NodeJS.ProcessEnv}} options * `autoStart` is kit.json rufloDaemon.autoStart !== false; `runner` is accepted and never used. - * @returns {{config: string, autostart: string, changed: boolean, held: HeldConfig}} + * @returns {{config: string, autostart: string, changed: boolean, configActiveChanged: boolean, held: HeldConfig}} * `held` names the desired keys ak cannot manage here (null when none). */ export function reconcileRufloDaemon(root, { rufloVersion = null, platform = process.platform, receipts, autoStart = true, dryRun = false, - desired = desiredDaemonKeys({ rufloVersion, platform }), + desired = desiredDaemonKeys({ rufloVersion, platform }), env = process.env, } = /** @type {any} */ ({})) { const key = path.resolve(root); const receipt = structuredClone(receipts[key] ?? {}); - const config = reconcileConfig(root, receipt, desired, dryRun); + const config = reconcileConfig(root, receipt, desired, dryRun, env); const autostart = reconcileAutostart(root, receipt, autoStart, dryRun); if (!dryRun) { if (hasOwnership(receipt)) receipts[key] = receipt; else delete receipts[key]; } return { - config: config.status, autostart: autostart.status, changed: config.changed || autostart.changed, held: config.held, + config: config.status, autostart: autostart.status, changed: config.changed || autostart.changed, + configActiveChanged: config.activeChanged, held: config.held, }; } @@ -254,16 +301,17 @@ export function daemonIntent(cfg) { * is left for Ruflo's start-on-use. Returns null outside a Ruflo project. * @param {string} cwd * @param {{cfg: any, rufloVersion?: (string|null), platform?: string, runner?: typeof run, - * alive?: (root: string) => boolean, dryRun?: boolean}} options + * alive?: (root: string) => boolean, dryRun?: boolean, env?: NodeJS.ProcessEnv}} options */ export async function applyRufloDaemon(cwd, { cfg, rufloVersion = null, platform = process.platform, runner = run, alive = projectDaemonAlive, dryRun = false, + env = process.env, }) { const root = rufloDaemonProjectRoot(cwd); if (!root) return null; const intent = daemonIntent(cfg); - const result = reconcileRufloDaemon(root, { rufloVersion, platform, receipts: intent.receipts, autoStart: intent.autoStart, dryRun }); - const configChanged = result.config === 'written' || result.config === 'removed'; + const result = reconcileRufloDaemon(root, { rufloVersion, platform, receipts: intent.receipts, autoStart: intent.autoStart, dryRun, env }); + const configChanged = result.configActiveChanged; // A config.json that keeps the floor from ak (unreadable, or the user's own // value) changed nothing a restart would pick up. const floorHeld = result.held?.entries.some((e) => e.key === MEMORY_FLOOR_KEY) ?? false; @@ -279,17 +327,17 @@ export async function applyRufloDaemon(cwd, { /** Status: the desired keys a user-managed config.json keeps from ak (a dry run). * @returns {HeldConfig} */ -export function daemonConfigHeld(root, { cfg, rufloVersion = null, platform = process.platform }) { +export function daemonConfigHeld(root, { cfg, rufloVersion = null, platform = process.platform, env = process.env }) { const intent = daemonIntent(structuredClone(cfg ?? {})); const receipts = structuredClone(intent.receipts); - return reconcileRufloDaemon(root, { rufloVersion, platform, receipts, autoStart: intent.autoStart, dryRun: true }).held; + return reconcileRufloDaemon(root, { rufloVersion, platform, receipts, autoStart: intent.autoStart, dryRun: true, env }).held; } /** Status: what a sync would change here, as phrases, or null when converged. */ -export function daemonDrift(root, { cfg, rufloVersion = null, platform = process.platform }) { +export function daemonDrift(root, { cfg, rufloVersion = null, platform = process.platform, env = process.env }) { const intent = daemonIntent(structuredClone(cfg ?? {})); const receipts = structuredClone(intent.receipts); - const r = reconcileRufloDaemon(root, { rufloVersion, platform, receipts, autoStart: intent.autoStart, dryRun: true }); + const r = reconcileRufloDaemon(root, { rufloVersion, platform, receipts, autoStart: intent.autoStart, dryRun: true, env }); if (!r.changed) return null; const parts = []; const desired = desiredDaemonKeys({ rufloVersion, platform }); diff --git a/src/lib/ruflo-memory.mjs b/src/lib/ruflo-memory.mjs index 358bdc9f..b7aa541e 100644 --- a/src/lib/ruflo-memory.mjs +++ b/src/lib/ruflo-memory.mjs @@ -36,9 +36,10 @@ import { componentById } from './ruflo-components/catalogue.mjs'; // are the same folder, as the filesystem treats them. const realOr = (p, file) => { try { return fs.realpathSync(file); } catch { return p.resolve(file); } }; const same = (p, a, b) => p.relative(a, b) === ''; +// Strict containment: equality cannot make a tool root a deeper temp boundary. const inside = (p, child, parent) => { const rel = p.relative(parent, child); - return rel === '' || (!rel.startsWith('..') && !p.isAbsolute(rel)); + return rel !== '' && !rel.startsWith('..') && !p.isAbsolute(rel); }; /** `~/…` for a folder under the home folder, else the absolute path. */ export function homeRelative(file, home = paths.home, p = path) { @@ -57,7 +58,8 @@ function unsuitableReason(dir, { home, env, platform, p }) { if (temps.some((temp) => same(p, dir, temp))) return 'a temporary folder'; const tool = paths.toolInternalDirs({ home, env, platform, p }).find((folder) => { const real = realOr(p, folder); - return inside(p, dir, real) && !temps.some((temp) => inside(p, temp, real) && inside(p, dir, temp)); + return (same(p, dir, real) || inside(p, dir, real)) + && !temps.some((temp) => inside(p, temp, real) && inside(p, dir, temp)); }); return tool ? `inside ${homeRelative(tool, home, p)}, a tool's own folder` : null; } @@ -75,14 +77,20 @@ export function rufloMemoryLocation(cwd = process.cwd(), { home = paths.home, env = process.env, platform = process.platform, p = path, } = {}) { const options = { home, env, platform, p }; - let reason = null; - for (const [kind, candidate] of [['project', paths.repoRoot(cwd, p)], ['folder', cwd]]) { - if (!candidate) continue; - const root = realOr(p, candidate); - const why = unsuitableReason(root, options); - if (!why) return { kind, root, dir: p.join(root, '.swarm'), db: paths.projectMemoryDb(root, p), reason: null }; - reason ??= why; + const repository = paths.repoRoot(cwd, p); + const projectRoot = repository && realOr(p, repository); + const projectReason = projectRoot && unsuitableReason(projectRoot, options); + if (projectRoot && !projectReason) { + return { kind: 'project', root: projectRoot, dir: p.join(projectRoot, '.swarm'), db: paths.projectMemoryDb(projectRoot, p), reason: null }; } + const folderRoot = realOr(p, cwd); + const folderReason = unsuitableReason(folderRoot, options); + if (!folderReason) { + return { kind: 'folder', root: folderRoot, dir: p.join(folderRoot, '.swarm'), db: paths.projectMemoryDb(folderRoot, p), reason: null }; + } + const reason = projectReason && !same(p, projectRoot, folderRoot) && projectReason !== folderReason + ? `${folderReason}, in a repository whose root is ${projectReason}` + : folderReason; const dir = paths.userMemoryDir(home, p); return { kind: 'user', root: dir, dir, db: p.join(dir, 'memory.db'), reason }; } diff --git a/src/lib/versions.mjs b/src/lib/versions.mjs index 7e541a9a..8617c98b 100644 --- a/src/lib/versions.mjs +++ b/src/lib/versions.mjs @@ -159,27 +159,26 @@ async function fetchSelfCandidate(tags, cachedBest, fetchLatest) { return { best, observed, answered }; } -/** The self record to save after a lookup, or null to save nothing. A live - * winner is an observation. When every lookup failed, the record stays and - * `last` is restamped, so the next lookup waits one TTL window; `observedAt` - * keeps when the recorded best was seen (a record written before it existed - * takes the previous `last`). A partial answer that leaves a cached candidate - * winning renews nothing. Neither does a failure when the recorded best is - * one this install cannot use (a `next` candidate on a stable install): such - * a record is never fresh, so a restamp would only rewrite kit.json on every - * call, and dropping the candidate would make an empty record look fresh and - * stop the lookup for a TTL window once the registry is back. */ -function selfRecord(cached, usable, { best, observed, answered }, now = Date.now()) { - if (observed) return { last: now, best, observedAt: now }; - if (answered || (cached?.best && !usable)) return null; - return { ...cached, last: now, observedAt: cached?.observedAt ?? cached?.last }; +/** `observedAt` describes the winning candidate; `last` and `lastTags` scope + * the latest completed check. `attempt` throttles a lookup that could not + * update that check. Neither stamp makes a cached winner newly observed. */ +function selfRecord(cached, usable, { best, observed, answered }, tags, now = Date.now()) { + if (observed) return { last: now, best, observedAt: now, lastTags: tags }; + if (answered || (cached?.best && !usable)) { + return { ...cached, attempt: { at: now, tags } }; + } + const { attempt: _priorAttempt, ...prior } = cached ?? {}; + return { ...prior, last: now, observedAt: cached?.observedAt ?? cached?.last, lastTags: tags }; } +const sameTags = (recorded, tags) => Array.isArray(recorded) && recorded.length === tags.length + && tags.every((tag, index) => recorded[index] === tag); + /** Look the kit up on its channels; with `record`, save what selfRecord keeps. * Returns the winning candidate. */ async function lookUpSelf(cfg, cached, { tags, cachedBest, fetchLatest, record }) { const candidate = await fetchSelfCandidate(tags, cachedBest, fetchLatest); - const entry = record ? selfRecord(cached, cachedBest, candidate) : null; + const entry = record ? selfRecord(cached, cachedBest, candidate, tags) : null; if (entry) { cfg.versionCheck = { ...cfg.versionCheck, self: entry }; try { saveKitConfig(cfg); } catch { /* read-only envs: next call re-fetches */ } @@ -191,8 +190,10 @@ async function lookUpSelf(cfg, cached, { tags, cachedBest, fetchLatest, record } * (pkgRoot). Prerelease installs also consult the `next` dist-tag — * prereleases publish there, so `latest` alone would never see them; the * higher of latest/next wins. Cached in kit.json alongside versionCheck. - * Failed lookups preserve eligible cached evidence (see selfRecord); `force` - * retries within the TTL. cacheOnly=true reports the recorded best with no + * Failed lookups preserve eligible cached evidence (see selfRecord); an + * `attempt` stamp limits partial/unusable retries to once per tag set and TTL. + * `lastTags` scopes completed checks. `force` retries within the TTL. + * cacheOnly=true reports the recorded best with no * network and no write (`ak sync --skip self`); record=false reports what the * lookup found without saving it (ADR-0063). * @param {{ pkgRoot?: string, force?: boolean, cacheOnly?: boolean, record?: boolean, @@ -208,8 +209,19 @@ export async function selfDrift({ pkgRoot, force = false, cacheOnly = false, rec const tags = installed?.includes('-') ? ['latest', 'next'] : ['latest']; const cachedBest = cached?.best && tags.includes(cached.best.tag) && isValidSemver(cached.best.version) ? cached.best : null; - const fresh = !force && cached?.last && Date.now() - cached.last < ttlMs - && (!cached.best || cachedBest); + const now = Date.now(); + const attempt = cached?.attempt; + const attemptFresh = Number.isSafeInteger(attempt?.at) && attempt.at > 0 && attempt.at <= now + && sameTags(attempt.tags, tags) + && now - attempt.at < ttlMs; + // Older records have no scope: a `next` winner proves both tags were tried; + // otherwise only the single latest channel is safe to reuse. + const lastScope = cached?.lastTags === undefined + ? (tags.length === 1 || cached?.best?.tag === 'next') + : sameTags(cached.lastTags, tags); + const lastFresh = Number.isSafeInteger(cached?.last) && cached.last > 0 && cached.last <= now + && now - cached.last < ttlMs && lastScope && (!cached.best || cachedBest); + const fresh = !force && (lastFresh || attemptFresh); const best = fresh || cacheOnly ? cachedBest : await lookUpSelf(cfg, cached, { tags, cachedBest, fetchLatest, record }); return { pkg: KIT_PKG, diff --git a/src/lib/windows-npm-shim.mjs b/src/lib/windows-npm-shim.mjs new file mode 100644 index 00000000..433ee81c --- /dev/null +++ b/src/lib/windows-npm-shim.mjs @@ -0,0 +1,83 @@ +// Recognize, never interpret, npm cmd-shim 8's plain `env node` wrapper pair. +// Fingerprints normalize only CRLF and the package entry path. Any flags, +// environment assignments, comments or custom behavior keep the PS fallback. +// Fixtures were emitted by cmd-shim 8.0.0; the ps1 hash matches the native +// Ruflo 3.48.0 CI artifact. No package manager or package code runs here. +import fs from 'node:fs'; +import path from 'node:path'; +import { createHash } from 'node:crypto'; + +const CMD_TEMPLATE = '46268d032014c41f0112ffc0f52a9b297a809c289587e984e5919c65f29dad79'; +const PS_TEMPLATE = '11a7c411407320ddc34a9ae9633d6372b1867994ff723693be5be16148dd45a8'; + +export function windowsEnvValue(env, key) { + return env[Object.keys(env).sort().find((name) => name.toUpperCase() === key.toUpperCase())]; +} + +/** Windows treats environment names case-insensitively. Remove a replaced + * spelling before applying the override, so Node's sorted env selection and + * the resolver agree about the caller's PATH and runtime. */ +export function mergeWindowsEnv(base, overrides) { + const merged = { ...base }; + for (const [key, value] of Object.entries(overrides)) { + for (const old of Object.keys(merged)) if (old.toUpperCase() === key.toUpperCase()) delete merged[old]; + merged[key] = value; + } + return merged; +} + +function readBounded(file) { + const stat = fs.statSync(file); + if (!stat.isFile() || stat.size > 65536) throw Error('not a bounded shim or manifest'); + return fs.readFileSync(file, 'utf8').replaceAll('\r\n', '\n'); +} +const fingerprint = (source, target) => createHash('sha256').update(source.replaceAll(target, '')).digest('hex'); +const isFile = (file) => { try { return fs.statSync(file).isFile(); } catch { return false; } }; +function inside(root, file) { + const relative = path.relative(root, file); + return relative && !relative.split(path.sep).includes('..') && !path.isAbsolute(relative); +} +function plainNodeShebang(file) { + const fd = fs.openSync(file, 'r'); + try { + const data = Buffer.alloc(128); + const size = fs.readSync(fd, data, 0, data.length, 0); + return /^#!\/usr\/bin\/env node\r?\n/.test(data.subarray(0, size).toString('utf8')); + } finally { fs.closeSync(fd); } +} + +/** Map ONLY the PATH-selected, unchanged npm wrapper pair to its own manifest's + * public bin. Never consult a global-root guess or bypass an internal bundle. + * Return null when ownership, containment, template or runtime is uncertain. */ +export function npmShimInvocation(candidate, args, env) { + try { + const cmd = readBounded(candidate); + const ps = readBounded(`${candidate.slice(0, -4)}.ps1`); + const target = ps.match(/"\$basedir\/(node_modules\/[A-Za-z0-9@_./-]+)"/)?.[1]; + if (!target || target.split('/').some((part) => !part || part === '.' || part === '..')) return null; + if (fingerprint(ps, target) !== PS_TEMPLATE + || fingerprint(cmd, target.replaceAll('/', '\\')) !== CMD_TEMPLATE) return null; + const parts = target.split('/'); + const packageName = parts[1].startsWith('@') ? `${parts[1]}/${parts[2]}` : parts[1]; + const base = path.resolve(path.dirname(candidate)); + const packageRoot = path.join(base, 'node_modules', ...packageName.split('/')); + const pkg = JSON.parse(readBounded(path.join(packageRoot, 'package.json'))); + if (pkg.name !== packageName) return null; + const name = path.basename(candidate).slice(0, -4); + const declared = typeof pkg.bin === 'string' + ? (packageName.split('/').at(-1) === name ? pkg.bin : null) : pkg.bin?.[name]; + if (typeof declared !== 'string' || !declared || path.win32.isAbsolute(declared) + || path.isAbsolute(declared) || declared.split(/[\\/]/).includes('..')) return null; + const entry = path.resolve(base, ...parts); + if (path.resolve(packageRoot, declared) !== entry || !isFile(entry)) return null; + const realPackage = fs.realpathSync(packageRoot); + if (!inside(fs.realpathSync(base), realPackage) + || !inside(realPackage, fs.realpathSync(entry)) || !plainNodeShebang(entry)) return null; + // npm chooses adjacent node.exe first, then node.exe on the caller's PATH. + // Do not substitute agentic-kit's current runtime or a different npm tree. + const adjacent = path.join(base, 'node.exe'); + const node = isFile(adjacent) ? adjacent : (windowsEnvValue(env, 'PATH') || '') + .split(path.delimiter).filter(Boolean).map((dir) => path.resolve(dir, 'node.exe')).find(isFile); + return node ? { command: node, args: [entry, ...args], resolved: true } : null; + } catch { return null; } +} diff --git a/tests/fixtures/npm-windows-shim/license.txt b/tests/fixtures/npm-windows-shim/license.txt new file mode 100644 index 00000000..20a47625 --- /dev/null +++ b/tests/fixtures/npm-windows-shim/license.txt @@ -0,0 +1,15 @@ +The ISC License + +Copyright (c) npm, Inc. and Contributors + +Permission to use, copy, modify, and/or distribute this software for any +purpose with or without fee is hereby granted, provided that the above +copyright notice and this permission notice appear in all copies. + +THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES +WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF +MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR +ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES +WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN +ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF OR +IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. diff --git a/tests/fixtures/npm-windows-shim/ruflo.cmd b/tests/fixtures/npm-windows-shim/ruflo.cmd new file mode 100644 index 00000000..ea9a8598 --- /dev/null +++ b/tests/fixtures/npm-windows-shim/ruflo.cmd @@ -0,0 +1,17 @@ +@ECHO off +GOTO start +:find_dp0 +SET dp0=%~dp0 +EXIT /b +:start +SETLOCAL +CALL :find_dp0 + +IF EXIST "%dp0%\node.exe" ( + SET "_prog=%dp0%\node.exe" +) ELSE ( + SET "_prog=node" + SET PATHEXT=%PATHEXT:;.JS;=;% +) + +endLocal & goto #_undefined_# 2>NUL || title %COMSPEC% & "%_prog%" "%dp0%\node_modules\ruflo\bin\ruflo.js" %* diff --git a/tests/fixtures/npm-windows-shim/ruflo.ps1 b/tests/fixtures/npm-windows-shim/ruflo.ps1 new file mode 100644 index 00000000..7de4de00 --- /dev/null +++ b/tests/fixtures/npm-windows-shim/ruflo.ps1 @@ -0,0 +1,28 @@ +#!/usr/bin/env pwsh +$basedir=Split-Path $MyInvocation.MyCommand.Definition -Parent + +$exe="" +if ($PSVersionTable.PSVersion -lt "6.0" -or $IsWindows) { + # Fix case when both the Windows and Linux builds of Node + # are installed in the same directory + $exe=".exe" +} +$ret=0 +if (Test-Path "$basedir/node$exe") { + # Support pipeline input + if ($MyInvocation.ExpectingInput) { + $input | & "$basedir/node$exe" "$basedir/node_modules/ruflo/bin/ruflo.js" $args + } else { + & "$basedir/node$exe" "$basedir/node_modules/ruflo/bin/ruflo.js" $args + } + $ret=$LASTEXITCODE +} else { + # Support pipeline input + if ($MyInvocation.ExpectingInput) { + $input | & "node$exe" "$basedir/node_modules/ruflo/bin/ruflo.js" $args + } else { + & "node$exe" "$basedir/node_modules/ruflo/bin/ruflo.js" $args + } + $ret=$LASTEXITCODE +} +exit $ret diff --git a/tests/kit/aqe-live-lock-process.test.mjs b/tests/kit/aqe-live-lock-process.test.mjs new file mode 100644 index 00000000..467f34b8 --- /dev/null +++ b/tests/kit/aqe-live-lock-process.test.mjs @@ -0,0 +1,138 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { mkdtempSync, existsSync, rmSync, mkdirSync } from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { createProcessScope } from '../live/aqe-live-lock-process.mjs'; +import { spawnEnv, envValue } from './helpers/home-sandbox.mjs'; +import { ownerRecord, writeOwner, readOwner, prepareRunRootHolds, + inspectRunRootHolds } from '../../scripts/run-roots.mjs'; + +function sandbox(root) { + const home = path.join(root, 'home'); + mkdirSync(home); + return spawnEnv(home); +} + +test('abort closes a call-owned child before its temporary root is removed', async () => { + const root = mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-abort-proof-')); + const controller = new AbortController(); + const scope = createProcessScope(controller.signal); + let closed = false; + try { + const marker = path.join(root, 'child-ready'); + const run = scope.launch(process.execPath, ['-e', `require('node:fs').writeFileSync(${JSON.stringify(marker)}, 'ready');setInterval(() => {}, 1000)`], { cwd: root, env: sandbox(root) }); + assert.ok(run.child.pid > 0); + const deadline = Date.now() + 2000; + while (!existsSync(marker) && Date.now() < deadline) await new Promise((resolve) => setTimeout(resolve, 10)); + assert.ok(existsSync(marker), 'the child must be running before cancellation'); + controller.abort(); + await scope.closeAll(); + closed = true; + assert.equal(run.closed, true); + assert.throws(() => scope.launch(process.execPath, [], { env: {} }), /cannot launch/); + assert.ok(existsSync(root), 'root must remain until closure is established'); + } finally { + if (!closed) await scope.closeAll(); + if (closed) rmSync(root, { recursive: true, force: true }); + } + assert.equal(existsSync(root), false); +}); + +test('spawn failure is retained without an unhandled rejection', async () => { + const scope = createProcessScope(new AbortController().signal); + const root = mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-spawn-failure-')); + const run = scope.launch(path.join(root, 'ak-missing-executable'), [], { env: sandbox(root) }); + await assert.rejects(scope.wait(run, 1000), /ENOENT/); + await scope.closeAll(); + rmSync(root, { recursive: true, force: true }); +}); + +test('requires explicit sandbox env and passes it to the child', async () => { + const root = mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-env-proof-')); + const scope = createProcessScope(new AbortController().signal); + try { + assert.throws(() => scope.launch(process.execPath, [], {}), /explicit sandbox env/); + assert.throws(() => scope.launch(process.execPath, [], { env: {} }), /requires sandbox home/); + const env = sandbox(root); + const run = scope.launch(process.execPath, + ['-e', 'console.log(JSON.stringify({home:process.env.HOME,tmp:process.env.TMPDIR,state:process.env.XDG_STATE_HOME}))'], + { cwd: root, env }); + const result = await scope.wait(run, 2000); + assert.equal(result.code, 0); + const observed = JSON.parse(result.stdout.trim()); + assert.equal(observed.home, env.HOME); + assert.equal(observed.tmp, env.TMPDIR); + assert.equal(observed.state, env.XDG_STATE_HOME); + } finally { + await scope.closeAll(); + rmSync(root, { recursive: true, force: true }); + } +}); + +test('Windows bootstrap validation accepts preserved mixed-case names without adding duplicates', async () => { + const root = mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-windows-env-')); + const home = path.join(root, 'home'); + mkdirSync(home); + const env = spawnEnv(home, {}, { platform: 'win32', env: { + SYSTEMROOT: process.env.SystemRoot ?? process.env.SYSTEMROOT ?? 'C:\\Windows', + COMSPEC: process.env.ComSpec ?? process.env.COMSPEC ?? 'C:\\Windows\\System32\\cmd.exe', + PaThExT: process.env.PATHEXT ?? '.COM;.EXE;.BAT;.CMD', + Path: process.env.PATH ?? '', + } }); + assert.equal(env.SystemRoot, undefined); + assert.equal(env.ComSpec, undefined); + assert.equal(envValue(env, 'SystemRoot', 'win32'), env.SYSTEMROOT); + const scope = createProcessScope(new AbortController().signal, { platform: 'win32' }); + try { + assert.throws(() => scope.launch(process.execPath, [], { env: { ...env, COMSPEC: '' } }), + /requires Windows process env/); + const run = scope.launch(process.execPath, ['-e', 'console.log("bootstrapped")'], { env }); + const result = await scope.wait(run, 2000); + assert.equal(result.code, 0); + assert.match(result.stdout, /bootstrapped/); + assert.equal(Object.keys(env).filter((key) => key.toUpperCase() === 'SYSTEMROOT').length, 1); + assert.equal(Object.keys(env).filter((key) => key.toUpperCase() === 'COMSPEC').length, 1); + } finally { + await scope.closeAll(); + rmSync(root, { recursive: true, force: true }); + } +}); + +test('a guarded scope holds the enclosing run root until owned children close', async () => { + const base = mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-hold-base-')); + const root = mkdtempSync(path.join(base, 'ak-suite-')); + const owner = ownerRecord(); + writeOwner(root, owner); + prepareRunRootHolds(root, owner.runId); + const beforeRoot = process.env.AK_SUITE_ROOT; + const beforeId = process.env.AK_SUITE_RUN_ID; + process.env.AK_SUITE_ROOT = root; + process.env.AK_SUITE_RUN_ID = owner.runId; + let scope; + try { + scope = createProcessScope(new AbortController().signal, { closeLimitMs: 25 }); + assert.equal(inspectRunRootHolds(root, readOwner(root).runId).unresolved, true); + const run = scope.launch(process.execPath, ['-e', 'setInterval(() => {}, 1000)'], + { env: sandbox(base) }); + assert.equal(run.closed, false); + const realKill = run.child.kill.bind(run.child); + run.child.kill = () => false; + try { + await assert.rejects(scope.closeAll(), /did not close/); + assert.equal(inspectRunRootHolds(root, owner.runId).unresolved, true); + } finally { + run.child.kill = realKill; + } + await scope.closeAll(); + assert.equal(run.closed, true); + assert.equal(inspectRunRootHolds(root, owner.runId).unresolved, false); + } finally { + if (scope) await scope.closeAll(); + if (beforeRoot === undefined) delete process.env.AK_SUITE_ROOT; + else process.env.AK_SUITE_ROOT = beforeRoot; + if (beforeId === undefined) delete process.env.AK_SUITE_RUN_ID; + else process.env.AK_SUITE_RUN_ID = beforeId; + rmSync(base, { recursive: true, force: true }); + } +}); diff --git a/tests/kit/aqe-verification.test.mjs b/tests/kit/aqe-verification.test.mjs index 1bad2acd..60e24c5b 100644 --- a/tests/kit/aqe-verification.test.mjs +++ b/tests/kit/aqe-verification.test.mjs @@ -6,39 +6,40 @@ import { aqeVerificationPassed } from '../../src/lib/aqe-verification.mjs'; test('a live lock does not hide an independent storage error', () => { assert.equal(classifyAqeStartup({ code: 0, stderr: 'locked by a live process; FsyncFailed' }).status, 'failed'); }); -// TEMPORARY (remove with the rule in classifyAqeStartup, pacphi/agentic-kit#240): the -// exact stderr agentic-qe 3.14.3 emits when a healthy patterns.rvf is held by a live -// owner (captured from a fixture; store and lock bytes were unchanged). The FsyncFailed -// comes from a create attempt AQE should not make (agentic-qe#574). Remove this test -// when a released agentic-qe fixes agentic-qe#574 and that release is the kit's floor; -// agentic-qe#719 (in 3.14.4) is a partial fix and does not remove it. -const LIVE_OWNER_CONTENTION = [ +// AQE 3.14.3 emitted this exact sequence under a live lock. The released 3.14.4 +// artifact omits FsyncFailed in macOS and Linux conformance probes; old output fails closed. +const OLD_LIVE_OWNER_CONTENTION = [ '[RVF] /p/.agentic-qe/patterns.rvf is locked by a live process (pid 70149) — not breaking the lock; degrading to SQLite for this run.', '[RVF] /p/.agentic-qe/patterns.rvf is unusable but its lock is held by a live process — leaving it alone and degrading to SQLite for this run.', '[RVF] Shared adapter init failed: RVF error 0x0303: FsyncFailed', ].join('\n'); -test('live-owner lock contention reads as busy, not a storage failure (agentic-qe#574)', () => { - const startup = classifyAqeStartup({ code: 0, stdout: '', stderr: LIVE_OWNER_CONTENTION }); - assert.equal(startup.status, 'busy'); - assert.match(startup.reason, /another live process/); - assert.match(startup.reason, /integrity unverified/); - assert.equal(aqeVerificationPassed(startup, { status: 'passed', corpus: { status: 'healthy' } }), true); +test('old live-owner FsyncFailed sequence fails closed and blocks verification', () => { + const startup = classifyAqeStartup({ code: 0, stdout: '', stderr: OLD_LIVE_OWNER_CONTENTION }); + assert.deepEqual(startup, { status: 'failed', reason: 'RVF backend failed' }); + assert.equal(aqeVerificationPassed(startup, { status: 'passed', corpus: { status: 'healthy' } }), false); }); -test('the contention rule needs all three lines; a partial match still fails', () => { - const [locked, unusable, fsync] = LIVE_OWNER_CONTENTION.split('\n'); +test('FsyncFailed fails with or without partial live-lock lines', () => { + const [locked, unusable, fsync] = OLD_LIVE_OWNER_CONTENTION.split('\n'); for (const stderr of [fsync, `${locked}\n${fsync}`, `${unusable}\n${fsync}`]) { assert.equal(classifyAqeStartup({ code: 0, stderr }).status, 'failed', stderr); } - assert.equal(classifyAqeStartup({ code: 1, stderr: LIVE_OWNER_CONTENTION }).status, 'failed'); + assert.equal(classifyAqeStartup({ code: 1, stderr: OLD_LIVE_OWNER_CONTENTION }).status, 'failed'); }); -test('live-owner contention does not hide failed embedding initialization', () => { - const stderr = `${LIVE_OWNER_CONTENTION}\nReasoningBank prewarm failed`; - assert.equal(classifyAqeStartup({ code: 0, stderr }).status, 'degraded'); +test('FsyncFailed takes precedence over failed embedding initialization', () => { + const stderr = `${OLD_LIVE_OWNER_CONTENTION}\nReasoningBank prewarm failed`; + assert.equal(classifyAqeStartup({ code: 0, stderr }).status, 'failed'); }); test('a live lock does not hide failed embedding initialization', () => { assert.equal(classifyAqeStartup({ code: 0, stderr: 'locked by a live process; prewarm failed' }).status, 'degraded'); }); +test('ordinary live lock remains busy with owner health and RVF integrity unknown', () => { + const startup = classifyAqeStartup({ code: 0, stderr: '[RVF] locked by a live process; 0x0300: LockHeld; degrading to SQLite' }); + assert.equal(startup.status, 'busy'); + assert.match(startup.reason, /SQLite fallback observed/); + assert.match(startup.reason, /owner health and RVF integrity unverified/); + assert.equal(aqeVerificationPassed(startup, { status: 'passed', corpus: { status: 'healthy' } }), true); +}); test('busy RVF permits qualified semantic proof when corpus is verified', () => { assert.equal(aqeVerificationPassed({ status: 'busy' }, { status: 'passed', corpus: { status: 'healthy' } }), true); }); diff --git a/tests/kit/cli-json-honesty.test.mjs b/tests/kit/cli-json-honesty.test.mjs index 1a4a0a34..f82b79bf 100644 --- a/tests/kit/cli-json-honesty.test.mjs +++ b/tests/kit/cli-json-honesty.test.mjs @@ -23,9 +23,9 @@ const BIN = path.join(PKG_ROOT, 'bin', 'agentic-kit.mjs'); const KIT_JSON = path.join(HOME, '.config', 'agentic-kit', 'kit.json'); /** Run `ak …args` in the sandbox. */ -function ak(args) { +function ak(args, envOverrides = {}) { return spawnSync(process.execPath, [BIN, ...args], { - cwd: PROJECT, env: spawnEnv(HOME), encoding: 'utf8', timeout: 120_000, + cwd: PROJECT, env: { ...spawnEnv(HOME), ...envOverrides }, encoding: 'utf8', timeout: 120_000, }); } @@ -124,6 +124,72 @@ for (const [args, message] of USAGE_ERRORS) { }); } +const COMMAND_USAGE_ERRORS = [ + [['usage', 'bogus', '--json'], /usage: ak usage/], + [['usage', 'score', '--window', '99', '--json'], /--window must be/], + [['usage', 'prompts', '--window', '99', '--json'], /--window must be/], + [['usage', 'score', 'extra', '--json'], /unexpected argument 'extra'/], + [['usage', 'prompts', 'extra', '--json'], /unexpected argument 'extra'/], + [['models', 'bogus', '--json'], /usage: ak models/], + [['models', 'explain', '--json'], /usage: ak models explain/], + [['models', 'plan', '--json'], /usage: ak models plan/], + [['models', 'status', 'extra', '--json'], /unexpected argument/], + [['models', 'status', '--host', 'bogus', '--json'], /unsupported model host/], + [['audit', 'bogus', '--json'], /requires the hooks or context subcommand/], + [['heal', 'bogus', '--json'], /requires the hooks subcommand/], + [['heal', 'hooks', '--yes', '--json'], /--yes requires --apply/], + [['x', 'aqe-store', 'bogus', '--json'], /usage: ak x aqe-store/], + [['x', 'aqe-embedding', 'bogus', '--json'], /aqe-embedding/], + [['x', 'codex-context', 'bogus', '--json'], /codex-context/], + [['x', 'skills', 'bogus', '--json'], /usage: ak x skills/], + [['x', 'reference', 'bogus', '--json'], /reference/], + [['x', 'statusline', 'bogus', '--json'], /usage: ak x statusline/], + [['x', 'daemon-gc', 'bogus', '--json'], /unexpected argument/], + [['x', 'harvest', 'bogus', '--json'], /unexpected argument/], + [['host', 'bogus', '--json'], /unknown host subcommand/], + [['host', 'status', 'extra', '--json'], /unexpected argument/], + [['host', 'adapters', '--dry-run', '--json'], /has no preview/], + [['host', 'adapters', 'unknown', '--json'], /experimental host-adapter surface is disabled/], +]; + +for (const [args, message] of COMMAND_USAGE_ERRORS) { + test(`ak ${args.join(' ')} reports one command-level JSON usage error`, () => { + const child = ak(args); + const out = oneJson(child); + assert.equal(child.status, 2, child.stderr); + assert.deepEqual(Object.keys(out), ['error', 'exitCode']); + assert.equal(out.exitCode, 2); + assert.match(out.error, message); + assert.match(child.stderr, message); + }); +} + +for (const enabled of ['0', '1']) { + for (const verb of ['revoke', 'revoke-grant']) { + test(`ak host adapters ${verb} --json without a name is JSON with feature flag ${enabled}`, () => { + const child = ak(['host', 'adapters', verb, '--json'], { AK_EXPERIMENTAL_HOST_ADAPTERS: enabled }); + const out = oneJson(child); + assert.equal(child.status, 2, child.stderr); + assert.deepEqual(Object.keys(out), ['error', 'exitCode']); + assert.equal(out.exitCode, 2); + assert.match(out.error, new RegExp(`usage: ak host adapters ${verb} `)); + assert.match(child.stderr, new RegExp(`usage: ak host adapters ${verb} `)); + }); + } +} + +test('models rejects an unknown verb even when no snapshot exists', () => { + const child = ak(['models', 'bogus', '--json']); + assert.equal(child.status, 2, child.stderr); + assert.match(oneJson(child).error, /usage: ak models/); +}); + +test('plain status rejects a stray positional with exit 2', () => { + const child = ak(['status', 'stray']); + assert.equal(child.status, 2, child.stderr); + assert.match(child.stdout, /unexpected argument 'stray'/); +}); + test('without --json a command-level usage error still prints on stdout', () => { const child = ak(['status', '--refresh=bogus']); assert.equal(child.status, 2); diff --git a/tests/kit/daemon-gc-rerecord.test.mjs b/tests/kit/daemon-gc-rerecord.test.mjs new file mode 100644 index 00000000..5adb05e5 --- /dev/null +++ b/tests/kit/daemon-gc-rerecord.test.mjs @@ -0,0 +1,62 @@ +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { sandboxHome, rmrf } from './helpers/home-sandbox.mjs'; + +const home = sandboxHome('ak-daemon-gc-rerecord'); +after(() => rmrf(home)); +const trapDir = path.join(home, 'no-such-bin'); +const processProbeMarker = path.join(home, 'process-probe-reached'); +fs.mkdirSync(trapDir, { recursive: true }); +for (const command of ['ps', 'powershell']) { + const trap = path.join(trapDir, command); + fs.writeFileSync(trap, `#!/bin/sh\n: > '${processProbeMarker}'\nexit 99\n`, { mode: 0o755 }); +} +const { run } = await import('../../src/commands/x/daemon-gc.mjs'); + +for (const [name, kill, killed, expected] of [ + ['successful reap', true, true, 2], + ['failed reap', true, false, 1], + ['list only', false, true, 1], +]) { + test(`${name} re-records only after a successful kill`, async () => { + const calls = []; + let reapCalls = 0; + const daemon = { pid: 4242, workspace: '/missing-ak-workspace', workspaceExists: false, ageSecs: 1 }; + const result = await run({ + flags: { kill, mcp: false, quiet: true }, + deps: { + daemonLifecycle: { + list: async opts => { calls.push(opts); return [daemon]; }, + reap: () => { reapCalls++; return [{ ...daemon, killed }]; }, + }, + mcpLifecycle: { list: async () => [], reap: () => { throw new Error('MCP reap forbidden'); } }, + }, + }); + assert.equal(result, 0); + assert.equal(reapCalls, kill ? 1 : 0); + assert.equal(calls.length, expected); + if (expected === 2) assert.deepEqual(calls[1], { refresh: true, record: true, source: 'daemon-gc' }); + assert.equal(fs.existsSync(processProbeMarker), false, 'real process discovery was reached'); + }); +} + +test('JSON listing observes MCP transports without reaping or re-recording', async () => { + const daemonCalls = []; + let mcpCalls = 0; + const oldLog = console.log; + console.log = () => {}; + try { + assert.equal(await run({ + flags: { json: true, kill: true, mcp: false }, + deps: { + daemonLifecycle: { list: async opts => { daemonCalls.push(opts); return []; }, reap: () => { throw new Error('reap forbidden'); } }, + mcpLifecycle: { list: async () => { mcpCalls++; return []; }, reap: () => { throw new Error('MCP reap forbidden'); } }, + }, + }), 0); + } finally { console.log = oldLog; } + assert.deepEqual(daemonCalls, [undefined]); + assert.equal(mcpCalls, 1); + assert.equal(fs.existsSync(processProbeMarker), false, 'real process discovery was reached'); +}); diff --git a/tests/kit/deja-vu-teardown-verify.test.mjs b/tests/kit/deja-vu-teardown-verify.test.mjs index 8f5041cf..c36e0ca2 100644 --- a/tests/kit/deja-vu-teardown-verify.test.mjs +++ b/tests/kit/deja-vu-teardown-verify.test.mjs @@ -80,7 +80,7 @@ test('deja-vu verify cleanly skips disabled, unowned integration without probing const { result, out } = await captureLog(() => verify.verifyDejaVu({ cfg: cfg({ enabled: false }), adapter, })); - assert.equal(result, true); + assert.deepEqual(result, { status: 'skipped', reason: 'disabled and unowned' }); assert.deepEqual(calls, []); assert.match(out, /disabled and unowned — skipped/); }); diff --git a/tests/kit/exec.test.mjs b/tests/kit/exec.test.mjs index 0526bcda..770ff176 100644 --- a/tests/kit/exec.test.mjs +++ b/tests/kit/exec.test.mjs @@ -3,7 +3,259 @@ import assert from 'node:assert/strict'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; -import { run, have, resolveShim } from '../../src/lib/exec.mjs'; +import { run, have, resolveShim, withAbortSignal } from '../../src/lib/exec.mjs'; + +const isAlive = (pid) => { + try { process.kill(pid, 0); return true; } catch { return false; } +}; +async function waitUntil(predicate) { + for (let n = 0; n < 60; n += 1) { + if (predicate()) return true; + await new Promise((resolve) => setTimeout(resolve, 50)); + } + return predicate(); +} + +for (const [name, options] of [ + ['no input with explicit signal', { input: undefined, inherited: false }], + ['input with explicit signal', { input: '', inherited: false }], + ['no input with inherited signal', { input: undefined, inherited: true }], + ['input with inherited signal', { input: '', inherited: true }], +]) { + test(`run() abort reaps owned grandchild: ${name}`, async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-abort-tree-')); + const pidFile = path.join(dir, 'pids.json'); + const controller = new AbortController(); + const code = `const {spawn}=require('node:child_process'); + const gc=spawn(process.execPath,['-e','setInterval(()=>{},1000)'],{stdio:'ignore'}); + require('node:fs').writeFileSync(${JSON.stringify(pidFile)},JSON.stringify([process.pid,gc.pid])); + setInterval(()=>{},1000);`; + let pids = []; + let pending; + try { + const launch = () => run(process.execPath, ['-e', code], { + input: options.input, timeout: 5_000, + ...(options.inherited ? {} : { signal: controller.signal }), + }); + pending = options.inherited ? withAbortSignal(controller.signal, launch) : launch(); + assert.equal(await waitUntil(() => fs.existsSync(pidFile)), true, 'owned children started'); + pids = JSON.parse(fs.readFileSync(pidFile, 'utf8')); + assert.equal(isAlive(pids[1]), true, 'grandchild alive before abort'); + controller.abort(); + const result = await pending; + assert.notEqual(result.code, 0); + assert.equal(await waitUntil(() => !isAlive(pids[1])), true, + 'abort must terminate the grandchild, not only its parent'); + } finally { + controller.abort(); + for (const pid of pids) { + if (isAlive(pid)) { + try { process.kill(pid, 'SIGKILL'); } catch { /* already exited */ } + } + } + await pending; + assert.equal(await waitUntil(() => pids.every((pid) => !isAlive(pid))), true, 'fixture children exited'); + fs.rmSync(dir, { recursive: true, force: true }); + } + }); +} + +test('run() does not spawn for a pre-aborted signal, including inherited abort', async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-preabort-')); + const marker = path.join(dir, 'spawned'); + const aborted = new AbortController(); + aborted.abort(); + const args = ['-e', `require('node:fs').writeFileSync(${JSON.stringify(marker)},'yes')`]; + try { + for (const input of [undefined, '']) { + const explicit = await run(process.execPath, args, { signal: aborted.signal, input }); + const inherited = await withAbortSignal(aborted.signal, + () => run(process.execPath, args, { input })); + assert.notEqual(explicit.code, 0); + assert.notEqual(inherited.code, 0); + assert.equal(fs.existsSync(marker), false); + } + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } +}); + +test('an explicit live signal takes precedence over an aborted inherited signal', async () => { + const inherited = new AbortController(); + inherited.abort(); + const explicit = new AbortController(); + for (const input of [undefined, '']) { + const result = await withAbortSignal(inherited.signal, () => run( + process.execPath, ['-e', 'process.stdout.write("ok")'], + { signal: explicit.signal, input }, + )); + assert.equal(result.code, 0, result.stderr); + assert.equal(result.stdout, 'ok'); + } +}); + +for (const stop of ['abort', 'timeout']) { + test(`Windows ${stop} after direct-child exit reports incomplete tree cleanup`, async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-exited-root-')); + const pidFile = path.join(dir, 'pids.json'); + const exitFile = path.join(dir, 'parent-exit'); + const readyFile = path.join(dir, 'descendant-ready'); + const controller = new AbortController(); + // libuv's Windows Job Object kills non-detached children when their + // parent exits. unref() alone only releases the event-loop reference. + // Deliberately escape that job while retaining the real output handles. + const descendant = `require('node:fs').writeFileSync(${JSON.stringify(readyFile)},'ready'); + setTimeout(()=>{},10000);`; + const code = `const {spawn}=require('node:child_process'); + const fs=require('node:fs'); + const gc=spawn(process.execPath,['-e',${JSON.stringify(descendant)}],{ + detached:process.platform==='win32',stdio:['ignore','inherit','inherit']}); + gc.unref(); + fs.writeFileSync(${JSON.stringify(pidFile)},JSON.stringify([process.pid,gc.pid])); + const ready=setInterval(()=>{ + if(fs.existsSync(${JSON.stringify(readyFile)})) clearInterval(ready); + },10); + setTimeout(()=>process.exit(2),4000).unref(); + process.on('exit',()=>fs.writeFileSync(${JSON.stringify(exitFile)},'yes'));`; + let pids = []; + let pending; + try { + const started = Date.now(); + pending = run(process.execPath, ['-e', code], { + windows: true, signal: controller.signal, timeout: stop === 'timeout' ? 1_200 : 5_000, + }); + assert.equal(await waitUntil(() => fs.existsSync(pidFile)), true); + pids = JSON.parse(fs.readFileSync(pidFile, 'utf8')); + assert.equal(await waitUntil(() => fs.existsSync(readyFile)), true, 'descendant initialized'); + assert.equal(await waitUntil(() => fs.existsSync(exitFile)), true, 'direct child exited'); + await new Promise((resolve) => setTimeout(resolve, 100)); + assert.equal(isAlive(pids[1]), true, 'descendant still owns the output pipe'); + const stoppedAt = Date.now(); + if (stop === 'abort') controller.abort(); + const result = await pending; + assert.notEqual(result.code, 0, 'an incomplete run cannot report success'); + assert.match(result.stderr, /incomplete.*tree cleanup/i); + assert.ok(Date.now() - (stop === 'abort' ? stoppedAt : started) + < (stop === 'abort' ? 1_400 : 2_600), 'return is bounded by abort or timeout'); + } finally { + controller.abort(); + for (const pid of pids) { + if (isAlive(pid)) { + try { process.kill(pid, 'SIGKILL'); } catch { /* already exited */ } + } + } + await pending; + assert.equal(await waitUntil(() => pids.every((pid) => !isAlive(pid))), true, 'fixture children exited'); + fs.rmSync(dir, { recursive: true, force: true }); + } + }); +} + +test('Windows taskkill stall has a bounded cleanup wait and reports uncertainty', { + skip: process.platform === 'win32', +}, async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-taskkill-stall-')); + const killer = path.join(dir, 'taskkill.exe'); + const oldPath = process.env.PATH; + fs.writeFileSync(killer, `#!${process.execPath}\nsetTimeout(() => process.exit(1), 1500);\n`, { mode: 0o755 }); + process.env.PATH = `${dir}${path.delimiter}${oldPath}`; + try { + const started = Date.now(); + const result = await run(process.execPath, ['-e', 'setInterval(()=>{},1000)'], { + windows: true, timeout: 100, + }); + assert.notEqual(result.code, 0); + assert.match(result.stderr, /incomplete.*tree cleanup/i); + assert.ok(Date.now() - started < 1400, 'does not wait for the stalled taskkill process'); + } finally { + process.env.PATH = oldPath; + fs.rmSync(dir, { recursive: true, force: true }); + } +}); + +test('run() enforces maxBuffer in UTF-8 bytes', async () => { + const result = await run(process.execPath, ['-e', 'process.stdout.write("ééé")'], { + maxBuffer: 4, + }); + assert.notEqual(result.code, 0); + assert.match(result.stderr, /maxBuffer/i); +}); + +function recordShimChildren(reported, observedShimPid, cleanupPids) { + // The reported parent is assertion evidence, never kill authority. + cleanupPids.push(reported[0], reported[1], observedShimPid); + assert.equal(reported[2], observedShimPid, 'Node is a child of the observed PowerShell shim'); +} + +test('a mismatched reported parent never becomes a fixture cleanup kill target', () => { + const cleanupPids = []; + const killTargets = []; + const unexpectedParent = 990003; + // Synthetic PIDs and an injected recording function: no real process signal. + const kill = (pid) => { killTargets.push(pid); }; + try { + assert.throws(() => recordShimChildren([990001, 990002, unexpectedParent], 990004, cleanupPids), + /Node is a child of the observed PowerShell shim/); + } finally { + for (const pid of cleanupPids) kill(pid); + } + assert.equal(killTargets.includes(unexpectedParent), false, 'unowned reported parent must never be signalled'); + assert.deepEqual(killTargets, [990001, 990002, 990004]); +}); + +test('Windows abort reaps the Node child behind a PowerShell shim and its grandchild', { + skip: process.platform !== 'win32', +}, async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-abort-shim-')); + const pidFile = path.join(dir, 'pids.json'); + const shimEntry = path.join(dir, 'powershell-pid'); + const controller = new AbortController(); + const quotedNode = process.execPath.replaceAll("'", "''"); + const pids = []; + let pending; + let outcome; + try { + fs.writeFileSync(path.join(dir, 'codex.cmd'), '@echo off\r\n'); + const code = `const {spawn}=require('node:child_process'); + const gc=spawn(process.execPath,['-e','setInterval(()=>{},1000)'],{stdio:'ignore'}); + require('node:fs').writeFileSync(${JSON.stringify(pidFile)},JSON.stringify([process.pid,gc.pid,process.ppid])); + setInterval(()=>{},1000);`; + // npm shims forward a script filename; PowerShell 5.1 reserializes + // native arguments, so multiline node -e source is not that interface. + const script = path.join(dir, 'fixture.cjs'); + fs.writeFileSync(script, code); + fs.writeFileSync(path.join(dir, 'codex.ps1'), + `[System.IO.File]::WriteAllText('${shimEntry.replaceAll("'", "''")}',[string]$PID)\n` + + `& '${quotedNode}' '${script.replaceAll("'", "''")}' $args\nexit $LASTEXITCODE\n`); + pending = run('codex', [], { + // PowerShell checks PATHEXT even for the absolute Node.exe path. + // Excluding .EXE changes native execution into document activation. + env: { PATH: dir, PATHEXT: '.COM;.EXE;.BAT;.CMD' }, signal: controller.signal, timeout: 10_000, + }); + pending.then((result) => { outcome = result; }); + assert.equal(await waitUntil(() => fs.existsSync(pidFile) || outcome), true, 'PowerShell launch settled or ready'); + assert.equal(fs.existsSync(shimEntry), true, `PowerShell entered owned shim: ${JSON.stringify(outcome)}`); + assert.equal(fs.existsSync(pidFile), true, `PowerShell launched Node: ${JSON.stringify(outcome)}`); + const reported = JSON.parse(fs.readFileSync(pidFile, 'utf8')); + recordShimChildren(reported, Number(fs.readFileSync(shimEntry, 'utf8')), pids); + assert.equal(isAlive(pids[1]), true); + controller.abort(); + const result = await pending; + assert.notEqual(result.code, 0); + assert.equal(await waitUntil(() => pids.every((pid) => !isAlive(pid))), true, + 'taskkill must remove the Node process and its grandchild'); + } finally { + controller.abort(); + for (const pid of pids) { + if (isAlive(pid)) { + try { process.kill(pid, 'SIGKILL'); } catch { /* already exited */ } + } + } + await pending; + assert.equal(await waitUntil(() => pids.every((pid) => !isAlive(pid))), true, 'fixture children exited'); + fs.rmSync(dir, { recursive: true, force: true }); + } +}); // code-quality Finding 2: exec.mjs used to set shell:true for a fixed set of // Windows .cmd shims (npm/npx/claude/ruflo/aqe/claude-flow), which handed diff --git a/tests/kit/helpers/home-sandbox.mjs b/tests/kit/helpers/home-sandbox.mjs index 65f8bad0..038c6dfd 100644 --- a/tests/kit/helpers/home-sandbox.mjs +++ b/tests/kit/helpers/home-sandbox.mjs @@ -287,7 +287,9 @@ export function offlineKitConfig(extra = {}) { ttlHours: 24, last: Date.now(), seen: { ruflo: '9.9.9', 'agentic-qe': '9.9.9' }, - self: { last: Date.now(), best: { version: '0.0.1', tag: 'latest' } }, + // This fixture runs against the prerelease kit, whose self check uses + // both channels. A legacy latest-only record must retry under A3. + self: { last: Date.now(), best: { version: '0.0.1', tag: 'latest' }, lastTags: ['latest', 'next'] }, }, ...extra, }; diff --git a/tests/kit/host-dry-run.test.mjs b/tests/kit/host-dry-run.test.mjs index a7db8ed2..68fec135 100644 --- a/tests/kit/host-dry-run.test.mjs +++ b/tests/kit/host-dry-run.test.mjs @@ -59,6 +59,23 @@ function ak(sb, ...args) { const readKit = (home) => fs.readFileSync(path.join(home, '.config', 'agentic-kit', 'kit.json'), 'utf8'); +for (const args of [ + ['host', 'status'], ['host'], ['x', 'host'], +]) { + test(`ak ${args.join(' ')} --dry-run --json reports status without recording evidence`, (t) => { + const sb = sandbox(t); + const beforeHome = snapshot(sb.home); + const beforeProject = snapshot(sb.project); + const r = ak(sb, ...args, '--dry-run', '--json'); + assert.equal(r.status, 0, r.all); + const out = JSON.parse(r.stdout); + assert.deepEqual(Object.keys(out), ['scope', 'config', 'hosts', 'providers']); + assert.ok(out.hosts.claude); + assertUnchanged(beforeHome, sb.home, 'status preview must not write host evidence or config'); + assertUnchanged(beforeProject, sb.project, 'status preview must not write project files'); + }); +} + test('ak host pick --dry-run previews and writes nothing', (t) => { const sb = sandbox(t); const beforeHome = snapshot(sb.home); @@ -119,6 +136,74 @@ test('ak host pick --dry-run --json carries previewOfCurrent, true only with no assert.equal(explicitJson.previewOfCurrent, false); }); +test('host pick refusal and x host alias dry-run emit one JSON object without writes', (t) => { + const sb = sandbox(t); + const beforeHome = snapshot(sb.home); + const beforeProject = snapshot(sb.project); + for (const command of ['host', 'x']) { + const args = command === 'x' ? ['x', 'host'] : ['host']; + const r = ak(sb, ...args, 'pick', '--host', 'claude,opencdoe', '--dry-run', '--json'); + assert.equal(r.status, 2, r.all); + const out = JSON.parse(r.stdout); + assert.deepEqual(Object.keys(out), ['error', 'exitCode']); + assert.equal(out.exitCode, 2); + assert.match(out.error, /unknown host\(s\): opencdoe/); + assert.match(r.stderr, /unknown host\(s\): opencdoe/); + } + assertUnchanged(beforeHome, sb.home, 'refused picks must not touch HOME'); + assertUnchanged(beforeProject, sb.project, 'refused picks must not touch the project'); +}); + +test('host off dry-run JSON previews teardown and preserves configuration', (t) => { + const sb = sandbox(t, divergedConfig()); + const beforeHome = snapshot(sb.home); + const beforeProject = snapshot(sb.project); + const r = ak(sb, 'host', 'off', '--dry-run', '--json'); + assert.equal(r.status, 0, r.all); + const out = JSON.parse(r.stdout); + assert.equal(out.dryRun, true); + assert.deepEqual(out.wouldDisable, ['claude', 'codex']); + assert.equal(out.primaryHost, 'claude'); + assert.deepEqual(out.wouldClear, ['aqe provider/fallback', 'ruflo providers', 'activity routing']); + assert.equal(out.wouldStripManagedProviderEnv, true); + assert.equal(out.wouldRestoreOrRemoveManagedAqeConfig, true); + assert.equal(out.wouldReconcileOpencodeGuidance, true); + assert.equal(out.wouldTeardownOpencode, false); + assert.equal(out.wouldRemoveManagedCodexMcp, false); + assert.match(r.stderr, /dry run/i); + assertUnchanged(beforeHome, sb.home, 'off preview must not touch HOME'); + assertUnchanged(beforeProject, sb.project, 'off preview must not touch the project'); +}); + +test('host reset-routes dry-run JSON reports selected and empty routes without writes', (t) => { + const sb = sandbox(t, divergedConfig()); + const beforeHome = snapshot(sb.home); + const beforeProject = snapshot(sb.project); + const selected = ak(sb, 'host', 'reset-routes', '--activity', 'architecture', '--dry-run', '--json'); + assert.equal(selected.status, 0, selected.all); + assert.deepEqual(JSON.parse(selected.stdout), { dryRun: true, activities: ['architecture'] }); + const empty = ak(sb, 'host', 'reset-routes', '--activity', 'design', '--dry-run', '--json'); + assert.equal(empty.status, 0, empty.all); + assert.deepEqual(JSON.parse(empty.stdout), { dryRun: true, activities: [] }); + const invalid = ak(sb, 'host', 'reset-routes', '--activity', 'not-an-activity', '--dry-run', '--json'); + assert.equal(invalid.status, 0, invalid.all); + assert.deepEqual(JSON.parse(invalid.stdout), { dryRun: true, activities: [] }); + assert.match(invalid.stderr, /unknown activity 'not-an-activity'/); + assertUnchanged(beforeHome, sb.home, 'route previews must not touch HOME'); + assertUnchanged(beforeProject, sb.project, 'route previews must not touch the project'); +}); + +test('host reset-routes dry-run JSON reports no divergence as an empty preview', (t) => { + const sb = sandbox(t); + const beforeHome = snapshot(sb.home); + const beforeProject = snapshot(sb.project); + const r = ak(sb, 'host', 'reset-routes', '--dry-run', '--json'); + assert.equal(r.status, 0, r.all); + assert.deepEqual(JSON.parse(r.stdout), { dryRun: true, activities: [] }); + assertUnchanged(beforeHome, sb.home, 'no-op route preview must not touch HOME'); + assertUnchanged(beforeProject, sb.project, 'no-op route preview must not touch the project'); +}); + test('ak host off --dry-run previews and writes nothing', (t) => { const sb = sandbox(t, divergedConfig()); const beforeHome = snapshot(sb.home); diff --git a/tests/kit/host-pick-rerecord.test.mjs b/tests/kit/host-pick-rerecord.test.mjs new file mode 100644 index 00000000..72887e89 --- /dev/null +++ b/tests/kit/host-pick-rerecord.test.mjs @@ -0,0 +1,57 @@ +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import { sandboxHome, rmrf, writeKitConfig, offlineKitConfig } from './helpers/home-sandbox.mjs'; + +const home = sandboxHome('ak-host-pick-rerecord'); +after(() => rmrf(home)); +const host = await import('../../src/commands/x/host.mjs'); +const cfg = { integrations: { hosts: { claude: false, codex: true, opencode: false } } }; +const cwd = '/disposable-project'; + +for (const [name, initial, ok, expected] of [ + ['successful install', 'absent', true, 2], + ['failed install', 'absent', false, 1], + ['present host', 'external', true, 1], +]) { + test(`host pick ${name} re-records only after success`, async () => { + const states = []; + const facts = []; + const installs = []; + await host.installPickAbsentHosts(cfg, cwd, { + installState: async (_host, opts) => { states.push(opts); return { method: states.length === 1 ? initial : 'npm', version: '1.0.0' }; }, + install: async id => { installs.push(id); return { ok, detail: ok ? 'installed' : 'failed' }; }, + collectFacts: async opts => { facts.push(opts); }, + }); + assert.equal(states.length, expected); + assert.deepEqual(installs, initial === 'absent' ? ['codex'] : []); + if (expected === 2) { + assert.deepEqual(states[1], { refresh: true, record: true, source: 'host-pick' }); + assert.deepEqual(facts, [{ cwd, cfg, refresh: true, record: true, source: 'host-pick' }]); + } else assert.deepEqual(facts, []); + }); +} + +test('disabled hosts do not probe, install, or record evidence', async () => { + await host.installPickAbsentHosts({ integrations: { hosts: {} } }, cwd, { + installState: async () => { throw new Error('disabled host probed'); }, + install: async () => { throw new Error('disabled host installed'); }, + collectFacts: async () => { throw new Error('disabled host recorded'); }, + }); +}); + +test('host command dispatch passes the lifecycle to pick before installation', async () => { + writeKitConfig(home, offlineKitConfig()); + const sentinel = new Error('injected host lifecycle reached'); + let calls = 0; + await assert.rejects(host.run({ + flags: { host: 'codex', yes: true, 'aqe-provider': 'none' }, + positionals: ['pick'], + pkgRoot: process.cwd(), + deps: { hostLifecycle: { + installState: async () => { calls++; throw sentinel; }, + install: async () => { throw new Error('installer reached'); }, + collectFacts: async () => { throw new Error('collector reached'); }, + } }, + }), error => error === sentinel); + assert.equal(calls, 1); +}); diff --git a/tests/kit/live-check-evidence.test.mjs b/tests/kit/live-check-evidence.test.mjs index f23f385b..7b010b60 100644 --- a/tests/kit/live-check-evidence.test.mjs +++ b/tests/kit/live-check-evidence.test.mjs @@ -7,6 +7,7 @@ import { test } from 'node:test'; import assert from 'node:assert/strict'; import fs from 'node:fs'; import path from 'node:path'; +import { DatabaseSync } from 'node:sqlite'; import { sandboxHome, assertSandboxed, snapshot, assertUnchanged, captureLog, rmrf, sandboxProject, writeKitConfig, offlineKitConfig, fakeGlobalRoot, @@ -18,9 +19,11 @@ const paths = await import('../../src/lib/paths.mjs'); const evidence = await import('../../src/lib/live-check-evidence.mjs'); const { writeEvidence } = await import('../../src/lib/evidence.mjs'); const aqeSection = (await import('../../src/commands/status/sections/aqe.mjs')).default; +const projectMemorySection = (await import('../../src/commands/status/sections/project-memory.mjs')).default; const { SYNC_STEPS } = await import('../../src/commands/sync.mjs'); assertSandboxed(paths, HOME); -paths._setGlobalRootForTest(fakeGlobalRoot(HOME, { ruflo: '9.9.9', 'agentic-qe': '9.9.9' })); +const GLOBAL_ROOT = fakeGlobalRoot(HOME, { ruflo: '9.9.9', 'agentic-qe': '9.9.9' }); +paths._setGlobalRootForTest(GLOBAL_ROOT); const PROJECT = sandboxProject('ak-live-evidence'); const NOW = Date.parse('2026-09-26T12:00:00Z'); @@ -33,6 +36,70 @@ test('the store lives under the kit state directory, one file per check', () => assert.deepEqual([...evidence.LIVE_CHECK_IDS].sort(), ['aqe-embedding', 'deja-vu', 'mcp', 'memory', 'providers', 'security']); assert.equal(evidence.LIVE_CHECK_TTL_MS, 24 * 3600_000); + assert.deepEqual(evidence.RECORDED_CHECK_IDS, [...evidence.LIVE_CHECK_IDS, 'memory-routes']); +}); + +test('routing evidence is separate from the generic memory round trip and bound to CLI version and platform', () => { + reset(); + const key = (routingVersion, platform) => evidence.liveCheckInputsKey('memory-routes', { routingVersion, platform }); + assert.notEqual(key('3.45.0', 'darwin'), key('3.45.1', 'darwin')); + assert.notEqual(key('3.45.0', 'darwin'), key('3.45.0', 'linux')); + evidence.recordLiveCheck({ id: 'memory', status: 'passed', source: 'status-refresh-live', inputsKey: 'generic' }, { now: NOW }); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: key('3.45.0', 'darwin'), now: NOW }), null); + evidence.recordLiveCheck({ id: 'memory-routes', status: 'passed', source: 'status-refresh-live', + inputsKey: key('3.45.0', 'darwin') }, { now: NOW }); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: key('3.45.1', 'darwin'), now: NOW }).invalidated, true); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: key('3.45.0', 'linux'), now: NOW }).invalidated, true); +}); + +test('the routing key follows the installed CLI even when the wrapper version stays fixed', (t) => { + const root = fs.mkdtempSync(path.join(HOME, 'route-version-')); + t.after(() => { paths._setGlobalRootForTest(GLOBAL_ROOT); rmrf(root); }); + const wrapper = path.join(root, 'ruflo'); + const cli = path.join(wrapper, 'node_modules', '@claude-flow', 'cli'); + fs.mkdirSync(cli, { recursive: true }); + fs.writeFileSync(path.join(wrapper, 'package.json'), JSON.stringify({ version: '3.45.0' })); + const setCli = (version) => fs.writeFileSync(path.join(cli, 'package.json'), JSON.stringify({ version })); + paths._setGlobalRootForTest(root); + setCli('3.45.0'); + const before = evidence.liveCheckInputsKey('memory-routes', { platform: 'darwin' }); + setCli('3.45.1'); + assert.notEqual(evidence.liveCheckInputsKey('memory-routes', { platform: 'darwin' }), before); +}); + +test('two-store row lowers only for applicable routing proof and warns after failure or upgrade', async (t) => { + reset(); + const cwd = sandboxProject('ak-route-row'); + t.after(() => rmrf(cwd)); + const dir = path.join(cwd, '.swarm'); + fs.mkdirSync(dir); + for (const name of ['memory.db', 'agentdb-memory.db']) { + const db = new DatabaseSync(path.join(dir, name)); + db.exec('CREATE TABLE memory_entries (status TEXT); INSERT INTO memory_entries VALUES (NULL)'); + db.close(); + } + const row = async (routingVersion = '3.45.0', platform = 'darwin', now = NOW) => + (await projectMemorySection.collect({ cwd, rufloVersion: routingVersion, platform, now })) + .find((item) => /two project memory stores/.test(item.message)); + const key = evidence.liveCheckInputsKey('memory-routes', { routingVersion: '3.45.0', platform: 'darwin' }); + assert.equal((await row()).level, 'warn'); + evidence.recordLiveCheck({ id: 'memory', status: 'passed', source: 'status-refresh-live', inputsKey: 'generic' }, { now: NOW }); + assert.equal((await row()).level, 'warn'); + evidence.recordLiveCheck({ id: 'memory-routes', status: 'passed', source: 'status-refresh-live', inputsKey: key }, { now: NOW }); + const beforeRead = snapshot(HOME); + assert.equal((await row()).level, 'info'); + assert.match((await row()).message, /existing-corpus access unverified/); + assertUnchanged(beforeRead, HOME, 'ordinary project-memory status reads neither probe nor write'); + assert.equal((await row('3.45.1')).level, 'warn'); + assert.equal((await row('3.45.0', 'linux')).level, 'warn'); + assert.equal((await row(null)).level, 'warn'); + assert.equal((await row('3.45.0', 'darwin', NOW + 2 * evidence.LIVE_CHECK_TTL_MS)).level, 'info'); + evidence.recordLiveCheck({ id: 'memory-routes', status: 'inconclusive', source: 'status-refresh-live', inputsKey: key }, { now: NOW + 1000 }); + assert.equal((await row('3.45.0', 'darwin', NOW + 2000)).level, 'warn'); + evidence.recordLiveCheck({ id: 'memory', status: 'passed', source: 'status-refresh-live', inputsKey: 'generic' }, { now: NOW + 3000 }); + assert.equal((await row('3.45.0', 'darwin', NOW + 4000)).level, 'warn'); + fs.writeFileSync(path.join(evidence.liveCheckDir(), 'memory-routes.json'), '{corrupt'); + assert.equal((await row()).level, 'warn'); }); test('a recorded result reads back with its source and age', () => { diff --git a/tests/kit/live-checks.test.mjs b/tests/kit/live-checks.test.mjs index 75a2b1e0..ad5db1a5 100644 --- a/tests/kit/live-checks.test.mjs +++ b/tests/kit/live-checks.test.mjs @@ -12,6 +12,7 @@ import assert from 'node:assert/strict'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; +import { DatabaseSync } from 'node:sqlite'; import { fileURLToPath } from 'node:url'; import { sandboxHome, assertSandboxed, snapshot, assertUnchanged, captureLog, rmrf, @@ -22,13 +23,15 @@ const HOME = sandboxHome('ak-live-checks'); const paths = await import('../../src/lib/paths.mjs'); const live = await import('../../src/lib/live-checks.mjs'); const evidence = await import('../../src/lib/live-check-evidence.mjs'); +const projectMemorySection = (await import('../../src/commands/status/sections/project-memory.mjs')).default; const { refreshRequestFromFlags, cliRefreshStages } = await import('../../src/lib/refresh.mjs'); const status = await import('../../src/commands/status.mjs'); const { loadKitConfig } = await import('../../src/lib/config.mjs'); assertSandboxed(paths, HOME); const PROJECT = sandboxProject('ak-live-checks'); -paths._setGlobalRootForTest(fakeGlobalRoot(HOME, { ruflo: '9.9.9' })); +const GLOBAL_ROOT = fakeGlobalRoot(HOME, { ruflo: '9.9.9' }); +paths._setGlobalRootForTest(GLOBAL_ROOT); const PKG_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..'); const seedHome = (cfg = offlineKitConfig()) => { @@ -93,7 +96,7 @@ test('each check carries its timeout and the evidence id its result is remembere assert.equal(entry('learning').evidenceId, null); assert.equal(entry('harvest').evidenceId, null); assert.equal(entry('aqe').evidenceId, null, 'the aqe proof remembers only its embedding request, itself'); - assert.equal(entry('memory-routes').evidenceId, 'memory'); + assert.equal(entry('memory-routes').evidenceId, 'memory-routes'); }); test('without --only the quick checks that apply run; mcp runs whenever Codex is enabled', async () => { @@ -219,9 +222,11 @@ test('a named check that does not apply says on its line that its result is not assert.match(out, /^⚠ {2}mcp failed \(20 ms\) — effective Codex MCP inventory unavailable; not remembered: this check does not apply to your setup$/m); assert.match(out, /^✓ security passed \(30 ms\)$/m, 'a check that applies says nothing more'); const skipped = await runStatus({ refresh: 'live', only: ['deja-vu'] }, [ - { id: 'deja-vu', status: 'passed', reason: null, elapsedMs: 2, entries: [], applies: false }, + { id: 'deja-vu', status: 'skipped', reason: 'disabled and unowned', elapsedMs: 2, entries: [], applies: false }, ]); - assert.match(skipped.out, /^✓ deja-vu passed \(2 ms\) — not remembered: this check does not apply to your setup$/m); + assert.equal(skipped.code, 1, 'a named skipped proof did not pass'); + assert.match(skipped.out, /Running live checks \(\d+ ms\): 1 skipped/); + assert.match(skipped.out, /^⚠ {2}deja-vu skipped \(2 ms\) — disabled and unowned; not remembered: this check does not apply to your setup$/m); }); test('one line per check; with --only each check\'s own lines are indented under it', async () => { @@ -585,25 +590,203 @@ test('a failed check is remembered for status with its first failure as the reas assert.equal(got.reason, '@claude-flow/security missing'); }); -test('the memory-routes proof is remembered as the memory check; learning and harvest are not remembered', async () => { +test('the memory-routes proof is remembered separately; learning and harvest are not remembered', async () => { seedHome(); rmrf(evidence.liveCheckDir()); await runOnly(['memory-routes', 'learning', 'harvest']); - const got = evidence.readLiveCheck('memory', {}); + const got = evidence.readLiveCheck('memory-routes', {}); assert.equal(got.status, 'failed'); assert.equal(got.source, 'status-refresh-live'); - assert.deepEqual(fs.readdirSync(evidence.liveCheckDir()), ['memory.json']); + assert.equal(evidence.readLiveCheck('memory', {}).status, 'failed'); + assert.deepEqual(fs.readdirSync(evidence.liveCheckDir()), ['memory-routes.json', 'memory.json']); +}); + +test('a failed or timed-out routing run replaces a previous pass; a later generic memory pass cannot revive it', async () => { + seedHome(); + rmrf(evidence.liveCheckDir()); + const cfg = offlineKitConfig(); + const route = (run, timeoutMs = 50) => ({ id: 'memory-routes', evidenceId: 'memory-routes', + timeoutMs, applies: () => true, run }); + const generic = { id: 'memory', evidenceId: 'memory', timeoutMs: 50, applies: () => true, + run: async () => true }; + await live.runLiveChecks({ cfg, cwd: PROJECT, checks: [route(async () => ({ status: 'passed' }))] }); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'passed'); + await live.runLiveChecks({ cfg, cwd: PROJECT, checks: [route(async () => false)] }); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'failed'); + await live.runLiveChecks({ cfg, cwd: PROJECT, checks: [route(() => new Promise(() => {}), 10)], graceMs: 1 }); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'inconclusive'); + await live.runLiveChecks({ cfg, cwd: PROJECT, checks: [generic] }); + assert.equal(evidence.readLiveCheck('memory').status, 'passed'); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'inconclusive'); +}); + +test('a route timeout after the CLI proof keeps the generic memory pass', async () => { + seedHome(); + rmrf(evidence.liveCheckDir()); + const check = { id: 'memory-routes', evidenceId: 'memory-routes', timeoutMs: 10, applies: () => true, + run: ({ onCliOutcome }) => { + onCliOutcome({ status: 'passed', reason: null }); + return new Promise(() => {}); + } }; + const [result] = await live.runLiveChecks({ cfg: offlineKitConfig(), cwd: PROJECT, checks: [check], graceMs: 1 }); + assert.equal(result.status, 'inconclusive'); + assert.equal(evidence.readLiveCheck('memory').status, 'passed'); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'inconclusive'); +}); + +test('a CLI upgrade during the route check cannot attribute the old observation to the new version', async (t) => { + seedHome(); + rmrf(evidence.liveCheckDir()); + const root = fs.mkdtempSync(path.join(HOME, 'routing-upgrade-')); + const cli = path.join(root, 'ruflo', 'node_modules', '@claude-flow', 'cli'); + fs.mkdirSync(cli, { recursive: true }); + const setVersion = (version) => fs.writeFileSync(path.join(cli, 'package.json'), JSON.stringify({ version })); + t.after(() => { paths._setGlobalRootForTest(GLOBAL_ROOT); rmrf(root); }); + paths._setGlobalRootForTest(root); + setVersion('3.45.0'); + const oldKey = evidence.liveCheckInputsKey('memory-routes'); + const check = { id: 'memory-routes', evidenceId: 'memory-routes', timeoutMs: 50, applies: () => true, + run: async ({ onCliOutcome }) => { + onCliOutcome({ status: 'passed', reason: null }); + setVersion('3.45.1'); + return { status: 'passed', reason: null }; + } }; + const [result] = await live.runLiveChecks({ cfg: offlineKitConfig(), cwd: PROJECT, checks: [check] }); + const currentKey = evidence.liveCheckInputsKey('memory-routes'); + assert.notEqual(currentKey, oldKey); + assert.equal(result.status, 'inconclusive'); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: oldKey }).status, 'inconclusive'); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: oldKey }).invalidated, false); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: currentKey }).invalidated, true); + assert.equal(evidence.readLiveCheck('memory').status, 'passed'); + + const project = sandboxProject('ak-route-upgrade-row'); + t.after(() => rmrf(project)); + const swarm = path.join(project, '.swarm'); + fs.mkdirSync(swarm); + for (const name of ['memory.db', 'agentdb-memory.db']) { + const db = new DatabaseSync(path.join(swarm, name)); + db.exec('CREATE TABLE memory_entries (status TEXT); INSERT INTO memory_entries VALUES (NULL)'); + db.close(); + } + const row = async (version) => (await projectMemorySection.collect({ cwd: project, + rufloVersion: version })).find((item) => /two project memory stores/.test(item.message)); + assert.equal((await row('3.45.0')).level, 'warn'); + assert.equal((await row('3.45.1')).level, 'warn'); }); test('a skipped deja-vu proof is not remembered as a pass', async () => { seedHome(); rmrf(evidence.liveCheckDir()); const [r] = await runOnly(['deja-vu']); - assert.equal(r.status, 'passed'); + assert.equal(r.status, 'skipped'); + assert.equal(r.reason, 'disabled and unowned'); assert.match(texts(r), /deja-vu disabled and unowned — skipped/); assert.equal(evidence.readLiveCheck('deja-vu', {}), null); }); +test('a skipped applicable check never writes conformance evidence', async () => { + seedHome(); + rmrf(evidence.liveCheckDir()); + const [result] = await live.runLiveChecks({ cfg: offlineKitConfig(), cwd: PROJECT, + checks: [{ id: 'deja-vu', evidenceId: 'deja-vu', applies: () => true, + run: async () => ({ status: 'skipped', reason: 'not installed' }) }] }); + assert.equal(result.status, 'skipped'); + assert.equal(evidence.readLiveCheck('deja-vu', {}), null); +}); + +test('learning removes only its call-owned folder after success, missing artifacts, thrown runner, and abort', async (t) => { + const tmpRoot = privateRoot(t); + const unrelated = path.join(tmpRoot, 'unrelated'); + fs.mkdirSync(unrelated); + const cases = [ + async (_cmd, _args, { cwd }) => { + const neural = path.join(cwd, '.claude-flow', 'neural'); + fs.mkdirSync(neural, { recursive: true }); + fs.writeFileSync(path.join(neural, 'stats.json'), '{"patternsLearned":1}'); + fs.writeFileSync(path.join(neural, 'patterns.json'), '[{"id":"one"}]'); + return { code: 0, stdout: '', stderr: '' }; + }, + async () => ({ code: 0, stdout: '', stderr: '' }), + async () => { throw new Error('runner failed'); }, + async () => { throw new DOMException('aborted', 'AbortError'); }, + ]; + for (const [i, runner] of cases.entries()) { + const { result } = await captureLog(() => live.verifyLearning({ tmpRoot, runner })); + assert.equal(result, i === 0, `case ${i}`); + assert.deepEqual(fs.readdirSync(tmpRoot), ['unrelated'], `case ${i} left a call-owned folder`); + } +}); + +test('memory removes its call-owned folder even when final purge throws', async (t) => { + const tmpRoot = privateRoot(t); + fs.mkdirSync(path.join(tmpRoot, 'unrelated')); + const runner = async (_cmd, args) => { + if (args[1] === 'init') return { code: 0, stdout: '', stderr: '' }; + if (args[1] === 'store') return { code: 1, stdout: '', stderr: 'store failed' }; + throw new Error('purge failed'); + }; + await assert.rejects(captureLog(() => live.verifyMemory({ tmpRoot, runner, haveCmd: async () => true })), /purge failed/); + assert.deepEqual(fs.readdirSync(tmpRoot), ['unrelated']); +}); + +test('memory removes only its call-owned folder when init throws or is aborted', async (t) => { + const tmpRoot = privateRoot(t); + fs.mkdirSync(path.join(tmpRoot, 'unrelated')); + for (const error of [new Error('runner failed'), new DOMException('aborted', 'AbortError')]) { + const { result } = await captureLog(() => live.verifyMemory({ tmpRoot, + runner: async () => { throw error; }, haveCmd: async () => true })); + assert.equal(result, false); + assert.deepEqual(fs.readdirSync(tmpRoot), ['unrelated']); + } +}); + +test('memory removes its call-owned folder after a successful mocked CLI round trip', async (t) => { + const tmpRoot = privateRoot(t); + fs.mkdirSync(path.join(tmpRoot, 'unrelated')); + let db; + let storedValue; + const runner = async (_cmd, args, { cwd }) => { + if (args[1] === 'init') { + fs.mkdirSync(path.join(cwd, '.swarm')); + db = new DatabaseSync(path.join(cwd, '.swarm', 'memory.db')); + db.exec('CREATE TABLE memory_entries (namespace TEXT, key TEXT)'); + } else if (args[1] === 'store') { + db.prepare('INSERT INTO memory_entries VALUES (?, ?)').run(args[args.indexOf('-n') + 1], args[args.indexOf('-k') + 1]); + storedValue = args[args.indexOf('--value') + 1]; + } else if (args[1] === 'retrieve') { + return { code: 0, stdout: storedValue, stderr: '' }; + } else if (args[1] === 'purge') { + db.exec('DELETE FROM memory_entries'); + db.close(); + } + return { code: 0, stdout: '', stderr: '' }; + }; + const { result } = await captureLog(() => live.verifyMemory({ + tmpRoot, runner, haveCmd: async () => true, observeRoutes: false, + })); + assert.equal(result, true); + assert.deepEqual(fs.readdirSync(tmpRoot), ['unrelated']); +}); + +test('harvest removes only its call-owned folder on success, nonzero result, thrown runner, and abort', async (t) => { + const tmpRoot = privateRoot(t); + fs.mkdirSync(path.join(tmpRoot, 'unrelated')); + for (const mode of ['success', 'nonzero', 'throw', 'abort']) { + const dirs = []; + const runner = async (_cmd, _args, { cwd }) => { + dirs.push(cwd); + if (mode === 'throw') throw new Error('runner failed'); + if (mode === 'abort') throw new DOMException('aborted', 'AbortError'); + return { code: mode === 'nonzero' ? 1 : 0, stdout: '', stderr: '' }; + }; + const { result } = await captureLog(() => live.verifyHarvest({ tmpRoot, runner, haveCmd: async () => true })); + assert.equal(result, mode === 'success', mode); + assert.ok(dirs.length >= 1 && dirs.every((dir) => path.dirname(dir) === tmpRoot), mode); + assert.deepEqual(fs.readdirSync(tmpRoot), ['unrelated'], mode); + } +}); + test('an aqe proof stopped before the embedding request remembers no embedding result', async () => { seedHome(); rmrf(evidence.liveCheckDir()); diff --git a/tests/kit/maintenance-discovery-orchestrator.test.mjs b/tests/kit/maintenance-discovery-orchestrator.test.mjs index 1d72f60c..c9d92ba2 100644 --- a/tests/kit/maintenance-discovery-orchestrator.test.mjs +++ b/tests/kit/maintenance-discovery-orchestrator.test.mjs @@ -145,6 +145,99 @@ test('pause and resume: a paused source keeps its checkpoint and completes once assert.equal(published.scanState, 'published'); }); +test('a paused scan survives restart without publishing partial evidence, then resumes', async (t) => { + const dir = fixture(t); + const root = fixture(t); + buildLargeTree(root, { dirs: 5, filesPerDir: 4 }); + const first = control(root, dir); + const [checkpointed] = await first.orchestrator.start({ sourceIds: [SOURCE.sourceId], maxSlices: 1 }); + assert.equal(checkpointed.scanState, 'checkpointed'); + const paused = first.orchestrator.pause({ sourceId: SOURCE.sourceId }); + assert.equal(paused.state, 'paused'); + assert.ok(paused.visited > 0); + const [summary] = first.historyStore.list(); + assert.equal(summary.state, 'paused'); + assert.equal(summary.visited, paused.visited); + assert.equal(summary.completedPartitions, paused.completedPartitions); + assert.equal(summary.pendingPartitions, paused.pendingPartitions); + assert.equal(summary.completedAt, null); + assert.ok(summary.recordedAt); + assert.equal(first.lastGoodStore.current().length, 0); + + const restarted = control(root, dir); + const [row] = restarted.orchestrator.coverage(); + assert.equal(row.state, 'paused'); + assert.equal(row.visited, paused.visited); + assert.equal(row.completedPartitions, paused.completedPartitions); + assert.equal(row.pendingPartitions, paused.pendingPartitions); + assert.equal(row.lastCompletedAt, null); + assert.equal(restarted.checkpointStore.list().length, 1); + const [published] = await restarted.orchestrator.resume({ sourceIds: [SOURCE.sourceId] }); + assert.equal(published.scanState, 'published'); + assert.equal(restarted.orchestrator.coverage()[0].state, 'complete'); + assert.equal(restarted.checkpointStore.list().length, 0); + assert.equal(restarted.lastGoodStore.current().length, 1); + assert.equal(control(root, dir).orchestrator.coverage()[0].state, 'complete'); +}); + +test('a confirmed stop after pause prevents old paused state from returning', async (t) => { + const dir = fixture(t); + const root = fixture(t); + buildLargeTree(root); + const first = control(root, dir); + await first.orchestrator.start({ sourceIds: [SOURCE.sourceId], maxSlices: 1 }); + first.orchestrator.pause({ sourceId: SOURCE.sourceId }); + first.orchestrator.stop({ sourceId: SOURCE.sourceId, confirmed: true }); + assert.equal(first.checkpointStore.list().length, 0); + assert.equal(control(root, dir).orchestrator.coverage()[0].state, 'not-scanned'); +}); + +test('pause does not claim durable state when its summary write fails', async (t) => { + const dir = fixture(t); + const root = fixture(t); + buildLargeTree(root); + const first = control(root, dir); + await first.orchestrator.start({ sourceIds: [SOURCE.sourceId], maxSlices: 1 }); + const historyStore = { ...first.historyStore, recordSummary: () => { throw new Error('history unavailable'); } }; + const second = control(root, dir, { historyStore }); + assert.throws(() => second.orchestrator.pause({ sourceId: SOURCE.sourceId }), /history unavailable/); + assert.equal(second.orchestrator.coverage()[0].state, 'scanning'); + assert.equal(second.historyStore.list().length, 0); + assert.equal(second.checkpointStore.list().length, 1); +}); + +test('pause refuses when no history store can record the boundary', async (t) => { + const dir = fixture(t); + const root = fixture(t); + buildLargeTree(root); + const first = control(root, dir); + await first.orchestrator.start({ sourceIds: [SOURCE.sourceId], maxSlices: 1 }); + const restarted = control(root, dir, { historyStore: null }); + assert.throws(() => restarted.orchestrator.pause({ sourceId: SOURCE.sourceId }), /history unavailable/); + assert.equal(restarted.orchestrator.coverage()[0].state, 'scanning'); +}); + +test('paused history restores only its matching environment and loses to a later failure at the same millisecond', (t) => { + const dir = fixture(t); + const root = fixture(t); + const fixed = Date.parse('2026-09-08T12:00:00.000Z'); + const first = control(root, dir, { now: () => fixed }); + first.historyStore.recordSummary({ scanId: 'one', sourceId: SOURCE.sourceId, + environmentId: 'other', state: 'paused', completedAt: null, + recordedAt: new Date(fixed).toISOString(), visited: 7 }); + assert.equal(control(root, dir).orchestrator.coverage()[0].state, 'not-scanned'); + first.historyStore.recordSummary({ scanId: 'two', sourceId: SOURCE.sourceId, + environmentId: SOURCE.environmentId, state: 'paused', completedAt: null, + recordedAt: new Date(fixed).toISOString(), visited: 8 }); + assert.equal(control(root, dir).orchestrator.coverage()[0].state, 'paused'); + first.historyStore.recordSummary({ scanId: 'two', sourceId: SOURCE.sourceId, + environmentId: SOURCE.environmentId, state: 'failed', completedAt: new Date(fixed).toISOString(), + visited: 9, limitingReason: 'io-failure' }); + const [row] = control(root, dir).orchestrator.coverage(); + assert.equal(row.state, 'failed'); + assert.equal(row.visited, 9); +}); + test('MNT-DSC-018: stop previews affected work before confirming, then removes the active scan and retains history', async (t) => { const controlDir = fixture(t); const root = fixture(t); @@ -662,3 +755,18 @@ test('scan history rolls over independently at ten records per source and surviv [2, 3, 4, 5, 6, 7, 8, 9, 10, 11]); } }); + +test('clearHistory keeps a paused summary while its continuation is open', (t) => { + const dir = fixture(t); + const checkpoints = createCheckpointStore(path.join(dir, 'checkpoints'), { fsImpl: fs }); + const history = createScanHistoryStore(dir, { fsImpl: fs }); + const scanId = 'paused-scan'; + checkpoints.write({ scanId, sourceId: SOURCE.sourceId, environmentId: SOURCE.environmentId, + scanEpoch: 1, completedPartitions: [], pendingPartitions: [], workCounts: { visited: 1 }, + sourceStamps: [], createdAt: new Date().toISOString() }); + history.recordSummary({ scanId, sourceId: SOURCE.sourceId, environmentId: SOURCE.environmentId, + state: 'paused', completedAt: null, recordedAt: new Date().toISOString(), visited: 1 }); + assert.deepEqual(history.clearHistory(), { removed: 0, kept: 1 }); + checkpoints.remove(scanId); + assert.deepEqual(history.clearHistory(), { removed: 1, kept: 0 }); +}); diff --git a/tests/kit/maintenance-management-service.test.mjs b/tests/kit/maintenance-management-service.test.mjs index 0a9da99f..562e49b5 100644 --- a/tests/kit/maintenance-management-service.test.mjs +++ b/tests/kit/maintenance-management-service.test.mjs @@ -696,6 +696,35 @@ test('scan lifecycle: pauseScan and resumeScan accept a bare sourceId (symmetric } }); +test('paused collection root survives a service restart in Discovery and the Inventory partial banner', async (t) => { + const controlRoot = fixtureRoot(t); + const root = fixtureRoot(t); + fs.writeFileSync(path.join(root, 'root-file.txt'), 'x'); + for (let i = 0; i < 5; i += 1) { + const child = path.join(root, `project-${i}`); + fs.mkdirSync(path.join(child, '.git'), { recursive: true }); + fs.writeFileSync(path.join(child, 'file.txt'), 'x'); + } + const sourceId = opaqueId('src', { kind: 'collection-root', root }, INSTALLATION_KEY); + const discovery = { collectionRoots: [{ root, sourceId, maxDepth: null, includeNetwork: false }] }; + const first = buildHarness(t, { controlRoot, discovery }); + await first.service.startScan({ sourceId, maxSlices: 1 }); + const paused = first.service.pauseScan({ sourceId }); + const pausedRow = paused.coverage.find((entry) => entry.sourceId === sourceId); + assert.equal(pausedRow.state, 'paused'); + assert.ok(pausedRow.visited > 0); + + const restarted = buildHarness(t, { controlRoot, discovery }); + const row = restarted.service.discovery().coverage.find((entry) => entry.sourceId === sourceId); + assert.equal(row.state, 'paused'); + assert.equal(row.visited, pausedRow.visited); + assert.equal(row.label, 'Collection root'); + await restarted.service.refreshInventory(); + const page = restarted.service.inventory({}); + assert.equal(page.partialSources.total, 5, 'the paused root joins four unscanned automatic roots'); + assert.ok(page.partialSources.entries.some((entry) => entry.sourceId === sourceId && entry.state === 'paused')); +}); + test('DSC: setAutomaticSource, addExclusion, and removeExclusion round-trip through kit.json', async (t) => { const h = buildHarness(t); await h.service.setAutomaticSource({ sourceId: 'ollama', enabled: false }); diff --git a/tests/kit/maintenance-refresh-injection.test.mjs b/tests/kit/maintenance-refresh-injection.test.mjs new file mode 100644 index 00000000..03a363fe --- /dev/null +++ b/tests/kit/maintenance-refresh-injection.test.mjs @@ -0,0 +1,78 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { spawnSync } from 'node:child_process'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..'); + +// The Maintenance service statically imports footprint/index.mjs, so blocking +// that module's import would reject a valid injected run. Instrument its real +// constructor instead; the management facade is lazy and must never import. +const guard = ` +export async function resolve(specifier, context, nextResolve) { + if (specifier.endsWith('/maintenance/management/service.mjs')) { + throw new Error('real management facade imported'); + } + return nextResolve(specifier, context); +} +export async function load(url, context, nextLoad) { + const loaded = await nextLoad(url, context); + if (!url.endsWith('/footprint/index.mjs')) return loaded; + const needle = ' const collect = {'; + const source = String(loaded.source); + if (source.split(needle).length !== 2) throw new Error('collector constructor guard no longer matches'); + return { ...loaded, source: source.replace(needle, + " throw new Error('real footprint collector constructed');\\n" + needle) }; +} +`; + +const child = ` +import assert from 'node:assert/strict'; +import { register } from 'node:module'; +import { pathToFileURL } from 'node:url'; +register('data:text/javascript,' + encodeURIComponent(process.env.AK_REFRESH_GUARD)); +const { run } = await import(pathToFileURL(process.env.AK_MAINTAIN_MODULE).href); +const calls = []; +const refreshStages = { + maintenance: async () => { calls.push('maintenance'); return { ok: true }; }, + inventory: async () => { calls.push('inventory'); return { ok: true }; }, + local: async () => { calls.push('local'); return { ok: true }; }, +}; +const service = { async report() { calls.push('report'); return { mode: 'read-only' }; } }; +const lines = []; +const originalLog = console.log; +console.log = (...args) => lines.push(args.join(' ')); +let code; +try { + code = await run({ flags: { json: true, refresh: '' }, positionals: ['report'], + deps: { refreshStages, service } }); +} finally { + console.log = originalLog; +} +assert.equal(code, 0, lines.join('\\n')); +assert.deepEqual(calls, ['maintenance', 'inventory', 'local', 'report']); +const result = JSON.parse(lines.join('\\n')); +assert.equal(result.mode, 'read-only'); +assert.equal(result.refresh.ok, true); +assert.deepEqual(result.refresh.stages.map(({ id }) => id), ['maintenance', 'inventory', 'local']); +originalLog('injected refresh used no real constructors'); +`; + +test('injected maintain refresh constructs no real collector or management facade', (t) => { + const home = tempDir('ak-maintain-injected-home', t); + const project = tempDir('ak-maintain-injected-project', t); + const result = spawnSync(process.execPath, ['--input-type=module', '--eval', child], { + cwd: project, + env: spawnEnv(home, { + AK_MAINTAIN_MODULE: path.join(ROOT, 'src/commands/maintain.mjs'), + AK_REFRESH_GUARD: guard, + }), + encoding: 'utf8', + }); + assert.ifError(result.error); + assert.equal(result.status, 0, `stdout: ${result.stdout}\nstderr: ${result.stderr}`); + assert.match(result.stdout, /injected refresh used no real constructors/); +}); diff --git a/tests/kit/models-command.test.mjs b/tests/kit/models-command.test.mjs index b4774fd4..fc815f6f 100644 --- a/tests/kit/models-command.test.mjs +++ b/tests/kit/models-command.test.mjs @@ -40,6 +40,15 @@ test('models status is a cache-only read', async () => { assert.equal(JSON.parse(result.output).inventory.snapshotId, 'models:test'); }); +test('models rejects an unknown verb with a populated snapshot store', async () => { + const result = await capture(() => run({ + flags: { json: true }, positionals: ['bogus'], + deps: { loadConfig: () => cfg, readStore: () => store }, + })); + assert.equal(result.code, 2); + assert.match(result.output, /usage: ak models status\|refresh\|diff\|explain\|plan/); +}); + test('models status host filter rejects unknown owners and narrows evidence', async () => { const invalid = await capture(() => run({ flags: { json: true, host: 'unknown' }, positionals: ['status'], diff --git a/tests/kit/persisted-file-identity.test.mjs b/tests/kit/persisted-file-identity.test.mjs new file mode 100644 index 00000000..087f80f8 --- /dev/null +++ b/tests/kit/persisted-file-identity.test.mjs @@ -0,0 +1,229 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; + +import { tempDir } from './helpers/temp-dir.mjs'; +import { planPartitions, partitionDrifted } from '../../src/lib/maintenance/discovery/partitions.mjs'; +import { inspectHookTarget, atomicReplaceHookTarget } from '../../src/lib/hook-remediation/fs-port.mjs'; +import { JsonlTailer } from '../../src/lib/live/jsonl-tailer.mjs'; +import { createHostHealthSnapshot } from '../../src/lib/host-health-evidence.mjs'; +import { HOOK_HEAL_RECEIPT_SCHEMA, readHookReceipt, sealHookReceipt } from '../../src/lib/hook-remediation/store.mjs'; +import { inspectHostAlignment } from '../../src/lib/host-alignment.mjs'; +import { TranscriptStreams } from '../../src/lib/live/transcript-streams.mjs'; + +const A = 9007199254740992n; +const B = 9007199254740993n; + +test('Discovery stamp survives JSON with adjacent 64-bit inodes and fractional mtime', (t) => { + const root = tempDir('ak-partition-id', t); + const lstatSync = fs.lstatSync; + let ino = A; + const fsImpl = { ...fs, lstatSync(target, options) { + const stat = lstatSync(target, options); + stat.ino = options?.bigint ? ino : Number(ino); + return stat; + } }; + const stamp = planPartitions(root, { fsImpl }).partitions[0].sourceStamps[0]; + assert.equal(stamp.ino, '9007199254740992'); + assert.equal(typeof stamp.mtimeMs, 'number'); + assert.equal(stamp.mtimeMs, lstatSync(root).mtimeMs); + const restored = JSON.parse(JSON.stringify(stamp)); + assert.equal(partitionDrifted({ sourceStamps: [{ ...restored, ino: Number(A) }] }, { fsImpl }), true, + 'unsafe legacy number cannot authorize a match'); + ino = B; + assert.equal(partitionDrifted({ sourceStamps: [restored] }, { fsImpl }), true); + for (const malformed of ['-1', '01', -1]) { + assert.equal(partitionDrifted({ sourceStamps: [{ ...restored, ino: malformed }] }, { fsImpl }), true); + } + ino = 42n; + const safeStamp = planPartitions(root, { fsImpl }).partitions[0].sourceStamps[0]; + assert.equal(partitionDrifted({ sourceStamps: [{ ...safeStamp, ino: 42 }] }, { fsImpl }), false); +}); + +test('hook target parent identity survives JSON and refuses adjacent replacement', (t) => { + const root = tempDir('ak-hook-parent-id', t); + const file = path.join(root, 'hook.json'); + fs.writeFileSync(file, '{}'); + const statSync = fs.statSync; + let ino = A; + const fsImpl = { ...fs, statSync(target, options) { + const stat = statSync(target, options); + if (target === root) stat.ino = options?.bigint ? ino : Number(ino); + return stat; + } }; + // Exercise the parent guard on every OS before any mutation is allowed. + // This injected branch is not evidence of native Windows atomic replacement. + fsImpl.openSync = (target, flags) => { + assert.equal(flags, fs.constants.O_RDONLY | (fs.constants.O_NOFOLLOW ?? 0)); + return fs.openSync(target, flags); + }; + fsImpl.renameSync = () => assert.fail('parent drift must refuse before rename'); + const options = { fsImpl, platform: 'linux' }; + const snapshot = inspectHookTarget(file, root, options); + assert.equal(snapshot.parent.ino, '9007199254740992'); + const savedParent = JSON.parse(JSON.stringify(snapshot.parent)); + ino = B; + assert.throws(() => atomicReplaceHookTarget({ ...snapshot, parent: savedParent }, Buffer.from('[]'), undefined, options), + /target parent changed/); + assert.equal(fs.readFileSync(file, 'utf8'), '{}'); +}); + +test('Windows hook mutation refuses before accessing the filesystem', (t) => { + const root = tempDir('ak-hook-windows-refusal', t); + const file = path.join(root, 'hook.json'); + fs.writeFileSync(file, '{}'); + const snapshot = inspectHookTarget(file, root, { platform: 'win32' }); + const fsImpl = { lstatSync: () => assert.fail('unsupported mutation must refuse before inspection') }; + assert.throws(() => atomicReplaceHookTarget(snapshot, Buffer.from('[]'), undefined, { fsImpl, platform: 'win32' }), + /hook mutation is unsupported on Windows until replace-existing atomicity is proven/); + assert.equal(fs.readFileSync(file, 'utf8'), '{}'); + assert.deepEqual(fs.readdirSync(root), ['hook.json']); +}); + +test('hook snapshot gets identity and fractional mtime from one descriptor stat', (t) => { + const root = tempDir('ak-hook-one-stat', t); + const file = path.join(root, 'hook.json'); + fs.writeFileSync(file, '{}'); + const fstatSync = fs.fstatSync; + let calls = 0; + const fsImpl = { ...fs, fstatSync(fd, options) { + calls++; + assert.equal(options?.bigint, true); + const stat = fstatSync(fd, options); + stat.ino = A; + return stat; + }, lstatSync(target, options) { + const stat = fs.lstatSync(target, options); + if (target === file) stat.ino = A; + return stat; + } }; + const snapshot = inspectHookTarget(file, root, { fsImpl }); + assert.equal(calls, 1); + assert.equal(snapshot.mtimeMs, fs.statSync(file).mtimeMs); +}); + +test('JSONL replacement with a colliding Number inode resets offset and emits new records', (t) => { + const root = tempDir('ak-tailer-id', t); + const file = path.join(root, 'events.jsonl'); + fs.writeFileSync(file, '{"a":1}\n'); + const statSync = fs.statSync; + let ino = A; + t.mock.method(fs, 'statSync', (target, options) => { + const stat = statSync(target, options); + if (target === file) stat.ino = options?.bigint ? ino : Number(ino); + return stat; + }); + const records = []; + const tailer = new JsonlTailer(file, { onRecord: record => records.push(record) }); + tailer.reconcile(); + fs.writeFileSync(file, '{"b":2}\n'); + ino = B; + tailer.reconcile(); + assert.deepEqual(records, [{ a: 1 }, { b: 2 }]); +}); + +test('host health evidence changes when adjacent 64-bit launcher inode changes', (t) => { + const root = tempDir('ak-health-id', t); + const launcher = path.join(root, process.platform === 'win32' ? 'codex.cmd' : 'codex'); + fs.writeFileSync(launcher, 'launcher'); + const observedIds = []; + const statSync = fs.statSync; + let ino = A; + t.mock.method(fs, 'statSync', (target, options) => { + const stat = statSync(target, options); + if (target === launcher) { + assert.equal(options?.bigint, true); + stat.ino = ino; + observedIds.push(ino); + } + return stat; + }); + const snapshot = createHostHealthSnapshot({ env: { PATH: root, PATHEXT: '.CMD' }, inputPaths: () => [] }); + const before = snapshot({ cwd: root, cfg: {} }).key; + assert.deepEqual(observedIds, [A], 'the fixture must be the launcher fingerprinted by this platform'); + ino = B; + assert.notEqual(snapshot({ cwd: root, cfg: {} }).key, before); + assert.deepEqual(observedIds, [A, B]); +}); + +test('hook receipt accepts exact parent IDs but refuses unsafe legacy Numbers', (t) => { + const root = tempDir('ak-receipt-id', t); + fs.chmodSync(root, 0o700); + const id = 'tx-2026-09-29T00-00-00.000Z-0123456789abcdef'; + const dir = path.join(root, id); + fs.mkdirSync(dir, { mode: 0o700 }); + const file = path.join(dir, 'receipt.json'); + const digest = 'a'.repeat(64); + const receipt = { + schemaVersion: HOOK_HEAL_RECEIPT_SCHEMA, id, createdAt: new Date().toISOString(), + status: 'prepared', planDigest: digest, auditId: 'audit', + authorization: { mechanism: 'explicit-action-selection', actionIds: ['one'], trustMutationAuthorized: false }, + actions: [{ + id: 'one', host: 'codex', hostVersion: '1', recipeId: 'recipe', profileId: 'profile', + target: path.join(root, 'hook'), containmentRoot: root, classification: 'safe-automatic', state: 'prepared', + preimage: { sha256: digest, size: 2, mode: 0o600, modeSupported: true, + uid: null, gid: null, specialMode: 0, parent: { realPath: root, dev: '1', ino: '9007199254740993' } }, + postimage: { sha256: digest, size: 2, mode: 0o600, modeSupported: true }, + backup: { relative: path.join('backups', '0000.bin'), sha256: digest, size: 2 }, + }], + }; + // Seed sealed on-disk inputs for reader validation. Durable receipt writes + // have a separate contract and require native directory fsync support. + const seedReceipt = () => fs.writeFileSync(file, JSON.stringify(sealHookReceipt(receipt))); + for (const ino of [String(A), String(B)]) { + receipt.actions[0].preimage.parent.ino = ino; + seedReceipt(); + assert.equal(readHookReceipt(root, id).receipt.actions[0].preimage.parent.ino, ino); + } + receipt.actions[0].preimage.parent.ino = Number(B); + seedReceipt(); + assert.throws(() => readHookReceipt(root, id), /receipt action image is invalid/); + for (const malformed of ['-1', '01', -1]) { + receipt.actions[0].preimage.parent.ino = malformed; + seedReceipt(); + assert.throws(() => readHookReceipt(root, id), /receipt action image is invalid/); + } + receipt.actions[0].preimage.parent.ino = 42; + seedReceipt(); + assert.equal(readHookReceipt(root, id).receipt.actions[0].preimage.parent.ino, 42); +}); + +test('host alignment snapshot serializes adjacent file IDs distinctly', (t) => { + const root = tempDir('ak-align-id', t); + const file = path.join(root, '.mcp.json'); + fs.writeFileSync(file, '{}'); + const lstatSync = fs.lstatSync; + let ino = A; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = lstatSync(target, options); + if (target === file) stat.ino = options?.bigint ? ino : Number(ino); + return stat; + }); + const options = { home: root, codexHome: path.join(root, '.codex'), projectRoots: [root] }; + const before = inspectHostAlignment(options); + assert.equal(before.snapshots.find(s => s.file === file).identity.inode, '9007199254740992'); + ino = B; + assert.notEqual(inspectHostAlignment(options).digest, before.digest); +}); + +test('transcript replay cursor distinguishes adjacent 64-bit file epochs', (t) => { + const root = tempDir('ak-transcript-id', t); + const file = path.join(root, 'session-1.jsonl'); + fs.writeFileSync(file, ''); + const statSync = fs.statSync; + let ino = A; + t.mock.method(fs, 'statSync', (target, options) => { + const stat = statSync(target, options); + if (target === file) stat.ino = options?.bigint ? ino : Number(ino); + return stat; + }); + const first = new TranscriptStreams({ roots: { claude: root }, mask: value => value }); + const before = first.open('claude', 'session-1').snapshot().cursor; + first.close(); + ino = B; + const second = new TranscriptStreams({ roots: { claude: root }, mask: value => value }); + const after = second.open('claude', 'session-1').snapshot().cursor; + second.close(); + assert.notEqual(before, after); +}); diff --git a/tests/kit/project-memory.test.mjs b/tests/kit/project-memory.test.mjs index 7579dbba..30b1cb80 100644 --- a/tests/kit/project-memory.test.mjs +++ b/tests/kit/project-memory.test.mjs @@ -274,7 +274,7 @@ test('the stray search is bounded and says when it stopped early', (t) => { assert.equal(capped.complete, false); assert.equal(capped.visited, 3); const deep = findStrayMemoryStores(root); - assert.equal(deep.complete, true); + assert.equal(deep.complete, false, 'a depth cutoff leaves descendants unchecked'); assert.deepEqual(deep.strays, [], 'folders deeper than the depth bound are not searched'); assert.deepEqual(findStrayMemoryStores(path.join(root, 'missing')), { strays: [], complete: true, visited: 0, nestedRepositories: [] }); }); @@ -300,3 +300,63 @@ test('the stray search stops at a nested repository or an in-checkout worktree: assert.deepEqual(strays.map((stray) => `${stray.kind} ${stray.path}`), ['aqe docs/.agentic-qe']); assert.deepEqual(nestedRepositories, ['.tools', 'packages/api', 'wt/feature']); }); + +test('the stray search finds AQE below ordinary dot folders without entering tool homes or linked folders', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-stray-dot-')); + const outside = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-stray-outside-')); + t.after(() => { fs.rmSync(root, { recursive: true, force: true }); fs.rmSync(outside, { recursive: true, force: true }); }); + fs.mkdirSync(path.join(root, '.superpowers', 'sdd', 'program', 'reports', '.agentic-qe'), { recursive: true }); + fs.mkdirSync(path.join(root, '.notes', 'drafts', '.agentic-qe'), { recursive: true }); + for (const dir of ['.git', '.claude', '.codex', '.agentic-qe', '.swarm', 'node_modules']) { + fs.mkdirSync(path.join(root, dir, 'nested', '.agentic-qe'), { recursive: true }); + } + fs.mkdirSync(path.join(outside, '.agentic-qe')); + fs.symlinkSync(outside, path.join(root, '.notes', 'linked')); + fs.symlinkSync(root, path.join(root, '.notes', 'loop')); + fs.mkdirSync(path.join(root, '.other-repo', '.agentic-qe'), { recursive: true }); + fs.writeFileSync(path.join(root, '.other-repo', '.git'), 'gitdir: elsewhere\n'); + + const result = findStrayMemoryStores(root); + assert.equal(result.complete, true); + assert.deepEqual(result.strays.map((stray) => stray.path), [ + '.notes/drafts/.agentic-qe', + '.superpowers/sdd/program/reports/.agentic-qe', + ]); + assert.deepEqual(result.nestedRepositories, ['.other-repo']); +}); + +test('depth-limited dot subtrees do not claim complete stray coverage', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-stray-limit-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + fs.mkdirSync(path.join(root, '.notes', 'a', 'b', 'c', 'd', '.agentic-qe'), { recursive: true }); + const result = findStrayMemoryStores(root); + assert.equal(result.complete, false); + assert.deepEqual(result.strays, []); +}); + +test('dependency root markers are excluded before stray inspection', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-stray-dependency-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + fs.mkdirSync(path.join(root, 'node_modules', '.agentic-qe'), { recursive: true }); + touch(root, 'node_modules/.swarm/memory.db'); + touch(root, 'node_modules/.swarm/agentdb-memory.db'); + const result = findStrayMemoryStores(root); + assert.equal(result.complete, true); + assert.deepEqual(result.strays, []); +}); + +test('a listed dot subtree with denied marker metadata reports incomplete coverage', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-stray-metadata-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + fs.mkdirSync(path.join(root, '.notes', '.agentic-qe'), { recursive: true }); + const denied = path.join(root, '.notes', '.agentic-qe'); + const original = fs.lstatSync; + fs.lstatSync = (file, ...args) => { + if (file === denied) throw Object.assign(new Error('permission denied'), { code: 'EACCES' }); + return original(file, ...args); + }; + let result; + try { result = findStrayMemoryStores(root); } finally { fs.lstatSync = original; } + assert.equal(result.complete, false); + assert.deepEqual(result.strays, []); +}); diff --git a/tests/kit/ruflo-components-status.test.mjs b/tests/kit/ruflo-components-status.test.mjs index 8e8be73b..15f468ae 100644 --- a/tests/kit/ruflo-components-status.test.mjs +++ b/tests/kit/ruflo-components-status.test.mjs @@ -1,6 +1,7 @@ import { test } from 'node:test'; import assert from 'node:assert/strict'; import { rufloComponentRows, formatComponentResults, componentResultReport, RESTART_REMINDER } from '../../src/commands/status/sections/ruflo-components.mjs'; +import { describeState } from '../../src/lib/ruflo-components/states.mjs'; import { rufloComponentsTrustGroup, trustManifestLines } from '../../src/lib/trust-manifest.mjs'; const view = (id, label, stateId, stateLabel, meaning, action = '') => ({ id, label, state: { id: stateId, label: stateLabel, meaning, action } }); @@ -23,6 +24,45 @@ test('rows lead with a summary and always carry the meaning', () => { assert.equal(rows[3].fix, null); }); +test('applied-unverified gives one restart instruction across message and manual fix', () => { + const [summary, unverified] = rufloComponentRows({ + rufloVersion: '3.44.0', summary: { active: 0, total: 1 }, components: [ + { id: 'minilmPicker', label: 'MiniLM agent picker', state: describeState('applied-unverified') }, + ], + }); + assert.equal(summary.level, 'info'); + assert.equal(unverified.state, 'applied-unverified'); + assert.equal(unverified.level, 'warn'); + assert.equal(unverified.repair, 'manual'); + assert.match(unverified.message, /MiniLM agent picker — applied, not verified: Set, but not yet confirmed/); + assert.match(unverified.fix, /restart Claude Code, Codex and OpenCode, then run ak status --refresh/i); + assert.equal(`${unverified.message} ${unverified.fix}`.match(/restart Claude Code, Codex and OpenCode/gi)?.length, 1); +}); + +test('neighboring component rows retain action, repair, and convergence contracts', () => { + const states = ['not-applied', 'drifted', 'blocked', 'active']; + const rows = rufloComponentRows({ + rufloVersion: '3.44.0', summary: { active: 1, total: 4 }, + components: states.map((id) => ({ id, label: `${id} component`, state: describeState(id) })), + }).slice(1); + for (const id of ['not-applied', 'drifted']) { + const rendered = rows.find((r) => r.state === id); + assert.equal(rendered.level, 'warn'); + assert.equal(rendered.repair, 'sync'); + assert.match(rendered.message, /Run ak sync/); + assert.match(rendered.fix, /sync applies/); + } + const blocked = rows.find((r) => r.state === 'blocked'); + assert.equal(blocked.level, 'fail'); + assert.equal(blocked.repair, 'sync'); + assert.match(blocked.message, /Follow the reason shown, then run ak sync/); + assert.match(blocked.fix, /sync applies/); + const active = rows.find((r) => r.state === 'active'); + assert.equal(active.level, 'ok'); + assert.equal(active.repair, null); + assert.equal(active.fix, null); +}); + test('setup results table lists state and meaning per component', () => { const lines = formatComponentResults(snapshot); assert.ok(lines.some((l) => /Learning profile.*user-managed.*leaves it alone/.test(l))); diff --git a/tests/kit/ruflo-daemon-config.test.mjs b/tests/kit/ruflo-daemon-config.test.mjs index 8c4ac413..90467e09 100644 --- a/tests/kit/ruflo-daemon-config.test.mjs +++ b/tests/kit/ruflo-daemon-config.test.mjs @@ -120,6 +120,174 @@ test('a malformed, non-object or symlinked config.json is user-managed and untou assert.equal(fs.readFileSync(target, 'utf8'), '{}'); }); +test('YAML markers hold JSON creation in apply and preview without hiding user daemon values', (t) => { + for (const extension of ['yaml', 'yml']) { + const root = tmpProject(t); + fs.mkdirSync(path.join(root, '.claude-flow')); + const yaml = path.join(root, '.claude-flow', `config.${extension}`); + fs.writeFileSync(yaml, 'daemon:\n maxConcurrent: 7\n'); + writeSettings(root, { claudeFlow: { daemon: { autoStart: false } } }); + const receipts = {}; + const preview = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts, dryRun: true }); + assert.equal(preview.config, 'user-managed'); + assert.equal(preview.held?.reason, 'yaml-shadow'); + assert.equal(preview.autostart, 'enabled'); + assert.deepEqual(receipts, {}); + const applied = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts }); + assert.deepEqual({ config: applied.config, held: applied.held, changed: applied.changed }, + { config: preview.config, held: preview.held, changed: preview.changed }); + assert.equal(fs.existsSync(configFile(root)), false); + assert.equal(fs.readFileSync(yaml, 'utf8'), 'daemon:\n maxConcurrent: 7\n'); + assert.equal(autoStart(root), true); + } +}); + +test('a root JSON config holds ineffective creation of lower-priority daemon JSON', (t) => { + const root = tmpProject(t); + fs.writeFileSync(path.join(root, 'claude-flow.config.json'), '{"daemon.maxConcurrent":7}'); + const result = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts: {} }); + assert.equal(result.held?.reason, 'higher-priority-json'); + assert.equal(fs.existsSync(configFile(root)), false); +}); + +test('a symlinked .claude-flow directory cannot redirect daemon JSON edits outside the project', (t) => { + const root = tmpProject(t); + const target = tmpProject(t); + fs.symlinkSync(target, path.join(root, '.claude-flow')); + const receipts = {}; + const result = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts }); + assert.equal(result.config, 'user-managed'); + assert.equal(result.held?.invalid, true); + assert.equal(fs.existsSync(path.join(target, 'config.json')), false); + assert.deepEqual(receipts, {}); + fs.writeFileSync(path.join(target, 'config.json'), '{}'); + const second = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts }); + assert.equal(second.config, 'user-managed'); + assert.equal(fs.readFileSync(path.join(target, 'config.json'), 'utf8'), '{}'); +}); + +test('root JSON becoming active holds new keys but permits receipted obsolete-key cleanup', (t) => { + const root = tmpProject(t); + const receipts = {}; + reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts }); + fs.writeFileSync(path.join(root, 'claude-flow.config.json'), '{"daemon.maxConcurrent":7}'); + const result = reconcileRufloDaemon(root, { rufloVersion: '3.46.1', platform: 'darwin', receipts }); + assert.equal(result.held?.reason, 'higher-priority-json'); + assert.deepEqual(readConfig(root), { 'daemon.resourceThresholds.minFreeMemoryPercent': 0 }); + assert.deepEqual(receipts[path.resolve(root)].configKeys, + { 'daemon.resourceThresholds.minFreeMemoryPercent': 0 }); +}); + +test('cleanup of inactive receipted JSON under root JSON does not restart a live daemon', async (t) => { + const root = rufloRepo(t); + const cfg = { rufloDaemon: { receipts: {} } }; + reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts: cfg.rufloDaemon.receipts }); + fs.writeFileSync(path.join(root, 'claude-flow.config.json'), '{"daemon.maxConcurrent":7}'); + const before = JSON.stringify(cfg.rufloDaemon.receipts); + const preview = reconcileRufloDaemon(root, { + rufloVersion: '3.46.1', platform: 'linux', receipts: cfg.rufloDaemon.receipts, dryRun: true, + }); + assert.equal(preview.config, 'removed'); + assert.equal(JSON.stringify(cfg.rufloDaemon.receipts), before); + const { calls, runner } = recorder(); + const applied = await applyRufloDaemon(root, { + cfg, rufloVersion: '3.46.1', platform: 'linux', runner, alive: () => true, + }); + assert.equal(applied.result.config, 'removed'); + assert.equal(applied.restarted, false); + assert.deepEqual(calls, []); + assert.equal(fs.existsSync(configFile(root)), false); + assert.deepEqual(cfg.rufloDaemon.receipts, {}); +}); + +test('an existing explicit config holds creation of daemon JSON in preview and apply', (t) => { + const root = tmpProject(t); + const external = tmpProject(t); + const custom = path.join(external, 'custom-config.json'); + fs.writeFileSync(custom, '{"daemon.maxConcurrent":7}'); + fs.mkdirSync(path.join(root, '.claude-flow')); + fs.writeFileSync(path.join(root, '.claude-flow', 'config.yaml'), 'daemon:\n maxConcurrent: 9\n'); + const receipts = {}; + const env = { CLAUDE_FLOW_CONFIG: custom }; + const before = fs.readFileSync(custom, 'utf8'); + const preview = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts, env, dryRun: true }); + assert.equal(preview.config, 'user-managed'); + assert.equal(preview.held?.reason, 'explicit-config'); + assert.deepEqual(receipts, {}); + const applied = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts, env }); + assert.deepEqual(applied, preview); + assert.equal(fs.existsSync(configFile(root)), false); + assert.equal(fs.readFileSync(custom, 'utf8'), before); +}); + +test('relative explicit config resolves from process cwd, and absent path does not hold JSON creation', (t) => { + const root = tmpProject(t); + const external = tmpProject(t); + const originalCwd = process.cwd(); + fs.writeFileSync(path.join(external, 'custom-config.json'), '{"daemon.maxConcurrent":7}'); + try { + process.chdir(external); + const held = reconcileRufloDaemon(root, { + rufloVersion: '3.45.0', platform: 'darwin', receipts: {}, + env: { CLAUDE_FLOW_CONFIG: './custom-config.json' }, dryRun: true, + }); + assert.equal(held.held?.reason, 'explicit-config'); + const writable = reconcileRufloDaemon(root, { + rufloVersion: '3.45.0', platform: 'darwin', receipts: {}, + env: { CLAUDE_FLOW_CONFIG: './missing-config.json' }, dryRun: true, + }); + assert.equal(writable.config, 'written'); + } finally { process.chdir(originalCwd); } +}); + +test('an existing daemon JSON takes precedence over CLAUDE_FLOW_CONFIG', async (t) => { + const root = rufloRepo(t); + const external = tmpProject(t); + const custom = path.join(external, 'custom-config.json'); + fs.writeFileSync(custom, '{"daemon.maxConcurrent":7}'); + fs.writeFileSync(configFile(root), '{}'); + const { calls, runner } = recorder(); + const result = await applyRufloDaemon(root, { + cfg: { rufloDaemon: { receipts: {} } }, rufloVersion: '3.46.1', platform: 'darwin', + env: { CLAUDE_FLOW_CONFIG: custom }, runner, alive: () => true, + }); + assert.equal(result.result.config, 'written'); + assert.equal(result.restarted, true); + assert.deepEqual(readConfig(root), { 'daemon.resourceThresholds.minFreeMemoryPercent': 0 }); + assert.equal(fs.readFileSync(custom, 'utf8'), '{"daemon.maxConcurrent":7}'); + assert.deepEqual(calls, [['ruflo', 'daemon', 'stop', root], ['ruflo', 'daemon', 'start', root]]); +}); + +test('an existing JSON keeps its ownership rules beside YAML, and release exposes YAML again', (t) => { + const root = tmpProject(t); + const receipts = {}; + reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts }); + const yaml = path.join(root, '.claude-flow', 'config.yml'); + fs.writeFileSync(yaml, 'daemon:\n maxConcurrent: 7\n'); + const preview = reconcileRufloDaemon(root, { rufloVersion: '3.46.1', platform: 'linux', receipts, dryRun: true }); + assert.equal(preview.config, 'removed'); + assert.equal(fs.existsSync(configFile(root)), true); + assert.equal(reconcileRufloDaemon(root, { rufloVersion: '3.46.1', platform: 'linux', receipts }).config, 'removed'); + assert.equal(fs.existsSync(configFile(root)), false); + assert.equal(fs.readFileSync(yaml, 'utf8'), 'daemon:\n maxConcurrent: 7\n'); + assert.deepEqual(receipts, {}); +}); + +test('a YAML hold stops repeated low-memory restarts while leaving the YAML and receipts alone', async (t) => { + const root = rufloRepo(t); + fs.writeFileSync(path.join(root, '.claude-flow', 'config.yaml'), 'daemon:\n maxConcurrent: 7\n'); + fs.mkdirSync(path.join(root, '.claude-flow', 'logs')); + fs.writeFileSync(path.join(root, '.claude-flow', 'logs', 'daemon.log'), + `[${new Date().toISOString()}] [INFO] Worker consolidate deferred: Memory too low: 3.9% free\n`); + const { calls, runner } = recorder(); + const cfg = { rufloDaemon: { receipts: {} } }; + const result = await applyRufloDaemon(root, { cfg, rufloVersion: '3.46.1', platform: 'darwin', runner, alive: () => true }); + assert.equal(result.restarted, false); + assert.deepEqual(calls, []); + assert.equal(fs.existsSync(configFile(root)), false); + assert.deepEqual(cfg.rufloDaemon.receipts, {}); +}); + test('the memory pin still wins and .swarm stays the memory root', (t) => { const root = tmpProject(t); fs.writeFileSync(path.join(root, 'claude-flow.config.json'), JSON.stringify({ memory: { persistPath: '.swarm' } })); diff --git a/tests/kit/ruflo-memory-location.test.mjs b/tests/kit/ruflo-memory-location.test.mjs index 212b8d43..32fa0c54 100644 --- a/tests/kit/ruflo-memory-location.test.mjs +++ b/tests/kit/ruflo-memory-location.test.mjs @@ -84,6 +84,30 @@ test('a disposable folder below a temporary root inside a tool folder keeps its assert.equal(launch.env.CLAUDE_FLOW_MEMORY_PATH, undefined); }); +test('a temporary root equal to a tool folder does not exempt its descendants', (t) => { + const home = sandbox(t); + const cache = mkdir(path.join(home, '.cache')); + const child = mkdir(path.join(cache, 'project')); + const at = (cwd) => rufloMemoryLocation(cwd, { home, env: { TMPDIR: cache } }); + assert.equal(at(cache).kind, 'user', 'the tool folder itself remains unsuitable'); + const location = at(child); + assert.equal(location.kind, 'user'); + assert.match(location.reason, /inside ~\/\.cache, a tool's own folder/); + assert.equal(location.db, path.join(userStore(home), 'memory.db')); +}); + +test('Windows same-path temp and tool roots are not deeper disposable boundaries', () => { + const home = 'C:\\Users\\Me'; + const local = `${home}\\AppData\\Local`; + const options = { home, platform: 'win32', p: path.win32, env: { LOCALAPPDATA: local, TMPDIR: local.toLowerCase() } }; + const tool = rufloMemoryLocation(local, options); + const child = rufloMemoryLocation(`${local}\\project`, options); + assert.equal(tool.kind, 'user'); + assert.equal(child.kind, 'user'); + assert.match(child.reason, /tool's own folder/); + assert.equal(child.db, `${home}\\.claude-flow\\memory\\memory.db`); +}); + test('the filesystem root, the home folder and a temporary root use the one user-level store', (t) => { const home = sandbox(t); for (const [cwd, reason] of [ @@ -167,6 +191,25 @@ test('a Git repository at the home folder does not pull a plain subfolder into ~ assert.equal(rufloMemoryLocation(home, { home }).kind, 'user'); }); +test('user-level location explains both an unsuitable folder and its unsuitable repository root', (t) => { + const home = sandbox(t); + fs.mkdirSync(path.join(home, '.git')); + const codex = mkdir(path.join(home, '.codex', 'sessions')); + const location = rufloMemoryLocation(codex, { home }); + assert.deepEqual([location.kind, location.root, location.dir, location.db], + ['user', userStore(home), userStore(home), path.join(userStore(home), 'memory.db')]); + assert.equal(location.reason, "inside ~/.codex, a tool's own folder, in a repository whose root is the home folder"); +}); + +test('location does not repeat a reason when root and folder share one tool area', (t) => { + const home = sandbox(t); + const root = mkdir(path.join(home, '.codex', 'workspace')); + fs.mkdirSync(path.join(root, '.git')); + const child = mkdir(path.join(root, 'src')); + assert.equal(rufloMemoryLocation(child, { home }).reason, "inside ~/.codex, a tool's own folder"); + assert.equal(rufloMemoryLocation(root, { home }).reason, "inside ~/.codex, a tool's own folder"); +}); + test('the launcher pins both memory variables to the user-level store and starts Ruflo inside it', (t) => { const home = sandbox(t); const launch = rufloMcpLaunch(home, { CLAUDE_FLOW_DB_PATH: '/.swarm/memory.db', KEEP: 'yes' }, { cfg, rufloVersion: '3.45.0', home }); @@ -215,6 +258,27 @@ test('status reports the user-level store and stray stores outside projects, for assert.match(strays[0].message, /leaves them in place/); }); +test('user status reports AQE home data separately without calling an empty folder healthy', async (t) => { + const home = sandbox(t); + const aqeDir = path.join(home, '.agentic-qe'); + fs.mkdirSync(aqeDir); + let rows = await userMemory.collect({ home, env: {} }); + let aqe = rows.find((r) => r.message.includes(aqeDir)); + assert.ok(aqe); + assert.equal(aqe.level, 'info'); + assert.equal(aqe.fix, null); + assert.match(aqe.message, /AQE.*memory\.db absent.*unverified/); + assert.doesNotMatch(aqe.message, /Ruflo|healthy|merge|move/i); + + fs.writeFileSync(path.join(aqeDir, 'memory.db'), 'placeholder'); + fs.writeFileSync(path.join(aqeDir, 'memory.db-wal'), 'wal'); + rows = await userMemory.collect({ home, env: {} }); + aqe = rows.find((r) => r.message.includes(aqeDir)); + assert.match(aqe.message, /AQE.*memory\.db present.*14 B.*unverified/); + assert.doesNotMatch(aqe.message, /Ruflo|healthy|merge|move/i); + assert.equal(rows.filter((r) => /stray Ruflo/.test(r.message)).length, 0); +}); + test('Claude mode from the home folder pins both memory variables to the user-level store', (t) => { const home = sandbox(t); const launch = rufloMcpLaunch(home, { RUFLO_INTELLIGENCE_MODE: 'fast' }, { cfg, rufloVersion: '3.46.1', home, host: 'claude' }); diff --git a/tests/kit/ruflo-windows-diagnostic-process.test.mjs b/tests/kit/ruflo-windows-diagnostic-process.test.mjs new file mode 100644 index 00000000..d902dcd2 --- /dev/null +++ b/tests/kit/ruflo-windows-diagnostic-process.test.mjs @@ -0,0 +1,191 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { EventEmitter } from 'node:events'; +import { PassThrough } from 'node:stream'; +import { spawn } from 'node:child_process'; +import { createDiagnosticScope } from '../live/ruflo-windows-diagnostic-process.mjs'; + +function fakeChild({ closes = true, code = 0 } = {}) { + const child = new EventEmitter(); + Object.assign(child, { pid: 123, exitCode: null, signalCode: null, + stdin: new PassThrough(), stdout: new PassThrough(), stderr: new PassThrough(), + unreferenced: false, killed: false }); + child.unref = () => { child.unreferenced = true; }; + child.kill = () => { + child.killed = true; + if (closes) queueMicrotask(() => { child.exitCode = code; child.emit('close', code, null); }); + }; + return child; +} + +function scopeFor(spawnFn) { + let released = false; + const scope = createDiagnosticScope({ spawnFn, platform: 'win32', + killTimeoutMs: 20, closeTimeoutMs: 20, + acquireHold: () => ({}), releaseHold: () => { released = true; } }); + return { scope, released: () => released }; +} + +test('every diagnostic role passes the exact environment, excluding parent-only credentials', async () => { + process.env.AK_DIAGNOSTIC_PARENT_ONLY = 'synthetic-do-not-inherit'; + const env = { PATH: 'fixture-only' }; + const seen = []; + let active; + const { scope } = scopeFor((command, _args, options) => { + seen.push({ command, env: options.env }); + const child = fakeChild(); + if (command === 'taskkill.exe') queueMicrotask(() => { active.kill(); child.kill(); }); + else active = child; + return child; + }); + try { + for (const command of ['version', 'legacy', 'initialize']) { + const r = await scope.launch({ command, args: [] }, { env, timeoutMs: 10 }); + assert.equal(r.cleanupComplete, true); + } + assert.deepEqual(seen.map((x) => x.command), ['version', 'taskkill.exe', 'legacy', 'taskkill.exe', 'initialize', 'taskkill.exe']); + for (const entry of seen) { + assert.strictEqual(entry.env, env); + assert.equal(entry.env.AK_DIAGNOSTIC_PARENT_ONLY, undefined); + } + assert.equal(scope.release(), true); + } finally { delete process.env.AK_DIAGNOSTIC_PARENT_ONLY; } +}); + +for (const failure of ['nonzero', 'spawn-error', 'stall']) { + test(`taskkill ${failure} always disposes handles and retains uncertainty`, async () => { + const children = []; + const { scope, released } = scopeFor((command) => { + const child = fakeChild({ closes: false }); children.push(child); + if (command === 'taskkill.exe' && failure !== 'stall') queueMicrotask(() => { + if (failure === 'spawn-error') child.emit('error', new Error('ENOENT')); + child.exitCode = 1; child.emit('close', 1, null); + }); + return child; + }); + const started = Date.now(); + const r = await scope.launch({ command: 'initialize', args: [] }, { env: {}, timeoutMs: 10 }); + assert.equal(r.cleanupComplete, false); + assert.ok(Date.now() - started < 1000); + assert.equal(children[0].killed, true, 'direct-child fallback attempted'); + assert.equal(children[0].unreferenced, true); + assert.equal(children[0].stdout.destroyed, true); + assert.equal(children[0].stderr.destroyed, true); + assert.equal(children[0].stdin.destroyed, true); + assert.equal(scope.release(), false); + assert.equal(released(), false); + if (failure === 'stall') assert.equal(children[1].unreferenced, true); + }); +} + +for (const command of ['version', 'legacy']) { + test(`${command} cleanup failure cannot be cleared by a later successful launch`, async () => { + const { scope, released } = scopeFor((cmd) => { + const child = fakeChild(); + if (cmd === 'later' || cmd === 'taskkill.exe') queueMicrotask(() => { + child.exitCode = cmd === 'taskkill.exe' ? 1 : 0; child.emit('close', child.exitCode, null); + }); + return child; + }); + assert.equal((await scope.launch({ command, args: [] }, { env: {}, timeoutMs: 10 })).cleanupComplete, false); + assert.equal((await scope.launch({ command: 'later', args: [] }, { env: {}, timeoutMs: 10 })).cleanupComplete, true); + assert.equal(scope.release(), false); + assert.equal(released(), false); + }); +} + +test('real version, legacy, initialize and taskkill children cannot see a parent-only sentinel', async () => { + const key = 'AK_DIAGNOSTIC_PARENT_ONLY'; + const previous = process.env[key]; + process.env[key] = 'synthetic-do-not-inherit'; + const env = { AK_DIAGNOSTIC_ALLOWED: 'yes' }; + const observed = []; + const scope = createDiagnosticScope({ + platform: 'win32', acquireHold: () => ({}), releaseHold: () => {}, + killTimeoutMs: 1000, closeTimeoutMs: 1000, + spawnFn: (command, args, options) => { + const payload = 'process.stdout.write(JSON.stringify({allowed:process.env.AK_DIAGNOSTIC_ALLOWED,sentinel:process.env.AK_DIAGNOSTIC_PARENT_ONLY}));'; + const code = command === 'taskkill.exe' + ? `${payload}process.kill(${Number(args[1])},'SIGKILL');` + : `${payload}${command === 'initialize' ? 'setInterval(()=>{},1000);' : ''}`; + const child = spawn(process.execPath, ['-e', code], { ...options, env: options.env }); + let output = ''; + child.stdout.on('data', (chunk) => { output += chunk; }); + child.on('close', () => observed.push({ command, ...JSON.parse(output) })); + return child; + }, + }); + try { + for (const command of ['version', 'legacy', 'initialize']) { + const result = await scope.launch({ command, args: [] }, { env, timeoutMs: 500 }); + assert.equal(result.cleanupComplete, true); + } + assert.deepEqual(observed.map((x) => x.command).sort(), ['initialize', 'legacy', 'taskkill.exe', 'version']); + for (const result of observed) { + assert.equal(result.allowed, 'yes'); + assert.equal(result.sentinel, undefined); + } + assert.equal(scope.release(), true); + } finally { + if (previous === undefined) delete process.env[key]; else process.env[key] = previous; + } +}); + +test('EOF response waits for natural pipe closure without racing taskkill', async () => { + let child; + const commands = []; + const { scope, released } = scopeFor((command) => { + commands.push(command); + child = fakeChild({ closes: false }); + queueMicrotask(() => child.stdout.write('initialized')); + setTimeout(() => { child.exitCode = 0; }, 5); + setTimeout(() => child.emit('close', 0, null), 60); + return child; + }); + const result = await scope.launch({ command: 'eof', args: [] }, { + env: {}, timeoutMs: 200, endInput: true, until: (text) => text === 'initialized', + }); + assert.equal(result.matched, true); + assert.equal(result.code, 0); + assert.equal(result.cleanupComplete, true, 'requires observed close, not just exit code'); + assert.equal(result.timedOut, false); + assert.equal(child.killed, false); + assert.deepEqual(commands, ['eof']); + assert.equal(scope.release(), true); + assert.equal(released(), true); +}); + +test('EOF match plus exit code zero without pipe closure retains uncertainty', async () => { + const commands = []; + let child; + const { scope, released } = scopeFor((command) => { + commands.push(command); + child = fakeChild({ closes: false }); + queueMicrotask(() => { child.stdout.write('initialized'); child.exitCode = 0; }); + return child; + }); + const result = await scope.launch({ command: 'eof', args: [] }, { + env: {}, timeoutMs: 30, endInput: true, until: (text) => text === 'initialized', + }); + assert.equal(result.matched, true); + assert.equal(result.timedOut, true); + assert.equal(result.cleanupComplete, false); + assert.deepEqual(commands, ['eof'], 'never target a reaped PID'); + assert.equal(child.unreferenced, true); + assert.equal(child.stdout.destroyed, true); + assert.equal(scope.release(), false); + assert.equal(released(), false); +}); + +test('a real EOF responder exits naturally after its reply before cleanup is accepted', async () => { + const scope = createDiagnosticScope({ acquireHold: () => ({}), releaseHold: () => {} }); + const code = "process.stdin.resume();process.stdin.on('end',()=>{process.stdout.write('initialized');setTimeout(()=>process.exit(0),80)});"; + const result = await scope.launch({ command: process.execPath, args: ['-e', code] }, { + env: {}, timeoutMs: 2000, endInput: true, until: (text) => text === 'initialized', + }); + assert.equal(result.matched, true); + assert.equal(result.code, 0); + assert.equal(result.signal, null, 'no forced termination after the reply'); + assert.equal(result.cleanupComplete, true); + assert.equal(scope.release(), true); +}); diff --git a/tests/kit/run-root-identities.test.mjs b/tests/kit/run-root-identities.test.mjs new file mode 100644 index 00000000..4b79d17b --- /dev/null +++ b/tests/kit/run-root-identities.test.mjs @@ -0,0 +1,139 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; +import { ownerRecord, writeOwner, readOwner, collectAbandonedRoots, OWNER_FILE } from '../../scripts/run-roots.mjs'; +import { runGuarded } from '../../scripts/run-tests.mjs'; + +const first = 2n ** 53n; +const second = first + 1n; +assert.equal(Number(first), Number(second), 'fixture must collide through Number'); + +// Model the OS exposing exact BigInt fields or lossy legacy Number fields. +function identity(stat, options, field, value) { + const exact = options?.bigint === true; + const key = !exact && field.endsWith('Ns') ? field.replace(/Ns$/, 'Ms') : field; + stat[key] = exact ? value : Number(value) / (field.endsWith('Ns') ? 1e6 : 1); + return stat; +} +function fixture(t) { + const parent = fs.realpathSync(tempDir('ak-exact-root', t)); + const root = fs.mkdtempSync(path.join(parent, 'ak-suite-')); + const home = path.join(parent, 'home'); + fs.mkdirSync(home); + writeOwner(root, ownerRecord()); + return { parent, root, home }; +} + +for (const field of ['dev', 'ino', 'ctimeNs']) { + for (const phase of ['open', 'read']) { + test(`owner ${field} collision at ${phase} refuses swapped file`, (t) => { + const { root } = fixture(t); + const file = path.join(root, OWNER_FILE); + const originalLstat = fs.lstatSync; + const originalFstat = fs.fstatSync; + let reads = 0; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = originalLstat(target, options); + return target === file ? identity(stat, options, field, ++reads === 1 ? first : second) : stat; + }); + t.mock.method(fs, 'fstatSync', (fd, options) => + identity(originalFstat(fd, options), options, field, phase === 'open' ? second : first)); + assert.equal(readOwner(root), null); + }); + } + + test(`sibling ${field} collision during injected proof retains root`, (t) => { + const { parent, root, home } = fixture(t); + const original = fs.lstatSync; + let swapped = false; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = original(target, options); + return target === root ? identity(stat, options, field, swapped ? second : first) : stat; + }); + let removals = 0; + const result = collectAbandonedRoots({ tmpdir: parent, homedir: home, log: () => {}, + probes: { alive: () => false, startedAfter: () => false, completeExit: () => { swapped = true; return true; } }, + remove: () => { removals++; }, + }); + assert.equal(removals, 0); + assert.deepEqual(result.removed, []); + assert.match(result.kept[0].reason, /changed/); + assert.ok(fs.existsSync(root)); + }); +} + +for (const field of ['dev', 'ino', 'birthtimeNs']) { + test(`call-owned ${field} collision retains replacement untouched`, (t) => { + const { parent, home } = fixture(t); + const tmpdir = path.join(parent, 'tmp'); + const repoRoot = path.join(home, 'repo'); + fs.mkdirSync(tmpdir); + fs.mkdirSync(repoRoot); + const env = spawnEnv(home); + t.mock.method(os, 'tmpdir', () => tmpdir); + const original = fs.lstatSync; + let swapped = false; + let ownRoot; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = original(target, options); + if (!stat || path.dirname(String(target)) !== tmpdir || !/^ak-suite-[A-Za-z0-9]{6}$/.test(path.basename(String(target)))) return stat; + ownRoot = target; + for (const key of ['dev', 'ino', 'birthtimeNs']) identity(stat, options, key, first); + return identity(stat, options, field, swapped ? second : first); + }); + const messages = []; + const code = runGuarded([], { env, homedir: home, repoRoot, log: (message) => { + messages.push(message); + if (message.startsWith('real-state tripwire:')) { + // Preserve metadata so the ownership/hold checks alone cannot catch this swap. + fs.cpSync(ownRoot, `${ownRoot}-saved`, { recursive: true }); + fs.rmSync(ownRoot, { recursive: true }); + fs.renameSync(`${ownRoot}-saved`, ownRoot); + swapped = true; + } + } }); + assert.equal(code, 4); + assert.match(messages.join('\n'), /directory identity changed/); + assert.ok(fs.existsSync(path.join(ownRoot, OWNER_FILE))); + assert.ok(fs.existsSync(path.join(ownRoot, '.ak-suite-holds'))); + }); +} + +test('unchanged exact owner identity accepts large IDs and preserves numeric attribution', (t) => { + const { root } = fixture(t); + const file = path.join(root, OWNER_FILE); + const lstat = fs.lstatSync; + const fstat = fs.fstatSync; + const patch = (stat, options) => { + for (const field of ['dev', 'ino', 'ctimeNs']) identity(stat, options, field, second); + return stat; + }; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = lstat(target, options); + return target === file ? patch(stat, options) : stat; + }); + t.mock.method(fs, 'fstatSync', (fd, options) => patch(fstat(fd, options), options)); + const owner = readOwner(root); + assert.equal(owner.pid, process.pid); + assert.equal(owner.uid, process.getuid?.() ?? null); + assert.equal(typeof owner.startedAt, 'number'); +}); + +test('exact owner stats reject foreign filesystem UID before opening', (t) => { + if (!process.getuid) return; // Windows has no UID boundary to validate. + const { root } = fixture(t); + const file = path.join(root, OWNER_FILE); + const lstat = fs.lstatSync; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = lstat(target, options); + if (target === file) stat.uid = options?.bigint ? BigInt(process.getuid()) + 1n : process.getuid() + 1; + return stat; + }); + const open = t.mock.method(fs, 'openSync', () => { throw Error('must not open foreign owner'); }); + assert.equal(readOwner(root), null); + assert.equal(open.mock.callCount(), 0); +}); diff --git a/tests/kit/run-tests-runner.test.mjs b/tests/kit/run-tests-runner.test.mjs index 6f7b9a6d..4d7cc9d9 100644 --- a/tests/kit/run-tests-runner.test.mjs +++ b/tests/kit/run-tests-runner.test.mjs @@ -445,7 +445,10 @@ function cleanupProbe(home) { if (own(p)) { ownStats++; if (mode === 'refusal' && ownStats >= 3) stat.isDirectory = () => false; - if (mode === 'identity' && ownStats >= 4) stat.birthtimeMs += 1; + if (mode === 'identity' && ownStats >= 4) { + if (typeof stat.birthtimeNs === 'bigint') stat.birthtimeNs += 1n; + else stat.birthtimeMs += 1; + } } return stat; }; diff --git a/tests/kit/setup-host-rerecord.test.mjs b/tests/kit/setup-host-rerecord.test.mjs new file mode 100644 index 00000000..80cb1110 --- /dev/null +++ b/tests/kit/setup-host-rerecord.test.mjs @@ -0,0 +1,106 @@ +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import { registerHooks } from 'node:module'; +import { sandboxHome, rmrf } from './helpers/home-sandbox.mjs'; +import { isolateProject, REPO_ROOT } from './helpers/project-isolation.mjs'; + +const home = sandboxHome('ak-setup-host-rerecord'); +after(() => rmrf(home)); +isolateProject('ak-setup-host-rerecord'); +const setup = await import('../../src/commands/setup.mjs'); +const cfg = { integrations: { hosts: { claude: false, codex: true, opencode: false } } }; + +for (const [name, initial, ok, expected] of [ + ['successful install', 'absent', true, 2], + ['failed install', 'absent', false, 1], + ['present host', 'npm', true, 1], +]) { + test(`${name} records fresh host facts only when installation succeeds`, async () => { + const stateCalls = []; + const facts = []; + const installs = []; + await setup.installEnabledAbsentHosts(cfg, { yes: true }, { + installState: async (_host, opts) => { stateCalls.push(opts); return { method: stateCalls.length === 1 ? initial : 'npm', version: '1.0.0' }; }, + install: async id => { installs.push(id); return { ok, detail: ok ? 'installed' : 'failed' }; }, + collectFacts: async opts => { facts.push(opts); }, + }); + assert.equal(stateCalls.length, expected); + assert.deepEqual(installs, initial === 'absent' ? ['codex'] : []); + if (expected === 2) { + assert.deepEqual(stateCalls[1], { refresh: true, record: true, source: 'setup' }); + assert.deepEqual(facts, [{ cfg, refresh: true, record: true, source: 'setup' }]); + } else assert.deepEqual(facts, []); + }); +} + +test('disabled hosts do not probe, install, or record evidence', async () => { + await setup.installEnabledAbsentHosts({ integrations: { hosts: {} } }, { yes: true }, { + installState: async () => { throw new Error('disabled host probed'); }, + install: async () => { throw new Error('disabled host installed'); }, + collectFacts: async () => { throw new Error('disabled host recorded'); }, + }); +}); + +test('setup run passes its host lifecycle through the machine setup boundary', async () => { + const lifecycle = { installState: async () => { throw new Error('sentinel'); } }; + let received; + await setup.run({ + flags: { 'dry-run': true, minimal: true, 'no-aqe': true, 'no-ruvnet-brain': true, 'no-agent-browser': true, yes: true }, + pkgRoot: REPO_ROOT, + deps: { hostLifecycle: lifecycle }, + machineSetup: async args => { received = args.deps?.hostLifecycle; return false; }, + }); + assert.equal(received, lifecycle); +}); + +test('real run_machine passes injected lifecycle to the host installation loop', async () => { + const stubs = new Map([ + ['../lib/versions.mjs', "export const installedVersion = () => '3.48.0'; export const cmpVersions = () => 0;"], + ['../lib/heal.mjs', "export const healNatives = async () => ({ ok: true, detail: 'stub' }); export const healAidefence = async () => ({ ok: true, detail: 'stub' });"], + ['../lib/exec.mjs', "export const have = async () => false; export const run = async () => { throw Error('real command reached'); };"], + ['../lib/providers.mjs', ` + export const HOSTS = [{ id: 'codex', pkg: '@openai/codex' }]; + export const hostInstallState = async () => { throw Error('default host probe reached'); }; + export const installHost = async () => { throw Error('default installer reached'); }; + export const collectIntegrationFacts = async () => { throw Error('default facts collector reached'); }; + export const migrateRetiredRoutesInConfig = () => {}; + export const printActivityRoutingTable = () => {}; + export const convergeProviderStack = () => {}; + export const applySetupHostFlags = () => {}; + export const guidanceContext = () => {}; + export const reportRetiredRouteChanges = () => {}; + `], + ['../lib/adapters/lifecycle-registry.mjs', ` + export const hostsWithLifecycle = () => []; + export const lifecycleAdapterFor = () => { throw Error('host lifecycle reached'); }; + export const lifecycleExecutionEnabled = () => false; + export const detectionBinFor = () => { throw Error('host lifecycle reached'); }; + `], + ['../lib/ruflo-components/apply.mjs', "export const reconcileRufloComponents = async () => { throw Error('components reached'); };"], + ['./status/sections/ruflo-components.mjs', "export const componentResultReport = () => [];"], + ]); + registerHooks({ + resolve(specifier, context, nextResolve) { + if (context.parentURL?.includes('/src/commands/setup.mjs?b2-machine') && stubs.has(specifier)) { + return { url: `data:text/javascript,${encodeURIComponent(stubs.get(specifier))}`, shortCircuit: true }; + } + return nextResolve(specifier, context); + }, + }); + const isolatedSetup = await import('../../src/commands/setup.mjs?b2-machine'); + const sentinel = new Error('injected host lifecycle reached'); + const machineCfg = { + agentBrowser: false, aqe: false, ruvnetBrain: false, security: false, + integrations: { hosts: { codex: true } }, + }; + let calls = 0; + await assert.rejects(isolatedSetup.run_machine({ + flags: { yes: true }, cfg: machineCfg, pkgRoot: home, + deps: { hostLifecycle: { + installState: async () => { calls++; throw sentinel; }, + install: async () => { throw Error('injected installer reached'); }, + collectFacts: async () => { throw Error('injected facts reached'); }, + } }, + }), error => error === sentinel); + assert.equal(calls, 1); +}); diff --git a/tests/kit/setup-memory-probe.test.mjs b/tests/kit/setup-memory-probe.test.mjs index d477d976..e9842afd 100644 --- a/tests/kit/setup-memory-probe.test.mjs +++ b/tests/kit/setup-memory-probe.test.mjs @@ -10,6 +10,7 @@ import path from 'node:path'; import { DatabaseSync } from 'node:sqlite'; import { verifyProjectMemoryWrite } from '../../src/commands/setup.mjs'; import { captureLog } from './helpers/home-sandbox.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; function store(file) { fs.mkdirSync(path.dirname(file), { recursive: true }); @@ -109,3 +110,114 @@ test('the setup probe leaves nothing in a redirected memory root', async (t) => assert.deepEqual(keys(path.join(redirected, 'agentdb-memory.db')), ['user-row'], 'the redirected MCP store kept a setup probe'); for (const name of ['memory.db', 'agentdb-memory.db']) assert.deepEqual(keys(path.join(root, '.swarm', name)), ['user-row']); }); + +// Ruflo 3.48 writes its native mirror under CLAUDE_FLOW_MEMORY_PATH even when +// CLAUDE_FLOW_DB_PATH points elsewhere. The fake follows those two routes. +function routedProject(t) { + const root = tempDir('ak-setup-route', t); + const primary = path.join(root, '.swarm', 'memory.db'); + const canonical = path.join(root, '.swarm', 'agentdb-memory.db'); + const redirected = path.join(root, 'data', 'memory', 'agentdb-memory.db'); + for (const file of [primary, canonical, redirected]) { + const db = store(file); + db.prepare("INSERT INTO memory_entries VALUES ('real', 'user-row', 'active')").run(); + db.close(); + } + const mirrors = []; + const runner = async (_cmd, args, { env }) => { + const key = args[args.indexOf('-k') + 1]; + assert.equal(env.RUFLO_DAEMON_AUTOSTART, '0'); + assert.equal(env.CLAUDE_FLOW_DB_PATH, primary); + const mirror = path.join(env.CLAUDE_FLOW_MEMORY_PATH, 'agentdb-memory.db'); + mirrors.push(mirror); + for (const file of [env.CLAUDE_FLOW_DB_PATH, mirror]) { + fs.mkdirSync(path.dirname(file), { recursive: true }); + const db = new DatabaseSync(file); + db.exec('CREATE TABLE IF NOT EXISTS memory_entries (namespace TEXT, key TEXT, status TEXT)'); + db.prepare("INSERT INTO memory_entries VALUES ('_setup', ?, 'active')").run(key); + db.close(); + } + return { code: 0, stdout: '', stderr: '' }; + }; + return { root, primary, canonical, redirected, mirrors, runner }; +} + +test('setup verifies the primary while removing its private mirror and preserving native corpora', async (t) => { + const p = routedProject(t); + const { out } = await captureLog(() => verifyProjectMemoryWrite(p.root, { CLAUDE_FLOW_DB_PATH: p.primary, + CLAUDE_FLOW_MEMORY_PATH: path.dirname(p.redirected) }, { runner: p.runner })); + assert.match(out, /memory write VERIFIED/); + assert.deepEqual(keys(p.primary), ['user-row']); + assert.deepEqual(keys(p.canonical), ['user-row']); + assert.deepEqual(keys(p.redirected), ['user-row']); + assert.equal(p.mirrors.length, 1); + assert.equal(fs.existsSync(path.dirname(p.mirrors[0])), false); +}); + +test('setup pins the primary when caller env is absent and removes the mirror on a failed store', async (t) => { + const p = routedProject(t); + const { out } = await captureLog(() => verifyProjectMemoryWrite(p.root, undefined, { + runner: async (cmd, args, options) => { + await p.runner(cmd, args, options); + return { code: 1, stdout: '', stderr: 'failed after write' }; + }, + })); + assert.match(out, /memory write verification FAILED/); + assert.doesNotMatch(out, /memory write VERIFIED/); + assert.deepEqual(keys(p.canonical), ['user-row']); + assert.deepEqual(keys(p.redirected), ['user-row']); + assert.equal(fs.existsSync(path.dirname(p.mirrors[0])), false); +}); + +test('setup removes its private mirror after a runner throws', async (t) => { + const p = routedProject(t); + const { out } = await captureLog(() => verifyProjectMemoryWrite(p.root, {}, { + runner: async (cmd, args, options) => { + await p.runner(cmd, args, options); + throw new Error('runner failed'); + }, + })); + assert.match(out, /memory write verification FAILED/); + assert.equal(fs.existsSync(path.dirname(p.mirrors[0])), false); + assert.deepEqual(keys(p.canonical), ['user-row']); +}); + +test('setup reports a missing primary row and removes the private mirror', async (t) => { + const p = routedProject(t); + let mirror; + const { out } = await captureLog(() => verifyProjectMemoryWrite(p.root, {}, { + runner: async (_cmd, _args, { env }) => { + mirror = env.CLAUDE_FLOW_MEMORY_PATH; + const db = store(path.join(mirror, 'agentdb-memory.db')); + db.close(); + return { code: 0, stdout: '', stderr: '' }; + }, + })); + assert.match(out, /memory write verification FAILED/); + assert.doesNotMatch(out, /memory write VERIFIED/); + assert.equal(fs.existsSync(mirror), false); + assert.deepEqual(keys(p.primary), ['user-row']); +}); + +test('setup does not claim complete verification when private mirror removal fails', { skip: process.platform === 'win32' }, async (t) => { + const p = routedProject(t); + let mirror; + t.after(() => { + if (mirror && fs.existsSync(mirror)) { + fs.chmodSync(mirror, 0o700); + fs.rmSync(mirror, { recursive: true, force: true }); + } + }); + const { out } = await captureLog(() => verifyProjectMemoryWrite(p.root, {}, { + runner: async (cmd, args, options) => { + await p.runner(cmd, args, options); + mirror = options.env.CLAUDE_FLOW_MEMORY_PATH; + fs.chmodSync(mirror, 0o000); + return { code: 0, stdout: '', stderr: '' }; + }, + })); + assert.match(out, /temporary mirror cleanup failed/); + assert.match(out, /memory write verification FAILED/); + assert.doesNotMatch(out, /memory write VERIFIED/); + assert.deepEqual(keys(p.primary), ['user-row']); +}); diff --git a/tests/kit/status-command.test.mjs b/tests/kit/status-command.test.mjs index 27744242..92ca6191 100644 --- a/tests/kit/status-command.test.mjs +++ b/tests/kit/status-command.test.mjs @@ -658,6 +658,7 @@ test('Codex MCP topology fails recursive self-registration and reports missing A 'args = ["x", "ruflo-mcp"]', ].join('\n')); try { + const codexConfigBefore = fs.readFileSync(path.join(PROJECT, '.codex', 'config.toml'), 'utf8'); const rows = rowsFor(await collect(), 'codex-mcp'); assert.equal(rows.find((r) => /recursive codex/.test(r.message))?.level, 'fail'); assert.equal(rows.find((r) => /agentic-qe MCP is not concretely/.test(r.message))?.level, 'warn'); @@ -666,7 +667,12 @@ test('Codex MCP topology fails recursive self-registration and reports missing A // can prove (codexMcpRepairPlan); Agentic-QE owns its own Codex registration. assert.equal(rows.find((r) => /recursive codex/.test(r.message))?.repair, 'sync'); assert.equal(rows.find((r) => /duplicate Ruflo/.test(r.message))?.repair, 'sync'); - assert.equal(rows.find((r) => /agentic-qe MCP is not concretely/.test(r.message))?.repair, 'manual'); + const aqe = rows.find((r) => /agentic-qe MCP is not concretely/.test(r.message)); + assert.equal(aqe?.repair, 'manual'); + assert.match(aqe.fix, /aqe init --auto --with-codex --codex-guidance compact/); + assert.doesNotMatch(aqe.fix, /platform setup|codex mcp-server/); + assert.equal(fs.readFileSync(path.join(PROJECT, '.codex', 'config.toml'), 'utf8'), codexConfigBefore, + 'status only reports the AQE-owned initialization flow'); } finally { rmrf(path.join(PROJECT, '.codex')); } @@ -716,7 +722,7 @@ test('a user-owned deprecated codex mcp-server entry is a manual removal', async }); test('Codex MCP topology does not ask an aqe:false machine to register agentic-qe in Codex', async () => { - // #237 N1: with AQE opted out, `aqe platform setup codex` is advice for a + // #237 N1: with AQE opted out, Codex initialization advice is for a // tool the user declined; the topology rows must honor kit.json like the // aqe section does. seedHome(offlineKitConfig({ @@ -970,7 +976,7 @@ test('--refresh prints one line per finished stage, then the status table', asyn assert.match(r.out, /ak status\n.*versions +fake versions row/); }); -test('plain status runs no refresh stage', async () => { +test('plain status rejects a stray positional before any refresh stage', async () => { seedHome(); let calls = 0; const refreshStages = new Proxy({}, { get: () => async () => { calls += 1; return { ok: true }; } }); @@ -981,7 +987,8 @@ test('plain status runs no refresh stage', async () => { r = await captureLog(() => status.run({ flags: {}, positionals: ['extra'], pkgRoot: PKG_ROOT, deps: { refreshStages } })); } finally { process.chdir(cwd); } assert.equal(calls, 0); - assert.notEqual(r.result, 2, 'plain status still ignores a positional, as it always has'); + assert.equal(r.result, 2); + assert.match(r.out, /unexpected argument 'extra'/); }); test('a refresh with a stray argument is a usage error that names the one-token spelling', async () => { diff --git a/tests/kit/status-version-drift-refresh.test.mjs b/tests/kit/status-version-drift-refresh.test.mjs index 0f9bc8f7..88dd9dc3 100644 --- a/tests/kit/status-version-drift-refresh.test.mjs +++ b/tests/kit/status-version-drift-refresh.test.mjs @@ -151,7 +151,7 @@ function seedSelfHome({ ageMs = 0 } = {}) { ttlHours: 24, last: Date.now(), seen: { ruflo: '9.9.9', 'agentic-qe': '9.9.9' }, - self: { last: Date.now() - ageMs, best: { version: '0.0.1', tag: 'latest' } }, + self: { last: Date.now() - ageMs, best: { version: '0.0.1', tag: 'latest' }, lastTags: ['latest', 'next'] }, }, }); writeKitConfig(HOME, cfg); diff --git a/tests/kit/sync-daemon-repair.test.mjs b/tests/kit/sync-daemon-repair.test.mjs index 616841da..79c9aa82 100644 --- a/tests/kit/sync-daemon-repair.test.mjs +++ b/tests/kit/sync-daemon-repair.test.mjs @@ -16,6 +16,7 @@ import { sandboxHome, rmrf } from './helpers/home-sandbox.mjs'; const SANDBOX_HOME = sandboxHome('ak-sync-daemon-repair'); after(() => rmrf(SANDBOX_HOME)); const sync = await import('../../src/commands/sync.mjs'); +const { applyRufloDaemon } = await import('../../src/lib/ruflo-daemon-config.mjs'); const daemonsStep = sync.SYNC_STEPS.find((s) => s.id === 'daemons'); @@ -75,3 +76,49 @@ test('a reap attempt that fails to kill anything is not re-recorded either', asy assert.equal(listCalls.length, 1, 'a failed reap (pid already gone/reused) leaves the process list unchanged; no re-record is needed'); } finally { rmrf(cwd); } }); + +test('daemon step uses injected version, apply and save seams after its sweep', async () => { + const cwd = freshCwd(); + const calls = []; + try { + await daemonsStep.run({ + cwd, cfg: {}, + daemonLifecycle: { list: async () => [], reap: () => [] }, + daemonVersion: () => { calls.push('version'); return '3.48.0'; }, + daemonApply: async (_cwd, options) => { + calls.push(['apply', options.rufloVersion]); + return { result: { config: 'converged', autostart: 'converged', changed: false, held: null }, restarted: false }; + }, + saveConfig: () => calls.push('save'), + }); + assert.deepEqual(calls, ['version', ['apply', '3.48.0'], 'save']); + } finally { rmrf(cwd); } +}); + +test('versions-triggered daemon step applies only fixture settings with no external processes', async () => { + const cwd = freshCwd(); + const calls = []; + try { + fs.mkdirSync(path.join(cwd, '.git')); + fs.mkdirSync(path.join(cwd, '.claude-flow')); + fs.mkdirSync(path.join(cwd, '.swarm')); + fs.writeFileSync(path.join(cwd, '.swarm', 'memory.db'), ''); + const cfg = { rufloDaemon: { receipts: {} } }; + assert.ok(sync.activeSteps(new Set(['versions']), {}, cfg).includes('daemons')); + await daemonsStep.run({ + cwd, cfg, + daemonLifecycle: { list: async () => { calls.push('list'); return []; }, reap: () => [] }, + daemonVersion: () => '3.45.0', + daemonApply: (root, options) => applyRufloDaemon(root, { + ...options, platform: 'darwin', alive: () => { calls.push('alive'); return true; }, + runner: async (_tool, args) => { calls.push(args.join(' ')); return { code: 0 }; }, + }), + saveConfig: () => calls.push('save'), + }); + assert.deepEqual(calls, ['list', 'alive', 'daemon stop', 'daemon start', 'save']); + assert.deepEqual(JSON.parse(fs.readFileSync(path.join(cwd, '.claude-flow', 'config.json'), 'utf8')), + { 'daemon.idleSecs': 0, 'daemon.resourceThresholds.minFreeMemoryPercent': 0 }); + assert.deepEqual(cfg.rufloDaemon.receipts[path.resolve(cwd)].configKeys, + { 'daemon.idleSecs': 0, 'daemon.resourceThresholds.minFreeMemoryPercent': 0 }); + } finally { rmrf(cwd); } +}); diff --git a/tests/kit/sync-dry-run-preview.test.mjs b/tests/kit/sync-dry-run-preview.test.mjs index f3def05e..19c1b9ec 100644 --- a/tests/kit/sync-dry-run-preview.test.mjs +++ b/tests/kit/sync-dry-run-preview.test.mjs @@ -136,6 +136,44 @@ const DRY = (over = {}) => ({ 'dry-run': true, 'no-upgrade': false, yes: false, const failedDates = async () => ({ code: 1, stdout: '', stderr: 'offline' }); const NOT_CHECKED = /versions not checked online \(offline or timed out\); this plan uses the versions ak recorded/; +test('versions-only dry run previews conditional daemon convergence in a Ruflo project without writes', async () => { + seed(); + const marker = path.join(PROJECT, '.claude-flow', 'config.yaml'); + fs.mkdirSync(path.dirname(marker), { recursive: true }); + fs.writeFileSync(marker, 'daemon:\n maxConcurrent: 7\n'); + const beforeHome = snapshot(HOME); + const beforeProject = snapshot(PROJECT); + try { + const { result, out } = await syncLines({ + flags: DRY(), fetchLatest: async () => null, releaseDatesRunner: failedDates, + collectFn: async () => [{ subsystem: 'versions', level: 'warn', message: 'ruflo update available', fix: 'sync upgrades', repair: 'sync' }], + }); + assert.equal(result, 0, out); + assert.match(out, /\[daemons\].*installed Ruflo version/); + assert.doesNotMatch(out, /daemon\.idleSecs.*0/, 'the target version is not yet known'); + assertUnchanged(beforeHome, HOME); + assertUnchanged(beforeProject, PROJECT); + } finally { fs.rmSync(marker); } +}); + +test('versions-triggered daemon preview honors skip daemons and skip versions', async () => { + seed(); + const marker = path.join(PROJECT, '.claude-flow', 'config.yaml'); + fs.mkdirSync(path.dirname(marker), { recursive: true }); + fs.writeFileSync(marker, 'daemon:\n maxConcurrent: 7\n'); + const collectFn = async () => [{ subsystem: 'versions', level: 'warn', message: 'update available', fix: 'sync upgrades', repair: 'sync' }]; + try { + const daemonSkipped = await syncLines({ + flags: DRY({ skip: ['daemons'] }), fetchLatest: async () => null, releaseDatesRunner: failedDates, collectFn, + }); + assert.match(daemonSkipped.out, /skipped by request: \[daemons\]/); + const versionSkipped = await syncLines({ + flags: DRY({ skip: ['versions'] }), fetchLatest: async () => null, releaseDatesRunner: failedDates, collectFn, + }); + assert.doesNotMatch(versionSkipped.out, /\[daemons\]/); + } finally { fs.rmSync(marker); } +}); + test('a dry run looks the versions up online and plans the upgrade a fresh cache does not know about, recording nothing', () => { const root = seed({ age: HOUR }); const env = spawnEnv(HOME); diff --git a/tests/kit/sync-self-freshness.test.mjs b/tests/kit/sync-self-freshness.test.mjs index 6d726340..2bfce5d2 100644 --- a/tests/kit/sync-self-freshness.test.mjs +++ b/tests/kit/sync-self-freshness.test.mjs @@ -113,6 +113,103 @@ test('a failed next lookup retains its cached candidate without claiming a fresh assert.equal(loadKitConfig().versionCheck.self.last, 1); }); +test('partial self answers retry once per TTL without renewing the cached observation', async t => { + seed(); + const cfg = loadKitConfig(); + cfg.versionCheck.self = { last: 100, observedAt: 80, best: { version: '4.0.0-alpha.50', tag: 'next' } }; + writeKitConfig(home, cfg); + let now = 200_000_000; + t.mock.method(Date, 'now', () => now); + const tags = []; + const fetchLatest = async (_pkg, tag) => { tags.push(tag); return tag === 'latest' ? '4.0.0-alpha.0' : null; }; + const first = await selfDrift({ pkgRoot, fetchLatest }); + assert.deepEqual([first.latest, first.tag], ['4.0.0-alpha.50', 'next']); + assert.deepEqual(loadKitConfig().versionCheck.self, { + last: 100, observedAt: 80, best: { version: '4.0.0-alpha.50', tag: 'next' }, + attempt: { at: now, tags: ['latest', 'next'] }, + }); + const afterFirst = fs.readFileSync(paths.kitConfigPath(), 'utf8'); + now += 1000; + assert.equal((await selfDrift({ pkgRoot, fetchLatest })).latest, first.latest); + assert.deepEqual(tags, ['latest', 'next']); + assert.equal(fs.readFileSync(paths.kitConfigPath(), 'utf8'), afterFirst); + await selfDrift({ pkgRoot, force: true, fetchLatest }); + assert.deepEqual(tags, ['latest', 'next', 'latest', 'next']); + now += 24 * 3600_000; + await selfDrift({ pkgRoot, fetchLatest }); + assert.deepEqual(tags, ['latest', 'next', 'latest', 'next', 'latest', 'next']); + assert.equal(loadKitConfig().versionCheck.self.observedAt, 80); + now += 1000; + const recovered = await selfDrift({ pkgRoot, force: true, fetchLatest: async (_pkg, tag) => + tag === 'next' ? '4.0.0-alpha.51' : null }); + assert.equal(recovered.latest, '4.0.0-alpha.51'); + assert.deepEqual(loadKitConfig().versionCheck.self, + { last: now, observedAt: now, best: { version: '4.0.0-alpha.51', tag: 'next' }, lastTags: ['latest', 'next'] }); +}); + +test('a stable lookup cannot make an untried prerelease channel fresh', async t => { + seed('4.0.0'); + const cfg = loadKitConfig(); + cfg.versionCheck.self = { last: 100, observedAt: 80, best: { version: '4.0.0', tag: 'latest' } }; + writeKitConfig(home, cfg); + let now = 200_000_000; + t.mock.method(Date, 'now', () => now); + const stableTags = []; + await selfDrift({ pkgRoot, fetchLatest: async (_pkg, tag) => { + stableTags.push(tag); return null; + } }); + assert.deepEqual(stableTags, ['latest']); + const stable = loadKitConfig().versionCheck.self; + assert.deepEqual(stable, { + last: now, observedAt: 80, best: { version: '4.0.0', tag: 'latest' }, lastTags: ['latest'], + }); + fs.writeFileSync(path.join(pkgRoot, 'package.json'), JSON.stringify({ name: KIT_PKG, version: '4.0.0-alpha.1' })); + now += 1000; + const prereleaseTags = []; + const offline = async (_pkg, tag) => { prereleaseTags.push(tag); return null; }; + const first = await selfDrift({ pkgRoot, fetchLatest: offline }); + assert.deepEqual(prereleaseTags, ['latest', 'next']); + assert.deepEqual([first.latest, first.tag], ['4.0.0', 'latest']); + const afterFirst = fs.readFileSync(paths.kitConfigPath(), 'utf8'); + await selfDrift({ pkgRoot, fetchLatest: offline }); + assert.deepEqual(prereleaseTags, ['latest', 'next']); + assert.equal(fs.readFileSync(paths.kitConfigPath(), 'utf8'), afterFirst); + assert.equal(loadKitConfig().versionCheck.self.observedAt, 80); +}); + +test('a legacy latest-only record does not suppress an untried next channel', async t => { + seed('4.0.0-alpha.1'); + const cfg = loadKitConfig(); + cfg.versionCheck.self = { last: 200_000_000, observedAt: 100, best: { version: '4.0.0', tag: 'latest' } }; + writeKitConfig(home, cfg); + t.mock.method(Date, 'now', () => 200_001_000); + const tags = []; + await selfDrift({ pkgRoot, fetchLatest: async (_pkg, tag) => { tags.push(tag); return null; } }); + assert.deepEqual(tags, ['latest', 'next']); + assert.equal(loadKitConfig().versionCheck.self.observedAt, 100); +}); + +test('a successful stable observation also leaves next untried after a channel switch', async t => { + seed('4.0.0'); + const cfg = loadKitConfig(); + cfg.versionCheck.self.last = 100; + writeKitConfig(home, cfg); + let now = 200_000_000; + t.mock.method(Date, 'now', () => now); + await selfDrift({ pkgRoot, fetchLatest: async () => '4.0.1' }); + assert.deepEqual(loadKitConfig().versionCheck.self.lastTags, ['latest']); + assert.equal(loadKitConfig().versionCheck.self.observedAt, now); + fs.writeFileSync(path.join(pkgRoot, 'package.json'), JSON.stringify({ name: KIT_PKG, version: '4.0.0-alpha.1' })); + now += 1000; + const tags = []; + const result = await selfDrift({ pkgRoot, fetchLatest: async (_pkg, tag) => { + tags.push(tag); return tag === 'next' ? '4.1.0-alpha.1' : null; + } }); + assert.deepEqual(tags, ['latest', 'next']); + assert.deepEqual([result.latest, result.tag], ['4.1.0-alpha.1', 'next']); + assert.deepEqual(loadKitConfig().versionCheck.self.lastTags, ['latest', 'next']); +}); + test('stable installs reject cached next-channel candidates when latest is unavailable', async () => { seed('4.0.0'); const cfg = loadKitConfig(); @@ -145,6 +242,68 @@ test('a stable install whose record holds only a next-channel candidate is not r assert.deepEqual(loadKitConfig().versionCheck.self.best, { version: '5.0.0-alpha.1', tag: 'next' }); }); +test('offline self attempts are scoped to tags and do not persist in read-only modes', async t => { + seed('4.0.0'); + const cfg = loadKitConfig(); + cfg.versionCheck.self = { last: 100, observedAt: 80, best: { version: '5.0.0-alpha.1', tag: 'next' } }; + writeKitConfig(home, cfg); + let now = 200_000_000; + t.mock.method(Date, 'now', () => now); + const tags = []; + const fetchLatest = async (_pkg, tag) => { tags.push(tag); return null; }; + assert.equal((await selfDrift({ pkgRoot, fetchLatest })).latest, null); + assert.deepEqual(loadKitConfig().versionCheck.self, { + last: 100, observedAt: 80, best: { version: '5.0.0-alpha.1', tag: 'next' }, + attempt: { at: now, tags: ['latest'] }, + }); + const afterFirst = fs.readFileSync(paths.kitConfigPath(), 'utf8'); + await selfDrift({ pkgRoot, fetchLatest }); + assert.deepEqual(tags, ['latest']); + assert.equal(fs.readFileSync(paths.kitConfigPath(), 'utf8'), afterFirst); + await selfDrift({ pkgRoot, force: true, fetchLatest }); + assert.deepEqual(tags, ['latest', 'latest']); + fs.writeFileSync(path.join(pkgRoot, 'package.json'), JSON.stringify({ name: KIT_PKG, version: '4.0.0-alpha.1' })); + await selfDrift({ pkgRoot, fetchLatest }); + assert.deepEqual(tags, ['latest', 'latest', 'latest', 'next'], 'the untried channel is probed'); + const beforeReadOnly = fs.readFileSync(paths.kitConfigPath(), 'utf8'); + await selfDrift({ pkgRoot, force: true, record: false, fetchLatest }); + assert.deepEqual(tags.slice(-2), ['latest', 'next']); + assert.equal(fs.readFileSync(paths.kitConfigPath(), 'utf8'), beforeReadOnly); + await selfDrift({ pkgRoot, force: true, cacheOnly: true, fetchLatest }); + assert.equal(fs.readFileSync(paths.kitConfigPath(), 'utf8'), beforeReadOnly); + assert.equal(tags.length, 6); + now += 24 * 3600_000; + await selfDrift({ pkgRoot, fetchLatest }); + assert.equal(tags.length, 8); +}); + +test('malformed self attempt stamps cannot suppress an offline retry', async t => { + seed('4.0.0'); + const cfg = loadKitConfig(); + cfg.versionCheck.self = { + last: 100, best: { version: '5.0.0-alpha.1', tag: 'next' }, + attempt: { at: Number.MAX_SAFE_INTEGER, tags: ['latest'] }, + }; + writeKitConfig(home, cfg); + t.mock.method(Date, 'now', () => 200_000_000); + let calls = 0; + await selfDrift({ pkgRoot, fetchLatest: async () => { calls += 1; return null; } }); + assert.equal(calls, 1); + assert.deepEqual(loadKitConfig().versionCheck.self.attempt, { at: 200_000_000, tags: ['latest'] }); + for (const attempt of [ + { at: '200000000', tags: ['latest'] }, + { at: 200_000_000.5, tags: ['latest'] }, + { at: 200_000_000, tags: ['next'] }, + { at: 200_000_000, tags: 'latest' }, + ]) { + const next = loadKitConfig(); + next.versionCheck.self.attempt = attempt; + writeKitConfig(home, next); + await selfDrift({ pkgRoot, fetchLatest: async () => { calls += 1; return null; } }); + } + assert.equal(calls, 5); +}); + test('successful registry observations supersede cached versions even after a channel rollback', async () => { seed(); const cfg = loadKitConfig(); diff --git a/tests/kit/telemetry-cli.test.mjs b/tests/kit/telemetry-cli.test.mjs index f7f2ad7b..a91132f2 100644 --- a/tests/kit/telemetry-cli.test.mjs +++ b/tests/kit/telemetry-cli.test.mjs @@ -66,6 +66,15 @@ test('should_keepErrorsGeneric_when_malformedFilesContainSecrets', t => { const result = cli(['validate', file]); assert.equal(result.status, 2); assert.doesNotMatch(result.stdout + result.stderr, /SECRET/); + assert.deepEqual(Object.keys(JSON.parse(result.stdout)), ['error', 'exitCode']); + assert.equal(JSON.parse(result.stdout).exitCode, 2); +}); +test('telemetry parser rejection remains private and machine-readable', () => { + const result = cli(['schema', '--private-path=/secret/SECRET']); + assert.equal(result.status, 2); + assert.deepEqual(Object.keys(JSON.parse(result.stdout)), ['error', 'exitCode']); + assert.equal(JSON.parse(result.stdout).exitCode, 2); + assert.doesNotMatch(result.stdout + result.stderr, /SECRET|\/secret/); }); test('should_degradeSourcesIndependently_when_usageReaderFails', async () => { const { collectSnapshot } = await import('../../src/lib/telemetry/collect.mjs'); diff --git a/tests/kit/upstream-watch-registry.test.mjs b/tests/kit/upstream-watch-registry.test.mjs index 7b1d43b9..18eca394 100644 --- a/tests/kit/upstream-watch-registry.test.mjs +++ b/tests/kit/upstream-watch-registry.test.mjs @@ -193,11 +193,10 @@ test('the tracking issues carry their whole upstream remainder', () => { const t240 = entry(doc, 'pacphi/agentic-kit#240'); assert.deepEqual([...t240.tracks].sort(), ['proffesor-for-testing/agentic-qe#574', 'proffesor-for-testing/agentic-qe#719']); assert.match(t240.adjustment, /agentic-qe#574/); - assert.ok(t240.kitImpact.files.includes('src/lib/aqe-readiness.mjs')); + assert.ok(t240.kitImpact.files.includes('docs/troubleshooting.md')); }); -// agentic-qe#719 is a partial fix for #574: releasing it alone must not dispatch removing the busy rule. -test('the partial fix agentic-qe#719 is context only; agentic-qe#574 drives the dispatch', () => { +test('the #574 exception retirement records released conformance without closing #240', () => { const doc = document(); const partial = entry(doc, 'proffesor-for-testing/agentic-qe#719'); assert.equal(partial.mapping, 'unmapped'); @@ -206,7 +205,13 @@ test('the partial fix agentic-qe#719 is context only; agentic-qe#574 drives the assert.match(partial.note, /agentic-qe#574/); const driver = entry(doc, 'proffesor-for-testing/agentic-qe#574'); assert.equal(driver.mapping, 'mapped'); - assert.match(driver.adjustment, /busy rule/); + assert.equal(driver.status, 'retired'); + assert.match(driver.adjustment, /macOS and Linux/); + assert.match(driver.adjustment, /not a universal AQE minimum/); + assert.ok(driver.history.some((item) => item.event === 'retired' && item.date === '2026-09-29')); + const tracker = entry(doc, 'pacphi/agentic-kit#240'); + assert.equal(tracker.status, 'watching'); + assert.match(tracker.adjustment, /final main PR/); }); test('stale threads are mapped to what ak carries, or retired with a reason', () => { diff --git a/tests/kit/verify-memory-routes.test.mjs b/tests/kit/verify-memory-routes.test.mjs index b1e87d23..cf530a51 100644 --- a/tests/kit/verify-memory-routes.test.mjs +++ b/tests/kit/verify-memory-routes.test.mjs @@ -241,8 +241,11 @@ test('the quick live memory check proves the CLI round trip without starting an assert.ok(!calls.includes('mcp start'), 'the live check stays quick: routing is observed only by the memory-routes proof'); }); -test('--only memory-routes runs the memory proof with the route observation and remembers it as memory', posix, async () => { +test('--only memory-routes runs the memory proof with the route observation and remembers it separately', posix, async () => { const cfg = offlineKitConfig(); + const evidence = await import('../../src/lib/live-check-evidence.mjs'); + evidence.recordLiveCheck({ id: 'memory', status: 'failed', source: 'status-refresh-live', + inputsKey: evidence.liveCheckInputsKey('memory') }); const { results, calls } = await withFakeRuflo('aligned', async () => ({ results: await verify.runLiveChecks({ cfg, cwd: PROJECT, only: ['memory-routes'] }), })); @@ -250,4 +253,21 @@ test('--only memory-routes runs the memory proof with the route observation and assert.equal(results[0].status, 'passed', JSON.stringify(results[0])); assert.ok(calls.includes('mcp start'), 'the named proof observes CLI↔MCP routing'); assert.ok(results[0].entries.some((e) => /see each other's writes/.test(e.text)), JSON.stringify(results[0].entries)); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'passed'); + assert.equal(evidence.readLiveCheck('memory').status, 'passed', + 'the successful CLI round trip refreshes the generic memory row'); +}); + +test('an unavailable route observation is inconclusive even after a successful CLI round trip', posix, async () => { + const cfg = offlineKitConfig(); + const evidence = await import('../../src/lib/live-check-evidence.mjs'); + evidence.recordLiveCheck({ id: 'memory', status: 'failed', source: 'status-refresh-live', + inputsKey: evidence.liveCheckInputsKey('memory') }); + const { results } = await withFakeRuflo('mcp-down', async () => ({ + results: await verify.runLiveChecks({ cfg, cwd: PROJECT, only: ['memory-routes'] }), + })); + assert.equal(results[0].status, 'inconclusive'); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'inconclusive'); + assert.equal(evidence.readLiveCheck('memory').status, 'passed', + 'MCP unavailability does not undo the successful CLI proof'); }); diff --git a/tests/kit/windows-npm-shim.test.mjs b/tests/kit/windows-npm-shim.test.mjs new file mode 100644 index 00000000..fc965df1 --- /dev/null +++ b/tests/kit/windows-npm-shim.test.mjs @@ -0,0 +1,151 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { resolveShim, run } from '../../src/lib/exec.mjs'; +import { callMcpTools } from '../../src/lib/mcp-tool-call.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; + +const templates = Object.fromEntries(['cmd', 'ps1'].map((ext) => [ext, + fs.readFileSync(new URL(`../fixtures/npm-windows-shim/ruflo.${ext}`, import.meta.url), 'utf8')])); +function fixture(t) { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-npm-shim-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + const entry = path.join(root, 'node_modules', 'ruflo', 'bin', 'ruflo.js'); + const manifest = path.join(root, 'node_modules', 'ruflo', 'package.json'); + fs.mkdirSync(path.dirname(entry), { recursive: true }); + fs.writeFileSync(entry, '#!/usr/bin/env node\n'); + fs.writeFileSync(manifest, JSON.stringify({ name: 'ruflo', bin: { ruflo: 'bin/ruflo.js' } })); + for (const ext of ['cmd', 'ps1']) fs.writeFileSync(path.join(root, `ruflo.${ext}`), templates[ext]); + const powershell = path.join(root, 'System32', 'WindowsPowerShell', 'v1.0', 'powershell.exe'); + fs.mkdirSync(path.dirname(powershell), { recursive: true }); fs.writeFileSync(powershell, 'fixture'); + const node = path.join(root, 'node.exe'); fs.writeFileSync(node, 'fixture'); + const env = { PATH: root, PATHEXT: '.EXE;.CMD', SystemRoot: root }; + return { root, entry, manifest, node, powershell, env }; +} + +test('recognized npm shim maps to its declared public bin with literal argv and adjacent Node', (t) => { + const f = fixture(t); + const args = ['mcp', 'start', 'a & b', 'quoted " argument', '$(ignored)', '', '\n']; + assert.deepEqual(resolveShim('ruflo', args, { windows: true, env: f.env }), { + command: f.node, args: [f.entry, ...args], resolved: true, + }); + assert.deepEqual(resolveShim(path.join(f.root, 'ruflo.cmd'), args, { windows: true, env: f.env }), { + command: f.node, args: [f.entry, ...args], resolved: true, + }); + assert.equal(resolveShim('ruflo', args, { windows: true, env: f.env, npmBin: false }).command, f.powershell); + for (const ext of ['cmd', 'ps1']) { + fs.writeFileSync(path.join(f.root, `ruflo.${ext}`), templates[ext].replaceAll('\r\n', '\n').replaceAll('\n', '\r\n')); + } + assert.equal(resolveShim('ruflo', args, { windows: true, env: f.env }).command, f.node, 'CRLF templates remain recognized'); +}); + +test('PATH-selected installation and case-insensitive Node lookup beat current runtime/global guesses', (t) => { + const f = fixture(t); const other = fixture(t); + fs.rmSync(f.node); + const env = { pAtH: `${f.root}${path.delimiter}${other.root}`, pAtHeXt: '.CMD', sYsTeMrOoT: f.root }; + assert.deepEqual(resolveShim('ruflo', ['mcp', 'start'], { windows: true, env }), { + command: other.node, args: [f.entry, 'mcp', 'start'], resolved: true, + }); +}); + +for (const mutation of ['cmd', 'ps1', 'manifest', 'foreign-package', 'undeclared-bin', 'traversal', 'shebang', 'missing-bin', 'missing-node']) { + test(`${mutation} cannot silently bypass a custom or unverified wrapper`, (t) => { + const f = fixture(t); + if (mutation === 'cmd' || mutation === 'ps1') fs.appendFileSync(path.join(f.root, `ruflo.${mutation}`), '\ncustom-behavior\n'); + if (mutation === 'manifest') fs.writeFileSync(f.manifest, 'broken json'); + if (mutation === 'foreign-package') fs.writeFileSync(f.manifest, JSON.stringify({ name: 'other', bin: { ruflo: 'bin/ruflo.js' } })); + if (mutation === 'undeclared-bin') fs.writeFileSync(f.manifest, JSON.stringify({ name: 'ruflo', bin: { other: 'bin/ruflo.js' } })); + if (mutation === 'traversal') fs.writeFileSync(f.manifest, JSON.stringify({ name: 'ruflo', bin: { ruflo: '../other.js' } })); + if (mutation === 'shebang') fs.writeFileSync(f.entry, '#!/usr/bin/env node --require injected\n'); + if (mutation === 'missing-bin') fs.rmSync(f.entry); + if (mutation === 'missing-node') fs.rmSync(f.node); + assert.equal(resolveShim('ruflo', [], { windows: true, env: f.env }).command, f.powershell); + }); +} + +test('custom wrapper first on PATH and native executable precedence remain authoritative', (t) => { + const first = fixture(t); const second = fixture(t); + fs.appendFileSync(path.join(first.root, 'ruflo.ps1'), '\n# user customization\n'); + const env = { ...first.env, PATH: `${first.root}${path.delimiter}${second.root}` }; + assert.equal(resolveShim('ruflo', [], { windows: true, env }).command, first.powershell); + fs.writeFileSync(path.join(first.root, 'ruflo.exe'), 'native'); + assert.deepEqual(resolveShim('ruflo', ['literal'], { windows: true, env }), { + command: path.join(first.root, 'ruflo.exe'), args: ['literal'], resolved: true, + }); +}); + +test('bin symlink escaping its package cannot authorize bypass', { skip: process.platform === 'win32' }, (t) => { + const f = fixture(t); + const outside = path.join(f.root, 'outside.js'); fs.writeFileSync(outside, '#!/usr/bin/env node\n'); + fs.rmSync(f.entry); fs.symlinkSync(outside, f.entry); + assert.equal(resolveShim('ruflo', [], { windows: true, env: f.env }).command, f.powershell); +}); + +test('recognized public entry point answers MCP initialize without stdin EOF', { skip: process.platform === 'win32' }, async (t) => { + const f = fixture(t); + fs.rmSync(f.node); fs.symlinkSync(process.execPath, f.node); + fs.writeFileSync(f.entry, `#!/usr/bin/env node +require('node:readline').createInterface({input:process.stdin}).on('line', line => { + const r=JSON.parse(line); + if(r.method==='initialize') process.stdout.write(JSON.stringify({jsonrpc:'2.0',id:r.id,result:{protocolVersion:'2024-11-05'}})+'\\n'); +}); +`); + const spec = resolveShim('ruflo', ['mcp', 'start'], { windows: true, env: f.env }); + assert.equal(spec.command, f.node); + const result = await callMcpTools({ ...spec, cwd: f.root, env: spawnEnv(f.root), calls: [], timeoutMs: 1000 }); + assert.equal(result.status, 'ok'); +}); + +test('scoped package aliases and string bin declarations require manifest agreement', (t) => { + const f = fixture(t); + fs.writeFileSync(f.manifest, JSON.stringify({ name: 'ruflo', bin: './bin/ruflo.js' })); + assert.equal(resolveShim('ruflo', [], { windows: true, env: f.env }).command, f.node); + const target = 'node_modules/@scope/tool/bin/cli.js'; + const entry = path.join(f.root, ...target.split('/')); + fs.mkdirSync(path.dirname(entry), { recursive: true }); + fs.writeFileSync(entry, '#!/usr/bin/env node\n'); + const manifest = path.join(f.root, 'node_modules', '@scope', 'tool', 'package.json'); + fs.writeFileSync(manifest, JSON.stringify({ name: '@scope/tool', bin: { ruflo: 'bin/cli.js' } })); + for (const ext of ['cmd', 'ps1']) { + const oldTarget = ext === 'cmd' ? 'node_modules\\ruflo\\bin\\ruflo.js' : 'node_modules/ruflo/bin/ruflo.js'; + fs.writeFileSync(path.join(f.root, `ruflo.${ext}`), templates[ext].replaceAll(oldTarget, ext === 'cmd' ? target.replaceAll('/', '\\') : target)); + } + assert.deepEqual(resolveShim('ruflo', [], { windows: true, env: f.env }), { + command: f.node, args: [entry], resolved: true, + }); + fs.writeFileSync(manifest, JSON.stringify({ name: '@scope/tool', bin: 'bin/cli.js' })); + assert.equal(resolveShim('ruflo', [], { windows: true, env: f.env }).command, f.powershell); +}); + +test('run honors a differently cased PATH override without duplicate environment keys', { skip: process.platform === 'win32' }, async (t) => { + const f = fixture(t); + fs.rmSync(f.node); fs.symlinkSync(process.execPath, f.node); + fs.writeFileSync(f.entry, `#!/usr/bin/env node +process.stdout.write(JSON.stringify({argv:process.argv.slice(2),paths:Object.keys(process.env).filter(k=>k.toUpperCase()==='PATH')})); +`); + const args = ['mcp', 'quoted " value', 'a & b', '']; + const result = await run('ruflo', args, { windows: true, env: { + pAtH: f.root, pAtHeXt: '.CMD', sYsTeMrOoT: f.root, + } }); + assert.equal(result.code, 0, result.stderr); + assert.deepEqual(JSON.parse(result.stdout), { argv: args, paths: ['pAtH'] }); +}); + +test('an earlier cmd with no safe sibling does not fall through to another installation', (t) => { + const first = fixture(t); const second = fixture(t); + fs.rmSync(path.join(first.root, 'ruflo.ps1')); + const env = { ...first.env, PATH: `${first.root}${path.delimiter}${second.root}` }; + assert.deepEqual(resolveShim('ruflo', ['mcp'], { windows: true, env }), { + command: 'ruflo', args: ['mcp'], resolved: false, + }); +}); + +test('package symlink escaping the selected installation cannot authorize bypass', { skip: process.platform === 'win32' }, (t) => { + const first = fixture(t); const second = fixture(t); + const pkg = path.join(first.root, 'node_modules', 'ruflo'); + fs.rmSync(pkg, { recursive: true }); + fs.symlinkSync(path.join(second.root, 'node_modules', 'ruflo'), pkg); + assert.equal(resolveShim('ruflo', [], { windows: true, env: first.env }).command, first.powershell); +}); diff --git a/tests/live/aqe-live-lock-conformance.test.mjs b/tests/live/aqe-live-lock-conformance.test.mjs new file mode 100644 index 00000000..bf5e6f2b --- /dev/null +++ b/tests/live/aqe-live-lock-conformance.test.mjs @@ -0,0 +1,150 @@ +// Opt-in native contract for agentic-qe#574. No installed package is patched. +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { createHash } from 'node:crypto'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { pathToFileURL } from 'node:url'; +import { createProcessScope } from './aqe-live-lock-process.mjs'; + +const required = process.env.AK_AQE_LOCK_LIVE === '1'; +const digest = (file) => createHash('sha256').update(fs.readFileSync(file)).digest('hex'); +const pause = (ms) => new Promise((resolve) => setTimeout(resolve, ms)); + +function installedRoot() { + if (process.env.AK_AQE_PACKAGE_ROOT) { + assert.ok(path.isAbsolute(process.env.AK_AQE_PACKAGE_ROOT), 'package root must be absolute'); + return fs.realpathSync(process.env.AK_AQE_PACKAGE_ROOT); + } + for (const dir of (process.env.PATH ?? '').split(path.delimiter)) { + const bin = path.join(dir, 'aqe'); + if (dir && fs.existsSync(bin)) { + const root = path.resolve(path.dirname(fs.realpathSync(bin)), '../..'); + if (fs.existsSync(path.join(root, 'package.json'))) return root; + } + } + throw new Error('AQE missing: set AK_AQE_PACKAGE_ROOT to absolute installed package root'); +} + +function childEnv(root, project) { + const home = path.join(root, 'home'); + const tmp = path.join(root, 'tmp'); + fs.mkdirSync(home); fs.mkdirSync(tmp); + return { + PATH: process.env.PATH ?? '', + ...(process.platform === 'win32' ? { + SystemRoot: process.env.SystemRoot ?? 'C:\\Windows', + ComSpec: process.env.ComSpec ?? 'C:\\Windows\\System32\\cmd.exe', + PATHEXT: process.env.PATHEXT ?? '.COM;.EXE;.BAT;.CMD', + } : {}), + HOME: home, USERPROFILE: home, TMPDIR: tmp, TMP: tmp, TEMP: tmp, + LANG: 'en_US.UTF-8', CI: '1', NO_COLOR: '1', + XDG_CONFIG_HOME: path.join(home, 'config'), XDG_STATE_HOME: path.join(home, 'state'), + XDG_DATA_HOME: path.join(home, 'data'), XDG_CACHE_HOME: path.join(home, 'cache'), + APPDATA: path.join(home, 'appdata'), LOCALAPPDATA: path.join(home, 'localappdata'), + CODEX_HOME: path.join(home, 'codex'), CLAUDE_CONFIG_DIR: path.join(home, 'claude'), + HERMES_HOME: path.join(home, 'hermes'), npm_config_prefix: path.join(home, 'npm-prefix'), + npm_config_cache: path.join(home, 'npm-cache'), + MISE_DATA_DIR: path.join(home, 'mise-data'), MISE_CONFIG_DIR: path.join(home, 'mise-config'), + MISE_CACHE_DIR: path.join(home, 'mise-cache'), AQE_PROJECT_ROOT: project, + AQE_MEMORY_PATH: path.join(project, '.agentic-qe', 'memory.db'), + AQE_STORAGE_PATH: path.join(project, '.agentic-qe'), + CLAUDE_FLOW_MEMORY_PATH: path.join(project, '.swarm'), + CLAUDE_FLOW_DB_PATH: path.join(project, '.swarm', 'memory.db'), + RUFLO_DAEMON_AUTOSTART: '0', + }; +} + +async function bounded(scope, file, args, options, limit) { + const run = scope.launch(process.execPath, [file, ...args], options); + return scope.wait(run, limit); +} + +function snapshot(store) { + return Object.fromEntries(fs.readdirSync(store).filter((n) => n.startsWith('patterns.rvf')) + .sort().map((name) => { + const file = path.join(store, name); + return [name, { sha256: digest(file), bytes: fs.statSync(file).size }]; + })); +} + +function check(label, result, owner, before, store) { + const output = result.stdout + result.stderr; + assert.equal(result.timedOut, false, `${label} timed out`); + assert.equal(result.code, 0, `${label} exited ${result.code}: ${output.slice(-1200)}`); + assert.equal(owner.closed, false, `native holder closed during ${label}`); + assert.equal(owner.child.exitCode, null, `native holder exited during ${label}`); + assert.equal(owner.child.signalCode, null, `native holder signaled during ${label}`); + assert.ifError(owner.error); + assert.match(output, /is locked by a live process/, `${label} missed live-lock warning`); + assert.match(output, /LockHeld|0x0300/, `${label} missed LockHeld fallback`); + assert.doesNotMatch(output, /FsyncFailed|0x0303/, `${label} emitted FsyncFailed`); + assert.deepEqual(snapshot(store), before, `${label} changed RVF bytes`); + assert.ok(!fs.readdirSync(store).some((n) => n.includes('.corrupt-')), `${label} quarantined RVF`); +} + +test('installed AQE degrades under a live native RVF lock without changing the store', + { skip: !required && 'set AK_AQE_LOCK_LIVE=1 for native proof', timeout: 420_000 }, async (t) => { + const root = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-live-lock-'))); + const scope = createProcessScope(t.signal); + t.after(async () => { + try { + await scope.closeAll(); + } catch (error) { + throw new Error(`owned child closure unverified; retained ${root}`, { cause: error }); + } + fs.rmSync(root, { recursive: true, force: true, maxRetries: 3 }); + }); + const project = path.join(root, 'project'); fs.mkdirSync(project); + fs.writeFileSync(path.join(project, 'package.json'), '{"name":"aqe-live-lock-probe","version":"1.0.0","type":"module"}\n'); + const env = childEnv(root, project); + const packageRoot = installedRoot(); + const pkg = JSON.parse(fs.readFileSync(path.join(packageRoot, 'package.json'), 'utf8')); + assert.equal(pkg.name, 'agentic-qe'); + if (process.env.AK_AQE_EXPECTED_VERSION) assert.equal(pkg.version, process.env.AK_AQE_EXPECTED_VERSION); + const entry = path.join(packageRoot, 'dist', 'cli', 'bundle.js'); + const adapter = path.join(packageRoot, 'dist', 'integrations', 'ruvector', 'shared-rvf-adapter.js'); + assert.ok(fs.existsSync(entry) && fs.existsSync(adapter), 'installed AQE CLI and adapter required'); + const sourceHashes = { entry: digest(entry), adapter: digest(adapter) }; + const options = { cwd: project, env }; + const init = await bounded(scope, entry, ['init', '--minimal', '--auto'], options, 150_000); + assert.equal(init.timedOut, false); + assert.equal(init.code, 0, `aqe init failed: ${(init.stdout + init.stderr).slice(-1200)}`); + const store = path.join(project, '.agentic-qe'); + const ready = path.join(root, 'ready'); + const holderFile = path.join(root, 'holder.mjs'); + const moduleUrl = pathToFileURL(adapter).href; + fs.writeFileSync(holderFile, `import { createRequire } from 'node:module'; import fs from 'node:fs';\nglobalThis.require=createRequire(${JSON.stringify(adapter)});\nconst {getSharedRvfAdapter}=await import(${JSON.stringify(moduleUrl)});\nglobalThis.hold=getSharedRvfAdapter(${JSON.stringify(store)},384);\nif(!globalThis.hold)process.exit(2);\nfs.writeFileSync(${JSON.stringify(ready)},String(process.pid));\nconst timer=setInterval(()=>{if(!globalThis.hold)process.exit(3)},1000);\nprocess.on('SIGTERM',()=>{clearInterval(timer);globalThis.hold.close();process.exit(0)});\n`); + const owner = scope.launch(process.execPath, [holderFile], options); + try { + const deadline = Date.now() + 30_000; + while (!fs.existsSync(ready) && !owner.closed && !owner.error && owner.child.exitCode === null + && owner.child.signalCode === null && Date.now() < deadline) await pause(50); + assert.ifError(owner.error); + assert.ok(fs.existsSync(ready), 'holder did not signal ready'); + assert.equal(Number(fs.readFileSync(ready, 'utf8')), owner.child.pid, 'holder PID mismatch'); + assert.equal(owner.closed, false, 'holder closed after ready'); + assert.equal(owner.child.exitCode, null, 'holder exited after ready'); + assert.equal(owner.child.signalCode, null, 'holder signaled after ready'); + const before = snapshot(store); + assert.ok(before['patterns.rvf'] && before['patterns.rvf.lock'], 'native RVF and lock required'); + const status = await bounded(scope, entry, ['status'], options, 150_000); + check('aqe status', status, owner, before, store); + const challengerFile = path.join(root, 'challenger.mjs'); + fs.writeFileSync(challengerFile, `import {createRequire} from 'node:module';\nglobalThis.require=createRequire(${JSON.stringify(adapter)});\nconst {getSharedRvfAdapter}=await import(${JSON.stringify(moduleUrl)});\nconst adapter=getSharedRvfAdapter(${JSON.stringify(store)},384);\nconsole.log(JSON.stringify({fallback:adapter===null}));\nif(adapter){adapter.close();process.exitCode=2}\n`); + const challenger = await bounded(scope, challengerFile, [], options, 30_000); + check('shipped adapter', challenger, owner, before, store); + assert.match(challenger.stdout, /"fallback":true/, 'adapter did not fall back'); + console.log(JSON.stringify({ aqeVersion: pkg.version, platform: process.platform, + node: process.version, sourceHashes, ownerPid: owner.child.pid, + status: { exit: status.code, liveLock: true, lockHeld: true, fsyncFailed: false }, + adapter: { exit: challenger.code, fallback: true, liveLock: true, lockHeld: true, fsyncFailed: false }, + before, after: snapshot(store) })); + } finally { + const closed = await scope.stop(owner, 10_000); + if (!t.signal.aborted) { + assert.equal(closed.code, 0, `holder failed to close: ${closed.stderr.slice(-1000)}`); + } + } + }); diff --git a/tests/live/aqe-live-lock-process.mjs b/tests/live/aqe-live-lock-process.mjs new file mode 100644 index 00000000..433ab8a3 --- /dev/null +++ b/tests/live/aqe-live-lock-process.mjs @@ -0,0 +1,107 @@ +import { spawn } from 'node:child_process'; +import { acquireRunRootHold, releaseRunRootHold } from '../../scripts/run-roots.mjs'; +import { envValue } from '../kit/helpers/home-sandbox.mjs'; + +const CLOSE_LIMIT_MS = 10_000; + +function closedWithin(run, ms) { + if (run.closed) return Promise.resolve(run.result); + let timer; + return Promise.race([ + run.done, + new Promise((_, reject) => { + timer = setTimeout(() => reject(new Error(`call-owned child ${run.child.pid ?? 'unspawned'} did not close`)), ms); + }), + ]).finally(() => clearTimeout(timer)); +} + +export function createProcessScope(signal, { closeLimitMs = CLOSE_LIMIT_MS, platform = process.platform } = {}) { + // Register uncertainty with the enclosing guarded runner before any child starts. + const hold = acquireRunRootHold(); + const runs = new Set(); + let closed = false; + let closing = false; + const killLive = () => { + for (const run of runs) if (!run.closed) run.child.kill('SIGKILL'); + }; + signal.addEventListener('abort', killLive, { once: true }); + + function launch(command, args, options) { + if (closing || signal.aborted) throw Error('cannot launch after process scope closing or aborted'); + if (!options?.env || typeof options.env !== 'object' || Array.isArray(options.env)) { + throw Error('call-owned child requires an explicit sandbox env'); + } + const env = options.env; + const missing = (key) => { + const value = envValue(env, key, platform); + return typeof value !== 'string' || !value; + }; + const required = ['HOME', 'USERPROFILE', 'TMPDIR', 'TEMP', 'TMP', 'XDG_CONFIG_HOME', + 'XDG_STATE_HOME', 'APPDATA', 'LOCALAPPDATA']; + if (required.some(missing)) { + throw Error('call-owned child requires sandbox home, temp, and state env'); + } + if (platform === 'win32' && ['SystemRoot', 'ComSpec', 'PATHEXT'].some(missing)) { + throw Error('call-owned child requires Windows process env'); + } + const child = spawn(command, args, { ...options, env, stdio: ['ignore', 'pipe', 'pipe'] }); + const run = { child, closed: false, error: null, stdout: '', stderr: '', done: null, result: null }; + runs.add(run); + child.stdout.setEncoding('utf8'); child.stderr.setEncoding('utf8'); + child.stdout.on('data', (s) => { run.stdout += s; }); + child.stderr.on('data', (s) => { run.stderr += s; }); + // Resolve on close even after spawn error. The caller can inspect error immediately + // while polling readiness, and no delayed rejection can go unhandled. + run.done = new Promise((resolve) => { + child.once('error', (error) => { run.error = error; }); + child.once('close', (code, childSignal) => { + run.closed = true; + run.result = { code, signal: childSignal, stdout: run.stdout, stderr: run.stderr }; + resolve(run.result); + }); + }); + if (signal.aborted) child.kill('SIGKILL'); + return run; + } + + async function wait(run, limit) { + let timedOut = false; + const timer = setTimeout(() => { + timedOut = true; + if (!run.closed) run.child.kill('SIGKILL'); + }, limit); + try { + const result = await closedWithin(run, limit + CLOSE_LIMIT_MS); + if (run.error) throw run.error; + return { ...result, timedOut }; + } finally { clearTimeout(timer); } + } + + async function stop(run, grace = CLOSE_LIMIT_MS) { + if (!run.closed) run.child.kill('SIGTERM'); + let timer; + try { + await Promise.race([ + run.done, + new Promise((resolve) => { timer = setTimeout(resolve, grace); }), + ]); + } finally { + clearTimeout(timer); + if (!run.closed) run.child.kill('SIGKILL'); + } + return closedWithin(run, CLOSE_LIMIT_MS); + } + + async function closeAll() { + if (closed) return; + closing = true; + killLive(); + // A failed close keeps the caller's temporary root intact for diagnosis. + await Promise.all([...runs].map((run) => closedWithin(run, closeLimitMs))); + releaseRunRootHold(hold); + closed = true; + signal.removeEventListener('abort', killLive); + } + + return { launch, wait, stop, closeAll }; +} diff --git a/tests/live/ruflo-windows-diagnostic-process.mjs b/tests/live/ruflo-windows-diagnostic-process.mjs new file mode 100644 index 00000000..5ec1b20a --- /dev/null +++ b/tests/live/ruflo-windows-diagnostic-process.mjs @@ -0,0 +1,103 @@ +// Test-only exact-environment launcher. Every attempt contributes independently +// to the scope's cleanup receipt; uncertainty is sticky until handoff. +import { spawn } from 'node:child_process'; +import { acquireRunRootHold, releaseRunRootHold } from '../../scripts/run-roots.mjs'; + +const delay = (ms) => new Promise((resolve) => setTimeout(resolve, ms)); + +export function createDiagnosticScope({ spawnFn = spawn, platform = process.platform, + killTimeoutMs = 3000, closeTimeoutMs = 5000, + acquireHold = acquireRunRootHold, releaseHold = releaseRunRootHold } = {}) { + const hold = acquireHold(); + const receipts = []; + let released = false; + + async function launch(spec, { cwd, env, timeoutMs, input = '', endInput = true, + until = () => false, tree = true }) { + if (released) throw Error('diagnostic scope already released'); + if (!env || typeof env !== 'object') throw Error('explicit diagnostic environment required'); + const result = { code: null, signal: null, stdout: '', stderr: '', error: null, + timedOut: false, overflow: false, matched: false, cleanupComplete: false, elapsedMs: 0 }; + receipts.push(result); + let child; + let closed = false; + let bytes = 0; + const started = Date.now(); + const live = () => child?.pid && child.exitCode === null && child.signalCode === null; + const capture = (key, chunk) => { + bytes += chunk.length; + result.overflow ||= bytes > 65536; + result[key] = (result[key] + chunk.toString('utf8')).slice(0, 16384); + }; + const waitClose = async () => { + const deadline = Date.now() + closeTimeoutMs; + while (!closed && Date.now() < deadline) await delay(10); + }; + try { + child = spawnFn(spec.command, spec.args, { + cwd, env, shell: false, stdio: ['pipe', 'pipe', 'pipe'], + detached: tree && platform !== 'win32', + }); + child.on('error', (e) => { result.error = e.message; }); + child.once('close', (code, signal) => { + closed = true; result.code = code; result.signal = signal; + }); + child.stdout.on('data', (chunk) => capture('stdout', chunk)); + child.stderr.on('data', (chunk) => capture('stderr', chunk)); + child.stdin.on('error', (e) => { result.error = e.message; }); + child.stdin.write(input); + if (endInput) child.stdin.end(); + while (!closed && !result.error && !result.overflow && Date.now() - started < timeoutMs) { + result.matched = until(result.stdout); + // EOF comparisons must observe natural closure. Stopping on the + // first response can race a server already exiting after stdin end. + if (result.matched && !endInput) break; + await delay(10); + } + result.matched ||= until(result.stdout); + result.timedOut = !closed && !(result.matched && !endInput) && !result.error && !result.overflow; + } catch (error) { + result.error = error.message; + } finally { + // Never throw before cleanup. Failed tree termination is uncertainty even + // if the direct-child fallback subsequently closes its own pipes. + let treeStopped = closed || !child; + if (child && !closed) { + if (live()) { + if (tree && platform === 'win32') { + const killed = await launch({ command: 'taskkill.exe', args: ['/PID', String(child.pid), '/T', '/F'] }, + { cwd, env, timeoutMs: killTimeoutMs, tree: false }); + treeStopped = killed.cleanupComplete && killed.code === 0 && !killed.timedOut && !killed.error; + } else { + try { + if (tree) process.kill(-child.pid, 'SIGKILL'); + else child.kill('SIGKILL'); + treeStopped = true; + } catch { treeStopped = false; } + } + if (!treeStopped && live()) { + try { child.kill('SIGKILL'); } catch { /* preserve uncertainty */ } + } + } + await waitClose(); + } + result.cleanupComplete = treeStopped && (closed || !child); + if (child && !closed) { + // Close local handles and release event-loop references without claiming + // the process tree exited. The scope hold remains active. + child.stdin.destroy(); child.stdout.destroy(); child.stderr.destroy(); + child.unref(); + } + result.elapsedMs = Date.now() - started; + } + return result; + } + + function release() { + if (receipts.some((receipt) => !receipt.cleanupComplete)) return false; + if (!released) releaseHold(hold); + released = true; + return true; + } + return { launch, release, receipts }; +} diff --git a/tests/live/ruflo-windows-transport.test.mjs b/tests/live/ruflo-windows-transport.test.mjs new file mode 100644 index 00000000..d4c91e08 --- /dev/null +++ b/tests/live/ruflo-windows-transport.test.mjs @@ -0,0 +1,95 @@ +// Opt-in diagnosis only. A backend response never substitutes for public routing +// proof. Run alongside (not instead of) ruflo-memory-routing.test.mjs. +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import crypto from 'node:crypto'; +import { spawnEnv, envValue } from '../kit/helpers/home-sandbox.mjs'; +import { resolveShim } from '../../src/lib/exec.mjs'; +import { createDiagnosticScope } from './ruflo-windows-diagnostic-process.mjs'; + +const sha = (file) => crypto.createHash('sha256').update(fs.readFileSync(file)).digest('hex'); +const initialized = (stdout) => stdout.split('\n').some((line) => { + try { const r = JSON.parse(line); return r.id === 1 && Boolean(r.result?.protocolVersion); } + catch { return false; } +}); +const input = `${JSON.stringify({ jsonrpc: '2.0', id: 1, method: 'initialize', params: { + protocolVersion: '2024-11-05', capabilities: {}, clientInfo: { name: 'ak-native-diagnostic', version: '1' }, +} })}\n`; + +test('diagnose installed Ruflo Windows public shim versus installed bin transport', { + skip: process.env.AK_RUFLO_WINDOWS_DIAGNOSTIC !== '1', timeout: 180000, +}, async () => { + assert.equal(process.platform, 'win32', 'diagnostic requires native Windows'); + const packageRoot = process.env.AK_RUFLO_PACKAGE_ROOT; + assert.ok(packageRoot && path.isAbsolute(packageRoot), 'explicit installed Ruflo package root required'); + const packageFile = path.join(packageRoot, 'package.json'); + const pkg = JSON.parse(fs.readFileSync(packageFile, 'utf8')); + assert.equal(pkg.name, 'ruflo'); + const bin = path.resolve(packageRoot, typeof pkg.bin === 'string' ? pkg.bin : pkg.bin.ruflo); + assert.ok(bin.startsWith(`${fs.realpathSync(packageRoot)}${path.sep}`), 'bin belongs to installed package'); + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-ruflo-win-diagnostic-')); + const scope = createDiagnosticScope(); + let clean = false; + try { + const bootstrap = {}; + for (const key of ['PATH', 'SystemRoot', 'WINDIR', 'ComSpec', 'PATHEXT']) { + const value = envValue(process.env, key); + if (value) bootstrap[key] = value; + } + const env = spawnEnv(root, { + CLAUDE_FLOW_DB_PATH: path.join(root, '.swarm', 'memory.db'), + CLAUDE_FLOW_MEMORY_PATH: path.join(root, '.swarm'), RUFLO_DAEMON_AUTOSTART: '0', + }, { env: bootstrap }); + fs.writeFileSync(path.join(root, 'claude-flow.config.json'), JSON.stringify({ daemon: { autostart: false } })); + const invocation = resolveShim('ruflo', ['mcp', 'start'], { env }); + assert.equal(invocation.resolved, true, 'installed public shim must resolve'); + const powershell = resolveShim('ruflo', ['mcp', 'start'], { env, npmBin: false }); + const scriptIndex = powershell.args.indexOf('-File'); + assert.ok(scriptIndex >= 0, 'receipt requires the native PowerShell transport'); + const shim = powershell.args[scriptIndex + 1]; + const cmdShim = shim.slice(0, -4) + '.cmd'; + const version = await scope.launch(resolveShim('ruflo', ['--version'], { env }), + { cwd: root, env, timeoutMs: 10000 }); + console.log(JSON.stringify({ diagnostic: 'inputs', node: process.version, uv: process.versions.uv, + platform: process.platform, packageVersion: pkg.version, version, + packageSha: sha(packageFile), bin, binSha: sha(bin), invocation, powershell, + cmdSha: sha(cmdShim), cmdSource: fs.readFileSync(cmdShim, 'utf8').slice(0, 8192), + shimSha: sha(shim), shimSource: fs.readFileSync(shim, 'utf8').slice(0, 8192), + sourceSha: sha(new URL(import.meta.url)), + launcherSha: sha(new URL('./ruflo-windows-diagnostic-process.mjs', import.meta.url)) })); + assert.equal(version.cleanupComplete, true, 'version launch cleanup incomplete'); + // Capture the original fixture's native argument forwarding without + // creating any descendants: this isolates quoting from tree ownership. + const legacyShim = path.join(root, 'legacy-fixture.ps1'); + const marker = path.join(root, 'legacy-marker'); + fs.writeFileSync(legacyShim, `& '${process.execPath.replaceAll("'", "''")}' -e $args[0]\nexit $LASTEXITCODE\n`); + const legacyCode = `const {spawn}=require('node:child_process'); + require('node:fs').writeFileSync(${JSON.stringify(marker)},JSON.stringify([process.pid]));`; + const legacy = await scope.launch({ command: powershell.command, args: [ + ...powershell.args.slice(0, scriptIndex + 1), legacyShim, legacyCode, + ] }, { cwd: root, env, timeoutMs: 5000 }); + console.log(JSON.stringify({ diagnostic: 'legacy-multiline-node-e', + marker: fs.existsSync(marker), ...legacy })); + assert.equal(legacy.cleanupComplete, true, 'legacy launch cleanup incomplete'); + for (const [transport, spec, eof] of [ + ['public-open-stdin', invocation, false], + ['public-eof', invocation, true], + ['powershell-open-stdin', powershell, false], + ['powershell-eof', powershell, true], + ['installed-bin-open-stdin', { command: process.execPath, args: [bin, 'mcp', 'start'] }, false], + ]) { + const observation = await scope.launch(spec, { cwd: root, env, input, endInput: eof, + timeoutMs: 15000, until: initialized }); + console.log(JSON.stringify({ diagnostic: transport, initialized: observation.matched, endInput: eof, ...observation })); + assert.equal(observation.cleanupComplete, true, `${transport} cleanup incomplete`); + } + clean = scope.release(); + assert.equal(clean, true, 'every diagnostic launch must prove cleanup before root removal'); + } finally { + if (clean) fs.rmSync(root, { recursive: true, force: true }); + else console.error(`diagnostic root retained: ${root}`); + } +}); From 069553586fb3ad1d824877259562eef0765a42a1 Mon Sep 17 00:00:00 2001 From: Chris Phillipson Date: Tue, 29 Sep 2026 11:50:46 -0700 Subject: [PATCH 08/10] fix(usage): complete session attribution and accounting remediation (#282) * docs(usage): plan usage accuracy capture units * feat(usage): add shared session surface vocabulary * fix(usage): preserve fixed initiators and bound raw evidence * docs(usage): map capture units to source and tests * fix(usage): classify local managed Claude statusline settings * fix(usage): keep ambiguous managed statusline values unknown * fix(usage): classify statusline shell wrappers as custom * fix(usage): reject option-shaped statusline targets * feat(usage): persist session surface classification in cache * fix(usage): retain bounded unfamiliar origin metadata * fix(usage): classify Codex child threads and unpriced reviews * fix(usage): preserve rejected source and imported origin * fix(footprint): count Claude sessions by declared identity * fix(footprint): keep recovered project evidence out of session counts * fix(runtime): distinguish desktop apps from hosted CLI sessions * fix(runtime): recognize Codex service after global options * fix(runtime): preserve quoted Codex config boundaries * fix(system): preserve project census count basis in management API * fix(system): label desktop applications in runtime views * fix(usage): bind Claude provider detail to session model evidence * fix(usage): tighten Claude provider evidence validation * fix(usage): exclude imported Codex turns individually * fix(usage): reject incomplete mixed-turn ownership * fix(usage): count Codex component usage without responses * fix(usage): attribute OpenCode totals by response provider * fix(usage): retain provider bucket session metrics * feat(usage): capture Codex effort timing and compaction * feat(usage): preserve session surface presentation evidence across project DTOs * feat(dashboard): render session surfaces and import census disclosures * fix(usage): bound Codex compactions and gate unprovable replay * fix(dashboard): preserve legacy surface filters and refresh selections * feat(usage): reconcile Claude cost-state checkpoints * fix(usage): qualify Claude cost-state comparison scope * test(ui): include session surfaces in default suite * fix(usage): mark OpenCode reported zero as unpriced when untrusted * fix(usage): deduplicate copied Claude messages across files * fix(usage): scope Claude dedup to display windows * fix(usage): elect one Claude message owner across windows * fix(usage): bind Claude ownership to source eligibility * fix(usage): gate OpenCode cache reuse on parse semantics * fix(usage): invalidate local buckets when timezone changes * fix(usage): count unknown Claude transcript records * fix(usage): classify valid Claude JSON record shapes * fix(usage): require unambiguous OpenCode database selection * feat(usage): detect unsupported OpenCode storage presence * fix(usage): reject incomplete OpenCode source discovery * fix(usage): disclose unsupported OpenCode storage coverage * fix(usage): capture OpenCode compactions and reconcile counters * fix(usage): refresh OpenCode evidence and retain uncertain bounds * fix(usage): exclude OpenCode children from prompt fingerprints * docs(usage): accept bounded session and accounting contracts * docs(adr): record accepted session surface contracts * docs(usage): correct window provider and delegation metrics * fix(usage): preserve legacy filters and integration contracts * test(dashboard): align served provider helper assertion * docs(usage): archive completed V6 plans and evidence * test(maintenance): use a native census project fixture * test(usage): model unavailable runtime timezone portably * test(hooks): isolate CLI probes and assert no drift launches * test(host): frame lifecycle logs as test diagnostics --- ...ge-scorecard-local-transcript-analytics.md | 7 + docs/adr/0052-codex-usage-attribution.md | 104 +++-- ...ion-surface-initiator-and-product-names.md | 159 ++++--- docs/adr/README.md | 14 +- .../archive/2026-09-28-plan-usage-accuracy.md | 200 ++++++++ .../2026-09-29-plan-v6-opencode-cost.md | 35 ++ .../archive/2026-09-29-plan-v6-surfaces-ui.md | 20 + ...6-09-29-v6-codex-thread-source-evidence.md | 31 ++ docs/archive/README.md | 4 + docs/codex-usage-diagnostic.md | 10 + docs/dashboard.md | 27 +- docs/ddd/machine-footprint.md | 25 +- docs/ddd/ubiquitous-language.md | 17 +- ...-09-28-remediation-v2-develop-execution.md | 5 + docs/transcripts.md | 118 ++--- docs/usage-scorecard-metrics.md | 426 +++++++++++------- scripts/run-tests.mjs | 3 +- src/commands/system.mjs | 17 +- src/lib/census-presentation.mjs | 8 + src/lib/codex-import-marker.mjs | 71 ++- src/lib/codex-rollout-reader.mjs | 19 +- src/lib/codex-usage-walk.mjs | 29 +- src/lib/dashboard-server.mjs | 7 +- src/lib/dashboard/client.mjs | 6 +- src/lib/dashboard/client/intelligence.mjs | 25 +- .../dashboard/client/maintenance-filters.mjs | 9 +- .../dashboard/client/maintenance-focus.mjs | 3 +- .../dashboard/client/session-presentation.mjs | 38 ++ src/lib/dashboard/client/system-projects.mjs | 15 +- src/lib/dashboard/client/usage.mjs | 33 +- src/lib/dashboard/intel-history.mjs | 3 +- src/lib/dashboard/maintenance-api.mjs | 11 +- src/lib/dashboard/page.mjs | 3 +- src/lib/dashboard/system-summary.mjs | 19 +- src/lib/footprint/codex-import-discovery.mjs | 80 ++++ src/lib/footprint/project-sources.mjs | 145 ++++-- src/lib/footprint/projects.mjs | 13 +- src/lib/footprint/runtime.mjs | 4 +- src/lib/footprint/session-origin.mjs | 36 +- src/lib/footprint/session-surfaces.mjs | 46 ++ src/lib/live/process-sessions.mjs | 132 +++++- .../management/focus-navigation.mjs | 2 +- .../management/projection-projects.mjs | 7 +- src/lib/maintenance/management/query.mjs | 15 +- src/lib/pricing.mjs | 3 + src/lib/project-census.mjs | 2 + src/lib/quota.mjs | 102 +++-- src/lib/session-surface.mjs | 186 ++++++++ src/lib/usage-aggregate.mjs | 195 +++++++- src/lib/usage-claude-dedup.mjs | 142 ++++++ src/lib/usage-cost.mjs | 136 +++++- src/lib/usage-index.mjs | 304 ++++++++++--- src/lib/usage-opencode-bounds.mjs | 4 +- src/lib/usage-opencode-cache.mjs | 52 +++ src/lib/usage-opencode-health.mjs | 18 + src/lib/usage-opencode-observations.mjs | 142 ++++++ src/lib/usage-opencode-source.mjs | 77 ++++ src/lib/usage-opencode-storage-coverage.mjs | 87 ++++ src/lib/usage-opencode.mjs | 78 ++-- src/lib/usage-parsers.mjs | 307 +++++++++++-- src/lib/usage-project-evidence.mjs | 23 +- tests/dashboard.test.cjs | 2 +- tests/kit/dashboard-project-identity.test.mjs | 5 +- tests/kit/footprint-projects.test.mjs | 4 +- tests/kit/footprint-windows.test.mjs | 28 ++ tests/kit/hook-audit.test.mjs | 78 ++-- tests/kit/host-pick-rerecord.test.mjs | 7 +- tests/kit/intel-history.test.mjs | 13 +- tests/kit/intelligence-picker-groups.test.mjs | 6 +- tests/kit/intelligence-table-groups.test.mjs | 10 +- tests/kit/live-process-sessions.test.mjs | 118 ++++- .../kit/maintenance-dashboard-v2-api.test.mjs | 2 +- tests/kit/maintenance-focus-client.test.mjs | 4 +- .../kit/maintenance-project-grouping.test.mjs | 105 +++++ .../project-sources-claude-sessions.test.mjs | 107 +++++ tests/kit/quota-codex-presence.test.mjs | 3 +- tests/kit/quota.test.mjs | 222 ++++++++- tests/kit/run-tests-runner.test.mjs | 4 +- tests/kit/session-presentation.test.mjs | 72 +++ tests/kit/session-surface-renderers.test.mjs | 104 +++++ tests/kit/session-surface.test.mjs | 139 ++++++ tests/kit/system-command.test.mjs | 8 + tests/kit/system-runtime-app-labels.test.mjs | 36 ++ tests/kit/system-summary.test.mjs | 33 +- tests/kit/usage-claude-cost-state.test.mjs | 174 +++++++ tests/kit/usage-claude-dedup.test.mjs | 354 +++++++++++++++ tests/kit/usage-claude-provider.test.mjs | 96 ++++ .../kit/usage-claude-record-coverage.test.mjs | 133 ++++++ ...ge-codex-effort-timing-compaction.test.mjs | 162 +++++++ tests/kit/usage-codex-import-gaps.test.mjs | 115 +++++ tests/kit/usage-codex-import-turns.test.mjs | 261 +++++++++++ tests/kit/usage-codex-thread-source.test.mjs | 150 ++++++ tests/kit/usage-codex-zero-response.test.mjs | 98 ++++ .../kit/usage-index-opencode-source.test.mjs | 265 +++++++++++ tests/kit/usage-index-opencode.test.mjs | 247 ++++++++++ tests/kit/usage-index-v6.test.mjs | 27 +- tests/kit/usage-index.test.mjs | 7 +- tests/kit/usage-limits-empty-state.test.mjs | 8 +- tests/kit/usage-opencode-compaction.test.mjs | 236 ++++++++++ tests/kit/usage-opencode-selection.test.mjs | 125 +++++ .../usage-opencode-storage-coverage.test.mjs | 103 +++++ tests/kit/usage-opencode.test.mjs | 146 +++++- tests/kit/usage-session-surface.test.mjs | 142 ++++++ tests/kit/usage-timezone-cache.test.mjs | 191 ++++++++ tests/ui/intelligence-picker.mjs | 8 +- tests/ui/maintenance-focus.mjs | 7 +- tests/ui/maintenance-host-alignment.mjs | 5 +- tests/ui/maintenance-projects.mjs | 12 +- tests/ui/session-surfaces.mjs | 64 +++ 109 files changed, 7323 insertions(+), 782 deletions(-) create mode 100644 docs/archive/2026-09-28-plan-usage-accuracy.md create mode 100644 docs/archive/2026-09-29-plan-v6-opencode-cost.md create mode 100644 docs/archive/2026-09-29-plan-v6-surfaces-ui.md create mode 100644 docs/archive/2026-09-29-v6-codex-thread-source-evidence.md create mode 100644 src/lib/census-presentation.mjs create mode 100644 src/lib/dashboard/client/session-presentation.mjs create mode 100644 src/lib/footprint/codex-import-discovery.mjs create mode 100644 src/lib/footprint/session-surfaces.mjs create mode 100644 src/lib/session-surface.mjs create mode 100644 src/lib/usage-claude-dedup.mjs create mode 100644 src/lib/usage-opencode-cache.mjs create mode 100644 src/lib/usage-opencode-health.mjs create mode 100644 src/lib/usage-opencode-observations.mjs create mode 100644 src/lib/usage-opencode-source.mjs create mode 100644 src/lib/usage-opencode-storage-coverage.mjs create mode 100644 tests/kit/project-sources-claude-sessions.test.mjs create mode 100644 tests/kit/session-presentation.test.mjs create mode 100644 tests/kit/session-surface-renderers.test.mjs create mode 100644 tests/kit/session-surface.test.mjs create mode 100644 tests/kit/system-runtime-app-labels.test.mjs create mode 100644 tests/kit/usage-claude-cost-state.test.mjs create mode 100644 tests/kit/usage-claude-provider.test.mjs create mode 100644 tests/kit/usage-claude-record-coverage.test.mjs create mode 100644 tests/kit/usage-codex-effort-timing-compaction.test.mjs create mode 100644 tests/kit/usage-codex-import-gaps.test.mjs create mode 100644 tests/kit/usage-codex-import-turns.test.mjs create mode 100644 tests/kit/usage-codex-thread-source.test.mjs create mode 100644 tests/kit/usage-codex-zero-response.test.mjs create mode 100644 tests/kit/usage-index-opencode-source.test.mjs create mode 100644 tests/kit/usage-opencode-compaction.test.mjs create mode 100644 tests/kit/usage-opencode-selection.test.mjs create mode 100644 tests/kit/usage-opencode-storage-coverage.test.mjs create mode 100644 tests/kit/usage-session-surface.test.mjs create mode 100644 tests/kit/usage-timezone-cache.test.mjs create mode 100644 tests/ui/session-surfaces.mjs diff --git a/docs/adr/0009-usage-scorecard-local-transcript-analytics.md b/docs/adr/0009-usage-scorecard-local-transcript-analytics.md index f55a270d..94eb15f8 100644 --- a/docs/adr/0009-usage-scorecard-local-transcript-analytics.md +++ b/docs/adr/0009-usage-scorecard-local-transcript-analytics.md @@ -2,6 +2,13 @@ - **Status:** Implemented - **Date:** 2026-07-25 +- **Updated:** 2026-09-29 — usage schema 25 → 26 delivers bounded session surface/provider + evidence, per-turn Codex import ownership, positive component usage without responses, + one Claude message charge owner across the bounded two-window pool, separate host-reported + reconciliation signals, explicit OpenCode source/cost/coverage semantics and timezone-aware + cache reuse. Footprint remains schema 8. The + [current accounting contracts](../usage-scorecard-metrics.md#current-accounting-and-cache-contracts) + define source bounds and compatibility; historical measurements below are not fresh results. - **Updated:** 2026-09-20 — ADR-0054 adds an explicit, offline, allowlisted fleet export boundary; local analytics and dashboard collection semantics remain unchanged. - **Earlier update:** 2026-09-09 — reconciled against repository source and tests for issue #211 diff --git a/docs/adr/0052-codex-usage-attribution.md b/docs/adr/0052-codex-usage-attribution.md index 2e5a6ec2..55343df8 100644 --- a/docs/adr/0052-codex-usage-attribution.md +++ b/docs/adr/0052-codex-usage-attribution.md @@ -12,6 +12,18 @@ record with at least one response (`usage-aggregate.mjs`'s `buildSessionRows`), so the file's 176,326 tokens never reach any total regardless of the explanation. The advisory is right in substance; classification is unchanged. Extends the "Not done" bullet below with the measured counts. +- **Updated:** 2026-09-29 — Unit 8 re-read the one gap candidate selected from a read-only schema-25 + cache under a 2 MiB source bound. Its current source has five token snapshots with full + input/cache/output components, no normalized assistant response, tool items and an abort. + The parser already retains its component row; aggregation now admits a Codex record with + positive component usage even when responses are zero. A positive total-only counter + remains unsupported: it supplies no input/cache/output split or price. Source health + discloses such events as `total-only-token-count` and reports zero-response records with + counted components separately from those without attributable component rows. +- **Updated:** 2026-09-29 — accepted V6 also classifies guardian reviews and other thread sources, + rolls up verified acyclic parent links, retains effort and host-reported first-token timing, + and exposes compaction bounds. Auto-review models without supported prices are unpriced. + The delivered cache migration is 25 → 26; earlier v23 evidence below describes the original fix. - **Deciders:** agentic-kit maintainers - **Related:** [ADR-0009](0009-usage-scorecard-local-transcript-analytics.md), [ADR-0038](0038-consistent-cross-host-session-metrics.md), @@ -121,16 +133,50 @@ lists them). Their turns are stamped `external-import-turn-N`, they have no "responses" and a large share of prompts, priced at model `unknown`, $0, while the real data lives in the Claude transcript. -The in-rollout marker (the `turn_id` prefix) is the signal, so detection works -without the imports file. Parsing stops at the first such line; the record is kept -out of aggregation, out of every yield statistic, and counted in -`diagnostics.importedExcluded` (796 on the reference machine). Nothing is dropped -silently. The record itself is still cached, so a rescan is cheap. - -Project discovery applies the same marker to each rollout's bounded head (256 KiB, 40 lines): an -imported copy names no project, host or Desktop origin, and the scan reports how many it set aside -(`importedExcluded`, 924 on the reference machine on 2026-09-27, every marker on the rollout's -second line). +The in-rollout marker (`payload.turn_id` prefix) is the signal; the import map is not a +runtime dependency. Exclusion is per turn. A valid native `task_started` or identified +`turn_context` opens native ownership; explicit record IDs must agree with that boundary. +Missing or conflicting IDs in mixed files remain unattributable. An absent context ID may +enrich an identified turn, but an explicitly invalid boundary ID breaks adjacency and marks +ownership incomplete. A foreign completion does +not close the active turn. Replayed parent history still cannot count as child activity. +Marker text in messages and later `session_meta` declarations establish no ownership. + +Copied prompts, responses, tools, context and tokens are excluded. Excluded cumulative +snapshots advance the baseline; native snapshots book only their deltas. Decreasing counters +or an explicit first native `last_token_usage == total_token_usage` reset start a new segment +(the latter excludes identical re-emissions). First session identity remains authoritative. +A native turn's cwd can establish its genuine project; otherwise the first declared cwd is +eligible only after own activity is proved. The first declared app surface applies, without +assuming every mixed file came from Desktop. + +Files without proven own activity remain `imported: true` with unknown/imported-copy origin +and are cached but excluded. Mixed files retain `importEvidence`: `importedTurns`, +`importedRecords`, `ambiguousRecords` and `nativeRecords`, with no turn IDs or copied content. +The unique imported-turn set retains at most 4,096 bounded IDs; `importedTurnCountComplete` +marks a lower-bound count when capped. Mixed sources with clipped lines, skipped nonblank +records or explicitly invalid turn-boundary IDs are conservatively excluded in their entirety +with `ownershipComplete: false`, pending a complete readable source. String and streaming +readers expose the last pass's skipped-record count as `importEvidence.malformedRecords`; +clipping retains its existing diagnostic. This may omit proven activity before a gap but +cannot carry native ownership across unreadable copied-turn metadata. A subagent +whose replay cannot be separated also has incomplete ownership. Source-health counters +`importOwnershipIncompleteFiles` and `importedTurnCountIncompleteFiles` retain these gaps +even when the file is excluded or served from cache. +Usage diagnostics expose `importedExcluded`, `importedMixed`, `importedTurnsExcluded` and +`importAmbiguousRecords`; the ambiguous count discloses excluded records without proved +ownership. Aggregate rows and session detail preserve the same evidence. Schema 26 remains +the single unreleased migration; no personal cache is rebuilt during implementation. + +Discovery first reads its usual 256 KiB/40-line head. Import-marked heads additionally read +at most a 256 KiB head and a 2 MiB tail, capped at 20,000 records per window and 512 MiB of +additional reads per scan. An unread middle resets ownership. Positive native activity can +establish a mixed sighting; imported-only exclusion requires a complete, unambiguous read. +Malformed envelopes or explicitly invalid turn-boundary IDs also leave an import candidate +unresolved. A sampled subagent without a complete replay boundary remains unresolved. Sources and the +discovery summary expose `importedMixed` and `importedUnresolved`; unresolved imports make +`complete` and `sessionCountComplete` false and cannot trigger encoded-directory recovery. +These are bounded observations, not an exhaustive turn census. ### 4. Cumulative counter restarts are summed, per event @@ -171,7 +217,7 @@ tools and `FunctionCallOutput` the known set. Only a type in none of them warns. ## Consequences -- Cache schema **v23**: every cached Codex record and its `parseStats` re-derive. +- Original implementation cache schema **v23**: every cached Codex record and its `parseStats` re-derive. Earlier records carry the wrong imports, subagent usage, replay counts, last-wins totals, single-day/model rows and permanent diagnostics. - Codex subagent sessions now have real tokens and cost. Cost totals, the @@ -204,28 +250,26 @@ subagent and previously dropped usage is now priced. ## Not done (recorded follow-ups) -- Guardian-review classification, unread host fields, and the context-coverage - denominator remain as audited; this ADR does not change them. +- Guardian-review classification, effort, first-token timing and compaction evidence are now + captured. Compaction lower/nullable upper bounds preserve uncertain pairing. This does not + establish support for every unread host field or change the context-coverage denominator. - A stream tee or push channel for live oversized rollouts is out of scope. - The cause of counter restarts is unknown (decision 4). - A subagent with no ordinals still reports no usage (decision 2). -- One rollout carries `token_count`s but no agent message, so the pre-existing - `partial-response-yield` warning remains — measured on the reference machine (2026-09-28): of 1,714 - Codex rollouts (734 token-bearing), exactly 1 is such a gap file. It is explained by a cached fact - (`session.aborts > 0`, tool-only activity) but that does not change what is counted: the aggregate - never builds a session row for a record with zero responses (`usage-aggregate.mjs`'s - `buildSessionRows`), so the file's usage (176,326 tokens, its own `last_token_usage.total_tokens` - sum across its `token_count` events) reaches no total either way. Counting that usage, or - documenting the shape more precisely, is left to the usage-accuracy branch. -- Whole-rollout exclusion may drop real usage (open, plausible, 2026-09-27). On the reference - machine 6 of 924 imported rollouts carry a later turn that is not an import: one `task_started` - whose `turn_id` starts with `rollout-`, no `user_message` event, `role: user` response items in - five of the six (2 to 76 per file) and non-zero `token_count` usage (the per-file sum of - `last_token_usage.total_tokens` is about 8k to 449k). Both usage and discovery set the whole file - aside at the marker, so this usage is not counted. With no `user_message`, the turn may be - automatic (a compaction or title pass). Measured from counts only. Decided 2026-09-27 (audit - decision 12): Branch 8 excludes per turn instead of per file, so imported turns are never counted - and later turns are, after it establishes whether they are the user's work or an automatic pass. +- The 2026-09-28 full-corpus count (1,714 rollouts, 734 token-bearing, one + zero-response gap) is historical. Unit 8's bounded 2026-09-29 re-read of that gap + candidate found 11,082 uncached input, 163,456 cached input and 1,788 output + tokens in a native, zero-response record. Those components now reach aggregate + totals and cost estimation. Total-only counters still cannot yield a split or + price and are diagnosed rather than silently treated as free usage. +- The historical 2026-09-27 observation (6 of 924 import-marked files) did not prove + that every token snapshot in those files belonged to a native turn. Unit 7 implements + decision 12 per turn. The 2026-09-29 metadata-only reproduction found 6 mixed files among + 945 import-marked candidates, with 64 native-turn responses. Only one token snapshot fell + inside a native interval, and its input/cache/output components were all zero; the other + snapshots were copied-turn evidence. No billable components are inferred from total-only + counters. The native turn's initiator remains unknown unless separately declared; this + does not prove whether it was user work or an automatic pass. ## Verification diff --git a/docs/adr/0060-session-surface-initiator-and-product-names.md b/docs/adr/0060-session-surface-initiator-and-product-names.md index 098892ff..63e29e2f 100644 --- a/docs/adr/0060-session-surface-initiator-and-product-names.md +++ b/docs/adr/0060-session-surface-initiator-and-product-names.md @@ -1,13 +1,13 @@ # ADR-0060 — Session surface, initiator and official product names -- **Status:** Proposed; §3 implemented for project discovery (2026-09-27), the rest staged follow-on +- **Status:** Accepted - **Date:** 2026-09-26 -- **Updated:** 2026-09-27 — §3 implemented for project discovery and the System projects note: - imported copies give no project, host or origin and are counted. The ledger-derived source labels - (Cursor, Cowork) and the other views remain proposed. +- **Updated:** 2026-09-29 — delivered shared classification, usage/cache and census evidence, + Runtime application attribution, and CLI/dashboard presentation. Dedicated Cowork storage remains + an optional follow-up (#257); acceptance covers the bounded sources described below. - **Deciders:** agentic-kit maintainers - **Related:** [ADR-0050](0050-dashboard-project-identity-and-context-reporting.md) (session origin - rule, superseded in part by this record once accepted), + rule, superseded in part by this record), [ADR-0052](0052-codex-usage-attribution.md) (imported Codex rollouts excluded from usage), [ADR-0025](0025-machine-footprint-metrics.md) (Runtime census labels), [ADR-0027](0027-shared-project-census.md) (project census), @@ -25,7 +25,7 @@ the product's official commercial name, with no duplicate or false categories. Research on 2026-09-26 read only enumerated log fields and counts (no prompt or response content), the vendors' current documentation, the openai/codex source at `7f6c0f9`, and the installed Claude -Code 2.1.283 and Claude Desktop 2.9939.2 builds. What it established: +Code 2.1.283 and Claude Desktop 2.9939.2 builds. What that historical sample established (not a fresh census or current behavior): 1. **The logs declare where a session came from.** Claude Code transcripts carry `entrypoint`; Codex rollouts carry `session_meta.originator`, `source` and `thread_source`. Folder location is @@ -66,9 +66,9 @@ Code 2.1.283 and Claude Desktop 2.9939.2 builds. What it established: copied in five files. The Runtime census counts `Claude.app` as the Claude Code host (basename match) while `ChatGPT.app` is invisible. -## Decision (proposed) +## Decision -### 1. Two dimensions from declared fields, raw value always kept +### 1. Separate dimensions from declared fields, bounded raw evidence Every session record carries: @@ -77,18 +77,23 @@ Every session record carries: - **Initiator** — `person`, `automation`, `agent` (a subagent or reviewer spawned by another session) or `imported-copy`. Claude: person when interactive or the entrypoint is one Claude Code itself treats as attended (`claude-vscode`, `claude-desktop*`, `local-agent`, `remote*`, - `ssh-remote`, Claude Tag values); automation for `sdk-*`, `mcp`, `claude-code-github-action`, and + `ssh-remote`, Claude Tag values); automation for `sdk-py`, `sdk-ts`, `sdk-cli`, `mcp`, `claude-code-github-action`, and for `sessionKind` `bg`/`daemon`/`daemon-worker`. Codex: `thread_source` `user` and `chatgpt_handoff` are person, except that `codex exec` top-level threads are automation; - `subagent` and `guardian_review` are agent; app feature values such as `automation` are + `subagent`, `guardian_review` and `agent_created_thread` are agent; app feature values such as `automation` are automation. -- **Raw evidence** — the exact declared values (`entrypoint:cli`, - `originator:codex_exec/source:exec`). An unrecognized value is shown as "Other" with its raw value, - never merged into a larger bucket. +- **Raw evidence** — bounded tokens from the named declaration fields, such as `entrypoint:cli` and + `originator:codex_exec/source:exec`. Unfamiliar valid tokens remain available in local detail, with Other or Unknown + classification and no inferred product or provider. Tokens must be at most 80 characters, + begin with a letter and contain only letters, digits, underscores, dots or hyphens; the known + `Codex Desktop` value is the sole space-containing exception. Malformed values are omitted. Folder class (ChatGPT Projects folder, projectless Work folder, Cowork data, temporary folder) is an -explanatory attribute, never the classifier. The first record that declares a value wins, and -the rule is stated in code and tests. +explanatory attribute, never the classifier. The first eligible declaring record wins within each reader's bounded evidence window; +invalid values remain Unknown and do not authorize a search for a preferred later identity. +A later replayed parent declaration cannot replace the child's identity. Git scope, host, +surface, initiator and provider are separate fields and filters. Cloud choices appear only +when a covered record declares a cloud surface. ### 2. Official names @@ -101,8 +106,8 @@ Claude (per code.claude.com and claude.com documentation): | `claude-desktop`, `claude-desktop-3p` | Claude Desktop (attribute "on 3P" for the second) | person | | `local-agent`, `local_agent`, `remote_cowork` | Cowork | person | | `remote`, `remote_desktop`, `remote_mobile`, `remote_projects` | Cloud session (attribute: started from Desktop, mobile, web or a project) | person | -| `remote_trigger`, `remote_cowork_trigger` | Cloud session (routine) | automation | -| `sdk-py`, `sdk-ts` | Claude Agent SDK (attribute: Python or TypeScript; "plugin hook" when evidenced) | automation | +| `remote_trigger`, `remote_cowork_trigger` | Cloud session | automation | +| `sdk-py`, `sdk-ts` | Claude Agent SDK | automation | | `sdk-cli` | Non-interactive mode (`claude -p`) | automation | | `claude-code-github-action` | GitHub Actions | automation | | `claude_in_slack`, `claude-in-slack`, `claude-in-teams` | Claude Tag (Slack or Teams) | person | @@ -126,7 +131,8 @@ OpenAI (per learn.chatgpt.com, which developers.openai.com/codex now redirects t | `codex_work_web`, `codex_work_mobile`, `codex_work_cca`, `chatgpt_cca` | ChatGPT Work (cloud) | by `thread_source` | | any other | Other OpenAI client (raw value shown) | by `thread_source` | -Subagents and Auto-review roll up under their parent surface ("Subagents", "Auto-review", OpenAI's +Codex subagents and Auto-review with a verified, acyclic parent link in the observed record set +roll up under their parent surface ("Subagents", "Auto-review", OpenAI's own labels) and are never separate products. `source="vscode"` never produces a "VS Code" label. Hosts are named **Claude Code**, **Codex**, **OpenCode** (by Anomaly) and **Hermes Agent** (Nous @@ -135,10 +141,20 @@ Desktop** and **ChatGPT desktop app**; they are applications, not hosts. ### 3. Imported copies are excluded everywhere, and counted -ADR-0052's rule extends to project discovery and every origin view: a rollout stamped -`external-import-turn-*` (or listed in the imports ledger when present) contributes no project -sighting, origin, facet count or Runtime attribution. Each view reports how many it excluded, labelled -"Imported from Claude Code" (or Cursor, or Cowork, from the ledger's source path). +ADR-0052's rule extends to project discovery and origin views **per turn**. The portable +signal is `payload.turn_id` beginning `external-import-turn`; an import map is not required. +Copied turns contribute no usage or project/origin sighting. A later proven native turn can +establish the first declared app surface and a genuine project; unknown or conflicting turn +boundaries remain excluded with diagnostics. A Desktop declaration is not inferred for +other declared products. First session identity and parent replay exclusion remain intact. + +Pure imports remain unknown/imported-copy. Mixed usage rows retain import-exclusion counts. +Discovery distinguishes confirmed exclusions (`importedExcluded`), proven mixed observations +(`importedMixed`) and bounded observations that cannot settle ownership (`importedUnresolved`). +The latter make coverage incomplete, without inventing a project from an encoded directory. +See ADR-0052 §3 for exact byte/record budgets and cumulative-counter rules. Intelligence, +System Projects and `ak system` disclose these populations and source incompleteness. +Optional ledger source labels remain follow-on work. ### 4. One vocabulary module and one label table @@ -151,56 +167,59 @@ Labels are tested once; views test that they use the shared table. - The Runtime census treats `Claude.app` and `ChatGPT.app` symmetrically as desktop applications, neither as a host; their bundled CLIs are attributed to the hosted session (lane D's Claude rule extended to the Codex bundle). -- Cowork transcripts become an optional discovery source; until then views say Cowork is not - covered. +- Dedicated Cowork storage remains uncovered (#257), and views disclose that limit. Covered + Claude transcript records may still declare the Cowork surface; that does not prove coverage + of the separate store. -### 6. Counting rules +### 6. Counting rules and compatibility -Claude sessions are counted by `sessionId`, excluding `subagents/` transcripts and non-conversation -records; Codex subagent and reviewer rollouts roll up to their parent; `thread_source` is classified -in full. +Claude project census sessions use declared `sessionId`, excluding subagent and bridge-only +transcripts. Rows expose `countBasis`: declared-session IDs, transcript files, database sessions, +recovered-project sightings or mixed observations. Encoded-directory recovery can establish a +project sighting with zero session weight; missing identity and bounded reads keep completeness +visible. Census session observations are distinct from billed Usage sessions. + +Project `sessionSurfaces` is additive: legacy `sessionOrigins` remains for compatibility. Raw +project detail unions retain at most 16 sorted values per named field per classification group; +`rawEvidenceComplete: false` discloses truncation. The raw-token policy was explicitly approved +by the maintainer for local detail; it does not authorize publishing private tokens. + +Old coarse `codex-desktop` snapshots render Unknown surface with a ChatGPT desktop app family +note because their mode was not recorded. Old `claude-desktop` snapshots retain Claude Desktop; +initiator and provider remain Unknown. Saved legacy origin filters preserve their membership +and use explicit legacy labels. Missing newer fields are not evidence of a precise mode. + +### 7. Provider evidence is independent + +Claude provider-specific assistant model IDs may establish Amazon Bedrock or Google Vertex AI +metadata (`assistant-model-id`); ordinary or conflicting IDs leave Unknown. Codex and OpenCode +may provide a recorded provider ID. Both are observed source metadata, not network attestation. +Current environment, routing configuration, application identity and price-table identity cannot +establish a historical serving provider. The `on 3P` attribute remains visible independently of +provider Unknown. Codex Auto-review tokens remain unpriced when no supported price exists. ## Consequences -- Usage, System → Projects, Maintenance facets, Intelligence designation (which today mixes Git - scope, origin and host in one enum) and the Runtime table change labels and counts. On this - machine 32 project folders lost a false Desktop origin when discovery began setting imports aside - (re-measured 2026-09-27; 23 on 2026-09-26). -- The usage cache schema changes (new session fields); a rebuild is expected. -- Tests that pin current names change together (inventory in the audit record, Addendum 3). -- `CLAUDE_CODE_ENTRYPOINT` and the transcript format are internal to Claude Code and may change; - keeping the raw value and an "Other" fallback bounds that risk. -- Privacy is unchanged: only enumerated values and counts are read. - -## Open questions for acceptance - -- Whether "Cloud session" should appear at all in local views, given none was observed locally. -- Whether the "on 3P" attribute is worth showing. -- How ADR-0057's role lenses consume surface and initiator. -- Whether a later turn inside an imported copy that is not itself an import (6 of 924 rollouts on - 2026-09-27, with real token usage) counts as the importing app's own session. Decided 2026-09-27 - (audit decision 12): it counts, excluded per turn in Branch 8 - ([ADR-0052](0052-codex-usage-attribution.md), "Not done"). - -## Verification (when implemented) - -Fixtures per raw value; an import-ledger join fixture; a census reproduction of the 2026-09-26 counts -from enumerated values; one-label-per-value UI assertions across views; no prompt content in any -fixture. - -## Implementation status - -§3 is implemented for project discovery and the System projects note (2026-09-27): an imported copy -gives no project, host or origin, and discovery counts it in `importedExcluded`. The per-source -labels from the imports ledger, Runtime attribution and §1, §2 and §4–§6 remain follow-on work -(the audit record's Addendum 3). - -Three views already show the smaller project counts but do not yet say how many imported copies were -set aside; §3's "each view reports how many it excluded" is still owed for them: - -- the Intelligence census line (`src/lib/dashboard/client/intelligence.mjs`, which prints - `everSeen`; the server's `readCensus` in `src/lib/dashboard-server.mjs` drops `importedExcluded`); -- the System → Projects liner (`sysProjectsLinerHtml` in - `src/lib/dashboard/client/system-projects.mjs`); -- the `ak system` text output (`renderProjects` in `src/commands/system.mjs`, which prints only the - count; `ak system --json` carries `importedExcluded`). +- Shared vocabulary in `src/lib/session-surface.mjs` supplies Usage, project details, Maintenance, + Intelligence and Runtime labels. Desktop applications have no host identity; bundled CLIs need + observed session attribution, and a Codex app-server process is a service. +- Usage cache schema changes exactly **25 → 26**; old entries require rebuilding. Footprint + snapshot schema remains **8**, with additive evidence and explicit legacy presentation. +- First-declaration, parent-link, import-ownership and source bounds prevent these observations + from establishing whole-corpus coverage. Historical research counts above are not release metrics. +- Internal host fields can change. Unknown, bounded raw evidence and source-health diagnostics + preserve uncertainty without deriving products from directories or `source="vscode"`. + +## Verification and remaining limits + +Synthetic fixtures cover shared vocabulary, first declaring records, parent/reviewer attribution, +legacy filters, raw-token caps, provider evidence, import ownership, session-count bases and +Runtime application/service distinctions. CLI/dashboard consumer assertions cover the shared +labels and disclosures. The implementation is bound to the accepted V6 source units; final +integration gates and publication are separate decisions. + +Dedicated Cowork storage (#257), optional import-ledger source labels, missing parent evidence +and records outside bounded readers remain uncovered. No new live corpus, provider, performance +or billing measurement is claimed by this documentation update. See +[Usage metrics](../usage-scorecard-metrics.md#current-accounting-and-cache-contracts) for the +bounded accounting, source selection and cache contracts delivered alongside this vocabulary. diff --git a/docs/adr/README.md b/docs/adr/README.md index 7b12aab7..8eb84908 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -66,7 +66,7 @@ Consequences**, and cites the grounded source it rests on where relevant. | [0054](0054-fleet-evidence-export.md) | Vendor-neutral fleet evidence export | Implemented | | [0055](0055-aqe-embedding-lifecycle.md) | AQE embedding lifecycle and qualified readiness | Implemented | | [0058](0058-managed-ruflo-components.md) | Managed ruflo components | Accepted (implementation in progress — see Implementation status) | -| [0060](0060-session-surface-initiator-and-product-names.md) | Session surface, initiator and official product names | Proposed; §3 implemented for project discovery (2026-09-27), the rest staged follow-on | +| [0060](0060-session-surface-initiator-and-product-names.md) | Session surface, initiator and official product names | Accepted | | [0061](0061-brain-reclaim-stuck-remediation.md) | RuvNet Brain "unresolved rollback state" remediation | Accepted | | [0062](0062-aqe-project-store-integrity.md) | AQE project store integrity | Accepted | | [0063](0063-evidence-store-and-refresh-vocabulary.md) | One evidence store and the refresh vocabulary | Accepted; CLI and dashboard refresh operation delivered | @@ -399,11 +399,13 @@ component out. ## ADR-0060 — Session surface, initiator and official product names -[ADR-0060](0060-session-surface-initiator-and-product-names.md) (Proposed; §3 implemented for -project discovery) derives a session's surface and initiator from the hosts' declared log fields, -keeps every raw value, uses official product names (Claude Desktop, ChatGPT desktop app, Codex CLI, -and others), and excludes imported session copies from every origin view. Project discovery already -sets imported copies aside and counts them; the other decisions remain proposed. +[ADR-0060](0060-session-surface-initiator-and-product-names.md) (Accepted; updated 2026-09-29) +separates Git scope, host, session surface, initiator and provider using declared source evidence +and shared official labels. Local detail retains approved bounded origin tokens; observed provider +metadata is not network attestation. Per-turn import ownership, count bases and completeness remain +visible, desktop applications are distinct from hosts, and legacy filters preserve their labeled +membership. Dedicated Cowork storage remains uncovered (#257). Usage schema changes 25 → 26; +footprint stays 8. ## ADR-0062 — AQE project store integrity diff --git a/docs/archive/2026-09-28-plan-usage-accuracy.md b/docs/archive/2026-09-28-plan-usage-accuracy.md new file mode 100644 index 00000000..0c0eae8c --- /dev/null +++ b/docs/archive/2026-09-28-plan-usage-accuracy.md @@ -0,0 +1,200 @@ +# Usage accuracy execution plan + +- **Branch:** `fix/usage-accuracy`, based on `develop@e2f9dcae0554ff63921df618a819fd5e6afe80d2` +- **Scope sources:** [v2 V6](../plans/2026-09-28-remediation-program-v2.md), + [v1 Wave 4](../plans/2026-09-26-remediation-program.md), + [ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md). + +## Status + +Implemented on `fix/usage-accuracy`; all 24 units are independently accepted. +The whole-branch review and its scoped correction review passed through `6d5979cc`. +A one-line legacy test correction at `1c02db91` passed controller review, followed +by all eight local gates on that exact clean commit: 6,405 unit tests passed, +seven skipped, legacy suites passed, and browser checks passed (514 legacy +assertions plus 16 native tests). Coverage was 94.19% lines, 83.82% branches and +93.50% functions. Feature PR CI, develop integration and final human main review +remain separate gates at this archival capture. No release, installation or real +store operation is claimed. Unit 11 captures Claude Code's +latest valid cumulative `cost-state` checkpoint as a separate reconciliation +signal. It reports provable time or token scope differences while preserving +message-derived cost totals. The observed checkpoint has no end time or serving +provider attestation, so equal counters remain unverified. Unit 12 now retains +bounded, hashed Claude API message identities per cached file and reconciles +copied charges after discovery. Aggregate responses/tokens/cost count a shared +message once; each session's `responses` still counts what its transcript +recorded, with `accountedResponses` showing its aggregate share. Unit 12 is +accepted, including its bounded global-message-owner policy. The Claude identity pool always covers the +displayed window and its equal-length predecessor, regardless of the +`previous` or `lookbackDays` options. One owner is elected per identity across +that pool before either window is projected; a copied message therefore +contributes to at most one of the two windows. Explicit deeper lookback can +support other history views but cannot change an eligible owner's charge. +Eligibility requires both the file mtime and transcript session end to reach +the fixed horizon; older-mtime copies discovered by a deeper lookback cannot +steal or enlarge it. Distinct historical messages remain visible under that +explicit request, with out-of-pool coverage reported in source health. Claude +reads may reach twice the displayed window (capped at 730 days for the +dashboard's 365-day maximum), with the cap reported for wider callers. +Unit 13 records an entry-level local calendar context: the resolved full timezone +identity plus Node's tzdata and ICU versions. A mismatch or missing/invalid context +reparses available source records; no timestamp is inferred from a cached day. +Process memo and single-flight keys include that context. Degraded OpenCode entries +retain their original marker but cannot contribute incompatible day/punchcard rows; +source health reports `timezoneCacheEntriesExcluded`. Unset `TZ` uses the runtime's +resolved machine zone; an unresolved zone declines cache and aggregate-memo reuse. +Schema remains 26, preserving Unit 15's independent OpenCode cost marker. Unit 13 +is accepted. Unit 14 adds count-only Claude record coverage for known handled, +known ignored, unknown, invalid-type and malformed JSON lines. A v26 cache entry +without those counters reparses; unknown or malformed records degrade source +health without changing message usage or cost. The bounded real-data sample and +focused verification are in the ignored task14 handoff report. Unit 14 and Units 15–19 are independently accepted. + +Pricing retains its existing local `row.day` contract (`usage-parsers.localDay`, +`pricing.costOf`, and the cache-saving probes documented in usage metrics). Cold +and warm reads within a zone must agree, including dated rates. A timezone change +can move a row across a dated rate boundary and therefore change its API-equivalent +estimate; this unit neither establishes a provider billing timezone nor freezes a +price from the old local day. Reported OpenCode cost stays observed, and token, +provider, model, Claude cost-state, Codex fields and duration semantics stay intact. + +Unit 2 is limited to the agreed classifier interface, parser fields and one +usage-cache schema bump. +The maintainer approved retaining unfamiliar, bounded tokens from named origin +fields as raw evidence for local detail. The classifier now retains those tokens +without inferring a product or provider; malformed, oversized and non-string +values remain excluded. Units 22/23 delivered the local detail UI and count/coverage disclosures. Unit 24 records +ADR-0060 acceptance against those implemented contracts, subject to documentation review. + +Unit 6 records Amazon Bedrock or Google Vertex AI only when a Claude assistant +message carries a provider-specific model ID. Conflicting or ordinary IDs leave +the provider unknown. Historical transcripts do not capture launch environment +or settings, so current configuration cannot identify their serving provider; +OpenRouter, other gateways and private endpoints remain unknown without bound +session evidence. The detail stays under `sessionOrigin.thirdPartyProvider` with +`thirdPartyProviderBasis: assistant-model-id` when known. Units 22/23 implement display. + +## Gates and ownership + +All source units have completed their assigned implementation and independent review. Unit 24 +owns the affected ADR bodies and living guides in the assigned worktree. The controller owns +shared indexes/manifests, final integration gates, whole-branch review, plan archival and any +separately authorized publication. Unit 24 runs documentation gates only. No private transcript +content, raw identifiers, paths or observed costs enter public documentation. + +## Capture units + +| Unit | Source boundary and acceptance | Focused evidence / dependency | +|---|---|---| +| 1 | Shared raw-value → surface → initiator → label vocabulary; bounded raw evidence, unknown and provider separate. Adapt footprint origin with legacy fields retained. | ADR table fixtures, first declaration, privacy, identity and imports regressions. Accepted. | +| 2 | Parser and usage cache integration, exactly one schema 25→26 bump. Carry new fields and rebuild old cache. | Parser, cache migration and aggregate tests; after unit 1. Footprint schema stays 8. | +| 3 | Full Codex `thread_source` classification, subagent/reviewer rollup and unpriced Auto-review models (X-7). | Per-value parser fixtures and counts; after unit 2. | +| 4 | Count Claude by `sessionId`, exclude subagent and bridge transcripts, and remeasure source-bound census. | Duplicate/session fixtures plus enumerated-count reproduction; after unit 2. | +| 5 | Runtime census symmetry for Claude.app and ChatGPT.app; attribute bundled CLIs to observed sessions. | Runtime fixtures on both app forms; after units 2–4. | +| 6 | Third-party Claude provider from session-bound provider-specific assistant model ID; unknown remains unknown and provider is a separate detail field. Cloud label renders only for observations. | Evidence-precedence and unknown fixtures; after unit 2. | +| 7 | Imported turn exclusion per turn; later genuine Codex turns count and establish an actual app origin (decision 12/B1-4). | Mixed import/real-turn fixture and enumerated-count reproduction; after unit 2. | +| 8 | Token-bearing Codex record with zero responses: count it or document exact unsupported shape (UA-5). | Minimal shape reproduction and reconciliation; after unit 2. | +| 9 | O-7 session `byProvider` last-wins repair. | Count-only reproduction, provider totals; after parser integration. | +| 10 | X-8 Codex effort, first-token time and compaction capture. | Field fixtures and aggregate reconciliation; after parser integration. | +| 11 | C-6 Claude `cost-state` reconciliation. | Cost-state fixture and count-only sample; after parser integration. | +| 12 | C-8 cross-file message-id dedup. | Duplicate message fixture and count-only sample; after parser integration. | +| 13 | C-9 local-timezone day bucketing frozen in cache. | Boundary-day fixtures in two zones; after cache integration. | +| 14 | C-11 unknown-record counter. | Known/unknown record fixtures; after parser integration. | +| 15–19 | O-6, O-9, O-10, O-11, O-12, each in a separate commit. | Accepted: cost trust and cache semantics, explicit database selection, V2/legacy coverage warnings, bounded compaction/reconciliation and child fingerprint exclusion. | +| 20 | StatusLine classifier reads local managed settings. | Settings fixture and classifier regression; independently reordered before V4 integration. Other managed policy channels remain unobserved. | +| 21 | Shell wrapper around footer helper is `custom` (UA-4). | Wrapper fixture; after unit 20. | +| 22 | Shared labels across Usage, System Projects, Maintenance and Intelligence; unknown has one label and designations use separate axes. | View assertions; **after V3 merges into develop**, then integrate develop. | +| 23 | Show imported exclusion count in Intelligence census, System Projects and `ak system`; disclose Cowork source coverage is absent. | Three render assertions and source-bound counts; after V3 and unit 7. | +| 24 | Accept ADR-0060 and align DDD/docs to verified implementation. | Docs links and drift checks; assigned documentation writer after all prior units; controller owns shared indexes. | + +## Source and test map + +This table preserves the initial candidate boundaries. Exact delivered files and source-bound +results are in the private unit reports; the candidates grant no new edit authority. All source +units are accepted and only the documentation candidate remains under review. + +| Unit | Candidate source | Test entrypoint | +|---|---|---| +| 2 | `src/lib/usage-parsers.mjs`, `usage-project-evidence.mjs`, `usage-index.mjs` | `tests/kit/usage-index.test.mjs`, `usage-codex-attribution.test.mjs` | +| 3 | `src/lib/usage-parsers.mjs`, `usage-aggregate.mjs` | `tests/kit/usage-codex-attribution.test.mjs`, `usage-classify.test.mjs` | +| 4 | `src/lib/usage-parsers.mjs`, `footprint/project-sources.mjs` | `tests/kit/usage-claude-dedup.test.mjs`, `dashboard-project-identity.test.mjs` | +| 5 | `src/lib/footprint/runtime.mjs`, `project-census.mjs` | `tests/kit/footprint-collectors.test.mjs`, `system-summary.test.mjs` | +| 6 | `src/lib/usage-parsers.mjs`, `usage-local-provider.mjs` | `tests/kit/usage-provenance.test.mjs`, `usage-local-pricing.test.mjs` | +| 7 | `src/lib/codex-import-marker.mjs`, `usage-parsers.mjs`, `footprint/project-sources.mjs` | `tests/kit/project-sources-imports.test.mjs`, `usage-codex-attribution.test.mjs` | +| 8 | `src/lib/usage-parsers.mjs`, `usage-aggregate.mjs` | `tests/kit/usage-codex-attribution.test.mjs`, `usage-codex-large-rollout.test.mjs` | +| 9 | `src/lib/usage-opencode.mjs`, `usage-aggregate.mjs`; parser row identity already exists | `tests/kit/usage-opencode.test.mjs`, `usage-index-opencode.test.mjs`, `usage-index.test.mjs` | +| 10 | `src/lib/usage-parsers.mjs`, `usage-insights.mjs` | `tests/kit/usage-codex-attribution.test.mjs`, `usage-context.test.mjs` | +| 11 | `src/lib/usage-parsers.mjs`, `usage-cost.mjs`, `usage-aggregate.mjs`, `usage-index.mjs` | `tests/kit/usage-claude-cost-state.test.mjs`, `usage-claude-dedup.test.mjs`, `usage-index.test.mjs` | +| 12 | `src/lib/usage-parsers.mjs`, `usage-index.mjs` | `tests/kit/usage-claude-dedup.test.mjs`, `usage-index.test.mjs` | +| 13 | `src/lib/usage-index.mjs`, `usage-aggregate.mjs` | `tests/kit/usage-index.test.mjs`, `usage-claude-window-pairing.test.mjs` | +| 14 | `src/lib/usage-parsers.mjs`, `usage-aggregate.mjs` | `tests/kit/usage-index.test.mjs`, `usage-telemetry.test.mjs` | +| 15 O-6 | `src/lib/usage-opencode.mjs`, `usage-cost.mjs` | `tests/kit/usage-opencode.test.mjs`; accepted | +| 16 O-9 | `src/lib/usage-opencode.mjs`, `usage-opencode-bounds.mjs` | `tests/kit/usage-index-opencode.test.mjs`; accepted | +| 17 O-10 | `src/lib/usage-opencode.mjs`, `usage-index.mjs` | `tests/kit/usage-opencode.test.mjs`; accepted | +| 18 O-11 | `src/lib/usage-opencode.mjs`, `usage-aggregate.mjs` | `tests/kit/usage-opencode.test.mjs`; accepted | +| 19 O-12 | `src/lib/usage-opencode.mjs`, `usage-parsers.mjs` | `tests/kit/usage-opencode.test.mjs`; accepted | +| 20 | `src/lib/quota.mjs` | `tests/kit/quota.test.mjs`, `usage-limits-empty-state.test.mjs` | +| 21 | `src/lib/quota.mjs` | `tests/kit/quota.test.mjs` | +| 22 | `src/lib/dashboard/client/usage.mjs`, `system-projects.mjs`, `intelligence.mjs`, `maintenance-filters.mjs` | `tests/kit/dashboard-project-groups.test.mjs`, `intelligence-table-groups.test.mjs`, `maintenance-dashboard-client-labels.test.mjs` | +| 23 | `src/lib/dashboard/client/intelligence.mjs`, `system-projects.mjs`, `src/commands/system.mjs` | `tests/kit/dashboard-intel-integration.test.mjs`, `system-command.test.mjs` | +| 24 | `docs/adr/0060-session-surface-initiator-and-product-names.md`, relevant DDD guide | `tests/kit/docs-layout.test.mjs` and Markdown lint | + +Units 9–19 recorded bounded source observations and synthetic affected-row evidence in their +unit reports. A sample without an affected row is not proof of current-user impact. Reference +counts in ADR-0060 remain historical. No new real-data probe runs in Unit 24. + +## Unit 18 accepted: OpenCode compaction and reconciliation + +The parser reads bounded compaction parts and selected session metadata. A user +compaction request plus an error-free assistant summary with a finish value and +that actual parent link establishes one completed observation per request. +Requests alone, orphan summaries and in-flight markers retain uncertainty in the +lower/upper bounds. Failed or aborted summaries do not establish completion. +The OpenCode aggregate projection now retains those bounds; Codex and Claude +projections retain their existing behavior. + +Session counters are diagnostic only. Exact OpenCode v1.18.33 source shows that +session totals accumulate step-finish parts, but assistant tokens hold the latest +step. Reconciliation therefore requires completed valid messages, exactly one +matching valid step per assistant, populated valid session counters, and no V2 +rows in that session. Multiple or missing steps, incomplete metadata, unsupported +versions/token bases and untrusted hosted zero costs remain unknown. Matching or +mismatching counters never replace or add to message usage. No steps are billed a +second time. The cache marker extends cost-trust-v2 with observations-v1; schema +26, source identity and timezone checks remain intact. + +The bounded local metadata sample contained three sessions, no compaction parts, +no in-flight markers and no populated session counters. Its database digest was +unchanged. This is not positive affected-user evidence; synthetic fixtures cover +the supported and failure cases. Detailed commands and evidence are in the +ignored task18 report. + +Unit 18 review fixes bind warm reuse to a SHA-256 digest of the selected session's +observation metadata, relevant message fields, compaction/step-finish parts, and +V2 scope presence. Both the probe and parser stay within the same per-session +acquisition ceilings and their own read snapshots; the persisted digest comes +from the parse snapshot. Unchanged inputs reuse the cache; same-count rewrites +and removals invalidate it even when upstream timestamps do not change. No +whole-database payload hash or prompt-body hash is used. + +Response-free request evidence now remains in the current/previous compaction +bounds. Refused acquisitions contribute only their unknown bound, preserving the +existing rule that they do not become ordinary zero-cost session rows. Neither +path manufactures responses, tokens or billing. Unit 18 was accepted before +Unit 19 began. + +## Unit 19 accepted: OpenCode child prompt fingerprints + +OpenCode sessions with a nonempty `parent_id` retain prompts, turns, tokens, +provider costs and subagent classification, but produce no prompt fingerprints. +Both scan and selected-session parsing apply this rule. Aggregation also excludes +old cached child fingerprints from typed-prompt metrics, current prompt patterns +and historical baselines; the cached bytes remain until normal invalidation. +Main-session behavior and Unit 18 cache identity, timezone and observation marker +checks remain intact. Schema 26 is unchanged. + +Synthetic native-schema tests cover matching and distinct parent/child text, +child-only historical windows, absent parent rows, scan/read parity, stale warm +cache consumption and retained provider usage. The prior bounded preflight found +no child sessions; it does not establish current-user impact. Unit 19 is independently accepted; Unit 24 is the documentation candidate. Commands and results are in the ignored +task19 report. diff --git a/docs/archive/2026-09-29-plan-v6-opencode-cost.md b/docs/archive/2026-09-29-plan-v6-opencode-cost.md new file mode 100644 index 00000000..d9cd70ff --- /dev/null +++ b/docs/archive/2026-09-29-plan-v6-opencode-cost.md @@ -0,0 +1,35 @@ +# V6 Unit 15: OpenCode reported-zero cost trust + +## Status + +The scoped parser/cost work and core cache invalidation handoff are implemented and independently accepted. The final OpenCode marker also includes observation semantics. +Full V6 verification passed at `1c02db91`; feature PR CI and develop integration +remain separate gates at this archival capture. No live store mutation is claimed. + +## Decision and scope + +OpenCode may record zero when a model has no configured rate. A positive-token +assistant response with recorded cost zero is therefore unpriced when its +provider is nonlocal or unknown. A known local provider's zero remains observed. +Positive recorded costs, zero-token responses, and missing-cost estimates keep +their existing treatment. Provider attribution remains per response. + +This unit changes only the OpenCode parser and the shared cost reader at its +OpenCode-specific row boundary. No rate is inferred from the model name, host, +or environment. + +## Evidence and acceptance + +- A bounded, read-only live-store query checks counts and shape, with a file + digest before and after. No affected positive-token zero-cost row was found. +- Synthetic SQLite messages with the real storage shape test remote, unknown, + and local provider IDs; mixed observed and missing cost; malformed costs; + and cold/warm cache conservation. +- Focused tests, typecheck, scoped lint, and diff checks gate the unit commit. + +## Integration dependency + +Existing schema-26 cached OpenCode records cannot reconstruct which responses +had an untrusted reported zero after per-response data was coalesced. The core +owner must invalidate or reparse those old records before release. This unit +does not edit the core-owned cache/index module. diff --git a/docs/archive/2026-09-29-plan-v6-surfaces-ui.md b/docs/archive/2026-09-29-plan-v6-surfaces-ui.md new file mode 100644 index 00000000..1f0c5d26 --- /dev/null +++ b/docs/archive/2026-09-29-plan-v6-surfaces-ui.md @@ -0,0 +1,20 @@ +# V6 session presentation execution + +## Status + +Units 22/23 and their consumer/default-suite handoff are implemented and independently accepted; final legacy Unknown filter compatibility is included. +Full V6 verification passed at `1c02db91`; feature PR CI and develop integration +remain separate gates at this archival capture. No live store mutation is claimed. + +Approved Units 22/23, based on `4a414024`; sole writer in `task/v6-surfaces-ui`. + +1. Add a shared presentation vocabulary and additive project session surface evidence. + Preserve legacy origin/count bases and honest uncertainty in old snapshots. +2. Render independent host, surface, initiator, provider and Git scope dimensions. + Retain bounded local raw details without inferring products or network providers. +3. Disclose pure imported exclusions, mixed files and unresolved ownership separately + in Intelligence, System Projects and the system command, including Cowork coverage. +4. Validate focused contracts and actual renderers, static checks and browser behavior. + Stop after unit commits for independent review; no publishing or personal data reads. + +Baseline: 29 focused renderer/import/empty-state tests passed before changes. diff --git a/docs/archive/2026-09-29-v6-codex-thread-source-evidence.md b/docs/archive/2026-09-29-v6-codex-thread-source-evidence.md new file mode 100644 index 00000000..9f8f9e5e --- /dev/null +++ b/docs/archive/2026-09-29-v6-codex-thread-source-evidence.md @@ -0,0 +1,31 @@ +# V6 Unit 3: Codex thread sources and Auto-review pricing + +Source state: `bf8babd4` plus this unit's changes. Checked 2026-09-29 10:57 UTC with `codex-cli 0.158.0`. + +## Findings and change + +- **Verified kit defect:** an unfamiliar `thread_source` on a known interactive originator inherited `person`. The classifier now reports `unknown`, while recognized user, handoff, agent, and automation values retain their declared initiators. The fixed SDK, exec, and MCP rules still apply. +- **Verified kit defect:** SQLite ledger backfill changed `threadSource` without rebuilding `sessionOrigin`. It now classifies from bounded raw declaration evidence and preserves the first rollout declaration over a later replayed parent declaration. +- **Verified kit gap:** the observed `source.subagent.thread_spawn.parent_thread_id` was not retained. The first metadata line now accepts only a UUID-shaped parent ID. Aggregate rollup uses a parent present in the same Codex record set, rejects missing links and cycles, and groups child and reviewer records under that parent's surface. It keeps child-owned usage and strips only a ledger-identified subagent whose rollout could not separate replay. +- **Verified kit defect:** `codex-auto-review` token rows were assigned unknown-model fallback dollars. This exact server-side alias now contributes tokens and unpriced-message coverage, with zero estimated dollars and no invented cache saving. Published known-model pricing is unchanged. + +## Bounded local metadata probe + +Read only the first JSONL line of 1,806 local Codex rollout files, plus `turn_context.model` for guardian-review files. This is a file census, not a unique-session census or a billing statement. The `thread_source` counts were: `subagent` 510, `user` 200, absent/null 994, `guardian_review` 90, `chatgpt_handoff` 10, `agent_created_thread` 2. None had a structured `thread_source`; 600 had structured `source.subagent` (500 `thread_spawn`, 90 `other` with guardian review, 10 `other` with subagent). The 500 observed `thread_spawn` parent IDs were UUID-shaped. Guardian-review files contained 843 `turn_context` declarations of `codex-auto-review`. The probe did not read prompt text, deduplicate sessions, inspect imports, or verify a published price. + +## Verification + +- RED: new focused tests failed on unfamiliar source classification, parent extraction, ledger origin, and Auto-review fallback pricing. A separate cycle fixture failed before the cycle guard. +- GREEN: `node scripts/run-tests.mjs exec -- --test tests/kit/usage-codex-attribution.test.mjs tests/kit/usage-session-surface.test.mjs tests/kit/session-surface.test.mjs tests/kit/usage-index-v6.test.mjs tests/kit/usage-codex-thread-source.test.mjs tests/kit/pricing.test.mjs` — 101 passed, 0 failed. +- `node node_modules/typescript/bin/tsc --noEmit` — passed. `node node_modules/eslint/bin/eslint.js` on the six changed source/test files — passed. `git diff --check` — passed. +- The synthetic aggregate fixture checks parent, child, and reviewer totals and cold/warm cache consistency without touching the real usage cache. + +## Accounting limits + +The 26 schema remains unchanged. Existing schema-26 caches created before this unit may lack parsed parent IDs until a fresh cache rebuild; the final pre-PR gate owns that rebuild. A reviewer without a verified parent remains on its declared surface because no parent can be inferred. No provider calls, full unit/UI suite, real cache migration, or user database writes were made. + +## Independent review fix round 1 — 2026-09-29 11:05 UTC + +- **P2 confirmed:** a present `thread_source` rejected by the bounded token parser (`{}` or a string containing spaces) became `null`, allowing a known interactive originator's `person` default. The classifier now distinguishes an absent declaration from a rejected one without retaining rejected content. Focused fixtures cover direct classification, the first rollout metadata line, ledger overlay, and the SDK, exec, and MCP fixed initiators. +- **P3 confirmed:** ledger backfill called `classifySessionSurface` without the imported-copy flag and could replace an imported copy's `initiator` with `agent`. `ledgerOrigin` now preserves the parser's imported-copy override. The regression fixture uses a real minimal imported rollout and a synthetic guardian-review ledger row; the exported helper remains safe even though `buildIndex` filters imported records before aggregation. +- RED: both new defect fixtures failed against `5fe016d1`. GREEN: the same six focused test files listed above passed, **103 tests, 0 failures**. TypeScript `--noEmit`, targeted ESLint on the three changed source/test files, and `git diff --check` passed. No full unit/UI run or local corpus repeat was performed. diff --git a/docs/archive/README.md b/docs/archive/README.md index 6e65c9bb..3929324e 100644 --- a/docs/archive/README.md +++ b/docs/archive/README.md @@ -69,6 +69,10 @@ reconfirmed by this metadata audit. The per-file inventory and limitations are r | File | Original location | What it was | Why it's historical | |---|---|---|---| | [2026-09-28-plan-follow-ups-v2.md](2026-09-28-plan-follow-ups-v2.md) | `docs/plans/2026-09-28-follow-ups-v2.md` | V4 product, CLI, memory, process-lifecycle and upstream integration follow-ups. | All eight local gates and independent whole-branch review passed at `29152654`; final-head PR CI and squash integration were pending at archival. Conditional Ruflo #3419 guidance remains deferred; native Windows AQE was not tested. | +| [2026-09-28-plan-usage-accuracy.md](2026-09-28-plan-usage-accuracy.md) | `docs/plans/2026-09-28-usage-accuracy.md` | Completed V6 usage and session evidence plan | All 24 units independently accepted; whole-branch review and all eight local gates passed through `1c02db91`. Feature PR CI and develop integration remain separate gates at archival. Current contracts: [Usage metrics](../usage-scorecard-metrics.md) and [ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md). | +| [2026-09-29-plan-v6-surfaces-ui.md](2026-09-29-plan-v6-surfaces-ui.md) | `docs/plans/2026-09-29-v6-surfaces-ui.md` | Completed V6 session presentation plan | Shared vocabulary, legacy filter compatibility, independent evidence dimensions and source-coverage disclosures; included in the V6 gates at `1c02db91`. | +| [2026-09-29-plan-v6-opencode-cost.md](2026-09-29-plan-v6-opencode-cost.md) | `docs/plans/2026-09-29-v6-opencode-cost.md` | Completed OpenCode reported-zero trust plan | Scoped cost work and mandatory core cache handoff accepted; versioned observation semantics are documented in the living usage guide. | +| [2026-09-29-v6-codex-thread-source-evidence.md](2026-09-29-v6-codex-thread-source-evidence.md) | `.superpowers/sdd/2026-09-28-usage-accuracy/task3-report.md` | Historical V6 Unit 3 evidence | Original bounded metadata census and initial/fix tests retained verbatim. Its interim cache note describes that capture stage, not a live-cache mutation or current release claim. | | [2026-09-29-plan-upstream-watch-followups.md](2026-09-29-plan-upstream-watch-followups.md) | `docs/plans/2026-09-29-upstream-watch-followups.md` | V4 C4 execution plan for bounded PR polling, deterministic retry failures and twelve deferred watcher minors. | Implementation and independent review complete; all eight local gates passed at `b75c1e3f`. Final PR CI and merge were pending at archival. Current contract: [Upstream watch](../upstream-watch.md). | | [2026-06-upstream-findings-f1-f6.md](2026-06-upstream-findings-f1-f6.md) | `docs/upstream/ruflo-self-improvement-findings.md` | The F1–F6 findings series: proofs/refutations of ruflo's self-improvement claims (Q-learning persistence, state-encoder collapse, SONA learn→inference wiring, native-training misreporting), with filed upstream issues. | Every finding is now fixed upstream: F2 in 3.10.6 ([#2222](https://github.com/ruvnet/ruflo/issues/2222)), F2b in 3.10.7, F3 in 3.10.11 ([#2239](https://github.com/ruvnet/ruflo/issues/2239)), F4 in `@ruvector/ruvllm` 2.5.6 ([RuVector#519](https://github.com/ruvnet/RuVector/issues/519)), F6 in 3.18.1/3.19.0 + ruvllm 2.5.7 ([#2549](https://github.com/ruvnet/ruflo/issues/2549), closed 2026-07-03). | | [2026-06-token-consumption-incident.md](2026-06-token-consumption-incident.md) | `docs/usage/token-consumption-findings-and-mitigation-2026-06.md` | Root-cause report for the June 2026 token-burn incident: six immortal auto-started daemons consumed ~8.1B tokens over 7 days via headless worker sessions. Produced the opt-in daemon policy, TTL reaper, ⚙ statusline alarm, and `ruflo-token-audit`. | The root cause was fixed upstream in ruflo 3.27/3.28 ([#2661](https://github.com/ruvnet/ruflo/issues/2661)): AI workers are opt-in, launches are governed by a machine-wide budget with telemetry, one supervisor daemon per repo, native daemon TTL. The kit's daemon policy flipped back to default-on (local-only workers) on that baseline; the reapers and token-audit remain as an independent check. | diff --git a/docs/codex-usage-diagnostic.md b/docs/codex-usage-diagnostic.md index 4151b979..d9284fcc 100644 --- a/docs/codex-usage-diagnostic.md +++ b/docs/codex-usage-diagnostic.md @@ -9,6 +9,16 @@ Everything you need is below: what was wrong, why the fix can be trusted without re-auditing the code yourself, how to run one script, and exactly what to send back. +**Updated 2026-09-29 — diagnostic scope.** This script remains an independent historical +cumulative-snapshot comparison. It does not reproduce the current parser's per-turn import +ownership, replay subtraction, counter segments, per-day/model attribution, or positive component +usage with zero responses. Use [ADR-0052](adr/0052-codex-usage-attribution.md) and the +[current accounting contracts](usage-scorecard-metrics.md#current-accounting-and-cache-contracts) +to interpret differences. Total-only counters remain unsupported for split/pricing, Auto-review +models may be unpriced, and host-reported first-token/compaction evidence is separate. A mismatch +with this script alone is not evidence of a present accounting defect. The figures below are +historical, not a new corpus measurement. + --- ## The short version diff --git a/docs/dashboard.md b/docs/dashboard.md index ad48fdea..0654494c 100644 --- a/docs/dashboard.md +++ b/docs/dashboard.md @@ -237,6 +237,24 @@ or proof of subscription billing. Claude Code writes one transcript line per con repeats the message's usage on each, so Claude tokens, cost, responses and context samples count each API message once. +**Evidence and compatibility (updated 2026-09-29).** Surface, initiator and provider are +separate from host and Git scope. Local session/project detail exposes bounded declared origin +tokens and observed provider basis; Unknown remains explicit, and provider metadata is not +network attestation. Legacy desktop filters retain their labeled membership. A coarse legacy +Codex desktop snapshot gives a ChatGPT desktop app family note without guessing its mode. +Cloud choices require observations; dedicated Cowork storage remains uncovered. Census views +show imported exclusions, mixed/unresolved observations, count basis and incomplete coverage. +Runtime distinguishes desktop applications from their observed CLI sessions. + +Usage counts proven native intervals in mixed imports and positive Codex token components even +without responses. Claude copied messages have one charge owner across the bounded current and +previous window pool; session-local observations remain qualified. OpenCode unknown/nonlocal +positive-token zero cost is unpriced. Source warnings remain visible without current-window +activity, and ambiguous databases require explicit selection. Older caches may rebuild; old +OpenCode child fingerprints are ignored by prompt metrics while usage remains. See the +[current contracts](usage-scorecard-metrics.md#current-accounting-and-cache-contracts) for exact +bounds, timezone behavior and the version-limited OpenCode reconciliation signal. + **Two hero rows.** The first carries sessions, api-equivalent cost, tokens, engaged time, and cache read. Each tile pairs its figure with a change against the previous window of the same length and a per-day sparkline, so the number and its direction arrive together. The change is read @@ -267,7 +285,8 @@ each with its percentile markers laid over the bars. A percentile that lands in top bucket renders with a `≥` prefix — the bucket has no upper edge, so the honest claim is a floor rather than a point. A window holding no samples reads `not measured` instead of a row of zero bars. **Response latency is the gap between a prompt and the response that answered it. It is not -time-to-first-token**, which no local transcript records. +time-to-first-token**. Codex can separately record host-reported first-token timing; that +observation does not rename or replace the completion-latency metric. **How you run** answers permission posture, who drove, and who served. Posture is a closed four-value vocabulary — guarded, auto-edit, plan, unrestricted — mapped from each host's own @@ -279,9 +298,9 @@ own nested transcript, so that cost is discovered, priced, and included; a forke rollout opens with its parent's replayed history, so only what follows the replay is counted — the subagent's own tokens are priced and the parent is never billed twice. A subagent from a host that records no event ordinals cannot have its replay separated and reads `$0.00`, which means not -measurable rather than cheap. Codex sessions imported from Claude Code are not Codex activity and are -excluded from every Codex figure. The panel does not rank window cost by inference provider: a transcript host is not a vendor. -Codex and OpenCode can record a serving provider, while Claude history lacks that field; +measurable rather than cheap. Copied turns imported from Claude Code are excluded from Codex figures; proven native +turns in mixed files remain eligible, with unresolved ownership disclosed. The panel does not rank window cost by inference provider: a transcript host is not a vendor. +Codex and OpenCode can record a serving provider, while Claude may expose provider-specific assistant model metadata; identity is reported per session on the Sessions detail strip — beside the provenance backing it — rather than as a window axis. diff --git a/docs/ddd/machine-footprint.md b/docs/ddd/machine-footprint.md index a27b8295..c7ad16f3 100644 --- a/docs/ddd/machine-footprint.md +++ b/docs/ddd/machine-footprint.md @@ -646,15 +646,22 @@ declarations and unclassified sightings. `countBasis` distinguishes transcript f sessions, recovered-project sightings and mixed observations; a recovered directory is not one verified session. -Imported session copies are not sightings. A Codex rollout stamped `external-import-turn-*` is a -Claude Code transcript that the ChatGPT desktop app imported; it is a copy, and the original Claude -Code session is counted where its transcript still exists. A folder that only an import names is -therefore not a project. Discovery skips it before reading its cwd, so it adds no project, host or Session -origin, and counts it in `importedExcluded` (per host scan and in total). `complete` is unaffected. - -**Proposed change ([ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md)).** -`sessionOrigins` is to be replaced by session surface and initiator, derived from the same declared -fields but keeping every raw value. +Updated 2026-09-29: imported Codex turns are not sightings. Pure copies add no project, host or +surface and count in `importedExcluded`; proven native activity in a mixed file can establish a +sighting and counts in `importedMixed`. Bounded observations that cannot settle ownership count +in `importedUnresolved` and make coverage incomplete. The budgets and malformed-record behavior +are specified in [ADR-0052](../adr/0052-codex-usage-attribution.md#3-sessions-codex-imported-from-claude-code-are-not-codex-sessions). + +Accepted [ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md) adds +`sessionSurfaces` alongside legacy `sessionOrigins`. Claude census counts declared session IDs, +excluding subagent and bridge-only records; recovery has zero session weight. Raw detail retains +at most 16 validated tokens per field per classification group and reports truncation. Cloud +surfaces appear only with observations; dedicated Cowork storage remains uncovered. Old coarse +Codex desktop origins cannot identify an app mode. Footprint schema remains 8. + +Runtime distinguishes Claude Desktop and the ChatGPT desktop app as applications with no host. +Their bundled CLI sessions require observed process attribution; Codex app-server is a service. +Application presence alone does not establish an active conversation or historical provider. `project-identity.mjs` relates directories through canonical Git metadata and, for linked worktrees, a verified common directory plus backlink. It preserves unknown association when evidence is diff --git a/docs/ddd/ubiquitous-language.md b/docs/ddd/ubiquitous-language.md index c62b9a92..dda55274 100644 --- a/docs/ddd/ubiquitous-language.md +++ b/docs/ddd/ubiquitous-language.md @@ -133,20 +133,21 @@ missing price. `Dual-host` describes two enabled peer hosts, not an execution command and not evidence that two inference vendors served a workflow. Generalized execution belongs to `ak run`. -## Session surface language (mostly proposed) +## Session surface language -These terms are proposed by [ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md). -Only **Imported session copy** is implemented so far, and only in usage and project discovery. For -the rest, the implemented contract is ADR-0050's **session origin** (`claude-desktop`, -`codex-desktop` or `unknown`), which an imported copy never supplies. +Updated 2026-09-29 to match accepted +[ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md). Git scope, host, surface, +initiator and provider are separate dimensions. Legacy origin keys remain compatibility evidence. | Term | Meaning | |------|---------| | Session surface | The product surface that started a session, read from the host's own declared field (Claude `entrypoint`; Codex `originator` with `source`) and shown by its official name, such as Claude Code CLI, Claude Desktop, ChatGPT desktop app · Codex, or Codex CLI | -| Initiator | Who started a session: a person, automation (scripts, SDKs, non-interactive runs, CI), an agent (a subagent or reviewer spawned by another session), or an imported copy | -| Imported session copy | A session one tool copied from another, such as a Claude Code transcript the ChatGPT desktop app imported as a Codex thread; excluded from usage, origin and project counts and reported as a count | -| Raw surface value | The exact declared value a surface was derived from; always kept, and shown for any value the vocabulary does not recognize | +| Initiator | Who started a session: a person, automation (scripts, SDKs, non-interactive runs, CI), an agent (a subagent or reviewer spawned by another session), an imported copy, or Unknown | +| Imported session copy | A session one tool copied from another, such as a Claude Code transcript the ChatGPT desktop app imported as a Codex thread; whose copied turns are excluded from usage and project/origin sightings; proven native turns in mixed files remain eligible, with exclusion and incompleteness counts | +| Raw surface value | A named declaration token retained only under the approved 80-character validation rule; unfamiliar valid tokens can appear in local detail without inferring a product or provider | | Tool workspace | A folder a tool creates for its own work outside the user's projects, such as `~/.codex/.chatgpt-projects/…` or `~/Documents/Codex/…`; an explanation attribute, never a surface | +| Session count basis | The counted unit: declared session IDs, transcript files, database sessions, recovered sightings or mixed observations; zero-weight recovery is not a verified session | +| Provider evidence basis | Recorded provider ID or provider-specific assistant model ID; observed metadata, not network attestation | | Desktop application | Claude Desktop or the ChatGPT desktop app; an application that can start sessions, not a host | Say **session surface** for where a session came from; the Live event `surface` field (native, ruflo, diff --git a/docs/plans/2026-09-28-remediation-v2-develop-execution.md b/docs/plans/2026-09-28-remediation-v2-develop-execution.md index bed7dd78..c0d5c16e 100644 --- a/docs/plans/2026-09-28-remediation-v2-develop-execution.md +++ b/docs/plans/2026-09-28-remediation-v2-develop-execution.md @@ -130,6 +130,11 @@ V4's temporary busy-rule/memory-routing CI probe is sandboxed and removed before V6 reads the actual cache schema before making exactly one migration; the old plan's 25 → 26 number must not overwrite a schema bump that has already landed. +V6 Unit 21 implements static `statusLine` command classification for direct helper +invocations and treats shell wrappers, inline programs, and chains as `custom`. +Its focused and unit gates passed; the unit commit is pending independent review. +V6's UI copy remains with V3 until that stream integrates. + ### Scheduling and team shape The session has four slots total: controller plus at most three workers. Ruflo or AQE diff --git a/docs/transcripts.md b/docs/transcripts.md index cb3a61b7..357ed048 100644 --- a/docs/transcripts.md +++ b/docs/transcripts.md @@ -30,29 +30,37 @@ checks run in the test suite --- +**Updated 2026-09-29.** The scan and reader share source selection and parse semantics, while +cross-file Claude charge ownership occurs only after discovery. Session-local transcript counts +need not equal aggregate accounted responses. Usage cache schema 26 additionally requires calendar, +source and semantic compatibility; the read-path figure below illustrates the original split, +not every current cache key. OpenCode child user turns remain readable but do not enter prompt +fingerprints. See the [current accounting contracts](usage-scorecard-metrics.md#current-accounting-and-cache-contracts) +for bounded imports, ownership, source coverage, reconciliation and warm-cache validation. + ## 1. Transcript stores Claude and Codex use JSONL files; OpenCode uses its SQLite session/message/part store. The kit reads source histories without rewriting them (transcripts are never -rewritten; rule 3 of the module header, `usage-index.mjs:22`): +rewritten; rule 3 of the module header, `usage-index.mjs:24`): | Host | Store | Discovered by | |---|---|---| -| Claude Code | `~/.claude/projects//.jsonl` | `listClaude` (`usage-index.mjs:374-388`) — exactly one level of project directories | -| Claude Code (subagent) | `~/.claude/projects///subagents/agent-.jsonl` | `listClaudeSubagents` (`usage-index.mjs:349-354`) — the one nested shape `listClaude` descends into | -| Codex CLI | `~/.codex/sessions///
    /rollout--.jsonl` | `listCodex` (`usage-index.mjs:391-411`) — the `yyyy/mm/dd` tree walk | +| Claude Code | `~/.claude/projects//.jsonl` | `listClaude` (`usage-index.mjs:460-474`) — exactly one level of project directories | +| Claude Code (subagent) | `~/.claude/projects///subagents/agent-.jsonl` | `listClaudeSubagents` (`usage-index.mjs:435-440`) — the one nested shape `listClaude` descends into | +| Codex CLI | `~/.codex/sessions///
    /rollout--.jsonl` | `listCodex` (`usage-index.mjs:477-497`) — the `yyyy/mm/dd` tree walk | | OpenCode | platform data root `opencode/opencode.db` (normally `~/.local/share/opencode/opencode.db` on Unix) | `usage-opencode.mjs` reads session/message/part rows with read-only SQLite queries | -Roots come from `defaultRoots()` (`usage-index.mjs:324-329`) and are injectable +Roots come from `defaultRoots()` (`usage-index.mjs:374-415`) and are injectable for tests. A malformed line is skipped, never fatal (`jsonLines`, -`usage-parsers.mjs:174-180` — one corrupt line must not cost a whole file). +`usage-parsers.mjs:183-191` — one corrupt line must not cost a whole file). A session's **delegated** work is a real transcript of its own, written beside the parent under `/subagents/`. Discovery is that one nested shape and no more — not a recursive walk — so a directory that is not a session-id directory with a `subagents` child contributes nothing rather than being crawled. Each such record takes a **namespaced** id, `/` -(`usage-index.mjs:354`), because Claude Code names every subagent file +(`usage-index.mjs:440`), because Claude Code names every subagent file `agent-.jsonl` and that stem is not unique across two parent sessions; an unnamespaced id would silently collide two unrelated records into one. §4.1 covers how a namespaced id is validated and resolved back to its file. @@ -68,22 +76,22 @@ evidence when their native records name it; Claude history normally leaves it un ### 1.1 Claude entry vocabulary Each line has a top-level `type`. The parser (`parseClaude`, -`usage-parsers.mjs:747-777`) reads: +`usage-parsers.mjs:822-871`) reads: | `type` | What the parser takes from it | |---|---| -| "ai-title" | The model-written session title (`usage-parsers.mjs:768`) — preferred over the first-prompt fallback | -| `user` | A user-**role** turn — which is *not* the same as "the human"; see §3. On a turn that passes isHumanPrompt, also its `permissionMode` — the session's permission posture, read on the person's own turn only (`usage-parsers.mjs:593-609`) — and the opening of the response-latency window | -| `assistant` | One content block of a model message: `model` id, the message's `usage` token counts (repeated on every block's line, so counted once per `message.id`, else `requestId`, last line winning), `tool_use` blocks (`usage-parsers.mjs:661-743`) | -| any | Side-band fields read regardless of type: `attributionSkill`/`attributionPlugin` (`usage-parsers.mjs:770-771`), `isSidechain` (`usage-parsers.mjs:772-773`), `cwd` for project derivation | +| "ai-title" | The model-written session title (`usage-parsers.mjs:852`) — preferred over the first-prompt fallback | +| `user` | A user-**role** turn — which is *not* the same as "the human"; see §3. On a turn that passes isHumanPrompt, also its `permissionMode` — the session's permission posture, read on the person's own turn only (`usage-parsers.mjs:628-644`) — and the opening of the response-latency window | +| `assistant` | One content block of a model message: `model` id, the message's `usage` token counts (repeated on every block's line, so counted once per `message.id`, else `requestId`, last line winning), `tool_use` blocks (`usage-parsers.mjs:708-818`) | +| any | Side-band fields read regardless of type: `attributionSkill`/`attributionPlugin` (`usage-parsers.mjs:854-855`), `isSidechain` (`usage-parsers.mjs:856-862`), `cwd` for project derivation | A real assistant completion also closes two pieces of per-entry evidence the transcript does not state outright. It **closes the latency window** the preceding human prompt opened, into one `noteLatencySample` call over the gap -between them (`usage-parsers.mjs:722-725`); and it **conditionally sets `ctxLastTokens`** to the +between them (`usage-parsers.mjs:779-782`); and it **conditionally sets `ctxLastTokens`** to the tokens actually in the model's window for that turn — fresh input plus what was served from cache — so the field always describes the last completion rather -than a running total (`usage-parsers.mjs:405-410`; the call site is at lines 658–668). That write is +than a running total (`usage-parsers.mjs:430-435`; the call site is at lines 658–668). That write is evidence-gated: an entry whose `message.usage` is absent decodes to all-zeros, and a zero is not a measurement of an empty context, so it must not overwrite a real prior value. Neither is a field Claude Code writes; both are derived, per @@ -93,7 +101,7 @@ An assistant entry with `isApiErrorMessage: true` is a **local placeholder** Claude Code writes when a request dies before a real completion (connection drop, rate limit, auth failure — `model: ""`, all-zero usage). It is real engaged time but not a model attempt: counted as an *exception*, never -pushed into `models`, priced, or counted as a response (`usage-parsers.mjs:700-735`; the full story is +pushed into `models`, priced, or counted as a response (`usage-parsers.mjs:757-792`; the full story is [`usage-scorecard-metrics.md`](usage-scorecard-metrics.md) §10). It is not a latency sample either — the pending window is deliberately left open, so the first *real* completion that eventually follows is what gets timed. @@ -101,22 +109,22 @@ first *real* completion that eventually follows is what gets timed. ### 1.2 Codex entry vocabulary Codex rollout lines carry `type` + `payload`. The parser (`parseCodex`, -`usage-parsers.mjs:1183-1240`) reads: +`usage-parsers.mjs:1355-1442`) reads: | `type` / `payload.type` | What the parser takes from it | |---|---| -| `session_meta` | Authoritative session id, `cwd`, and `thread_source` — the FIRST such line in the file wins for all three, AND for `inferenceProvider`/`providerProvenance` too (`usage-parsers.mjs:831-841`, gate; `:767-784`, why); a subagent rollout replays its PARENT thread's own session_meta line later in the same file, and a later-wins rule let that relabel the record `subagent`→`user` and re-key its id to the parent's — `"subagent"` marks a delegated thread whose rollout may open with its parent's replayed history; events before the replay boundary (`codex-replay.mjs`) count for nothing, so the subagent's own tokens are counted and the replay is not, and the record stays flagged `subagent` (`usage-parsers.mjs:1188-1195`; `usage-scorecard-metrics.md` Appendix A, Bug B) | -| `turn_context` | The model id in effect from this point on, plus `approval_policy` (a string) and `sandbox_policy` (an **object** keyed `.type`, e.g. `{"type":"danger-full-access"}`) — the permission posture, last evidence winning, since a session may renegotiate mid-run (`usage-parsers.mjs:843-864`) | -| `event_msg` → `token_count` | A **cumulative** usage snapshot, turned into its increase over the previous one and booked on the event's local day under the model of the turn in effect; a snapshot lower than its predecessor starts a new segment (the host's counter restarted), so every segment counts. A replayed snapshot only advances the running total (`usage-parsers.mjs:898-909`) | -| `event_msg` → `task_started` | `model_context_window` — denominator-only compatibility evidence, not proof of a paired input/window sample — and the turn's start time (`usage-parsers.mjs:915-934`) | -| `event_msg` → `task_complete` | The host's own `duration_ms` for the turn, taken as a latency sample only when no prompt-to-response gap already covered it; a non-null `error` counts as an exception (`usage-parsers.mjs:936-954`) | -| `event_msg` → `turn_aborted` | An explicit interrupt: counted in `aborts`, and it clears both latency states so an unanswered prompt is never timed against a later, unrelated response (`usage-parsers.mjs:1048-1089`) | +| `session_meta` | Authoritative session id, `cwd`, and `thread_source` — the FIRST such line in the file wins for all three, AND for `inferenceProvider`/`providerProvenance` too (`usage-parsers.mjs:925-941`, gate; `:1343-1373`, why); a subagent rollout replays its PARENT thread's own session_meta line later in the same file, and a later-wins rule let that relabel the record `subagent`→`user` and re-key its id to the parent's — `"subagent"` marks a delegated thread whose rollout may open with its parent's replayed history; events before the replay boundary (`codex-replay.mjs`) count for nothing, so the subagent's own tokens are counted and the replay is not, and the record stays flagged `subagent` (`usage-parsers.mjs:1360-1367`; `usage-scorecard-metrics.md` Appendix A, Bug B) | +| `turn_context` | The model id in effect from this point on, plus `approval_policy` (a string) and `sandbox_policy` (an **object** keyed `.type`, e.g. `{"type":"danger-full-access"}`) — the permission posture, last evidence winning, since a session may renegotiate mid-run (`usage-parsers.mjs:943-964`) | +| `event_msg` → `token_count` | A **cumulative** usage snapshot, turned into its increase over the previous one and booked on the event's local day under the model of the turn in effect; a snapshot lower than its predecessor starts a new segment (the host's counter restarted), so every segment counts. A replayed snapshot only advances the running total (`usage-parsers.mjs:1003-1022`) | +| `event_msg` → `task_started` | `model_context_window` — denominator-only compatibility evidence, not proof of a paired input/window sample — and the turn's start time (`usage-parsers.mjs:1028-1047`) | +| `event_msg` → `task_complete` | The host's own `duration_ms` for the turn, taken as a latency sample only when no prompt-to-response gap already covered it; a non-null `error` counts as an exception (`usage-parsers.mjs:1049-1076`) | +| `event_msg` → `turn_aborted` | An explicit interrupt: counted in `aborts`, and it clears both latency states so an unanswered prompt is never timed against a later, unrelated response (`usage-parsers.mjs:1170-1218`) | | `event_msg` → `user_message` | A legacy-format prompt CANDIDATE — Codex does not route tool output through this event, but the text still needs the human-prompt gate below before it counts | | `event_msg` → `agent_message` | A legacy-format model response | | `event_msg` → `item_completed` → `UserMessage` | A current-format prompt candidate; text blocks use the observed lowercase `text` discriminator; also gated below | | `event_msg` → `item_completed` → `AgentMessage` | A current-format model response; text blocks use the observed uppercase `Text` discriminator | -| Human-prompt gate (`isCodexHumanMessage`, `usage-parsers.mjs:970-974`) | Codex carries no discipline of its own for telling a typed prompt apart from harness output or a mirrored cross-host envelope replayed into the rollout rather than typed there. Reuses "HARNESS_OUTPUT_RE" verbatim (Claude's own envelope markers reproduce byte-for-byte inside a mirrored rollout) plus two Codex-specific machine markers (`CODEX_MACHINE_ENVELOPE_RE`, `usage-parsers.mjs:956-973`): a `/` with one slash, where the parent half reuses `VALID_ID`'s own charset and the child half must match the real on-disk `agent-…` shape. The namespaced grammar is a **narrowing** of the plain one, never a loosening: both are the same path-traversal guard, and a traversal shape is rejected at either tier. -2. **Locate by id** across both roots (`locate`, `usage-index.mjs:1031-1050`), +2. **Locate by id** across both roots (`locate`, `usage-index.mjs:1227-1246`), consulting the scan cache when present but never requiring it — `readSession` works with no prior buildIndex. A namespaced id resolves through this call: `locateSubagent(nested.parentId, nested.stem, r.claude, id)` - (`usage-index.mjs:1048`), which builds the + (`usage-index.mjs:1244`), which builds the nested path from the two **already-validated capture groups** rather than from raw request text. -3. **Realpath containment** (`usage-index.mjs:1122-1135`) — the resolved file +3. **Realpath containment** (`usage-index.mjs:1318-1331`) — the resolved file must live under a transcript root *after* `realpathSync` collapses symlinks; a symlink planted inside a root pointing at `/etc/anything` passes a lexical `startsWith` but fails this. Roots are realpath'd too so a symlinked dotfiles setup still works. -4. **Size cap** — `MAX_SESSION_BYTES` (64 MB, `usage-index.mjs:195`): a +4. **Size cap** — `MAX_SESSION_BYTES` (64 MB, `usage-index.mjs:241`): a transcript is read whole and JSON-expands ~5×, so an unbounded read is a memory-amplification primitive. Oversized reads as unavailable, not risky. ### 4.2 Parse and price The file is parsed with `withTurns: true` by the provider's parser -(`usage-index.mjs:1157-1164`), and `meta` is assembled by `sessionPayload` -(`usage-aggregate.mjs:1324-1361`). Its call builds a narrower subset of the Sessions +(`usage-index.mjs:1330-1339`), and `meta` is assembled by `sessionPayload` +(`usage-aggregate.mjs:1479-1518`). Its call builds a narrower subset of the Sessions view fields: `prompts`, `responses`, `exceptions`, `sidechain`, `threadSource`, `models`, `tools`, `skill`/`plugin`, worktree — plus a `cost` priced from the same per-model usage rows `aggregate()` uses. OpenCode recorded row cost wins over @@ -421,11 +429,11 @@ never renames a retained session model, changes historical token pricing, or rew ### 4.3 Mask, then truncate — both marked, differently -Every turn body is passed through `maskSecrets` (`usage-aggregate.mjs:168-173` — the +Every turn body is passed through `maskSecrets` (`usage-aggregate.mjs:171-176` — the configured secret shapes) **server-side, before serialization**, then length-capped at `MAX_TURN_CHARS` (40,000, -`usage-aggregate.mjs:73`) with the marker appended at the truncation call -("originalChars is measured", `usage-aggregate.mjs:1346-1355`). Two invariants: +`usage-aggregate.mjs:76`) with the marker appended at the truncation call +("originalChars is measured", `usage-aggregate.mjs:1503-1512`). Two invariants: * **Presence is the signal.** `truncated`/`originalChars` are emitted only when the slice fired, so a complete turn cannot be misread as abridged. @@ -604,7 +612,7 @@ was wrong before, for the curious. `isHumanPrompt` once counted `harness-output` envelopes as human prompts — 32 claimed vs 20 real on the reference session. Cached session records carried the inflated counts, hence the wholesale `SCHEMA_VERSION` 5 cache - invalidation ("no longer count as human prompts", `usage-index.mjs:70-73`). + invalidation ("no longer count as human prompts", `usage-index.mjs:78-81`). * **Session expander fields shipped but unrendered.** The per-session fields §6.1's expander now renders (classification `basis` + confidence, the token split, flags) once travelled on the wire and rendered nowhere. @@ -612,7 +620,7 @@ was wrong before, for the curious. assembled `meta` left cost undefined, and `fmtUsd(undefined)` renders the truthy string `"$0.00"` — a fixed-looking zero on a panel whose whole subject is cost. `meta.cost` is now priced via this call: `sessionCost(rec, deps)` - (`usage-aggregate.mjs:1318`) — over the same per-model usage rows aggregate() reads. + (`usage-aggregate.mjs:1473`) — over the same per-model usage rows aggregate() reads. * **Aggregate-side incidents** (the v4/v5 cache bumps, the Codex parsing defects) are recorded in `usage-scorecard-metrics.md` Appendix A. diff --git a/docs/usage-scorecard-metrics.md b/docs/usage-scorecard-metrics.md index ed402269..b5da1724 100644 --- a/docs/usage-scorecard-metrics.md +++ b/docs/usage-scorecard-metrics.md @@ -52,6 +52,98 @@ this document to check an arithmetic claim. --- +## Current accounting and cache contracts + +Updated 2026-09-29 against the delivered V6 source. Older dated reproductions later in this +reference remain historical measurements. These contracts qualify the formulas below. + +**Classification and counting.** [ADR-0060](adr/0060-session-surface-initiator-and-product-names.md) +separates Git scope, host, surface, initiator and provider. The shared vocabulary uses declared +fields, never folder location or Codex `source="vscode"` alone. Observed metadata can identify +an assistant model's provider family or a recorded provider ID, but neither is network +attestation. Price-table identity and current configuration cannot establish historical serving +provider. Auto-review models without supported rates contribute tokens and unpriced coverage. +Census `countBasis` and source bounds remain distinct from Usage's billed-session rules; +zero-weight project recovery is not a verified session. Dedicated Cowork storage is uncovered. + +**Codex accounting.** Copied turns are excluded individually; proven native intervals in mixed +files remain eligible. Incomplete ownership, malformed mixed input and unseparable replay remain +conservatively excluded with diagnostics. Discovery reports pure exclusions, mixed observations +and unresolved bounded observations separately. A native record with positive input/cache/output +components can count even with zero responses. A total-only counter cannot establish that split +or a price; `total-only-token-count` reports the gap. Child-owned usage still counts after parent +replay exclusion. Effort and host-reported first-token timing are separate observations from +completion latency. Compactions expose a lower bound and nullable upper bound; missing pairing +never becomes an invented exact count. See [ADR-0052](adr/0052-codex-usage-attribution.md). + +**Claude charge ownership.** `usage-claude-dedup.mjs` reconciles bounded, hashed API message +identities across files after discovery. One owner is elected in a fixed pool covering twice the +displayed day window, capped at 730 days, before projecting either the current or previous window. +Both file mtime and session end must reach the pool cutoff. Requesting a previous comparison or +more historical lookback cannot make an eligible copied message charge twice or change its owner. +This is not whole-corpus uniqueness: source health qualifies outside-pool history and incomplete +identity coverage. Transcript-local `responses`, context samples and tools retain their local +meaning; `accountedResponses` records the aggregate charge share. The extra cold-read horizon is +a bound, not a measured performance guarantee. + +Claude's latest valid cumulative `cost-state` is a separate host-reported reconciliation signal, +never an addition to message-derived cost. Bounded model keys and timestamp/token-scope diagnostics +identify supported differences. A missing checkpoint end and missing serving-provider evidence +prevent an equality claim even when counters agree. Record coverage separately counts known +handled, known ignored, unknown, invalid-type and malformed JSON records. Unknown/malformed +records degrade source health without rewriting valid message cost. + +**OpenCode cost and source.** Positive reported cost remains observed. A reported zero with +positive tokens from a nonlocal or unknown provider is unpriced, rather than evidence of a free +request; a recognized local provider's zero retains its observed meaning. Missing-cost estimation +and unpriced coverage remain separate. No recorded or estimated amount is an invoice. + +The shared source resolver serves Usage, selected-session detail and project discovery. An +explicit root takes precedence; `OPENCODE_DB` can select an absolute path or a path relative to +the OpenCode data root. `:memory:` opens nothing. `OPENCODE_DISABLE_CHANNEL_DB=1` or `true` selects +the main database. Otherwise bounded discovery requires one canonical candidate: multiple +main/channel candidates require explicit selection, without a silent main/mtime preference or +combining copied stores. Canonical path identity gates cache and memo reuse; it is not a database +content digest. Conventional footprint storage sizing is a separate measurement boundary. + +V2 `session_message` presence and legacy JSON storage are observed as unsupported coverage, +without reading those formats into usage. Unknown observation scope is explicit. Those warnings +refresh on warm scans and with no current-window records: source availability and completeness +are independent of current-window usage, and presence does not prove lost current-window usage. + +OpenCode compaction evidence pairs requests and completed summaries and preserves failed, +pending, orphaned and in-flight uncertainty. Session counter reconciliation is supported only +for recorded version **1.18.33**, no V2 scope, complete valid messages, trusted costs and exactly +one matching `step-finish` per assistant message. Other versions, multiple steps, unpopulated +counters or unproved scope report unknown; session counters never add a second charge. + +Warm reuse requires `cost-trust-v2-observations-v1`, source identity, calendar compatibility and a +matching `observationFingerprint`. The fingerprint comes from bounded allowlisted session, +message and part observations in a read-only snapshot; it detects relevant same-count rewrites +and removals without hashing prompt bodies or persisting their raw fields. Missing digests, +budget refusal or read failures do not authorize normal reuse. Compatible last-good records may +still be retained under explicit degraded-source semantics. These digest probes add warm-scan +work; no performance improvement is claimed. + +OpenCode sessions with a nonempty `parent_id` keep their prompts/turns, tokens and cost, but their +user turns do not enter typed-prompt fingerprints. Aggregate session prompt metrics, current +prompt patterns and historical baselines also ignore old cached child fingerprints. Those cached +bytes can remain until normal invalidation; they do not force another global schema bump. + +**Cache compatibility.** Usage schema changes exactly **25 → 26**; footprint stays **8**. A v25 +usage cache requires rebuilding. Current v26 entries also need each source's semantic markers, +identity/coverage fields and calendar context; a valid schema number alone does not authorize +reuse. Local calendar context includes the resolved full timezone plus Node tzdata and ICU +versions. Missing, invalid or changed context reparses available sources; incompatible degraded +OpenCode records cannot contribute cached day/punchcard rows and are counted in +`timezoneCacheEntriesExcluded`. Unresolved zones decline cache and aggregate-memo reuse. +Local-day rebucketing can change a dated API-equivalent rate; it does not establish a provider's +billing timezone. Legacy surface fallback remains explicitly labeled, not precise new evidence. + +Implementation boundaries: `session-surface.mjs`, `usage-index.mjs`, `usage-claude-dedup.mjs`, +`usage-opencode-source.mjs`, `usage-opencode-storage-coverage.mjs`, +`usage-opencode-observations.mjs`, `usage-opencode-cache.mjs` and `usage-aggregate.mjs`. + ## 0. How to read an entry Every metric section below follows the same shape: @@ -71,9 +163,10 @@ Every metric section below follows the same shape: ## 1. Data provenance Two JSONL transcript stores plus OpenCode's SQLite store, read without source edits — the derived -record is cached "keyed by (path, mtime, size)" (`src/lib/usage-index.mjs:10`), +record is cached using source identity, file metadata, calendar context and source-specific +semantic evidence (see the current contracts above), and the whole cache is invalidated on a `SCHEMA_VERSION` change -(`usage-index.mjs:177`). A main Claude session's entry is additionally keyed on its statusline +(`usage-index.mjs:198`). A main Claude session's entry is additionally keyed on its statusline window ledger's own mtime and size, so a ledger that appears or changes re-parses the session: | Transcript host | Store | Format | @@ -84,22 +177,22 @@ window ledger's own mtime and size, so a ledger that appears or changes re-parse | OpenCode | platform data root `opencode/opencode.db` | SQLite session/message/part rows; per-assistant token fields and optional recorded cost/provider identity | Discovery is **one level of project directories plus that one nested shape**, -not a recursive walk: `listClaude` (`usage-index.mjs:374-388`) descends into a +not a recursive walk: `listClaude` (`usage-index.mjs:460-474`) descends into a session-id directory only through "listClaudeSubagents" -(`usage-index.mjs:349-354`), which reads exactly +(`usage-index.mjs:435-440`), which reads exactly `//subagents/*.jsonl`. A directory that is not a session-id dir with a `subagents` child — Claude Code's own `memory` dir, say — contributes nothing rather than being crawled. Each subagent record takes a -**namespaced** id, `/` (`usage-index.mjs:354`), because Claude +**namespaced** id, `/` (`usage-index.mjs:440`), because Claude Code names every such file `agent-.jsonl` and that stem is not guaranteed unique across two parents; an unnamespaced id would silently collide two unrelated subagent records into one. `locateSubagent` -(`usage-index.mjs:1015-1034`) resolves that id back to the nested path when a +(`usage-index.mjs:1211-1230`) resolves that id back to the nested path when a reader opens the session, building the candidate path from the two validated capture groups rather than from raw request input. -The parsers are `parseClaude` (`usage-parsers.mjs:742-777`) and `parseCodex` -(`usage-parsers.mjs:1183-1240`). They normalize raw JSONL bytes; project evidence also consults the local filesystem. +The parsers are `parseClaude` (`usage-parsers.mjs:817-871`) and `parseCodex` +(`usage-parsers.mjs:1355-1442`). They normalize raw JSONL bytes; project evidence also consults the local filesystem. Missing-time fallback paths can consult the clock. Their output is local evidence, not a network response or an invoice. Nothing in this transcript pipeline calls a provider API or a billing endpoint; **no transcript @@ -196,22 +289,27 @@ turns`. **Formula:** ```text -sessions = count of session records with responses > 0 AND end >= cutoff -responses = Σ over included sessions of session.responses +eligible = responses > 0 OR positive Codex component usage OR retained OpenCode observations +current.sessions = count of eligible records with non-null end >= cutoff (no upper bound) +windowStart = now - days × DAY_MS +previous.sessions = count of eligible records with windowStart - days × DAY_MS <= end < windowStart +responses = Σ over included sessions of accountedResponses (else responses) ``` **Source:** -- Filter: a parsed record with zero assistant turns is dropped entirely — "no - assistant turn → not a session" (`usage-aggregate.mjs:888-899`) — and a record whose - last activity falls outside the requested window is dropped too - (`usage-aggregate.mjs:898-899`). +- Filter: `buildSessionRows` (`usage-aggregate.mjs:970-982`) accepts response-bearing records, + positive Codex component usage or retained OpenCode observations. An unknown end or an end + outside the requested window is excluded. Refused OpenCode acquisitions retain uncertainty + separately without becoming ordinary session rows. The current projection supplies no + upper bound; `previousWindow` (`usage-aggregate.mjs:1274-1281`) supplies the exclusive + `windowStart` upper bound and derives both bounds from displayed `days` and `now`. - `responses` accumulation: Claude increments once per API message id — every transcript line of one message counts once, the last line's usage winning - (`usage-parsers.mjs:661-696`); Codex increments per `agent_message` event - (`usage-parsers.mjs:1021-1025`). -- Totals: `totals.responses += s.responses` per included session -(`usage-aggregate.mjs:928`). + (`usage-parsers.mjs:708-753`); Codex increments per `agent_message` event + (`usage-parsers.mjs:1143-1147`). +- Totals: `totals.responses += s._accountedResponses` per included session +(`usage-aggregate.mjs:1050`). - Render: `kpi("sessions", fmtNum(t.sessions), fmtNum(t.responses)+" assistant turns", "")` (`dashboard/client.mjs`). @@ -332,22 +430,22 @@ same sentence fingerprints identically whichever host recorded it: | Symbol | Location | Notes | |---|---|---| -| `normalizePromptText` | `src/lib/usage-parsers.mjs:261` | lowercased, whitespace-collapsed, trailing punctuation stripped | -| `promptFingerprint` | `src/lib/usage-parsers.mjs:296` | the `{h, t, th}` hash/count/token-hash triple | -| `promptShape` | `src/lib/usage-parsers.mjs:342` | the `q`/`o` flags, anchored on the question and persona-opener rules below | -| `QUESTION_WH_RE` | `src/lib/usage-parsers.mjs:311` | one of two rules the `q` flag checks | -| `QUESTION_AUX_RE` | `src/lib/usage-parsers.mjs:312` | the other | -| `PERSONA_OPENER_RE` | `src/lib/usage-parsers.mjs:318` | what the `o` flag checks | -| `notePromptFingerprint` | `src/lib/usage-parsers.mjs:359` | records one fingerprint, or counts overflow past the caps below | -| `MAX_PROMPT_FPS` | `src/lib/usage-parsers.mjs:240` | the per-session fingerprint cap | -| `MAX_TOKEN_HASHES` | `src/lib/usage-parsers.mjs:255` | the per-fingerprint token-hash cap | +| `normalizePromptText` | `src/lib/usage-parsers.mjs:286` | lowercased, whitespace-collapsed, trailing punctuation stripped | +| `promptFingerprint` | `src/lib/usage-parsers.mjs:321` | the `{h, t, th}` hash/count/token-hash triple | +| `promptShape` | `src/lib/usage-parsers.mjs:367` | the `q`/`o` flags, anchored on the question and persona-opener rules below | +| `QUESTION_WH_RE` | `src/lib/usage-parsers.mjs:336` | one of two rules the `q` flag checks | +| `QUESTION_AUX_RE` | `src/lib/usage-parsers.mjs:337` | the other | +| `PERSONA_OPENER_RE` | `src/lib/usage-parsers.mjs:343` | what the `o` flag checks | +| `notePromptFingerprint` | `src/lib/usage-parsers.mjs:384` | records one fingerprint, or counts overflow past the caps below | +| `MAX_PROMPT_FPS` | `src/lib/usage-parsers.mjs:265` | the per-session fingerprint cap | +| `MAX_TOKEN_HASHES` | `src/lib/usage-parsers.mjs:280` | the per-fingerprint token-hash cap | | `PROVENANCE_TAGS` | `src/lib/usage-provenance.mjs:21` | the closed four-tag vocabulary | | the ordered provenance rules | `src/lib/usage-provenance.mjs:33-77` | matched against, in order, to resolve a tag | | `provenanceOf` | `src/lib/usage-provenance.mjs:93` | resolves one turn's provenance tag | Wired on the Claude path where userTurnKind is called — -`src/lib/usage-parsers.mjs:600-621`; on the Codex path inside -`handleCodexUserMessage` — `src/lib/usage-parsers.mjs:985-1006`; on the +`src/lib/usage-parsers.mjs:635-656`; on the Codex path inside +`handleCodexUserMessage` — `src/lib/usage-parsers.mjs:1107-1128`; on the opencode path inside `recordUserMessage` — `src/lib/usage-opencode.mjs:184-197` **What this does not model:** @@ -455,12 +553,12 @@ this day carried the fingerprint layer", so a zero here is *measured*. | Symbol | Location | |---|---| -| `TAP_MAX_TOKENS` | `src/lib/usage-aggregate.mjs:287` | -| the baseline window and floor | `src/lib/usage-aggregate.mjs:294` | -| `v16Projection` (per session) | `src/lib/usage-aggregate.mjs:330` | -| `foldSessionPrompts` | `src/lib/usage-aggregate.mjs:369` | -| `sealPromptHosts` | `src/lib/usage-aggregate.mjs:387` | -| `buildPromptBaselines` | `src/lib/usage-aggregate.mjs:417` | +| `TAP_MAX_TOKENS` | `src/lib/usage-aggregate.mjs:290` | +| the baseline window and floor | `src/lib/usage-aggregate.mjs:297` | +| `v16Projection` (per session) | `src/lib/usage-aggregate.mjs:333` | +| `foldSessionPrompts` | `src/lib/usage-aggregate.mjs:374` | +| `sealPromptHosts` | `src/lib/usage-aggregate.mjs:392` | +| `buildPromptBaselines` | `src/lib/usage-aggregate.mjs:422` | | `detectSupervisionTapShare` | `src/lib/usage-insights.mjs:780` | | `detectHeadlessShare` | `src/lib/usage-insights.mjs:808` | | `detectHostPromptAsymmetry` | `src/lib/usage-insights.mjs:845` | @@ -557,7 +655,8 @@ call has no per-token rate, and the unknown-model fallback would invent one (1M input + 1M output tokens would read $18). Those messages count toward `unpricedMessages`, add no dollars, and are left out of the cache-saving estimate; the gap is coverage the reader can see, not a silent $0. A local -provider that reports a cost, including 0, stays an observed figure. A +provider that reports a cost, including 0, stays an observed figure. For nonlocal or unknown +providers, reported zero with positive tokens is unpriced. A custom-named local provider cannot be recognised from its id and keeps the fallback rate. @@ -598,7 +697,7 @@ lexicographically so no `Date` parsing is involved and the module stays clock-free. `foldSessionUsageRow` passes each usage row's own `day` to `costOf` -(`usage-aggregate.mjs:739-741`), which +(`usage-aggregate.mjs:746-748`), which it already has because rows are keyed by `(day, model)`. **This is the whole point:** tokens metered in August must still read as August's rate when the panel is opened in December. Pricing by *today's* date instead would restate a @@ -668,9 +767,9 @@ tokens = input + output + cacheRead + cacheWrite (summed across all rows in wi ``` **Source:** `t.tokens` from `totals`, accumulated per row at -`usage-aggregate.mjs:765` (`rowTokens = row.input + row.output + row.cacheRead + +`usage-aggregate.mjs:775` (`rowTokens = row.input + row.output + row.cacheRead + row.cacheWrite`) and rolled into `totals.tokens` via `addTo` -(`usage-aggregate.mjs:656-665`). Rendered with `fmtTok()` +(`usage-aggregate.mjs:663-672`). Rendered with `fmtTok()` (`dashboard/client.mjs`): `≥1e9` → `"X.XB"`, `≥1e6` → `"X.XM"`, `≥1e3` → `"X.XK"`, else the rounded integer. @@ -684,7 +783,7 @@ per row is **gross input minus cached input** — Claude's parser reads `cache_read_input_tokens` and `cache_creation_input_tokens` as separate fields the provider already reports separately (`telemetry-records.mjs:216-224`); Codex's parser subtracts `cached_input_tokens` from `input_tokens` explicitly -at this call site (`usage-parsers.mjs:1132-1183`, `input: Math.max(0, gross - cacheRead)`) because +at this call site (`usage-parsers.mjs:1296-1355`, `input: Math.max(0, gross - cacheRead)`) because Codex's own `input_tokens` field **includes** cached tokens and would double-count them against the separately-reported `cacheRead` figure if left as-is. This is asserted by test: @@ -775,25 +874,25 @@ session data, and each needs its own fix: human, or genuinely idle) donates its *entire* idle stretch to the span, even though no work happened during it. Fix: split each session into active sub-intervals wherever the gap between two consecutive timestamps - exceeds `IDLE_GAP_MS` (15 minutes, `usage-parsers.mjs:29`), then union + exceeds `IDLE_GAP_MS` (15 minutes, `usage-parsers.mjs:32`), then union *those* sub-intervals — this is `engagedSeconds`. **Source:** -- `mergeIntervals()` (`usage-aggregate.mjs:40-65`) — the pure union primitive, +- `mergeIntervals()` (`usage-aggregate.mjs:43-68`) — the pure union primitive, sorts intervals and merges any two that are "overlapping OR exactly touching" - (`s <= curEnd`, `usage-aggregate.mjs:56`), returning total covered seconds + (`s <= curEnd`, `usage-aggregate.mjs:59`), returning total covered seconds rounded to the nearest second. -- `activeIntervals()` (`usage-parsers.mjs:469-487`) — splits one session's +- `activeIntervals()` (`usage-parsers.mjs:494-512`) — splits one session's sorted timestamp list into sub-intervals wherever a gap exceeds IDLE_GAP_MS; "a run of one timestamp yields a zero-length interval and so - contributes nothing" (comment, `usage-parsers.mjs:469-475`). + contributes nothing" (comment, `usage-parsers.mjs:494-500`). - Aggregation, each its own call to `mergeIntervals`: `engagedSeconds` over - every session's active sub-intervals (`usage-aggregate.mjs:1064-1069`); + every session's active sub-intervals (`usage-aggregate.mjs:1174-1179`); `spanUnionSeconds` over whole spans instead - (`usage-aggregate.mjs:1044-1068`); spanMs is a running sum of - "s._span[1] - s._span[0]" across the loop (`usage-aggregate.mjs:963-981`), - finalized into `spanMinutes` (`usage-aggregate.mjs:1064-1068`). + (`usage-aggregate.mjs:1154-1178`); spanMs is a running sum of + "s._span[1] - s._span[0]" across the loop (`usage-aggregate.mjs:1045-1076`), + finalized into `spanMinutes` (`usage-aggregate.mjs:1174-1178`). - Render: `fmtHours()` (`dashboard/client.mjs`, `≥10h` rounds to the nearest hour, else one decimal place) and `fmtMins()` (`dashboard/client.mjs`, `≥60min` rounds to hours, else whole @@ -846,12 +945,12 @@ byDay[day].sessionsActive = count of distinct sessions with any usage row that d **Source:** the day key is the row's own `row.day`, computed once at parse time as **local calendar day**, not UTC -(`usage-parsers.mjs:35`/`usage-parsers.mjs:1174` call `localDay(at)`) — so a +(`usage-parsers.mjs:38`/`usage-parsers.mjs:1332` call `localDay(at)`) — so a session that runs from 23:58 local to 00:05 local has its session count attributed to the day its *first* usage row landed on (test: `tests/kit/usage-index.test.mjs:738`, "a session that opens before midnight is counted on its first billed day"). Accumulation, at this call: `dayBucket(byDay, -row.day)` then `d.cost = round(d.cost + rowCost)` (`usage-aggregate.mjs:764-771`). Bar height: +row.day)` then `d.cost = round(d.cost + rowCost)` (`usage-aggregate.mjs:774-781`). Bar height: `h = maxDay ? max(2, cost/maxDay*100) : 2` (`dashboard/client.mjs`) — every non-empty day gets a visually nonzero bar (floor of 2%), so a very cheap day is never rendered as invisible. @@ -879,11 +978,11 @@ renders "no sessions in window" instead of zeroed figures (`dashboard/client.mjs`). **Formula:** identical aggregation to every other bucket -(`byHost[s.host]`, populated via `addTo()` (`usage-aggregate.mjs:667-676`), - called once per session at this call: `usage-aggregate.mjs:968-993`), keyed by the literal string +(`byHost[s.host]`, populated via `addTo()` (`usage-aggregate.mjs:674-683`), + called once per session at this call: `usage-aggregate.mjs:1050-1103`), keyed by the literal string `"claude"` or `"codex"` assigned at parse time (this call: `blankSession(id, 'claude')` / `blankSession(id, 'codex')`, -`usage-parsers.mjs:197-225`, `:1220`, `parseClaude`/`parseCodex` entry points). +`usage-parsers.mjs:216-250`, `:1383`, `parseClaude`/`parseCodex` entry points). OpenCode's SQLite reader builds the same record shape and contributes a third host key. @@ -895,22 +994,27 @@ The historical Codex incidents in Appendix A are examples, not an exhaustive dia **Two identity maps, with separate evidence.** The aggregate buckets window spend by two identities, and reading one as the other is the -mistake this split exists to prevent (`usage-aggregate.mjs:913`, -`usage-aggregate.mjs:941-942`): +mistake this split exists to prevent (`usage-aggregate.mjs:994`, +`usage-aggregate.mjs:1022-1023`): - **`byHost`** — the execution host: which CLI wrote the transcript (`claude`, `codex`, `opencode`). This is what the host cards render. It is a fact about the file's provenance on disk, and it proves nothing about which vendor served the tokens. -- **`byProvider`** — the inference-provider string **as recorded**, ungated: - `s.provider ?? 'unknown'`. This map keeps its historical name and its - historical shape for callers that want the raw string, whatever its - evidence. A session that recorded no provider keys to `'unknown'`. - -`byProvider` uses the served session row's recorded inference provider, or `unknown`. -Codex session metadata/turn context and OpenCode assistant `providerID` can establish -that value with observed provenance; Claude history normally leaves it absent. The -Scorecard UI does not rank this map, but session details expose provider/provenance. +- **`byProvider`** — OpenCode response, token and cost totals are split by the provider on each + assistant usage row, with missing row providers in `unknown`. `foldSessionUsageRows` + (`usage-aggregate.mjs:791-815`) accumulates these per-provider shares; the second pass applies + them at this call: `addTo(bucket(byProvider, provider), { ...s, ...usage })` + (`usage-aggregate.mjs:1083-1092`). Each response/token/cost share lands once, but one session + counts once under **each** provider it used. Provider session counts therefore need not sum + to the overall session count; they are not disjoint session populations. + +When there are no per-row provider shares, the fallback uses the session's recorded provider, +`s.provider ?? 'unknown'`, with its accounted response count at this call: `addTo` +(`usage-aggregate.mjs:1091`). +Codex session metadata/turn context and OpenCode assistant `providerID` supply observed metadata; +Claude history normally leaves the session provider absent. These facts are not network +attestation. The Scorecard UI does not rank this map, but session details expose provider/provenance. The source's former parser field is retained separately as `transcriptProvider`. **What this does not model:** a workflow that hands off between Claude and @@ -935,9 +1039,9 @@ punchcard[dow + "-" + hour] += 1 per assistant/agent_message response, at its **Source:** incremented once per Claude API message (all of a message's transcript lines are one hit) -(`usage-parsers.mjs:42`, keyed by this call: `punchKey(at)`) and once per Codex -`agent_message` (`usage-parsers.mjs:1021-1025`), merged into the window-level -`punchcard` object per session (`usage-aggregate.mjs:943-1007`). Cell intensity is +(`usage-parsers.mjs:45`, keyed by this call: `punchKey(at)`) and once per Codex +`agent_message` (`usage-parsers.mjs:1143-1147`), merged into the window-level +`punchcard` object per session (`usage-aggregate.mjs:1024-1117`). Cell intensity is linear against the single busiest cell in the window: `v = pcMax ? n/pcMax : 0` (`dashboard/client.mjs`) — this is a **relative**, not absolute, scale, so the heatmap's brightest cell is always @@ -979,9 +1083,9 @@ byModel[model].sessions = count of DISTINCT sessions whose s.models includes th ``` **Source:** cost/tokens/responses accumulate inside the usage-row loop -(`usage-aggregate.mjs:752-779`); the per-model session count is deliberately computed +(`usage-aggregate.mjs:762-790`); the per-model session count is deliberately computed **separately**, once per session over its `s.models` array -(`usage-aggregate.mjs:910-919`) rather than inside the cost loop, precisely +(`usage-aggregate.mjs:991-1000`) rather than inside the cost loop, precisely **so that a model can appear in `byModel` — with a nonzero session count — even in a session that contributed zero cost/tokens/responses for that model.** This is not an edge case invented for this document: it is the @@ -991,11 +1095,11 @@ excluded subagent-replay session still shows up as "used," at zero cost, rather than vanishing. `byModel[...].responses` is populated from each usage row's response field at -this call (`usage-aggregate.mjs:775-778`). The shared usage-row accumulator is -defined at `usage-parsers.mjs:499-515`; Claude passes one response per API -message at its call site (`usage-parsers.mjs:677`). Codex passes the session's +this call (`usage-aggregate.mjs:786-789`). The shared usage-row accumulator is +defined at `usage-parsers.mjs:524-550`; Claude passes one response per API +message at its call site (`usage-parsers.mjs:723`). Codex passes the session's whole response count once, where finalizeCodexUsage makes the corresponding -call (`usage-parsers.mjs:1136-1179`). +call (`usage-parsers.mjs:1303-1337`). The two parsers therefore hand the aggregate the same response-bearing row shape, despite their different per-turn and cumulative transcript formats. @@ -1014,12 +1118,12 @@ split `server_error` 27, `authentication_failed` 3, `rate_limit` 3 — three distinct underlying causes, one placeholder shape). The parser recognizes the decoded API-error placeholder and returns before model -or usage attribution (`usage-parsers.mjs:701-720`). The turn does **not** increment -the response count or punchcard — it is not a model response (`usage-parsers.mjs:661-696`) — it *is* real +or usage attribution (`usage-parsers.mjs:758-777`). The turn does **not** increment +the response count or punchcard — it is not a model response (`usage-parsers.mjs:708-753`) — it *is* real engaged time (its timestamp still extends the session span), someone was genuinely waiting on it — and increments the record's exception count instead. Aggregation rolls that count into the window -total (`usage-aggregate.mjs:968-979`) and keeps it on the session row beside the -delegation-source fields (`usage-aggregate.mjs:838-867`), so it remains +total (`usage-aggregate.mjs:1050-1074`) and keeps it on the session row beside the +delegation-source fields (`usage-aggregate.mjs:909-944`), so it remains inspectable in Sessions without creating a fake model row. When `totals.exceptions > 0`, the panel header shows a small `"· N dropped/errored turns excluded"` note (`dashboard/client.mjs`); @@ -1378,7 +1482,7 @@ both credential-free for ak: `windowDurationMins: 10080` (the weekly). Windows are therefore keyed and labelled by duration (`windowLabel`, `quota.mjs:52`), never by slot name. The same rule applies to the historical snapshots parsed out of rollouts: the -normalizer at `usage-parsers.mjs:866-884` keeps a flat `windows` list keyed by +normalizer at `usage-parsers.mjs:971-989` keeps a flat `windows` list keyed by `window_minutes`. **Freshness is part of the number.** Both sides carry `fetchedAt`; the view @@ -1424,7 +1528,7 @@ Codex ≥0.140 maintains its own SQLite thread ledger (`~/.codex/state_N.sqlite` globs and takes the newest). `readCodexState` (`:62`, whose own delegate call reads the db file) reads per-thread `thread_source` (`user` vs `subagent`) plus `thread_spawn_edges`, and -`applyCodexLedger` (`usage-aggregate.mjs:1280-1310`) overlays that onto parsed +`applyCodexLedger` (`usage-aggregate.mjs:1392-1465`) overlays that onto parsed sessions: a thread that ONLY the ledger identifies as a subagent has its token usage stripped — with no `thread_source` in its own rollout its parsed usage is the unsubtracted cumulative total, which replays the parent's entire token history @@ -1432,7 +1536,7 @@ unsubtracted cumulative total, which replays the parent's entire token history visible. A rollout that says `thread_source: subagent` itself is left untouched: the parser already reduced it to the subagent's own usage (§16.2). **The parser is primary, the ledger is the fallback**: `rec.threadSource ?? -t?.threadSource ?? fromEdges` (`usage-aggregate.mjs:1280`) reads the rollout's +t?.threadSource ?? fromEdges` (`usage-aggregate.mjs:1392`) reads the rollout's own `session_meta.thread_source` first, and only consults the ledger when that line is missing entirely. This is sound because `thread_source` is now the FIRST session_meta line's value (§1.2) rather than whichever meta @@ -1443,7 +1547,7 @@ migration generation its `N` reflects; a rollout the ledger cannot resolve (an older Codex build, a migrated-beyond-recognition state file) still gets a correct `threadSource` straight from its own transcript rather than falling through unclassified. Codex sessions also carry -`reasoningOutput` (`usage-parsers.mjs:1136-1183`) — reasoning tokens are a **subset** +`reasoningOutput` (`usage-parsers.mjs:1303-1355`) — reasoning tokens are a **subset** of output tokens and are annotation only, never added to any sum. ## 14. Known limitations, restated as a single checklist @@ -1468,7 +1572,7 @@ the same list: - [x] A percentile taken from the overflow bucket of a histogram is printed with `≥`, and an unmeasured one is `null` rather than `0` (§15). - [x] A latency figure is never called TTFT: it is a prompt-to-answer gap or - a host-measured turn duration, and neither transcript records TTFT (§15). + a host-measured turn duration. Codex host-reported first-token evidence is a separate metric (§15). - [x] Permission posture keeps `not-recorded` as a first-class bucket — unmapped evidence is never folded into a real posture — and the inference provider is separately reported when native evidence exists (§8, §16). @@ -1509,23 +1613,23 @@ p(q), over N samples, landing in bucket i (count n_i, running total `cum` before **Source:** -- Edges: `LAT_BUCKET_EDGES` and `LEN_BUCKET_EDGES` (`usage-aggregate.mjs:212`, `:215`). - The parsers carry their own copies (`usage-parsers.mjs:365-370`) and the +- Edges: `LAT_BUCKET_EDGES` and `LEN_BUCKET_EDGES` (`usage-aggregate.mjs:215`, `:215`). + The parsers carry their own copies (`usage-parsers.mjs:390-395`) and the browser bundle a third pair (`LAT_EDGES`/`LEN_EDGES`), because the payload ships bucket *counts* and never the edges they were binned on. -- Slotting: `bucketIndex` (`usage-parsers.mjs:379-381`) — one definition of a +- Slotting: `bucketIndex` (`usage-parsers.mjs:404-406`) — one definition of a boundary, shared by every histogram built on these edges. -- Sampling: `noteLatencySample` (`usage-parsers.mjs:372-377`) allocates +- Sampling: `noteLatencySample` (`usage-parsers.mjs:397-402`) allocates `latHist` lazily, so a session that never observed a latency keeps `latHist: null` — absent, not a fabricated row of zeroes. - Session length: `seal` derives each session's `lenSeconds` from its own - active intervals (`usage-parsers.mjs:489-496`) — the §6 engaged figure for + active intervals (`usage-parsers.mjs:514-521`) — the §6 engaged figure for one session, never its first-to-last span. This is a per-session parse result, kept distinct from the next window-level fold. -- Window merge: `buildRhythm` (`usage-aggregate.mjs:1084-1109`) adds the +- Window merge: `buildRhythm` (`usage-aggregate.mjs:1194-1219`) adds the per-session `latHist` slot-wise and buckets each session's `lenSeconds`. -- Percentiles: `percentileFromBuckets` (`usage-aggregate.mjs:241-258`). The +- Percentiles: `percentileFromBuckets` (`usage-aggregate.mjs:244-261`). The browser re-implementation `bucketPercentile` (`usage-rhythm.mjs:106-126`) is pinned to byte-identical output, and the browser's edge copies to the server constants, by `tests/kit/dashboard-usage-telemetry.test.mjs:924-942` and @@ -1558,14 +1662,14 @@ median target 50 lands in bucket 1 (running total 40, n = 25), giving **Overflow floors, and why `≥` is not decoration.** The last bucket of either histogram has no upper edge to interpolate towards, so a percentile landing in it reports that bucket's **floor** and nothing more — -`if (i >= edges.length) return round(lo, 2)` (`usage-aggregate.mjs:252`). +`if (i >= edges.length) return round(lo, 2)` (`usage-aggregate.mjs:255`). A p95 printed as `≥60s` therefore means *at least 60 seconds* — the counts cannot say whether the real figure is 61 seconds or 61 minutes, and printing a bare `60s` would state a precision they do not carry. Both renderers apply the prefix by the same rule (`v >= lastEdge`), so a value that reaches the last edge by interpolation and one that came from the overflow slot print identically — the two are the same claim. An empty histogram is `null`, never -`0` (`usage-aggregate.mjs:244`): "nothing was measured" and "measured zero" are +`0` (`usage-aggregate.mjs:247`): "nothing was measured" and "measured zero" are different statements and only the first is true, so the cards print `not measured` and the CLI prints `no samples`. @@ -1575,13 +1679,13 @@ same way, and the panel says so rather than implying a single clock: | Host | How a latency sample is produced | |---|---| -| codex | **Host-measured.** `task_started` remembers the turn's start (`usage-parsers.mjs:915-934`) and `task_complete` samples Codex's own `duration_ms` (`usage-parsers.mjs:936-954`) — but only if no prompt-gap already covered that turn (so a turn is never sampled twice) and only within the same 3600 s cap the derived paths apply. | -| codex | Also derives a prompt-gap when one is available: `handleCodexUserMessage` opens the window (`usage-parsers.mjs:976-1006`) and the next agent message closes it, clearing `turnStartedAt` so the `duration_ms` fallback cannot double-fire (`usage-parsers.mjs:1008-1019`). | -| claude | **Derived from event gaps.** A human prompt sets `latState.pendingMs` (`pendingMs`, `usage-parsers.mjs:593-608`); the first real assistant turn closes that gap into a `noteLatencySample` call (`usage-parsers.mjs:722-725`). | +| codex | **Host-measured.** `task_started` remembers the turn's start (`usage-parsers.mjs:1028-1047`) and `task_complete` samples Codex's own `duration_ms` (`usage-parsers.mjs:1049-1076`) — but only if no prompt-gap already covered that turn (so a turn is never sampled twice) and only within the same 3600 s cap the derived paths apply. | +| codex | Also derives a prompt-gap when one is available: `handleCodexUserMessage` opens the window (`usage-parsers.mjs:1098-1128`) and the next agent message closes it, clearing `turnStartedAt` so the `duration_ms` fallback cannot double-fire (`usage-parsers.mjs:1130-1141`). | +| claude | **Derived from event gaps.** A human prompt sets `latState.pendingMs` (`pendingMs`, `usage-parsers.mjs:628-643`); the first real assistant turn closes that gap into a `noteLatencySample` call (`usage-parsers.mjs:779-782`). | | opencode | Derived from its message stream, measured to **completion**: `rec.pendingPromptMs` is the user message's `time.created`, and the first assistant row closes it at that row's `time.completed` (`closeLatencyWindow`, `usage-opencode.mjs:318-324`) — OpenCode inserts the assistant row ~15 ms after the prompt and fills it in as it generates, so its own `time.created` is not a response time. A row with no completed stamp yields no sample. | **Every** path is capped: a sample above `MAX_LATENCY_SAMPLE_SECONDS` -(3600 s, `usage-parsers.mjs:445-460`) is an idle resume — the person walked away +(3600 s, `usage-parsers.mjs:470-485`) is an idle resume — the person walked away and came back — not a wait for a reply, so it is dropped from sampling entirely rather than parked in the overflow bucket beside genuinely slow turns. That includes Codex's host-measured `duration_ms`. An earlier ruling exempted @@ -1594,11 +1698,11 @@ reference corpus before the fix: 12 of 835 durations exceeded the cap, the largest 94,079,450 ms ≈ 26.1 hours, all of them landing in the `≥60s` overflow bucket and dragging `latP95` into it. An interrupted turn contributes nothing at all — `turn_aborted` clears both -pending states (`usage-parsers.mjs:1048-1089`), so a prompt that was never +pending states (`usage-parsers.mjs:1170-1218`), so a prompt that was never answered can never be timed against a later, unrelated reply. A dropped API turn is likewise never a sample: the error branch returns before the latency block and deliberately leaves `pendingMs` set, so the first real completion -that eventually follows is what gets timed (`usage-parsers.mjs:601-610`). +that eventually follows is what gets timed (`usage-parsers.mjs:636-645`). **This figure is never labeled TTFT, in any surface.** Time-to-first-token measures when a stream *starts*; every figure here measures when a turn @@ -1611,7 +1715,8 @@ TTFT" beside the per-host note it qualifies. A true TTFT exists for Claude Code, but only as a span in its opt-in OpenTelemetry beta ([monitoring](https://code.claude.com/docs/en/monitoring-usage)) — a different, non-transcript evidence class that this scorecard does not read. -Neither transcript store records it, so no panel here may borrow the name. +Codex can separately record host-reported first-token timing. That evidence is retained +independently and does not rename or replace the completion-latency metric. **What this does not model:** @@ -1639,8 +1744,8 @@ Neither transcript store records it, so no panel here may borrow the name. means a window dominated by one host is really reporting that host's instrument; - the bucketing *function* is implemented twice: the parsers export - `bucketIndex` (`usage-parsers.mjs:379-381`) and the aggregate keeps a private - copy of the same loop (`usage-aggregate.mjs:222-225`), because the dependency + `bucketIndex` (`usage-parsers.mjs:404-406`) and the aggregate keeps a private + copy of the same loop (`usage-aggregate.mjs:225-228`), because the dependency between the two modules is deliberately one-way. The *edges* they run on are pinned equal by test — `AGG_LAT_EDGES` against `LAT_BUCKET_EDGES` (`tests/kit/usage-index.test.mjs:15-19`) — but the two function bodies @@ -1670,13 +1775,13 @@ mode = normalizeMode(host, raw evidence) or 'not-recorded' **Source:** `normalizeMode` (`usage-modes.mjs:23-35`) is the whole taxonomy; `MODES` (`usage-modes.mjs:4`) is the closed four-value vocabulary. Per-day folding is "addCost(d.byMode, rec.mode ?? 'not-recorded', rowCost)" -(`usage-aggregate.mjs:752-772`) — in the usage-row pass, because only a row knows +(`usage-aggregate.mjs:762-782`) — in the usage-row pass, because only a row knows which day its dollars landed on. The window bucket is -this call: `addTo(bucket(byMode, s.mode ?? 'not-recorded'), s)` (`usage-aggregate.mjs:983-993`). +this call: `addTo(bucket(byMode, s.mode ?? 'not-recorded'), s)` (`usage-aggregate.mjs:1093-1103`). The evidence each parser reads: Claude's `permissionMode`, off the human prompt -only (`usage-parsers.mjs:593-608`); Codex's `approval_policy`/`sandbox_policy` +only (`usage-parsers.mjs:628-643`); Codex's `approval_policy`/`sandbox_policy` off each `turn_context`, last one wins since a session may renegotiate mid-run -(`usage-parsers.mjs:843-864`); OpenCode's `mode` off each assistant message +(`usage-parsers.mjs:943-964`); OpenCode's `mode` off each assistant message (`usage-opencode.mjs:344-345`). Render is `modeChart` in `src/lib/dashboard/client/usage.mjs`; the CLI table is `printScoreModeTable` (`src/commands/usage.mjs:270-272`). @@ -1708,7 +1813,7 @@ or `{"type":"workspace-write", …}` with sibling fields such as `network_access` — never the bare string the taxonomy is written against. A survey of this machine's rollouts (400 files, 2026-08-28) found 1,110 object occurrences and **zero** string ones. `handleCodexTurnContext` -(`usage-parsers.mjs:843-864`) therefore reads `sandbox_policy.type` and passes +(`usage-parsers.mjs:943-964`) therefore reads `sandbox_policy.type` and passes that to `normalizeMode`, which is unchanged and still accepts the string form. Before this extraction the object reached `normalizeMode` intact, matched no rule, and stringified into `modeRaw` as `"never/[object Object]"`: the `plan`, @@ -1731,11 +1836,11 @@ second field. (`usage-modes.mjs:25`, `:32`), so an unrecognised raw value — a future `permissionMode`, a policy this taxonomy has not been taught — yields no mode. The raw string is kept beside the normalized one as `modeRaw` -(`usage-parsers.mjs:208`) precisely because the mapping is a judgement call and +(`usage-parsers.mjs:240`) precisely because the mapping is a judgement call and a reader checking it needs the evidence it was made from. `not-recorded` is a first-class bucket key rather than a display fallback, folded at this call: `addTo(bucket(byMode, s.mode ?? 'not-recorded'), s)` -(`usage-aggregate.mjs:983-993`), it is always offered as a row by the CLI table +(`usage-aggregate.mjs:1093-1103`), it is always offered as a row by the CLI table even at zero (`printBucketTable`, `src/commands/usage.mjs:257-264`), and `segColor` (`src/lib/dashboard/client/usage-rhythm.mjs:191-192`) forces it to the de-emphasis ink rather than letting a palette give @@ -1747,17 +1852,21 @@ evidence must never read as a posture. **Formula:** ```text -source = (session.sidechain || session.threadSource == 'subagent') - ? 'subagent' : 'main' +delegated = isSubagentSession(session) + OR session.threadSource IN ['guardian_review', 'agent_created_thread'] +source = delegated ? 'subagent' : 'main' bySource[k].cost = Σ over sessions with that source of session.cost centre of the donut = round(main / (main + subagent) × 100) % ``` -**Source:** `sourceKey` (`usage-aggregate.mjs:932-936`). Both rows are created, +**Source:** `sourceKey` (`usage-aggregate.mjs:1013-1017`) uses the shared +`isSubagentSession` predicate (`usage-context.mjs:20-23`) and explicitly includes guardian reviews +and agent-created threads. The same source classification gates `humanPrompts`: only main-session +prompts enter the autonomy denominator (`usage-aggregate.mjs:1051-1054`). Both rows are created, at this call to it, before the fold -("Both source rows always exist", `usage-aggregate.mjs:962-966`) so "no subagent sessions" renders +("Both source rows always exist", `usage-aggregate.mjs:1044-1048`) so "no subagent sessions" renders as a zero rather than a row the UI silently drops. Claude's evidence is the -`isSidechain` flag on any entry in the file (`usage-parsers.mjs:758-763`, decoded at +`isSidechain` flag on any entry in the file (`usage-parsers.mjs:842-847`, decoded at `telemetry-records.mjs:267`); Codex's is the ledger-backed `thread_source` (§13c). Render is `sourceDonut` in `src/lib/dashboard/client/usage.mjs`. @@ -1768,7 +1877,7 @@ replay cannot be separated reports none. A zero cannot establish whether actual **Claude — real, priced, included.** A session's delegated work is written to its own transcript under `//subagents/`, and those files are -discovered by `listClaudeSubagents` (`usage-index.mjs:349-354`, §1) and parsed +discovered by `listClaudeSubagents` (`usage-index.mjs:435-440`, §1) and parsed like any other. `parseClaude` already prices those bytes and marks the record `sidechain` from its own `isSidechain` entries, so the cost is real, is included in `totals.cost`, and the session opens in the Sessions tab like a main-thread @@ -1856,10 +1965,10 @@ costPerSessionP90 = nearest-rank P90 of the same set cacheSavedUsd = Σ rows (costOf(1M as input) - costOf(1M as cacheRead)) × cacheRead / 1e6 ``` -**Source:** the derived block is `finishTotals` (`usage-aggregate.mjs:1042-1082`), +**Source:** the derived block is `finishTotals` (`usage-aggregate.mjs:1152-1192`), which the previous-window projection calls too so a baseline is never derived a second, drifting way. `median` and `percentile` are exact over the values -(`usage-aggregate.mjs:1029-1041`), unlike §15's bucketed percentiles. +(`usage-aggregate.mjs:1125-1151`), unlike §15's bucketed percentiles. Active days come from `byDay`'s key count and the streak from `activeStreak` in `src/lib/dashboard/client/usage.mjs`; the tiles are `cadenceCells` there, and `printScoreCadence` (`src/commands/usage.mjs:219-242`) in the CLI. @@ -1868,13 +1977,13 @@ Active days come from `byDay`'s key count and the streak from `activeStreak` in `totals.humanPrompts` is not the fingerprint-based `typedPrompts` count. It can include control records and unrecognized machine-authored prompts. It is accumulated under an explicit main-thread guard -(`usage-aggregate.mjs:953`): a subagent's prompts are written by the harness, +(`usage-aggregate.mjs:1035`): a subagent's prompts are written by the harness, so counting them would report a person as having typed work nobody asked for by hand — and would grow the denominator exactly in the windows where delegation was heaviest, making autonomy fall as automation rose. `totals.prompts` still -records every prompt beside it (`usage-aggregate.mjs:928`); the two are +records every prompt beside it (`usage-aggregate.mjs:1009`); the two are different questions and both are on the wire. Touch rate is those same human -prompts per engaged hour (`usage-aggregate.mjs:1030`), so both per-prompt figures +prompts per engaged hour (`usage-aggregate.mjs:1140`), so both per-prompt figures share one denominator. A rate whose denominator is zero is `null`, never `0` — no engaged time means the rate was never measured, which is not what "zero per hour" claims. @@ -1900,8 +2009,8 @@ presence, not verified billing. A session with no usage rows contributes to neit that map nor its per-day session count. **Cost per session is a median over priced sessions only.** A session carries -`_priced` when it had any usage rows at all (`usage-aggregate.mjs:873-879`), and -only those costs enter the distribution (`usage-aggregate.mjs:962-980`). A session +`_priced` when it had any usage rows at all (`usage-aggregate.mjs:950-956`), and +only those costs enter the distribution (`usage-aggregate.mjs:1044-1075`). A session with no usage rows costs `$0` *structurally* — nothing was ever measured for it, the common case being a Codex subagent that only its ledger row identifies, whose tokens are stripped as a double-count (§16.2, §13c) — and letting those in would report "the typical session @@ -1915,9 +2024,9 @@ positive figure that rounds away at two decimals prints `<$0.01`, never "nothing" are different claims. **What the cache saved, asked as a difference.** `cacheSavingPerMillion` -(`usage-aggregate.mjs:732-746`) prices one million tokens twice through the +(`usage-aggregate.mjs:739-756`) prices one million tokens twice through the *injected* pricer — once as fresh input, once as cache reads — and takes the -gap; `cacheSavedFor` (`usage-aggregate.mjs:732-753`) scales that to the tokens +gap; `cacheSavedFor` (`usage-aggregate.mjs:739-763`) scales that to the tokens a row actually read from cache. Nothing in that path knows what the cache multiplier is, so the saving cannot drift out of step with §3's table the way a hard-coded "0.9 × input" would the day the multiplier changed. Both probes @@ -1936,15 +2045,15 @@ this row = $4.50 × 2,000,000 / 1e6 = $9.00 ``` The window total is the sum of those per-row figures -(`usage-aggregate.mjs:943-979`), carried on each session row as `cacheSavedUsd` -(`usage-aggregate.mjs:836-857`) so it is auditable a row at a time rather than only +(`usage-aggregate.mjs:1024-1074`), carried on each session row as `cacheSavedUsd` +(`usage-aggregate.mjs:907-933`) so it is auditable a row at a time rather than only in aggregate, and rendered in the cache tile's subtitle as `saved ≈ $X vs uncached`. **Deltas: what "the previous window" is, exactly.** For a displayed window of `d` days ending at `now`, the baseline is the equal-length window immediately before it — the half-open interval `[now − 2d, now − d)` -(`previousWindow`, `usage-aggregate.mjs:1157-1181`). Both bounds are derived from +(`previousWindow`, `usage-aggregate.mjs:1267-1292`). Both bounds are derived from `now` and `d`, the window the UI is *showing*, and never from the parse cutoff: the caller widens that cutoff on purpose so older records survive to be aggregated here, and deriving the baseline from a widened bound would silently @@ -1958,26 +2067,26 @@ BASELINE_MIN_ACTIVE_DAYS of history BEFORE the displayed window and returns null without it — while this depth is a strict superset of the previous window at every supported width. A delta against an unknown-length window is not a delta. The upper bound is exclusive so a session ending exactly at the boundary belongs to the current -window and is not counted in both (`endMs`, `usage-aggregate.mjs:888-899`). Asking for +window and is not counted in both (`endMs`, `usage-aggregate.mjs:966-980`). Asking for `previous` without widening the lookback yields an all-zero baseline — the older records were never read off disk — and every chip self-suppresses against it rather than claiming a change it cannot measure. Leaving `previous` off entirely leaves `agg.previous` as `null` — "not requested", which a zeroed totals object would misreport as "measured nothing" -(`usage-aggregate.mjs:1229`). A chip self-suppresses when the baseline is null +(`usage-aggregate.mjs:1313`). A chip self-suppresses when the baseline is null or zero, and a magnitude that rounds to zero prints flat rather than drawing an arrow the printed number does not support (`deltaChip`, `usage-rhythm.mjs:36-53`; `fmtDelta`, `src/commands/usage.mjs:184-195`). **Engaged time by day is a sibling map, not a `byDay` field.** `byDay`'s presence contract is **days with retained usage rows** — a key exists exactly when tokens -landed on that day (`dayBucket`, `usage-aggregate.mjs:698-705`) — and that is +landed on that day (`dayBucket`, `usage-aggregate.mjs:705-712`) — and that is what the active-day count and the streak above are counted from. Engaged time does not share that key set: a session that runs past midnight, or a day spent reading, produces worked time on a day that billed nothing. So -`buildEngagedByDay` (`usage-aggregate.mjs:1133-1153`) keys its own map, cutting +`buildEngagedByDay` (`usage-aggregate.mjs:1243-1263`) keys its own map, cutting each active interval at every local midnight it crosses -(`splitAtLocalMidnight`, `usage-aggregate.mjs:1119-1130`) and unioning the pieces +(`splitAtLocalMidnight`, `usage-aggregate.mjs:1229-1240`) and unioning the pieces per day, which makes the map sum exactly to `totals.engagedSeconds`. Folding it into `byDay` would have forced one of two lies: inventing zero-token `byDay` rows, or dropping real worked time. The consequence is visible on the tiles — @@ -2013,8 +2122,8 @@ byDay[day].exceptions += session.exceptions attributed to the session's FIRST ``` **Source:** `exceptions` and `aborts` accumulate together onto totals -(`totals.exceptions += s.exceptions`, `usage-aggregate.mjs:954`); the per-day series lands on `byDay` itself — -`byDay[s._day].exceptions` (`usage-aggregate.mjs:977`). Render is +(`totals.exceptions += s.exceptions`, `usage-aggregate.mjs:1055`); the per-day series lands on `byDay` itself — +`byDay[s._day].exceptions` (`usage-aggregate.mjs:1106`). Render is `relRate`/`relStat`/`relTrend` in `src/lib/dashboard/client/usage.mjs` (that bundle shares a basename with the CLI command module, so cited here by name, no line); `printScoreReliability` (`src/commands/usage.mjs:270-295`) prints @@ -2025,9 +2134,9 @@ model, and each host signals that differently: | Host | What is counted, and where | |---|---| -| claude | The API-error placeholder — Claude Code synthesizes a local turn with no completion behind it when a connection drops, a rate limit rejects, or auth fails. The decoder sets `isApiError` from either `isApiErrorMessage` or the literal `` model marker (`telemetry-records.mjs:269`), because the flag is not set on every build that emits the placeholder; the parser counts it as an exception, not a response, and returns before any model or usage attribution (`usage-parsers.mjs:701-720`). | -| codex | A `task_complete` event carrying a non-null `error` (`usage-parsers.mjs:936-954`). | -| codex | `turn_aborted` is counted **separately**, into `rec.aborts` (`usage-parsers.mjs:1048-1089`) — not into exceptions. | +| claude | The API-error placeholder — Claude Code synthesizes a local turn with no completion behind it when a connection drops, a rate limit rejects, or auth fails. The decoder sets `isApiError` from either `isApiErrorMessage` or the literal `` model marker (`telemetry-records.mjs:269`), because the flag is not set on every build that emits the placeholder; the parser counts it as an exception, not a response, and returns before any model or usage attribution (`usage-parsers.mjs:758-777`). | +| codex | A `task_complete` event carrying a non-null `error` (`usage-parsers.mjs:1049-1076`). | +| codex | `turn_aborted` is counted **separately**, into `rec.aborts` (`usage-parsers.mjs:1170-1218`) — not into exceptions. | | opencode | An assistant message carrying a non-null `error` other than `MessageAbortedError` (`usage-opencode.mjs:344-348`). | | opencode | `MessageAbortedError` — how OpenCode records a turn the user stopped — is counted **separately**, into `rec.aborts` (name at `usage-opencode.mjs:307-308`), and keeps the row's tokens and cost. | @@ -2036,7 +2145,7 @@ recorded interruption; it does not independently prove who initiated it. An exce is the turn failing. Summing them would report a deliberate interruption as a reliability problem and move a number that is supposed to mean "how often did this break". They are counted, carried -(`aborts`, `usage-aggregate.mjs:809-833`) and displayed side by side, with the +(`aborts`, `usage-aggregate.mjs:836-904`) and displayed side by side, with the distinction stated on the tile rather than left to the label. **Aborts are CODEX-AND-OPENCODE normalized evidence.** This counter consumes Codex @@ -2056,7 +2165,7 @@ treatment `latHist` (§15) and the context chip (§16) already get. **Exceptions ride the session's first-billed day.** The per-day series uses the same attribution as the session count — `byDay[s._day].exceptions += s.exceptions` -(`usage-aggregate.mjs:954-957`) — which is *not* the moment a turn dropped: a session spanning midnight lands all of its +(`usage-aggregate.mjs:1106`) — which is *not* the moment a turn dropped: a session spanning midnight lands all of its exceptions on the day its tokens first billed. That keeps the reliability trend and the session trend drawn on one convention — the alternative, attributing each exception to its own timestamp, would have made the two lines disagree @@ -2073,15 +2182,12 @@ measurement behind §10 breaks 33 such placeholder turns down as `server_error` placeholder shape, which is why the panel counts them together and §10 excludes them from the model ranking rather than showing a `$0` model row. -**What this does not model:** the rate's denominator is *responses*, which -includes the exception turns themselves (they increment `rec.responses` before -the error branch returns, `usage-parsers.mjs:568`) — they were real engaged -time, someone was genuinely waiting on them. A retry that eventually succeeded -appears as one exception plus one successful response, not as a single -recovered turn; nothing in either transcript links the two. And the worst-day -flag names the day with the most exceptions without inventing a threshold for -what counts as a spike, because any constant chosen here would be a judgement -the data never made. +**What this does not model:** the rate's denominator is accounted responses. Claude API-error +placeholders increment `rec.exceptions` and return before response, usage or punchcard accounting +(`recordClaudeAssistantTurn`, `usage-parsers.mjs:749-777`); they do not enter that denominator. +A later successful retry can contribute one response alongside the earlier exception, but the +metric does not pair them into a single recovered turn. The worst-day flag names the day with +the most exceptions without inventing a threshold for what counts as a spike. --- @@ -2101,7 +2207,7 @@ byTool[name] += session.tools[name] summed across sessions byDay[day].byModelFamily[fam] += rowCost fam = modelFamily(row.model) ``` -**Source:** the tool tally is folded into `byTool` at `usage-aggregate.mjs:943-1007`; +**Source:** the tool tally is folded into `byTool` at `usage-aggregate.mjs:1024-1117`; the per-day family split is this call: `addCost(d.byModelFamily, modelFamily(row.model), rowCost)` (`usage-aggregate.mjs:776`), inside the usage-row pass because only a row knows its day. Render is `toolRows`/`modelMix` in @@ -2109,10 +2215,10 @@ row knows its day. Render is `toolRows`/`modelMix` in **Tool names are the host's own, never renamed.** Claude's tally is keyed by the `tool_use` block's own `name` (`collectClaudeToolNames`, -`usage-parsers.mjs:627-638`). Codex's five tallied item types — +`usage-parsers.mjs:662-673`). Codex's five tallied item types — `CommandExecution`, `McpToolCall`, `FileChange`, `CollabAgentToolCall`, -`DynamicToolCall` (`CODEX_TOOL_ITEM_TYPES`, `usage-parsers.mjs:1040-1046`, tallied at this -call: `CODEX_TOOL_ITEM_TYPES.has(decoded.unknownItemType)` — `usage-parsers.mjs:1092-1103`) — +`DynamicToolCall` (`CODEX_TOOL_ITEM_TYPES`, `usage-parsers.mjs:1162-1168`, tallied at this +call: `CODEX_TOOL_ITEM_TYPES.has(decoded.unknownItemType)` — `usage-parsers.mjs:1221-1235`) — keep those exact spellings in the ranking. Mapping `CommandExecution` onto `Bash`, or `FileChange` onto `Edit`, would be a claim about equivalence that neither host makes: the vocabularies are host-specific, the semantics do not @@ -2126,12 +2232,12 @@ above it is read against — and the fold row is dimmed because `Other` is a residue, not a tool. **Model-family folding, with the rules pinned.** `modelFamily` -(`usage-aggregate.mjs:272-279`) lowercases the id, keeps only the segment after +(`usage-aggregate.mjs:275-282`) lowercases the id, keeps only the segment after the last `/` so a namespaced id still ends on the same tokens, then: | Rule | Example id | Family | |---|---|---| -| contains an Anthropic family name (`CLAUDE_FAMILIES`, `usage-aggregate.mjs:264`) | `claude-opus-5-20260401` | `opus` | +| contains an Anthropic family name (`CLAUDE_FAMILIES`, `usage-aggregate.mjs:267`) | `claude-opus-5-20260401` | `opus` | | — matched by containment, not position, since the id shape has moved | `claude-3-5-sonnet-20241022` | `sonnet` | | — and after the last slash, so a namespaced id still folds | `openrouter/anthropic/claude-haiku-4-5` | `haiku` | | otherwise matches `gpt-(\d+)` | `gpt-5.6-sol` | `gpt-5` | @@ -2360,14 +2466,14 @@ commit `540be18` in the historical fix. `parseCodex`'s single `addUsage()` call never included a `responses` field — Claude's parser passes `responses: 1` per API message at this call: -`usage-parsers.mjs:677` (the current equivalent), but Codex's call +`usage-parsers.mjs:723` (the current equivalent), but Codex's call passed no such field at all. Because `byModel[model].responses` is summed -directly from each usage row's `responses` field (`usage-aggregate.mjs:775-778`, +directly from each usage row's `responses` field (`usage-aggregate.mjs:786-789`, `m.responses += row.responses`), **every** Codex model in §10's Models-in-Play list displayed `0 resp` regardless of real token/cost volume or actual `agent_message` count. **Fix:** parseCodex now passes `responses: rec.responses` (the session's own tallied response count, inside -`finalizeCodexUsage`, `usage-parsers.mjs:1136-1183`) on its `addUsage()` call. +`finalizeCodexUsage`, `usage-parsers.mjs:1303-1355`) on its `addUsage()` call. #### Bug B — subagent thread-replay could double-bill tokens @@ -2396,10 +2502,10 @@ confirmed as a real Codex rollout field by **[C7]**) and skips the `addUsage()` call entirely when its value is `'subagent'` — `finalizeCodexUsage` returns early at this call: `if (!lastUsage || rec.threadSource === 'subagent')` -(`usage-parsers.mjs:1136-1173`). The session record itself is **not** +(`usage-parsers.mjs:1303-1331`). The session record itself is **not** dropped — it remains visible in the Sessions tab with `threadSource` surfaced (mirroring the existing `sidechain` flag Claude sessions already -carry, `usage-parsers.mjs:760-763`), so a maintainer auditing the raw data can +carry, `usage-parsers.mjs:844-847`), so a maintainer auditing the raw data can still see it; it simply contributes zero tokens/cost, exactly as intended by the "models still shows up in §10's list, with zero cost" mechanism §10 describes. @@ -2461,8 +2567,8 @@ parity: it does not apply ledger fallback or allocate costs per turn/day/model. `byModel` on the first run after the change, purely because the cache predated it; every unit test still passed, since tests only exercise a fresh parse. `SCHEMA_VERSION` went to `4` specifically to force the one-time - re-parse; the constant now reads `24` (`usage-index.mjs:177`), each bump since - having forced its own re-parse the same way. + re-parse. That is the historical v4 migration; the current `SCHEMA_VERSION` is `26` + (`usage-index.mjs:198`), with the present compatibility contract stated above. Re-querying the same live server after the bump returned `totals.exceptions: 20` with `` absent from `byModel` — measured, not projected. diff --git a/scripts/run-tests.mjs b/scripts/run-tests.mjs index d0587a8d..2335a1d5 100644 --- a/scripts/run-tests.mjs +++ b/scripts/run-tests.mjs @@ -28,7 +28,8 @@ export const SUITES = { ['--test', 'tests/ui/dashboard-project-context.mjs', 'tests/ui/maintenance-projects.mjs', 'tests/ui/maintenance-host-alignment.mjs', 'tests/ui/intelligence-picker.mjs', 'tests/ui/usage-project-groups.mjs', 'tests/ui/context-coverage.mjs', 'tests/ui/host-readiness.mjs', - 'tests/ui/maintenance-focus.mjs', 'tests/ui/maintenance-guidance.mjs'], + 'tests/ui/maintenance-focus.mjs', 'tests/ui/maintenance-guidance.mjs', + 'tests/ui/session-surfaces.mjs'], ], }; diff --git a/src/commands/system.mjs b/src/commands/system.mjs index c417951d..7b583540 100644 --- a/src/commands/system.mjs +++ b/src/commands/system.mjs @@ -1,3 +1,4 @@ +import { censusDisclosure } from '../lib/census-presentation.mjs'; // ak system — the machine footprint in the terminal (ADR-0025). // // The CLI twin of the dashboard's System area, driving the SAME composed @@ -231,24 +232,25 @@ function renderRuntime(runtime) { } const rows = census?.value ?? []; if (!rows.length) { - console.log(` ${dim('no agent processes are running')}`); + console.log(` ${dim('no coding-agent or desktop-application processes are running')}`); return; } const sink = reasonSink(); console.log(''); - table(['HOST', 'PID', 'CPU', 'RSS', 'UPTIME', 'PROJECT'], rows.map((row) => [ - row.host, + table(['CODING-AGENT HOST / DESKTOP APPLICATION', 'PID', 'CPU', 'RSS', 'UPTIME', 'WORKING CONTEXT'], rows.map((row) => [ + row.application ?? row.host ?? 'Unknown process', String(row.pid), sink.cell(row.cpuPercent, fmtPercent), sink.cell(row.rssBytes, fmtBytes), sink.cell(row.uptimeMs, fmtDuration), - row.project?.status === UNKNOWN ? 'unattributed' : (row.project?.value?.label ?? 'unattributed'), + row.source?.status === UNKNOWN ? 'unattributed' + : (row.source?.value?.label ?? row.project?.value?.label ?? 'unattributed'), ])); sink.report(); - // `project` degrades per process (a cwd the platform will not disclose); its + // `source` degrades per process (a cwd the platform will not disclose); its // reason lives on the row, not in the numeric sink above. - for (const reason of new Set(rows.filter((row) => row.project?.status === UNKNOWN) - .map((row) => row.project.reason))) { + for (const reason of new Set(rows.filter((row) => (row.source ?? row.project)?.status === UNKNOWN) + .map((row) => (row.source ?? row.project).reason))) { console.log(` ${dim(`unattributed: ${reason}`)}`); } } @@ -348,6 +350,7 @@ function renderProjects(projects, now) { info(dim('not measured yet — run: ak system --refresh=machine')); return; } + info(dim(censusDisclosure(projects))); field('discovered', `${meas(projects.count)}${projects.truncated ? dim(' · list truncated') : ''}`); if (!projects.locMeasured) field('lines of code', dim('not measured in this scan')); diff --git a/src/lib/census-presentation.mjs b/src/lib/census-presentation.mjs new file mode 100644 index 00000000..dbb5c752 --- /dev/null +++ b/src/lib/census-presentation.mjs @@ -0,0 +1,8 @@ +/** An absent legacy field is unknown, never a measured zero. Shared by CLI/UI. */ +export function censusDisclosure(census = {}) { + const count = (field) => Number.isInteger(census[field]) && census[field] >= 0 ? String(census[field]) : 'Unknown number of'; + return `${count('importedExcluded')} confirmed pure imported copies excluded (no project, host or origin contribution); ` + + `${count('importedMixed')} mixed files retain proven native activity; ` + + `${count('importedUnresolved')} files have unresolved bounded ownership (not confirmed exclusions). ` + + 'The dedicated Cowork transcript source is not covered; Cowork declarations in covered transcripts remain valid observations.'; +} diff --git a/src/lib/codex-import-marker.mjs b/src/lib/codex-import-marker.mjs index 1be22e44..afa1ab0f 100644 --- a/src/lib/codex-import-marker.mjs +++ b/src/lib/codex-import-marker.mjs @@ -12,14 +12,15 @@ export const CODEX_IMPORT_TURN_PREFIX = 'external-import-turn'; -/** One decoded rollout record: is it a turn of an imported thread? Only a +/** One decoded rollout record: does it explicitly belong to an imported turn? Only a * string `payload.turn_id` counts; the marker text inside a message does not. */ export function isCodexImportedLine(e) { const turnId = e?.payload?.turn_id; return typeof turnId === 'string' && turnId.startsWith(CODEX_IMPORT_TURN_PREFIX); } -/** A rollout's bounded head (raw JSON lines): is the rollout an imported copy? +/** Does a bounded head contain an imported turn? This does not prove the + * whole rollout is imported-only; later native turns require a separate scan. * A line without the marker text is not parsed, which keeps this cheap on the * large native heads; unparseable and non-string lines are skipped. */ export function isImportedCodexRollout(headLines) { @@ -31,3 +32,69 @@ export function isImportedCodexRollout(headLines) { } return false; } + +/** Stateful per-turn ownership. Explicit turn metadata outranks adjacency; + * a foreign completion never closes the active turn. Missing IDs in a mixed + * file cannot open a native turn. No IDs or payloads escape the state. */ +export function newCodexTurnOwnership({ hasImports = true } = {}) { + return { hasImports, ownershipComplete: true, activeId: null, activeOwner: hasImports ? 'ambiguous' : 'native', + importedIds: new Set(), importedTurnCountComplete: true, importedRecords: 0, ambiguousRecords: 0, nativeRecords: 0 }; +} + +const validTurnId = (id) => typeof id === 'string' && id.length > 0 + && id.length <= 256 && !/\s/u.test(id); + +function noteBoundary(state, p, imported) { + // Absence can mean an enriching context. An explicitly invalid declaration + // cannot preserve adjacency to the prior native turn in a mixed source. + if (state.hasImports && Object.hasOwn(p, 'turn_id') && !validTurnId(p.turn_id)) { + state.activeId = null; + state.activeOwner = 'ambiguous'; + state.ownershipComplete = false; + return; + } + if (!validTurnId(p.turn_id) && p.type !== 'task_started') return; + state.activeId = validTurnId(p.turn_id) ? p.turn_id : null; + state.activeOwner = imported ? 'imported' : state.activeId ? 'native' + : state.hasImports ? 'ambiguous' : 'native'; +} + +function endsActiveTurn(record, p, activeId) { + return record?.type === 'event_msg' && ['task_complete', 'turn_aborted'].includes(p.type) + && validTurnId(p.turn_id) && p.turn_id === activeId; +} + +function noteImportedId(state, id) { + if (state.importedIds.has(id)) return; + if (validTurnId(id) && state.importedIds.size < 4096) state.importedIds.add(id); + else state.importedTurnCountComplete = false; +} + +export function codexTurnOwner(state, record) { + const p = record?.payload ?? {}; + const id = p.turn_id; + const imported = isCodexImportedLine(record); + const boundary = record?.type === 'turn_context' + || (record?.type === 'event_msg' && p.type === 'task_started'); + // An ID-less context enriches an identified turn but cannot open one. + if (boundary) noteBoundary(state, p, imported); + let owner = imported ? 'imported' : state.activeOwner; + if (state.hasImports && id !== undefined && (!validTurnId(id) || id !== state.activeId) && !imported) owner = 'ambiguous'; + if (imported) { noteImportedId(state, id); state.importedRecords++; } + else if (owner === 'imported') state.importedRecords++; + else if (owner === 'ambiguous' && record?.type !== 'session_meta') state.ambiguousRecords++; + if (owner === 'native' && record?.type === 'event_msg' + && ['user_message', 'agent_message', 'item_completed', 'token_count'].includes(p.type)) state.nativeRecords++; + if (endsActiveTurn(record, p, state.activeId)) { + state.activeId = null; + state.activeOwner = state.hasImports ? 'ambiguous' : 'native'; + } + return owner; +} + +/** Only enumerated counts are persisted; turn IDs remain local to the walk. */ +export function codexImportEvidence(state) { + return { importedTurns: state.importedIds.size, importedTurnCountComplete: state.importedTurnCountComplete, + importedRecords: state.importedRecords, + ambiguousRecords: state.ambiguousRecords, nativeRecords: state.nativeRecords }; +} diff --git a/src/lib/codex-rollout-reader.mjs b/src/lib/codex-rollout-reader.mjs index fba25d8d..989cb113 100644 --- a/src/lib/codex-rollout-reader.mjs +++ b/src/lib/codex-rollout-reader.mjs @@ -57,12 +57,16 @@ function clippedStub(head) { } /** Parse one whole line, or `null` for anything that is not a JSON object. */ -function parseLine(buf) { - if (!buf.length || buf[0] !== OPEN_BRACE) return null; +function parseLine(buf, stats) { + if (!buf.length) return null; + if (buf[0] !== OPEN_BRACE) { + if (buf.toString('utf8').trim()) stats.malformedRecords++; + return null; + } try { const obj = JSON.parse(buf.toString('utf8')); return obj && typeof obj === 'object' ? obj : null; - } catch { return null; } + } catch { stats.malformedRecords++; return null; } } /** @@ -75,6 +79,7 @@ function parseLine(buf) { */ function* passOver(file, { chunkBytes, maxLineBytes, stats }) { stats.clippedLines = 0; + stats.malformedRecords = 0; const fd = fs.openSync(file, 'r'); try { const limit = fs.fstatSync(fd).size; @@ -102,7 +107,7 @@ function* passOver(file, { chunkBytes, maxLineBytes, stats }) { stats.clippedLines++; out = clippedStub(head); } else if (partsLen) { - out = parseLine(parts.length === 1 ? parts[0] : Buffer.concat(parts, partsLen)); + out = parseLine(parts.length === 1 ? parts[0] : Buffer.concat(parts, partsLen), stats); } parts = []; partsLen = 0; oversize = false; head = null; return out; @@ -119,7 +124,7 @@ function* passOver(file, { chunkBytes, maxLineBytes, stats }) { if (nl < 0) break; let obj; if (partsLen === 0 && !oversize && nl - start <= maxLineBytes) { - obj = parseLine(view.subarray(start, nl)); // whole line inside this chunk: no copy + obj = parseLine(view.subarray(start, nl), stats); // whole line inside this chunk: no copy } else { take(Buffer.from(view.subarray(start, nl))); // `chunk` is reused, so keep a copy obj = finish(); @@ -140,7 +145,7 @@ function* passOver(file, { chunkBytes, maxLineBytes, stats }) { * Open a rollout for parsing. Returns a source `parseCodex` accepts in place of * a string: `head` (the first 256 KiB, for session-origin detection), `lines` * (re-iterable — each iteration re-reads the file, which the subagent replay - * pre-pass needs) and `stats` (`clippedLines` of the LAST pass). + * pre-pass needs) and `stats` (`clippedLines` and `malformedRecords` of the LAST pass). * * Throws (ENOENT, EACCES, …) when the file cannot be opened or its head read; * the caller decides how to report that. @@ -151,7 +156,7 @@ function* passOver(file, { chunkBytes, maxLineBytes, stats }) { export function openCodexRollout(file, limits = {}) { const chunkBytes = limits.chunkBytes ?? DEFAULT_CHUNK_BYTES; const maxLineBytes = limits.maxLineBytes ?? DEFAULT_MAX_LINE_BYTES; - const stats = { clippedLines: 0 }; + const stats = { clippedLines: 0, malformedRecords: 0 }; const fd = fs.openSync(file, 'r'); let head; try { diff --git a/src/lib/codex-usage-walk.mjs b/src/lib/codex-usage-walk.mjs index da7e8a2e..92a35a06 100644 --- a/src/lib/codex-usage-walk.mjs +++ b/src/lib/codex-usage-walk.mjs @@ -33,9 +33,12 @@ const ZERO = Object.freeze({ input_tokens: 0, cached_input_tokens: 0, output_tok const MONOTONIC = ['input_tokens', 'cached_input_tokens', 'output_tokens', 'total_tokens']; const snapshotOf = (t) => Object.fromEntries(FIELDS.map((f) => [f, Number(t?.[f]) || 0])); +const isTotalOnly = (t) => t.total_tokens > 0 && !t.input_tokens && !t.cached_input_tokens && !t.output_tokens; /** * @typedef {object} CodexUsageWalk + * @property {boolean} excludedBaseline + * @property {boolean} unknownBaseline * @property {boolean} unattributable * @property {Record|null} prev * @property {string|null} model @@ -51,6 +54,8 @@ const snapshotOf = (t) => Object.fromEntries(FIELDS.map((f) => [f, Number(t?.[f] export function newCodexUsageWalk({ unattributable = false } = {}) { return { unattributable, + excludedBaseline: false, + unknownBaseline: false, // a total-only counter hid the component baseline prev: null, // the previous cumulative snapshot, replayed ones included model: null, // the model of the turn_context in effect lastMs: null, // the last finite event time seen on a token_count @@ -94,15 +99,31 @@ export function noteCodexWalkResponse(walk, ms, dayOf) { * the running total; an own one books its delta on `ms`'s day under the current * model. */ -export function walkCodexTokenCount(walk, total, ms, replay, dayOf) { +export function walkCodexTokenCount(walk, total, ms, replay, dayOf, last = null) { if (!total) return; const cur = snapshotOf(total); const prev = walk.prev; - walk.prev = cur; if (Number.isFinite(ms)) walk.lastMs = ms; - const restarted = prev !== null && MONOTONIC.some((f) => cur[f] < prev[f]); + if (isTotalOnly(cur)) { + // A total-only snapshot has no component baseline. Treat the next full + // snapshot as a new baseline unless that call itself proves a reset. + walk.unknownBaseline = true; + walk.excludedBaseline = replay; + return; + } + // A first native call can reset ABOVE the copied baseline. Matching + // last/total counters prove that reset; an identical re-emission does not. + const lastSnapshot = last ? snapshotOf(last) : null; + const explicitReset = (walk.excludedBaseline || walk.unknownBaseline) && lastSnapshot + && FIELDS.every((f) => cur[f] === lastSnapshot[f]) + && (prev === null || FIELDS.some((f) => cur[f] !== prev[f])); + const restarted = explicitReset || (prev !== null && MONOTONIC.some((f) => cur[f] < prev[f])); + const unknownBaseline = walk.unknownBaseline; + walk.unknownBaseline = false; + walk.prev = cur; + walk.excludedBaseline = replay; if (restarted) walk.segments++; - if (replay || walk.unattributable) return; + if (replay || walk.unattributable || (unknownBaseline && !restarted)) return; const base = prev === null || restarted ? ZERO : prev; const d = Object.fromEntries(FIELDS.map((f) => [f, Math.max(0, cur[f] - base[f])])); if (!d.input_tokens && !d.output_tokens && !d.cached_input_tokens) return; diff --git a/src/lib/dashboard-server.mjs b/src/lib/dashboard-server.mjs index cef33622..b81cf2ee 100644 --- a/src/lib/dashboard-server.mjs +++ b/src/lib/dashboard-server.mjs @@ -287,7 +287,8 @@ function censusBackedDiscovery() { readCensus: () => (last ? { everSeen: last.everSeen, onDisk: last.onDisk, gitRepos: last.gitRepos, learning: last.learning, - complete: last.complete, + complete: last.complete, importedExcluded: last.importedExcluded, + importedMixed: last.importedMixed, importedUnresolved: last.importedUnresolved, } : null), }; } @@ -378,8 +379,8 @@ async function collectData({ cwd, fetchStatus, projectParam, getProjectSnapshot, intel: { selectedProjectKey: selected?.key ?? null, selectedProjectLabel: selected?.label ?? null, - projects: projects.map(({ key, label, path: projectPath, source, learningScope, learningScopeEvidence, learningOrigins, learningObservedAt }) => ( - { key, label, path: projectPath, source, + projects: projects.map(({ key, label, path: projectPath, source, hosts, sessionOrigins, sessionSurfaces, learningScope, learningScopeEvidence, learningOrigins, learningObservedAt }) => ( + { key, label, path: projectPath, source, hosts, sessionOrigins, sessionSurfaces, learningScope: ['repository', 'worktree', 'user'].includes(learningScope) ? learningScope : 'unknown', learningScopeEvidence: learningScopeEvidence ?? 'unclassified', learningObservedAt: learningObservedAt ?? null, learningOrigins: ['claude-desktop', 'codex-desktop'].filter((origin) => learningOrigins?.includes(origin)) } diff --git a/src/lib/dashboard/client.mjs b/src/lib/dashboard/client.mjs index 47c217fb..167e835e 100644 --- a/src/lib/dashboard/client.mjs +++ b/src/lib/dashboard/client.mjs @@ -1,3 +1,5 @@ +import { censusDisclosure } from '../census-presentation.mjs'; +import { SESSION_SURFACE_LABELS, SESSION_HOST_LABELS, SESSION_INITIATOR_LABELS, SESSION_PROVIDER_LABELS, sessionPresentation, sessionProviderPresentation } from '../session-surface.mjs'; import { repositoryTree } from './project-groups.mjs'; import { contextCard } from './context-card.mjs'; import { contextHostCard } from './context-host-card.mjs'; @@ -92,6 +94,8 @@ aboutSrc = inject(aboutSrc, 'var ABOUT = []; // PLACEHOLDER:ABOUT_JS', `var ABOU const datetimeSrc = readSplit('datetime.mjs'); const hostReadinessSrc = readSplit('host-readiness.mjs'); +const sessionPresentationSrc = readSplit('session-presentation.mjs'); +const sessionVocabularySrc = `const SESSION_SURFACE_LABELS=${JSON.stringify(SESSION_SURFACE_LABELS)},SESSION_HOST_LABELS=${JSON.stringify(SESSION_HOST_LABELS)},SESSION_INITIATOR_LABELS=${JSON.stringify(SESSION_INITIATOR_LABELS)},SESSION_PROVIDER_LABELS=${JSON.stringify(SESSION_PROVIDER_LABELS)};${sessionPresentation.toString()}${sessionProviderPresentation.toString()}`; const intelligenceSrc = readSplit('intelligence.mjs'); const pollSrc = readSplit('poll.mjs'); const refreshControlSrc = readSplit('refresh-control.mjs'); @@ -165,5 +169,5 @@ const bootSrc = readSplit('boot.mjs'); export const JS = ` (function(){ ${bootstrapSrc}${contextCard.toString()}${contextHostCard.toString()}${repositoryTree.toString()}${overviewSrc}${datetimeSrc}${hostReadinessSrc} -${intelligenceSrc}${pollSrc}${refreshControlSrc}${usageRhythmSrc}${usagePromptsSrc}${usageContextHooksSrc}${usageSrc}${modelLifecycleSrc}${usageOrchestratorsSrc}${rufloComponentsSrc}${aboutSrc}${systemReadoutSrc}${systemProjectsSrc}${maintenanceWorkspaceSrc}${maintenanceFiltersSrc}${maintenanceCardsSrc}${maintenanceOperationSrc}${maintenanceLanguageLogosSrc}${maintenanceFocusSrc}${maintenanceInventorySrc}${maintenanceRelationshipsSrc}${maintenanceInspectorSrc}${maintenanceGuidanceSrc}${maintenanceDiscoverySrc}${maintenanceActivitySrc}${systemMaintenanceActionsSrc}${systemMaintenanceSrc}${bootSrc}})(); +${censusDisclosure.toString()}${sessionVocabularySrc}${sessionPresentationSrc}${intelligenceSrc}${pollSrc}${refreshControlSrc}${usageRhythmSrc}${usagePromptsSrc}${usageContextHooksSrc}${usageSrc}${modelLifecycleSrc}${usageOrchestratorsSrc}${rufloComponentsSrc}${aboutSrc}${systemReadoutSrc}${systemProjectsSrc}${maintenanceWorkspaceSrc}${maintenanceFiltersSrc}${maintenanceCardsSrc}${maintenanceOperationSrc}${maintenanceLanguageLogosSrc}${maintenanceFocusSrc}${maintenanceInventorySrc}${maintenanceRelationshipsSrc}${maintenanceInspectorSrc}${maintenanceGuidanceSrc}${maintenanceDiscoverySrc}${maintenanceActivitySrc}${systemMaintenanceActionsSrc}${systemMaintenanceSrc}${bootSrc}})(); `; diff --git a/src/lib/dashboard/client/intelligence.mjs b/src/lib/dashboard/client/intelligence.mjs index 239641be..5a3e2308 100644 --- a/src/lib/dashboard/client/intelligence.mjs +++ b/src/lib/dashboard/client/intelligence.mjs @@ -1,6 +1,8 @@ // @ts-nocheck — browser bundle source (never node-imported; client.mjs // reads it as text). See src/lib/dashboard/client/**'s eslint.config.mjs // override comment for why this directory isn't run through the node lib. +import { censusDisclosure } from '../../census-presentation.mjs'; +import { projectSurfacesHtml, surfaceNames } from './session-presentation.mjs'; import { renderHostReadiness } from './host-readiness.mjs'; import { renderAbout } from './about.mjs'; import { DASH_TOKEN, activeTab, esc, overviewView, positionThumb } from './bootstrap.mjs'; @@ -54,20 +56,18 @@ import { fmtNum, kpi } from './usage.mjs'; html+='

    At least one transcript could not be read, ' +"so every figure above is a lower bound.

    "; } + html+='

    '+esc(censusDisclosure(c))+'

    '; body.innerHTML=html; box.hidden=false; } - var INTEL_SCOPE_GROUPS=[['repository','Git repositories'],['worktree','Git worktrees'],['user','User-level learning'],['unknown','Other / unclassified']]; - var machineWideDesignationFilter='all'; + var INTEL_SCOPE_GROUPS=[['repository','Git repositories'],['worktree','Git worktrees'],['user','User-level learning'],['unknown','Unknown']]; + var machineWideDesignationFilter='all',machineWideSurfaceFilter='all'; function machineWideDesignation(p){ if(p.learningScope==='repository')return 'Git repository'; if(p.learningScope==='worktree')return 'Git worktree'; - var origins=Array.isArray(p.learningOrigins)?p.learningOrigins:[]; - if(origins.includes('codex-desktop'))return 'ChatGPT Desktop'; - if(origins.includes('claude-desktop'))return 'Claude Desktop'; - if(Array.isArray(p.hosts)&&p.hosts.includes('opencode'))return 'OpenCode'; - return 'Directory'; + if(p.learningScope==='user')return 'User-level learning'; + return 'Unknown'; } function intelScopeRows(rows,scope){ return rows.filter(function(p){ @@ -86,7 +86,7 @@ import { fmtNum, kpi } from './usage.mjs'; var label=p.label||'(unlabeled)'; return '
    ' +''+esc(label)+storeHtml+'' - +''+esc(machineWideDesignation(p))+'' + +''+esc(machineWideDesignation(p))+''+projectSurfacesHtml(p)+'' +''+esc(fmtNum(p.patternsLearned))+'' +''+esc(fmtNum(p.patternStoreCount))+'' +''+esc(lastTxt)+'
    '; @@ -110,12 +110,15 @@ import { fmtNum, kpi } from './usage.mjs'; +kpi("most active project",totals.mostActiveProject||"—","by most recent learning adaptation","accent"); var table=document.getElementById("mw-table"); if(!table)return; - if(!perProject.length){table.innerHTML='
    no projects discovered on this machine.
    ';return;} + if(!perProject.length){machineWideSurfaceFilter='all';table.innerHTML='
    no projects discovered on this machine.
    ';return;} var designations=['all'].concat(Array.from(new Set(perProject.map(machineWideDesignation))).sort()); - var visible=(machineWideDesignationFilter==='all'?perProject:perProject.filter(function(row){return machineWideDesignation(row)===machineWideDesignationFilter;})).sort(function(a,b){return String(a.label||'').localeCompare(String(b.label||''),undefined,{sensitivity:'base',numeric:true})||String(a.key||a.path||'').localeCompare(String(b.key||b.path||''));}); + var surfaceChoices=Array.from(new Set(perProject.flatMap(surfaceNames))).sort(); + if(machineWideSurfaceFilter!=='all'&&!surfaceChoices.includes(machineWideSurfaceFilter))machineWideSurfaceFilter='all'; + var visible=(machineWideDesignationFilter==='all'?perProject:perProject.filter(function(row){return machineWideDesignation(row)===machineWideDesignationFilter;})).filter(function(row){return machineWideSurfaceFilter==='all'||surfaceNames(row).includes(machineWideSurfaceFilter);}).sort(function(a,b){return String(a.label||'').localeCompare(String(b.label||''),undefined,{sensitivity:'base',numeric:true})||String(a.key||a.path||'').localeCompare(String(b.key||b.path||''));}); table.innerHTML='
    ' +designations.map(function(designation){var label=designation==='all'?'All':designation;return '';}).join('') - +'
    '+machineWideTable(visible); + +''+machineWideTable(visible); + var surfaceSelect=document.getElementById('mw-surface-filter');if(surfaceSelect)surfaceSelect.onchange=function(){machineWideSurfaceFilter=surfaceSelect.value;renderMachineWide(mw);}; if(table.querySelectorAll)Array.from(table.querySelectorAll('.mw-filter-pill')).forEach(function(button){button.addEventListener('click',function(){machineWideDesignationFilter=button.getAttribute('data-designation')||'all';renderMachineWide(mw);});}); } diff --git a/src/lib/dashboard/client/maintenance-filters.mjs b/src/lib/dashboard/client/maintenance-filters.mjs index 4f854743..54a4d9aa 100644 --- a/src/lib/dashboard/client/maintenance-filters.mjs +++ b/src/lib/dashboard/client/maintenance-filters.mjs @@ -1,4 +1,5 @@ // @ts-nocheck — dashboard browser bundle. +import { surfaceFacetLabel } from './session-presentation.mjs'; import { mntIcon, mntProjectDesignation } from './maintenance-cards.mjs'; import { esc } from './bootstrap.mjs'; import { MNT, MNT_GUIDANCE_LANE_LABELS, MNT_CONFLICT_EXPLANATIONS, MNT_CREDENTIAL_READINESS_LABELS, MNT_CURATED_VIEW_LABELS, MNT_SCOPE_LABELS, mntHumanize, mntKindLabel } from './maintenance-workspace.mjs'; @@ -9,13 +10,13 @@ import { MNT, MNT_GUIDANCE_LANE_LABELS, MNT_CONFLICT_EXPLANATIONS, MNT_CREDENTIA "evidenceFields","recentlyChanged", ]; var MNT_FACET_LABEL={ - family:"Resource",scope:"Scope",environment:"Environment",project:"Project",sessionOrigin:"Session origin",projectType:"Project type",kind:"Type",adapter:"Adapters",consumer:"Hosts", + family:"Resource",scope:"Scope",environment:"Environment",project:"Project",sessionOrigin:"Session surface",projectType:"Project type",kind:"Type",adapter:"Adapters",consumer:"Hosts", carrier:"Carrier",provenance:"Source",packageManager:"Package manager",versionState:"Version state", guidance:"Guidance",dependencyRole:"Dependency role",conflict:"Conflict", credentialReadiness:"Credential",channel:"Channel",evidenceFields:"Evidence available", recentlyChanged:"Recently changed", }; - var MNT_ADAPTER_LABELS={claude:'Claude',codex:'Codex',opencode:'OpenCode',hermes:'Hermes'}; + var MNT_ADAPTER_LABELS={claude:'Claude Code',codex:'Codex',opencode:'OpenCode',hermes:'Hermes Agent'}; var MNT_DEPENDENCY_ROLE_LABEL={"depends-on":"Depends on","depended-on-by":"Depended on by","none":"No dependency role"}; function mntFacetLabelsFor(facet){ @@ -24,8 +25,8 @@ import { MNT, MNT_GUIDANCE_LANE_LABELS, MNT_CONFLICT_EXPLANATIONS, MNT_CREDENTIA } export function mntFacetValueLabel(facet,value){ if(facet==="adapter"||facet==="consumer")return MNT_ADAPTER_LABELS[value]||mntHumanize(value); - if(facet==="sessionOrigin")return ({"claude-desktop":"Claude Desktop","codex-desktop":"ChatGPT Desktop",unknown:"Unclassified"})[value]||"Unclassified"; - if(facet==="projectType")return ({git:'Git',folder:'Folder',worktree:'Worktree',unknown:'Not checked'})[value]||'Not checked'; + if(facet==="sessionOrigin")return surfaceFacetLabel(value); + if(facet==="projectType")return ({git:'Git',folder:'Folder',worktree:'Worktree',unknown:'Unknown'})[value]||'Unknown'; if(facet==="scope")return MNT_SCOPE_LABELS[value]||mntHumanize(value); if(facet==="kind")return mntKindLabel(value); if(facet==="guidance")return MNT_GUIDANCE_LANE_LABELS[value]||mntHumanize(value); diff --git a/src/lib/dashboard/client/maintenance-focus.mjs b/src/lib/dashboard/client/maintenance-focus.mjs index d4518f30..567089f2 100644 --- a/src/lib/dashboard/client/maintenance-focus.mjs +++ b/src/lib/dashboard/client/maintenance-focus.mjs @@ -1,4 +1,5 @@ // @ts-nocheck — classic browser bundle source. +import { projectSurfacesHtml } from './session-presentation.mjs'; import { mntLanguageLogo } from './maintenance-language-logos.mjs'; import { esc } from './bootstrap.mjs'; import { MNT, MNT_SCOPE_LABELS, mntKindLabel } from './maintenance-workspace.mjs'; @@ -68,7 +69,7 @@ import { mntFacetValueLabel } from './maintenance-filters.mjs'; +(level==='project'?mntProjectKindBadge(node.projectKind):'') +(level==='project'&&node.languages&&node.languages.length?mntLanguageBadges(node.languages):'') +(node.description?''+esc(node.description)+'':'') - +(note?''+esc(note)+'':'')+''+esc(node.count)+' installation'+(node.count===1?'':'s')+''+mntIcon('chevron')+'
  • '; + +(note?''+esc(note)+'':'')+''+esc(node.count)+' installation'+(node.count===1?'':'s')+''+mntIcon('chevron')+''+(level==='project'?projectSurfacesHtml(node):'')+''; } function mntFocusInstallation(row,index){ var crumbs=(row.breadcrumb||[]).slice(),scope=row.scope||{}; diff --git a/src/lib/dashboard/client/session-presentation.mjs b/src/lib/dashboard/client/session-presentation.mjs new file mode 100644 index 00000000..dedbee79 --- /dev/null +++ b/src/lib/dashboard/client/session-presentation.mjs @@ -0,0 +1,38 @@ +// @ts-nocheck — bundled with shared vocabulary by client.mjs. +import { SESSION_HOST_LABELS, SESSION_SURFACE_LABELS, sessionPresentation } from '../../session-surface.mjs'; +import { esc } from './bootstrap.mjs'; + +export function surfaceEntries(project){ + if(Array.isArray(project.sessionSurfaces))return project.sessionSurfaces.filter(function(row){return row.sessions>0;}); + return (project.sessionOrigins||[]).filter(function(row){return row.sessions>0;}); +} +export function surfaceNames(project){ + var entries=surfaceEntries(project); + return Array.from(new Set(entries.map(function(row){return sessionPresentation(row).label;}))).sort(); +} +export function surfaceRawText(origin){ + var raw=origin.rawEvidence||{},parts=[]; + ['entrypoint','originator','source','threadSource','sessionKind'].forEach(function(key){ + var values=Array.isArray(raw[key])?raw[key]:[raw[key]]; + values.slice(0,16).forEach(function(value){ + if(typeof value==='string'&&value.length<=80&&(value==='Codex Desktop'||/^[A-Za-z][A-Za-z0-9_.-]*$/.test(value)))parts.push(key+': '+value); + }); + }); + if(origin.rawEvidenceComplete===false)parts.push('additional raw declarations omitted by bound'); + return parts.join(' · '); +} +export function surfaceDetailHtml(origin){ + var p=sessionPresentation(origin),raw=surfaceRawText(origin); + return esc(p.label)+' · initiator: '+esc(p.initiator)+(p.note?' · '+esc(p.note):'')+(raw?' · '+esc(raw):'')+(Array.isArray(origin.attributes)?' · '+origin.attributes.filter(function(value){return ['on 3P','started from Claude Desktop','started from mobile','started from a project','started from web'].includes(value);}).map(esc).join(', '):''); +} +export function projectSurfacesHtml(project){ + var entries=surfaceEntries(project); + return '
    Session surfaces: '+esc(surfaceNames(project).join(', ')||'Unknown')+'' + +(entries.length?entries.map(function(row){return '
    Host: '+esc(Object.hasOwn(SESSION_HOST_LABELS,row.host)?SESSION_HOST_LABELS[row.host]:'Unknown')+' · '+surfaceDetailHtml(row)+' · provider: '+esc(sessionPresentation(row).provider)+' ('+esc(sessionPresentation(row).providerBasis)+')' + +' · '+esc(row.sessions)+' sessions ('+esc(row.countBasis||'legacy count basis unknown')+')
    ';}).join(''):'
    Host, initiator and provider: Unknown
    ')+'
    '; +} +export function surfaceFacetLabel(value){ + if(value==='unknown')return 'Legacy origin: no declared desktop origin'; + if(value==='surface-unknown')return 'Unknown session surface'; + if(value==='codex-desktop')return sessionPresentation({origin:value}).note; + return Object.hasOwn(SESSION_SURFACE_LABELS,value)?SESSION_SURFACE_LABELS[value]:'Unknown';} diff --git a/src/lib/dashboard/client/system-projects.mjs b/src/lib/dashboard/client/system-projects.mjs index d94fc149..ec77eba4 100644 --- a/src/lib/dashboard/client/system-projects.mjs +++ b/src/lib/dashboard/client/system-projects.mjs @@ -2,6 +2,8 @@ // @ts-nocheck — browser bundle source (never node-imported; client.mjs // reads it as text). See src/lib/dashboard/client/**'s eslint.config.mjs // override comment for why this directory isn't run through the node lib. +import { censusDisclosure } from '../../census-presentation.mjs'; +import { projectSurfacesHtml } from './session-presentation.mjs'; import { authHeaders, esc } from './bootstrap.mjs'; import { formatLocalDateTime, formatLocalDateTimeLong, shortSessionId } from './datetime.mjs'; import { ago } from './intelligence.mjs'; @@ -256,7 +258,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; if(!pm||pm.status==="unknown"||!Array.isArray(pm.value)){ procs.innerHTML=sysEmpty((pm&&pm.reason)||"the process census is unavailable."); }else if(!pm.value.length){ - procs.innerHTML=sysEmpty("no host process is running right now \u2014 a measured zero."); + procs.innerHTML=sysEmpty("no coding-agent or desktop-application process is running right now \u2014 a measured zero."); }else{ var rows=pm.value,maxRss=0,body=""; for(i=0;imaxRss)maxRss=rv;} @@ -271,7 +273,8 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; +esc(source.value.label||source.value.path)+"" : '' +esc(String((source&&source.reason)||"not attributable").split("\u2014")[0].trim())+""; - body+=''+esc(p.host)+"" + body+='' + +esc(p.application||p.host||"Unknown process")+"" +''+esc(String(p.pid))+"" +""+proj+"" +''+mhtml(p.uptimeMs,fmtDur)+"" @@ -282,7 +285,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; } // pid is right-aligned in the body, so its header is too — a numeric // column whose header hangs off the far side reads as a different column. - procs.innerHTML='
    ' + procs.innerHTML='
    Host
    ' +'' +'' +""+body+"
    Coding-agent host / desktop applicationpidWorking contextUptimeCPURSS
    "; @@ -783,7 +786,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; var control=opts.expandable?'':''; var worktreeMark=opts.worktree?'':''; var display=opts.worktree?''+esc(pr.label||'worktree')+'':name; - return ''+worktreeMark+display+''+esc(pr.path||'not measured yet')+'' + return ''+worktreeMark+display+''+esc(pr.path||'not measured yet')+''+projectSurfacesHtml(pr)+'' +''+mhtml(pr.loc&&pr.loc.total,function(v){return "~"+fmtTok(v);})+"" +""+langCell(pr.loc)+"" +''+mhtml(pr.totalBytes,fmtBytes)+"" @@ -801,7 +804,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; +". This view shows "+esc(fmtNum(tree.repositories.length))+" verified repositor"+(tree.repositories.length===1?"y":"ies") +" with "+esc(fmtNum(worktrees))+" nested worktree"+(worktrees===1?"":"s")+"; "+esc(fmtNum(tree.excludedDirectories))+" non-repository directories are excluded." +" Line counts are approximate: extension-bucketed, with node_modules and vendored " - +"trees excluded. Disk is the whole project directory, .git and node_modules included."; + +"trees excluded. Disk is the whole project directory, .git and node_modules included. "+esc(censusDisclosure(p))+""; } export function renderSysProjects(d){ @@ -810,7 +813,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; var p=d.projects; if(!p){el.innerHTML=sysEmpty(NOT_SCANNED);return;} var all=p.projects||[]; - if(!all.length&&!(p.discoveryProjects||[]).length){el.innerHTML=sysEmpty("no repository was discovered on this machine.");return;} + if(!all.length&&!(p.discoveryProjects||[]).length){el.innerHTML=sysEmpty("no repository was discovered on this machine.")+'
    '+esc(censusDisclosure(p))+"
    ";return;} var tree=repositoryTree({projects:all,discoveryProjects:p.discoveryProjects}); var byPath={};tree.repositories.forEach(function(group){byPath[group.repository.path]=group;}); var repositories=sortProjects(tree.repositories.map(function(group){return group.repository;}),projSort.key,projSort.dir); diff --git a/src/lib/dashboard/client/usage.mjs b/src/lib/dashboard/client/usage.mjs index 09d34e48..7c5f2359 100644 --- a/src/lib/dashboard/client/usage.mjs +++ b/src/lib/dashboard/client/usage.mjs @@ -1,6 +1,8 @@ // @ts-nocheck — browser bundle source (never node-imported; client.mjs // reads it as text). See src/lib/dashboard/client/**'s eslint.config.mjs // override comment for why this directory isn't run through the node lib. +import { SESSION_HOST_LABELS, sessionProviderPresentation } from '../../session-surface.mjs'; +import { surfaceDetailHtml } from './session-presentation.mjs'; import { formatLocalDateTime } from './datetime.mjs'; import { VIEWS, authHeaders, esc, setTab, syncHash } from './bootstrap.mjs'; import { ago } from './intelligence.mjs'; @@ -1103,23 +1105,18 @@ import { renderUsage } from './usage-orchestrators.mjs'; +''+esc(sub||resetTxt(resetSec))+""; } - // An empty Claude panel is explained by WHICH statusLine a session runs - // (#238 M3): the tee lives only in the kit footer. The server sends the - // user-level statusLine's class (claudeChannel, never its path); the copy - // states Claude Code's precedence rule, because a project's own statusLine - // overrides the user-level one — which is how a footer-carrying project - // still fills this panel when the user-level script cannot. An unknown or - // missing class (an older server) gets the generic sentence. - var CLAUDE_PRECEDENCE="a project’s own statusLine takes precedence over your user-level one"; + // The channel describes the resolved settings context; local and managed + // settings can override the project and user settings. + var CLAUDE_PRECEDENCE="local or managed settings may override the project and user settings; the effective statusLine takes precedence"; var CLAUDE_SETUP="Set a project up with ak setup --project (ak sync keeps its footer current), then run a Pro/Max session there."; var CLAUDE_EMPTY={ - "kit-footer":"your user-level statusLine carries the kit footer, so limits arrive after the first response " + "kit-footer":"the effective statusLine carries the kit footer, so limits arrive after the first response " +"of a Claude Code session on a Pro/Max plan. Run one session, then revisit.", - "custom":"your user-level statusLine runs a custom script without the kit footer, so it does not report limits " - +"to ak. They arrive only from sessions in projects whose own statusLine carries the footer: "+CLAUDE_PRECEDENCE+". "+CLAUDE_SETUP, - "none":"you have no user-level statusLine, so only sessions in projects whose own statusLine carries the kit footer " + "custom":"the effective statusLine runs a custom script without the kit footer, so it does not report limits " + +"to ak. They arrive from sessions whose effective statusLine carries the footer: "+CLAUDE_PRECEDENCE+". "+CLAUDE_SETUP, + "none":"there is no effective statusLine, so only sessions whose effective statusLine carries the kit footer " +"report limits. "+CLAUDE_SETUP, - "project-helper":"your user-level statusLine runs each project’s own ruflo helper, so limits arrive from sessions " + "project-helper":"the effective statusLine runs each project’s own ruflo helper, so limits arrive from sessions " +"in projects where that helper carries the kit footer. Run ak sync in such a project to re-inject it, " +"then run a Pro/Max session there." }; @@ -1328,7 +1325,7 @@ import { renderUsage } from './usage-orchestrators.mjs'; // not exist, when in fact it was measured and found absent (ADR-0009 §5). function dash(v){return (v==null||v==="")?"—":String(v);} function reportedIdentity(v){v=String(v==null?"":v).trim();return v&&!/^unknown$/i.test(v)?v:null;} - function identityName(v){var raw=reportedIdentity(v);if(!raw)return"Not recorded";return{claude:"Claude Code",codex:"Codex",opencode:"OpenCode",anthropic:"Anthropic",openai:"OpenAI",openrouter:"OpenRouter",bedrock:"AWS Bedrock",vertex:"Google Vertex AI",foundry:"Microsoft Foundry",gateway:"Custom gateway",ollama:"Ollama",lmstudio:"LM Studio"}[raw.toLowerCase()]||raw;} + function identityName(v){return Object.hasOwn(SESSION_HOST_LABELS,v)?SESSION_HOST_LABELS[v]:'Unknown';} // ── per-session chips ───────────────────────────────────────────────────── // Evidence the row already carries, shown only where the transcript @@ -1403,7 +1400,7 @@ import { renderUsage } from './usage-orchestrators.mjs'; ? ' (conf '+esc(sx.confidence.toFixed(2))+")" : ""; var modelList=(Array.isArray(sx.models)?sx.models:[]).filter(function(model){return reportedIdentity(model);}); var models=modelList.length?modelList.join(", "):"Not recorded"; - var providerRaw=reportedIdentity(sx.provider),provider=identityName(providerRaw),provenance=reportedIdentity(sx.providerProvenance)||"unknown",providerContext=providerRaw?provenance+" evidence":"not established by source"; + var providerPresentation=sessionProviderPresentation(sx),provider=providerPresentation.label,providerContext=providerPresentation.basis; var toks="in "+fmtTok(sx.input)+" · out "+fmtTok(sx.output) +" · cache r "+fmtTok(sx.cacheRead)+" / w "+fmtTok(sx.cacheWrite) // Codex-only detail: reasoning tokens are a SUBSET of output (they bill @@ -1431,7 +1428,7 @@ import { renderUsage } from './usage-orchestrators.mjs'; // codex or opencode transcript can record an interrupt, so a claude row // reads "not recorded" rather than a measured-looking 0. +" · aborts "+(sx.host==="codex"||sx.host==="opencode"?fmtNum(Number(sx.aborts)||0):"not recorded for this host"); - var rows=[["execution host",esc(identityName(sx.host))],["inference provider",esc(provider)+" ("+esc(providerContext)+")"],["models",esc(models)],["posture",posture],["rhythm",esc(rhythm)],["basis",esc(basis)+conf],["tokens",esc(toks)], + var rows=[["session surface",surfaceDetailHtml(sx.sessionOrigin||{})],["execution host",esc(identityName(sx.host))],["inference provider",esc(provider)+" ("+esc(providerContext)+")"],["models",esc(models)],["posture",posture],["rhythm",esc(rhythm)],["basis",esc(basis)+conf],["tokens",esc(toks)], ["tools",esc(tools)],["flags",esc(flags)]]; return '