diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 80141851..75936b0a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -2,7 +2,7 @@ name: ci on: push: - branches: [main, npm-kit] + branches: [main, develop, npm-kit] pull_request: workflow_dispatch: @@ -60,7 +60,7 @@ jobs: # The smoke test proves local CLI behavior, not npm reachability. # Seed a fresh empty drift cache so an offline Windows runner does # not pay four sequential 20s `npm view` timeouts inside status. - node -e 'const fs=require("node:fs"),p=require("node:path"),base=process.platform==="win32"?process.env.APPDATA:process.env.XDG_CONFIG_HOME,dir=p.join(base,"agentic-kit"),last=Date.now();fs.mkdirSync(dir,{recursive:true});fs.writeFileSync(p.join(dir,"kit.json"),JSON.stringify({versionCheck:{last,seen:{},self:{last,best:null}}}))' + node -e 'const fs=require("node:fs"),p=require("node:path"),base=process.platform==="win32"?process.env.APPDATA:process.env.XDG_CONFIG_HOME,dir=p.join(base,"agentic-kit"),last=Date.now();fs.mkdirSync(dir,{recursive:true});fs.writeFileSync(p.join(dir,"kit.json"),JSON.stringify({versionCheck:{last,seen:{},self:{last,best:null,lastTags:["latest","next"]}}}))' node bin/agentic-kit.mjs --version node bin/agentic-kit.mjs --help --all > /dev/null # status must emit valid JSON and exit deterministically even on a @@ -89,6 +89,8 @@ jobs: - name: Install devDependencies run: pnpm install --frozen-lockfile + - name: Tracked JavaScript comment and test-title guard + run: pnpm run test:quality - name: Typecheck (tsc --checkJs) run: pnpm run typecheck - name: Lint (eslint) diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index 92ffece7..be9bd50d 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -88,7 +88,12 @@ jobs: # same teardown abort on store-touching commands (`memory search` → correct # output, rc 134), so an `--only memory-routes` step added here would need the same # guard. Remove continue-on-error once that issue closes. + - name: Require Node 22.15+ for the macOS resolution hook + if: matrix.os == 'macos-latest' + run: node -e "const [major, minor] = process.versions.node.split('.').map(Number); if (major < 22 || (major === 22 && minor < 15)) { console.error('trace-ort requires Node 22.15+'); process.exit(1); }" + - name: Deep proof against the live packages (learning) + if: matrix.os != 'macos-latest' continue-on-error: true env: HOME: ${{ runner.temp }}/kit-home @@ -97,6 +102,52 @@ jobs: APPDATA: ${{ runner.temp }}/kit-home/AppData/Roaming run: node bin/agentic-kit.mjs status --refresh=live --only learning + - name: Deep proof against the live packages (learning, traced macOS) + if: matrix.os == 'macos-latest' + continue-on-error: true + shell: bash + env: + HOME: ${{ runner.temp }}/kit-home + USERPROFILE: ${{ runner.temp }}/kit-home + XDG_CONFIG_HOME: ${{ runner.temp }}/kit-home/.config + APPDATA: ${{ runner.temp }}/kit-home/AppData/Roaming + NODE_OPTIONS: --import=${{ github.workspace }}/scripts/trace-ort.mjs + TRACE_ORT_LOG: ${{ runner.temp }}/trace-ort.jsonl + run: | + set +e + node bin/agentic-kit.mjs status --refresh=live --only learning + learning_rc=$? + LEARNING_RC="$learning_rc" NODE_OPTIONS='' node --input-type=module -e ' + import fs from "node:fs"; + import crypto from "node:crypto"; + const source = fs.readFileSync("scripts/trace-ort.mjs"); + fs.writeFileSync(process.env.RUNNER_TEMP + "/trace-ort-receipt.json", JSON.stringify({ + sourceSha: process.env.GITHUB_SHA, + hookSha256: crypto.createHash("sha256").update(source).digest("hex"), + node: process.version, platform: process.platform, arch: process.arch, + learningExitCode: Number(process.env.LEARNING_RC), + tracePresent: fs.existsSync(process.env.TRACE_ORT_LOG), + }) + "\n"); + ' + exit "$learning_rc" + + - name: Check macOS trace artifact + if: always() && matrix.os == 'macos-latest' + shell: bash + run: | + test -s "$RUNNER_TEMP/trace-ort.jsonl" || { echo '::error::trace-ort artifact absent or empty'; exit 1; } + test -s "$RUNNER_TEMP/trace-ort-receipt.json" || { echo '::error::trace-ort receipt absent or empty'; exit 1; } + + - name: Upload macOS learning resolution trace + if: always() && matrix.os == 'macos-latest' + uses: actions/upload-artifact@v7 + with: + name: macos-learning-ort-trace + if-no-files-found: error + path: | + ${{ runner.temp }}/trace-ort.jsonl + ${{ runner.temp }}/trace-ort-receipt.json + clean-mac-setup: name: clean macOS setup (packed artifact) runs-on: macos-latest @@ -134,7 +185,7 @@ jobs: ollama serve > "$RUNNER_TEMP/ollama.log" 2>&1 & AK_OLLAMA_PID=$! trap 'kill "$AK_OLLAMA_PID" 2>/dev/null || true' EXIT - for attempt in {1..30}; do + for ((ak_readiness_attempt=0; ak_readiness_attempt<30; ak_readiness_attempt++)); do curl --fail --silent http://127.0.0.1:11434/api/version >/dev/null && break sleep 1 done @@ -142,6 +193,7 @@ jobs: git -C "$AK_PROJECT" init (cd "$AK_PROJECT" && ak setup --yes --no-ruvnet-brain --aqe-embedding-mode local) | tee "$RUNNER_TEMP/setup.log" (cd "$AK_PROJECT" && ak x aqe-embedding verify --json) | tee "$RUNNER_TEMP/embedding-proof.json" + # shellcheck disable=SC2016 # JavaScript template literals are evaluated by Node. node --input-type=module -e ' import fs from "node:fs"; import path from "node:path"; diff --git a/.github/workflows/upstream-watch.yml b/.github/workflows/upstream-watch.yml index 79f96873..38f361c3 100644 --- a/.github/workflows/upstream-watch.yml +++ b/.github/workflows/upstream-watch.yml @@ -58,8 +58,10 @@ jobs: set -e { echo "## Upstream watch preview (exit $code)" - jq -r '"since \(.since) (\(.sinceSource)), new records \(.records | length), could not check \(.fetchErrors | length), blind \(.blind), notice \(.notice.post), would fire \(.wouldFire | length)"' watch.json + jq -r 'if .blind then "blind \(.blind): \(.error // "unknown error"), could not check \(.fetchErrors | length), dispatch errors \(.dispatchErrors | length), deferred \(.deferred | length)" else "since \(.since) (\(.sinceSource)), new records \(.records | length), could not check \(.fetchErrors | length), dispatch errors \(.dispatchErrors | length), blind \(.blind), notice \(.notice.post), would fire \(.wouldFire | length), deferred \(.deferred | length)" end' watch.json jq -r '(.wouldFire // [])[] | "- would fire \(.id) \(.version) \(.branch)"' watch.json + jq -r '(.deferred // [])[] | "- deferred \(.id) \(.version) \(.branch)"' watch.json + jq -r '(.fired // [])[] | "- observed session before ledger failure: \(.id) \(.fields.session)"' watch.json echo; echo '```text'; cat errors.txt; echo '```' } >> "$GITHUB_STEP_SUMMARY" exit "$code" @@ -102,9 +104,10 @@ jobs: set -e { echo "## Upstream watch (exit $code)" - jq -r '"since \(.since) (\(.sinceSource)), new records \(.records | length), could not check \(.fetchErrors | length), dispatch errors \(.dispatchErrors | length), blind \(.blind), commit \(.commit // "none"), would fire \(.wouldFire | length), deferred \(.deferred | length)"' watch.json + jq -r 'if .blind then "blind \(.blind): \(.error // "unknown error"), could not check \(.fetchErrors | length), dispatch errors \(.dispatchErrors | length), deferred \(.deferred | length)" else "since \(.since) (\(.sinceSource)), new records \(.records | length), could not check \(.fetchErrors | length), dispatch errors \(.dispatchErrors | length), blind \(.blind), commit \(.commit // "none"), would fire \(.wouldFire | length), deferred \(.deferred | length)" end' watch.json jq -r '(.wouldFire // [])[] | "- would fire \(.id) \(.version) \(.branch)"' watch.json jq -r '(.deferred // [])[] | "- deferred \(.id) \(.version) \(.branch)"' watch.json + jq -r '(.fired // [])[] | "- observed session before ledger failure: \(.id) \(.fields.session)"' watch.json echo; echo '```text'; cat errors.txt; echo '```' jq -r '.notice.body' watch.json } >> "$GITHUB_STEP_SUMMARY" @@ -121,7 +124,8 @@ jobs: if: env.RECORD == 'true' run: | [ "$(jq -r '.notice.post' watch.json)" = true ] || { echo 'Nothing needs the maintainer.'; exit 0; } - commit=$(jq -r '.commit' watch.json) + commit=$(jq -r '.commit // empty' watch.json) + [ -n "$commit" ] || { echo 'Notice requested without a ledger commit.' >&2; exit 1; } jq -r '.notice.body' watch.json > notice.md test -s notice.md gh api "repos/$GITHUB_REPOSITORY/commits/$commit/comments" -F body=@notice.md --jq .html_url diff --git a/AGENTS.md b/AGENTS.md index c0d3f693..8f613e30 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -307,7 +307,7 @@ compression, or neural-routing targets are not measured agentic-kit guarantees. pnpm test # One focused suite -node --test tests/kit/dispatch-surface.test.mjs +node scripts/run-tests.mjs focus tests/kit/dispatch-surface.test.mjs # Browser verification pnpm run test:ui @@ -320,6 +320,8 @@ pnpm run lint:md pnpm run build ``` +A plain `node --test` run lacks the wrapper's real-state tripwire and temp-root checks. + `pnpm test` and `pnpm run test:ui` run through `scripts/run-tests.mjs`, which fingerprints `~/.config/agentic-kit`, `~/.local/state/agentic-kit` (or `%APPDATA%`/`%LOCALAPPDATA%` on Windows), `~/.claude/CLAUDE.md`, `~/.claude/settings.json`, `~/.claude.json`, @@ -332,8 +334,20 @@ them, and `sandboxHome()` and `redirectToolState()` do the same for in-process c Code's own `~/.claude.json`) are listed as "concurrent writers" and do not fail a local run; CI (or `AK_TRIPWIRE_STRICT=1`) fails on them too. Every command also runs with `TMPDIR`/`TEMP`/`TMP` pointed at a fresh `ak-suite-*` folder: anything left in it afterwards fails the run and is -listed, and the runner refuses to start when that folder sits inside a git repository (point -`TMPDIR` elsewhere). The runner also drops `FORCE_COLOR` (Claude Code shells set it), because +listed (excluding its private atomic `.ak-suite-owner.json`, child-hold directory and Node compile cache). The runner +refuses home/filesystem-root temp bases before allocation and refuses roots inside a git +repository (point `TMPDIR` elsewhere). A completed run removes only its own validated direct, +canonical, nonsymlink, current-owner root. Tests with known child lifetime uncertainty acquire +`acquireRunRootHold()` before launching those children and release only after proving their exits. +An unresolved or unreadable hold retains the own root; it is not a general descendant-exit proof. +The runner then lists sibling suite roots: missing, invalid, +foreign or uncertain owner metadata means keep. Sibling handling is list-only on macOS, Linux +and Windows because no installed probe proves all descendants have exited; even a dead owner +is insufficient. Interrupted runs remove and collect nothing. Sibling listing/collection errors +do not change the suite's exit code. Own-root inspection failure retains the root; inspection, +removal or safety-refusal failure returns hygiene exit 4 unless a command or tripwire failure +already takes precedence. Removal errors may leave a partially removed own root. The runner also +drops `FORCE_COLOR` (Claude Code shells set it), because tests read plain text from pipes. Tests make temporary folders with `tempDir()` from `tests/kit/helpers/temp-dir.mjs`, and spawned children get their environment from `spawnEnv()` in `tests/kit/helpers/home-sandbox.mjs`. UI tests launch Chrome with `launchChrome()` from diff --git a/README.md b/README.md index bc502fb3..cefb605d 100644 --- a/README.md +++ b/README.md @@ -147,7 +147,7 @@ and current platform limits. | **setup** | Installs/updates ruflo + agentic-qe globally (handling npm ≥11.17's `allow-scripts` so natives build; AgentDB ships inside ruflo, so ak installs no separate copy), installs and verifies the exact Ruflo-compatible **agent-browser** native executor without adding its plugin/skills (`--no-agent-browser` disables it), installs the **RuvNet Brain** (an offline knowledge base over the rUv stack, powering the `search_ruvnet` MCP — a ~2 GB one-time download, prompted; skip with `--no-ruvnet-brain`), deploys the token-audit skill, merges the managed guidance blocks into the machine-wide guidance files (`~/.claude/CLAUDE.md`, plus `~/.codex/AGENTS.md` on codex machines), offers one-time MCP registration (user scope, with a tool-family picker), and — inside a repo — initializes the project: sanitized `ruflo init`, absolute memory-path pin, a **verified** store→disk write, statusline footer, and a background daemon with **local-only ($0) workers** (token-spending AI workers stay opt-in behind upstream's machine-wide budget). Project scope triggers on a `.git` entry in the current folder; without one it's skipped with a note. `--project` forces the same project setup in the current directory (e.g. a not-yet-`git init`-ed folder); it does not locate an ancestor repository. Project initialization runs `ruflo init --full --force` and can replace existing agent configuration, so read the [setup scope and project mutation contract](docs/setup.md) before using it on an existing project. `--minimal` skips it, `--yes` accepts all prompts (non-interactive), `--no-aqe` / `--no-agent-browser` / `--no-ruvnet-brain` / `--no-security` disable those subsystems, and `--reconfigure` re-offers MCP registration. `--codex` enables + installs the Codex host during setup (ambidextrous dual-host mode; both hosts become available for routing), and `--primary-host claude\|codex` picks which host leads (codex implies `--codex`). | | **status** | Per-subsystem ✓/⚠/✗ (versions, the kit's own version, **ruvnet-brain**, natives, **memory-pin**, security, learning, aqe/RVF, the managed **agent-browser** package/native/config/browser readiness, MCP, **hosts**, **providers**, **routing**, daemons, guidance blocks, statusline), each drift row naming what `sync` would do about it, or marking a step you must take yourself as `→ manual:` (sync never plans those) — plus a **health-history** line that flags regressions since the last sync. Browser status is filesystem-only: it never runs doctor or launches Chrome. | | **sync** | The one convergence verb: upgrades first when a new release exists, then re-heals everything an upgrade wipes, then re-checks and reports. Included in that heal: it **installs any enabled frontier host** (claude/codex/opencode) that's entirely absent — never touching an external (mise/brew/native) install — and **re-applies provider wiring** (the `ENABLE_*` host env, OpenCode's native configuration, the AQE default/fallback/agent overrides, admitted Agentic-QE 3.13.12+ `externalProviders`, and ruflo API providers) whenever it has drifted. External-provider reconciliation preserves foreign entries, refuses same-id conflicts, and prunes only entries whose exact value still matches an agentic-kit ownership receipt. On a dual-host project, sync also **seeds/heals the Claude/Codex default routing policy**. It appends a health-history snapshot, refreshes RuvNet Brain when enabled, and self-updates the kit last. A planned fix whose status row is still there afterwards is reported `unresolved:` and sync exits 1. A row whose fix you do by hand (`→ manual:`) never changes sync's exit code; a failing or warning one is listed under "needs your action". `--no-upgrade` skips self-update and package upgrades. `--skip ` (repeatable) leaves one subsystem out of this run only, including the step it owns; it is reported "skipped by request" and never counts as a failure. `--json` prints one JSON result on stdout (`plan`, `steps`, `unresolved`, `skipped`, `needsYourAction`, `converged`, `exitCode`) and sends the human lines to stderr. Model refresh/diff/plan findings remain advisory. | -| **dashboard** | Opens the local web dashboard (`127.0.0.1:7431`, localhost-only, never detaches) with five primary areas: **About · Overview · Usage · Observability · System**. Ordinary views remain observation-only. System's **Full scan** remeasures local inventory and then chains one provider check. **System → Maintenance** is the sole action surface, with four destinations: **Inventory** (scope → repository where applicable → type → resource → exact installation), **Guidance** (only outcomes the kit can ground), **Discovery** (where it looks), and **Activity** (receipts, undo, interruption audits). Every write is one exact placement and one action, with a server-derived short-lived plan, explicit confirmation, and a one-use capability. Advisory remains a measurement; the former Catalog tab redirects to Inventory and its cards now sit in System Summary. The page is self-contained, offline-first, protected by a per-session token, and never executes a browser-supplied command. Action targets resolve server-side; Discovery accepts validated source-root configuration. Full navigation and security semantics: [Dashboard guide](docs/dashboard.md); provider and recovery limits: [Maintenance runbook](docs/maintenance.md). **Auto-opens your browser** (`--no-open` for headless/SSH); `--port N` changes the port. Stop with Ctrl-C. (Also available as `ak x dashboard`.) | +| **dashboard** | Opens the local web dashboard (`127.0.0.1:7431`, localhost-only, never detaches) with five primary areas: **About · Overview · Usage · Observability · System**. Ordinary views remain observation-only. Choose **Refresh machine** in the header and press **Refresh** to remeasure the machine, refresh Maintenance evidence, and rebuild the inventory. **System → Maintenance** is the sole action surface, with four destinations: **Inventory** (scope → repository where applicable → type → resource → exact installation), **Guidance** (only outcomes the kit can ground), **Discovery** (where it looks), and **Activity** (receipts, undo, interruption audits). Every write is one exact placement and one action, with a server-derived short-lived plan, explicit confirmation, and a one-use capability. Advisory remains a measurement; the former Catalog tab redirects to Inventory and its cards now sit in System Summary. The page is self-contained, offline-first, protected by a per-session token, and never executes a browser-supplied command. Action targets resolve server-side; Discovery accepts validated source-root configuration. Full navigation and security semantics: [Dashboard guide](docs/dashboard.md); provider and recovery limits: [Maintenance runbook](docs/maintenance.md). **Auto-opens your browser** (`--no-open` for headless/SSH); `--port N` changes the port. Stop with Ctrl-C. (Also available as `ak x dashboard`.) | | **usage** | `score` and `prompts` summarize retained local transcript evidence. `status` reads provider-account analytics from cache; `refresh openrouter` explicitly contacts the OpenRouter management API using `OPENROUTER_MANAGEMENT_KEY`, then writes a credential-free mode-`0600` cache. Cache reads make no OpenRouter request. Account rows have no grounded host/session/project correlation and are never merged into transcript totals. | | **models** | Builds a private, host-scoped model inventory from Claude, Codex, OpenCode, Ollama, bounded local usage evidence, and a dated bundled record of Anthropic's public model/lifecycle facts. `status`, `diff`, `explain`, and `plan` are cache-only and read-only; `refresh --online` is the sole online-catalogue boundary. Public facts never imply account or OpenRouter routability. Swap plans enumerate routes plus Agentic QE/Ruflo consumers and print a copyable canonical action without executing it. The CLI exposes exact local evidence deliberately; the Dashboard exposes source-proven public catalogue identity and uses the owner-visible model read contract; secret-shaped values remain masked. See [Model lifecycle intelligence](docs/models.md). | | **admin** | Opens the **maintainer admin** (`127.0.0.1:7432`, localhost-only, foreground) — the project-telemetry sibling of `dashboard`, with the same dark/light visual theme and persisted theme preference: unique repo visitors and cloners (GitHub traffic API, needs a push-access token via `GITHUB_TOKEN`/`GH_TOKEN`/`gh auth token` — panels degrade honestly without one), contributors and watchers, npm download momentum (last 7d vs prior 7d, sparklines — shown as trend only, never an absolute reach number, since mirrors/CI inflate the raw count), latest CI run status and open Dependabot alerts, a **"since you last looked"** delta strip over a local baseline, open issues/PRs from others (oldest first), and external humans ranked by recency (bots excluded). Access is gated by a **per-session token** carried in the URL fragment and sent header-only; the page makes **zero external fetches** (the server proxies GitHub/npm; your credential never reaches the page or the payload). Where `dashboard` is offline-first, `admin` does deliberate GitHub/npm egress — that contract split is why they're siblings, not tabs. `--port N`, `--no-open`; Ctrl-C stops. (Also available as `ak x admin`.) | diff --git a/bin/agentic-kit.mjs b/bin/agentic-kit.mjs index d18c1ba6..b73b87c9 100755 --- a/bin/agentic-kit.mjs +++ b/bin/agentic-kit.mjs @@ -194,7 +194,9 @@ async function main() { } catch (err) { if (!String(err?.code ?? '').startsWith('ERR_PARSE_ARGS_')) throw err; if (cmd === 'telemetry') { - console.error('Telemetry failed: invalid command options.'); + const error = 'Telemetry failed: invalid command options.'; + console.error(error); + console.log(JSON.stringify({ error, exitCode: 2 })); return 2; } // Under --json a rejected option still answers with one JSON object (the diff --git a/docs/adr/0009-usage-scorecard-local-transcript-analytics.md b/docs/adr/0009-usage-scorecard-local-transcript-analytics.md index f55a270d..94eb15f8 100644 --- a/docs/adr/0009-usage-scorecard-local-transcript-analytics.md +++ b/docs/adr/0009-usage-scorecard-local-transcript-analytics.md @@ -2,6 +2,13 @@ - **Status:** Implemented - **Date:** 2026-07-25 +- **Updated:** 2026-09-29 — usage schema 25 → 26 delivers bounded session surface/provider + evidence, per-turn Codex import ownership, positive component usage without responses, + one Claude message charge owner across the bounded two-window pool, separate host-reported + reconciliation signals, explicit OpenCode source/cost/coverage semantics and timezone-aware + cache reuse. Footprint remains schema 8. The + [current accounting contracts](../usage-scorecard-metrics.md#current-accounting-and-cache-contracts) + define source bounds and compatibility; historical measurements below are not fresh results. - **Updated:** 2026-09-20 — ADR-0054 adds an explicit, offline, allowlisted fleet export boundary; local analytics and dashboard collection semantics remain unchanged. - **Earlier update:** 2026-09-09 — reconciled against repository source and tests for issue #211 diff --git a/docs/adr/0012-observability.md b/docs/adr/0012-observability.md index 90adc04f..c095b68b 100644 --- a/docs/adr/0012-observability.md +++ b/docs/adr/0012-observability.md @@ -670,3 +670,15 @@ turns the dashboard into a fleet service nor makes telemetry collection continuo - **Exact-folder leases (#238 item 2).** A runtime lease no longer requires a Git repository when an exact-folder match joins the process to its transcript; see the 2026-08-03 runtime identity amendment above. + +### 2026-09-29 follow-up: bounded re-entry and structured-source evidence + +An idle stop retains active tailer offsets. A native transcript displaced from the newest-file +window also retains its reader state in memory, up to twice the configured file bound (at least +two dormant readers). Re-entry within that bound resumes without replaying accepted records. +After dormant eviction, re-entry reads from the start; a new dashboard process has no persisted +native offset and bootstraps existing-file metadata before following new appends. This does not +create durable exactly-once delivery. + +The optional Ruflo and agentic-qe structured live-events input is experimental. Parser fixtures +exist, but no real producer has been verified. An explicit source path does not prove activity. diff --git a/docs/adr/0025-machine-footprint-metrics.md b/docs/adr/0025-machine-footprint-metrics.md index fa112028..5dacb494 100644 --- a/docs/adr/0025-machine-footprint-metrics.md +++ b/docs/adr/0025-machine-footprint-metrics.md @@ -1,7 +1,11 @@ # ADR-0025 — Machine footprint: infrastructure metrics for install, runtime, storage, and catalog - **Status:** Implemented -- **Updated:** 2026-09-28 — CLI parity (§5) is now `ak system [--refresh[=live|machine]] +- **Updated:** 2026-09-29 — §5 deep measurement now starts with `POST /api/refresh` at + `machine` strength; GET routes are passive. The old GET-started scan rationale is withdrawn + because a read request must not start measurement work. See + [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md). +- **Earlier update:** 2026-09-28 — CLI parity (§5) is now `ak system [--refresh[=live|machine]] [--project-trees] [--json]` (`src/lib/refresh.mjs`'s shared strengths, replacing the retired `--deep`); §5's `GET /api/system?refresh=deep` rationale is untouched by this branch and belongs to a later remediation program's dashboard work (remediation program, branch 6b; see @@ -322,14 +326,14 @@ silent "Other" slice into a to-do list a release can close. ([ADR-0014](0014-dashboard-auth-and-remediation.md)), and zero egress ([ADR-0007](0007-maintainer-admin-local-telemetry.md)'s offline side of the line) as every other dashboard route. -- `GET /api/system?refresh=deep` — starts or attaches to the single-flight deep scan. The - dashboard server is deliberately GET-only; a refresh is a re-*measurement* of local state, not - a mutation of user data, so it stays within that contract. `&trees=1|0` sets whether that scan - walks project working trees; it is a **measurement** parameter, not a view filter, because - trees that were never walked cannot be un-hidden client-side. +- `POST /api/refresh` with `{"strength":"machine","projectTrees":true}` starts + the staged single-flight refresh; `projectTrees` is an optional boolean measurement choice. + The earlier `GET /api/system?refresh=deep&trees=1|0` trigger and its GET-only rationale + are withdrawn: a GET must never start scan work, even when the work only measures local + state. Trees that were never walked cannot be un-hidden client-side. - `ak system [--refresh[=live|machine]] [--project-trees] [--json]` — CLI parity sharing the same collector, following the usage-scorecard precedent of one collector behind both surfaces - (`--refresh=machine` is the CLI equivalent of the `?refresh=deep` route below). + (`--refresh=machine` selects the same strength as the dashboard POST). - `GET /api/system/summary` (amendment, 2026-09-26; extended 2026-09-28) — the page's read: the same payload and parameters with `catalog`, `storage`, `install`, `projects` and `consumers` each projected to an allow-list of keys and items cut to what the page draws (including @@ -568,7 +572,8 @@ The draft left four points open. All four are decided; this section is the recor large corpus — the surprise cost is worse than a stale figure that says how stale it is. The snapshot's `asOf` is always rendered, and beyond `SNAPSHOT_STALE_AFTER_MS` (7 days) the freshness label turns amber and reads "stale, rescan". Opening the System tab issues a plain - `GET /api/system/summary`; only the Rescan control adds `?refresh=deep`. + `GET /api/system/summary`; only explicit Refresh machine starts a measurement with + `POST /api/refresh`. 4. **Windows ships a current-user census plus a best-effort true `cwd`, degrading honestly, with no dependency added.** The draft's "unsupported on win32" answer would have blanked the whole Runtime view on a supported platform. Instead `src/lib/live/win-process-survey.ps1` — a plain text diff --git a/docs/adr/0044-receipt-aware-maintenance-control-plane.md b/docs/adr/0044-receipt-aware-maintenance-control-plane.md index 63b02baa..9280e4aa 100644 --- a/docs/adr/0044-receipt-aware-maintenance-control-plane.md +++ b/docs/adr/0044-receipt-aware-maintenance-control-plane.md @@ -2,7 +2,9 @@ - **Status:** Implemented - **Date:** 2026-09-03 -- **Updated:** 2026-09-28 — the CLI verb for the explicit scan this ADR's v1 `ak maintain scan` +- **Updated:** 2026-09-29 — the dashboard explicit scan starts with `POST /api/refresh`; + the retired `GET /api/maintenance?refresh=scan` is rejected. See ADR-0063. +- **Earlier update:** 2026-09-28 — the CLI verb for the explicit scan this ADR's v1 `ak maintain scan` contract describes is now `ak maintain --refresh[=machine]` (the exact `?refresh=scan` dashboard route below is unaffected — that half of the vocabulary belongs to a later remediation program's dashboard work; see [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md)) @@ -232,9 +234,9 @@ a body no larger than 64 KiB. The SSE query-token exception does not apply. The read-only interruption audit and separately confirmed single-receipt reconciliation. Plain GET /api/maintenance reads the latest persisted scan report and never polls a -provider. The exact ?refresh=scan query performs and atomically persists a provider scan; -other or duplicate query parameters are rejected. The global browser poll remains passive. A -successful persisted System deep rescan chains exactly one Maintenance scan. See ADR-0045. +provider. An explicit `POST /api/refresh` runs the provider scan as its Maintenance stage and persists +the report; `GET /api/maintenance?refresh=scan` is rejected. The global browser poll remains +passive. A successful machine measurement precedes one Maintenance scan. See ADR-0045. The view groups **Updates ready**, **Safe cleanup**, **Needs review**, **Unsupported or blocked**, and **Recent changes / Undo**. Every row exposes a direct imperative; the selected finding adds its diff --git a/docs/adr/0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md b/docs/adr/0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md index e7959521..5c1cef1b 100644 --- a/docs/adr/0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md +++ b/docs/adr/0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md @@ -2,7 +2,10 @@ - **Status:** Implemented - **Date:** 2026-09-03 -- **Updated:** 2026-09-09 — reconciled against repository source and tests for issue #211 +- **Updated:** 2026-09-29 — the explicit provider scan now starts through + `POST /api/refresh`; `GET /api/maintenance` only reads, and the retired + `?refresh=scan` GET trigger is superseded by ADR-0063. +- **Earlier update:** 2026-09-09 — reconciled against repository source and tests for issue #211 - **Earlier update:** 2026-09-04 — proposed ADR-0048 retains physical artifact and consumer-binding identity, adds exact management placements, and plans configurable/resumable discovery; current explicit provider-scan behavior remains authoritative until implementation @@ -24,9 +27,9 @@ Artifact/consumer identity and the explicit provider-scan contract remain current. The **Browser refresh / Scan now** wording below describes the v1 view; current -Maintenance uses the consolidated measurement toolbar and v2 scan routes under -ADR-0048. Neither passive report/query reads nor filesystem discovery grant -provider mutation authority. FootprintSnapshot is now v7; v6 below records the +Maintenance uses the single Refresh control and explicit POST operation under +ADR-0063, alongside ADR-0048's separate v2 scan routes. Neither passive report/query +reads nor filesystem discovery grant provider mutation authority. FootprintSnapshot is now v7; v6 below records the Catalog v4 migration. The independent configurable Discovery scan does not replace the Footprint deep worker or the explicit provider scan. @@ -93,15 +96,16 @@ a session. Those would require host-native runtime receipts. ### Make scanning explicit and browser refresh passive -Maintenance has two read paths: +Maintenance has one report read and one explicit refresh start: - GET /api/maintenance reads the latest private persisted report. It does not call a host CLI, provider, registry, network source, or version detector. -- GET /api/maintenance?refresh=scan performs one explicit provider scan, persists the - resulting report atomically, and returns it. Unknown or duplicate query parameters fail closed. +- `POST /api/refresh` explicitly runs the Maintenance evidence stage, persists its provider + scan report, and then rebuilds inventory. A GET with the retired `?refresh=scan` + query is rejected. -The dashboard labels these controls **Browser refresh** and **Scan now**. The global poll clock uses -the first path. **Scan now** uses the second. A successful persisted System deep rescan chains one +The dashboard uses **Reload** to re-read a view and **Refresh** to start the staged POST +operation. The global poll clock only reads. A successful persisted System deep rescan chains one Maintenance provider scan so inventory and provider evidence converge without double-scanning concurrent callers. diff --git a/docs/adr/0048-inventory-led-maintenance-resource-management.md b/docs/adr/0048-inventory-led-maintenance-resource-management.md index a3f43266..06e67562 100644 --- a/docs/adr/0048-inventory-led-maintenance-resource-management.md +++ b/docs/adr/0048-inventory-led-maintenance-resource-management.md @@ -1,8 +1,13 @@ # ADR-0048 — Inventory-led Maintenance resource management - **Status:** Accepted — implementation delivered 2026-09-05; Implemented withheld pending - human-evaluation and cross-platform gates -- **Updated:** 2026-09-28 — Branch 6b gives this ADR's two scan controls CLI equivalents: + human usability, screen-reader, and cross-platform evaluation gates deferred to v5; + the former Refresh evidence and Re-measure machine controls are superseded by ADR-0063 +- **Updated:** 2026-09-29 — [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md) replaces the + two dashboard scan controls with one Refresh control; its three visible choices start the + shared staged POST operation. D-15 moves the human usability, screen-reader, and + cross-platform evaluation gates to v5; automated checks do not satisfy them. +- **Earlier update:** 2026-09-28 — Branch 6b gives this ADR's two scan controls CLI equivalents: `ak maintain --refresh` and `ak maintain --refresh=machine` (`src/lib/refresh.mjs`'s `--refresh[=live|machine]`, `ak status --help` for the shared stages). The `scan` verb, the `scans start`/`plan --deep`/`--refresh-inventory` re-measure flags, and `ak maintain recipes @@ -113,7 +118,8 @@ The Focus browser amendment approved on 2026-09-08 is implemented and passes foc approved prototype establishes interaction intent, not production or adapter completeness. It is not yet Implemented in this record's own sense, because the [live acceptance criteria](../maintenance-acceptance.md)'s human-evaluation and -cross-platform acceptance gates have not run on this machine. The dashboard's Maintenance panel now +cross-platform acceptance gates have not run on this machine and are deferred to v5 under +D-15. The dashboard's Maintenance panel now renders this ADR's Inventory/Guidance/Discovery/Activity workspace; ADR-0044's v1 HTTP routes and CLI verbs remain available as a documented compatibility surface until those gates pass and this record is updated again. ADR-0044 is not marked Superseded; see "Implementation status" for why. @@ -121,7 +127,8 @@ record is updated again. ADR-0044 is not marked Superseded; see "Implementation ## Current implementation boundary (2026-09-09) This stays **Accepted with implementation delivered**, not a completed human or -cross-platform release certification. ADR-0050 adds evidence-backed repository +cross-platform release certification. D-15 defers the usability, screen-reader, and +cross-platform evaluation gates to v5. ADR-0050 adds evidence-backed repository groups, independent Desktop-origin filtering, and wrapping language icons. Current cards have no three-icon disclosure or repeated uncertainty labels. Installation counts and exact action identities are unchanged. diff --git a/docs/adr/0052-codex-usage-attribution.md b/docs/adr/0052-codex-usage-attribution.md index 2e5a6ec2..55343df8 100644 --- a/docs/adr/0052-codex-usage-attribution.md +++ b/docs/adr/0052-codex-usage-attribution.md @@ -12,6 +12,18 @@ record with at least one response (`usage-aggregate.mjs`'s `buildSessionRows`), so the file's 176,326 tokens never reach any total regardless of the explanation. The advisory is right in substance; classification is unchanged. Extends the "Not done" bullet below with the measured counts. +- **Updated:** 2026-09-29 — Unit 8 re-read the one gap candidate selected from a read-only schema-25 + cache under a 2 MiB source bound. Its current source has five token snapshots with full + input/cache/output components, no normalized assistant response, tool items and an abort. + The parser already retains its component row; aggregation now admits a Codex record with + positive component usage even when responses are zero. A positive total-only counter + remains unsupported: it supplies no input/cache/output split or price. Source health + discloses such events as `total-only-token-count` and reports zero-response records with + counted components separately from those without attributable component rows. +- **Updated:** 2026-09-29 — accepted V6 also classifies guardian reviews and other thread sources, + rolls up verified acyclic parent links, retains effort and host-reported first-token timing, + and exposes compaction bounds. Auto-review models without supported prices are unpriced. + The delivered cache migration is 25 → 26; earlier v23 evidence below describes the original fix. - **Deciders:** agentic-kit maintainers - **Related:** [ADR-0009](0009-usage-scorecard-local-transcript-analytics.md), [ADR-0038](0038-consistent-cross-host-session-metrics.md), @@ -121,16 +133,50 @@ lists them). Their turns are stamped `external-import-turn-N`, they have no "responses" and a large share of prompts, priced at model `unknown`, $0, while the real data lives in the Claude transcript. -The in-rollout marker (the `turn_id` prefix) is the signal, so detection works -without the imports file. Parsing stops at the first such line; the record is kept -out of aggregation, out of every yield statistic, and counted in -`diagnostics.importedExcluded` (796 on the reference machine). Nothing is dropped -silently. The record itself is still cached, so a rescan is cheap. - -Project discovery applies the same marker to each rollout's bounded head (256 KiB, 40 lines): an -imported copy names no project, host or Desktop origin, and the scan reports how many it set aside -(`importedExcluded`, 924 on the reference machine on 2026-09-27, every marker on the rollout's -second line). +The in-rollout marker (`payload.turn_id` prefix) is the signal; the import map is not a +runtime dependency. Exclusion is per turn. A valid native `task_started` or identified +`turn_context` opens native ownership; explicit record IDs must agree with that boundary. +Missing or conflicting IDs in mixed files remain unattributable. An absent context ID may +enrich an identified turn, but an explicitly invalid boundary ID breaks adjacency and marks +ownership incomplete. A foreign completion does +not close the active turn. Replayed parent history still cannot count as child activity. +Marker text in messages and later `session_meta` declarations establish no ownership. + +Copied prompts, responses, tools, context and tokens are excluded. Excluded cumulative +snapshots advance the baseline; native snapshots book only their deltas. Decreasing counters +or an explicit first native `last_token_usage == total_token_usage` reset start a new segment +(the latter excludes identical re-emissions). First session identity remains authoritative. +A native turn's cwd can establish its genuine project; otherwise the first declared cwd is +eligible only after own activity is proved. The first declared app surface applies, without +assuming every mixed file came from Desktop. + +Files without proven own activity remain `imported: true` with unknown/imported-copy origin +and are cached but excluded. Mixed files retain `importEvidence`: `importedTurns`, +`importedRecords`, `ambiguousRecords` and `nativeRecords`, with no turn IDs or copied content. +The unique imported-turn set retains at most 4,096 bounded IDs; `importedTurnCountComplete` +marks a lower-bound count when capped. Mixed sources with clipped lines, skipped nonblank +records or explicitly invalid turn-boundary IDs are conservatively excluded in their entirety +with `ownershipComplete: false`, pending a complete readable source. String and streaming +readers expose the last pass's skipped-record count as `importEvidence.malformedRecords`; +clipping retains its existing diagnostic. This may omit proven activity before a gap but +cannot carry native ownership across unreadable copied-turn metadata. A subagent +whose replay cannot be separated also has incomplete ownership. Source-health counters +`importOwnershipIncompleteFiles` and `importedTurnCountIncompleteFiles` retain these gaps +even when the file is excluded or served from cache. +Usage diagnostics expose `importedExcluded`, `importedMixed`, `importedTurnsExcluded` and +`importAmbiguousRecords`; the ambiguous count discloses excluded records without proved +ownership. Aggregate rows and session detail preserve the same evidence. Schema 26 remains +the single unreleased migration; no personal cache is rebuilt during implementation. + +Discovery first reads its usual 256 KiB/40-line head. Import-marked heads additionally read +at most a 256 KiB head and a 2 MiB tail, capped at 20,000 records per window and 512 MiB of +additional reads per scan. An unread middle resets ownership. Positive native activity can +establish a mixed sighting; imported-only exclusion requires a complete, unambiguous read. +Malformed envelopes or explicitly invalid turn-boundary IDs also leave an import candidate +unresolved. A sampled subagent without a complete replay boundary remains unresolved. Sources and the +discovery summary expose `importedMixed` and `importedUnresolved`; unresolved imports make +`complete` and `sessionCountComplete` false and cannot trigger encoded-directory recovery. +These are bounded observations, not an exhaustive turn census. ### 4. Cumulative counter restarts are summed, per event @@ -171,7 +217,7 @@ tools and `FunctionCallOutput` the known set. Only a type in none of them warns. ## Consequences -- Cache schema **v23**: every cached Codex record and its `parseStats` re-derive. +- Original implementation cache schema **v23**: every cached Codex record and its `parseStats` re-derive. Earlier records carry the wrong imports, subagent usage, replay counts, last-wins totals, single-day/model rows and permanent diagnostics. - Codex subagent sessions now have real tokens and cost. Cost totals, the @@ -204,28 +250,26 @@ subagent and previously dropped usage is now priced. ## Not done (recorded follow-ups) -- Guardian-review classification, unread host fields, and the context-coverage - denominator remain as audited; this ADR does not change them. +- Guardian-review classification, effort, first-token timing and compaction evidence are now + captured. Compaction lower/nullable upper bounds preserve uncertain pairing. This does not + establish support for every unread host field or change the context-coverage denominator. - A stream tee or push channel for live oversized rollouts is out of scope. - The cause of counter restarts is unknown (decision 4). - A subagent with no ordinals still reports no usage (decision 2). -- One rollout carries `token_count`s but no agent message, so the pre-existing - `partial-response-yield` warning remains — measured on the reference machine (2026-09-28): of 1,714 - Codex rollouts (734 token-bearing), exactly 1 is such a gap file. It is explained by a cached fact - (`session.aborts > 0`, tool-only activity) but that does not change what is counted: the aggregate - never builds a session row for a record with zero responses (`usage-aggregate.mjs`'s - `buildSessionRows`), so the file's usage (176,326 tokens, its own `last_token_usage.total_tokens` - sum across its `token_count` events) reaches no total either way. Counting that usage, or - documenting the shape more precisely, is left to the usage-accuracy branch. -- Whole-rollout exclusion may drop real usage (open, plausible, 2026-09-27). On the reference - machine 6 of 924 imported rollouts carry a later turn that is not an import: one `task_started` - whose `turn_id` starts with `rollout-`, no `user_message` event, `role: user` response items in - five of the six (2 to 76 per file) and non-zero `token_count` usage (the per-file sum of - `last_token_usage.total_tokens` is about 8k to 449k). Both usage and discovery set the whole file - aside at the marker, so this usage is not counted. With no `user_message`, the turn may be - automatic (a compaction or title pass). Measured from counts only. Decided 2026-09-27 (audit - decision 12): Branch 8 excludes per turn instead of per file, so imported turns are never counted - and later turns are, after it establishes whether they are the user's work or an automatic pass. +- The 2026-09-28 full-corpus count (1,714 rollouts, 734 token-bearing, one + zero-response gap) is historical. Unit 8's bounded 2026-09-29 re-read of that gap + candidate found 11,082 uncached input, 163,456 cached input and 1,788 output + tokens in a native, zero-response record. Those components now reach aggregate + totals and cost estimation. Total-only counters still cannot yield a split or + price and are diagnosed rather than silently treated as free usage. +- The historical 2026-09-27 observation (6 of 924 import-marked files) did not prove + that every token snapshot in those files belonged to a native turn. Unit 7 implements + decision 12 per turn. The 2026-09-29 metadata-only reproduction found 6 mixed files among + 945 import-marked candidates, with 64 native-turn responses. Only one token snapshot fell + inside a native interval, and its input/cache/output components were all zero; the other + snapshots were copied-turn evidence. No billable components are inferred from total-only + counters. The native turn's initiator remains unknown unless separately declared; this + does not prove whether it was user work or an automatic pass. ## Verification diff --git a/docs/adr/0053-host-setup-evidence-and-usage-diagnostics.md b/docs/adr/0053-host-setup-evidence-and-usage-diagnostics.md index 3890587c..685042ea 100644 --- a/docs/adr/0053-host-setup-evidence-and-usage-diagnostics.md +++ b/docs/adr/0053-host-setup-evidence-and-usage-diagnostics.md @@ -12,7 +12,10 @@ 15-minute-capped, consent-gated, untouched by this branch (its `host-health-evidence.mjs` input-fingerprint helper, used only to invalidate that in-memory cache, is also untouched). See [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md) (remediation program, branch 6a tasks 5 and 7) -- **Updated:** 2026-09-28 — `ak host check-connection ` is the CLI twin of +- **Updated:** 2026-09-29 — the dashboard **Check again** local re-check is folded into + header Refresh; `POST /api/host-health/local` was removed. The separate consent-gated + connection check remains. See [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md). +- **Earlier update:** 2026-09-28 — `ak host check-connection ` is the CLI twin of this connection check: it reuses `createHostReadinessReader`, so it refuses for exactly the same hosts and reasons the dashboard dialog would (managed-only, `canCheckConnection`), and applies the same consent rule (`--yes` or an interactive y/N; a non-TTY without `--yes` is refused; @@ -112,8 +115,9 @@ the requesting client cancels the owned connected subprocess. ### HTTP and presentation boundaries -`GET /api/host-health` reads local/cached evidence. The separate POST allowlist is -`/api/host-health/local` and `/api/host-health/connection`. Both require the session +`GET /api/host-health` reads local/cached evidence. The local re-check now runs within +`POST /api/refresh`, while the separate consent-gated connection check uses +`POST /api/host-health/connection`. Both require the session token header and exact same-origin fetch metadata; query tokens cannot authorize POST. Requests are size-bounded and accept fixed fields, never arbitrary commands, paths, prompts, environment or client-selected models. Connection checks additionally @@ -168,8 +172,8 @@ from one module, `src/lib/host-management.mjs`. - **The hint is the complete host list.** `ak host pick --host` replaces the enabled set, so the hint names every currently enabled host (including admitted external hosts) plus the one to add, for example `ak host pick --host claude,codex`. -- **The paid connection check stays managed-only.** The local re-check button reads - **Check again** and runs for every host. +- **The paid connection check stays managed-only.** The local re-check is part of header + **Refresh** and runs for every host. **Consequences.** Automatic local checks now spawn at most the same bounded, read-only commands for up to three hosts per minute while the dashboard is open. The previous diff --git a/docs/adr/0055-aqe-embedding-lifecycle.md b/docs/adr/0055-aqe-embedding-lifecycle.md index 0f5bacd4..2475c1df 100644 --- a/docs/adr/0055-aqe-embedding-lifecycle.md +++ b/docs/adr/0055-aqe-embedding-lifecycle.md @@ -14,6 +14,7 @@ - **Updated:** 2026-09-27 — the recognizer accepts every plain npx spelling of AQE's server (optional `-y`/`--yes`; unversioned, `@latest` or an exact version), audit item 5 choice A - **Updated:** 2026-09-27 — a passing embedding check reads "embedder verified"; status, `ak x verify aqe` and setup say AQE's pattern index binding stays unverified (agentic-qe#754) and corpus compatibility stays separate - **Updated:** 2026-09-27 — the busy rule's removal condition is agentic-qe#574 fixed in a released agentic-qe that is the kit floor; agentic-qe#719 (carried by 3.14.4) is only a partial fix +- **Updated:** 2026-09-29 — the later approved N-1 criterion supersedes that floor condition: released AQE 3.14.4 passed native macOS and Linux live-owner probes without `FsyncFailed`, so ak removed the exact 3.14.3 `FsyncFailed`-as-busy exception. Any `FsyncFailed`/`0x0303` now fails, including the old sequence. Ordinary `LockHeld` SQLite fallback remains busy with owner health and RVF integrity unknown. AQE 3.14.4 is the verified baseline for this decision, not a universal minimum; native Windows AQE conformance was not run - **Updated:** 2026-09-27 — beside these projections, `ak sync` and `ak setup` pin AQE to the project root (absolute `AQE_PROJECT_ROOT`, `AQE_MEMORY_PATH`, `AQE_STORAGE_PATH`) in `.claude/settings.local.json`, the recognized `.mcp.json` entry and both AQE tables of the project `.codex/config.toml`, under receipts from the same owned-env engine; a file git tracks is not pinned, and AQE's own relative `AQE_MEMORY_PATH` is taken back even after an AQE re-init. Stray AQE stores are merged and archived by `ak x aqe-store merge`. See [ADR-0062](0062-aqe-project-store-integrity.md) (remediation Branch 5, B5-D1 to B5-D5, B5-M5) - **Updated:** 2026-09-28 — live-check evidence storage relocated from `/agentic-kit/live-checks/.json` to the shared `/agentic-kit/evidence/live-check/.json` layout; this ADR's own live-check BEHAVIOR (TTL, statuses, remembered-check display) is unchanged, only where the evidence file lives. `ak x aqe-embedding verify` (distinct from `ak x verify`'s `aqe-embedding` row) does not persist evidence either way — it is a one-shot, unpersisted synthetic-backend proof. See [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md) (remediation program, branch 6a task 2) - **Updated:** 2026-09-28 — `ak x verify` is retired; its live checks (this ADR's own quick @@ -21,9 +22,9 @@ full `aqe` proof runs with `--only aqe`. A live-check evidence row now carries the source id `status-refresh-live`, labelled "ak status --refresh=live"; a row recorded before this rename under the retired `verify`/`status-live` source ids still reads back, labelled "an earlier live - check" — the label never names a retired command. Evidence ids are unchanged: `memory-routes` - still records under the `memory` id, and the full `aqe` proof still records only its embedding - request under `aqe-embedding` (remediation program, branch 6b; see + check" — the label never names a retired command. `memory-routes` records its CLI round trip + under `memory` and its routing observation under `memory-routes`; the full `aqe` proof still + records only its embedding request under `aqe-embedding` (remediation program, branch 6b; see [ADR-0063](0063-evidence-store-and-refresh-vocabulary.md)) - **Related:** [ADR-0023](0023-fail-closed-operations-and-explicit-degradation.md), [September repair](https://github.com/pacphi/agentic-kit/blob/main/docs/archive/2026-09-09-audit-aqe-integration-repair.md) diff --git a/docs/adr/0060-session-surface-initiator-and-product-names.md b/docs/adr/0060-session-surface-initiator-and-product-names.md index 098892ff..63e29e2f 100644 --- a/docs/adr/0060-session-surface-initiator-and-product-names.md +++ b/docs/adr/0060-session-surface-initiator-and-product-names.md @@ -1,13 +1,13 @@ # ADR-0060 — Session surface, initiator and official product names -- **Status:** Proposed; §3 implemented for project discovery (2026-09-27), the rest staged follow-on +- **Status:** Accepted - **Date:** 2026-09-26 -- **Updated:** 2026-09-27 — §3 implemented for project discovery and the System projects note: - imported copies give no project, host or origin and are counted. The ledger-derived source labels - (Cursor, Cowork) and the other views remain proposed. +- **Updated:** 2026-09-29 — delivered shared classification, usage/cache and census evidence, + Runtime application attribution, and CLI/dashboard presentation. Dedicated Cowork storage remains + an optional follow-up (#257); acceptance covers the bounded sources described below. - **Deciders:** agentic-kit maintainers - **Related:** [ADR-0050](0050-dashboard-project-identity-and-context-reporting.md) (session origin - rule, superseded in part by this record once accepted), + rule, superseded in part by this record), [ADR-0052](0052-codex-usage-attribution.md) (imported Codex rollouts excluded from usage), [ADR-0025](0025-machine-footprint-metrics.md) (Runtime census labels), [ADR-0027](0027-shared-project-census.md) (project census), @@ -25,7 +25,7 @@ the product's official commercial name, with no duplicate or false categories. Research on 2026-09-26 read only enumerated log fields and counts (no prompt or response content), the vendors' current documentation, the openai/codex source at `7f6c0f9`, and the installed Claude -Code 2.1.283 and Claude Desktop 2.9939.2 builds. What it established: +Code 2.1.283 and Claude Desktop 2.9939.2 builds. What that historical sample established (not a fresh census or current behavior): 1. **The logs declare where a session came from.** Claude Code transcripts carry `entrypoint`; Codex rollouts carry `session_meta.originator`, `source` and `thread_source`. Folder location is @@ -66,9 +66,9 @@ Code 2.1.283 and Claude Desktop 2.9939.2 builds. What it established: copied in five files. The Runtime census counts `Claude.app` as the Claude Code host (basename match) while `ChatGPT.app` is invisible. -## Decision (proposed) +## Decision -### 1. Two dimensions from declared fields, raw value always kept +### 1. Separate dimensions from declared fields, bounded raw evidence Every session record carries: @@ -77,18 +77,23 @@ Every session record carries: - **Initiator** — `person`, `automation`, `agent` (a subagent or reviewer spawned by another session) or `imported-copy`. Claude: person when interactive or the entrypoint is one Claude Code itself treats as attended (`claude-vscode`, `claude-desktop*`, `local-agent`, `remote*`, - `ssh-remote`, Claude Tag values); automation for `sdk-*`, `mcp`, `claude-code-github-action`, and + `ssh-remote`, Claude Tag values); automation for `sdk-py`, `sdk-ts`, `sdk-cli`, `mcp`, `claude-code-github-action`, and for `sessionKind` `bg`/`daemon`/`daemon-worker`. Codex: `thread_source` `user` and `chatgpt_handoff` are person, except that `codex exec` top-level threads are automation; - `subagent` and `guardian_review` are agent; app feature values such as `automation` are + `subagent`, `guardian_review` and `agent_created_thread` are agent; app feature values such as `automation` are automation. -- **Raw evidence** — the exact declared values (`entrypoint:cli`, - `originator:codex_exec/source:exec`). An unrecognized value is shown as "Other" with its raw value, - never merged into a larger bucket. +- **Raw evidence** — bounded tokens from the named declaration fields, such as `entrypoint:cli` and + `originator:codex_exec/source:exec`. Unfamiliar valid tokens remain available in local detail, with Other or Unknown + classification and no inferred product or provider. Tokens must be at most 80 characters, + begin with a letter and contain only letters, digits, underscores, dots or hyphens; the known + `Codex Desktop` value is the sole space-containing exception. Malformed values are omitted. Folder class (ChatGPT Projects folder, projectless Work folder, Cowork data, temporary folder) is an -explanatory attribute, never the classifier. The first record that declares a value wins, and -the rule is stated in code and tests. +explanatory attribute, never the classifier. The first eligible declaring record wins within each reader's bounded evidence window; +invalid values remain Unknown and do not authorize a search for a preferred later identity. +A later replayed parent declaration cannot replace the child's identity. Git scope, host, +surface, initiator and provider are separate fields and filters. Cloud choices appear only +when a covered record declares a cloud surface. ### 2. Official names @@ -101,8 +106,8 @@ Claude (per code.claude.com and claude.com documentation): | `claude-desktop`, `claude-desktop-3p` | Claude Desktop (attribute "on 3P" for the second) | person | | `local-agent`, `local_agent`, `remote_cowork` | Cowork | person | | `remote`, `remote_desktop`, `remote_mobile`, `remote_projects` | Cloud session (attribute: started from Desktop, mobile, web or a project) | person | -| `remote_trigger`, `remote_cowork_trigger` | Cloud session (routine) | automation | -| `sdk-py`, `sdk-ts` | Claude Agent SDK (attribute: Python or TypeScript; "plugin hook" when evidenced) | automation | +| `remote_trigger`, `remote_cowork_trigger` | Cloud session | automation | +| `sdk-py`, `sdk-ts` | Claude Agent SDK | automation | | `sdk-cli` | Non-interactive mode (`claude -p`) | automation | | `claude-code-github-action` | GitHub Actions | automation | | `claude_in_slack`, `claude-in-slack`, `claude-in-teams` | Claude Tag (Slack or Teams) | person | @@ -126,7 +131,8 @@ OpenAI (per learn.chatgpt.com, which developers.openai.com/codex now redirects t | `codex_work_web`, `codex_work_mobile`, `codex_work_cca`, `chatgpt_cca` | ChatGPT Work (cloud) | by `thread_source` | | any other | Other OpenAI client (raw value shown) | by `thread_source` | -Subagents and Auto-review roll up under their parent surface ("Subagents", "Auto-review", OpenAI's +Codex subagents and Auto-review with a verified, acyclic parent link in the observed record set +roll up under their parent surface ("Subagents", "Auto-review", OpenAI's own labels) and are never separate products. `source="vscode"` never produces a "VS Code" label. Hosts are named **Claude Code**, **Codex**, **OpenCode** (by Anomaly) and **Hermes Agent** (Nous @@ -135,10 +141,20 @@ Desktop** and **ChatGPT desktop app**; they are applications, not hosts. ### 3. Imported copies are excluded everywhere, and counted -ADR-0052's rule extends to project discovery and every origin view: a rollout stamped -`external-import-turn-*` (or listed in the imports ledger when present) contributes no project -sighting, origin, facet count or Runtime attribution. Each view reports how many it excluded, labelled -"Imported from Claude Code" (or Cursor, or Cowork, from the ledger's source path). +ADR-0052's rule extends to project discovery and origin views **per turn**. The portable +signal is `payload.turn_id` beginning `external-import-turn`; an import map is not required. +Copied turns contribute no usage or project/origin sighting. A later proven native turn can +establish the first declared app surface and a genuine project; unknown or conflicting turn +boundaries remain excluded with diagnostics. A Desktop declaration is not inferred for +other declared products. First session identity and parent replay exclusion remain intact. + +Pure imports remain unknown/imported-copy. Mixed usage rows retain import-exclusion counts. +Discovery distinguishes confirmed exclusions (`importedExcluded`), proven mixed observations +(`importedMixed`) and bounded observations that cannot settle ownership (`importedUnresolved`). +The latter make coverage incomplete, without inventing a project from an encoded directory. +See ADR-0052 §3 for exact byte/record budgets and cumulative-counter rules. Intelligence, +System Projects and `ak system` disclose these populations and source incompleteness. +Optional ledger source labels remain follow-on work. ### 4. One vocabulary module and one label table @@ -151,56 +167,59 @@ Labels are tested once; views test that they use the shared table. - The Runtime census treats `Claude.app` and `ChatGPT.app` symmetrically as desktop applications, neither as a host; their bundled CLIs are attributed to the hosted session (lane D's Claude rule extended to the Codex bundle). -- Cowork transcripts become an optional discovery source; until then views say Cowork is not - covered. +- Dedicated Cowork storage remains uncovered (#257), and views disclose that limit. Covered + Claude transcript records may still declare the Cowork surface; that does not prove coverage + of the separate store. -### 6. Counting rules +### 6. Counting rules and compatibility -Claude sessions are counted by `sessionId`, excluding `subagents/` transcripts and non-conversation -records; Codex subagent and reviewer rollouts roll up to their parent; `thread_source` is classified -in full. +Claude project census sessions use declared `sessionId`, excluding subagent and bridge-only +transcripts. Rows expose `countBasis`: declared-session IDs, transcript files, database sessions, +recovered-project sightings or mixed observations. Encoded-directory recovery can establish a +project sighting with zero session weight; missing identity and bounded reads keep completeness +visible. Census session observations are distinct from billed Usage sessions. + +Project `sessionSurfaces` is additive: legacy `sessionOrigins` remains for compatibility. Raw +project detail unions retain at most 16 sorted values per named field per classification group; +`rawEvidenceComplete: false` discloses truncation. The raw-token policy was explicitly approved +by the maintainer for local detail; it does not authorize publishing private tokens. + +Old coarse `codex-desktop` snapshots render Unknown surface with a ChatGPT desktop app family +note because their mode was not recorded. Old `claude-desktop` snapshots retain Claude Desktop; +initiator and provider remain Unknown. Saved legacy origin filters preserve their membership +and use explicit legacy labels. Missing newer fields are not evidence of a precise mode. + +### 7. Provider evidence is independent + +Claude provider-specific assistant model IDs may establish Amazon Bedrock or Google Vertex AI +metadata (`assistant-model-id`); ordinary or conflicting IDs leave Unknown. Codex and OpenCode +may provide a recorded provider ID. Both are observed source metadata, not network attestation. +Current environment, routing configuration, application identity and price-table identity cannot +establish a historical serving provider. The `on 3P` attribute remains visible independently of +provider Unknown. Codex Auto-review tokens remain unpriced when no supported price exists. ## Consequences -- Usage, System → Projects, Maintenance facets, Intelligence designation (which today mixes Git - scope, origin and host in one enum) and the Runtime table change labels and counts. On this - machine 32 project folders lost a false Desktop origin when discovery began setting imports aside - (re-measured 2026-09-27; 23 on 2026-09-26). -- The usage cache schema changes (new session fields); a rebuild is expected. -- Tests that pin current names change together (inventory in the audit record, Addendum 3). -- `CLAUDE_CODE_ENTRYPOINT` and the transcript format are internal to Claude Code and may change; - keeping the raw value and an "Other" fallback bounds that risk. -- Privacy is unchanged: only enumerated values and counts are read. - -## Open questions for acceptance - -- Whether "Cloud session" should appear at all in local views, given none was observed locally. -- Whether the "on 3P" attribute is worth showing. -- How ADR-0057's role lenses consume surface and initiator. -- Whether a later turn inside an imported copy that is not itself an import (6 of 924 rollouts on - 2026-09-27, with real token usage) counts as the importing app's own session. Decided 2026-09-27 - (audit decision 12): it counts, excluded per turn in Branch 8 - ([ADR-0052](0052-codex-usage-attribution.md), "Not done"). - -## Verification (when implemented) - -Fixtures per raw value; an import-ledger join fixture; a census reproduction of the 2026-09-26 counts -from enumerated values; one-label-per-value UI assertions across views; no prompt content in any -fixture. - -## Implementation status - -§3 is implemented for project discovery and the System projects note (2026-09-27): an imported copy -gives no project, host or origin, and discovery counts it in `importedExcluded`. The per-source -labels from the imports ledger, Runtime attribution and §1, §2 and §4–§6 remain follow-on work -(the audit record's Addendum 3). - -Three views already show the smaller project counts but do not yet say how many imported copies were -set aside; §3's "each view reports how many it excluded" is still owed for them: - -- the Intelligence census line (`src/lib/dashboard/client/intelligence.mjs`, which prints - `everSeen`; the server's `readCensus` in `src/lib/dashboard-server.mjs` drops `importedExcluded`); -- the System → Projects liner (`sysProjectsLinerHtml` in - `src/lib/dashboard/client/system-projects.mjs`); -- the `ak system` text output (`renderProjects` in `src/commands/system.mjs`, which prints only the - count; `ak system --json` carries `importedExcluded`). +- Shared vocabulary in `src/lib/session-surface.mjs` supplies Usage, project details, Maintenance, + Intelligence and Runtime labels. Desktop applications have no host identity; bundled CLIs need + observed session attribution, and a Codex app-server process is a service. +- Usage cache schema changes exactly **25 → 26**; old entries require rebuilding. Footprint + snapshot schema remains **8**, with additive evidence and explicit legacy presentation. +- First-declaration, parent-link, import-ownership and source bounds prevent these observations + from establishing whole-corpus coverage. Historical research counts above are not release metrics. +- Internal host fields can change. Unknown, bounded raw evidence and source-health diagnostics + preserve uncertainty without deriving products from directories or `source="vscode"`. + +## Verification and remaining limits + +Synthetic fixtures cover shared vocabulary, first declaring records, parent/reviewer attribution, +legacy filters, raw-token caps, provider evidence, import ownership, session-count bases and +Runtime application/service distinctions. CLI/dashboard consumer assertions cover the shared +labels and disclosures. The implementation is bound to the accepted V6 source units; final +integration gates and publication are separate decisions. + +Dedicated Cowork storage (#257), optional import-ledger source labels, missing parent evidence +and records outside bounded readers remain uncovered. No new live corpus, provider, performance +or billing measurement is claimed by this documentation update. See +[Usage metrics](../usage-scorecard-metrics.md#current-accounting-and-cache-contracts) for the +bounded accounting, source selection and cache contracts delivered alongside this vocabulary. diff --git a/docs/adr/0062-aqe-project-store-integrity.md b/docs/adr/0062-aqe-project-store-integrity.md index 2790861d..dcd5b6d5 100644 --- a/docs/adr/0062-aqe-project-store-integrity.md +++ b/docs/adr/0062-aqe-project-store-integrity.md @@ -6,6 +6,7 @@ re-init value taken back, clean release; holder checks that time out refuse; nested repositories are not strays; stores fingerprinted at copy time; root checked before backup; applying receipt; starter patterns the root holds keep their usage +- **Updated:** 2026-09-29 — released AQE 3.14.4 passed native macOS and Linux live-owner conformance with `LockHeld` and no `FsyncFailed`; ak retired only the old exact `FsyncFailed`-as-busy exception in AQE startup classification. This is a verified baseline for that rule, not a universal AQE minimum. The 3.14.4 minimum below applies only to store merge. Native Windows AQE conformance remains unverified - **Deciders:** agentic-kit maintainers - **Related:** [ADR-0016](0016-capability-driven-integration-adapters.md) (project memory status and stray stores), [ADR-0055](0055-aqe-embedding-lifecycle.md) (the AQE embedding projections this diff --git a/docs/adr/0063-evidence-store-and-refresh-vocabulary.md b/docs/adr/0063-evidence-store-and-refresh-vocabulary.md index c3a15207..8a0e1752 100644 --- a/docs/adr/0063-evidence-store-and-refresh-vocabulary.md +++ b/docs/adr/0063-evidence-store-and-refresh-vocabulary.md @@ -1,7 +1,8 @@ # ADR-0063 — One evidence store and the refresh vocabulary - **Status:** Accepted -- **Updated:** 2026-09-28 — Branch 6b delivered the CLI refresh vocabulary +- **Updated:** 2026-09-29 — Branch 6c delivered the dashboard refresh operation and retired GET-started scans; 2026-09-29 V4 A3: self-version retry attempts and last freshness are scoped to checked channels +- **Earlier update:** 2026-09-28 — Branch 6b delivered the CLI refresh vocabulary - **Date:** 2026-09-28 - **Deciders:** agentic-kit maintainers - **Related:** [ADR-0025](0025-machine-footprint-metrics.md) (`/api/system/summary` projection), @@ -15,10 +16,11 @@ already recorded its own `Updated:` line), [ADR-0055](0055-aqe-embedding-lifecycle.md) (live-check evidence storage relocation, mechanical), [ADR-0053](0053-host-setup-evidence-and-usage-diagnostics.md) (host setup checks are now persisted evidence with an age rule) -- **Supersedes:** [ADR-0048](0048-inventory-led-maintenance-resource-management.md)'s use of the - word "evidence" for its own, separate **Refresh evidence** / **Re-measure machine** scan - controls — terminology only; see "Relationship to ADR-0048" below. Their UI, backing code - (`scan-store.mjs`), and evidence semantics are unchanged by this branch. +- **Supersedes:** [ADR-0048](0048-inventory-led-maintenance-resource-management.md)'s separate + **Refresh evidence** / **Re-measure machine** dashboard controls; + [ADR-0025](0025-machine-footprint-metrics.md) §5's GET-started deep refresh rationale; and + [ADR-0045](0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md)'s + `GET /api/maintenance?refresh=scan` trigger. Their underlying measurements and stores remain. ## Context @@ -404,7 +406,7 @@ the [issues 237–239 audit](../plans/2026-09-26-issues-237-238-239-verification Item 4 named. It closed CLI-only: the dashboard's own controls are untouched, and the dashboard half of this work moved to the next remediation program — see [the branch 6b plan](../archive/2026-09-28-superpowers-plan-branch-6b-one-refresh-flag.md)'s "Closing -this branch" section, and "Ahead: the dashboard half" below. +this branch" section and "Delivered in 6c" below. - **One flag, three strengths, one ordered stage table** — `--refresh[=live|machine]` across `ak status`, `ak system` and `ak maintain [report]`; see "The `--refresh` flag's three strengths" @@ -428,9 +430,9 @@ this branch" section, and "Ahead: the dashboard half" below. `learning`, `harvest`, `aqe`, `memory-routes` — slow, `--only`-only) now live in `src/lib/live-checks.mjs` and run as `--refresh=live`'s `live` stage. `memory` is the quick store/retrieve/purge round trip; `memory-routes` additionally observes whether the CLI and MCP - see each other's writes, and its result is remembered under the `memory` evidence id, not its - own. A live-check evidence row now carries the source id `status-refresh-live`, labelled "ak - status --refresh=live"; a row recorded before this branch under the retired `verify` or + see each other's writes. Its CLI result is remembered under `memory` and its routing result + under `memory-routes`. A live-check evidence row carries the source id + `status-refresh-live`, labelled "ak status --refresh=live"; a row recorded before this branch under the retired `verify` or `status-live` source ids still reads back, labelled "an earlier live check" — the label never names a retired command (`live-check-evidence.mjs`'s `SOURCE_LABEL`). - **The Codex quota presence gate.** `/api/limits` asks `codex app-server` for its @@ -485,64 +487,59 @@ this branch" section, and "Ahead: the dashboard half" below. and stopping before any write; `ak host adapters` refuses `--dry-run` outright (exit 2) because its verbs have no preview. -## Ahead: the dashboard half - -6b closed CLI-only — see -[the branch 6b plan](../archive/2026-09-28-superpowers-plan-branch-6b-one-refresh-flag.md)'s "Closing -this branch" section. The dashboard's own controls are unchanged by this branch and remain future -work for the next remediation program: - -- One dashboard **Refresh** control offering the same three strengths the CLI now has, and a - **Reload** control that only re-reads the current view. -- A `POST /api/refresh` route driving that control through the same `runRefresh`/stage machinery - this branch built for the CLI, and the dashboard's existing read-only `GET` routes. -- The eventual supersession of ADR-0048's **Refresh evidence** / **Re-measure machine** controls - and of [ADR-0025](0025-machine-footprint-metrics.md) §5's `GET ?refresh=deep` rationale, once - that dashboard work lands — neither is superseded by this branch, and both remain exactly as - their own ADRs describe them today. - -What stays out of scope regardless: this store's storage does not unify with Maintenance's own -`scan-store.mjs`/deep-snapshot system (R1 — one flag and, eventually, one dashboard control drive -the existing chain; the footprint snapshot and its storage stay where they are), and -`src/lib/host-health-evidence.mjs` (ADR-0053's setup-proof input-fingerprint helper) is still not -folded into this store — see "What is deliberately not folded into this store" above, unchanged -by 6b. +## Delivered in 6c + +Branch 6c adds one dashboard **Refresh** control. Its visible choices are **Refresh**, +**Refresh live**, and **Refresh machine**; the operation request calls their conceptual strengths +`local`, `live`, and `machine`. The separate header **Reload** re-reads the active view and +starts no checks. The former **Check again** local host-health button is folded into Refresh. + +`POST /api/refresh` starts one explicit operation with a bounded JSON body: `strength` is +required; `projectTrees` is an optional boolean for `machine` only. The response is 202 with +`started: true` and the operation state, or 409 with the current state when work is already +running. The server requires the per-session `x-dash-token` header and same-origin mutation +metadata; a query token cannot authorize this POST. `GET /api/refresh` reads the latest state +without starting work. That state has an `operationId`, strength, timestamps, running/completion +fields, and sanitized stage progress, but no stage results. `GET /api/system`, +`GET /api/system/summary`, `GET /api/maintenance`, and the host-health read remain reads: +legacy refresh/scan query arguments are rejected, and `/api/host-health/local` was removed. +The separate consent-gated `POST /api/host-health/connection` remains. + +The dashboard runs the shared `runRefresh` stage order from `src/lib/refresh.mjs`: + +| Strength | Ordered stages | +|---|---| +| `local` (Refresh) | Maintenance evidence → inventory → local evidence and versions | +| `live` (Refresh live) | Maintenance evidence → inventory → live checks → local evidence and versions | +| `machine` (Refresh machine) | Machine measurement → Maintenance evidence → inventory → local evidence and versions | + +A failed machine measurement skips its dependent Maintenance and inventory stages; the local +stage still runs. Other stage failures do not stop later stages. Machine measurement can include +project trees when explicitly selected. The Maintenance stage scans provider evidence; inventory +rebuilds after it; local collects status and forces the host-readiness check. These stages use +the existing Maintenance scan and footprint stores, not a merged evidence store. + +`createRefreshOperation` holds one current operation in one dashboard server process. +Two tabs connected to that server share its single-flight state, so a concurrent POST gets 409. +This is volatile process state, with neither durable operation history nor distributed mutual +exclusion across servers. The client retains its accepted `operationId` and reports a result +only for that identity. If its POST response is lost or another operation supersedes the +server's latest state, the page may say its outcome is unavailable; it does not claim the +other operation's result. ## Relationship to ADR-0048 -ADR-0048's **Refresh evidence** and **Re-measure machine** dashboard controls predate this branch -and use the word "evidence" in the Maintenance-inventory sense (provider probes feeding the -Inventory/Guidance/Discovery/Activity workspace), which is conceptually adjacent to — but a -genuinely separate system from — the evidence store this ADR describes. This ADR's evidence store -is the eventual "live" tier's technical precursor and this branch's own interim `--refresh` boolean -is its no-suffix groundwork; ADR-0048's own controls, their backing code (`scan-store.mjs`), and -their UI are unchanged by this branch. No file under Maintenance's own scan system was touched by -Tasks 1–11. A reader should not infer that ADR-0048's controls now share code, storage, or an age -rule with this ADR's evidence store — they do not, yet. Branch 6b gives those two controls CLI -equivalents — `ak maintain --refresh` and `ak maintain --refresh=machine` — without changing the -controls, their backing code, or their UI themselves; see "Delivered in 6b" above. +ADR-0048's separate **Refresh evidence** and **Re-measure machine** dashboard controls are +superseded by the single Refresh control above. Its Inventory/Guidance/Discovery/Activity +workspace, scan storage (`scan-store.mjs`), evidence semantics, and guarded management +actions remain. The CLI equivalents delivered in 6b are `ak maintain --refresh` and +`ak maintain --refresh=machine`. The control unifies the user's start path, not the +underlying storage or age rules. `src/lib/host-health-evidence.mjs` also remains outside the +shared evidence envelope. ## Known limitations (recorded, not fixed, by this branch) -1. **Resolved in Branch 6b: the failed-lookup rule is now the same in all four version-drift - functions.** This item originally recorded that `ruvector.mjs`/`ruvnet-brain.mjs`'s `drift()` - could silently drop a known update on a failed forced fetch, unlike `versions.mjs`'s - `driftReport()`/`selfDrift()`. Branch 6b fixed both (`fix(versions): a failed lookup keeps the - cached version and waits one TTL window before retrying`, and its follow-ups), so all four now - share one rule: on a total lookup failure, the cached `latest`/`best`/`installedRelease` value - is kept — never overwritten with `null` — and the TTL stamp (`last`) is restamped, so the next - unforced call waits one more TTL window before retrying (`force` bypasses this and retries - immediately). `observedAt` records the real time a value was last actually observed, not the - time of a failed retry: `ruvector.mjs`'s `drift()` keeps `observedAt: cached.observedAt ?? - cached.last` on failure (`:70`); `ruvnet-brain.mjs`'s `drift()` does the same - (`:288`, `recordedRelease()`); `versions.mjs`'s `driftReport()`'s `lookUpLatest()` restamps - `observedAt` from the prior `last` only for packages that were never individually observed - (`:82`); its `selfDrift()`'s `selfRecord()` restamps on a *total* failure, including one with no - cached candidate at all — only a partial answer (something answered live but did not win), or a - total failure whose cached candidate is unusable (a `next` candidate on a stable install), saves - nothing (`:172-176`). None of the four applies this rule under `record: false` (`ak sync - --dry-run`, ADR-0063's own `record` parameter) or a cache-only read (`cacheOnly: true`, `ak - sync --skip `): both skip the network and the write entirely, by design. +1. **Resolved in Branch 6b, refined in V4 A3: failed version lookups retain recorded evidence without claiming a new observation.** A total failed lookup of managed packages, Brain, or ruvector keeps its cached candidate and restamps `last` for one retry per configured TTL; `observedAt` remains the time that candidate was actually seen. The kit follows that rule when its cached candidate is usable. A partial answer that leaves the kit's cached candidate winning, or a stable install whose cached `next` candidate is unusable, keeps `last`, `observedAt`, and `best` unchanged and separately records `versionCheck.self.attempt` with its time and exact channel tags. Successful and total-failure kit lookups record `lastTags`, the channel scope of `last`; changing from stable to prerelease therefore probes an untried `next` channel even within the prior TTL. Legacy records without `lastTags` are reused for the single `latest` channel or where a `next` winner proves it was checked; a legacy `latest` winner cannot suppress an untried `next`. Malformed or future attempt metadata cannot suppress retries. `force` bypasses freshness. `record: false` permits a lookup without saving its result or attempt, while `cacheOnly: true` performs neither a lookup nor a write. 2. **`globalRoot()`'s `record`-persistence structural fragility** — see "The `npm-global-root` exception" above. 3. **Task 9's dashboard timeout bounds async hangs only** — see "The dashboard poll's two cost diff --git a/docs/adr/README.md b/docs/adr/README.md index 24982403..8eb84908 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -58,7 +58,7 @@ Consequences**, and cites the grounded source it rests on where relevant. | [0045](0045-artifact-consumer-bindings-and-explicit-maintenance-scans.md) | Physical artifacts, host consumers, and explicit Maintenance scans | Implemented | | [0046](0046-scan-local-observation-reuse-and-nonblocking-deep-scans.md) | Scan-local observation reuse and nonblocking deep scans | Implemented | | [0047](0047-streaming-observation-forest.md) | Streaming observation forest for deep scans | Accepted; Projects pilot and separate Discovery continuation implemented | -| [0048](0048-inventory-led-maintenance-resource-management.md) | Inventory-led Maintenance resource management | Accepted; Focus browser implemented and focused checks pass; human/cross-platform gates pending | +| [0048](0048-inventory-led-maintenance-resource-management.md) | Inventory-led Maintenance resource management | Accepted; two dashboard controls superseded by 0063; human usability, screen-reader and cross-platform gates deferred to v5 | | [0050](0050-dashboard-project-identity-and-context-reporting.md) | Dashboard project identity and context reporting | Implemented | | [0051](0051-supported-peer-delegation-and-host-realignment.md) | Supported peer delegation and scoped host realignment | Accepted; implemented locally | | [0052](0052-codex-usage-attribution.md) | Codex usage attribution: own usage, imports, segments, streaming | Accepted | @@ -66,10 +66,10 @@ Consequences**, and cites the grounded source it rests on where relevant. | [0054](0054-fleet-evidence-export.md) | Vendor-neutral fleet evidence export | Implemented | | [0055](0055-aqe-embedding-lifecycle.md) | AQE embedding lifecycle and qualified readiness | Implemented | | [0058](0058-managed-ruflo-components.md) | Managed ruflo components | Accepted (implementation in progress — see Implementation status) | -| [0060](0060-session-surface-initiator-and-product-names.md) | Session surface, initiator and official product names | Proposed; §3 implemented for project discovery (2026-09-27), the rest staged follow-on | +| [0060](0060-session-surface-initiator-and-product-names.md) | Session surface, initiator and official product names | Accepted | | [0061](0061-brain-reclaim-stuck-remediation.md) | RuvNet Brain "unresolved rollback state" remediation | Accepted | | [0062](0062-aqe-project-store-integrity.md) | AQE project store integrity | Accepted | -| [0063](0063-evidence-store-and-refresh-vocabulary.md) | One evidence store and the refresh vocabulary | Accepted | +| [0063](0063-evidence-store-and-refresh-vocabulary.md) | One evidence store and the refresh vocabulary | Accepted; CLI and dashboard refresh operation delivered | Theme: ADRs **0001–0006** define **dual-host LLM routing and leadership** — how `ak` lets ruflo route each development activity (architecture, implementation, testing, review, …) to the right host (Claude @@ -399,11 +399,13 @@ component out. ## ADR-0060 — Session surface, initiator and official product names -[ADR-0060](0060-session-surface-initiator-and-product-names.md) (Proposed; §3 implemented for -project discovery) derives a session's surface and initiator from the hosts' declared log fields, -keeps every raw value, uses official product names (Claude Desktop, ChatGPT desktop app, Codex CLI, -and others), and excludes imported session copies from every origin view. Project discovery already -sets imported copies aside and counts them; the other decisions remain proposed. +[ADR-0060](0060-session-surface-initiator-and-product-names.md) (Accepted; updated 2026-09-29) +separates Git scope, host, session surface, initiator and provider using declared source evidence +and shared official labels. Local detail retains approved bounded origin tokens; observed provider +metadata is not network attestation. Per-turn import ownership, count bases and completeness remain +visible, desktop applications are distinct from hosts, and legacy filters preserve their labeled +membership. Dedicated Cowork storage remains uncovered (#257). Usage schema changes 25 → 26; +footprint stays 8. ## ADR-0062 — AQE project store integrity diff --git a/docs/archive/2026-09-28-plan-dashboard-refresh.md b/docs/archive/2026-09-28-plan-dashboard-refresh.md new file mode 100644 index 00000000..de0a8107 --- /dev/null +++ b/docs/archive/2026-09-28-plan-dashboard-refresh.md @@ -0,0 +1,28 @@ +# Dashboard Refresh delivery plan + +## Status + +**Implemented; final integration gates pending** — Captured 2026-09-29 after `d2c1833b`. Tasks 6c-1 through 6c-5, the live-view follow-up, native plain-folder proof, and the paused Activity timestamp handoff passed scoped independent reviews. The branch incorporates green `develop@af825c9f`. Full branch gates, whole-branch review and feature PR CI remain at this archival capture; this status does not claim a merge or release. + +Implement the deferred 6c work in order, with one reviewed task and unit commit at a time. The [remediation program](../plans/2026-09-28-remediation-program-v2.md) and [6b handoff](2026-09-28-superpowers-plan-branch-6b-one-refresh-flag.md) define the contracts. A passing exact-head develop CI gate precedes production edits. Tests use sandbox state and injected services; final branch gates and integration belong to the controller. + +| Task | Dependency | Code and proof | +| --- | --- | --- | +| 6c-1: additive server operation | 6b Tasks 2 and 4 | Add `dashboard/refresh-api.mjs` with shared stage runner, single-flight state, validated POST and GET; wire it into `dashboard-server.mjs` without changing the client or poll cost. Add API, security, stage and cost tests. This dispatch only. | +| 6c-2: one Refresh control | 6c-1, 6b Tasks 2, 3 and 5 | Add `client/refresh-control.mjs`; wire page, boot and client views to POST once, poll only the page's operation, and reload the active view. Remove retired controls. Prove request counts, stage progress, blocked Maintenance writes, host consent and narrow-screen UI. | +| 6c-3: read-only GET routes | 6c-2 | Reject scan-starting GET query parameters, remove GET scan paths and `/api/host-health/local`; keep the 6c-1 stage's Maintenance and inventory refresh dependencies. Prove all GET routes have no scan side effects. | +| 6c-4: current vocabulary | 6c-1 through 6c-3, 6b Task 13 | Extend the vocabulary guard for retired dashboard strings and legacy refresh URLs; update dashboard, Maintenance, upgrading, DDD and installed guidance. Include the remaining README/dashboard noun and `maintenance-discovery.mjs` string. Check links and Markdown. | +| 6c-5: decisions and supersessions | 6c-1 through 6c-4, 6b Task 14 | Reconcile ADR-0063 with ADR-0048, ADR-0025, ADR-0045, ADR-0044 and ADR-0053, including index/status rows and the decision log; record actual POST behavior and read-only GETs. Run docs and final branch gates. | + +V3 carry-ins from the program: + +- **6c-2 client fixes:** B0-22 shows the Claude Code badge as "Unknown" when Configuration was not assessed. B0-23 measures the Codex header icon against WCAG's 3:1 non-text contrast minimum and changes it only if the measurement fails. B6a-12 makes `mntSyncHash` and `mntApplyHashState` re-derive state from `location.hash` if 6c-2 touches that code; otherwise open a small issue. +- **6c-5 decision record:** B6a-9 requires ADR-0063 to state that two dashboard tabs share one server process's module state. If 6c-1's tests build the `ruflo-components` path, cover its cwd case; otherwise record the ruling that drops that test case. ADR-0048's status line moves its human-evaluation gates to v5 under D-15. +- **Live view (#256):** Confirm the session re-read changes no total before documenting D-18; if it does, apply D-18's offset alternative. Label the structured live-events input experimental under D-19. Observe a plain-folder, non-Git bind on a real machine once and fix what the observation shows. +- **Live-view follow-up (2026-09-29):** A nonzero regression showed that bounded-window re-entry increased the accepted-record total, so D-18 now retains displaced native readers and offsets in a bounded in-memory map. A genuine Codex transcript kept accepted, session, and project totals stable across idle restart and window re-entry in an isolated service probe. Eviction can still replay old records, and a new service process has no persisted offset; this does not establish exactly-once delivery or unchanged historical token accounting. D-19's structured live-events input is experimental; no real producer was verified. See the V3 live-view report. +- **Plain-folder native proof (2026-09-29):** An independently reviewed, genuine Codex `0.159.0` native `thread/fork` created a new transcript in an isolated plain folder. The real process survey and `LiveSessionsService` joined the actual host PID/cwd to that new transcript, with observed presence. The copied parent retained its exact upstream bytes and remained presence-unknown. The five-session service snapshot included unrelated controllers; it does not show five plain-folder joins. This was an idle fork with no `turn/start`, inference, billing measurement, detailed activity proof, or browser journey. The fork's empty RPC `turns` field does not mean its copied history was empty. See the V3 plain-native report. +- **Evidence-gated issue #254:** Diagnose the "CONNECTING" stall only if the browser network trace specified by the program arrives; otherwise leave it to V7. +- **Paused Activity handoff:** V4 B8 records pauses with a distinct `recordedAt` and no completion time. This branch carries that validated timestamp through Activity projection and the API, then sorts and displays pause records without claiming completion. Impossible calendar dates are rejected; valid completion timestamps remain a fallback. The V4 producer is integrated separately. +- **Pre-PR flags check (2026-09-29):** The latest Codex remained `0.159.0` at the 11:16 UTC registry recheck. Its native artifact passed integrity verification, help checks and `-s read-only -a never app-server` initialization in an isolated home. This proves flag acceptance and startup, not inference or billing. The installed global CLI was not changed. + +Extend 6c-4's guard exclusions to `docs/plans/` and the current archive/proposal/ADR layout. Move this finished plan to `docs/archive/` in the completing pull request with an archive index row. diff --git a/docs/archive/2026-09-28-plan-follow-ups-v2.md b/docs/archive/2026-09-28-plan-follow-ups-v2.md new file mode 100644 index 00000000..fc1c8a5e --- /dev/null +++ b/docs/archive/2026-09-28-plan-follow-ups-v2.md @@ -0,0 +1,55 @@ +# Follow ups v2: V4 branch plan + +## Status at archival + +**Implementation and local verification complete (2026-09-29).** All eight local +gates passed at `2915265402b758ddcd73d0dd663db9308637f3b2`: 6,159 unit tests passed +with seven skips and no failures, 514 browser assertions plus 15 UI tests passed, +and typecheck, lint, complexity, Markdown, build and offline links passed. +Measured coverage was 94.00% lines, 83.36% branches and 93.27% functions. +Independent whole-branch review found no actionable findings; its additional +focused run passed 131 tests with two Windows-only skips. + +B1 merged in PR #273, V3 dashboard changes in #276, C3 trace in #277 and C4 watch +in #278. This branch includes green `develop@989c5e56`, the reviewed exact runner +identity follow-up `f80bc55e`, temporary C1 job removal `257e6940`, A3 ADR amendment +`44dc4e9` and C6 evidence alignment `29152654`. B13 required no product fix after +the approved conditional check. B6's extra Codex hook fix line remains deferred +pending a Ruflo-supported answer to #3419. + +Native macOS/Linux AQE live-lock conformance passed on the named released +artifacts; native Windows AQE was not run. Native Windows Ruflo 3.48.0 memory +visibility was observed with its native bridge disabled. Final-head feature PR +CI, including the corrected Windows identity fixtures, and squash integration +remain pending at capture. This archive does not claim main merge, release, +installation or operational cleanup. + +The [remediation program V4](../plans/2026-09-28-remediation-program-v2.md#v4-fixfollow-ups-v2-every-small-product-cli-and-upstream-item) defines scope. The [archived Branch 9 plan](2026-09-28-superpowers-plan-branch-9-follow-ups.md) supplies task details. Paths below name current source seams and focused test targets. After an explicit directory prefix, subsequent bare filenames in the same cell use that directory. A new test named below is a proposed file. Later implementers must verify dependencies before editing. + +| Row | Source or artifact mapping | Focused proof and prerequisite | +| --- | --- | --- | +| A1 | `bin/agentic-kit.mjs`; `src/commands/usage.mjs`, `models.mjs`, `audit.mjs`, `heal.mjs`, `telemetry.mjs`, `x/host.mjs` | `tests/kit/cli-json-honesty.test.mjs`, `usage-cli.test.mjs`, `models-command.test.mjs`, `telemetry-cli.test.mjs`, `status-command.test.mjs`; include unknown models verb and status positional | +| A2 | `src/commands/x/host.mjs`; `bin/agentic-kit.mjs` | `tests/kit/host-dry-run.test.mjs`, `host-cli-migration.test.mjs`; pick refusal, off, reset-routes under `--dry-run --json` | +| A3 | `src/lib/versions.mjs`; `docs/adr/0063-evidence-store-and-refresh-vocabulary.md` | Accepted in `175677a6`; `versionCheck.self.attempt` and `lastTags` scope offline retries. ADR-0063 item 1 is amended; local gates passed at `29152654`, with final PR CI pending | +| A4 | `src/commands/status.mjs`; `src/lib/refresh.mjs` | `tests/kit/refresh.test.mjs`, `status-version-drift-refresh.test.mjs`; injected `refreshStages` plus `service` builds no collector | +| B1 | `src/lib/paths.mjs`; `src/lib/footprint/index.mjs`, `storage.mjs`, `consumers.mjs`, `storage-reclaim-detectors.mjs`, `install.mjs`; `src/lib/host-readiness-local.mjs`, `live/process-sessions.mjs`, `hook-audit/providers/opencode.mjs`, `usage-opencode.mjs`; `src/commands/uninstall.mjs` | `tests/kit/xdg-relative.test.mjs` and specified regressions; exact-head CI gate passed before edit; preserve nullable OpenCode fallback | +| B2 | `src/commands/x/daemon-gc.mjs`, `src/commands/x/host.mjs`, `src/commands/setup.mjs` | New `tests/kit/daemon-gc-rerecord.test.mjs`, `setup-host-rerecord.test.mjs`, `host-pick-rerecord.test.mjs`; Branch 9 Task 8 plus deferred host pick; compare `sync-host-repair.test.mjs` | +| B3 | `src/lib/ruflo-memory.mjs`, `paths.mjs` | `tests/kit/ruflo-memory-location.test.mjs`, `project-memory-status.test.mjs`; compose both unsuitable reasons and make `inside()` exclude equality | +| B4 | `src/commands/status/sections/project-memory.mjs`; `src/lib/live-check-evidence.mjs`, `live-checks.mjs` | **Accepted:** distinct `memory-routes` evidence binds installed CLI version and platform; generic `memory` cannot lower the row. Focused evidence, runner, status, and routing tests cover pass, upgrade, failure, timeout, and read-only render. | +| B5 | `src/lib/project-memory.mjs`; `src/commands/status/sections/user-memory.mjs`, `codex-mcp.mjs`; #757 registry entry | **Accepted:** bounded ordinary dot-folder discovery, read-only AQE home data row, and an AQE-owned init hint. `tests/kit/project-memory.test.mjs`, `project-memory-status.test.mjs`, `ruflo-memory-location.test.mjs`, `status-command.test.mjs` cover the three units. No real store was merged or moved. | +| B6 | `src/commands/status/sections/ruflo-components.mjs` | **Accepted:** applied-but-unverified keeps its state and meaning in the message and gives one restart/recheck instruction in its manual fix. Rendered-row and neighboring-state tests cover the contract. The Codex-hooks fix line remains conditional on a Ruflo-supported answer to #3419 and a pre-PR recheck. | +| B7 | `src/lib/ruflo-daemon-config.mjs`; `src/commands/sync.mjs`, `sync/plan-versions.mjs` | `tests/kit/sync-daemon-repair.test.mjs`, `sync-dry-run-preview.test.mjs`, `sync-skip-versions.test.mjs`; F6 hidden YAML keys and F7 versions-only preview parity | +| B8 | `src/lib/maintenance/discovery/orchestrator.mjs`, `history.mjs` | `tests/kit/maintenance-discovery-orchestrator.test.mjs`, `maintenance-recovery.test.mjs`; restart after pause shows paused history | +| B9 | `src/lib/exec.mjs`, `execution/process-tree.mjs` | `tests/kit/process-tree.test.mjs`; abort kills descendants; Windows CI required | +| B10 | `src/lib/maintenance/discovery/partitions.mjs`; inventory `src/lib/live/jsonl-tailer.mjs`, `live/transcript-streams.mjs`, `telemetry/store.mjs`, `maintenance/management/service-store.mjs` for additional persisted IDs | `tests/kit/file-identity-bigint.test.mjs`; distinguish IDs above `2^53`; enumerate the exact sites before edit | +| B11 | `src/lib/live-checks.mjs` | `tests/kit/live-checks.test.mjs`; skipped deja-vu check says skipped and check-created temp folders are cleaned | +| B12 | `src/commands/setup.mjs`; `src/lib/memory-probe-cleanup.mjs` unchanged | **Fixed:** `tests/kit/setup-memory-probe.test.mjs`; Ruflo 3.48.0 seeded reproduction created an unused native side file, and a disposable candidate run confirmed a private mirror leaves no canonical side file or probe row | +| B13 | `src/commands/sync.mjs`; `src/lib/aqe-project-pin.mjs` | Conditional check found the AQE pin converged across all four targets; no B13 product fix was made | +| C1 | `.github/workflows/ci.yml`; `src/lib/aqe-readiness.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | **Accepted; temporary CI job removed:** native macOS and Linux live-owner probes on released AQE 3.14.4 omitted `FsyncFailed`; the exact exception is retired. Ordinary `LockHeld` remains busy, while any `FsyncFailed` fails. The temporary CI job was removed in `257e6940` after its evidence was reviewed. #240 closure waits for the final main PR. No native Windows AQE conformance is claimed. | +| C2 | `docs/host-support.md`; `src/lib/hook-audit/agentic-dependency-constraints.json` | `tests/kit/ruflo-support-window.test.mjs` plus link check; verify AQE 3.14.4 #528/#532/#535 and Ruflo #2356/#420 first | +| C3 | `.github/workflows/nightly.yml`; `scripts/trace-ort.mjs` | Merged #277; native macOS trace run 36567908852. Exact approved comment posted and body verified at [Ruflo #2885](https://github.com/ruvnet/ruflo/issues/2885#issuecomment-5891510508). The post-heal trace is noncausal; the learning step remains nonblocking even though its outer command exited 1. | +| C4 | `scripts/upstream-watch/classify.mjs`, `fetch.mjs`, `ledger.mjs`, `dispatch.mjs`, `render.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | Merged #278 as `989c5e56`; develop CI 36578593054 passed all 13 jobs. M10 remained declined. | +| C5 | `src/lib/aqe-guidance.mjs`; `src/commands/setup.mjs`; ignored `.superpowers/sdd/2026-09-28-follow-ups-v2/c5-issue-draft.md` | Approved exact AQE repeated-init issue posted as [#778](https://github.com/proffesor-for-testing/agentic-qe/issues/778); the 3.14.4 disposable repro does not establish 3.14.5 behavior | +| C6 | `docs/host-support.md`; `src/lib/ruflo-support-window.mjs`, `aqe-readiness.mjs`; `src/lib/hook-audit/agentic-dependency-constraints.json` | 2026-09-29 13:42 UTC registry: Ruflo 3.48.0, AQE 3.14.5, Codex 0.159.0. Narrow disposable Ruflo one-file scan and integrity-verified native Codex read-only App Server initialize passed; AQE 3.14.4/3.14.5 live-lock proof passed on macOS/Linux. No provider turn or native Windows AQE proof. Local V4 gates passed at `29152654`; final PR CI pending | + +B1 used disposable homes, guarded focused tests, and the ignored B1 report at `.superpowers/sdd/2026-09-28-follow-ups-v2/b1-report.md`. No shared manifests, lockfiles, ADR index, or decision log change belongs to this plan update. The controller owns integration and the whole-branch gate. diff --git a/docs/archive/2026-09-28-plan-runner-hygiene.md b/docs/archive/2026-09-28-plan-runner-hygiene.md new file mode 100644 index 00000000..1385f29b --- /dev/null +++ b/docs/archive/2026-09-28-plan-runner-hygiene.md @@ -0,0 +1,57 @@ +# Runner hygiene execution plan + +## Status + +**Implemented and independently reviewed**, captured before PR integration on 2026-09-29. +Code head `e0fcc2eb` passed all local unit, UI, typecheck, lint, complexity, Markdown, +build and internal-link gates. Unit results: 5,947 passed, zero failed, six native +Windows-only skips; coverage 94.33% lines, 83.27% branches and 93.47% functions. +System Chrome passed 495 dashboard checks and 15 Node UI tests. Native Windows +and Linux Chrome proof remain the feature PR's CI gate at this capture point. + +Delivered: the About renderer regression, exact concurrent-writer exceptions, +owner records and guarded focused runs, list-only sibling handling on every platform, +Chrome environment isolation, tool-selector propagation regression, and owned-process +exit/cancellation checks. A participating test acquires a run-bound hold before launching +its children; uncertainty retains its own fixture and the enclosing run root. Holds do +not discover unregistered descendants or authorize sibling removal. + +Task 13's immutable inventory and literal-path list were independently reviewed against +snapshot `c4aa2015ccd66c47f99f9204d0444faac007e83e9a8b709ddcee04791819e6bb`: +33,978 entries before the runner cutoff, four afterward and three legacy ownerless roots; +27 unattributed, 21 recent and one lsof-matched entry were retained separately. The private +handoff contains machine paths and stays outside git. Age, prefix attribution and lsof +absence do not establish deletion safety. No backlog removal was performed. + +The source-bound research census recorded 647 sites in 279 files at its stated baseline; +it does not count later edits or prove historical leak causes. LQ-1's propagation-defect +premise was refuted, and a mutation-tested regression preserves the existing behavior. +Final review found one cancellation-order defect; `e0fcc2eb` fixed it and passed scoped +rereview. Releases, global installation and the aggregate main merge remain separately gated. + +## Contract and dependencies + +The [v2 scope](../plans/2026-09-28-remediation-program-v2.md#v5-testrunner-hygiene-suites-that-clean-up-after-themselves) inherits [archived Branch 9](2026-09-28-superpowers-plan-branch-9-follow-ups.md) Tasks 5, 7 and 10–13, under B9-R1–R8. V1 is integrated at the baseline. V4/V6 must coordinate before changing environment-helper consumers. Worktree ownership is limited to this branch; shared manifests remain the integration owner's responsibility. + +The [cleanup design](2026-09-28-research-test-temp-folder-cleanup.md) selects B9-R5's explicit list-only fallback. Task 11 must not interpret a dead PID, empty process group, empty registry or empty handle scan as proof of abandonment. Task 12 must keep the interrupted root even after its known child exits on list-only platforms. This conditions the archived example's removal assertion; it does not relax B9-R5. + +## File, test and dependency map + +| Unit | Exact owned files or proposed files | Validation and acceptance | Depends on | +|---|---|---|---| +| Task 10 research | This plan; `docs/plans/2026-09-28-test-temp-folder-cleanup-design.md`; ignored scratch report/probes | Real macOS orphan observation, official docs, explicit unmeasured platforms; Markdown, links, docs citations/layout | Baseline and brief | +| Task 5 About render | New `tests/kit/about-install-edit-render.test.mjs`; read `src/lib/dashboard/client/about.mjs`, `src/lib/install-edits.mjs`, `src/commands/status/sections/natives.mjs` | Real About renderer, escaped single Ruflo pin line, no AgentDB line/no-edit line; wording mutation fails; existing `about-install-edits.test.mjs`, `about-agentdb-join.test.mjs` | Research handoff; fresh file claims | +| Task 7 concurrent writers | `scripts/real-state-tripwire.mjs`, `tests/kit/real-state-tripwire.test.mjs` | Absent-to-present `.claude-flow`, two proven-config files are concurrent locally, fail in strict mode; config.json still fails; `run-tests-runner.test.mjs`, `home-sandbox-tripwire.test.mjs` | Reverify installed supported Ruflo sources; no fabricated version | +| Task 11 owner record and safe paths | New `scripts/run-roots.mjs`, `tests/kit/run-roots.test.mjs`; `scripts/run-tests.mjs`, `tests/kit/run-tests-runner.test.mjs`, `AGENTS.md` | Red then green path/owner/schema/symlink/UID/host/invalid data tests; listing does not change exit; interrupted roots never removed by defaults; `real-state-tripwire.test.mjs`, `temp-dir-helper.test.mjs` | Task 10 decisions accepted; preserve synchronous spawn/signal behavior | +| Task 12 focus and exit proof | `scripts/run-tests.mjs`, `tests/kit/run-tests-runner.test.mjs`, `AGENTS.md` | Focus pass=0, leftover=4, missing args=2; killed runner's idle child and concurrently live runner retained; root still retained after child death on list-only platforms; exact-PID cleanup finally | Task 11; controller updates external brief template | +| Windows smoke lifetime diagnosis | Proposed `tests/kit/status-zero-spawn.test.mjs`; new disposable fixture only if needed | Handshake shows whether fork outlives parent; compare explicit close wait; preserve assertion and cleanup errors; Windows Node 24 regression required | Controller authorizes implementation; native Windows CI, not fixture simulation | +| LQ-1 environment premise | Read `scripts/run-tests.mjs`, `tests/kit/helpers/home-sandbox.mjs`; if proven defect, those files plus `tests/kit/run-tests-runner.test.mjs`, `tests/kit/spawn-env-guard.test.mjs` | Current runner deletes FORCE_COLOR only; demonstrate sentinel AQE variables across each actual child boundary before any patch; retain state isolation | Task 12; reconcile scope text with source | +| LQ-4 Chrome environment | `tests/ui/helpers/launch-chrome.mjs`; new `tests/kit/launch-chrome-env.test.mjs`; environment helper only with exact claim | Preserve required display/path/platform variables and private Chrome temp; exclude user state; mocked launch failure and close cleanup; native UI smoke | Task 12; coordinate helper users; installed Playwright available | +| Task 13 reviewed inventory | Ignored source/list/report in controller-approved report folder, no tracked cleanup program | Literal absolute paths, prefix attribution, exclusion counts, independent review of same snapshot; no removal | Last focused run; controller provides report destination/reviewer | +| Branch handoff | Update this plan and design status; archive via `scripts/docs-relocate.mjs` in completion PR with index rows | Focused gates, type/lint/build and hermetic full gate appropriate to eventual implementation; exact commit/evidence receipt | All implementation units complete; integration approval | + +## Execution boundaries + +Use `node scripts/run-tests.mjs focus ` with disposable home/state roots for focused tests. Do not use pnpm in a worktree with symlinked dependencies. Run focused failure-path checks before wider gates; do not repeat green gates without a new concern. Each code unit needs its own failing/passing evidence and conventional commit after authorization. Before editing/staging `AGENTS.md`, verify there is no injected drift. + +Task 13 excludes recently modified entries, live-handle matches, unattributed prefixes, product-created `ak-sync-preview-npm-*`, and valid owner roots. Age is a manual-review filter only. Task 13 does not implement deletion. No unit claims interrupted-run backlog reclamation until a platform can prove every descendant gone. diff --git a/docs/archive/2026-09-28-plan-usage-accuracy.md b/docs/archive/2026-09-28-plan-usage-accuracy.md new file mode 100644 index 00000000..0c0eae8c --- /dev/null +++ b/docs/archive/2026-09-28-plan-usage-accuracy.md @@ -0,0 +1,200 @@ +# Usage accuracy execution plan + +- **Branch:** `fix/usage-accuracy`, based on `develop@e2f9dcae0554ff63921df618a819fd5e6afe80d2` +- **Scope sources:** [v2 V6](../plans/2026-09-28-remediation-program-v2.md), + [v1 Wave 4](../plans/2026-09-26-remediation-program.md), + [ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md). + +## Status + +Implemented on `fix/usage-accuracy`; all 24 units are independently accepted. +The whole-branch review and its scoped correction review passed through `6d5979cc`. +A one-line legacy test correction at `1c02db91` passed controller review, followed +by all eight local gates on that exact clean commit: 6,405 unit tests passed, +seven skipped, legacy suites passed, and browser checks passed (514 legacy +assertions plus 16 native tests). Coverage was 94.19% lines, 83.82% branches and +93.50% functions. Feature PR CI, develop integration and final human main review +remain separate gates at this archival capture. No release, installation or real +store operation is claimed. Unit 11 captures Claude Code's +latest valid cumulative `cost-state` checkpoint as a separate reconciliation +signal. It reports provable time or token scope differences while preserving +message-derived cost totals. The observed checkpoint has no end time or serving +provider attestation, so equal counters remain unverified. Unit 12 now retains +bounded, hashed Claude API message identities per cached file and reconciles +copied charges after discovery. Aggregate responses/tokens/cost count a shared +message once; each session's `responses` still counts what its transcript +recorded, with `accountedResponses` showing its aggregate share. Unit 12 is +accepted, including its bounded global-message-owner policy. The Claude identity pool always covers the +displayed window and its equal-length predecessor, regardless of the +`previous` or `lookbackDays` options. One owner is elected per identity across +that pool before either window is projected; a copied message therefore +contributes to at most one of the two windows. Explicit deeper lookback can +support other history views but cannot change an eligible owner's charge. +Eligibility requires both the file mtime and transcript session end to reach +the fixed horizon; older-mtime copies discovered by a deeper lookback cannot +steal or enlarge it. Distinct historical messages remain visible under that +explicit request, with out-of-pool coverage reported in source health. Claude +reads may reach twice the displayed window (capped at 730 days for the +dashboard's 365-day maximum), with the cap reported for wider callers. +Unit 13 records an entry-level local calendar context: the resolved full timezone +identity plus Node's tzdata and ICU versions. A mismatch or missing/invalid context +reparses available source records; no timestamp is inferred from a cached day. +Process memo and single-flight keys include that context. Degraded OpenCode entries +retain their original marker but cannot contribute incompatible day/punchcard rows; +source health reports `timezoneCacheEntriesExcluded`. Unset `TZ` uses the runtime's +resolved machine zone; an unresolved zone declines cache and aggregate-memo reuse. +Schema remains 26, preserving Unit 15's independent OpenCode cost marker. Unit 13 +is accepted. Unit 14 adds count-only Claude record coverage for known handled, +known ignored, unknown, invalid-type and malformed JSON lines. A v26 cache entry +without those counters reparses; unknown or malformed records degrade source +health without changing message usage or cost. The bounded real-data sample and +focused verification are in the ignored task14 handoff report. Unit 14 and Units 15–19 are independently accepted. + +Pricing retains its existing local `row.day` contract (`usage-parsers.localDay`, +`pricing.costOf`, and the cache-saving probes documented in usage metrics). Cold +and warm reads within a zone must agree, including dated rates. A timezone change +can move a row across a dated rate boundary and therefore change its API-equivalent +estimate; this unit neither establishes a provider billing timezone nor freezes a +price from the old local day. Reported OpenCode cost stays observed, and token, +provider, model, Claude cost-state, Codex fields and duration semantics stay intact. + +Unit 2 is limited to the agreed classifier interface, parser fields and one +usage-cache schema bump. +The maintainer approved retaining unfamiliar, bounded tokens from named origin +fields as raw evidence for local detail. The classifier now retains those tokens +without inferring a product or provider; malformed, oversized and non-string +values remain excluded. Units 22/23 delivered the local detail UI and count/coverage disclosures. Unit 24 records +ADR-0060 acceptance against those implemented contracts, subject to documentation review. + +Unit 6 records Amazon Bedrock or Google Vertex AI only when a Claude assistant +message carries a provider-specific model ID. Conflicting or ordinary IDs leave +the provider unknown. Historical transcripts do not capture launch environment +or settings, so current configuration cannot identify their serving provider; +OpenRouter, other gateways and private endpoints remain unknown without bound +session evidence. The detail stays under `sessionOrigin.thirdPartyProvider` with +`thirdPartyProviderBasis: assistant-model-id` when known. Units 22/23 implement display. + +## Gates and ownership + +All source units have completed their assigned implementation and independent review. Unit 24 +owns the affected ADR bodies and living guides in the assigned worktree. The controller owns +shared indexes/manifests, final integration gates, whole-branch review, plan archival and any +separately authorized publication. Unit 24 runs documentation gates only. No private transcript +content, raw identifiers, paths or observed costs enter public documentation. + +## Capture units + +| Unit | Source boundary and acceptance | Focused evidence / dependency | +|---|---|---| +| 1 | Shared raw-value → surface → initiator → label vocabulary; bounded raw evidence, unknown and provider separate. Adapt footprint origin with legacy fields retained. | ADR table fixtures, first declaration, privacy, identity and imports regressions. Accepted. | +| 2 | Parser and usage cache integration, exactly one schema 25→26 bump. Carry new fields and rebuild old cache. | Parser, cache migration and aggregate tests; after unit 1. Footprint schema stays 8. | +| 3 | Full Codex `thread_source` classification, subagent/reviewer rollup and unpriced Auto-review models (X-7). | Per-value parser fixtures and counts; after unit 2. | +| 4 | Count Claude by `sessionId`, exclude subagent and bridge transcripts, and remeasure source-bound census. | Duplicate/session fixtures plus enumerated-count reproduction; after unit 2. | +| 5 | Runtime census symmetry for Claude.app and ChatGPT.app; attribute bundled CLIs to observed sessions. | Runtime fixtures on both app forms; after units 2–4. | +| 6 | Third-party Claude provider from session-bound provider-specific assistant model ID; unknown remains unknown and provider is a separate detail field. Cloud label renders only for observations. | Evidence-precedence and unknown fixtures; after unit 2. | +| 7 | Imported turn exclusion per turn; later genuine Codex turns count and establish an actual app origin (decision 12/B1-4). | Mixed import/real-turn fixture and enumerated-count reproduction; after unit 2. | +| 8 | Token-bearing Codex record with zero responses: count it or document exact unsupported shape (UA-5). | Minimal shape reproduction and reconciliation; after unit 2. | +| 9 | O-7 session `byProvider` last-wins repair. | Count-only reproduction, provider totals; after parser integration. | +| 10 | X-8 Codex effort, first-token time and compaction capture. | Field fixtures and aggregate reconciliation; after parser integration. | +| 11 | C-6 Claude `cost-state` reconciliation. | Cost-state fixture and count-only sample; after parser integration. | +| 12 | C-8 cross-file message-id dedup. | Duplicate message fixture and count-only sample; after parser integration. | +| 13 | C-9 local-timezone day bucketing frozen in cache. | Boundary-day fixtures in two zones; after cache integration. | +| 14 | C-11 unknown-record counter. | Known/unknown record fixtures; after parser integration. | +| 15–19 | O-6, O-9, O-10, O-11, O-12, each in a separate commit. | Accepted: cost trust and cache semantics, explicit database selection, V2/legacy coverage warnings, bounded compaction/reconciliation and child fingerprint exclusion. | +| 20 | StatusLine classifier reads local managed settings. | Settings fixture and classifier regression; independently reordered before V4 integration. Other managed policy channels remain unobserved. | +| 21 | Shell wrapper around footer helper is `custom` (UA-4). | Wrapper fixture; after unit 20. | +| 22 | Shared labels across Usage, System Projects, Maintenance and Intelligence; unknown has one label and designations use separate axes. | View assertions; **after V3 merges into develop**, then integrate develop. | +| 23 | Show imported exclusion count in Intelligence census, System Projects and `ak system`; disclose Cowork source coverage is absent. | Three render assertions and source-bound counts; after V3 and unit 7. | +| 24 | Accept ADR-0060 and align DDD/docs to verified implementation. | Docs links and drift checks; assigned documentation writer after all prior units; controller owns shared indexes. | + +## Source and test map + +This table preserves the initial candidate boundaries. Exact delivered files and source-bound +results are in the private unit reports; the candidates grant no new edit authority. All source +units are accepted and only the documentation candidate remains under review. + +| Unit | Candidate source | Test entrypoint | +|---|---|---| +| 2 | `src/lib/usage-parsers.mjs`, `usage-project-evidence.mjs`, `usage-index.mjs` | `tests/kit/usage-index.test.mjs`, `usage-codex-attribution.test.mjs` | +| 3 | `src/lib/usage-parsers.mjs`, `usage-aggregate.mjs` | `tests/kit/usage-codex-attribution.test.mjs`, `usage-classify.test.mjs` | +| 4 | `src/lib/usage-parsers.mjs`, `footprint/project-sources.mjs` | `tests/kit/usage-claude-dedup.test.mjs`, `dashboard-project-identity.test.mjs` | +| 5 | `src/lib/footprint/runtime.mjs`, `project-census.mjs` | `tests/kit/footprint-collectors.test.mjs`, `system-summary.test.mjs` | +| 6 | `src/lib/usage-parsers.mjs`, `usage-local-provider.mjs` | `tests/kit/usage-provenance.test.mjs`, `usage-local-pricing.test.mjs` | +| 7 | `src/lib/codex-import-marker.mjs`, `usage-parsers.mjs`, `footprint/project-sources.mjs` | `tests/kit/project-sources-imports.test.mjs`, `usage-codex-attribution.test.mjs` | +| 8 | `src/lib/usage-parsers.mjs`, `usage-aggregate.mjs` | `tests/kit/usage-codex-attribution.test.mjs`, `usage-codex-large-rollout.test.mjs` | +| 9 | `src/lib/usage-opencode.mjs`, `usage-aggregate.mjs`; parser row identity already exists | `tests/kit/usage-opencode.test.mjs`, `usage-index-opencode.test.mjs`, `usage-index.test.mjs` | +| 10 | `src/lib/usage-parsers.mjs`, `usage-insights.mjs` | `tests/kit/usage-codex-attribution.test.mjs`, `usage-context.test.mjs` | +| 11 | `src/lib/usage-parsers.mjs`, `usage-cost.mjs`, `usage-aggregate.mjs`, `usage-index.mjs` | `tests/kit/usage-claude-cost-state.test.mjs`, `usage-claude-dedup.test.mjs`, `usage-index.test.mjs` | +| 12 | `src/lib/usage-parsers.mjs`, `usage-index.mjs` | `tests/kit/usage-claude-dedup.test.mjs`, `usage-index.test.mjs` | +| 13 | `src/lib/usage-index.mjs`, `usage-aggregate.mjs` | `tests/kit/usage-index.test.mjs`, `usage-claude-window-pairing.test.mjs` | +| 14 | `src/lib/usage-parsers.mjs`, `usage-aggregate.mjs` | `tests/kit/usage-index.test.mjs`, `usage-telemetry.test.mjs` | +| 15 O-6 | `src/lib/usage-opencode.mjs`, `usage-cost.mjs` | `tests/kit/usage-opencode.test.mjs`; accepted | +| 16 O-9 | `src/lib/usage-opencode.mjs`, `usage-opencode-bounds.mjs` | `tests/kit/usage-index-opencode.test.mjs`; accepted | +| 17 O-10 | `src/lib/usage-opencode.mjs`, `usage-index.mjs` | `tests/kit/usage-opencode.test.mjs`; accepted | +| 18 O-11 | `src/lib/usage-opencode.mjs`, `usage-aggregate.mjs` | `tests/kit/usage-opencode.test.mjs`; accepted | +| 19 O-12 | `src/lib/usage-opencode.mjs`, `usage-parsers.mjs` | `tests/kit/usage-opencode.test.mjs`; accepted | +| 20 | `src/lib/quota.mjs` | `tests/kit/quota.test.mjs`, `usage-limits-empty-state.test.mjs` | +| 21 | `src/lib/quota.mjs` | `tests/kit/quota.test.mjs` | +| 22 | `src/lib/dashboard/client/usage.mjs`, `system-projects.mjs`, `intelligence.mjs`, `maintenance-filters.mjs` | `tests/kit/dashboard-project-groups.test.mjs`, `intelligence-table-groups.test.mjs`, `maintenance-dashboard-client-labels.test.mjs` | +| 23 | `src/lib/dashboard/client/intelligence.mjs`, `system-projects.mjs`, `src/commands/system.mjs` | `tests/kit/dashboard-intel-integration.test.mjs`, `system-command.test.mjs` | +| 24 | `docs/adr/0060-session-surface-initiator-and-product-names.md`, relevant DDD guide | `tests/kit/docs-layout.test.mjs` and Markdown lint | + +Units 9–19 recorded bounded source observations and synthetic affected-row evidence in their +unit reports. A sample without an affected row is not proof of current-user impact. Reference +counts in ADR-0060 remain historical. No new real-data probe runs in Unit 24. + +## Unit 18 accepted: OpenCode compaction and reconciliation + +The parser reads bounded compaction parts and selected session metadata. A user +compaction request plus an error-free assistant summary with a finish value and +that actual parent link establishes one completed observation per request. +Requests alone, orphan summaries and in-flight markers retain uncertainty in the +lower/upper bounds. Failed or aborted summaries do not establish completion. +The OpenCode aggregate projection now retains those bounds; Codex and Claude +projections retain their existing behavior. + +Session counters are diagnostic only. Exact OpenCode v1.18.33 source shows that +session totals accumulate step-finish parts, but assistant tokens hold the latest +step. Reconciliation therefore requires completed valid messages, exactly one +matching valid step per assistant, populated valid session counters, and no V2 +rows in that session. Multiple or missing steps, incomplete metadata, unsupported +versions/token bases and untrusted hosted zero costs remain unknown. Matching or +mismatching counters never replace or add to message usage. No steps are billed a +second time. The cache marker extends cost-trust-v2 with observations-v1; schema +26, source identity and timezone checks remain intact. + +The bounded local metadata sample contained three sessions, no compaction parts, +no in-flight markers and no populated session counters. Its database digest was +unchanged. This is not positive affected-user evidence; synthetic fixtures cover +the supported and failure cases. Detailed commands and evidence are in the +ignored task18 report. + +Unit 18 review fixes bind warm reuse to a SHA-256 digest of the selected session's +observation metadata, relevant message fields, compaction/step-finish parts, and +V2 scope presence. Both the probe and parser stay within the same per-session +acquisition ceilings and their own read snapshots; the persisted digest comes +from the parse snapshot. Unchanged inputs reuse the cache; same-count rewrites +and removals invalidate it even when upstream timestamps do not change. No +whole-database payload hash or prompt-body hash is used. + +Response-free request evidence now remains in the current/previous compaction +bounds. Refused acquisitions contribute only their unknown bound, preserving the +existing rule that they do not become ordinary zero-cost session rows. Neither +path manufactures responses, tokens or billing. Unit 18 was accepted before +Unit 19 began. + +## Unit 19 accepted: OpenCode child prompt fingerprints + +OpenCode sessions with a nonempty `parent_id` retain prompts, turns, tokens, +provider costs and subagent classification, but produce no prompt fingerprints. +Both scan and selected-session parsing apply this rule. Aggregation also excludes +old cached child fingerprints from typed-prompt metrics, current prompt patterns +and historical baselines; the cached bytes remain until normal invalidation. +Main-session behavior and Unit 18 cache identity, timezone and observation marker +checks remain intact. Schema 26 is unchanged. + +Synthetic native-schema tests cover matching and distinct parent/child text, +child-only historical windows, absent parent rows, scan/read parity, stale warm +cache consumption and retained provider usage. The prior bounded preflight found +no child sessions; it does not establish current-user impact. Unit 19 is independently accepted; Unit 24 is the documentation candidate. Commands and results are in the ignored +task19 report. diff --git a/docs/archive/2026-09-28-research-test-temp-folder-cleanup.md b/docs/archive/2026-09-28-research-test-temp-folder-cleanup.md new file mode 100644 index 00000000..a4a58dc0 --- /dev/null +++ b/docs/archive/2026-09-28-research-test-temp-folder-cleanup.md @@ -0,0 +1,449 @@ +# Test temp folder cleanup design + +## Status + +**Research complete.** The selected list-only policy was implemented and independently +reviewed in the [execution plan](2026-09-28-plan-runner-hygiene.md), through `e0fcc2eb`. +The remainder of this document preserves the initial research snapshot and its limits. + +Research snapshot: 2026-09-28, `e2f9dcae0554ff63921df618a819fd5e6afe80d2`, macOS Darwin 27.0.0, Node 26.4.0; Node 22.22.3 used for CLI checks. Selected behavior is **list-only for abandoned sibling roots on macOS, Linux and Windows**. No candidate establishes complete descendant liveness. This uses B9-R5's explicit fallback, preserves B9-R1–R8, and introduces no native sweeper or deletion authority. + +The [execution plan](2026-09-28-plan-runner-hygiene.md) maps subsequent work. Raw commands, JSON results, full lexical census, counts and literal experiment paths are retained in ignored `.superpowers/sdd/2026-09-28-runner-hygiene/`. No production/test code changed during this research unit. Creator lifecycles at that baseline are fully classified below; native Windows/Linux behavior remains explicitly unmeasured. + +## Verified runner and creator behavior + +At `scripts/run-tests.mjs:57-85`, the runner creates a unique suite root, redirects all three temp variables, runs synchronous commands, reports leftovers and removes its root after commands finish, including failure. SIGKILL cannot reach this cleanup. The source has no owner record or sibling collector. It strips FORCE_COLOR only (`scripts/run-tests.mjs:68`); LQ-1's premise that this layer strips AQE_EMBEDDER variables is refuted at this revision. Downstream boundaries still need sentinel tests. + +The completed AST and source census below classifies 647 creator sites across 279 files. The original 634-hit lexical inventory is superseded; two comment hits were excluded, three inline allocations and seven aliased calls were recovered, five sandboxConfigBase calls were added, and two executable child templates were retained separately. + +| Lifecycle class | Source evidence | Failure boundary | +|---|---|---| +| Cleanup registered before caller assertions | `tests/kit/helpers/temp-dir.mjs:16-20` creates then registers t.after or file after; `tests/kit/run-tests-runner.test.mjs:15-16` registers immediately | Assertion failure is covered after registration; process kill, allocation-to-registration failure and removal error remain | +| Manual cleanup after successful operations | `tests/kit/status-zero-spawn.test.mjs:86-90` removes its ledger after exec/read | Failed exec or parsing skips ledger removal; the parent run still diagnoses it | +| Module-level creator with file hook | `tests/kit/evidence.test.mjs:9` and `:231`; `tests/kit/refresh.test.mjs:17-19` | Abrupt exit misses hooks; in evidence, an early import/assertion may occur before hook registration | +| Exit hook in plain scripts | `tests/kit/helpers/private-tmpdir.cjs:11-16`; dashboard/statusline callers | Normal exit runs synchronous removal; SIGKILL never does; registration is after allocation | +| Caller-owned tools state | `tests/kit/helpers/home-sandbox.mjs:108-130` | redirectToolState returns restore; caller must arrange finally/hook before assertions; helper itself registers none | +| Caller-owned home/project | `tests/kit/helpers/home-sandbox.mjs:141-157` and `:303-306` | Creation does not register cleanup; consumers determine lifetime | +| Child temp base | `tests/kit/helpers/home-sandbox.mjs:83-97` | spawnEnv uses home/tmp and permits extra overrides; containment requires caller's home and final TMP values to be inside run root | +| Non-Node child | `tests/ui/helpers/launch-chrome.mjs:20-33` | Launch failure and browser.close remove private root; killed test or omitted close bypasses cleanup; Chrome is outside a Node-only registry | +| Hardcoded /tmp strings | `tests/kit/dashboard-live-source.test.mjs:7-21` | Path parsing and assertions only; these lines create no directory or file | + +Every site now has a current cleanup mechanism and evidence reference. This does not establish historical leak causality, successful native cleanup, or freedom from setup-before-registration gaps. The success-path classes identify assertion/error leak exposure; the parent-hook classes distinguish allocations already covered by enclosing cleanup. + +## Child temp routing and paths outside the run root + +The retained `temp-routing-sites.txt` records explicit temp-variable sites. `spawnEnv` places child temp under the supplied home; both it and redirectToolState remain beneath the outer suite when the home/base was created there. Nested runner refusal uses a temp base deliberately inside a disposable project (`tests/kit/run-tests-runner.test.mjs:119`); harvest redirects all three variables into its disposable repository (`tests/kit/agentdb-retirement.test.mjs:221-225`). Neither is a real repository escape. Chrome pins all four platform variables at `tests/ui/helpers/launch-chrome.mjs:22`. + +The source does contain child environments that lose outer-root containment: the live AQE version probe at `tests/live/aqe-stop-hook-conformance.test.mjs:40` passes only PATH and NO_COLOR, so it does not inherit the runner's temp variables. Shell command-discovery probes at `tests/live/aqe-stop-hook-conformance.test.mjs:24` and `tests/live/aqe-codex-guidance-conformance.test.mjs:43` likewise use PATH-only environments. These are source-confirmed routing gaps, not measured folder leaks. Later live fixtures explicitly set TMPDIR but do not establish Windows TEMP/TMP containment. The proof-key guard test sets only TMPDIR (`tests/kit/aqe-live-proof-key-guard.test.mjs:24`); its expected early refusal does not make that a portable temp-isolation contract. + +Literal Windows temp paths in ruflo-memory-location tests are injected path-classification inputs, not spawned child environments. dashboard-live-source's /tmp values are also parsing inputs. No cleanup design may assume every subprocess preserves the root simply because the top runner sets it. + +## The fourteen reported post-runner leaks + +The [archived premise table](2026-09-28-superpowers-plan-branch-9-follow-ups.md#premise-verification-done-by-the-planner-tasks-carry-the-evidence-forward) records fourteen newer folders by prefix. These are historical observations, not fourteen reproduced failures today. + +| Historical entries | Attributed creator and cleanup | What can be concluded | +|---|---|---| +| ak-evidence-home ×2 | `tests/kit/evidence.test.mjs:9`, file after at `:231` | Creation precedes imports/assertions and late hook. Early failure or killed process can leak; exact historical cause unknown | +| ak-ruflo-components-evidence-location-home ×2 | `tests/kit/ruflo-components-evidence-location.test.mjs:6`, after at `:18` | Early import/assertion before hook or interruption possible; exact cause unknown | +| ak-refresh-home ×3; ak-refresh-proj ×3 | `tests/kit/refresh.test.mjs:17-19` | Hook is early, so ordinary later assertion failure should clean. Kill, failure before registration or failed removal remain hypotheses | +| ak-status-live-home ×3 | `tests/kit/status-live.test.mjs:17`, after at `:267` | Late registration leaves early initialization failure window; interruption/removal failure possible | +| ak-live-checks-home ×1 | `tests/kit/live-checks.test.mjs:21`, after at `:688` | Late registration has the same exposure; no historical exit trace identifies cause | + +A basename and mtime cannot tell whether an assertion, import, kill or cleanup error caused a specific leak. Reconstructing that requires corresponding process/test logs. Do not rewrite this attribution as proof that every ordinary failed test leaks. + +## Concurrent runs and race windows + +Each mkdtemp root is unique, but sibling worktrees share the temp parent. An owner file is proposed to be written atomically via a temporary file and rename immediately after root creation. Until a valid record exists, keep the root (B9-R3). Old runners also have no owner and remain untouched. Malformed records, wrong host/user/platform, unreadable entries and unknown schema all mean keep. + +An owner may finish during inspection, a PID may be reused, a descendant may start after a snapshot, and a root could change between validation and deletion. Owner records and start identities do not close these races. Current selection performs no sibling removal, so racing observations cannot authorize it. A future collector needs complete process containment plus a stable filesystem identity/revalidation protocol; a path name and PID are insufficient. + +## Abandonment candidates and proof limits + +| Candidate | macOS | Linux | Windows | Signal and cost assessment | +|---|---|---|---|---| +| A: process group / parent descent | Reject: detached descendants escape recorded group | Same source counterexample; no native execution here | Reject: snapshots can lose exited intermediate parents; PID reuse complicates descent | detached changes session/group and requires signal forwarding; no equivalence proof | +| B: parent identity plus handle scan / descent | Reject: measured live orphan has no root handle | Same logical gap; native /proc and lsof not measured | ParentProcessId/CreationDate cannot recover every missing intermediate ancestor | Avoiding signal changes is possible; fast probes still cannot prove absence | +| C: every Node process imports PID registrar | Reject: non-Node children; env replacement; startup registration race | Same coverage gap; no native execution | Same coverage gap; no native execution | NODE_OPTIONS affects children and can be removed; no transparent behavior proof | +| Selected: list-only | Never returns abandoned | Never returns abandoned | Never returns abandoned | No process launch/signal change; no expensive liveness probe required for removal | + +`tests/kit/process-tree.test.mjs:52`, `tests/kit/exec-kill-tree.test.mjs:53` and `tests/kit/mcp-tool-call.test.mjs:21` deliberately use detached children. POSIX detached children create a new group/session; unref and stdio determine parent waiting behavior. Thus even an empty original group is insufficient. [Node child process documentation](https://nodejs.org/api/child_process.html#optionsdetached). + +Windows ParentProcessId may refer to a dead or reused parent. CreationDate helps disambiguate identity, but a snapshot cannot reconstruct an already vanished chain of intermediate processes. This is a design inference from the documented fields, not a native Windows experiment. [Microsoft Win32_Process](https://learn.microsoft.com/en-us/windows/win32/cimwin32prov/win32-process). + +B9-R6's startedAt milliseconds and 2-second reuse tolerance can distinguish a later process, but never establishes that the original process's children exited. Clock precision, permission errors and unparseable identity must resolve to unknown/keep. A newly started owner record should bind observed process start identity, not simply assume its file-write timestamp is the process start. + +## Real macOS orphan experiment and timings + +Scratch `probe.mjs` created an isolated temp parent, home, project and tmp. It launched the existing runner using `exec --repo -- `. The child wrote PID/cwd/TMPDIR to a handshake file then idled for 60 seconds. The controller killed only the exact runner PID, waited for its exit, inspected the known child and finally sent SIGTERM to that child. No process search was used to select kill targets. + +Trimmed evidence: + +```text +runner PID 57521: SIGKILL +child PID 57522: alive=true + PID PPID PGID COMMAND +57522 1 57490 node /sleep.mjs +child TMPDIR: /tmp/ak-suite-tHEZEe +child cwd: /project +root exists=true +lsof -nP +D : exit 1, stdout empty, stderr empty +cleanup ps -p 57522: header only; child gone +``` + +The child cwd was the disposable project, outside its suite root. It had no open root file. The root was retained for review. This proves candidate B unsound for this case and disproves parent-death-only collection. The script's finally targets only its exact owned PIDs; its bounded timer is secondary protection. + +Five sequential samples on this host (no cross-platform claim): + +| Probe/workload | Measured range | Median | +|---|---|---| +| process.kill(childPid, 0) | 0.00075–0.008 ms | 0.00108 ms | +| ps -o lstart= -p childPid | 1.625–2.054 ms | 1.742 ms | +| lsof -nP +D empty suite root | 144.769–150.222 ms | 145.634 ms | +| Plain node --test inert.test.mjs | 64.153–66.150 ms | 64.621 ms | +| Guarded exec of same one-file test | 91.135–100.622 ms | 96.753 ms | + +Observed median guarded overhead was 32.132 ms with 22 empty sandbox tripwire roots. This is not a measurement of a populated real home, a large backlog or the future collector. `/proc` and PowerShell CIM cost are **unmeasured** because there is no native Linux/Windows runtime in this task. A list-only inventory still needs bounded I/O and later performance validation against many roots; no production latency claim is made. + +## Windows locks and CI failure + +The controller supplied CI run **36520154869**, Windows Node 24: `status-zero-spawn.test.mjs` failed in inSandbox rmSync(project), line 55, EPERM; the leftover was ak-spawn-guard-smoke-proj. This failure is attributed evidence from the brief, not a freshly downloaded CI log. + +Verified source: the smoke script forks at `tests/kit/status-zero-spawn.test.mjs:82` then calls process.exit(0) at `:83`, explicitly avoiding waiting for the grandchild. execFileSync waits for that immediate child; it does not establish that the grandchild released its project cwd. A surviving grandchild causing the observed EPERM is a plausible hypothesis, not proven root cause. The fork target itself exits immediately, making scheduling relevant. + +Proposed regression: in a copied disposable fixture, hold the fork on a bounded handshake; record exact child/grandchild PID and spawn/exit/close timestamps; attempt cleanup while held; compare an explicit wait-for-close variant. Capture Windows error code/path and remaining entries. Preserve both assertion and cleanup errors rather than letting finally mask the first. The ignored `windows-probe-plan.md` specifies native Windows Node 24/26 diagnostics, one timed CIM snapshot and exact-PID cleanup. No workflow was edited or run. + +Recursive rm is not atomic; an error can leave a partially removed tree. On a removal error, report retained/partially removed and never claim an intact preserved root. Windows cwd/open-handle behavior depends on handle sharing; not every open handle universally blocks deletion. Native locking behavior remains unmeasured here. The proof must precede any future removal attempt. + +Node documents recursive rm retries for EBUSY, EMFILE, ENFILE, ENOTEMPTY and EPERM with linear backoff; maxRetries defaults to 0 and retryDelay to 100 ms. These options are ignored without recursive mode. Retries cannot establish ownership or abandonment. [Node fs.rmSync](https://nodejs.org/api/fs.html#fsrmsyncpath-options). + +## Focused runs on Node 22 and 26 + +Both installed binaries were executed in the sandbox with an invalid node.config.json and one inert test. Plain `node --test inert.test.mjs` passed on 22.22.3 and 26.4.0, demonstrating that neither loaded the config by default. Adding `--experimental-default-config-file` failed with exit 9 and invalid-content diagnostics on both. Explicit `--experimental-config-file` and `--import` are also opt-in; external NODE_OPTIONS can inject imports, but a repository file cannot silently establish that environment. + +`--test-global-setup=./missing.mjs` is rejected as a bad option (exit 9) by installed 22.22.3; 26.4.0 recognizes it and fails resolving the intentionally missing module (exit 7). Thus the inherited wording must not imply global setup exists on Node 22.22.3. B9-R7's conclusion stands: plain node --test is unguarded; use the explicit wrapper. [Node 22.22.3 CLI](https://nodejs.org/download/release/v22.22.3/docs/api/cli.html), [Node 26.4.0 CLI](https://nodejs.org/download/release/v26.4.0/docs/api/cli.html). + +## Backlog and classification boundary + +A read-only direct-child listing of the real temp parent observed **34,013** current-user, non-symlink entries beginning ak-, grouped into 127 suffix-normalized prefixes. These raw counts include this research's own new root and concurrent activity; they are not removal candidates. The full per-prefix snapshot is `backlog-counts.json`, generated by retained `census.py`. Leading counts: ak-usage 2,520; ak-adapter-conformance-cli 2,301; ak-quota 2,152; ak-live-service 2,150; ak-adapter-consent 2,124; ak-intel-history 2,106; ak-adapter-grants 1,836; ak-stamp 1,525; ak-conformance-tiers-grants 1,512; ak-usage-solo 1,400; ak-host-cli 855; ak-host-project 855. + +Task 13 must take a fresh snapshot after the last focused run. Direct children only, real absolute parent, current owner, no symlinks; exclude valid owner roots, product ak-sync-preview-npm roots, all paths found by one lsof snapshot, anything changed in 24 hours, and unattributed prefixes. A: before 2026-09-27 12:01 local runner landing; B: later attributable leaks; C: legacy ownerless suite roots. Keep all exclusion counts and script source, independently rederive counts, and submit literal paths for maintainer judgment. A missing lsof result due to error is incomplete evidence, not proof that nothing is in use. Idle orphans can evade lsof, so this remains a manual review list with no deletion authority. + +## Completed creator census + +The final census classifies **647 creator sites in 279 files, with zero unclassified current lifecycle sites**: 645 AST calls plus two executable child-template sites. It adds seven `makeTempDir` aliases, five sandboxConfigBase calls and three inline allocator definitions, and removes two comment-only lexical hits. Local factory invocations are represented by their allocator definition and caller contract, rather than counted as additional allocations. The retained AST/parser and manual override scripts reproduce the census without executing tests or tools. + +This is a source lifecycle classification, not a guarantee that cleanup runs after SIGKILL or that recursive removal succeeds. Each returned fixture is traced to caller cleanup; mixed callers stay mixed. Assertions before hook registration were checked in the allocating scope, excluding callbacks that run later. `sandboxHome` and `sandboxProject` register **no** cleanup themselves; they must not inherit tempDir's safe-return contract. Allocation, setup and multiple cleanup operations can still throw before protection or skip a later cleanup. + +| Code | Sites | Current lifecycle | +|---|---:|---| +| H | 130 | Shared helper registers cleanup before return; see temp-dir:19 or home-sandbox:176. | +| R | 168 | Local test/file hook registration; cleanup line shown. No preceding direct assert/assertSandboxed call in the allocating scope was found. | +| M | 87 | Module allocation with file hook; early imports/assertions can precede registration. | +| F | 113 | Removal in finally; setup before entering try remains exposed. | +| S | 67 | Direct cleanup reached only on normal execution, often after assertions. | +| CF | 20 | Factory returns root; callers remove in finally; pre-return setup remains exposed. | +| CH | 1 | Factory returns root; caller registers a hook; pre-registration setup remains exposed. | +| CM | 5 | Factory callers have mixed success-only/finally/hook lifecycles; exact examples in JSON. | +| O | 2 | Helper transfers ownership without registering cleanup; callers classified separately. | +| CS | 15 | Factory returns root; callers remove only on normal execution. | +| E | 6 | Process exit handler; normal exit only. | +| XS | 1 | Executable child template uses success-path cleanup. | +| XL | 1 | Executable child template deliberately leaks to exercise runner detection. | +| PE | 10 | Allocation covered transitively by enclosing private temp exit handler. | +| P | 16 | Allocation covered transitively by parent folder cleanup hook. | +| G | 2 | Root added to collection consumed by already registered/file cleanup hook. | +| C | 3 | Returns a cleanup method; construction failure handling and caller obligations described in JSON. | + +Source index below uses `allocation line:code→cleanup/caller evidence line` within the named file. H and E also use the shared helper references in the legend. Full cleanup expressions, all evidence locations, caller notes and scope boundaries are in ignored `creator-lifecycle-final.json`; reproducible scripts are `ast-census.mjs`, `lifecycle.mjs`, `classify.py` and `render-census.py`. No site is deemed safe simply because its file contains an unrelated hook. + +| Source file | Every creator site and lifecycle evidence | +|---|---| +| `tests/dashboard.test.cjs` | 19:E→helper; 70:PE→19 | +| `tests/kit/about-install-edits.test.mjs` | 11:M→12 | +| `tests/kit/about-security.test.mjs` | 10:M→11 | +| `tests/kit/adapter-admission.test.mjs` | 253:H→helper; 274:H→helper; 303:H→helper; 506:F→540 | +| `tests/kit/adapter-aqe-provider.test.mjs` | 89:CM→115,206,252; 247:H→helper; 350:H→helper; 498:F→514 | +| `tests/kit/adapter-conformance.test.mjs` | 127:S→240; 333:F→345; 356:F→369 | +| `tests/kit/adapter-execution.test.mjs` | 481:F→499 | +| `tests/kit/adapter-grants.test.mjs` | 16:H→helper; 169:H→helper | +| `tests/kit/adapter-hook-runner.test.mjs` | 197:F→208; 304:CS→305,313,320 | +| `tests/kit/adapter-integrity.test.mjs` | 48:F→61; 66:F→76; 81:F→96; 109:F→129; 134:F→152; 157:F→173; 158:F→174; 179:F→187 | +| `tests/kit/adapter-registries.test.mjs` | 74:H→helper | +| `tests/kit/adapter-sources.test.mjs` | 150:F→159; 171:F→182; 187:F→194; 199:F→208; 349:F→361; 367:F→376; 382:F→389; 395:F→402; 439:F→453; 458:F→470; 475:F→487; 492:F→505; 512:F→528 | +| `tests/kit/agent-browser-runtime.test.mjs` | 23:CS→23,88,126 | +| `tests/kit/agentdb-retirement.test.mjs` | 26:M→27; 40:M→41; 218:R→225 | +| `tests/kit/ak-launcher-evidence.test.mjs` | 17:H→helper | +| `tests/kit/aqe-embedding-probe.test.mjs` | 10:R→11 | +| `tests/kit/aqe-embedding-projection.test.mjs` | 14:R→15 | +| `tests/kit/aqe-embedding-transport.test.mjs` | 93:R→94; 109:R→110 | +| `tests/kit/aqe-guidance.test.mjs` | 38:R→39; 50:R→51; 71:H→helper | +| `tests/kit/aqe-lifecycle-migration.test.mjs` | 18:R→19 | +| `tests/kit/aqe-live-proof-key-guard.test.mjs` | 15:H→helper | +| `tests/kit/aqe-project-pin.test.mjs` | 24:H→helper; 283:H→helper; 349:H→helper | +| `tests/kit/aqe-readiness.test.mjs` | 11:F→17 | +| `tests/kit/aqe-store-holders.test.mjs` | 28:H→helper; 39:H→helper; 47:H→helper; 57:H→helper; 78:H→helper; 93:H→helper; 131:H→helper; 149:H→helper; 157:H→helper; 167:H→helper; 183:H→helper; 196:H→helper | +| `tests/kit/aqe-store-merge-fixture.test.mjs` | 30:H→helper; 49:H→helper; 71:H→helper | +| `tests/kit/aqe-store-merge-preview.test.mjs` | 44:H→helper | +| `tests/kit/aqe-store-merge.test.mjs` | 174:H→helper | +| `tests/kit/blocks-drift-parity.test.mjs` | 19:M→20; 30:M→31,36 | +| `tests/kit/blocks-dual-mode.test.mjs` | 33:S→53; 58:S→74; 79:S→96 | +| `tests/kit/blocks.test.mjs` | 99:S→106; 110:S→135; 139:S→146; 178:S→185; 189:S→199 | +| `tests/kit/brain-held-refresh-sync.test.mjs` | 21:M→36; 31:M→36 | +| `tests/kit/brain-held-refresh.test.mjs` | 13:M→14 | +| `tests/kit/claude-env-projection.test.mjs` | 13:R→15; 14:R→15 | +| `tests/kit/claude-window-ledger.test.mjs` | 14:H→helper | +| `tests/kit/clean-machine-setup.test.mjs` | 14:S→46; 50:S→76 | +| `tests/kit/cli-help.test.mjs` | 13:M→14 | +| `tests/kit/cli-json-honesty.test.mjs` | 16:M→18; 17:M→18 | +| `tests/kit/codex-context-command.test.mjs` | 7:M→67 | +| `tests/kit/codex-context.test.mjs` | 11:R→12 | +| `tests/kit/codex-mcp-convergence.test.mjs` | 11:M→12; 23:M→24 | +| `tests/kit/codex-mcp.test.mjs` | 14:CF→18,38,46; 94:F→126; 131:F→145; 150:F→164; 169:F→178; 183:F→196; 201:F→219; 224:F→256; 261:F→297; 302:F→327; 332:F→354; 360:F→398 | +| `tests/kit/codex-plugins.test.mjs` | 10:M→269; 214:F→231 | +| `tests/kit/codex-state.test.mjs` | 13:H→helper | +| `tests/kit/codex-statusline.test.mjs` | 13:G→14,103 | +| `tests/kit/codex-usage-diagnostic.test.mjs` | 13:R→14 | +| `tests/kit/conformance-tiers.test.mjs` | 24:H→helper; 260:H→helper; 298:H→helper; 317:H→helper; 355:H→helper; 389:H→helper; 673:H→helper | +| `tests/kit/context-audit.test.mjs` | 151:F→178 | +| `tests/kit/daemon-sweep-evidence.test.mjs` | 33:H→helper; 49:F→54 | +| `tests/kit/daemons-status.test.mjs` | 11:R→12 | +| `tests/kit/dashboard-context-hooks.test.mjs` | 203:F→262; 267:F→304 | +| `tests/kit/dashboard-hermetic-defaults.test.mjs` | 15:M→16; 56:P→14,16,56 | +| `tests/kit/dashboard-intel-integration.test.mjs` | 129:H→helper | +| `tests/kit/dashboard-project-identity.test.mjs` | 16:R→17 | +| `tests/kit/dashboard-status-cost.test.mjs` | 27:F→75; 29:F→76 | +| `tests/kit/dashboard-status-inprocess.test.mjs` | 36:CF→48,101,104; 38:CF→48,101,104 | +| `tests/kit/deja-vu-lifecycle.test.mjs` | 17:H→helper | +| `tests/kit/deja-vu-teardown-verify.test.mjs` | 15:M→503 | +| `tests/kit/deja-vu.test.mjs` | 160:F→197; 202:F→237; 242:F→277; 282:F→339; 344:F→359 | +| `tests/kit/dispatch-surface.test.mjs` | 79:H→helper | +| `tests/kit/disposable-memory-project.test.mjs` | 40:R→41; 130:R→131 | +| `tests/kit/drift-freshness.test.mjs` | 16:M→17; 26:M→27 | +| `tests/kit/dry-run-nudge.test.mjs` | 26:CF→43,64,65; 28:CF→43,64,65; 37:CF→43,64,65 | +| `tests/kit/evidence.test.mjs` | 9:M→231 | +| `tests/kit/exec-kill-tree.test.mjs` | 23:R→24; 81:R→82 | +| `tests/kit/exec.test.mjs` | 93:F→111; 147:F→185; 190:F→215; 222:F→234 | +| `tests/kit/execution-runner.test.mjs` | 470:H→helper | +| `tests/kit/external-lifecycle.test.mjs` | 31:M→445; 172:F→188; 193:F→209; 214:F→230; 235:F→251; 258:F→271; 276:F→288; 293:F→306; 313:F→328; 335:F→356; 369:F→440; 370:F→441 | +| `tests/kit/file-identity-bigint.test.mjs` | 151:H→helper | +| `tests/kit/footprint-collectors.test.mjs` | 50:R→51 | +| `tests/kit/footprint-executable-paths.test.mjs` | 9:R→10 | +| `tests/kit/footprint-known-files.test.mjs` | 12:F→19 | +| `tests/kit/footprint-observation-forest.test.mjs` | 11:R→12 | +| `tests/kit/footprint-performance.test.mjs` | 20:R→21 | +| `tests/kit/footprint-projects.test.mjs` | 44:R→45 | +| `tests/kit/footprint-snapshot-v2.test.mjs` | 12:R→13 | +| `tests/kit/footprint-stack.test.mjs` | 47:R→48 | +| `tests/kit/guidance-targets.test.mjs` | 34:S→40; 74:S→82; 88:S→95; 100:S→109; 101:S→110; 239:S→259; 265:S→279; 283:S→296 | +| `tests/kit/heal-natives.test.mjs` | 30:H→helper; 34:H→helper; 52:H→helper; 114:H→helper; 139:H→helper; 150:H→helper; 166:H→helper; 214:H→helper; 254:H→helper; 277:H→helper; 297:H→helper | +| `tests/kit/helper-stamp.test.mjs` | 85:H→helper | +| `tests/kit/helpers/aqe-store-merge-fixture.mjs` | 14:H→helper | +| `tests/kit/helpers/aqe-store-merge-harness.mjs` | 119:H→helper | +| `tests/kit/helpers/codex-rollout.mjs` | 130:H→helper | +| `tests/kit/helpers/home-sandbox.mjs` | 110:C→124,125,129; 140:O→139,158,303; 169:R→181; 304:O→139,158,303 | +| `tests/kit/helpers/private-tmpdir.cjs` | 12:E→16 | +| `tests/kit/helpers/project-isolation.mjs` | 117:R→122 | +| `tests/kit/helpers/temp-dir.mjs` | 17:H→18,19,20 | +| `tests/kit/home-sandbox-tripwire.test.mjs` | 19:R→20; 27:XS→49,50,63 | +| `tests/kit/hook-audit-hosts.test.mjs` | 21:CF→138,139,164 | +| `tests/kit/hook-audit.test.mjs` | 20:CF→61,62,90 | +| `tests/kit/hook-auto-memory-retirement.test.mjs` | 123:CF→176,177,192 | +| `tests/kit/hook-legacy-retirement.test.mjs` | 84:CF→125,126,157 | +| `tests/kit/hook-remediation-cli.test.mjs` | 27:F→83; 88:F→141 | +| `tests/kit/hook-remediation.test.mjs` | 30:CF→68,69,98; 104:F→164; 169:F→204; 240:F→274; 279:F→305; 432:F→456; 481:F→506; 531:F→537; 542:F→553 | +| `tests/kit/hook-upstream.test.mjs` | 17:F→25 | +| `tests/kit/host-adapters-cli.test.mjs` | 77:H→helper; 754:H→helper | +| `tests/kit/host-alignment.test.mjs` | 7:M→8; 9:M→10; 11:M→12 | +| `tests/kit/host-cli-migration.test.mjs` | 13:H→helper; 14:H→helper | +| `tests/kit/host-dry-run.test.mjs` | 38:R→39; 41:R→42 | +| `tests/kit/host-executable.test.mjs` | 9:H→helper | +| `tests/kit/host-health-connected.test.mjs` | 202:R→203 | +| `tests/kit/host-health-evidence.test.mjs` | 9:R→10; 29:R→30; 39:R→40 | +| `tests/kit/host-readiness-local.test.mjs` | 9:R→12 | +| `tests/kit/host-setup-evidence.test.mjs` | 28:H→helper; 44:F→52 | +| `tests/kit/hosts.test.mjs` | 152:H→helper | +| `tests/kit/install-edits.test.mjs` | 21:R→22 | +| `tests/kit/integration-command-facts.test.mjs` | 9:M→10 | +| `tests/kit/intel-history.test.mjs` | 21:H→helper | +| `tests/kit/intelligence-picker-groups.test.mjs` | 51:R→52; 67:R→68 | +| `tests/kit/intelligence-table-groups.test.mjs` | 10:R→11 | +| `tests/kit/intelligence-watch.test.mjs` | 8:H→helper | +| `tests/kit/language-coverage.test.mjs` | 10:R→10 | +| `tests/kit/live-check-evidence.test.mjs` | 15:M→260; 25:M→260 | +| `tests/kit/live-checks.test.mjs` | 21:M→688; 30:M→688; 401:F→415; 421:R→422; 440:F→449; 468:R→469; 701:R→702; 733:R→734 | +| `tests/kit/live-core.test.mjs` | 149:H→helper; 178:H→helper; 202:H→helper; 226:H→helper; 248:H→helper; 263:H→helper; 274:H→helper; 275:H→helper; 288:H→helper; 299:H→helper; 309:H→helper | +| `tests/kit/live-folder-correlator.test.mjs` | 14:R→15 | +| `tests/kit/live-process-sessions.test.mjs` | 250:F→293; 299:F→314 | +| `tests/kit/live-qe-contract.test.mjs` | 40:H→helper | +| `tests/kit/live-service.test.mjs` | 9:H→helper | +| `tests/kit/live-tailer.test.mjs` | 9:H→helper; 150:H→helper | +| `tests/kit/live-transcript.test.mjs` | 15:H→helper | +| `tests/kit/maintenance-action-service.test.mjs` | 25:R→26 | +| `tests/kit/maintenance-cli.test.mjs` | 15:H→helper | +| `tests/kit/maintenance-dashboard-api.test.mjs` | 600:R→601 | +| `tests/kit/maintenance-dashboard-e2e.test.mjs` | 177:R→179 | +| `tests/kit/maintenance-discovery-checkpoint.test.mjs` | 14:R→15 | +| `tests/kit/maintenance-discovery-configuration.test.mjs` | 21:R→22 | +| `tests/kit/maintenance-discovery-coverage.test.mjs` | 15:R→16 | +| `tests/kit/maintenance-discovery-orchestrator.test.mjs` | 23:R→24; 649:R→650 | +| `tests/kit/maintenance-discovery-preview.test.mjs` | 17:R→18 | +| `tests/kit/maintenance-git-project-patch.test.mjs` | 26:R→27; 41:R→42; 178:R→179 | +| `tests/kit/maintenance-host-alignment.test.mjs` | 6:M→7; 8:M→9 | +| `tests/kit/maintenance-interruption-audit.test.mjs` | 21:R→22 | +| `tests/kit/maintenance-management-activity.test.mjs` | 15:H→helper; 149:H→helper | +| `tests/kit/maintenance-management-procedures.test.mjs` | 21:R→22 | +| `tests/kit/maintenance-management-service.test.mjs` | 42:R→43 | +| `tests/kit/maintenance-native-findings.test.mjs` | 24:R→25 | +| `tests/kit/maintenance-one-action.test.mjs` | 20:R→21 | +| `tests/kit/maintenance-owned-providers.test.mjs` | 25:R→26 | +| `tests/kit/maintenance-persistence-support.test.mjs` | 22:R→23 | +| `tests/kit/maintenance-project-kind.test.mjs` | 13:R→14 | +| `tests/kit/maintenance-read-model.test.mjs` | 87:R→88; 104:R→105; 144:R→145; 187:R→188; 205:R→206; 236:R→237; 251:R→252; 269:R→270 | +| `tests/kit/maintenance-recovery.test.mjs` | 20:R→21 | +| `tests/kit/maintenance-transaction.test.mjs` | 19:R→20 | +| `tests/kit/mcp-scopes.test.mjs` | 21:R→26 | +| `tests/kit/mcp-tool-call.test.mjs` | 64:R→65 | +| `tests/kit/memory-maintenance.test.mjs` | 17:R→18 | +| `tests/kit/memory-probe-cleanup.test.mjs` | 16:M→17,186; 59:R→60 | +| `tests/kit/model-dashboard-read-model.test.mjs` | 553:F→604; 609:F→645 | +| `tests/kit/model-inventory-store.test.mjs` | 27:CS→38,60,73 | +| `tests/kit/natives-probe.test.mjs` | 16:CF→31,40,52 | +| `tests/kit/natives-runtime.test.mjs` | 23:H→helper; 26:CM→39,47,57; 74:S→82 | +| `tests/kit/natives.test.mjs` | 13:CS→29,41,47; 33:S→35; 53:S→60; 68:S→79; 88:S→95; 99:S→106; 110:S→113; 117:CF→128,130,143 | +| `tests/kit/node-runtime.test.mjs` | 23:H→helper | +| `tests/kit/npx.test.mjs` | 33:CS→40,47,54; 109:S→116 | +| `tests/kit/nudge.test.mjs` | 15:H→helper | +| `tests/kit/opencode-agents-stale-reason.test.mjs` | 25:M→39; 35:M→39 | +| `tests/kit/opencode-aqe-embedding.test.mjs` | 9:R→10 | +| `tests/kit/opencode-ruflo-gateway.test.mjs` | 8:CF→9,237,238 | +| `tests/kit/opencode-state-hermeticity.test.mjs` | 27:F→32; 43:R→44 | +| `tests/kit/opencode-stock-ruflo-gateway.test.mjs` | 33:CH→280,289,297 | +| `tests/kit/opencode-version-drift.test.mjs` | 14:M→56 | +| `tests/kit/opencode.test.mjs` | 20:M→21; 23:P→20,21,23 | +| `tests/kit/output-progress.test.mjs` | 119:H→helper | +| `tests/kit/owned-env-backup-prune.test.mjs` | 13:M→14 | +| `tests/kit/owned-env-projection.test.mjs` | 11:R→12; 131:R→132 | +| `tests/kit/paths-global-root-evidence.test.mjs` | 37:H→helper; 156:H→helper; 159:F→198 | +| `tests/kit/paths-global-root.test.mjs` | 94:F→106; 111:F→117 | +| `tests/kit/project-census.test.mjs` | 27:H→helper | +| `tests/kit/project-guidance.test.mjs` | 20:R→21 | +| `tests/kit/project-isolation.test.mjs` | 27:R→28 | +| `tests/kit/project-memory-status.test.mjs` | 11:R→12; 37:R→38; 83:R→92; 118:R→119; 139:R→140; 156:R→157; 170:R→171; 195:R→196; 211:R→212; 235:R→236; 328:R→329 | +| `tests/kit/project-memory.test.mjs` | 11:M→51,65,79; 82:R→83; 98:R→99; 107:R→108; 120:R→121; 133:R→134; 143:R→144; 160:R→168; 190:R→191; 201:R→202; 222:R→223; 269:R→270; 283:R→284 | +| `tests/kit/project-sources-imports.test.mjs` | 21:R→22 | +| `tests/kit/prompts-mainline-boundary.test.mjs` | 14:F→22 | +| `tests/kit/provider-cli.test.mjs` | 51:CS→86,103,109; 62:CS→86,103,109; 193:CF→297,335,336; 215:CF→297,335,336 | +| `tests/kit/provider-credentials.test.mjs` | 24:M→25; 33:P→24,25,33; 40:P→24,25,42 (local S→42) | +| `tests/kit/provider-refresh-cli.test.mjs` | 28:CS→48,84,89; 43:CS→48,84,89 | +| `tests/kit/provider-teardown-preservation.test.mjs` | 9:R→11 | +| `tests/kit/providers-drift-parity.test.mjs` | 22:M→113; 62:R→63; 96:R→97 | +| `tests/kit/providers-external.test.mjs` | 22:G→16,18,23 | +| `tests/kit/providers.test.mjs` | 170:CS→175,193,210; 178:S→180; 371:S→378; 437:F→466; 475:S→483 | +| `tests/kit/qeCourt.test.mjs` | 214:S→216; 220:S→226; 230:F→253; 257:F→279; 283:R→284 | +| `tests/kit/quota-codex-presence.test.mjs` | 18:M→25 | +| `tests/kit/quota.test.mjs` | 17:H→helper | +| `tests/kit/real-state-tripwire.test.mjs` | 16:R→17 | +| `tests/kit/reference-command.test.mjs` | 11:M→64; 19:F→60 | +| `tests/kit/refresh.test.mjs` | 17:M→19; 18:M→19 | +| `tests/kit/reverse-bridge.test.mjs` | 10:H→helper; 67:H→helper | +| `tests/kit/routing-config.test.mjs` | 147:S→176; 180:S→189; 193:S→211 | +| `tests/kit/routing-projection.test.mjs` | 13:CS→17,49,50; 21:CS→17,49,50 | +| `tests/kit/routing-retirement-convergence.test.mjs` | 29:M→30; 109:R→110,124; 128:R→129,154; 165:R→166,185 | +| `tests/kit/ruflo-components-apply.test.mjs` | 132:R→133; 164:R→165; 189:R→190; 211:R→212; 234:R→235; 245:R→246; 262:R→263; 276:R→277 | +| `tests/kit/ruflo-components-catalogue.test.mjs` | 81:R→82 | +| `tests/kit/ruflo-components-convergence.test.mjs` | 12:M→13; 42:R→43; 197:R→198; 213:R→214 | +| `tests/kit/ruflo-components-env.test.mjs` | 15:R→16 | +| `tests/kit/ruflo-components-evidence-location.test.mjs` | 6:M→18 | +| `tests/kit/ruflo-components-evidence.test.mjs` | 67:R→68; 188:R→189 | +| `tests/kit/ruflo-components-git-exclude.test.mjs` | 19:H→helper; 33:R→34; 100:R→101 | +| `tests/kit/ruflo-components-hosts.test.mjs` | 19:R→20; 45:R→46; 142:R→143; 161:F→170 | +| `tests/kit/ruflo-components-snapshot.test.mjs` | 194:F→210 | +| `tests/kit/ruflo-daemon-config.test.mjs` | 19:R→20 | +| `tests/kit/ruflo-mcp-launcher.test.mjs` | 21:R→22; 83:R→84 | +| `tests/kit/ruflo-memory-location.test.mjs` | 19:R→20; 50:R→51 | +| `tests/kit/ruflo-memory-root-pin.test.mjs` | 24:R→25 | +| `tests/kit/ruflo-memory.test.mjs` | 11:F→31; 39:R→40 | +| `tests/kit/run-tests-runner.test.mjs` | 15:R→16; 108:XL→16,110 | +| `tests/kit/ruvector.test.mjs` | 18:M→214; 25:M→214 | +| `tests/kit/ruvnet-brain-plugin.test.mjs` | 10:R→11 | +| `tests/kit/ruvnet-brain.test.mjs` | 87:H→helper; 118:H→helper; 181:H→helper; 196:H→helper; 213:H→helper; 222:H→helper; 238:H→helper; 257:H→helper | +| `tests/kit/rvf.test.mjs` | 22:CS→49,56,64 | +| `tests/kit/scaffold.test.mjs` | 19:CF→34,42,52; 25:CF→34,42,52; 85:F→91; 143:F→157 | +| `tests/kit/security-status.test.mjs` | 12:M→13 | +| `tests/kit/settings-config.test.mjs` | 12:S→19; 29:S→34; 38:S→45; 49:S→57; 61:S→67; 71:S→77; 81:S→99; 103:S→119; 130:S→141; 146:S→153; 178:S→188; 192:S→197; 201:S→209; 213:S→224; 228:S→234 | +| `tests/kit/setup-command.test.mjs` | 19:M→104,775; 122:S→136; 191:S→197; 208:S→220; 225:S→239; 474:F→484; 489:F→506; 512:F→541; 546:F→588; 593:S→601; 606:S→622; 607:P→775; 627:P→775; 783:S→798; 803:S→811 | +| `tests/kit/setup-host-flags.test.mjs` | 103:S→107 | +| `tests/kit/setup-memory-probe.test.mjs` | 26:R→27; 83:R→84 | +| `tests/kit/spawn-env-guard.test.mjs` | 176:R→177 | +| `tests/kit/sqlite.test.mjs` | 18:S→27; 31:S→42; 46:S→60 | +| `tests/kit/status-agent-browser.test.mjs` | 13:R→17 | +| `tests/kit/status-aqe-drift.test.mjs` | 17:M→18; 58:R→59; 71:R→72; 83:R→84; 103:H→helper; 127:R→128; 145:R→146; 160:R→161; 171:R→172 | +| `tests/kit/status-command.test.mjs` | 19:M→1451; 29:M→386,536,671; 340:F→387 | +| `tests/kit/status-golden.test.mjs` | 21:M→22; 29:M→30 | +| `tests/kit/status-live.test.mjs` | 17:M→267; 31:M→144,267; 125:F→144 | +| `tests/kit/status-manual-fixes.test.mjs` | 17:M→28; 27:M→28 | +| `tests/kit/status-repair-contract.test.mjs` | 17:M→18; 30:M→31,146 | +| `tests/kit/status-setup-hints.test.mjs` | 14:M→38,62; 22:M→62; 23:P→14,23,62 | +| `tests/kit/status-version-drift-refresh.test.mjs` | 27:M→40; 37:M→40; 47:P→40,47 | +| `tests/kit/status-viability.test.mjs` | 24:M→347; 35:M→347 | +| `tests/kit/status-zero-spawn.test.mjs` | 41:F→54; 43:F→55 | +| `tests/kit/statusline-config-dir-parity.test.mjs` | 42:H→helper | +| `tests/kit/statusline-version.test.mjs` | 20:M→212; 62:P→62,212 | +| `tests/kit/statusline.test.mjs` | 22:M→23; 44:H→helper | +| `tests/kit/sync-command.test.mjs` | 20:M→35,1320; 31:M→1320 | +| `tests/kit/sync-daemon-repair.test.mjs` | 16:M→17; 27:CF→46,61,76 | +| `tests/kit/sync-dry-run-preview.test.mjs` | 20:M→42; 33:M→42; 183:P→40,42,339; 184:P→40,42,339; 254:P→40,42,339; 325:P→40,42,339; 339:P→40,42,339 | +| `tests/kit/sync-host-repair.test.mjs` | 5:M→6 | +| `tests/kit/sync-needs-your-action.test.mjs` | 25:M→26; 35:M→36 | +| `tests/kit/sync-self-freshness.test.mjs` | 10:M→11 | +| `tests/kit/sync-skip-versions.test.mjs` | 22:M→43; 34:M→43; 107:P→41,43,107 | +| `tests/kit/system-command.test.mjs` | 21:H→helper; 22:H→helper | +| `tests/kit/system-summary.test.mjs` | 644:H→helper; 666:H→helper | +| `tests/kit/telemetry-cli.test.mjs` | 15:R→16; 19:H→helper | +| `tests/kit/telemetry-source-bounds.test.mjs` | 19:R→20 | +| `tests/kit/temp-dir-helper.test.mjs` | 10:H→helper | +| `tests/kit/uninstall-command.test.mjs` | 18:M→493; 170:S→187; 192:S→203; 449:CS→448,452,474; 458:S→474; 501:S→519; 507:S→519 | +| `tests/kit/upstream-watch-fixtures.mjs` | 31:F→48 | +| `tests/kit/upstream-watch-ledger-branch.test.mjs` | 128:F→167 | +| `tests/kit/upstream-watch-registry.test.mjs` | 30:F→38 | +| `tests/kit/upstream-watch-script.test.mjs` | 878:F→904 | +| `tests/kit/usage-audit-211.test.mjs` | 15:R→16 | +| `tests/kit/usage-claude-dedup.test.mjs` | 193:H→helper | +| `tests/kit/usage-cli.test.mjs` | 16:H→helper | +| `tests/kit/usage-codex-large-rollout.test.mjs` | 22:H→helper | +| `tests/kit/usage-deps-contract.test.mjs` | 42:H→helper | +| `tests/kit/usage-git-projects.test.mjs` | 37:R→38 | +| `tests/kit/usage-index-claude-window.test.mjs` | 31:CM→70,81,138 | +| `tests/kit/usage-index-opencode.test.mjs` | 15:CM→118,130,268 | +| `tests/kit/usage-index-v6.test.mjs` | 87:H→helper | +| `tests/kit/usage-index.test.mjs` | 33:H→helper; 66:H→helper; 929:H→helper; 970:H→helper | +| `tests/kit/usage-local-pricing.test.mjs` | 20:CF→126,133,173 | +| `tests/kit/usage-opencode.test.mjs` | 14:CM→95,137,325 | +| `tests/kit/usage-openrouter.test.mjs` | 13:CS→137,192,211 | +| `tests/kit/usage-project-groups.test.mjs` | 81:R→82 | +| `tests/kit/usage-truncation.test.mjs` | 43:H→helper | +| `tests/kit/verify-memory-routes.test.mjs` | 91:M→92; 96:M→97; 106:H→helper; 157:H→helper; 217:R→219; 218:P→217,218,219 | +| `tests/kit/version-lookup-record.test.mjs` | 22:M→41 | +| `tests/kit/versions.test.mjs` | 112:F→126 | +| `tests/kit/working-context.test.mjs` | 9:R→10 | +| `tests/live/aqe-codex-guidance-conformance.test.mjs` | 84:R→85 | +| `tests/live/aqe-external-provider-transport.test.mjs` | 258:R→280 | +| `tests/live/aqe-stop-hook-conformance.test.mjs` | 44:R→45 | +| `tests/live/codex-context-contract.test.mjs` | 16:R→17 | +| `tests/live/disposable-memory-project.mjs` | 49:C→41,43,57 | +| `tests/live/ruflo-memory-routing.test.mjs` | 26:M→28 | +| `tests/statusline-brain.test.cjs` | 18:E→helper; 41:PE→18 | +| `tests/statusline-segments.test.cjs` | 19:E→helper; 64:PE→19; 346:PE→19; 380:PE→19 | +| `tests/statusline-window-ledger.test.cjs` | 21:E→helper; 45:F→54; 176:PE→21; 186:PE→21 | +| `tests/ui/dashboard-ui.mjs` | 68:E→helper; 162:PE→68; 1102:PE→68; 1103:PE→68 | +| `tests/ui/helpers/launch-chrome.mjs` | 20:C→21,29,31 | + +## Decisions for Tasks 11–13 + +1. Implement one proveAbandoned interface with **abandoned=false, reason=cannot prove complete descendant exit** by default on macOS/Linux/Windows. Unit fixtures may exercise collector plumbing but cannot enable a production deletion path or constitute native platform proof. +2. Proposed owner fields: schema=1, random runId, absolute canonical root/temp parent, pid, startedAt milliseconds, hostname, platform, uid (null only when unavailable), proofMode=list-only. Validate schema, finite positive PID/timestamp, host/user identity, path and file type. Atomic record writing improves attribution only. +3. Retain B9-R4 path rules for own-root removal; refuse real home/filesystem-root temp bases before allocation, check absolute canonical direct parent, exact suite basename, lstat/non-symlink, POSIX owner. Unknown/error means refuse. List sibling roots without deleting them; do not change command/tripwire/leftover exit precedence (B9-R2). +4. Preserve synchronous runner and Ctrl-C/tool-kill behavior. Owner files are ignored in own leftovers. Interrupted runners remove nothing; completed runs retain current own-root semantics under B9-R1. Own-root cleanup is not a descendant-exit proof; the Windows lifetime regression must address unfinished children explicitly. +5. Task 12's list-only exit proof keeps a killed runner's root while its child lives **and after that child exits**. Do not execute the archived example's unconditional run-3 removal on a list-only platform. This follows B9-R5; controller was notified before dependent implementation. +6. B9-R1–R8 need no safety-rule relaxation. R7 has a factual clarification: global setup is unavailable in installed 22.22.3; config/import still require explicit opt-in. Task 13 remains report-only. Native Windows diagnostics and historical-cause limitations stay visible; no claim of exhaustive leak causes or automatic backlog cleanup is justified. diff --git a/docs/archive/2026-09-29-aqe-released-artifact-receipt.md b/docs/archive/2026-09-29-aqe-released-artifact-receipt.md new file mode 100644 index 00000000..e96ee014 --- /dev/null +++ b/docs/archive/2026-09-29-aqe-released-artifact-receipt.md @@ -0,0 +1,46 @@ +# AQE 3.14.5 released-artifact receipt (2026-09-29) + +Capture: 2026-09-29. Documentation basis: `b5a0a946ae0aca2e1d435c78b12a98bc587d1266`. Probe observations: 2026-09-29T14:27:38Z; compilation of retained evidence, without new runtime probes. The upstream issues [#655](https://github.com/proffesor-for-testing/agentic-qe/issues/655), [#753](https://github.com/proffesor-for-testing/agentic-qe/issues/753), and [#778](https://github.com/proffesor-for-testing/agentic-qe/issues/778) are closed, but closure does not establish installed behavior. This receipt reports only the selected tests below. It is not an issue-closing request. + +## Exact artifact and environment + +The controller registry capture at 2026-09-29 14:16:56 UTC listed `agentic-qe@3.14.5` as latest. The registry tarball had SHA-512 integrity `sha512-mDvWB1fUaGevg6F3QNWK7ESXHeZnlvc0MiQ4Ue9do4SJwggM6aP5dl3DZNWKmO2g5GGFjYJuikl53KeuT3uazg==` and SHA-256 `32cd29f00201cbcba19313b62bac20ffaa5c7d032b5246e5d08d94d6d24b9221`. Its installed selected source files matched tarball bytes: + +| Package-relative path | SHA-256 | +| --- | --- | +| `package.json` | `6a854da5e68636739a400b6237ad77b2250480471d5a4e643cc6474f297ee3a0` | +| `dist/cli/bundle.js` | `a8b61993b4286336f79f9d7db9195b26d86626924adbd522a158fad222b08a84` | +| `dist/init/codex-installer.js` | `79dda90da0faa7202ea61fc0c449d9ef4062d96f6062324bb98ecd6e9980d78a` | +| `dist/audit/witness-chain.js` | `a712fa97849efdfe160c3c2e440901d25e48450e2d6bdaa69af4225c824f3458` | + +The tarball was installed in a disposable private prefix with lifecycle scripts ignored; only `better-sqlite3` was rebuilt there. Both commands exited 0. Acquisition and native dependency preparation ran unsandboxed with a private constructed environment; the sandbox claim applies only to the later probes. The before/after global check covered package metadata, not a complete global installation fingerprint. Public `aqe` bin reported 3.14.5. Runtime: Node 26.4.0, `better-sqlite3` 12.11.1, SQLite 3.53.2, macOS arm64. Probe invocations ran with a credential-free, private environment, denied network, and sandboxed writes. The probe script relied on an earlier setup step that created the empty project and environment; the retained setup commands and logs support that initial state, but the probe script is not a self-contained fresh-root reproducer. Optional Vibium bootstrap attempted network access and failed as expected under the sandbox during the three init runs. No global install was made: global AQE metadata was 3.14.4 both before and after. The release's npm publication timestamp was 2026-09-29 10:08:33 UTC; [PR #783's merge commit](https://github.com/proffesor-for-testing/agentic-qe/commit/63d3debab54105dbc93030eb0b0088e2f9bebd9c) is dated 13:05:19 UTC, after that publication. The release metadata supplied no `gitHead`, so this ordering and the artifact probe bound the conclusion rather than an assumed source tag. + +## Selected results + +| Issue | Verdict on released 3.14.5 | Observed contract and limit | +| --- | --- | --- | +| [#655](https://github.com/proffesor-for-testing/agentic-qe/issues/655) | Verified selected compact path | `aqe init --auto --with-codex --codex-guidance compact` exited 0. The AQE-owned sentinel region was 315 UTF-8 bytes, below the released 512-byte constant. A separate foreign prefix/suffix fixture survived two identical calls with complete `AGENTS.md` bytes and mtime stable between calls 1 and 2. The fixture creation mtime was not captured. Full/none modes, CRLF and malformed/duplicate sentinels, all-platform verification, and receipt correctness remain unverified. Prior 3.14.4 broader conformance failed; this narrow result does not sunset the registry constraint. | +| [#753](https://github.com/proffesor-for-testing/agentic-qe/issues/753) | Verified fresh-chain append case | Released `createWitnessChain` places tail read and insert inside an immediate SQLite transaction. One native WAL sequential control and three synchronized two-process rounds each ended at 4001 rows (seed plus 2000 per process), `valid: true`, `signatureFailures: 0`, and SQLite integrity check `ok`. Only concurrent round 1 demonstrably interleaved the writers. Entries were synthesized unsigned `PATTERN_CREATE` records; signature validation, old-fork repair, import splices, live host concurrency, and Windows/Linux remain unverified. The kit's live-holder refusal still protects stray-store merge and must not be removed from this proof. | +| [#778](https://github.com/proffesor-for-testing/agentic-qe/issues/778) | Refuted fixed-in-3.14.5 claim | Three same-option public init calls on one unchanged minimal project each exited 0, yet `.claude/settings.json` hashes and mtimes changed on every call. Run 2 added a backup and changed domain/learning defaults; run 3 still changed `aqe.initialized`. `AGENTS.md` and `CLAUDE.md` bytes/mtimes were stable after run 1. The merged source fix is newer than the tested release. Subsequent-release conformance is unverified; keep installed/released convergence outstanding under own [#239](https://github.com/pacphi/agentic-kit/issues/239). | + +The source-bound report's private retained logs, snapshots, lockfile, scripts and synthesized databases have SHA-256 receipts. The digest of its artifact binding record is `200d11393cab0617bdd92b7218e5dcc5ab146e26e55ba7f58efee20d17471b1b`; selected result records are `init-results.json` `c03ee35435c5946bb18afa2bab96bbf5a29d2c3761bb102e80083d8e492789e4`, `foreign-results.json` `933883e6b9720ad01fb9b560919077247d65ab601750bf3fa4213a0c4ee3fb7b`, and `witness-results.json` `95b38b5c30380a53c1f7284fcaafb257a6b8c513381c9c824ad3a0748451c3f4`. This receipt intentionally omits private acquisition locations and raw project data. Time bounds were 30/60 s registry requests, 240 s dependency install, 180 s native rebuild, 150 s per init, 45 s per native child, and 20 s readiness; none expired in successful observed runs. The retained witness harness has a failure-path cleanup gap: if the first child fails, it can raise before killing and reaping its sibling. Do not reuse it without fixing that gap. + +There is no evidence here for broad Windows AQE support, repaired old forks, signed-entry verification, all guidance modes, or installed 3.14.4 conformance with the selected 3.14.5 behaviors. No upstream post, local store merge, or issue closure is authorized by this receipt. + +## Native holder and host-support boundary + +The [dated host-support addendum](../host-support.md) records released AQE 3.14.4 and +3.14.5 live-owner lock checks on macOS and Linux: status and the shipped adapter +reported `LockHeld`, without `FsyncFailed`; holder and storage bytes remained intact. +V4 [PR #275](https://github.com/pacphi/agentic-kit/pull/275), squash +`bb2e7efe88abd3cb2d73baf26530386549e3dc2e`, removed the exact temporary +`FsyncFailed`-as-busy exception. Ordinary `LockHeld` remains busy. Older 3.14.3 +error sequences now fail closed; no universal managed AQE minimum was introduced. + +Native Windows AQE was untested. Ruflo 3.48.0 native Windows CLI-to-MCP and +MCP-to-CLI visibility used one database with reported sql.js + HNSW and its native +bridge disabled. That separate proof establishes neither Windows AQE behavior nor +one native backend guarantee across platforms. The earlier Linux Ruflo result was +asymmetric. Own [#240](https://github.com/pacphi/agentic-kit/issues/240) remains open +on its literal managed-floor and combined live-MCP/`ak x verify aqe` criteria; +the native holder probe does not prove that combined final-artifact path. diff --git a/docs/archive/2026-09-29-dashboard-paused-time-evidence.md b/docs/archive/2026-09-29-dashboard-paused-time-evidence.md new file mode 100644 index 00000000..8682c06f --- /dev/null +++ b/docs/archive/2026-09-29-dashboard-paused-time-evidence.md @@ -0,0 +1,35 @@ +# V3 Activity paused-time consumer report + +## Source and contract + +- Base: `66982f22` in `feat/dashboard-refresh`. +- Read the V4 B8 producer in `agentic-kit-v4-rest` without changing it. A paused history row has `recordedAt` and `completedAt: null`; completed rows retain `completedAt`. + +## Change + +- Activity projection keeps a bounded, valid ISO `recordedAt` separately from `completedAt`, uses it to choose the latest state per source and environment, and falls back to a valid completion time. Invalid scan timestamps become `null`; unrelated metadata is not projected. +- The v2 Activity API allowlists a bounded ISO pause timestamp and omits invalid stamps and private metadata. +- The Activity history table sorts, groups, and displays by the valid recorded time or completion time. Its labels and empty copy describe scan records without claiming every scan completed. + +## Evidence + +- Red phase: guarded focused suites had 3 expected failures for missing pause time and stale latest-state selection. +- Green phase: `env -u FORCE_COLOR node scripts/run-tests.mjs exec -- --test tests/kit/maintenance-management-activity.test.mjs tests/kit/maintenance-dashboard-v2-api.test.mjs`: 58 passed, 0 failed. +- Guarded actual dashboard browser run, `env -u FORCE_COLOR node scripts/run-tests.mjs exec -- tests/ui/dashboard-ui.mjs`: 514 passed, 0 failed. The new assertion inspects actual rendered table rows for a newer paused record and older completed records. +- `./node_modules/.bin/tsc -p tsconfig.json --noEmit`: passed. +- Focused ESLint: 0 errors, one pre-existing `max-lines` warning in `maintenance-api.mjs` (file has 1021 lines, threshold 1000). +- `git diff --check`: passed. + +## Limits + +- V4 B8 producer remains on its separate branch. This change is the consumer contract only, pending integration and independent review. +- Full unit and UI suites were not repeated after the final timestamp-validation refinement; focused unit tests, lint, and typecheck passed after it. The guarded browser run covered the Activity renderer before that refinement. + +## Scoped review fix: impossible calendar dates + +- Review found that the prior ISO shape plus `Date.parse` accepted `2026-09-31T12:00:00Z`, letting a paused row outrank a real September 30 completion and render on October 1. +- The Activity projection now checks the calendar day against its month and leap year, plus clock component bounds, before accepting a scan timestamp. The v2 Activity API uses the same validator for `recordedAt`. +- New cases reject September 31 and a non-leap February 29 at both projection boundaries, retain a valid completion timestamp when the recorded time is rejected, and retain a valid leap day with an offset and fractional seconds. +- Guarded focused tests: `env -u FORCE_COLOR node scripts/run-tests.mjs exec -- --test tests/kit/maintenance-management-activity.test.mjs tests/kit/maintenance-dashboard-v2-api.test.mjs` passed 61 tests after the validator change. The final fallback assertions were added afterward and rerun before this fix commit. +- `./node_modules/.bin/tsc -p tsconfig.json --noEmit` passed. Focused ESLint had 0 errors and the existing file-length warning in `maintenance-api.mjs`. +- Browser and full suites were not repeated for this scoped validation fix; the prior guarded browser run remains the renderer evidence, with full gates assigned to integration. diff --git a/docs/archive/2026-09-29-native-learning-trace.md b/docs/archive/2026-09-29-native-learning-trace.md new file mode 100644 index 00000000..00aef9c2 --- /dev/null +++ b/docs/archive/2026-09-29-native-learning-trace.md @@ -0,0 +1,32 @@ +# Hosted macOS learning resolution trace + +## Status and inputs + +Captured 2026-09-29. This is an observation, not a native-readiness or causal verdict. + +- [Workflow run](https://github.com/pacphi/agentic-kit/actions/runs/36567908852), [macOS job](https://github.com/pacphi/agentic-kit/actions/runs/36567908852/job/109404326296), [artifact](https://github.com/pacphi/agentic-kit/actions/runs/36567908852/artifacts/11031869828). +- Source `ed8f4cb96836e1911aae9a89bbb3f015f033e115`; hook SHA-256 `98d9cc2c54e570eb18e24af83a7c401779c6eec12d58fe740e8c1a7622276dae`. +- macOS arm64, Node 22.23.2; Ruflo 3.48.0 and Agentic QE 3.14.5 installed in a disposable CI prefix. +- The kit's `sync --no-upgrade` healing step ran before `node bin/agentic-kit.mjs status --refresh=live --only learning`. This is a post-heal observation, not a pristine npm-tree measurement. +- The hook follows [vidaunited's resolution-hook proposal](https://github.com/ruvnet/ruflo/issues/2885#issuecomment-5867331087), with metadata-only JSONL and explicit artifact paths. + +## Observations + +The artifact contains 22 records: 16 preload starts and six package-resolution records across three processes. The two distinct package roots, with the runner-specific prefix omitted, are: + +```text +@huggingface/transformers@3.8.1 + npm-prefix/lib/node_modules/ruflo/node_modules/@claude-flow/cli/node_modules/@huggingface/transformers +onnxruntime-node@1.21.0 + npm-prefix/lib/node_modules/ruflo/node_modules/@claude-flow/cli/node_modules/@huggingface/transformers/node_modules/onnxruntime-node +``` + +The learning step invokes `ruflo neural train -p coordination -e 50`. Its output contains the default `fp32` dtype warning and `libc++abi` / `mutex lock failed: Invalid argument`. The outer kit command exited 1. The trace and receipt were retained despite that failure; the existing `continue-on-error` boundary explains the green macOS job. + +The separate clean-setup job failed at an AQE embedding process probe, and external link checking failed. Those results are not evidence that the trace worked or that setup is healthy. The trace was enabled only in the macOS learning step. + +## What this establishes + +The old package roots were resolved during the failing traced step. It remains unproved which installation/healing/dependency action introduced them and whether they caused the mutex failure. Resolution observations are not an exhaustive native-module census. A start-only artifact would establish only that preload reached a registration attempt. `learningExitCode` records the outer kit command, not a separately captured native Ruflo exit. + +Synthetic ESM/CommonJS, nested-copy, child-inheritance, unknown-version and failed-log-target tests passed. A later fixture-only correction uses file URLs for Windows preload paths and adds a real space-path test; it does not change the hook bytes captured here. The exact sanitized upstream comment was awaiting maintainer approval at this capture. No upstream message or user-global installation is implied by this record. diff --git a/docs/archive/2026-09-29-plan-main-watch-reconciliation.md b/docs/archive/2026-09-29-plan-main-watch-reconciliation.md new file mode 100644 index 00000000..5aaea1fa --- /dev/null +++ b/docs/archive/2026-09-29-plan-main-watch-reconciliation.md @@ -0,0 +1,47 @@ +# Main watcher reconciliation + +## Status + +Implemented and independently reviewed through `7c2f3054`. The controller integrated +validated V6 develop at `3bbee599`; the reviewed watcher files were unchanged by +that merge. All eight local gates passed on that exact clean commit: 6,440 unit +tests passed, seven skipped, legacy suites passed, and browser checks passed. +Coverage was 94.18% lines, 83.80% branches and 93.50% functions. Feature PR CI, +squash integration and final human main review remain separate gates at archival. +No real routine trigger was performed. The limits below describe the worker scope. + +## Scope and acceptance + +Combine develop PR observation and blind reporting with main #280 dispatch pacing. +Preserve pending eligibility, the strict seven day observation window, two recorded +firings per thread, three day cooldown, three fixes per run, and 15 second spacing. +Then restrict trigger retries to documented HTTP 500/503, sanitize token-bearing +error metadata, preserve deferred work in bounded previews and blind results, and +align the living guide and workflow summaries. Use focused synthetic regression +tests before behavior changes and scoped static checks. + +## Ownership and policy receipt + +The controller assigned sole writing ownership of watcher source, workflow, tests, +and guide in `fix/main-watch-reconciliation`, and authorized the prepared two-parent +merge commit followed by three conventional unit commits. Shared manifests, +lockfiles, V6 fixtures, develop/main integration, archive index changes, publication, +and final full gates remain controller-owned. This plan is ready for the controller +to archive with its index update in the completing pull request. + +## Worker limits + +No real trigger, provider turn, external mutation, full suite, UI tests, pnpm, +installed CLI changes, user data writes, or delegated workers. Inject fetch, +execution, and sleeps. Retry evidence does not establish exactly-once execution or +absence of server-side sessions. Deferred work remains eligible for later checks; +repeated earlier failures can starve it. + +## Units + +1. Reconcile and commit both parent contracts; establish focused baseline. +2. Test and fix bounded retry and sanitized error diagnostics. +3. Test and fix preview cap, deferred blind result, and later-run discoverability. +4. Clarify guide, workflow preview/record summaries, and evidence limits. + +Record command evidence and limitations in the ignored worker implementation report. diff --git a/docs/archive/2026-09-29-plan-remediation-v2-closeout.md b/docs/archive/2026-09-29-plan-remediation-v2-closeout.md new file mode 100644 index 00000000..6566d520 --- /dev/null +++ b/docs/archive/2026-09-29-plan-remediation-v2-closeout.md @@ -0,0 +1,41 @@ +# Remediation v2 closeout execution + +## Status and authority + +Active under the confirmed [develop execution plan](../plans/2026-09-28-remediation-v2-develop-execution.md). +Base: `develop@88ce597f`, incorporating reviewed V1–V6 and the main watcher reconciliation. +All 13 develop CI checks passed. Source scope remains the 189 Appendix A and 57 Appendix B +entries in [the program](../plans/2026-09-28-remediation-program-v2.md). New release, installation, +real-store merge, deletion, personal-memory writes and future #239 work remain outside this +implementation authority. The final develop → main PR stays open for human review. + +## Units and ownership + +1. **Comment-label guard:** a sole writer in this closeout worktree implements a parser-based + guard and removes transient planning labels from comments/test titles without changing + behavior. Scope: tracked JavaScript in src, scripts, bin and tests; no allowlist. Preserve + durable audit IDs and literal strings/regex/templates. Use RED/GREEN boundary fixtures. +2. **Scope and evidence documentation:** a separate task worktree prepares dated public + receipts for every original scope ID, source-bound Windows and AQE evidence, and all + recorded rulings. It preserves failed observations, limits and explicit deferrals. It + submits exact shared-index/program-status handoffs and private issue drafts to the controller. + No source/test edits or publication. This unit can run alongside the guard because their + paths are disjoint; final citation checks follow their integration. +3. **Controller integration:** review each unit, combine their accepted commits, apply shared + index/status handoffs, relocate the V3 paused-time report without losing its historical body, + and reconcile current issue conditions. Keep the parent program/execution plans active while + human approval and operational gates remain. Preserve all worktrees and branches. +4. **Final gates and delivery:** full guarded unit/legacy and browser suites, types, lint, + complexity, Markdown, build, offline links and independent branch review. Open the closeout + PR into develop, squash only after required CI/review, verify tree equality and develop CI, + then open the final main PR for the human. Recheck main before that final PR. + +## Evidence and limits + +The fixed Windows ten-before/ten-after cohort is historical, with all outcomes retained. +A separate three-consecutive-PR Windows timing rule remains pending: the current latest three +include a 358-second leg. Refresh that gate after closeout PR CI; do not manufacture green runs. +AQE #655/#753 proofs are narrow released-artifact observations; #778 settings churn still occurred +in the release tested before its source fix. No issue closure or installed-state inference follows +from upstream closed status alone. Exact current evidence and unresolved gates belong in the final +receipt and PR. No real routine trigger or paid provider run is required for this closeout. diff --git a/docs/archive/2026-09-29-plan-upstream-native-trace.md b/docs/archive/2026-09-29-plan-upstream-native-trace.md new file mode 100644 index 00000000..b2391e68 --- /dev/null +++ b/docs/archive/2026-09-29-plan-upstream-native-trace.md @@ -0,0 +1,19 @@ +# Upstream native trace plan + +## Status + +**Implemented; final integration pending** — Captured 2026-09-29. The hook and nightly workflow passed independent review and all eight local gates. Hosted macOS run 36567908852 captured package resolutions and the learning failure. The Windows preload fixture was corrected to use a file URL and independently reviewed. Final documentation/PR CI and the exact upstream comment approval remain separate gates. + +## Scope + +Add a passive Node resolution hook for the nightly macOS learning probe. The hook records each resolved `@huggingface/transformers`, `@xenova/transformers`, and `onnxruntime-node` package root once per process, including its on-disk version when readable. It records no command arguments, environment values, or prompt content. The nightly workflow edit and native CI run belong to the integration owner. + +## Steps + +1. Add focused child-process tests with synthetic packages for ESM and CommonJS resolution, nested copies, missing/malformed versions, and an unusable log path. +2. Implement `scripts/trace-ort.mjs` from vidaunited's hook in [ruflo issue 2885](https://github.com/ruvnet/ruflo/issues/2885#issuecomment-5867331087). Require `TRACE_ORT_LOG` to be an absolute, explicit target. Emit a hook-start receipt and one JSONL record per package root per process. Observation failures must not change the observed command's exit result. +3. Run the focused guarded test and static checks, then provide exact nightly workflow hunks to the integration owner. Retain the trace, source/version, and exit receipt on failure. + +## Acceptance and limits + +The synthetic tests must prove observed resolution behavior without downloads or native ORT. The hosted macOS learning trace is captured in the accompanying dated evidence record. An empty trace is not proof that no relevant package loaded; a start receipt proves that preload reached the registration attempt, not that registration succeeded. Resolution hooks do not prove native addon teardown or capture packages loaded before registration. The receipt records the outer kit command exit, not necessarily Ruflo's raw exit. diff --git a/docs/archive/2026-09-29-plan-upstream-watch-followups.md b/docs/archive/2026-09-29-plan-upstream-watch-followups.md new file mode 100644 index 00000000..ae6d6e52 --- /dev/null +++ b/docs/archive/2026-09-29-plan-upstream-watch-followups.md @@ -0,0 +1,30 @@ +# Upstream watch followups + +## Status at archival + +Implemented and independently reviewed on `fix/upstream-watch-followups` at +`b75c1e3fec4f645700f0f6d18220956867b7e5d0`. All eight local gates passed at that +source: unit/legacy tests, browser UI, typecheck, lint, complexity, Markdown, +build, and offline links. Controller workflow commits `aebe2ae6` and `b75c1e3f` +complete items 11/12; their shell and jq behavior was independently exercised. +Green `develop@84307574` was subsequently merged at `ff1eff27` without conflicts. +Final-head PR CI and merge remain pending at capture; no real dispatch or +notification was performed for this lane. + +M7 distinguishes deterministic validation failures from transient retries. +M8 bounds eligible fired-PR polling to seven days after the latest firing, +retaining the ledger; a later PR requires manual reconciliation. All twelve +minors have a fix, regression proof of existing behavior, or explicit no-change +disposition. M10 remains declined. Current behavior is documented in +[Upstream watch](../upstream-watch.md). + +## Execution and acceptance + +Scope: M7, M8, and the twelve numbered deferred minors in the recovered PR #253 report. M10 is declined because the numeric PR field matches emitted records. This lane owns watcher scripts, focused tests, one Codex registry entry, and current watcher documentation. The integration owner owns workflow edits. + +1. Add focused failing tests for deterministic retry failures, bounded dispatch polling, ledger errors, and the numbered edge cases. Keep synthetic fetch, dispatch, and git boundaries. +2. Implement the smallest watcher changes that make those tests pass. Check already-correct behavior and record no-change findings without empty commits. +3. Commit independently verifiable items separately. Run focused Node tests and static checks without a full suite, dispatch call, or GitHub write. +4. Put item 11/12 workflow hunks, commands, per-item dispositions, and residual limits in the ignored C4 report. Stop for independent review. + +Acceptance: deterministic errors do not sleep; transient errors retry at most twice; fired PR polling stops after seven days or when registry state is ineligible; invalid ledger registry exits nonzero; notices retain their body and maximum length semantics; all twelve minors are either fixed, proved already handled, or handed off as exact workflow hunks. diff --git a/docs/archive/2026-09-29-plan-v6-opencode-cost.md b/docs/archive/2026-09-29-plan-v6-opencode-cost.md new file mode 100644 index 00000000..d9cd70ff --- /dev/null +++ b/docs/archive/2026-09-29-plan-v6-opencode-cost.md @@ -0,0 +1,35 @@ +# V6 Unit 15: OpenCode reported-zero cost trust + +## Status + +The scoped parser/cost work and core cache invalidation handoff are implemented and independently accepted. The final OpenCode marker also includes observation semantics. +Full V6 verification passed at `1c02db91`; feature PR CI and develop integration +remain separate gates at this archival capture. No live store mutation is claimed. + +## Decision and scope + +OpenCode may record zero when a model has no configured rate. A positive-token +assistant response with recorded cost zero is therefore unpriced when its +provider is nonlocal or unknown. A known local provider's zero remains observed. +Positive recorded costs, zero-token responses, and missing-cost estimates keep +their existing treatment. Provider attribution remains per response. + +This unit changes only the OpenCode parser and the shared cost reader at its +OpenCode-specific row boundary. No rate is inferred from the model name, host, +or environment. + +## Evidence and acceptance + +- A bounded, read-only live-store query checks counts and shape, with a file + digest before and after. No affected positive-token zero-cost row was found. +- Synthetic SQLite messages with the real storage shape test remote, unknown, + and local provider IDs; mixed observed and missing cost; malformed costs; + and cold/warm cache conservation. +- Focused tests, typecheck, scoped lint, and diff checks gate the unit commit. + +## Integration dependency + +Existing schema-26 cached OpenCode records cannot reconstruct which responses +had an untrusted reported zero after per-response data was coalesced. The core +owner must invalidate or reparse those old records before release. This unit +does not edit the core-owned cache/index module. diff --git a/docs/archive/2026-09-29-plan-v6-surfaces-ui.md b/docs/archive/2026-09-29-plan-v6-surfaces-ui.md new file mode 100644 index 00000000..1f0c5d26 --- /dev/null +++ b/docs/archive/2026-09-29-plan-v6-surfaces-ui.md @@ -0,0 +1,20 @@ +# V6 session presentation execution + +## Status + +Units 22/23 and their consumer/default-suite handoff are implemented and independently accepted; final legacy Unknown filter compatibility is included. +Full V6 verification passed at `1c02db91`; feature PR CI and develop integration +remain separate gates at this archival capture. No live store mutation is claimed. + +Approved Units 22/23, based on `4a414024`; sole writer in `task/v6-surfaces-ui`. + +1. Add a shared presentation vocabulary and additive project session surface evidence. + Preserve legacy origin/count bases and honest uncertainty in old snapshots. +2. Render independent host, surface, initiator, provider and Git scope dimensions. + Retain bounded local raw details without inferring products or network providers. +3. Disclose pure imported exclusions, mixed files and unresolved ownership separately + in Intelligence, System Projects and the system command, including Cowork coverage. +4. Validate focused contracts and actual renderers, static checks and browser behavior. + Stop after unit commits for independent review; no publishing or personal data reads. + +Baseline: 29 focused renderer/import/empty-state tests passed before changes. diff --git a/docs/archive/2026-09-29-remediation-v2-integration-evidence.md b/docs/archive/2026-09-29-remediation-v2-integration-evidence.md new file mode 100644 index 00000000..f25e909f --- /dev/null +++ b/docs/archive/2026-09-29-remediation-v2-integration-evidence.md @@ -0,0 +1,86 @@ +# Remediation v2 integration evidence + +Capture: 2026-09-29. Documentation basis: `b5a0a946ae0aca2e1d435c78b12a98bc587d1266`. +These are sanitized controller receipts, checked against local commit objects; they +record integration and the cited CI observations, not fresh runtime probes. +The [scope matrix](2026-09-29-remediation-v2-scope-matrix.md) retains original +dispositions and row-specific limits. + +## Inherited documentation integration + +The controller baseline receipt establishes these commits as ancestors of +`94890a00ec6f5869c186d1066076c5a2e17dfee4`. Titles and ancestry establish +provenance; they do not allocate every moved file to one PR or test current behavior. + +| PR | Integration commit | Scope | +| --- | --- | --- | +| [#264](https://github.com/pacphi/agentic-kit/pull/264) | `e7cfe9ca4d49cb5e94fba698cd0e8376e0c829b7` | docs: one purpose per docs folder, lower-case names, archive point-in-time material | +| [#266](https://github.com/pacphi/agentic-kit/pull/266) | `7936ca48942e4ed1b629641eb61dc0dfbcf12951` | docs(archive): repair and validate historical links | +| [#269](https://github.com/pacphi/agentic-kit/pull/269) | `159c1c98d86d5fbb22b5752413f5889d8b8d172b` | docs: archive completed taxonomy and Sonnet routing plans | + +## Reviewed feature integrations + +| Work | PR | Reviewed source | Squash commit | Reviewed and squash tree | +| --- | --- | --- | --- | --- | +| V3 dashboard | [#276](https://github.com/pacphi/agentic-kit/pull/276) | `cd5cd08b52a190e2a91540b404b648f6b43a365c` | `eb963f503f806f80df7a8173d7c71314adbd7cfa` | `b5544923cee35a3e0217de1e22d9f6b7f7ddaf4c` | +| V5 intelligence | [#274](https://github.com/pacphi/agentic-kit/pull/274) | `36d78caf2fca6a61b7fd924499da9541ae85060c` | `af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b` | `7966ba916018af69fc4f4908e2d7202da5c507d6` | +| V4 follow-ups | [#275](https://github.com/pacphi/agentic-kit/pull/275) | `f19566adf2d0f0d7027bc4ea45f9f990e2b0355e` | `bb2e7efe88abd3cb2d73baf26530386549e3dc2e` | `0674d0762339037dcf24c30b9adbf58056b01db5` | +| V6 usage | [#282](https://github.com/pacphi/agentic-kit/pull/282) | `fa0cb7b788791a90577a2dfd9f9ca61e07d0591e` | `069553586fb3ad1d824877259562eef0765a42a1` | `78de12a7c593fc79c889cf853ff6eedf39cb583e` | +| Main watcher reconciliation | [#283](https://github.com/pacphi/agentic-kit/pull/283) | `09b198b6960c2dae14819d08729caa8b3794129b` | `9dc018b903ac074af0d4dbfcea0891c4e2a34e94` | `9db3da69020c147a798032b2cbde74e4eb88fdda` | + +V4 prerequisite [#273](https://github.com/pacphi/agentic-kit/pull/273), native trace +[#277](https://github.com/pacphi/agentic-kit/pull/277), and watcher follow-up +[#278](https://github.com/pacphi/agentic-kit/pull/278) remain separately reviewed +parts of the delivery. Public implementation detail is retained in the archived +[dashboard](2026-09-28-plan-dashboard-refresh.md), +[follow-ups](2026-09-28-plan-follow-ups-v2.md), +[usage](2026-09-28-plan-usage-accuracy.md), and +[native trace](2026-09-29-native-learning-trace.md) records. + +## Latest source-bound CI and ancestry receipts + +| Source | Run | Recorded result | +| --- | --- | --- | +| V6 reviewed `fa0cb7b788791a90577a2dfd9f9ca61e07d0591e` | [36614260309](https://github.com/pacphi/agentic-kit/actions/runs/36614260309) | 15 applicable checks passed; published-release consumer skipped | +| V6 develop squash `069553586fb3ad1d824877259562eef0765a42a1` | [36614973862](https://github.com/pacphi/agentic-kit/actions/runs/36614973862) | 13 checks passed | +| Main watcher reviewed `09b198b6960c2dae14819d08729caa8b3794129b` | [36616775739](https://github.com/pacphi/agentic-kit/actions/runs/36616775739) | 14 applicable checks passed; real watch skipped | +| Develop `88ce597f444d487a34d9871a447cf8942f76f760` | [36617683730](https://github.com/pacphi/agentic-kit/actions/runs/36617683730) | 13 checks passed | + +The last develop commit records parents `9dc018b903ac074af0d4dbfcea0891c4e2a34e94` +and `cb54f42e11fb6ff152ec278eb2dc1e119ee05ac5`. Its tree remains +`9db3da69020c147a798032b2cbde74e4eb88fdda`, equal to the reviewed watcher tree +and squash. The main input is an ancestor of the reviewed feature. This records +already-reviewed content while retaining ancestry; it is not final main approval. + +## Bounded dashboard observations + +The V3 controller receipt supporting [#256](https://github.com/pacphi/agentic-kit/issues/256) +records actual native-transcript replay on bounded re-entry, a failing-before and +passing-after fixture, and stable accepted/session/project counts for that scenario. +Offsets live in bounded memory: eviction or a new process can replay, so this is +not durable exactly-once delivery. A separate real Codex CLI 0.159.0 idle-fork +process/transcript join in a plain folder used the real survey. It establishes +plain-folder binding for that case, without a model turn, billing or browser claim. +Structured live input remains experimental with no verified producer. These three +answers make #256 a candidate final-main closure, subject to controller reconciliation. +No Network event-stream trace or diagnosed fix is recorded for +[#254](https://github.com/pacphi/agentic-kit/issues/254); the final trace check remains. + +## Remaining release and compatibility boundaries + +Node.js [#65934](https://github.com/nodejs/node/issues/65934) remains an upstream +Node 22 runner-protocol/backport boundary. The V6 test-local change replaced a +`beforeEach` console emission with `t.diagnostic`, preserving Unicode and assertions. +An official SHA-verified Node 22.23.2 binary and a coalesced actual-child V8 replay +matched the CI error; native historical chunk boundaries were not proved. Five +focused checks and five fixed replays passed in the retained receipt. This does +not establish an upstream runtime fix or permit dropping Windows assertions. + +The [Windows timing gate](2026-09-29-windows-ci-evidence.md) fails its recorded +three-run snapshot. AQE release and installed-state limitations are in the +[artifact receipt](2026-09-29-aqe-released-artifact-receipt.md). No real routine +trigger, paid provider run, global upgrade or real-store operation was performed +for these receipts. The memory loop remains OFF until main approval. Final +closeout CI, aggregate main PR review, releases, installations and approved +operational work remain separate gates. Attended time was not instrumented; +commit and run timestamps establish observed elapsed intervals only. diff --git a/docs/archive/2026-09-29-remediation-v2-rulings.md b/docs/archive/2026-09-29-remediation-v2-rulings.md new file mode 100644 index 00000000..c01cbbf3 --- /dev/null +++ b/docs/archive/2026-09-29-remediation-v2-rulings.md @@ -0,0 +1,152 @@ +# Remediation v2 ordered execution rulings + +Capture: 2026-09-29. Documentation basis: +`b5a0a946ae0aca2e1d435c78b12a98bc587d1266`. +This record transcribes the controller's current execution log into readable +language, retaining source order, decisions, reasons and stated costs. It includes +17 `Ruling:` entries and the separately worded ADR coordination ruling, for 18 +entries total. The earlier 13-entry rollup was incomplete. These execution rulings +operate within the [confirmed execution plan](../plans/2026-09-28-remediation-v2-develop-execution.md). +They do not create further release, data-operation or provider authority. + +## Maintainer-approved decisions + +The confirmed execution plan authorizes feature PRs into develop and conditional +squash integration after required CI and independent review. Final main approval +remains human-owned. D-3 through D-19 retain the plan's qualified dispositions; +D-5 moves execution identity, terminal outcomes, cancellation semantics and the +upgraded-install matrix to later v5 work. D-18 depends on unchanged totals and +D-19 labels structured live input experimental. D-9 preserves both v5 branches +and reserved ADR numbers. Personal Codex memory writes need a direct request. + +The maintainer also approved bounded unknown raw-origin metadata for local detail +under ADR-0060, and exact sanitized AQE init and Ruflo upstream messages recorded +by the controller. Those specific approvals do not authorize new posts. No future +Issue #239 W1–W8 or D-20+ work is approved by this record. The controller rulings below +are implementation and coordination choices within those boundaries. + +## Controller rulings in source order + +1. **Split the V4 B1 prerequisite PR.** V4 B1 overlapped V6 OpenCode and footprint + readers, so integrate it into develop before the remaining branches consume it. + Freeze the original V4 branch until merge, then continue from updated develop. + Remaining V4 scope is retained. Stated cost: one extra coordination PR. + +2. **Stage the A3 ADR amendment for the controller.** Implement and test the retry + stamp first. Apply its exact ADR-0063 amendment after V3 dashboard ADR work and + include the aligned ADR in the same V4 feature PR. Correct the old statement: + `record: false` still performs lookup without writing; `cacheOnly` skips both. + Reason: serialized ADR ownership and accurate behavior. Cost was not stated. + +3. **Run V6 Unit 20 before the remaining V4 merge.** Its quota/status-line classifier + paths do not overlap V4; B1 is integrated and C6 flags are verified. Raw-origin + policy still blocks dependent parser work. Stated cost if wrong: later quota + overlap needs serialized reconciliation; schema, UI and provider boundaries stay. + +4. **Integrate V6 Unit 2's agreed interface while raw-origin policy is pending.** + Both policy choices use the same factory return shape. Pass the factory result + through without inserting a policy filter or assuming approval; use synthetic + disposable caches. Final policy, ADR and delivery remain gated. Stated cost: + reshaping or schema staging may need revision before merge; no real cache or + privacy policy is deployed by this intermediate decision. + +5. **Add `declared-session-ids` without a schema bump.** Preserve the honest label + on legacy `transcript-files` snapshots. V3 exclusively owns the maintenance API, + so stage the API enum/test change for V6 consumer integration. Reason: avoid + concurrent contract edits. Stated cost: a pending public API handoff. + +6. **Pass explicit false for unchecked project trees.** Machine refresh must honor + the current selection across tabs. Correct the old plan's undefined/sticky + setting. Stated cost: a prior true selection is no longer inherited, intentionally. + +7. **Retire the precise C1 busy exception on the approved native criterion.** + The macOS/Linux AQE 3.14.4 live-owner criterion was independently satisfied. + Remove only `FsyncFailed`-as-busy; ordinary `LockHeld` remains. Do not invent a + global AQE floor from the older comment: only store merge has that feature floor. + Stated compatibility cost: the older 3.14.3 error sequence fails closed. At this + decision Windows Ruflo routing remained a separate failed gate; the later + qualified Windows proof is in the [AQE/native receipt](2026-09-29-aqe-released-artifact-receipt.md). + +8. **Exclude incomplete or corrupt mixed-file ownership as a whole.** Surface the + coverage as incomplete, following the fix brief and ADR, to avoid false imported + spend. Clean observed mixed files retain proved response/origin attribution. + Stated cost: undercounting a proved prefix in damaged files; prefix salvage is + not claimed. Investigate actual zero-response/token shapes separately. + +9. **Add `sessionSurfaces` while preserving legacy semantics.** Retain + `sessionOrigins`, `countBasis` and uncertain legacy precise mode. Assign narrow + ownership for management query/focus navigation, intelligence history and + footprint/projects DTOs. Reason: avoid a schema bump and false legacy origin + inference. Stated cost: additive compatibility complexity, reviewed at integration. + +10. **Exclude ignored scratch evidence from local whole-repository ESLint.** + Preserve immutable measurement scripts when default lint discovery includes + generated ignored probes. All tracked targets and clean-checkout CI lint remain. + Stated cost: local scratch is not linted and is not shipped. This cannot hide + tracked defects. + +11. **Run independent V6 Unit 15 before Unit 12 completes.** Give its isolated branch + exclusive usage-cost/OpenCode paths; Unit 12 excludes them. Integrate only after + both reviews and a clean handoff, through the final V6 feature PR. Stated cost + if wrong: delayed narrow integration handoff; concurrent writers remain forbidden. + +12. **Accept one charge owner in a fixed Claude identity pool.** The pool spans twice + the display days, capped at 730; eligibility requires both modification time and + end time. Outside-pool duplicates cannot steal or augment a charge. Uncovered + current data is excluded with coverage counts. Comparison-toggle invariance and + unique-charge conservation were checked. The earlier per-cohort Claude policy + was explicitly rejected and superseded. Stated cost: up to one preceding + display-width of extra cold reads or cached-claim validation. This is neither + a measured runtime benchmark nor a whole-corpus uniqueness guarantee. + +13. **Keep the conventional OpenCode path for storage census.** Actual usage and + project readers use `selectOpencodeSource`; the conventional default still + supplies the directory used by footprint/storage inventory. Returning null on + history ambiguity would break unrelated census callers. Stated cost: that + conventional location is not selected-history authority. Verify actual history + readers use the selector; no storage inventory redesign follows. + +14. **Build Unit 17's pure storage detector independently.** Reserve only its two + new paths in a separate worktree, using a caller-owned database or explicit + legacy root. It cannot select sources or change the index. Full acceptance waits + for Unit 16 and consumers. Reason: independent work without shared writers. + Stated cost: an explicit interface handoff. + +15. **Start main watcher reconciliation alongside V6 Windows fixes.** The paths are + disjoint; integrate updated develop only after V6 merges and develop is green. + Preserve both input sets, documented HTTP retry scope, token redaction, deferred + preview and blind-backlog handling. Reason: avoid serial idle time. Stated cost + if wrong: later integration rework, never concurrent writer paths. + +16. **Limit trigger retries to documented HTTP 500/503.** Do not retry body transport + errors or after an observed session URL. The routine API offers no idempotency + guarantee, and bounded retry is not proof of no side effects. Stated cost if + wrong: other transient 5xx responses wait for a later check. No real trigger + was performed. The [watcher integration receipt](2026-09-29-remediation-v2-integration-evidence.md) + binds the implemented source and CI. + +17. **Record already-reviewed main ancestry after feature squash.** Permit a + content-identical merge only when the reviewed feature contains the exact main + input and the squash tree equals the reviewed tree. Reason: preserve ancestry + without repeating the final main conflict. Stated cost if wrong: hiding an + unincorporated main change. The controller checked ancestor, tree, parents and + current-main preconditions; latest develop CI remained the dependency gate. + +18. **Include shared test-helper comments in the whole-tree guard sweep.** The + approved no-allowlist sweep permits comment-only label rewrites in helpers, + resolving the narrower worker brief. No competing writer owns those paths. + Runtime statements, exports and behavior must remain unchanged. Stated cost + if wrong: an accidental helper behavior change; source-token comparison and + focused checks must retain the code contract. + +## Limits carried into closeout + +The [integration receipt](2026-09-29-remediation-v2-integration-evidence.md) records +Node 22's upstream runner limitation and the test-local diagnostic mitigation. +Source fixes, released-artifact observations and installed behavior remain distinct. +The [Windows receipt](2026-09-29-windows-ci-evidence.md) retains a failing timing +gate snapshot; the [AQE receipt](2026-09-29-aqe-released-artifact-receipt.md) retains +untested platforms, unsigned fresh-chain limits and setup churn in 3.14.5. +No paid runs were performed. The memory loop remains OFF until main approval. +Attended time was not instrumented. Program and execution plans stay active while +final main review and operational gates remain. diff --git a/docs/archive/2026-09-29-remediation-v2-scope-matrix.md b/docs/archive/2026-09-29-remediation-v2-scope-matrix.md new file mode 100644 index 00000000..e88980e7 --- /dev/null +++ b/docs/archive/2026-09-29-remediation-v2-scope-matrix.md @@ -0,0 +1,269 @@ +# Remediation v2 scope reconciliation + +Capture: 2026-09-29. Documentation basis: `b5a0a946ae0aca2e1d435c78b12a98bc587d1266`. +Implementation evidence ends at develop `88ce597f444d487a34d9871a447cf8942f76f760`. +This records 189 Appendix A and 57 Appendix B identifiers exactly once, in source order. +Original disposition is retained separately from the current evidence disposition. + +The [approved program](../plans/2026-09-28-remediation-program-v2.md) and +[execution plan](../plans/2026-09-28-remediation-v2-develop-execution.md) govern authority. +Grouped Appendix B findings keep their original group evidence; the expanded identifiers +do not imply separate runtime probes. Historical DONE and declined/superseded entries +are provenance, not a new test of current behavior. Original ruling line references +refer to the source reviews identified by the program. + +See the [integration evidence](2026-09-29-remediation-v2-integration-evidence.md), +[Windows evidence](2026-09-29-windows-ci-evidence.md), +[AQE evidence](2026-09-29-aqe-released-artifact-receipt.md), and +[ordered rulings](2026-09-29-remediation-v2-rulings.md). +V7 guard work, final integration gates, the final main PR and human approval remain. +No release, installation, real-store mutation, deletion or paid provider run is proved. + +| ID | Appendix | Original disposition | Current disposition | Source evidence / PR / commit | Remaining condition and evidence limit | +| --- | --- | --- | --- | --- | --- | +| B0-1 | A | DONE | inherited done | #241 (`476717b63333802c7a5bc55267556303fd1ea025`); historical commit 476717b63333802c7a5bc55267556303fd1ea025 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-2 | A | DONE | inherited done | #241 (`476717b63333802c7a5bc55267556303fd1ea025`); historical commit 476717b63333802c7a5bc55267556303fd1ea025 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-3 | A | DONE | inherited done | #241 (`476717b63333802c7a5bc55267556303fd1ea025`); historical commit 476717b63333802c7a5bc55267556303fd1ea025 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-6 | A | DONE | inherited done | #241 (`476717b63333802c7a5bc55267556303fd1ea025`); historical commit 476717b63333802c7a5bc55267556303fd1ea025 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-11 | A | DONE | inherited done | #241 (`476717b63333802c7a5bc55267556303fd1ea025`); historical commit 476717b63333802c7a5bc55267556303fd1ea025 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-12 | A | DONE | inherited done | #241 (`476717b63333802c7a5bc55267556303fd1ea025`); historical commit 476717b63333802c7a5bc55267556303fd1ea025 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-5 | A | DONE | inherited done | `links (external)` in `.github/workflows/nightly.yml`; `links (internal)` in `ci.yml` | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-7 | A | DONE | inherited done | #252 (`31a1a39b621d19adc793edf0eef26c36ba09773e`), 6a Task 11 (about a 96 % cut); historical commit 31a1a39b621d19adc793edf0eef26c36ba09773e | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-8 | A | DONE | inherited done | Check ran in #241; the About render test moved to B9-11 (V5) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-9 | A | DECLINED | declined by prior ruling | Ruling `L:35` (normal dashboard operation); the `readLimits` part is B6b-15 | Retain original ruling and citation; no implementation. | +| B0-10 | A | DONE | inherited done | `locators.json` has 0 matches for `` (re-checked 2026-09-28) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. Source example path sanitized for export; no real personal path is asserted. | +| B0-4 | A | D-17 | answered/deferred design decision | program v2 Appendix A row B0-4 | Drop one-off AQE coverage-gap analysis; preserve actual 70/70/70 gate proof. | +| B0-13 | A | V4 | implemented on develop via V4 | Items B.9, B.12, B.3, B.6; accepted unit 53f6a17f8f79262035048e6c4dfd129103d32042; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B0-14 | A | V4 | implemented on develop via V4 | Items B.9, B.12, B.3, B.6; accepted unit b94d19f5f9dbde61a67fbddc62afd1d31d1d8478; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B0-15 | A | V4 | implemented on develop via V4 | Items B.9, B.12, B.3, B.6; accepted unit b4cc7d37bd198ff5d99e1e0d3b18baf4b19fff96; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B0-21 | A | V4 | implemented on develop via V4 | Items B.9, B.12, B.3, B.6; accepted unit 07e9c6738a2573a0dfceafacc0e2e6857d2e9c6a; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B0-16 | A | D-16 | answered/deferred design decision | program v2 Appendix A row B0-16 | Accepted Ruflo table-missing-before-quick_check limitation; no new fix. | +| B0-17 | A | DECLINED | declined by prior ruling | Documented trade-off (`ak sync --help`, ADR-0033 §9) | Retain original ruling and citation; no implementation. | +| B0-18 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `docs/TROUBLESHOOTING.md:94` now reads "Windows Limits data appears after `ak sync`"; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-19 | A | DONE | inherited done | Windows CI green on #241 through #263 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B0-20 | A | SUPERSEDED | superseded by cited decision/work | #241's squash dropped them; briefs forbid trailer-like lines | Retain source ruling; verify cited replacement in final reconciliation. | +| B0-22 | A | V3 | implemented on develop via V3 | Client fixes; accepted unit 8420f9d9428a4041f11fd8c034921f172275eda3; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Final main PR/review remains. | +| B0-23 | A | V3 | implemented on develop via V3 | Client fixes; accepted unit 8420f9d9428a4041f11fd8c034921f172275eda3; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Final main PR/review remains. | +| B0-24 | A | issue #256 → V3 | V3 code/evidence on develop; #256 closure pending | The controller adds the evidence in Wave 0; V3 closes #256 (D-18, D-19); accepted unit e173bab9d32d350f9f7818aae24392d70791022c; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Compare all three #256 exit conditions and disclose replay/experimental/no-producer limits before closing. | +| B0-25 | A | D-5 | decision D-5 carried to v5 | own #239 | Execution identity, terminal outcomes, cancellation semantics and upgraded-install matrix remain out of v2 completion. | +| B1-1 | A | DONE | inherited done | #244 (`5f5ca175bb6d9a4108402f4634be746e4d166ce1`); historical commit 5f5ca175bb6d9a4108402f4634be746e4d166ce1 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B1-2 | A | DONE | inherited done | #244 (`5f5ca175bb6d9a4108402f4634be746e4d166ce1`); historical commit 5f5ca175bb6d9a4108402f4634be746e4d166ce1 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B1-3 | A | §2 step 3 | snapshot/schema one-off pending | Snapshot still schema 7 | Current schema snapshot and operational approval to apply; no operation inferred from old schema-7 observation. | +| B1-4 | A | V6 | integrated on develop via V6 | Items 3, 2 and 6; [V6 integration receipt](2026-09-29-remediation-v2-integration-evidence.md) | Closeout integration gates and final main PR/human review remain. Aggregate PR/source-plan evidence; no invented fresh per-item runtime proof. | +| B1-5 | A | V6 | integrated on develop via V6 | Items 3, 2 and 6; [V6 integration receipt](2026-09-29-remediation-v2-integration-evidence.md) | Closeout integration gates and final main PR/human review remain. Aggregate PR/source-plan evidence; no invented fresh per-item runtime proof. | +| B1-6 | A | V6 | integrated on develop via V6 | Items 3, 2 and 6; [V6 integration receipt](2026-09-29-remediation-v2-integration-evidence.md) | Closeout integration gates and final main PR/human review remain. Aggregate PR/source-plan evidence; no invented fresh per-item runtime proof. | +| B2-1 | A | DONE | inherited done | #245 (`a0b75494789d45eb8d2e08fc42c4de477a2a8151`); strict Windows tripwire on #252; historical commit a0b75494789d45eb8d2e08fc42c4de477a2a8151 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B2-2 | A | DONE | inherited done | #245 (`a0b75494789d45eb8d2e08fc42c4de477a2a8151`); strict Windows tripwire on #252; historical commit a0b75494789d45eb8d2e08fc42c4de477a2a8151 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B2-3 | A | DONE | inherited done | #245 (`a0b75494789d45eb8d2e08fc42c4de477a2a8151`); strict Windows tripwire on #252; historical commit a0b75494789d45eb8d2e08fc42c4de477a2a8151 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B2-4 | A | DONE | inherited done | #245 (`a0b75494789d45eb8d2e08fc42c4de477a2a8151`); strict Windows tripwire on #252; historical commit a0b75494789d45eb8d2e08fc42c4de477a2a8151 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B2-5 | A | DONE | inherited done | #263: `tests/kit/verify-command.test.mjs` is gone, and its stricter replacements showed 0 flakes in 20 runs (`L:380`) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B2-6 | A | V4 | implemented on develop via V4 | Item B.1; reviewed head 404642628d57eac1e42f26286f7db763f0fce704; PR #273; integration commit bd6b4f0fc33b68812497922ab9bc8467e3d4d443; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B3-1 | A | DONE | inherited done | #247 (`264609992aacac4302c3dd8d9ca769d5222fc1a9`); historical commit 264609992aacac4302c3dd8d9ca769d5222fc1a9 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B3-2 | A | DONE | inherited done | #247 (`264609992aacac4302c3dd8d9ca769d5222fc1a9`); historical commit 264609992aacac4302c3dd8d9ca769d5222fc1a9 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B3-11 | A | DONE | inherited done | #247 (`264609992aacac4302c3dd8d9ca769d5222fc1a9`); historical commit 264609992aacac4302c3dd8d9ca769d5222fc1a9 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B3-12 | A | DONE | inherited done | #247 (`264609992aacac4302c3dd8d9ca769d5222fc1a9`); historical commit 264609992aacac4302c3dd8d9ca769d5222fc1a9 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B3-13 | A | DONE | inherited done | #247 (`264609992aacac4302c3dd8d9ca769d5222fc1a9`); historical commit 264609992aacac4302c3dd8d9ca769d5222fc1a9 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B3-3 | A | DONE | inherited done | Evidence on ruvnet/ruflo#3196; filed ruvnet/ruflo#3509 and #3508. All three are in the registry (`src/lib/hook-audit/agentic-dependency-constraints.json`) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B3-4 | A | DONE | inherited done | Evidence on ruvnet/ruflo#3196; filed ruvnet/ruflo#3509 and #3508. All three are in the registry (`src/lib/hook-audit/agentic-dependency-constraints.json`) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B3-5 | A | DONE | inherited done | Evidence on ruvnet/ruflo#3196; filed ruvnet/ruflo#3509 and #3508. All three are in the registry (`src/lib/hook-audit/agentic-dependency-constraints.json`) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B3-7 | A | DONE | inherited done | Re-checks recorded (`reports/b3-plan.md`; `L:195`) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B3-8 | A | DONE | inherited done | Re-checks recorded (`reports/b3-plan.md`; `L:195`) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B3-6 | A | V4 | implemented on develop via V4 | Items C.1, B.7; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| B3-9 | A | V4 | implemented on develop via V4 | Items C.1, B.7; accepted unit 03af512a7e42bec9e67744ff9b0f384c35ccbb28; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B3-10 | A | V4 | implemented on develop via V4 | Items C.1, B.7; accepted unit 03af512a7e42bec9e67744ff9b0f384c35ccbb28; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B4-1 | A | SUPERSEDED | superseded by cited decision/work | Decisions 14 and 15: #249 (`88e26999e6aecc61d3a1d09afbe51dd7173c93e7`), #253 (`82d1211b5526c9cc78702571c13813eead0c89ce`); #243 closed | Retain source ruling; verify cited replacement in final reconciliation. | +| B4-2 | A | SUPERSEDED | superseded by cited decision/work | Decisions 14 and 15: #249 (`88e26999e6aecc61d3a1d09afbe51dd7173c93e7`), #253 (`82d1211b5526c9cc78702571c13813eead0c89ce`); #243 closed | Retain source ruling; verify cited replacement in final reconciliation. | +| B4-3 | A | DONE | inherited done | #246 (`fca6b2bc208d718f74ebaef53ebbd438cbb34f3d`; the reconciliation's `29f93dfc042f443b588572aa0418229f23f23fa3` is a branch commit); historical commit fca6b2bc208d718f74ebaef53ebbd438cbb34f3d; historical commit 29f93dfc042f443b588572aa0418229f23f23fa3 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B4-5 | A | DONE | inherited done | #246 (`fca6b2bc208d718f74ebaef53ebbd438cbb34f3d`; the reconciliation's `29f93dfc042f443b588572aa0418229f23f23fa3` is a branch commit); historical commit fca6b2bc208d718f74ebaef53ebbd438cbb34f3d; historical commit 29f93dfc042f443b588572aa0418229f23f23fa3 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B4-4 | A | V4 | implemented on develop via V4 | Items C.1, C.2, C.4. B4-10's registry part is DONE in #258 (`a77bfdee3e3948fa1812ac1ce9199be43e0cc8e1`); PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| B4-10 | A | V4 | implemented on develop via V4 | Items C.1, C.2, C.4. B4-10's registry part is DONE in #258 (`a77bfdee3e3948fa1812ac1ce9199be43e0cc8e1`); PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| B4-11 | A | V4 | implemented on develop via V4 | Items C.1, C.2, C.4. B4-10's registry part is DONE in #258 (`a77bfdee3e3948fa1812ac1ce9199be43e0cc8e1`); PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| B4-6 | A | SUPERSEDED | superseded by cited decision/work | #617 was handled by hand in #250, and dispatch was redesigned in #253. The first live dispatch is tracked in DoD 6 | Retain source ruling; verify cited replacement in final reconciliation. | +| B4-7 | A | DONE | inherited done | Replies posted (`L:56`, `L:117-118`) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B4-8 | A | DONE | inherited done | Replies posted (`L:56`, `L:117-118`) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B4-9 | A | DONE | inherited done | #249 (`88e26999e6aecc61d3a1d09afbe51dd7173c93e7`); historical commit 88e26999e6aecc61d3a1d09afbe51dd7173c93e7 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B4-12 | A | DONE | inherited done | #250 (`3929cdd527231561bf6509f1405e6f446a057347`); historical commit 3929cdd527231561bf6509f1405e6f446a057347 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-1 | A | DONE | inherited done | #250 (`3929cdd527231561bf6509f1405e6f446a057347`); historical commit 3929cdd527231561bf6509f1405e6f446a057347 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-2 | A | DONE | inherited done | #250 (`3929cdd527231561bf6509f1405e6f446a057347`); historical commit 3929cdd527231561bf6509f1405e6f446a057347 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-14 | A | DONE | inherited done | #250 (`3929cdd527231561bf6509f1405e6f446a057347`); historical commit 3929cdd527231561bf6509f1405e6f446a057347 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-16 | A | DONE | inherited done | #250 (`3929cdd527231561bf6509f1405e6f446a057347`); historical commit 3929cdd527231561bf6509f1405e6f446a057347 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-3 | A | issue agentic-qe#655 | external closed; selected released compact path verified | Kept on conformance evidence; #755–#758 watched; [AQE artifact receipt](2026-09-29-aqe-released-artifact-receipt.md) (observed 2026-09-29T14:27:38Z): AQE #655 selected compact contract in private released 3.14.5 | Full/none guidance, complete receipt, cross-platform criteria and installed conformance remain unverified. Selected compact path only: 315-byte owned region and unchanged complete AGENTS bytes/mtime across two identical fixture invocations; initial creation mtime not recorded. | +| B5-4 | A | DONE | inherited done | #248 (`1c3c4d354433f444999916d6f641a8b83dafcc1b`); historical commit 1c3c4d354433f444999916d6f641a8b83dafcc1b | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-5 | A | issue agentic-qe#753 | external closed; released fresh append verified | AQE repairs the chain; [AQE artifact receipt](2026-09-29-aqe-released-artifact-receipt.md) (observed 2026-09-29T14:27:38Z): AQE #753 fresh-chain append in private released 3.14.5 | Keep holder refusal; old-fork repair, signature validation, import-splice repair, Windows/Linux and installed conformance remain unverified. Three synchronized two-writer rounds completed, but only one demonstrably interleaved; entries were unsigned synthesized PATTERN_CREATE events. | +| B5-6 | A | DONE | inherited done | Filed agentic-qe#754; wording in #250 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-7 | A | DONE | inherited done | Historical one-off receipt recorded in the approved program (private receipt identifier omitted) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-8 | A | §2 step 2 | pin reconciliation evidence; operation separate | Pin still absent (verified) | Check current pinned surfaces and status; real-store changes require separate approval. | +| B5-9 | A | V4 and D-7 | V4 scan integrated; real-store operation gated | Item B.5 (the scan) + D-7 (the store); accepted unit 77557d4f3095c62f447b54d6fbd0d9d23ba7aceb; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | D-7 real stray-store merge/import decision remains separate; no real-store action without explicit approval and backup. | +| B5-10 | A | V4 | implemented on develop via V4 | Items B.5, C.4; accepted unit 4d376a11677f7a779c14f34a3012c85aa1649e5b; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B5-11 | A | V4 | implemented on develop via V4 | Items B.5, C.4; accepted unit da3b820bdfadb408898b9f728614318cbfad9c88; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B5-17 | A | V4 | implemented on develop via V4 | Items B.5, C.4; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| B5-12 | A | DONE | inherited done | Folded into the agentic-qe#735 comment | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-13 | A | DECLINED | declined by prior ruling | Ruling `L:188` | Retain original ruling and citation; no implementation. | +| B5-15 | A | D-6 | external closed; 3.14.5 still reproduces | [AQE artifact receipt](2026-09-29-aqe-released-artifact-receipt.md) (observed 2026-09-29T14:27:38Z): AQE #778 settings churn reproduces in released 3.14.5; release at 10:08Z predates fix PR #783 merge 63d3debab54105dbc93030eb0b0088e2f9bebd9c at 13:05Z | Subsequent released-artifact conformance unverified; global installed 3.14.4 metadata unchanged; keep #239 open for readiness. Closed upstream issue and later merged fix source do not establish a fixed released or installed artifact. | +| B5-18 | A | DONE | inherited done | vibium uninstalled with approval (`L:176`, `L:180`) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-19 | A | DONE | inherited done | Registry lists agentic-qe#561 and #755–#759 (grep) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-20 | A | DONE | inherited done | 21 posts read back (`L:215-239`) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B5-21 | A | DECLINED | declined by prior ruling | Maintainer decision `L:213` | Retain original ruling and citation; no implementation. | +| B6a-1 | A | DONE | inherited done | #252 (`31a1a39b621d19adc793edf0eef26c36ba09773e`), released in alpha.59; historical commit 31a1a39b621d19adc793edf0eef26c36ba09773e | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6a-13 | A | DONE | inherited done | #252 (`31a1a39b621d19adc793edf0eef26c36ba09773e`), released in alpha.59; historical commit 31a1a39b621d19adc793edf0eef26c36ba09773e | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6a-2 | A | DONE | inherited done | #263: ADR-0063 `:102`; the `guarded` probe around `refreshPlanHosts` in `sync.mjs`; ADR-0063 "Known limitations" item 1 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6a-4 | A | DONE | inherited done | #263: ADR-0063 `:102`; the `guarded` probe around `refreshPlanHosts` in `sync.mjs`; ADR-0063 "Known limitations" item 1 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6a-8 | A | DONE | inherited done | #263: ADR-0063 `:102`; the `guarded` probe around `refreshPlanHosts` in `sync.mjs`; ADR-0063 "Known limitations" item 1 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6a-3 | A | V4 | implemented on develop via V4 | Items B.2, B.1; accepted unit 001225daa57662bfbc77510b39383e606b865823; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B6a-6 | A | V4 | implemented on develop via V4 | Items B.2, B.1; reviewed head 404642628d57eac1e42f26286f7db763f0fce704; PR #273; integration commit bd6b4f0fc33b68812497922ab9bc8467e3d4d443; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B6a-5 | A | V7 | pending V7 label cleanup | Item 1 (6b removed its own labels in #263) | Run tree-wide label guard after V6, preserve original 6b removal history. | +| B6a-7 | A | V5 | implemented on develop via V5 | Task 7; accepted unit 705bd5ff51c431a36e7b2b1e22a5c699c4315fa9; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b | Final main PR/review remains. | +| B6a-9 | A | V3 | implemented on develop via V3 | PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Final main PR/review remains. Source-plan 6c group-level mapping; PR #276 is aggregate integration evidence. | +| B6a-12 | A | V3 | implemented on develop via V3 | PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Final main PR/review remains. Source-plan 6c group-level mapping; PR #276 is aggregate integration evidence. | +| B6a-10 | A | DECLINED | declined by prior ruling | `L:313`, `L:323` | Retain original ruling and citation; no implementation. | +| B6a-11 | A | DECLINED | declined by prior ruling | `L:313`, `L:323` | Retain original ruling and citation; no implementation. | +| B6b-1 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-2 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-4 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-5 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-6 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-7 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-9 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-10 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-11 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-15 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-16 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-17 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-20 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-23 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-26 | A | DONE | inherited done | #263 (`e957737bb3b2d763aa09379992d2f6a90a574b68`): `src/lib/refresh.mjs`, `src/lib/live-checks.mjs`, `x/host-connection.mjs`, `reset-routes`, `--show-text`, `sync/plan-versions.mjs`, `configErrorRecovery`, `tests/kit/host-dry-run.test.mjs`, `quota.mjs` `host-not-found`, ADR index rows 0053–0063; historical commit e957737bb3b2d763aa09379992d2f6a90a574b68 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B6b-3 | A | V3 | implemented on develop via V3 | 6c-1 … 6c-3; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Final main PR/review remains. Source-plan 6c group-level mapping; PR #276 is aggregate integration evidence. | +| B6b-8 | A | V3 | implemented on develop via V3 | 6c-1 … 6c-3; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Final main PR/review remains. Source-plan 6c group-level mapping; PR #276 is aggregate integration evidence. | +| B6b-18 | A | V3 | implemented on develop via V3 | 6c-1 … 6c-3; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Final main PR/review remains. Source-plan 6c group-level mapping; PR #276 is aggregate integration evidence. | +| B6b-13 | A | DONE (CLI half) and V3 (dashboard half) | CLI inherited done; dashboard implemented on develop | #263 Tasks 13–14; 6c-4, 6c-5; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Final main PR/review remains. | +| B6b-19 | A | DONE (CLI half) and V3 (dashboard half) | CLI inherited done; dashboard implemented on develop | #263 Tasks 13–14; 6c-4, 6c-5; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Final main PR/review remains. | +| B6b-12 | A | SUPERSEDED | superseded by cited decision/work | False premise (R11) | Retain source ruling; verify cited replacement in final reconciliation. | +| B6b-14 | A | SUPERSEDED | superseded by cited decision/work | R17 | Retain source ruling; verify cited replacement in final reconciliation. | +| B6b-21 | A | V4 | implemented on develop via V4 | Items A.3, A.1, C.6; accepted unit 175677a682ec028c775eb1afe440add846827527; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B6b-22 | A | V4 | implemented on develop via V4 | Items A.3, A.1, C.6; accepted unit 557ba358e18a3c5ec50f806f5b69dae529a1cf98; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B6b-25 | A | V4 | implemented on develop via V4 | Items A.3, A.1, C.6; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| B6b-24 | A | DONE (PR body) and D-2 (release notes) | PR body done; release-notes operation pending | #263's body | D-2 release notes only after release approval and exact artifact gate. | +| UA-1 | A | V6 | integrated on develop via V6 | —; [V6 integration receipt](2026-09-29-remediation-v2-integration-evidence.md) | Closeout integration gates and final main PR/human review remain. Aggregate PR/source-plan evidence; no invented fresh per-item runtime proof. | +| UA-3 | A | V6 | integrated on develop via V6 | —; [V6 integration receipt](2026-09-29-remediation-v2-integration-evidence.md) | Closeout integration gates and final main PR/human review remain. Aggregate PR/source-plan evidence; no invented fresh per-item runtime proof. | +| UA-4 | A | V6 | integrated on develop via V6 | —; [V6 integration receipt](2026-09-29-remediation-v2-integration-evidence.md) | Closeout integration gates and final main PR/human review remain. Aggregate PR/source-plan evidence; no invented fresh per-item runtime proof. | +| UA-5 | A | V6 | integrated on develop via V6 | —; [V6 integration receipt](2026-09-29-remediation-v2-integration-evidence.md) | Closeout integration gates and final main PR/human review remain. Aggregate PR/source-plan evidence; no invented fresh per-item runtime proof. | +| UA-2 | A | issue #257: disclosure in V6; the source after v2 | V6 disclosure integrated; Cowork reader deferred | See "After v2"; own #257; [V6 integration receipt](2026-09-29-remediation-v2-integration-evidence.md) | Keep #257 open: source discovery and optional reader remain future work. | +| B9-1 | A | DONE | inherited done | #259 (`172f23074ecd8c5a21177ef58e4d58af0b7888ed`): orchestrator restore, "Partial data" card, round-trip prune, `docs/SETUP.md:126` and the `docs/UPGRADING.md` section of 2026-09-28; historical commit 172f23074ecd8c5a21177ef58e4d58af0b7888ed | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B9-4 | A | DONE | inherited done | #259 (`172f23074ecd8c5a21177ef58e4d58af0b7888ed`): orchestrator restore, "Partial data" card, round-trip prune, `docs/SETUP.md:126` and the `docs/UPGRADING.md` section of 2026-09-28; historical commit 172f23074ecd8c5a21177ef58e4d58af0b7888ed | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B9-8 | A | DONE | inherited done | #259 (`172f23074ecd8c5a21177ef58e4d58af0b7888ed`): orchestrator restore, "Partial data" card, round-trip prune, `docs/SETUP.md:126` and the `docs/UPGRADING.md` section of 2026-09-28; historical commit 172f23074ecd8c5a21177ef58e4d58af0b7888ed | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B9-17 | A | DONE | inherited done | #259 (`172f23074ecd8c5a21177ef58e4d58af0b7888ed`): orchestrator restore, "Partial data" card, round-trip prune, `docs/SETUP.md:126` and the `docs/UPGRADING.md` section of 2026-09-28; historical commit 172f23074ecd8c5a21177ef58e4d58af0b7888ed | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| B9-2 | A | V4 | implemented on develop via V4 | Items B.8, B.4, B.3, C.3, B.11, C.1; accepted unit b5570e1ed22717648a109c0dc2c4ff9418a992c8; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B9-3 | A | V4 | implemented on develop via V4 | Items B.8, B.4, B.3, C.3, B.11, C.1; accepted unit c8bfd64f7322718778499bf068352f3c7f723457; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B9-12 | A | V4 | implemented on develop via V4 | Items B.8, B.4, B.3, C.3, B.11, C.1; accepted unit b4cc7d37bd198ff5d99e1e0d3b18baf4b19fff96; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. | +| B9-15 | A | V4 | implemented on develop via V4 | Items B.8, B.4, B.3, C.3, B.11, C.1; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| B9-16 | A | V4 | implemented on develop via V4 | Items B.8, B.4, B.3, C.3, B.11, C.1; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| B9-18 | A | V4 | implemented on develop via V4 | Items B.8, B.4, B.3, C.3, B.11, C.1; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| B9-5 | A | issue #254: evidence-gated (§2 step 5) | evidence-gated issue #254 | V3 fixes it if a trace arrives; V7 closes it otherwise; own #254 | Recheck all available browser trace evidence; close with no-repro invitation only if issue criteria permit, otherwise leave open. | +| B9-6 | A | issue #255: after v2 | deferred named issue | See "After v2"; own #255 | Design-first worker warning after v2; no implementation claim. | +| B9-7 | A | DECLINED | declined by prior ruling | B9-OQ1 | Retain original ruling and citation; no implementation. | +| B9-9 | A | DECLINED | declined by prior ruling | B9-R12 | Retain original ruling and citation; no implementation. | +| B9-10 | A | issue #256 → V3 | V3 code/evidence on develop; #256 closure pending | D-18, D-19 and one real-machine observation; accepted unit e173bab9d32d350f9f7818aae24392d70791022c; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Compare all three #256 exit conditions and disclose replay/experimental/no-producer limits before closing. | +| B9-11 | A | V5 | implemented on develop via V5 | B9-14 then goes to §2 step 7; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b | Final main PR/review remains. | +| B9-13 | A | V5 | implemented on develop via V5 | B9-14 then goes to §2 step 7; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b | Final main PR/review remains. | +| B9-14 | A | V5 | V5 inventory integrated; deletion gated | B9-14 then goes to §2 step 7; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b | Read-only inventory is complete via PR #274; actual cleanup/deletion requires exact approval. | +| B9-20 | A | V5 | implemented on develop via V5 | B9-14 then goes to §2 step 7; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b | Final main PR/review remains. | +| B9-19 | A | DONE | inherited done | #261 (`0fa5e4890aabd830ee9d41878dd4e41bc85d3c73`); historical commit 0fa5e4890aabd830ee9d41878dd4e41bc85d3c73 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| LQ-1 | A | V5 | implemented on develop via V5 | PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b | Final main PR/review remains. | +| LQ-4 | A | V5 | implemented on develop via V5 | PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b | Final main PR/review remains. | +| LQ-2 | A | V4 | V4 core integrated; conditional #3419 line deferred | Items B.6, B.10; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Ruflo #3419 has no maintainer-supported answer; add conditional manual-fix line only if upstream evidence warrants it. Do not call the conditional line completed. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| LQ-3 | A | V4 | implemented on develop via V4 | Items B.6, B.10; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. Source-plan item-level mapping only; PR #275 is aggregate integration evidence. | +| HK-1 | A | DONE | inherited done | `git ls-remote --heads origin feat/managed-ruflo-components` is empty | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| HK-2 | A | DONE | inherited done | `git worktree list` and `git stash list` | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| HK-3 | A | DONE | inherited done | `git worktree list` and `git stash list` | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| HK-4 | A | DONE | inherited done | Dependabot #231 closed | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| HK-5 | A | DECLINED | declined by prior ruling | Stay documented or excluded (v1 plan; `H1:159-160`) | Retain original ruling and citation; no implementation. | +| HK-7 | A | DECLINED | declined by prior ruling | Stay documented or excluded (v1 plan; `H1:159-160`) | Retain original ruling and citation; no implementation. | +| HK-6 | A | D-9 | answered/deferred design decision | program v2 Appendix A row HK-6 | Retain two v5 branches under D-9; no deletion/migration. | +| HK-8 | A | DONE | inherited done | No `ak-*` folder beside the repository (`L:513`, re-checked) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| HK-9 | A | DONE | inherited done | No `ak-*` folder beside the repository (`L:513`, re-checked) | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| HK-10 | A | SUPERSEDED | superseded by cited decision/work | M-2 | Retain source ruling; verify cited replacement in final reconciliation. | +| HK-11 | A | issue stuinfla/ruvnet-brain#335 | external open issue | stuinfla/ruvnet-brain#335 | No local uninstall action. | +| HK-12 | A | DONE | inherited done | alpha.58 (`f3106a8ce53d5206898e8324cce9ba7c741f75c3`), alpha.59 (`b84b5a7ec8948daf21c04f7a4e91b9f9fe92e727`); historical commit f3106a8ce53d5206898e8324cce9ba7c741f75c3; historical commit b84b5a7ec8948daf21c04f7a4e91b9f9fe92e727 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| HK-13 | A | DONE | inherited done | #258 (`a77bfdee3e3948fa1812ac1ce9199be43e0cc8e1`); historical commit a77bfdee3e3948fa1812ac1ce9199be43e0cc8e1 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| HK-14 | A | DONE | inherited done | Probe repository returns 404; ruleset `24138491` active | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| HK-15 | A | D-8 | answered/deferred design decision | program v2 Appendix A row HK-15 | D-8 decision source and issue criteria must be cited at final V7 review. | +| DG-1 | A | D-10 … D-15 | answered/deferred design decision | program v2 Appendix A row DG-1 | Retain approved D-series decision and named future gate; no v2 code claim. | +| DG-2 | A | D-10 … D-15 | answered/deferred design decision | program v2 Appendix A row DG-2 | Retain approved D-series decision and named future gate; no v2 code claim. | +| DG-3 | A | D-10 … D-15 | answered/deferred design decision | program v2 Appendix A row DG-3 | Retain approved D-series decision and named future gate; no v2 code claim. | +| DG-4 | A | D-10 … D-15 | answered/deferred design decision | program v2 Appendix A row DG-4 | Retain approved D-series decision and named future gate; no v2 code claim. | +| DG-5 | A | D-10 … D-15 | answered/deferred design decision | program v2 Appendix A row DG-5 | Retain approved D-series decision and named future gate; no v2 code claim. | +| DG-6 | A | D-10 … D-15 | answered/deferred design decision | program v2 Appendix A row DG-6 | Retain approved D-series decision and named future gate; no v2 code claim. | +| AU-1 | A | DONE | inherited done | #240 exists; agentic-qe#734 (retired in #258), stuinfla/ruvnet-brain#330 and #331, ruvnet/ruflo#3446 and agentic-qe#735 are in the registry; the #574 note is issuecomment-5863219357 (`L:229`); stages 4–5 shipped in #241 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| AU-2 | A | DONE | inherited done | #240 exists; agentic-qe#734 (retired in #258), stuinfla/ruvnet-brain#330 and #331, ruvnet/ruflo#3446 and agentic-qe#735 are in the registry; the #574 note is issuecomment-5863219357 (`L:229`); stages 4–5 shipped in #241 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| AU-3 | A | DONE | inherited done | #240 exists; agentic-qe#734 (retired in #258), stuinfla/ruvnet-brain#330 and #331, ruvnet/ruflo#3446 and agentic-qe#735 are in the registry; the #574 note is issuecomment-5863219357 (`L:229`); stages 4–5 shipped in #241 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| AU-4 | A | DONE | inherited done | #240 exists; agentic-qe#734 (retired in #258), stuinfla/ruvnet-brain#330 and #331, ruvnet/ruflo#3446 and agentic-qe#735 are in the registry; the #574 note is issuecomment-5863219357 (`L:229`); stages 4–5 shipped in #241 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| AU-5 | A | DONE | inherited done | #240 exists; agentic-qe#734 (retired in #258), stuinfla/ruvnet-brain#330 and #331, ruvnet/ruflo#3446 and agentic-qe#735 are in the registry; the #574 note is issuecomment-5863219357 (`L:229`); stages 4–5 shipped in #241 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| AU-6 | A | DONE | inherited done | #240 exists; agentic-qe#734 (retired in #258), stuinfla/ruvnet-brain#330 and #331, ruvnet/ruflo#3446 and agentic-qe#735 are in the registry; the #574 note is issuecomment-5863219357 (`L:229`); stages 4–5 shipped in #241 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| AU-7 | A | DONE | inherited done | #240 exists; agentic-qe#734 (retired in #258), stuinfla/ruvnet-brain#330 and #331, ruvnet/ruflo#3446 and agentic-qe#735 are in the registry; the #574 note is issuecomment-5863219357 (`L:229`); stages 4–5 shipped in #241 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| AU-8 | A | DONE | inherited done | #240 exists; agentic-qe#734 (retired in #258), stuinfla/ruvnet-brain#330 and #331, ruvnet/ruflo#3446 and agentic-qe#735 are in the registry; the #574 note is issuecomment-5863219357 (`L:229`); stages 4–5 shipped in #241 | Confirm current behavior through final integration gates; no new ancestry claim. Historical ancestry confirms provenance, not current behavior; final integration gates remain. | +| AU-9 | A | DECLINED | declined by prior ruling | Paid qe-court live test stays manual | Retain original ruling and citation; no implementation. | +| AU-10 | A | SUPERSEDED | superseded by cited decision/work | Decision 15 | Retain source ruling; verify cited replacement in final reconciliation. | +| HO-1 | A | SUPERSEDED | superseded by cited decision/work | #253 merged; the hold lifted for N-5 only | Retain source ruling; verify cited replacement in final reconciliation. | +| DoD-1 | A | v2 DoD | operational definition-of-done gate | 1, 2, 6, 8, 7, and §5 with DoD 1; program v2 §6 | V6 integrated; V7 integration, final gates and main PR/human approval remain. External main watcher reconciliation is recorded separately. | +| DoD-2 | A | v2 DoD | operational definition-of-done gate | 1, 2, 6, 8, 7, and §5 with DoD 1; program v2 §6 | Source DoD-2 maps to program §6 item 2: all 246 rows need final citation/disposition after V6 and V7. | +| DoD-3 | A | v2 DoD | operational definition-of-done gate | 1, 2, 6, 8, 7, and §5 with DoD 1; program v2 §6 | Source DoD-3 maps to program §6 item 6: upstream watch live/quiet, #240 disposition and first real release dispatch remain observation-gated. | +| DoD-4 | A | v2 DoD | operational definition-of-done gate | 1, 2, 6, 8, 7, and §5 with DoD 1; program v2 §6 | Program records and archive reconciliation remain. Attended time was not instrumented; commit/run elapsed times cannot establish it. Personal Codex memory writes are outside this approval. | +| DoD-5 | A | v2 DoD | operational definition-of-done gate | 1, 2, 6, 8, 7, and §5 with DoD 1; program v2 §6 | Source DoD-5 maps to program §6 item 7: retained worktree/branch cleanup is an approved post-main operation, not done now. | +| DoD-6 | A | v2 DoD | operational definition-of-done gate | 1, 2, 6, 8, 7, and §5 with DoD 1; program v2 §6 | Source DoD-6 maps to program §5 plus DoD-1: full branch gates, integration review, final open main PR and human approval remain. | +| AB-row-1 | B | V1 | V1 integrated; 20-run evidence captured; issue publication pending | #262, Windows CI slowdown — V1 squash ab2fc5cb5104c2cf260fb9e3ff1721bd3b074202; [Windows receipt](2026-09-29-windows-ci-evidence.md) | Issue comment/closure require controller reconciliation. Current three-PR snapshot fails: Windows Node 24 took 358 seconds. Refresh after closeout PR CI. Observational medians meet the issue threshold; failed runs and a 359 s job remain, so no every-job-under-five-minutes claim. No #262 update posted from this draft. | +| AB-row-2 | B | V2 | V2 inherited documentation scope integrated on main | The docs taxonomy plan — D-3; [inherited integration receipt](2026-09-29-remediation-v2-integration-evidence.md): merged PR #264 e7cfe9ca4d49cb5e94fba698cd0e8376e0c829b7; PR #266 7936ca48942e4ed1b629641eb61dc0dfbcf12951; taxonomy/plan archive PR #269 159c1c98d86d5fbb22b5752413f5889d8b8d172b; all recorded ancestors of main@94890a00ec6f5869c186d1066076c5a2e17dfee4 | Retain final integration/docs checks before V7 signoff. Inherited receipt proves merged ancestry and titles, not current behavior or an exact per-file allocation among PRs. | +| AB-N-1 | B | V4 | implemented on develop via V4 | N-1, N-2, N-3, N-5 — Items C.1 to C.4; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N-2 | B | V4 | implemented on develop via V4 | N-1, N-2, N-3, N-5 — Items C.1 to C.4; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N-3 | B | V4 | implemented on develop via V4 | N-1, N-2, N-3, N-5 — Items C.1 to C.4; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N-5 | B | V4 | implemented on develop via V4 | N-1, N-2, N-3, N-5 — Items C.1 to C.4; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-row-4 | B | DONE | inherited done | N-4 — #261 (`0fa5e4890aabd830ee9d41878dd4e41bc85d3c73`) | Confirm current behavior through final integration gates; no new ancestry claim. | +| AB-N5-M7 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-M8 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-1 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-2 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-3 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-4 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-5 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-6 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-7 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-8 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-9 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-10 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-11 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-N5-minor-12 | B | V4 | implemented on develop via V4 | N-5 minors M7, M8 and 1–12 (14 items) — C4; each is listed in `reports/n5-253-deferred-minors.md` §2 and was present on `main`; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-row-6 | B | DECLINED | declined by prior ruling | N-5 minor M10 — #253's final review: "None needed"; the doc matches the code | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-6b-parked-1 | B | V4 | implemented on develop via V4 | 6b parked items 1, 2, 3 and m-4 — Items A.3, A.1, A.2; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-6b-parked-2 | B | V4 | implemented on develop via V4 | 6b parked items 1, 2, 3 and m-4 — Items A.3, A.1, A.2; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-6b-parked-3 | B | V4 | implemented on develop via V4 | 6b parked items 1, 2, 3 and m-4 — Items A.3, A.1, A.2; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-6b-m-4 | B | V4 | implemented on develop via V4 | 6b parked items 1, 2, 3 and m-4 — Items A.3, A.1, A.2; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-row-8 | B | V4 | implemented on develop via V4 | 6b re-review Minor (no test pins the injected path) — Item A.4; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-models-unknown-verb | B | V4 | implemented on develop via V4 | 6b declined-to-judge: `ak models` exits 0 on an unknown verb; stray positional on plain `ak status` — Item A.1; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-status-stray-positional | B | V4 | implemented on develop via V4 | 6b declined-to-judge: `ak models` exits 0 on an unknown verb; stray positional on plain `ak status` — Item A.1; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-row-10 | B | V3 | implemented on develop via V3 | 6b declined-to-judge: the dashboard noun left in README and DASHBOARD — 6c-4; PR #276; squash eb963f503f806f80df7a8173d7c71314adbd7cfa | Final main PR/review remains. | +| AB-row-11 | B | V4 | implemented on develop via V4 | 6b declined-to-judge: the newest in-window re-check — Item C.6 (B6b-25); PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-row-12 | B | DONE | inherited done | 6b declined-to-judge: Windows behavior — #263's Windows CI legs green | Confirm current behavior through final integration gates; no new ancestry claim. | +| AB-parallel-named-proofs | B | DECLINED | declined by prior ruling | 6b declined-to-judge: the other 9 (parallel named proofs, live summary on system and maintain, the live-check label, no ticker for fast stages, concurrent CLI and dashboard scans, `--refresh=local` undocumented, maintenance writes under `--refresh`, the not-scanned step text, `check-connection --dry-run`'s local checks) — The reviewer found each by design, specified by the plan, or pre-existing and harmless | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-live-summary-system-maintain | B | DECLINED | declined by prior ruling | 6b declined-to-judge: the other 9 (parallel named proofs, live summary on system and maintain, the live-check label, no ticker for fast stages, concurrent CLI and dashboard scans, `--refresh=local` undocumented, maintenance writes under `--refresh`, the not-scanned step text, `check-connection --dry-run`'s local checks) — The reviewer found each by design, specified by the plan, or pre-existing and harmless | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-live-check-label | B | DECLINED | declined by prior ruling | 6b declined-to-judge: the other 9 (parallel named proofs, live summary on system and maintain, the live-check label, no ticker for fast stages, concurrent CLI and dashboard scans, `--refresh=local` undocumented, maintenance writes under `--refresh`, the not-scanned step text, `check-connection --dry-run`'s local checks) — The reviewer found each by design, specified by the plan, or pre-existing and harmless | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-fast-stage-ticker | B | DECLINED | declined by prior ruling | 6b declined-to-judge: the other 9 (parallel named proofs, live summary on system and maintain, the live-check label, no ticker for fast stages, concurrent CLI and dashboard scans, `--refresh=local` undocumented, maintenance writes under `--refresh`, the not-scanned step text, `check-connection --dry-run`'s local checks) — The reviewer found each by design, specified by the plan, or pre-existing and harmless | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-concurrent-cli-dashboard | B | DECLINED | declined by prior ruling | 6b declined-to-judge: the other 9 (parallel named proofs, live summary on system and maintain, the live-check label, no ticker for fast stages, concurrent CLI and dashboard scans, `--refresh=local` undocumented, maintenance writes under `--refresh`, the not-scanned step text, `check-connection --dry-run`'s local checks) — The reviewer found each by design, specified by the plan, or pre-existing and harmless | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-refresh-local-docs | B | DECLINED | declined by prior ruling | 6b declined-to-judge: the other 9 (parallel named proofs, live summary on system and maintain, the live-check label, no ticker for fast stages, concurrent CLI and dashboard scans, `--refresh=local` undocumented, maintenance writes under `--refresh`, the not-scanned step text, `check-connection --dry-run`'s local checks) — The reviewer found each by design, specified by the plan, or pre-existing and harmless | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-maintenance-writes-refresh | B | DECLINED | declined by prior ruling | 6b declined-to-judge: the other 9 (parallel named proofs, live summary on system and maintain, the live-check label, no ticker for fast stages, concurrent CLI and dashboard scans, `--refresh=local` undocumented, maintenance writes under `--refresh`, the not-scanned step text, `check-connection --dry-run`'s local checks) — The reviewer found each by design, specified by the plan, or pre-existing and harmless | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-not-scanned-text | B | DECLINED | declined by prior ruling | 6b declined-to-judge: the other 9 (parallel named proofs, live summary on system and maintain, the live-check label, no ticker for fast stages, concurrent CLI and dashboard scans, `--refresh=local` undocumented, maintenance writes under `--refresh`, the not-scanned step text, `check-connection --dry-run`'s local checks) — The reviewer found each by design, specified by the plan, or pre-existing and harmless | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-check-connection-dry-run | B | DECLINED | declined by prior ruling | 6b declined-to-judge: the other 9 (parallel named proofs, live summary on system and maintain, the live-check label, no ticker for fast stages, concurrent CLI and dashboard scans, `--refresh=local` undocumented, maintenance writes under `--refresh`, the not-scanned step text, `check-connection --dry-run`'s local checks) — The reviewer found each by design, specified by the plan, or pre-existing and harmless | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-row-14 | B | V4 | implemented on develop via V4 | Branch 9 Task 1's parked `pause()` summary — Item B.8 (same as B9-2); PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Final main PR/review remains. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | +| AB-row-15 | B | V4, V5 | V4/V5 code integrated on develop | Branch 9 declined-to-judge: Tasks 5–14 and N-1 to N-5 — Mapped above; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e; PR #274; squash af825c9f165575b3ef6e1ce0a5836fcf40a1ca3b | Map each underlying Task 5–14 and N item to its exact PR during final V7 review. Source is grouped; per-task mapping is not present in this expanded row. | +| AB-row-16 | B | DONE | inherited done | Branch 9 declined-to-judge: Windows behavior (EBUSY, junctions) — #259's Windows CI legs green | Confirm current behavior through final integration gates; no new ancestry claim. | +| AB-row-17 | B | SUPERSEDED | superseded by cited decision/work | Branch 9 declined-to-judge: 6b's uncommitted `DASHBOARD.md` hunks — #263 merged | Retain source ruling; verify cited replacement in final reconciliation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-row-18 | B | V6 | integrated on develop via V6 | Branch 9 declined-to-judge: census numbers not re-run — Item 4 re-measures; [V6 integration receipt](2026-09-29-remediation-v2-integration-evidence.md) | Closeout integration gates and final main PR/human review remain. Aggregate PR/source-plan evidence; no invented fresh per-item runtime proof. | +| AB-history-coverage-read | B | DECLINED | declined by prior ruling | Branch 9 declined-to-judge: the other 9 (history read per `coverage()`, non-canonical copies kept, trailing newline, `sudo` uid, `env` key position, commented TOML line, CRLF fixtures, `UPGRADING.md` date order, copies of a released file) — The reviewer found each safe by design or out of scope | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-noncanonical-copies | B | DECLINED | declined by prior ruling | Branch 9 declined-to-judge: the other 9 (history read per `coverage()`, non-canonical copies kept, trailing newline, `sudo` uid, `env` key position, commented TOML line, CRLF fixtures, `UPGRADING.md` date order, copies of a released file) — The reviewer found each safe by design or out of scope | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-trailing-newline | B | DECLINED | declined by prior ruling | Branch 9 declined-to-judge: the other 9 (history read per `coverage()`, non-canonical copies kept, trailing newline, `sudo` uid, `env` key position, commented TOML line, CRLF fixtures, `UPGRADING.md` date order, copies of a released file) — The reviewer found each safe by design or out of scope | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-sudo-uid | B | DECLINED | declined by prior ruling | Branch 9 declined-to-judge: the other 9 (history read per `coverage()`, non-canonical copies kept, trailing newline, `sudo` uid, `env` key position, commented TOML line, CRLF fixtures, `UPGRADING.md` date order, copies of a released file) — The reviewer found each safe by design or out of scope | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-env-key-position | B | DECLINED | declined by prior ruling | Branch 9 declined-to-judge: the other 9 (history read per `coverage()`, non-canonical copies kept, trailing newline, `sudo` uid, `env` key position, commented TOML line, CRLF fixtures, `UPGRADING.md` date order, copies of a released file) — The reviewer found each safe by design or out of scope | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-commented-toml | B | DECLINED | declined by prior ruling | Branch 9 declined-to-judge: the other 9 (history read per `coverage()`, non-canonical copies kept, trailing newline, `sudo` uid, `env` key position, commented TOML line, CRLF fixtures, `UPGRADING.md` date order, copies of a released file) — The reviewer found each safe by design or out of scope | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-crlf-fixtures | B | DECLINED | declined by prior ruling | Branch 9 declined-to-judge: the other 9 (history read per `coverage()`, non-canonical copies kept, trailing newline, `sudo` uid, `env` key position, commented TOML line, CRLF fixtures, `UPGRADING.md` date order, copies of a released file) — The reviewer found each safe by design or out of scope | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-upgrading-date-order | B | DECLINED | declined by prior ruling | Branch 9 declined-to-judge: the other 9 (history read per `coverage()`, non-canonical copies kept, trailing newline, `sudo` uid, `env` key position, commented TOML line, CRLF fixtures, `UPGRADING.md` date order, copies of a released file) — The reviewer found each safe by design or out of scope | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-released-file-copies | B | DECLINED | declined by prior ruling | Branch 9 declined-to-judge: the other 9 (history read per `coverage()`, non-canonical copies kept, trailing newline, `sudo` uid, `env` key position, commented TOML line, CRLF fixtures, `UPGRADING.md` date order, copies of a released file) — The reviewer found each safe by design or out of scope | Retain original ruling and citation; no implementation. Only original grouped review ruling cited; no fresh per-item verification claimed. | +| AB-row-20 | B | DONE | inherited done | EPIPE in `runWithInput` (seen on #261's CI) — #260 (`ad0170d97a2427c1fb79f179aa8632d13d372238`) | Confirm current behavior through final integration gates; no new ancestry claim. | +| AB-row-21 | B | V4 | conditional V4 pin item | AQE pin not planned by sync (conditional) — Item B.13, only if §2 step 2 shows it; PR #275; squash bb2e7efe88abd3cb2d73baf26530386549e3dc2e | Recheck §2 pin result; only apply if pin absent. No unconditional sync pin claim. V4 mapping is source-plan item/group-level; final reviewer should resolve exact unit evidence if required. | diff --git a/docs/archive/2026-09-29-v6-codex-thread-source-evidence.md b/docs/archive/2026-09-29-v6-codex-thread-source-evidence.md new file mode 100644 index 00000000..9f8f9e5e --- /dev/null +++ b/docs/archive/2026-09-29-v6-codex-thread-source-evidence.md @@ -0,0 +1,31 @@ +# V6 Unit 3: Codex thread sources and Auto-review pricing + +Source state: `bf8babd4` plus this unit's changes. Checked 2026-09-29 10:57 UTC with `codex-cli 0.158.0`. + +## Findings and change + +- **Verified kit defect:** an unfamiliar `thread_source` on a known interactive originator inherited `person`. The classifier now reports `unknown`, while recognized user, handoff, agent, and automation values retain their declared initiators. The fixed SDK, exec, and MCP rules still apply. +- **Verified kit defect:** SQLite ledger backfill changed `threadSource` without rebuilding `sessionOrigin`. It now classifies from bounded raw declaration evidence and preserves the first rollout declaration over a later replayed parent declaration. +- **Verified kit gap:** the observed `source.subagent.thread_spawn.parent_thread_id` was not retained. The first metadata line now accepts only a UUID-shaped parent ID. Aggregate rollup uses a parent present in the same Codex record set, rejects missing links and cycles, and groups child and reviewer records under that parent's surface. It keeps child-owned usage and strips only a ledger-identified subagent whose rollout could not separate replay. +- **Verified kit defect:** `codex-auto-review` token rows were assigned unknown-model fallback dollars. This exact server-side alias now contributes tokens and unpriced-message coverage, with zero estimated dollars and no invented cache saving. Published known-model pricing is unchanged. + +## Bounded local metadata probe + +Read only the first JSONL line of 1,806 local Codex rollout files, plus `turn_context.model` for guardian-review files. This is a file census, not a unique-session census or a billing statement. The `thread_source` counts were: `subagent` 510, `user` 200, absent/null 994, `guardian_review` 90, `chatgpt_handoff` 10, `agent_created_thread` 2. None had a structured `thread_source`; 600 had structured `source.subagent` (500 `thread_spawn`, 90 `other` with guardian review, 10 `other` with subagent). The 500 observed `thread_spawn` parent IDs were UUID-shaped. Guardian-review files contained 843 `turn_context` declarations of `codex-auto-review`. The probe did not read prompt text, deduplicate sessions, inspect imports, or verify a published price. + +## Verification + +- RED: new focused tests failed on unfamiliar source classification, parent extraction, ledger origin, and Auto-review fallback pricing. A separate cycle fixture failed before the cycle guard. +- GREEN: `node scripts/run-tests.mjs exec -- --test tests/kit/usage-codex-attribution.test.mjs tests/kit/usage-session-surface.test.mjs tests/kit/session-surface.test.mjs tests/kit/usage-index-v6.test.mjs tests/kit/usage-codex-thread-source.test.mjs tests/kit/pricing.test.mjs` — 101 passed, 0 failed. +- `node node_modules/typescript/bin/tsc --noEmit` — passed. `node node_modules/eslint/bin/eslint.js` on the six changed source/test files — passed. `git diff --check` — passed. +- The synthetic aggregate fixture checks parent, child, and reviewer totals and cold/warm cache consistency without touching the real usage cache. + +## Accounting limits + +The 26 schema remains unchanged. Existing schema-26 caches created before this unit may lack parsed parent IDs until a fresh cache rebuild; the final pre-PR gate owns that rebuild. A reviewer without a verified parent remains on its declared surface because no parent can be inferred. No provider calls, full unit/UI suite, real cache migration, or user database writes were made. + +## Independent review fix round 1 — 2026-09-29 11:05 UTC + +- **P2 confirmed:** a present `thread_source` rejected by the bounded token parser (`{}` or a string containing spaces) became `null`, allowing a known interactive originator's `person` default. The classifier now distinguishes an absent declaration from a rejected one without retaining rejected content. Focused fixtures cover direct classification, the first rollout metadata line, ledger overlay, and the SDK, exec, and MCP fixed initiators. +- **P3 confirmed:** ledger backfill called `classifySessionSurface` without the imported-copy flag and could replace an imported copy's `initiator` with `agent`. `ledgerOrigin` now preserves the parser's imported-copy override. The regression fixture uses a real minimal imported rollout and a synthetic guardian-review ledger row; the exported helper remains safe even though `buildIndex` filters imported records before aggregation. +- RED: both new defect fixtures failed against `5fe016d1`. GREEN: the same six focused test files listed above passed, **103 tests, 0 failures**. TypeScript `--noEmit`, targeted ESLint on the three changed source/test files, and `git diff --check` passed. No full unit/UI run or local corpus repeat was performed. diff --git a/docs/archive/2026-09-29-windows-ci-evidence.md b/docs/archive/2026-09-29-windows-ci-evidence.md new file mode 100644 index 00000000..ff248962 --- /dev/null +++ b/docs/archive/2026-09-29-windows-ci-evidence.md @@ -0,0 +1,68 @@ +# Windows CI historical cohort and current timing gate + +Capture: 2026-09-29; compiled from retained GitHub job metadata, without new CI runs. +Documentation basis: `b5a0a946ae0aca2e1d435c78b12a98bc587d1266`. + +## Fixed historical cohort + +Preserve the ten before runs and the ten after runs, ending at run **36582013896**. These are fixed historical observations, not the current/latest ten runs. The complete observations appear below; retained controller JSON binds workload inputs and ancestry checks. + +| Cohort | Node | All-job median s | Noncancelled median s | Successful-job median s | Success / failure / cancelled | +| --- | --- | ---: | ---: | ---: | --- | +| Before | 22 | 649 | 659 | 659 | 9 / 0 / 1 | +| Before | 24 | 705.5 | 723 | 705.5 | 8 / 1 / 1 | +| Before | 26 | 875 | 888 | 888 | 9 / 0 / 1 | +| Fixed after | 22 | 221.5 | 221.5 | 240 | 7 / 3 / 0 | +| Fixed after | 24 | 225 | 225 | 215 | 7 / 3 / 0 | +| Fixed after | 26 | 210 | 210 | 210 | 7 / 3 / 0 | + +All-job medians in the earlier private draft are correct. Its “Median completed jobs” column mixes definitions: before excludes cancelled jobs but includes failure; after excludes failures. Rename/split the column as above. Cancelled observations are partial elapsed time, not completed workload performance. The after cohort includes three failed workflows and maximum Windows duration 359 s; preserve both. The earlier mixed after cohort containing pre-V1 run 36519742714 remains provenance, not a pure post-V1 comparison. Different source/workload/runner states prevent a controlled causal speedup claim. + +All ten fixed after heads independently passed local `git merge-base --is-ancestor ab2fc5cb5104c2cf260fb9e3ff1721bd3b074202 ` (exit 0). No object fetch was needed. + +## Separate current PR timing gate snapshot + +The approved source plan `docs/plans/2026-09-28-remediation-program-v2.md:261` says “three consecutive PR runs with every Windows leg under 5 minutes.” Selected the newest three distinct completed `pull_request` runs for `.github/workflows/ci.yml` in creation order, with no outcome or branch filtering. Latest attempt per run; all three are attempt 1. Snapshot: 2026-09-29T19:20:24Z, from a repository-wide inventory of 100 entries; full head hashes appear below. + +| Run (newest first) | Head | Windows 22 s | Windows 24 s | Windows 26 s | Workflow | +| --- | --- | ---: | ---: | ---: | --- | +| [36616775739](https://github.com/pacphi/agentic-kit/actions/runs/36616775739) | 09b198b6960c2dae14819d08729caa8b3794129b | 245 | 227 | 254 | success | +| [36614260309](https://github.com/pacphi/agentic-kit/actions/runs/36614260309) | fa0cb7b788791a90577a2dfd9f9ca61e07d0591e | 239 | 233 | 262 | success | +| [36612394617](https://github.com/pacphi/agentic-kit/actions/runs/36612394617) | 87234bdb2499200ce3fbe280894f6bb7ace74cde | 187 | **358** | 241 | **failure** | + +All nine Windows jobs completed successfully; all three heads pass the same V1 ancestry check. The third run failed on `test (macos-latest, node 22)`. Its Windows Node 24 duration independently violates the literal timing gate. Therefore this snapshot supports **two consecutive qualifying PR runs, not three**, and **two green workflows, not three**. Keep the failed third run. Refresh this separate gate after V7 PR CI; do not replace historical cohort membership. + +Raw per-run metadata and complete latest-attempt job lists are saved as `-run.json` and `-jobs.json`; total_count matched fetched job count for each. Job records bind URLs, start/end timestamps, duration, status, conclusion, attempt and head. Durations exclude queue time. + +API caveat: querying `actions/workflows/ci.yml/runs` returned a stale September 28 subset. The repository-wide `actions/runs?event=pull_request&status=completed&per_page=100` inventory correctly includes September 29 runs with workflow ID 313185208 and exact path `.github/workflows/ci.yml`; use that inventory plus attempt-specific job endpoints for refresh. Do not infer PR identity from the empty `pull_requests` arrays. + +## Complete fixed observations + +Cells retain seconds and conclusion for each Windows Node job. Full heads bind source; +job links retain timestamps and runner metadata. Cancelled values are partial elapsed times. + +| Cohort | Run | Source head | Node 22 | Node 24 | Node 26 | +| --- | --- | --- | --- | --- | --- | +| before | [36512746752](https://github.com/pacphi/agentic-kit/actions/runs/36512746752) | `ec7497179a2a6321ce2906b4346d5720ba50cba2` | [832 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36512746752/job/109228289572) | [588 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36512746752/job/109228289560) | [695 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36512746752/job/109228289704) | +| before | [36512738230](https://github.com/pacphi/agentic-kit/actions/runs/36512738230) | `bd8335748fbf117147a930322d3cc032009dfce2` | [604 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36512738230/job/109228262744) | [752 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36512738230/job/109228262957) | [968 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36512738230/job/109228263062) | +| before | [36512520521](https://github.com/pacphi/agentic-kit/actions/runs/36512520521) | `7936ca48942e4ed1b629641eb61dc0dfbcf12951` | [639 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36512520521/job/109227580342) | [591 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36512520521/job/109227580275) | [1027 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36512520521/job/109227580466) | +| before | [36507895935](https://github.com/pacphi/agentic-kit/actions/runs/36507895935) | `4d654d81dbaf59b475ff86d77e7a7ed3d10d4caa` | [755 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36507895935/job/109213354743) | [688 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36507895935/job/109213354777) | [776 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36507895935/job/109213354580) | +| before | [36507430463](https://github.com/pacphi/agentic-kit/actions/runs/36507430463) | `4f4f0201527952a99e4d5c165dd37c6a94cbc334` | [469 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36507430463/job/109211906632) | [512 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36507430463/job/109211906387) | [862 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36507430463/job/109211906551) | +| before | [36507067003](https://github.com/pacphi/agentic-kit/actions/runs/36507067003) | `174e5032e2dc36b8280941bc98593da18280e234` | [187 s, cancelled](https://github.com/pacphi/agentic-kit/actions/runs/36507067003/job/109210765385) | [157 s, cancelled](https://github.com/pacphi/agentic-kit/actions/runs/36507067003/job/109210765383) | [149 s, cancelled](https://github.com/pacphi/agentic-kit/actions/runs/36507067003/job/109210765403) | +| before | [36507043837](https://github.com/pacphi/agentic-kit/actions/runs/36507043837) | `b3761ba62f8b6253b3908a25c677a4ebbb565663` | [711 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36507043837/job/109210682350) | [723 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36507043837/job/109210682327) | [525 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36507043837/job/109210682409) | +| before | [36506687833](https://github.com/pacphi/agentic-kit/actions/runs/36506687833) | `2992c93c92ce1d8ea84f26283d6ccac2d146c8ee` | [607 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36506687833/job/109209580335) | [743 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36506687833/job/109209580181) | [1044 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36506687833/job/109209580209) | +| before | [36506455012](https://github.com/pacphi/agentic-kit/actions/runs/36506455012) | `e7cfe9ca4d49cb5e94fba698cd0e8376e0c829b7` | [659 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36506455012/job/109208858912) | [1413 s, failure](https://github.com/pacphi/agentic-kit/actions/runs/36506455012/job/109208858930) | [888 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36506455012/job/109208858967) | +| before | [36505126704](https://github.com/pacphi/agentic-kit/actions/runs/36505126704) | `c959db36a9c01e0a6454fb3386747bbe908971ea` | [795 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36505126704/job/109204728092) | [1243 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36505126704/job/109204728189) | [1088 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36505126704/job/109204728047) | +| after | [36582013896](https://github.com/pacphi/agentic-kit/actions/runs/36582013896) | `bb2e7efe88abd3cb2d73baf26530386549e3dc2e` | [247 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36582013896/job/109452341642) | [235 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36582013896/job/109452341600) | [214 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36582013896/job/109452341616) | +| after | [36581206489](https://github.com/pacphi/agentic-kit/actions/runs/36581206489) | `f19566adf2d0f0d7027bc4ea45f9f990e2b0355e` | [247 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36581206489/job/109449555718) | [174 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36581206489/job/109449556023) | [210 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36581206489/job/109449555782) | +| after | [36578593054](https://github.com/pacphi/agentic-kit/actions/runs/36578593054) | `989c5e563c8aa0527b1497fb64643f8277664c0a` | [214 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36578593054/job/109440525366) | [215 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36578593054/job/109440525563) | [200 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36578593054/job/109440525398) | +| after | [36577987831](https://github.com/pacphi/agentic-kit/actions/runs/36577987831) | `e8751224857f1301a350c8fa6e51dfc999c81206` | [225 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36577987831/job/109438467919) | [209 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36577987831/job/109438468373) | [222 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36577987831/job/109438468285) | +| after | [36573303236](https://github.com/pacphi/agentic-kit/actions/runs/36573303236) | `8430757488d1c154611d0f857d6c32db73e6fcca` | [240 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36573303236/job/109422417956) | [241 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36573303236/job/109422417926) | [220 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36573303236/job/109422417766) | +| after | [36573174471](https://github.com/pacphi/agentic-kit/actions/runs/36573174471) | `00a6faaa2622396309a231ef794628c856fc1071` | [209 s, failure](https://github.com/pacphi/agentic-kit/actions/runs/36573174471/job/109421970920) | [355 s, failure](https://github.com/pacphi/agentic-kit/actions/runs/36573174471/job/109421971175) | [184 s, failure](https://github.com/pacphi/agentic-kit/actions/runs/36573174471/job/109421971019) | +| after | [36572475641](https://github.com/pacphi/agentic-kit/actions/runs/36572475641) | `f440c1ec2e055f633438ea329ce629156439b3ba` | [214 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36572475641/job/109419613374) | [292 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36572475641/job/109419613443) | [195 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36572475641/job/109419613517) | +| after | [36569982241](https://github.com/pacphi/agentic-kit/actions/runs/36569982241) | `087a67154dccba79399900545574c72c86815041` | [218 s, failure](https://github.com/pacphi/agentic-kit/actions/runs/36569982241/job/109411273607) | [165 s, failure](https://github.com/pacphi/agentic-kit/actions/runs/36569982241/job/109411274140) | [187 s, failure](https://github.com/pacphi/agentic-kit/actions/runs/36569982241/job/109411274198) | +| after | [36569976640](https://github.com/pacphi/agentic-kit/actions/runs/36569976640) | `886b565dd5304f485af4c079c0c7c06aa83a13cd` | [283 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36569976640/job/109411256908) | [195 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36569976640/job/109411257316) | [210 s, success](https://github.com/pacphi/agentic-kit/actions/runs/36569976640/job/109411257237) | +| after | [36567912820](https://github.com/pacphi/agentic-kit/actions/runs/36567912820) | `ed8f4cb96836e1911aae9a89bbb3f015f033e115` | [197 s, failure](https://github.com/pacphi/agentic-kit/actions/runs/36567912820/job/109404334440) | [359 s, failure](https://github.com/pacphi/agentic-kit/actions/runs/36567912820/job/109404334368) | [231 s, failure](https://github.com/pacphi/agentic-kit/actions/runs/36567912820/job/109404334534) | + +Issue publication remains pending controller review; no posted table or closure is proved. +[Issue #262](https://github.com/pacphi/agentic-kit/issues/262) remains open. diff --git a/docs/archive/README.md b/docs/archive/README.md index d3a40b56..5512a1fd 100644 --- a/docs/archive/README.md +++ b/docs/archive/README.md @@ -68,6 +68,13 @@ reconfirmed by this metadata audit. The per-file inventory and limitations are r | File | Original location | What it was | Why it's historical | |---|---|---|---| +| [2026-09-28-plan-follow-ups-v2.md](2026-09-28-plan-follow-ups-v2.md) | `docs/plans/2026-09-28-follow-ups-v2.md` | V4 product, CLI, memory, process-lifecycle and upstream integration follow-ups. | All eight local gates and independent whole-branch review passed at `29152654`; final-head PR CI and squash integration were pending at archival. Conditional Ruflo #3419 guidance remains deferred; native Windows AQE was not tested. | +| [2026-09-29-plan-main-watch-reconciliation.md](2026-09-29-plan-main-watch-reconciliation.md) | `docs/plans/2026-09-29-main-watch-reconciliation.md` | Completed main #280 and develop watcher reconciliation | Both contracts preserved; bounded documented retry, sanitized diagnostics and faithful deferred preview. Independent review and all eight local gates passed through `3bbee599`; PR CI and squash integration remained pending at archival. Current contract: [Upstream watch](../upstream-watch.md). | +| [2026-09-28-plan-usage-accuracy.md](2026-09-28-plan-usage-accuracy.md) | `docs/plans/2026-09-28-usage-accuracy.md` | Completed V6 usage and session evidence plan | All 24 units independently accepted; whole-branch review and all eight local gates passed through `1c02db91`. Feature PR CI and develop integration remain separate gates at archival. Current contracts: [Usage metrics](../usage-scorecard-metrics.md) and [ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md). | +| [2026-09-29-plan-v6-surfaces-ui.md](2026-09-29-plan-v6-surfaces-ui.md) | `docs/plans/2026-09-29-v6-surfaces-ui.md` | Completed V6 session presentation plan | Shared vocabulary, legacy filter compatibility, independent evidence dimensions and source-coverage disclosures; included in the V6 gates at `1c02db91`. | +| [2026-09-29-plan-v6-opencode-cost.md](2026-09-29-plan-v6-opencode-cost.md) | `docs/plans/2026-09-29-v6-opencode-cost.md` | Completed OpenCode reported-zero trust plan | Scoped cost work and mandatory core cache handoff accepted; versioned observation semantics are documented in the living usage guide. | +| [2026-09-29-v6-codex-thread-source-evidence.md](2026-09-29-v6-codex-thread-source-evidence.md) | `.superpowers/sdd/2026-09-28-usage-accuracy/task3-report.md` | Historical V6 Unit 3 evidence | Original bounded metadata census and initial/fix tests retained verbatim. Its interim cache note describes that capture stage, not a live-cache mutation or current release claim. | +| [2026-09-29-plan-upstream-watch-followups.md](2026-09-29-plan-upstream-watch-followups.md) | `docs/plans/2026-09-29-upstream-watch-followups.md` | V4 C4 execution plan for bounded PR polling, deterministic retry failures and twelve deferred watcher minors. | Implementation and independent review complete; all eight local gates passed at `b75c1e3f`. Final PR CI and merge were pending at archival. Current contract: [Upstream watch](../upstream-watch.md). | | [2026-06-upstream-findings-f1-f6.md](2026-06-upstream-findings-f1-f6.md) | `docs/upstream/ruflo-self-improvement-findings.md` | The F1–F6 findings series: proofs/refutations of ruflo's self-improvement claims (Q-learning persistence, state-encoder collapse, SONA learn→inference wiring, native-training misreporting), with filed upstream issues. | Every finding is now fixed upstream: F2 in 3.10.6 ([#2222](https://github.com/ruvnet/ruflo/issues/2222)), F2b in 3.10.7, F3 in 3.10.11 ([#2239](https://github.com/ruvnet/ruflo/issues/2239)), F4 in `@ruvector/ruvllm` 2.5.6 ([RuVector#519](https://github.com/ruvnet/RuVector/issues/519)), F6 in 3.18.1/3.19.0 + ruvllm 2.5.7 ([#2549](https://github.com/ruvnet/ruflo/issues/2549), closed 2026-07-03). | | [2026-06-token-consumption-incident.md](2026-06-token-consumption-incident.md) | `docs/usage/token-consumption-findings-and-mitigation-2026-06.md` | Root-cause report for the June 2026 token-burn incident: six immortal auto-started daemons consumed ~8.1B tokens over 7 days via headless worker sessions. Produced the opt-in daemon policy, TTL reaper, ⚙ statusline alarm, and `ruflo-token-audit`. | The root cause was fixed upstream in ruflo 3.27/3.28 ([#2661](https://github.com/ruvnet/ruflo/issues/2661)): AI workers are opt-in, launches are governed by a machine-wide budget with telemetry, one supervisor daemon per repo, native daemon TTL. The kit's daemon policy flipped back to default-on (local-only workers) on that baseline; the reapers and token-audit remain as an independent check. | | [2026-06-11-token-consumption-recurrence.md](2026-06-11-token-consumption-recurrence.md) | `docs/usage/token-consumption-recurrence-and-cleanup-2026-06-11.md` | Follow-up audit 10 days later: 17 daemons had accumulated but the TTL auto-reaper had already contained them; cleanup of daemon-state files, logs, and two plugin MCP servers. | Same incident class as above — governed upstream since 3.27/3.28. Kept as evidence the TTL-reaper safety net worked. | @@ -154,12 +161,17 @@ reconfirmed by this metadata audit. The per-file inventory and limitations are r | [2026-09-27-superpowers-plan-branch-4b-upstream-watch-actions.md](2026-09-27-superpowers-plan-branch-4b-upstream-watch-actions.md) | `docs/superpowers/plans/2026-09-27-branch-4b-upstream-watch-actions.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | | [2026-09-27-superpowers-plan-branch-5-aqe-store-integrity.md](2026-09-27-superpowers-plan-branch-5-aqe-store-integrity.md) | `docs/superpowers/plans/2026-09-27-branch-5-aqe-store-integrity.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | | [2026-09-27-superpowers-plan-branch-6a-evidence-store.md](2026-09-27-superpowers-plan-branch-6a-evidence-store.md) | `docs/superpowers/plans/2026-09-27-branch-6a-evidence-store.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | +| [2026-09-28-plan-runner-hygiene.md](2026-09-28-plan-runner-hygiene.md) | `docs/plans/2026-09-28-runner-hygiene.md` | Completed V5 execution plan | Local implementation and independent review through `e0fcc2eb`; feature CI and integration tracked by the PR. Private backlog list is a manual review aid, not deletion authority. | +| [2026-09-28-research-test-temp-folder-cleanup.md](2026-09-28-research-test-temp-folder-cleanup.md) | `docs/plans/2026-09-28-test-temp-folder-cleanup-design.md` | Runner cleanup research | Source-bound creator census and experiments supporting list-only sibling handling; no complete native descendant proof or cleanup authority. | | [2026-09-28-superpowers-plan-branch-6b-one-refresh-flag.md](2026-09-28-superpowers-plan-branch-6b-one-refresh-flag.md) | `docs/superpowers/plans/2026-09-28-branch-6b-one-refresh-flag.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | | [2026-09-28-superpowers-plan-branch-9-follow-ups.md](2026-09-28-superpowers-plan-branch-9-follow-ups.md) | `docs/superpowers/plans/2026-09-28-branch-9-follow-ups.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | | [2026-09-28-superpowers-plan-upstream-watch-ledger-branch.md](2026-09-28-superpowers-plan-upstream-watch-ledger-branch.md) | `docs/superpowers/plans/2026-09-28-upstream-watch-ledger-branch.md` | Finished Superpowers plan | Implemented work record preserved after its implementing PR merged. | | [2026-09-28-superpowers-spec-upstream-watch-ledger-branch-design.md](2026-09-28-superpowers-spec-upstream-watch-ledger-branch-design.md) | `docs/superpowers/specs/2026-09-28-upstream-watch-ledger-branch-design.md` | Finished Superpowers specification | Implemented work record preserved after its implementing PR merged. | | [2026-09-28-plan-docs-taxonomy-and-archive.md](2026-09-28-plan-docs-taxonomy-and-archive.md) | `docs/plans/2026-09-28-docs-taxonomy-and-archive.md` | Finished documentation taxonomy and archive plan | Implemented in [PR #264](https://github.com/pacphi/agentic-kit/pull/264) and [PR #266](https://github.com/pacphi/agentic-kit/pull/266). Current layout rules live in the [docs index](../README.md) and [layout guard](../../scripts/docs-layout.mjs); this plan records the build steps. | | [2026-09-28-plan-sonnet-5-5-routing-refresh.md](2026-09-28-plan-sonnet-5-5-routing-refresh.md) | `docs/plans/2026-09-28-sonnet-5-5-routing-refresh.md` | Routing and pricing research with the implemented model-catalog decision | Implemented in [PR #268](https://github.com/pacphi/agentic-kit/pull/268). The operative tier decision is in [ADR-0006](../adr/0006-primary-host-and-ambidextrous-mirroring.md); prices and benchmarks here are dated evidence. | +| [2026-09-28-plan-dashboard-refresh.md](2026-09-28-plan-dashboard-refresh.md) | `docs/plans/2026-09-28-dashboard-refresh.md` | Completed V3 implementation plan | Scoped implementation and independent review through `d2c1833b`, including native plain-folder evidence and paused Activity timestamps. Full branch gates, PR CI and develop integration remain separate gates at capture. | +| [2026-09-29-plan-upstream-native-trace.md](2026-09-29-plan-upstream-native-trace.md) | `docs/plans/2026-09-29-upstream-native-trace.md` | Completed C3 instrumentation plan | Hook, workflow and native trace captured; final PR integration and exact upstream-post approval remain separate gates at capture. | +| [2026-09-29-native-learning-trace.md](2026-09-29-native-learning-trace.md) | — (new) | Hosted macOS package-resolution evidence | Binds the observed 3.8.1/1.21.0 roots and learning failure to one source/run; introduction and causality remain unproved. | ## Naming convention @@ -213,3 +225,15 @@ incident reports intentionally keep their original, now-dangling paths. | [2026-09-07-validation-maintenance-option-a.md](2026-09-07-validation-maintenance-option-a.md) | `docs/design/maintenance-overhaul/option-a-validation.md` | Maintenance overhaul option-a-validation.md | Dated verification evidence; does not prove the current source state or close open release gates. | | [2026-09-08-design-maintenance-project-metadata-adapters.md](2026-09-08-design-maintenance-project-metadata-adapters.md) | `docs/design/maintenance-overhaul/project-metadata-adapters.md` | Maintenance overhaul project-metadata-adapters.md | Live successor: [PROJECT-METADATA-ADAPTERS.md](../proposals/project-metadata-adapters.md); original snapshot retained as provenance. | | [2026-09-04-design-maintenance-overhaul-provider-and-action-policy.md](2026-09-04-design-maintenance-overhaul-provider-and-action-policy.md) | `docs/design/maintenance-overhaul/provider-and-action-policy.md` | Maintenance overhaul provider-and-action-policy.md | Frozen design snapshot; current behavior and remaining gates live in ADR-0048 and the maintained Maintenance guide. | + +## Added 2026-09-29 — remediation v2 evidence capture + +| File | Original location | What it was | Why it is historical | +| --- | --- | --- | --- | +| [2026-09-29-plan-remediation-v2-closeout.md](2026-09-29-plan-remediation-v2-closeout.md) | `docs/plans/2026-09-29-remediation-v2-closeout.md` | V7 closeout execution plan | Implementation and local gates completed; later PR, timing and human-review receipts remain separate. | +| [2026-09-29-aqe-released-artifact-receipt.md](2026-09-29-aqe-released-artifact-receipt.md) | Written here | AQE 3.14.5 released-artifact receipt (2026-09-29) | Dated capture; final main approval and operational gates remain. | +| [2026-09-29-remediation-v2-integration-evidence.md](2026-09-29-remediation-v2-integration-evidence.md) | Written here | Remediation v2 integration evidence | Dated capture; final main approval and operational gates remain. | +| [2026-09-29-remediation-v2-rulings.md](2026-09-29-remediation-v2-rulings.md) | Written here | Remediation v2 ordered execution rulings | Dated capture; final main approval and operational gates remain. | +| [2026-09-29-dashboard-paused-time-evidence.md](2026-09-29-dashboard-paused-time-evidence.md) | `.superpowers/sdd/2026-09-28-dashboard-refresh/paused-time-report.md` | V3 paused-time consumer proof | Body preserved from its pre-integration capture; later integration evidence is separate. | +| [2026-09-29-remediation-v2-scope-matrix.md](2026-09-29-remediation-v2-scope-matrix.md) | Written here | Remediation v2 scope reconciliation | Dated capture; final main approval and operational gates remain. | +| [2026-09-29-windows-ci-evidence.md](2026-09-29-windows-ci-evidence.md) | Written here | Windows CI historical cohort and current timing gate | Dated capture; final main approval and operational gates remain. | diff --git a/docs/codex-usage-diagnostic.md b/docs/codex-usage-diagnostic.md index 4151b979..d9284fcc 100644 --- a/docs/codex-usage-diagnostic.md +++ b/docs/codex-usage-diagnostic.md @@ -9,6 +9,16 @@ Everything you need is below: what was wrong, why the fix can be trusted without re-auditing the code yourself, how to run one script, and exactly what to send back. +**Updated 2026-09-29 — diagnostic scope.** This script remains an independent historical +cumulative-snapshot comparison. It does not reproduce the current parser's per-turn import +ownership, replay subtraction, counter segments, per-day/model attribution, or positive component +usage with zero responses. Use [ADR-0052](adr/0052-codex-usage-attribution.md) and the +[current accounting contracts](usage-scorecard-metrics.md#current-accounting-and-cache-contracts) +to interpret differences. Total-only counters remain unsupported for split/pricing, Auto-review +models may be unpriced, and host-reported first-token/compaction evidence is separate. A mismatch +with this script alone is not evidence of a present accounting defect. The figures below are +historical, not a new corpus measurement. + --- ## The short version diff --git a/docs/dashboard.md b/docs/dashboard.md index 9c6ccc16..0654494c 100644 --- a/docs/dashboard.md +++ b/docs/dashboard.md @@ -53,7 +53,7 @@ permanent. | System | Sessions | `#system/sessions` | Sessions | The largest retained sessions, with a localized two-line native identity, working context, and share of that host's retained bytes | | System | Storage | `#system/storage` | Storage | Where the retained bytes are, by category and host — learning stores counted separately because they dwarf everything else — plus per-series growth | | System | Runtime | `#system/runtime` | Runtime | Live host processes, their CPU and memory, background daemons, and machine denominators — refreshed on the header's poll clock while open | -| System | Catalog | `#system/catalog` | (redirect) | Retired as a visible destination. The link redirects to Maintenance › Inventory. Full scan still collects the catalog measurement, and its cards now sit in Summary | +| System | Catalog | `#system/catalog` | (redirect) | Retired as a visible destination. The link redirects to Maintenance › Inventory. Refresh machine still collects the catalog measurement, and its cards now sit in Summary | | System | Projects | `#system/projects` | Projects | Every repository with a remote that a host has recorded a session in — its approximate lines of code, language mix, total disk size and last activity. Worktrees, sub-folders and remote-less repositories are counted below the table, not listed | | System | Maintenance | `#system/maintenance` | Maintenance | Four destinations: **Inventory** (`#system/maintenance/inventory`, Focus navigation from scope through resource family to exact installation details), **Guidance** (`/guidance`, only outcomes the kit can ground, in five lanes), **Discovery** (`/discovery`, automatic sources, exact projects, collection roots, exclusions, scan coverage), and **Activity** (`/activity`, receipts, undo, interruption audits, dispositions, recipe changes, scan records). Inventory links carry scope, view, sort, `facet.` values, and the selected placement as opaque state | @@ -237,6 +237,24 @@ or proof of subscription billing. Claude Code writes one transcript line per con repeats the message's usage on each, so Claude tokens, cost, responses and context samples count each API message once. +**Evidence and compatibility (updated 2026-09-29).** Surface, initiator and provider are +separate from host and Git scope. Local session/project detail exposes bounded declared origin +tokens and observed provider basis; Unknown remains explicit, and provider metadata is not +network attestation. Legacy desktop filters retain their labeled membership. A coarse legacy +Codex desktop snapshot gives a ChatGPT desktop app family note without guessing its mode. +Cloud choices require observations; dedicated Cowork storage remains uncovered. Census views +show imported exclusions, mixed/unresolved observations, count basis and incomplete coverage. +Runtime distinguishes desktop applications from their observed CLI sessions. + +Usage counts proven native intervals in mixed imports and positive Codex token components even +without responses. Claude copied messages have one charge owner across the bounded current and +previous window pool; session-local observations remain qualified. OpenCode unknown/nonlocal +positive-token zero cost is unpriced. Source warnings remain visible without current-window +activity, and ambiguous databases require explicit selection. Older caches may rebuild; old +OpenCode child fingerprints are ignored by prompt metrics while usage remains. See the +[current contracts](usage-scorecard-metrics.md#current-accounting-and-cache-contracts) for exact +bounds, timezone behavior and the version-limited OpenCode reconciliation signal. + **Two hero rows.** The first carries sessions, api-equivalent cost, tokens, engaged time, and cache read. Each tile pairs its figure with a change against the previous window of the same length and a per-day sparkline, so the number and its direction arrive together. The change is read @@ -267,7 +285,8 @@ each with its percentile markers laid over the bars. A percentile that lands in top bucket renders with a `≥` prefix — the bucket has no upper edge, so the honest claim is a floor rather than a point. A window holding no samples reads `not measured` instead of a row of zero bars. **Response latency is the gap between a prompt and the response that answered it. It is not -time-to-first-token**, which no local transcript records. +time-to-first-token**. Codex can separately record host-reported first-token timing; that +observation does not rename or replace the completion-latency metric. **How you run** answers permission posture, who drove, and who served. Posture is a closed four-value vocabulary — guarded, auto-edit, plan, unrestricted — mapped from each host's own @@ -279,9 +298,9 @@ own nested transcript, so that cost is discovered, priced, and included; a forke rollout opens with its parent's replayed history, so only what follows the replay is counted — the subagent's own tokens are priced and the parent is never billed twice. A subagent from a host that records no event ordinals cannot have its replay separated and reads `$0.00`, which means not -measurable rather than cheap. Codex sessions imported from Claude Code are not Codex activity and are -excluded from every Codex figure. The panel does not rank window cost by inference provider: a transcript host is not a vendor. -Codex and OpenCode can record a serving provider, while Claude history lacks that field; +measurable rather than cheap. Copied turns imported from Claude Code are excluded from Codex figures; proven native +turns in mixed files remain eligible, with unresolved ownership disclosed. The panel does not rank window cost by inference provider: a transcript host is not a vendor. +Codex and OpenCode can record a serving provider, while Claude may expose provider-specific assistant model metadata; identity is reported per session on the Sessions detail strip — beside the provenance backing it — rather than as a window axis. @@ -557,6 +576,8 @@ updates. See [Observability](https://github.com/pacphi/agentic-kit/blob/main/docs/observability.md) for the map legend, workspace facts, host capability coverage, History/Review semantics, privacy limits, and troubleshooting. +The optional `--live-source` structured input is experimental: its Ruflo and +agentic-qe examples have fixture coverage, with no verified real producer. ## System @@ -573,13 +594,11 @@ Opening System costs almost nothing. The cheap tier — the live process census, file sizes, and the figures carried forward from the last full scan — is served on every read and cached briefly. -Everything else comes from the **Full scan**—the dashboard name for the deep tier—which walks -install trees, retained-data roots, host catalog surfaces, and the eligible hosted-repository -population. That is real I/O and can take minutes on a large machine, so it runs **only when you -press Full scan** (or run `ak system --refresh=machine`). Opening the tab never triggers it. Production runs -the synchronous collectors in one worker thread so the page can report phases and remain usable -while they run. Its status names the current phase, bounded count when available, and elapsed time. -Worker containment does not claim that the filesystem work itself completes faster. +Everything else comes from choosing **Refresh machine** in the header and pressing **Refresh**. +It walks install trees, retained-data roots, host catalog surfaces, and eligible hosted +repositories. That is real I/O and can take minutes, so opening System never starts it. +Production runs the synchronous collectors in one worker thread so the page can report phases +and remain usable. Worker containment does not make the filesystem work itself faster. One scan may reuse a complete physical observation when another section asks the same bounded question. Catalog reads one physical surface once per compatible reader contract even when several @@ -591,17 +610,19 @@ the same scan. Reuse is confined to that scan. Incomplete, older, differently rooted, or differently scoped evidence falls back to a fresh bounded walk rather than being treated as equivalent. -Measurement views fetch once, then again only while a scan you started is running (Runtime also -refreshes on the header's poll clock). They read `GET /api/system/summary`, which carries only what -the page draws; `GET /api/system` and `ak system --json` keep the complete payload. Maintenance -loads when you open it, and also reloads on the shared status poll while it stays the open view -(and no measurement or provider check is already running), reading the last complete inventory each -time; opening it checks no host provider and executes nothing. **Refresh evidence** on the -Maintenance workspace is the explicit control that runs provider probes, and it rebuilds the -Inventory afterwards; **Re-measure machine** beside it runs the System Full scan, walks every -discovery source to completion, then refreshes evidence. Full scan from the System rail chains the -same provider check after the snapshot is persisted. `ak maintain --refresh` and -`ak maintain --refresh=machine` are the CLI equivalents. +Measurement views read `GET /api/system/summary`, which carries only what the page draws; +`GET /api/system` and `ak system --json` keep the complete payload. Opening Maintenance reads +the last complete inventory and starts no provider or machine check. The header has one +**Refresh** button. Its selector shows **Refresh** (local strength), **Refresh live** (live +strength), and **Refresh machine** (machine strength). The local strength refreshes Maintenance +evidence, rebuilds the inventory, and re-checks local evidence and versions. The live strength +adds bounded live checks. The machine strength first measures the machine, then refreshes +Maintenance evidence; its inventory stage walks the discovery sources and rebuilds from the new +measurement. A failed machine measurement skips those two dependent stages. +The control starts an operation with `POST /api/refresh` and reads its progress with +`GET /api/refresh`. The System and Maintenance GET routes remain read-only. +`ak maintain --refresh` and `ak maintain --refresh=machine` offer the CLI equivalents. +**Reload** re-reads the active view; it starts no machine or provider check. ### Session identity and local time @@ -632,7 +653,7 @@ while provider-backed actions live only in Maintenance. ### Catalog cards in Summary -The cross-host capability catalog is still measured by Full scan, and its three cards now live at +The cross-host capability catalog is still measured by Refresh machine, and its three cards now live at the bottom of Summary: **Host inventory profile**, **Unique across hosts** (the presence matrix, filtered by what to show, which host carries it, and which source scope), and **Project skill pressure** (a project-by-host table with per-project host disclosure). Project, user, and @@ -669,15 +690,17 @@ Guidance and Activity tabs carry a count only when something is admitted or need A fresh installation shows an empty Inventory and every installed automatic source as **Not scanned yet**; a host that is not installed reads **Not installed**. -Two actions sit side by side above the tabs, each with its helper text: **Refresh evidence** runs -provider probes on the saved measurement and rebuilds the inventory in seconds, and **Re-measure -machine** walks the filesystem, then every discovery source, then refreshes evidence, which takes -minutes. Choose Refresh evidence to build the inventory; `ak maintain --refresh` -does the same from a terminal, together with the local status checks. While either runs, both -buttons are disabled, the status line names what is running ("Refreshing evidence…"; during -Re-measure machine, each phase in turn, from "Preparing measurement…" through "Machine measured · -refreshing evidence…"), and apply, undo, and record are refused; if the work does not finish, the -previous evidence is kept. +Select **Refresh** in the header selector and press the **Refresh** button to build the inventory +from saved measurement. **Refresh live** adds bounded live checks. **Refresh machine** measures +the machine, then refreshes Maintenance evidence; the inventory stage walks the discovery sources +and rebuilds the inventory. `ak maintain --refresh` runs the local stages from a terminal. The +ordered progress labels are **Measuring the machine** (machine strength only), +**Refreshing Maintenance evidence**, **Rebuilding the inventory** (including the discovery walk +for machine strength), **Running live checks** (live strength only), and +**Re-checking local evidence and versions**. While an operation runs, another cannot start, +and Maintenance apply, undo, and record are refused. If work does not finish, the previous +complete evidence is kept. + After the probes settle the inventory builds in the background: the empty state reads **Building the inventory…** until rows appear, or names the reason if the build did not complete. @@ -743,8 +766,8 @@ A saved root starts scanning at once. Scan progress reads as visited work, never added offer **Pause** and **Stop** while running, **Resume** and **Stop** while paused, **Retry scan** after a failure, and **Scan this root** if never run; stopping shows what would be affected and asks **Stop this source?**. Automatic sources carry no per-source control: each reads Not -scanned yet with "measured by Re-measure machine", or Complete with "covered by the last -measurement". A host source whose folder is not on this machine reads **Not installed** and is +scanned yet, or Complete after the last measurement. A host source whose folder is not on this +machine reads **Not installed** and is not counted in the progress sentence or the Inventory banner. A started source keeps running until it completes, pauses, stops, or fails. Host configuration sources skip transcript, session, log, and cache trees by name so they can complete. @@ -783,12 +806,11 @@ the totals. Every parent with breakdowns also gets an "everything else" row, so adds up to its parent. Roots that do not exist on this machine are listed as absent rather than ranked at 0 B, and roots that could not be read say so with their reason. -**Project trees** are excluded by default, and the chip that includes them is a *scan* control, -not a filter. One large repository can outweigh every shared cache combined, and a chart -containing it is a chart of one repository — so the ranking says, in the panel, that they were -left out. Turning the chip on starts a new Full scan that walks them (and turning it off starts -one that does not); it is disabled while a scan is running. `ak system --refresh=machine` scans -without project trees; add `--project-trees` to include them. +**Project trees** are excluded by default. The **Include project trees** option is +available only when **Refresh machine** is selected; it changes the measurement scope. +One large repository can outweigh every shared cache combined, so the panel says when +project trees were left out. Select that option and press Refresh to measure them. +`ak system --refresh=machine` omits them unless `--project-trees` is added. ### Two reclaimable tiers, never one total @@ -810,7 +832,7 @@ removes anything; where a CLI already owns the cleanup, the row names it. ### Reading the numbers honestly -- **A section that has never been scanned says so.** It reads "not measured yet — run Full scan", +- **A section that has never been scanned says so.** It reports an unmeasured state, never `0`. A zero here means a real, measured zero. - **A total whose inputs were incomplete renders as `≥ N`.** If one subtree could not be read or a walk hit its cap, the sum is a floor, not a total, and is labeled that way. @@ -954,8 +976,9 @@ provider, model/default agent, and applicable credentials or local endpoint. Automatic checks do not invoke its config-debug command, which can install dependencies. Unresolved remote configuration and native overrides stay Unknown. -**Check again** refreshes the local evidence for any host. **Check connection** -runs only for hosts managed by ak, and requires +Select **Refresh** in the header selector and press the **Refresh** button to re-check local +evidence for any host. +**Check connection** runs only for hosts managed by ak, and requires checking a confirmation box first: it sends one small provider request, using normal billing and native context. Native startup may initialize dependencies and update local cache/session files. Agent tools are restricted, and no repair diff --git a/docs/ddd/machine-footprint.md b/docs/ddd/machine-footprint.md index 28effd21..c7ad16f3 100644 --- a/docs/ddd/machine-footprint.md +++ b/docs/ddd/machine-footprint.md @@ -193,7 +193,8 @@ FootprintSnapshot { asOf, completeness, install, runtime, storage, catalog, pro v Delivery GET /api/system → cheap tier + persisted snapshot (token auth, loopback, no egress) - GET /api/system?refresh=deep → start-or-attach the single-flight deep scan + POST /api/refresh → start the single-flight staged refresh + GET /api/refresh → read that operation's progress GET /api/system/summary → the same read, catalog/storage/install/projects/consumers each projected to what the System page draws ak system [--refresh[=live|machine]] [--project-trees] [--json] → the same collector, CLI-rendered @@ -521,7 +522,7 @@ That identity also governs acquisition cost inside one Catalog collection. Compa keyed by normalized physical path and reader contract, so Claude and OpenCode bindings to the same skill surface share one bounded observation while retaining two ConsumerBindings. Markdown and file-stem entrypoints likewise compute one digest per file. The observation map is created and -discarded inside the collection; a later Full scan always observes the filesystem again. A path +discarded inside the collection; a later Refresh machine always observes the filesystem again. A path match under a different reader contract is not reusable evidence. Every occurrence retains host, surface, source scope (`user`, `project`, or `plugin`), project @@ -645,15 +646,22 @@ declarations and unclassified sightings. `countBasis` distinguishes transcript f sessions, recovered-project sightings and mixed observations; a recovered directory is not one verified session. -Imported session copies are not sightings. A Codex rollout stamped `external-import-turn-*` is a -Claude Code transcript that the ChatGPT desktop app imported; it is a copy, and the original Claude -Code session is counted where its transcript still exists. A folder that only an import names is -therefore not a project. Discovery skips it before reading its cwd, so it adds no project, host or Session -origin, and counts it in `importedExcluded` (per host scan and in total). `complete` is unaffected. +Updated 2026-09-29: imported Codex turns are not sightings. Pure copies add no project, host or +surface and count in `importedExcluded`; proven native activity in a mixed file can establish a +sighting and counts in `importedMixed`. Bounded observations that cannot settle ownership count +in `importedUnresolved` and make coverage incomplete. The budgets and malformed-record behavior +are specified in [ADR-0052](../adr/0052-codex-usage-attribution.md#3-sessions-codex-imported-from-claude-code-are-not-codex-sessions). -**Proposed change ([ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md)).** -`sessionOrigins` is to be replaced by session surface and initiator, derived from the same declared -fields but keeping every raw value. +Accepted [ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md) adds +`sessionSurfaces` alongside legacy `sessionOrigins`. Claude census counts declared session IDs, +excluding subagent and bridge-only records; recovery has zero session weight. Raw detail retains +at most 16 validated tokens per field per classification group and reports truncation. Cloud +surfaces appear only with observations; dedicated Cowork storage remains uncovered. Old coarse +Codex desktop origins cannot identify an app mode. Footprint schema remains 8. + +Runtime distinguishes Claude Desktop and the ChatGPT desktop app as applications with no host. +Their bundled CLI sessions require observed process attribution; Codex app-server is a service. +Application presence alone does not establish an active conversation or historical provider. `project-identity.mjs` relates directories through canonical Git metadata and, for linked worktrees, a verified common directory plus backlink. It preserves unknown association when evidence is @@ -704,38 +712,28 @@ The link is user-initiated browser navigation; the kit itself never fetches the token auth ([ADR-0014](../adr/0014-dashboard-auth-and-remediation.md)), `no-store`, zero egress. The response is the cheap tier computed fresh (TTL ~60s, shared-cache pattern like the project-snapshot cache) merged with the persisted deep snapshot and its `asOf`. -`?refresh=deep` starts the dashboard's **Full scan** or attaches to the one in flight -(single-flight, like the usage index's coalesced builds). In production, `index.mjs` retains the -single-flight promise and public activity state while `deep-scan-worker.mjs` runs the synchronous -runner in one worker thread. Phase and Projects progress messages return to the main thread, so -ordinary reads remain responsive while the worker is busy. Injected collectors and filesystem -implementations run the same `deep-scan-runner.mjs` inline rather than attempting to serialize test -functions. This containment is not evidence that total scan duration decreased. - -The System measurement routes stay GET-only: a Full scan re-measures local state and writes only this domain's own -snapshot file — it mutates no user data. - -`GET /api/system` is the complete read model, the same shape as `ak system --json`. -`GET /api/system/summary` is the page's read: the same payload (and the same `?refresh=deep` and -`&trees=` parameters) with `catalog`, `storage`, `install`, `projects` and `consumers` each -projected by `dashboard/system-summary.mjs` to an allow-list of keys — the catalog's items cut to -key, kind, name, hosts, source scopes, digest coverage, and distinct plugin providers -(`presence[].provider` with `ref` and `version`); storage's category/host/project/session tree cut -to key, label, bytes and children; a measured project row's per-tool native-addon lists and -per-project framework/dependency stack detection dropped entirely (never rendered). The catalog's -repeated presence copies (`item.presence` details, `consumerBindings`, `artifacts`) grow with -items × projects × hosts and are not drawn, so the page and its 30-second Runtime poll never -download them; the other four sections carried the same shape of excess and were the majority of -the endpoint's real-machine bytes once the catalog alone was slimmed. The projection is a -Dashboard-delivery view; this domain's collector output is unchanged. - -**A deep scan never runs on its own.** Opening the System area issues a plain `GET /api/system/summary`; -only **Full scan** adds `?refresh=deep`. A deep scan can cost minutes of I/O on a large corpus, and -making the act of *looking* cost that is a worse trade than a stale figure that states -how stale it is. Staleness is therefore surfaced rather than pre-empted: every deep-tier figure -renders with its snapshot's `asOf`, and past `SNAPSHOT_STALE_AFTER_MS` (7 days) the freshness -label turns amber and reads "stale, scan again". The client polls only while a user-started scan is -running, and stops when it finishes. +The header's **Refresh** control starts an operation through `POST /api/refresh`. +Choosing **Refresh machine** runs the deep collector first; `GET /api/refresh` reads the operation's +stage progress. In production, `index.mjs` retains the single-flight promise and public +activity state while `deep-scan-worker.mjs` runs the synchronous runner in one worker thread. +Phase and Projects progress messages return to the main thread so ordinary reads remain +responsive. Injected collectors run the same `deep-scan-runner.mjs` inline in tests. +Worker containment does not show that total scan duration decreased. + +The System measurement routes are read-only GETs. `GET /api/system` returns the complete +read model, the same shape as `ak system --json`. `GET /api/system/summary` projects +`catalog`, `storage`, `install`, `projects`, and `consumers` to the allow-listed keys +drawn by the page. The catalog omits repeated presence details, consumer bindings and +artifacts; Projects omits native-addon and stack details. This projection reduces the +payload downloaded by the page and its Runtime poll while the collector output stays +unchanged. + +Opening System reads the saved snapshot and starts no measurement. A deep scan can cost +minutes of I/O, so the user explicitly chooses **Refresh machine** and presses **Refresh**. +Every deep-tier figure renders with its snapshot's `asOf`; after `SNAPSHOT_STALE_AFTER_MS` (7 days), +the freshness label turns amber. The client reads progress for the user-started operation +and stops polling when it finishes. **Reload** re-reads the active view without starting +machine or provider checks. One deliberate divergence from Observability's delivery: absolute paths are **part of this payload**. `publicLivePayload`'s leaf-only rule exists to keep incidental provenance out of @@ -746,7 +744,7 @@ nothing to leak. The CLI twin (`ak system`) renders the same collector output, `--json` emitting the collector's payload verbatim, following the one-collector-two-surfaces precedent of the usage scorecard. -`ak system --refresh=machine` is the terminal spelling of **Full scan** and writes the same snapshot. +`ak system --refresh=machine` is the terminal spelling of **Refresh machine** and writes the same snapshot. ## Invariants @@ -856,7 +854,7 @@ normative and this table restates it for readers of this document. | Definition digest | SHA-256 over one complete bounded observed capability definition; equality proves those files match, not host selection, ownership, usage, or removal safety | | ProjectCapabilityPressure | Project/user/plugin contributions and exact overlap per project and host; context inclusion remains unknown | | ProjectFootprint | One eligible hosted repository's size facts: approximate LOC by language, tree/`.git`/`node_modules` bytes, last activity, and a proven HTTPS web link | -| Deep scan | The explicit, user-triggered, single-flight measurement pass called **Full scan** in the dashboard; it produces a FootprintSnapshot over the stated bounded populations | +| Deep scan | The explicit, user-triggered, single-flight measurement pass selected by **Refresh machine** in the dashboard; it produces a FootprintSnapshot over the stated bounded populations | | Cheap tier | The per-request census + known-file stats + snapshot carry-forward served on every read | ## References diff --git a/docs/ddd/observability.md b/docs/ddd/observability.md index 513ff96e..33e7aaeb 100644 --- a/docs/ddd/observability.md +++ b/docs/ddd/observability.md @@ -901,6 +901,8 @@ agentic-qe source adapters require explicit, repeatable `--live-source 'surface= registration, where `surface` is `ruflo` or `aqe`. Explicit sources consume tailer capacity before Claude/Codex discovery. Paths resolve against the startup working directory and are not confined to the project, so registration is an operator authorization to read that file. +The structured input remains experimental: parser fixtures exist, but no real producer has been +verified. Registration alone supplies no activity or runtime coverage. Independent plugin, skill, MCP, and gate registries are not implemented. This is an explicit source coverage limitation; ADR-0012 is Implemented because the supported adapter contract does not claim automatic upstream discovery. diff --git a/docs/ddd/ubiquitous-language.md b/docs/ddd/ubiquitous-language.md index d84bbe64..dda55274 100644 --- a/docs/ddd/ubiquitous-language.md +++ b/docs/ddd/ubiquitous-language.md @@ -88,7 +88,7 @@ token estimates never become observed token evidence. An absent or incompatible | Quick live checks | The bounded, no-cost subset of checks that a plain `ak status --refresh=live` runs in parallel: AQE embedding request, Codex MCP handshake, provider wiring, security packages, deja-vu structure, and the `memory` check's temp-dir CLI store/retrieve/purge round trip (no MCP tool calls; that is the separate `memory-routes` slow proof); a timeout is `inconclusive`. `--only CHECK[,CHECK...]` names exactly which checks to run, including a slow proof | | Slow proof | A live check that runs only when named with `--only`, up to six minutes each: `learning` (trains a temporary fixture, asserts patterns persist), `harvest` (records an outcome and distills through Ruflo in an isolated store), `aqe` (storage, embedding configuration/provenance, and the browser payload), `memory-routes` (the memory round trip plus whether CLI and MCP see each other's writes, remembered as the `memory` check). A named check runs even when it would not otherwise apply; `learning` and `harvest` are never remembered | | Connection check | `ak host check-connection [--yes] [--json] [--dry-run]`: the consent-gated, paid probe of a managed host's live connection. It is never a strength of `--refresh` and runs only for a managed host whose local checks already pass, after printing the shared disclosure and asking `[y/N]` on a TTY (refused without `--yes` off a TTY) | -| Machine measurement | The full re-measure `--refresh=machine` runs: walking install trees, storage, the cross-host catalog, and (with `--project-trees`) your projects' working trees, then persisting the result and rebuilding the inventory. The dashboard's Full scan and Re-measure machine controls run the same chain | +| Machine measurement | The full re-measure `--refresh=machine` runs: walking install trees, storage, the cross-host catalog, and (with `--project-trees`) your projects' working trees, then persisting the result and rebuilding the inventory. Choosing **Refresh machine** in the dashboard and pressing **Refresh** runs the same chain; **Reload** only re-reads the active view | | Memory route observation | What `ak status --refresh=live --only memory-routes` reports after its CLI proof, in its throwaway project only: whether a key written through Ruflo's CLI is readable through MCP and the reverse, where each landed, and the MCP backend seen. A split is a warning and an unusable interface is "not observed"; it never fails the suite, is not part of the quick live checks, and says nothing about an existing corpus | | Observed routing pair | An exact `@claude-flow/cli` release and platform on which the memory route observation was recorded. Only for such a pair does status say which Ruflo interface reads which project-memory store; a neighbouring, prerelease or build-tagged version, or another platform, stays unverified | | Canonical memory store | `/.swarm` for the root every ak memory launch contract pins (the repository root, else the folder, unless that is an unsuitable memory folder): `memory.db` and, with the native bridge, `agentdb-memory.db`. Status reports it from any subfolder, with each file's size, live WAL, largest namespace and how much of it is set to expire | @@ -133,20 +133,21 @@ missing price. `Dual-host` describes two enabled peer hosts, not an execution command and not evidence that two inference vendors served a workflow. Generalized execution belongs to `ak run`. -## Session surface language (mostly proposed) +## Session surface language -These terms are proposed by [ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md). -Only **Imported session copy** is implemented so far, and only in usage and project discovery. For -the rest, the implemented contract is ADR-0050's **session origin** (`claude-desktop`, -`codex-desktop` or `unknown`), which an imported copy never supplies. +Updated 2026-09-29 to match accepted +[ADR-0060](../adr/0060-session-surface-initiator-and-product-names.md). Git scope, host, surface, +initiator and provider are separate dimensions. Legacy origin keys remain compatibility evidence. | Term | Meaning | |------|---------| | Session surface | The product surface that started a session, read from the host's own declared field (Claude `entrypoint`; Codex `originator` with `source`) and shown by its official name, such as Claude Code CLI, Claude Desktop, ChatGPT desktop app · Codex, or Codex CLI | -| Initiator | Who started a session: a person, automation (scripts, SDKs, non-interactive runs, CI), an agent (a subagent or reviewer spawned by another session), or an imported copy | -| Imported session copy | A session one tool copied from another, such as a Claude Code transcript the ChatGPT desktop app imported as a Codex thread; excluded from usage, origin and project counts and reported as a count | -| Raw surface value | The exact declared value a surface was derived from; always kept, and shown for any value the vocabulary does not recognize | +| Initiator | Who started a session: a person, automation (scripts, SDKs, non-interactive runs, CI), an agent (a subagent or reviewer spawned by another session), an imported copy, or Unknown | +| Imported session copy | A session one tool copied from another, such as a Claude Code transcript the ChatGPT desktop app imported as a Codex thread; whose copied turns are excluded from usage and project/origin sightings; proven native turns in mixed files remain eligible, with exclusion and incompleteness counts | +| Raw surface value | A named declaration token retained only under the approved 80-character validation rule; unfamiliar valid tokens can appear in local detail without inferring a product or provider | | Tool workspace | A folder a tool creates for its own work outside the user's projects, such as `~/.codex/.chatgpt-projects/…` or `~/Documents/Codex/…`; an explanation attribute, never a surface | +| Session count basis | The counted unit: declared session IDs, transcript files, database sessions, recovered sightings or mixed observations; zero-weight recovery is not a verified session | +| Provider evidence basis | Recorded provider ID or provider-specific assistant model ID; observed metadata, not network attestation | | Desktop application | Claude Desktop or the ChatGPT desktop app; an application that can start sessions, not a host | Say **session surface** for where a session came from; the Live event `surface` field (native, ruflo, diff --git a/docs/host-support.md b/docs/host-support.md index 6bbfb259..4448fdf5 100644 --- a/docs/host-support.md +++ b/docs/host-support.md @@ -26,6 +26,24 @@ Claude Code `2.1.222`, Codex CLI `0.146.0`, and OpenCode `1.18.x`. Host and upstream behavior changes quickly; open issues below are a risk snapshot, not a promise that an issue remains open forever. +**2026-09-29 support-window addendum.** Registry metadata at 13:42 UTC listed +Ruflo 3.48.0, Agentic QE 3.14.5, and Codex CLI 0.159.0. In a network-denied, +disposable scan, the installed Ruflo 3.48.0 `security secrets --action scan +--path ` reported one synthetic file scanned and exited 0 without +changing the target. The npm-integrity-verified native Codex 0.159.0 binary +accepted `-s read-only -a never app-server`; an `initialize` request answered +successfully without a provider turn. These are narrow command checks, not +end-to-end host conformance. + +Released AQE 3.14.4 passed disposable live-owner lock checks on macOS and Linux: +the status command and shipped adapter reported `LockHeld` without +`FsyncFailed`, while the holder and storage bytes remained intact. The same +check passed on AQE 3.14.5 on macOS and Linux. Native Windows AQE was not run. +Ruflo 3.48.0 showed CLI-to-MCP and MCP-to-CLI memory visibility on native +Windows with one `memory.db`; the reported backend was sql.js + HNSW with its +native bridge disabled. The earlier Linux result was asymmetric, so these +observations do not establish one cross-platform native-backend guarantee. + The stock OpenCode gateway acceptance test currently covers the stable compatibility window **`>=1.18.18 <1.19.0`**. This is a tested release-line window, not a claim that every future OpenCode release is compatible and not a target for `ak sync` to @@ -118,16 +136,21 @@ Official extension references: [Claude hooks](https://code.claude.com/docs/en/ho | Upgrade convergence | `ak sync` heals managed assets | `ak sync` heals Ruflo/AQE access and retires owned legacy MCP | `ak sync` regenerates the embedded catalogue and repairs exact-receipted plugins/config | | Teardown | Managed blocks and registrations | Receipt-based managed teardown | Value- and hash-receipt teardown; user-owned values survive | -Ruflo MCP access and Ruflo-backed inference are different contracts. In -particular, Ruflo's [`agent_execute` provider-key behavior](https://github.com/ruvnet/ruflo/issues/2356) -can still require a separate provider credential even when invoked from Codex. +Ruflo MCP access and Ruflo-backed inference are different contracts. +In Ruflo 3.48.0's shipped `agent_execute` path, execution uses a separately +configured inference provider; its no-provider branch still returns an error +instead of delegating to the MCP host +([ruflo #2356](https://github.com/ruvnet/ruflo/issues/2356)). This is a source +check, not a credentialed runtime probe from Codex. `ak run` avoids that conflation by executing the selected host directly and using Ruflo for tools, memory, routing context, and orchestration assets. The dated upstream risk inventory includes: -- force initialization can overwrite unrelated `.mcp.json` content - ([ruflo #420](https://github.com/ruvnet/ruflo/issues/420)); +- force initialization still writes a generated `.mcp.json` over the existing + file in Ruflo 3.48.0's shipped source, without merging unrelated servers + ([ruflo #420](https://github.com/ruvnet/ruflo/issues/420)); this was not + exercised on a real project; - generated Claude and Codex instructions can diverge ([#2638](https://github.com/ruvnet/ruflo/issues/2638)); - init and plugin installation can duplicate assets or hooks @@ -139,6 +162,12 @@ The dated upstream risk inventory includes: - hierarchical AgentDB writes can report success without durable persistence ([#2887](https://github.com/ruvnet/ruflo/issues/2887)). +The additional Codex hook-environment fix line remains conditional. At the +2026-09-29 check, [Ruflo #3419](https://github.com/ruvnet/ruflo/issues/3419) +was open with only agentic-kit's Codex source-analysis comment, not a +maintainer-supported answer or a live hook observation. Version tags alone do +not close that evidence gap. + ## Agentic QE support | AQE capability | Claude Code | Codex | OpenCode | @@ -171,12 +200,30 @@ Current AQE includes a subscription-backed `codex` provider. Agentic-kit accepts `ak host pick --aqe-provider codex`, admits Codex fallback rungs, enables Codex providers referenced by `agentOverrides`, and projects Codex activity routes. -The dated AQE risk inventory includes its -[MCP entrypoint double-spawn](https://github.com/proffesor-for-testing/agentic-qe/issues/528), -[multi-platform initialization behavior](https://github.com/proffesor-for-testing/agentic-qe/issues/532), -[MCP tool correctness gaps](https://github.com/proffesor-for-testing/agentic-qe/issues/535), +AQE 3.14.4 adopted fixes for the +[MCP entrypoint double-spawn](https://github.com/proffesor-for-testing/agentic-qe/issues/528) +and [exclusive platform initialization](https://github.com/proffesor-for-testing/agentic-qe/issues/532) +(`aqe init --no-claude`). Both upstream issues remain open; versions below +3.14.4 retain those gaps. The remaining dated AQE risk inventory includes +[GOAP `maxSteps` and world-state, test-generation quality, and coherence recommendation-text gaps](https://github.com/proffesor-for-testing/agentic-qe/issues/535) +(the 3.14.4 recheck did not exercise `goap_execute`), [RVF recovery loop](https://github.com/proffesor-for-testing/agentic-qe/issues/574), and [local-embedding audit findings](https://github.com/proffesor-for-testing/agentic-qe/issues/615). +A 2026-09-29 disposable probe of an npm-integrity-verified AQE 3.14.5 tarball found that +[#655](https://github.com/proffesor-for-testing/agentic-qe/issues/655)'s selected compact +guidance path emits a 315-byte owned sentinel and preserves a foreign `AGENTS.md` +prefix/suffix across two repeated calls. Full/none and complete receipt/platform +conformance remain unverified. +[#753](https://github.com/proffesor-for-testing/agentic-qe/issues/753)'s fresh native +macOS witness append produced valid 4001-row chains in one sequential and three +synchronized two-process rounds; only one concurrent round demonstrably interleaved. That +does not justify removing the kit's stray-store live-holder refusal or claim old-fork +repair, signature validation, import safety, or Windows/Linux behavior. +[#778](https://github.com/proffesor-for-testing/agentic-qe/issues/778) still reproduces +settings churn on 3.14.5 across three same-option init runs; its source fix merged after +this release was published. A later released-artifact retest is pending. See the [dated artifact +receipt](archive/2026-09-29-aqe-released-artifact-receipt.md) for source binding and +limits. The Codex QE-Court investigation in [agentic-kit #108](https://github.com/pacphi/agentic-kit/issues/108) is a diff --git a/docs/maintenance.md b/docs/maintenance.md index fa9bafb4..f77d26d7 100644 --- a/docs/maintenance.md +++ b/docs/maintenance.md @@ -23,7 +23,7 @@ environment; native Windows mutation support remains an integration gate. ## Host alignment in User and Project views Select **Host alignment** under **More views**, then choose **User** or **Projects** -and an optional project filter. Use **Refresh evidence** to inspect current host +and an optional project filter. Use **Refresh** to inspect current host configuration. The rows identify retired peer transports and other host-alignment anomalies without exposing configuration contents or local paths in the inventory. @@ -58,25 +58,22 @@ The dashboard workspace has four tabs. Each answers a different question. | **Activity** | What changed? Receipts, undo, interruption audits, dispositions, recipe changes, and scan records. | Opening Maintenance reads the last complete inventory and opens **Inventory** across all scopes. -Nothing scans on open. A fresh installation has no inventory yet; the empty state reads **No -inventory has been built yet. Use Refresh evidence, above, to build it.** and every installed -automatic source reads **Not scanned yet** (a host that is not installed reads **Not installed**). Two actions sit side by side above the tabs, each with its own helper -text: - -- **Refresh evidence** runs provider probes on the saved measurement and rebuilds the inventory. - It takes seconds. The CLI equivalent is `ak maintain --refresh`. -- **Re-measure machine** walks the filesystem to re-measure installs, storage, projects, and every - discovery source, then refreshes evidence. It takes minutes. The CLI equivalent is - `ak maintain --refresh=machine`. - -Both controls run provider probes: Re-measure machine includes the Refresh evidence stage. While either action -runs, both buttons are disabled, the status line names what is running ("Refreshing evidence…"; -during Re-measure machine, each phase in turn, from "Preparing measurement…" through "Machine -measured · refreshing evidence…"), and apply, undo, and record are refused. If -the work does not finish, the previous evidence is kept. The inventory build runs after the probes -settle and can take a few seconds on a large footprint; the empty state reads **Building the -inventory…** until the rows appear, and **The last inventory build did not complete** with a short -reason if it fails. The retired Catalog link (`#system/catalog`) redirects to Inventory. +Nothing scans on open. A fresh installation has no inventory yet; the empty state asks you to +use **Refresh** in the header, and every installed automatic source reads **Not scanned yet** +(a host that is not installed reads **Not installed**). + +The header has one **Refresh** button. Its selector shows **Refresh** (local strength), +**Refresh live** (live strength), and **Refresh machine** (machine strength). The local strength +runs provider probes on the saved measurement, rebuilds the inventory, and re-checks local +evidence and versions. The live strength adds bounded live checks. The machine strength first +re-measures installs, storage, and projects, then refreshes Maintenance evidence. Its inventory +stage walks the discovery sources before rebuilding the inventory from the new measurement. +The CLI equivalents are `ak maintain --refresh`, `ak maintain --refresh=live`, and +`ak maintain --refresh=machine`. **Include project trees** applies only to **Refresh machine**. +While an operation runs, another refresh cannot start; apply, undo, and record are refused. +The prior complete evidence is retained if work does not finish. The empty state reads +**Building the inventory…** while the inventory builds, or names the failure reason. +The retired Catalog link (`#system/catalog`) redirects to Inventory. ## Inventory @@ -299,7 +296,7 @@ Discovery is where you tell Agentic Kit where to look. Configuration is user int configuration, Codex user configuration, OpenCode user configuration, Hermes user configuration, Projects (every project a recorded host session has visited), Runtimes, Package managers, Ollama (over loopback only), and Providers. Each states what it inspects, never a path. Automatic - sources have no per-source scan control: their coverage comes from **Re-measure machine**, and + sources have no per-source scan control: their coverage comes from **Refresh machine**, and the non-filesystem ones (Runtimes, Package managers, Ollama, Providers) are covered by the provider check. Asking `ak maintain scans start` to walk one of those is refused with `SOURCE_NOT_SCANNABLE`. A host source whose folder is not on this machine (for example Hermes @@ -337,7 +334,7 @@ Scans are resumable and completion-oriented. run. A started root keeps running through its work slices until it completes, pauses, stops, or fails; you never have to resume it yourself. `ak maintain scans start --source ID` does the same from the CLI and waits for the final state. Automatic sources show instead whether they are - measured by Re-measure machine or covered by the last measurement. + measured by choosing **Refresh machine** or covered by the last measurement. Progress is factual: "Scanned N entries. X of Y sources are complete. N sources have not been scanned yet." Counts are visited work, never totals. @@ -460,7 +457,7 @@ ak maintain undo --receipt RECEIPT_ID --yes Undo needs the recorded provider and version, a reversible or compensating operation, and an exact current postimage. If anything changed after apply, undo refuses instead of overwriting the new state. If no inventory has been built yet, plan and apply refuse with `SCAN_REQUIRED`; choose -**Refresh evidence** or run `ak maintain --refresh` first. If a placement cannot be +**Refresh** or run `ak maintain --refresh` first. If a placement cannot be bound to an exact executable finding, they refuse with `PLACEMENT_FINDING_UNRESOLVED`. Rollback classes are separate from safety: **reversible** (the provider restores and verifies the @@ -494,7 +491,8 @@ server registration and are not separate MCP installations. Measured user instruction files retain their resolved configuration location through **Reveal exact path**. Their paths remain private in ordinary inventory responses. -Refresh evidence after upgrading to populate locations missing from an older snapshot. +Select **Refresh** in the header selector and press the **Refresh** button after upgrading to +populate locations missing from an older snapshot. Inventory can relate a standalone skill and a plugin-contributed skill by exact name, bounded entrypoint digest, or bounded full-definition digest. Full-definition equality includes the @@ -542,7 +540,7 @@ transaction applies; follow the verb-specific options below. | Verb | What it does | |------|--------------| -| `[report] [--refresh[=live\|machine]] [--project-trees]` | `report` (the default verb) reads the last measurement. A bare `--refresh` refreshes Maintenance evidence and rebuilds the Inventory first (the dashboard's **Refresh evidence**); `--refresh=machine` re-measures System first and walks every discovery source to completion before rebuilding (the dashboard's **Re-measure machine**); `--project-trees` with `--refresh=machine` also measures your projects' working trees. | +| `[report] [--refresh[=live\|machine]] [--project-trees]` | `report` (the default verb) reads the last measurement. A bare `--refresh` refreshes Maintenance evidence and rebuilds the Inventory first (the dashboard's **Refresh**); `--refresh=machine` re-measures System first and walks every discovery source to completion before rebuilding (the dashboard's **Refresh machine**); `--project-trees` with `--refresh=machine` also measures your projects' working trees. | | `inventory [--scope S] [--view V] [--facet name=value ...] [--search TEXT] [--sort ORDER] [--cursor TOKEN] [--limit N]` | Queries placements. | | `show --placement ID [--reveal]` | Prints the inspector; `--reveal` prints the exact, owner-only path. | | `guidance [--lane LANE]` | Lists admitted Guidance entries and per-lane counts. | @@ -627,10 +625,13 @@ or an export contains a local path unless you reveal or export it deliberately. ## Dashboard security boundary -Maintenance is the only dashboard mutation surface. The v1 routes remain as compatibility: +Maintenance actions are the dashboard's exact-placement mutation surface. The v1 routes remain +for reads and explicit actions; refresh starts through the shared operation route: ```text -GET /api/maintenance (?refresh=scan runs the provider check, then rebuilds the Inventory) +GET /api/maintenance (reads the saved report) +POST /api/refresh (starts the selected refresh strength) +GET /api/refresh (reads operation progress) POST /api/maintenance/plans POST /api/maintenance/apply POST /api/maintenance/undo @@ -771,13 +772,14 @@ See the [dated top-50 coverage list](https://github.com/pacphi/agentic-kit/blob/ Activity presents scan history as a table grouped by the browser’s local calendar date, newest first. Each date has a chevron toggle to expand or collapse its source rows; groups -start collapsed and retain their state while the dashboard stays open. Source rows show completion time and timezone, source, status, and entry count. +start collapsed and retain their state while the dashboard stays open. Source rows show a completion time when available, plus source, status, and entry count. Discovery focuses on source configuration and current coverage; historical scans appear only in Activity. Groups represent dates, not inferred shared scan runs. Missing dates remain explicitly unknown. Version measurements, update checks, and snooze deadlines also use local date/time formatting. -Scan history retains the latest 10 completed records per source and environment, within the -90-day retention limit. Each new record replaces the oldest retained record for that source. +Scan history retains the latest 10 records per source and environment, within the +90-day retention limit. A pause records a continuation boundary with a recorded time, +not a scan completion time. Each new record replaces the oldest retained record for that source. Activity places recovery, work in progress, receipts, dispositions, and recipe changes in a responsive card grid above the full-width scan history. Narrow screens use a single column. @@ -788,5 +790,6 @@ rows. Column headers stay pinned while dates and records scroll. Executable installations expose their measured launcher through **Reveal exact path**. Detection checks PATH (including Windows PATHEXT), resolves symlinks, and reads bounded npm `bin` metadata when a launcher is not on PATH. When only the installation root was measured, -that root remains revealable. Paths stay out of the public inventory. Re-measure machine -to collect new launcher evidence; discovery covers the environment running the scan. +that root remains revealable. Paths stay out of the public inventory. Select **Refresh machine** +in the header selector and press **Refresh** to collect new launcher evidence; discovery covers +the environment running the scan. diff --git a/docs/models.md b/docs/models.md index a24aa393..8c9c1906 100644 --- a/docs/models.md +++ b/docs/models.md @@ -236,7 +236,7 @@ browser never derives a provider or publisher from a name and never invents an e An `unknown` cell explains which evidence is absent. Model details separately name published or discovered status, account access, local routability, and the operator's next evidence step. The table labels the discovery dimension **Catalogued**, not **Available**, so provider publication or -local discovery cannot be mistaken for account entitlement. A local refresh now resolves OpenCode's +local discovery cannot be mistaken for account entitlement. A local model refresh resolves OpenCode's effective configuration, removing unknowns caused only by ignored global, JSONC, agent, or command layers. Discovery still does not establish entitlement; configuration does not establish successful use; and a model id never establishes the serving provider. Catalog Explorer model details show an diff --git a/docs/observability.md b/docs/observability.md index ee049371..d8a1ff98 100644 --- a/docs/observability.md +++ b/docs/observability.md @@ -23,8 +23,9 @@ ak dashboard \ > OpenCode process presence is observed. What you don't get: ruflo and agentic-qe activity, which are never > auto-discovered and only appear once you register their event file > explicitly (see [Evidence and limitations](#evidence-and-limitations)). -> `--live-source` reads a file that something else already writes. It does not make Ruflo or -> agentic-qe produce events. +> `--live-source` is **experimental**. Its schema and parser have fixture coverage, but no +> verified Ruflo or agentic-qe producer currently writes a structured live-events file. +> Registering a path reads that file; it does not start a producer or establish runtime coverage. Open `#observability/live` or `#observability/history`, for example `http://127.0.0.1:7431/#observability/live` — once the dashboard's per-session token is already in @@ -37,6 +38,14 @@ leaves, collectors stop after 30 seconds by default. The next request resumes ea stopped, so work written while nobody was watching still appears. Stopping the dashboard closes the live service and all clients. +When a native transcript leaves the newest-file window, the service keeps a bounded in-memory +reader state. A file that returns while that state is retained resumes at its prior byte offset; +its accepted-record count does not rise from replay alone. The retained state is limited to twice +the configured file bound, with a minimum of two readers. Once evicted, a returning file is read +from its start. A new dashboard process has no saved native offset; its first scan bootstraps +metadata and begins following existing files at their ends. The live view does not provide a +durable, exactly-once event archive. + Model lifecycle is a separate read model under **Usage → Models**. It may consume bounded model ids already derived by the historical usage index, but it never consumes live transcript content or changes Observability state. Public catalogue enrichment cannot rename, add, remove, or alter @@ -358,7 +367,8 @@ paths remain absolute. The parser rejects an unsupported/missing surface or an empty path, but registration does not prove that the file exists, is a regular file, is inside the current project, or is produced by the named subsystem. Registration also does not turn on event output: `--live-source` observes a file an -existing producer writes. Unreadable/malformed sources degrade their adapter rather than +external producer may write. No real Ruflo or agentic-qe producer has been verified for this +structured format. Unreadable/malformed sources degrade their adapter rather than crashing the dashboard. Only register a local file you trust the dashboard process to read. The structured adapter still constructs allowlisted events, so arbitrary JSON fields do not pass through to the browser. diff --git a/docs/plans/2026-09-26-issues-237-238-239-verification-and-decisions.md b/docs/plans/2026-09-26-issues-237-238-239-verification-and-decisions.md index bf641d53..07114b43 100644 --- a/docs/plans/2026-09-26-issues-237-238-239-verification-and-decisions.md +++ b/docs/plans/2026-09-26-issues-237-238-239-verification-and-decisions.md @@ -2540,3 +2540,20 @@ in a throwaway repository (run 36451224053) showed that the job token pushes the email). Each firing is recorded as a `fired` line, so a routine that fails is fired again at most once; the Actions API's last successful run replaces a daily heartbeat commit; #243 closes. Design: `docs/archive/2026-09-28-superpowers-spec-upstream-watch-ledger-branch-design.md`. + +### V3 dashboard refresh implementation status (2026-09-29) + +The V3 branch at `19953b6d` delivers Addendum 3 Item 4's dashboard half: one header +Refresh control with Refresh, Refresh live, and Refresh machine choices; Reload only +re-reads the active view. `POST /api/refresh` starts the ordered operation; +`GET /api/refresh` reads the latest process-local state. GET system and Maintenance +routes reject retired scan-starting query arguments. ADR-0063 records the operation +identity and volatile single-flight boundary, and supersedes ADR-0048's two old +controls, ADR-0025 §5's GET-started scan rationale, and ADR-0045's GET scan trigger. +ADR-0044 and ADR-0053 now point to the explicit POST and header Refresh paths. + +D-15 keeps ADR-0048 Accepted with partial delivery: the human usability, screen-reader, +and cross-platform evaluation gates move to v5. The V4 offline retry change is on a +separate unmerged branch; this V3 source retains ADR-0063 Known limitations item 1. +This entry records V3 task 6c-5's source-bound documentation integration, not final +branch acceptance or completion of the separately dispatched live-view work. diff --git a/docs/plans/2026-09-28-remediation-program-v2.md b/docs/plans/2026-09-28-remediation-program-v2.md index 9ce59f9a..a69e52ad 100644 --- a/docs/plans/2026-09-28-remediation-program-v2.md +++ b/docs/plans/2026-09-28-remediation-program-v2.md @@ -4,9 +4,60 @@ ## Status -Active, not yet started. The §1 decision batch is awaiting the maintainer's answers. D-2's first -release (`4.0.0-alpha.60`, carrying #263's breaking changes) is being cut ahead of branch V1, per -D-2 option A. +Active under the maintainer-confirmed [develop execution plan](2026-09-28-remediation-v2-develop-execution.md) +(2026-09-28). V1 #267 and V2 #264/#266/#269 are delivered; #262 still needs residual +Windows timing evidence. #251 is done, and the alpha.60 release commit is on `main`. +Publication and global installation were not verified in the execution-plan baseline. +The historical decision batch and schedule below remain as scope/evidence references; +the approved execution section governs wherever their authority or timing differs. +At the 2026-09-29 closeout checkpoint, reviewed V1–V6 and main watcher reconciliation +are integrated through `develop@88ce597f444d487a34d9871a447cf8942f76f760` with all +13 develop CI checks passing. V7 guard and evidence units are independently accepted; +V7 shared integration, independent reviews and all nine local gates passed at +`795a0f7266de0bf6a04358359b75a1ad1b609e1d`; closeout PR CI and integration remain pending. +The [246-row scope matrix](../archive/2026-09-29-remediation-v2-scope-matrix.md), +[integration receipt](../archive/2026-09-29-remediation-v2-integration-evidence.md), +[ordered rulings](../archive/2026-09-29-remediation-v2-rulings.md), +[Windows evidence](../archive/2026-09-29-windows-ci-evidence.md) and +[AQE proof](../archive/2026-09-29-aqe-released-artifact-receipt.md) retain the limits. +The dated 19:20 UTC three-PR timing snapshot includes a 358-second Windows leg; +refresh this separate gate after closeout PR CI. +The final develop → main PR and human approval remain pending. This plan stays +active; no release, installation, real-store mutation, deletion or personal-memory +write follows from documentation completion. Attended time was not instrumented. + +### Approved execution and precedence + +The [confirmed execution plan](2026-09-28-remediation-v2-develop-execution.md) governs +branch sources, integration, releases, operations and completion. All Appendix A and B +rows remain in scope. In particular: + +- V1 and both V2 taxonomy PRs are already delivered; use their evidence and do not + replay their implementation. #262 remains open until its ten-run before/after + evidence is recovered or the missing evidence is explicitly reported. +- Bootstrap establishes `develop` from a verified current `main`. Every new feature + branch starts from current `develop`; every feature PR targets `develop`. The + controller may squash-merge only after required CI, including Windows, and + independent review pass. The aggregate `develop` → `main` PR is opened for human + review and left unmerged. References below to cutting branches from, merging into, + or fast-forwarding `main` for intermediate work mean `develop` for new work. +- D-1's cleanup language grants no automatic deletion. Preserve existing work and + unit-commit/evidence records. Real store operations, upstream submissions, paid + runs, and destructive cleanup retain their explicit approval gates. +- The alpha.60 commit is already on `main`; verify its actual published state before + any release decision. D-2's intermediate and close-out release schedule, including + alpha.61, and §2's install prerequisite are deferred until after human approval of + the final main PR and an independent release gate. Do not publish, install or sync + a new artifact merely to satisfy the historical schedule. +- D-3 through D-19 follow the confirmed execution plan's dispositions and current + evidence. D-3 and D-8 have landed. Execution uses bounded concurrency and exact + worktree ownership, rather than the original five-way Wave 2 schedule. +- For review readiness, reconcile every Appendix row to delivered evidence, an + approved disposition, a named open issue or a separately gated operation. Keep + the final main PR open. DoD §6's publication, global install, live-store work, + machine cleanup, future release dispatch and archive/memory operations are + operational completion gates after human review, not prerequisites to opening + that PR. Codex personal-memory updates require a direct user request. **Goal:** Finish everything left over from [remediation program v1](2026-09-26-remediation-program.md) in seven branches. The maintainer's attention goes into one decision sitting up front and a short list of named interrupts. v1 took about 28 attended hours; v2 aims for under 4 (§4 adds it up). @@ -272,7 +323,7 @@ One unit commit per line, test-first. The source text is the Branch 9 plan where 1. A relative `XDG_*` value is ignored (Branch 9 Task 6; B6a-6, B2-6). 2. Re-record seams for `x/daemon-gc.mjs` and `setup.mjs` (Task 8), plus the one for `x/host.mjs` `pick` (deferred 13a) (B6a-3). -3. `rufloMemoryLocation` names both reasons when the root and the folder are both unsuitable (deferred item 11). `inside()` stops treating equality as "inside", so a `TMPDIR` set to a tool folder is read correctly (B9-12, B0-15). +3. `rufloMemoryLocation` names both reasons when the root and the folder are both unsuitable (deferred item 11). `inside()` stops treating equality as "inside", so a `TMPDIR` set to a tool folder is read correctly (B9-12, B0-15). Implemented in V4 B3; isolated-branch review pending (`.superpowers/sdd/2026-09-28-follow-ups-v2/b3-report.md`). 4. N4, as D-4 decides (B9-3). 5. The stray scan walks dot folders below the root (B5-9, D-7). The real `~/.agentic-qe` home store is listed (B5-11). The `codex-mcp` hint stops suggesting AQE's broken Codex platform setup (agentic-qe#757) (B5-10). 6. `ruflo-components`: the applied-but-unverified row stops repeating the restart instruction (B0-21). The rows reading "partial — missing: Codex hooks" get a fix line once ruvnet/ruflo#3419 answers; if it is still unanswered at the pre-PR check, this part waits (LQ-2). diff --git a/docs/plans/2026-09-28-remediation-v2-develop-execution.md b/docs/plans/2026-09-28-remediation-v2-develop-execution.md new file mode 100644 index 00000000..c6ec6c33 --- /dev/null +++ b/docs/plans/2026-09-28-remediation-v2-develop-execution.md @@ -0,0 +1,283 @@ +# Remediation v2 develop execution plan + +> **For agentic workers:** Use superpowers:subagent-driven-development after maintainer +> confirmation. Write each branch's code-level plan with superpowers:writing-plans against +> its actual starting revision. This document governs program sequencing and authority. + +## Status + +**Active and confirmed** — the maintainer approved this execution plan on 2026-09-28. +Bootstrap and V1–V6 are integrated; V7 is in final validation. Approval covers isolated implementation work, +unit commits, feature PRs into `develop`, and conditional squash integration; the final +`develop` → `main` PR will be left open for human review. Releases, installation, real-data +operations and cleanup remain separately gated. Baseline inspected: `main@94890a00`. + +**2026-09-29 execution update:** Reviewed V1–V6 and main watcher reconciliation +are integrated through `develop@88ce597f444d487a34d9871a447cf8942f76f760` with all +13 develop CI checks passing. V7 guard and evidence units are independently accepted; +V7 shared integration, independent reviews and all nine local gates passed at +`795a0f7266de0bf6a04358359b75a1ad1b609e1d`; closeout PR CI and integration remain pending. +The [246-row scope matrix](../archive/2026-09-29-remediation-v2-scope-matrix.md), +[integration receipt](../archive/2026-09-29-remediation-v2-integration-evidence.md), +[ordered rulings](../archive/2026-09-29-remediation-v2-rulings.md), +[Windows evidence](../archive/2026-09-29-windows-ci-evidence.md) and +[AQE proof](../archive/2026-09-29-aqe-released-artifact-receipt.md) retain the limits. +The dated 19:20 UTC three-PR timing snapshot includes a 358-second Windows leg. +The controller will refresh that separate gate after closeout PR CI. +The final develop → main PR and human approval remain pending. This plan stays +active; no release, installation, real-store mutation, deletion or personal-memory +write follows from documentation completion. Attended time was not instrumented. + +**Goal:** Complete all remaining v2 remediation through feature PRs into `develop`, then +open one aggregate `develop` → `main` PR for human review. + +**Architecture:** One integration controller owns the merge queue and shared records. +Ruflo records coordination; bounded Codex workers implement in isolated worktrees. +Agentic QE supplies scoped quality work through its real installed tools. + +**Tech stack:** Node.js ES modules, existing zero-runtime-dependency CLI, GitHub Actions. + +**Spec:** [Remediation program v2](2026-09-28-remediation-program-v2.md), including every +Appendix A/B row and its referenced v1 plans, rulings and evidence. + +## Confirmed decisions and authority + +- Feature PRs target `develop`; the controller may squash-merge them after all required + CI and independent review pass. The final `develop` → `main` merge is human-owned. +- Make one conventional unit commit per independently verifiable task on feature branches. + Squashing deliberately produces one integration commit per PR. Preserve the unit commit + list and evidence mapping in the PR and ledger before any eventual branch cleanup. +- Defer all new releases and global installation until final main approval and the release + gate. The existing alpha.60 commit is already on main; publication and local installation + were not verified in this planning pass. Do not repeat or overwrite that release. +- Use recommendations D-3 through D-19, subject to current evidence. D-3 and D-8's work has + already landed. D-4 B uses successful version-bound memory-route evidence; D-5 A defers + the four unscoped #239 items; D-6 A drafts the AQE init issue; D-7 A fixes stray discovery. + D-9 A preserves both v5 branches and reserved ADR numbers. D-10–D-17 use recommended + deferrals/acceptances; D-18 A remains conditional on totals being unchanged; D-19 A labels + structured live events experimental. +- Real store merges, file deletion, upstream submissions and paid runs retain their + approval gates. Upstream approval is for exact sanitized text. No automatic worktree + deletion is inferred from approval to squash-merge PRs. +- Use program records for execution continuity. Updating Codex's personal memory requires + a direct user request; the original V7 auto-memory line does not override that boundary. +- This plan replaces the original program's main-targeted branch flow, intermediate + releases, unlimited Wave 2 fan-out, and cleanup-dependent pre-review completion criteria. + All other scope and gates remain in force unless explicitly reconciled by evidence. + +## Verified baseline and remaining uncertainty + +| Item | Current observation | Treatment | +| --- | --- | --- | +| V1 | #267 merged as `ab2fc5cb` | Verify residual timing evidence; do not repeat implementation | +| Windows stability | #262 open; #267 documents two passing runs at its head | Recover current run history; retain missing ten-run evidence as an open gate | +| V2 A/B | #264 and #266 merged; taxonomy plan marked Implemented; #269 archived it | Verify layouts/links and reuse completed work | +| D-8 | #251 merged as `ec749717` | Complete by existing evidence | +| Release | `94890a00` is the alpha.60 release commit | Verify status only; defer further release actions | +| GitHub PR inventory | No open PRs returned by the API | Recheck immediately before dispatch | +| Develop | No local or remote `develop` at planning baseline | Create under confirmed Phase A authority; verify current refs first | +| Existing work | Separate docs archive worktree; untracked `.harness/` in main | Preserve; inspect ownership and changes before dispatch | +| CI | `pull_request` enabled; push branches are `main` and `npm-kit` | Add develop push validation in the bootstrap PR | +| Ruflo | Guidance and memory calls succeeded; active claims empty | Prove scoped claims/runtime behavior before worker dispatch | +| AQE | Fleet status reports healthy, zero active agents/tasks | Verify each requested tool's real output against exact source | + +The original program's "not yet started" status and main-only flow are stale. Reconcile +them in the bootstrap PR, retaining historical evidence rather than replaying completed work. + +ADR-0063 is **Accepted**, updated 2026-09-29, with CLI and V3 dashboard delivery +recorded; V4 A3's offline retry limitation was refined and integrated in #275. +ADR-0048 is **Accepted**, updated 2026-09-28, with human evaluation gates +outstanding; V3 records their approved v5 deferral. +ADR-0060 is **Accepted**, updated 2026-09-29, with its delivered V6 contracts recorded. +Dedicated Cowork storage remains uncovered under #257; observed provider metadata is not +network attestation. + +## Phase A: establish the integration baseline + +Owner: controller. Dependencies: plan confirmation. Size: S, high confidence. + +- [x] Refresh GitHub and local refs; inventory worktrees, dirty paths, claims and open PRs. + Preserve unrelated work. Check the ignored v1 reconciliation, N-5 list, reviews and ledger + referenced by the spec are available; copy their relevant facts into scoped worker briefs. +- [x] Create `develop` from verified current main in a dedicated integration worktree. + Record base commit, user authority, controller and allowed actions in the execution ledger. +- [x] Create a bootstrap feature branch from develop. Commit this plan and reconcile the + source program's Status, decisions, branch targets and definition of done. +- [x] Add `develop` to `.github/workflows/ci.yml` push validation. Review other workflow + filters and concurrency keys so develop integration is tested without enabling publishing. + Check repository rules/check requirements; proposed changes to repository protection must + be explicit. Enforce the same merge gate in the controller even if develop is unprotected. +- [x] Validate docs layout, Markdown, links and workflow syntax; open bootstrap PR to develop. + Merge only after review and CI. Subsequent feature branches start at this integrated base. +- [x] Reconcile V1/V2 completion and #262 run evidence. Capture source revision, run URL, + OS/Node, job duration and result. Do not substitute a two-run observation for ten runs or + claim the old three-consecutive-PR-run rule was historically met without its evidence. + +Acceptance: develop exists; bootstrap checks pass; source scope is reconciled; remaining +rows have owners; existing work is preserved. Rollback: close an unmerged bootstrap PR; +after merge, use a reviewed revert PR rather than resetting shared history. + +## Phase B: execute independent workstreams + +All feature branches start from current `develop`, not main. Each has one writing owner, +one absolute worktree path, a code-level plan, exact path claims, dependency list and +acceptance evidence. Each plan re-reads source and inherited task references before coding. + +| Stream / branch | Deliverable | Dependencies and exclusive boundaries | Acceptance | +| --- | --- | --- | --- | +| V3 `feat/dashboard-refresh` | 6c-1–6c-5; POST refresh, single Refresh control, read-only GETs, vocabulary and client corrections; #256 decisions; #254 only with trace | Bootstrap; owns dashboard server/refresh contracts and client changes first | GETs cause no refresh side effects; refresh contracts, live-view behavior, UI and applicable contrast checks pass | +| V4 `fix/follow-ups-v2` | All A.1–A.4, B.1–B.13 and C.1–C.6, with evidence-based conditional dispositions | Bootstrap; CLI/status/setup/memory/discovery/upstream watch; shared dashboard or test helper paths require explicit handoff | JSON/exit contracts, offline retries, path handling, file IDs, cancellation on Windows, store discovery and upstream conformance proven | +| V5 `test/runner-hygiene` | Branch 9 Tasks 5, 7, 10–13; LQ-1/LQ-4; focused runner and reviewed cleanup inventory | V1 already integrated; bootstrap; owns `scripts/run-tests.mjs`, environment helpers and Chrome launch helper | Clean real-state tripwire and suite temp root; environment isolation, focused runner and cleanup ownership proven | +| V6 `fix/usage-accuracy` | Branch 7 items 1–5; B7-D1–D3; per-turn import exclusion; UA-5; all listed Branch 8 capture fixes; UA-4 | Bootstrap for core work; V3 integration for UI work; owns usage parsers/cache, session vocabulary and census counting | Reproductions using enumerated metadata/counts; imported turns excluded and later genuine turns counted; unknown provider stays unknown | + +Each stream includes every item named in the source program, not just the table summary. +V4 conditional B.13 uses a read-only plan/preview or a disposable-copy reproduction before +proposing changes to the user's configuration. Product fixes need no real-store mutation. +V4's temporary busy-rule/memory-routing CI probe is sandboxed and removed before its PR merges. +V6 reads the actual cache schema before making exactly one migration; the old plan's 25 → 26 +number must not overwrite a schema bump that has already landed. + +V6 Unit 21 implements static `statusLine` command classification for direct helper +invocations and treats shell wrappers, inline programs, and chains as `custom`. +It and the V6 UI are independently reviewed and integrated through #282. Usage schema +25 → 26 and the bounded accounting/coverage contracts are documented in the usage guide. + +### Scheduling and team shape + +The session has four slots total: controller plus at most three workers. Ruflo or AQE +dispatch must not create a second, uncounted fleet or expand spend or delegation limits. + +Start V3 and V6 core work, then V4 when exact file claims are disjoint. Queue V5 for the +next free slot. When a task is ready for review, release/pause that worker's writing scope +and use a free slot for an independent reviewer or AQE specialist. Do not keep three +implementers running while launching an additional reviewer. Queued streams may prepare +read-only briefs only when a slot is available. + +Use the real Ruflo coordination tools for task ownership/dependencies and actual Codex +workers for implementation. Use AQE for scoped test planning, risk/coverage assessment and +quality-gate work when its verified tool schemas can bind the worktree and source state. +A registration, success envelope or numerical score alone is not execution evidence. +If routing or isolation cannot be demonstrated, report the limitation before using a fallback. + +### Conflict prevention + +- One writer per worktree. Claims are exact paths plus resources/ports, not broad overlapping + globs. Recheck the work graph at each dispatch and integration boundary. +- V3 owns `src/lib/dashboard/client/*` first. V6 core excludes these files until V3 merges; + then V6 incorporates develop, reacquires paths and runs its UI task/review cycle. +- V3/V4 serialize changes to `src/lib/dashboard-server.mjs`, any shared Discovery modules, + shared tests and ADR-0063. V3 lands the refresh contract first wherever a consumer needs it. +- V1 → V5 orders `scripts/run-tests.mjs`. V5's helper changes require V4/V6 consumers to + incorporate develop and rerun affected tests. Stop dependent work if a contract changes. +- The controller alone integrates `docs/adr/README.md`, the shared decision log, program + ledger, manifests and lockfiles. Workers submit task-specific handoff text for these paths; + claim transfer happens only after their writing session ends. +- User guides and ADR-0063 get a named owner per task; different sections do not count as + separate file ownership. Update accepted ADR status/date/delivery notes in the same PR. +- Never message a running workflow agent. Use bounded task briefs and ledger handoffs. + Stop dependents on failed gates; independent work may continue in its own scope. + +## Phase C: integrate and finish V6's UI + +- [x] Review every unit commit after its RED/GREEN evidence. Allow at most five task fix + rounds, then raise the specific unresolved issue rather than looping indefinitely. +- [x] Freeze the candidate branch, integrate the latest develop and resolve conflicts in + its own worktree with no other writer. Run required full gates and whole-branch review. +- [x] Queue one feature PR merge at a time. Required evidence covers exact head/base SHAs; + a changed base invalidates the previous integration result and requires appropriate reruns. +- [x] Squash-merge only after required CI, including Windows, and independent review pass. + Verify the squash commit's tree equals the reviewed, develop-integrated feature tree. + Run/check develop CI before releasing dependent streams. +- [x] After V3 integrates, complete V6's labels across Usage, Projects, Maintenance and + Intelligence plus imported-copy counts in UI and `ak system`; keep these in the V6 PR. + +Task validation uses the narrowest meaningful tests first. Before V5 introduces `focus`, +use the existing guarded runner interface, for example +`node scripts/run-tests.mjs exec -- --test tests/kit/.test.mjs`. +After its integration use the verified `focus` interface in new briefs. + +Branch gates include the guarded unit/legacy suites, required UI suites, typecheck, lint, +complexity/Markdown/link checks, build and applicable policy/security checks. Use the +package scripts' actual underlying commands; no pnpm in symlinked-node_modules worktrees. +Keep 70/70/70 enforcement and report measured coverage against the 80% line target. +Fixtures and child processes use sandboxed homes and temp roots; state drift or leftovers +fail the gate. Preserve FORCE_COLOR scrubbing and platform redirects. Provider calls that +incur cost are excluded unless explicitly approved. + +## Phase D: V7 close-out and final human review + +This checkpoint records completed implementation and prepared issue drafts. The final +review PR will carry subsequent CI, timing and issue-action receipts; unchecked publication +gates below are deliberately not claimed ahead of those actions. A scheduled attempt at +`94890a00` returned seven HTTP 503 responses without parsed session URLs; successful routine +execution remains unverified. The later watcher fixes were validated synthetically and in +read-only PR preview, with no live trigger by this program. + +Dependencies: V3–V6 merged and develop green. Branch: `chore/v2-close-out`. Size: S/M. + +- [x] Apply the tree-wide comment-label guard after competing code edits finish. +- [x] Reconcile every original Appendix A/B row to an exact PR/evidence record, explicit + declined/superseded ruling, named open issue, or approved post-merge operation. No row + disappears because V1/V2 had already landed. +- [x] Complete #262's ten-run before/after analysis or keep its evidence gate visibly open. + Classify source/runtime changes in samples; never cherry-pick green runs into a false series. +- [x] Verify upstream watch behavior on the applicable branch/runtime. A first real release + dispatch may remain observation-pending until an actual release event; do not fabricate one. +- [x] Prepare issue updates (#239/#240/#254/#257/#262) with evidence. Distinguish integrated + on develop from shipped on main; close only when the issue's own completion condition holds. +- [x] Run docs alignment and archive completed branch plans with `scripts/docs-relocate.mjs`. + Keep this program's plan active while main approval and operational work remain pending. +- [ ] Open final `develop` → `main` PR with scope/decision matrix, feature PRs and unit commit + mappings, exact final source, tests, Windows evidence, release notes and known limitations. + Refresh main into develop first if necessary, review resulting changes, and rerun gates. + Leave the final PR open for human review; do not merge it. + +Review-ready means all code and documentation changes are integrated and validated, with +remaining human/external actions named. It does not claim publication, machine cleanup, +live-store consolidation or a future upstream event has happened. + +## Phase E: separately gated operational completion + +After human approval and merge to main, apply the independent release gate before tagging, +publishing or installing. Verify the existing published versions first; choose the next +version from current state, not this plan's stale alpha.61 schedule. Include breaking CLI +and dashboard changes in release notes. + +Complete approved install/sync/footprint verification against the released artifact. +For the real AQE stray store: exact preview, holder checks, explicit seed/import decision, +backup, disposable-copy import rehearsal, approved application/archive and receipt. +V5 supplies the reviewed literal-path temp inventory; deletion awaits approval. +Remove only approved program-owned worktrees/branches after confirming integration and +preserving unit-commit/evidence records. Keep develop for final review and until its retention +is decided; do not enforce the old "only main and v5 branches" rule during this workflow. + +## Review focus and failure handling + +1. Shared helpers and dashboard contracts changing under a concurrent consumer: exact claims, + explicit dependencies, serialized handoffs and consumer reruns. +2. A feature passing while its develop integration fails: latest-base validation, serialized + merges and develop push CI; stop dependents on failure. +3. Tests or real probes touching user data: guarded runner, isolated stores/homes, read-only + metadata/count observations and explicit operational gates. +4. Session relabeling or import fixes silently changing totals: fixture contracts plus + counts-only real-data reproductions and one controlled cache migration. +5. Completion claims exceeding evidence: source-bound results, no score-only approval, + explicit deferred human gates and a separate operational completion phase. + +On an unmerged failure retain the branch and evidence. After an integrated regression, +stop downstream promotion and submit a corrective or revert PR to develop. Never force-reset +develop/main, delete unrelated work, or weaken a gate to keep the schedule moving. + +## Planning receipt + +The maintainer confirmed this execution plan on 2026-09-28 and authorized isolated +implementation, unit commits, feature PRs into `develop`, and conditional squash +integration after required checks and review. The final `develop` → `main` PR is for +human review and remains unmerged. Release, installation, real-data operations and +cleanup retain their separate gates. The planning worktree started from `94890a00`; +the statements below describe the planning pass, before bootstrap implementation. +Current-state evidence came from local source/history and read-only GitHub API queries. +Ruflo memory search returned historical commands, not a current ownership decision. +Ruflo guidance/claims and AQE fleet status responded; no workers were launched. +Source grounding included `ruflo/plugins/ruflo-swarm/agents/coordinator.md` and +`agentic-qe/kb/capability-cards.md#agentic-qe` (the latter is a summary card, not runtime proof). diff --git a/docs/transcripts.md b/docs/transcripts.md index cb3a61b7..357ed048 100644 --- a/docs/transcripts.md +++ b/docs/transcripts.md @@ -30,29 +30,37 @@ checks run in the test suite --- +**Updated 2026-09-29.** The scan and reader share source selection and parse semantics, while +cross-file Claude charge ownership occurs only after discovery. Session-local transcript counts +need not equal aggregate accounted responses. Usage cache schema 26 additionally requires calendar, +source and semantic compatibility; the read-path figure below illustrates the original split, +not every current cache key. OpenCode child user turns remain readable but do not enter prompt +fingerprints. See the [current accounting contracts](usage-scorecard-metrics.md#current-accounting-and-cache-contracts) +for bounded imports, ownership, source coverage, reconciliation and warm-cache validation. + ## 1. Transcript stores Claude and Codex use JSONL files; OpenCode uses its SQLite session/message/part store. The kit reads source histories without rewriting them (transcripts are never -rewritten; rule 3 of the module header, `usage-index.mjs:22`): +rewritten; rule 3 of the module header, `usage-index.mjs:24`): | Host | Store | Discovered by | |---|---|---| -| Claude Code | `~/.claude/projects//.jsonl` | `listClaude` (`usage-index.mjs:374-388`) — exactly one level of project directories | -| Claude Code (subagent) | `~/.claude/projects///subagents/agent-.jsonl` | `listClaudeSubagents` (`usage-index.mjs:349-354`) — the one nested shape `listClaude` descends into | -| Codex CLI | `~/.codex/sessions///
/rollout--.jsonl` | `listCodex` (`usage-index.mjs:391-411`) — the `yyyy/mm/dd` tree walk | +| Claude Code | `~/.claude/projects//.jsonl` | `listClaude` (`usage-index.mjs:460-474`) — exactly one level of project directories | +| Claude Code (subagent) | `~/.claude/projects///subagents/agent-.jsonl` | `listClaudeSubagents` (`usage-index.mjs:435-440`) — the one nested shape `listClaude` descends into | +| Codex CLI | `~/.codex/sessions///
/rollout--.jsonl` | `listCodex` (`usage-index.mjs:477-497`) — the `yyyy/mm/dd` tree walk | | OpenCode | platform data root `opencode/opencode.db` (normally `~/.local/share/opencode/opencode.db` on Unix) | `usage-opencode.mjs` reads session/message/part rows with read-only SQLite queries | -Roots come from `defaultRoots()` (`usage-index.mjs:324-329`) and are injectable +Roots come from `defaultRoots()` (`usage-index.mjs:374-415`) and are injectable for tests. A malformed line is skipped, never fatal (`jsonLines`, -`usage-parsers.mjs:174-180` — one corrupt line must not cost a whole file). +`usage-parsers.mjs:183-191` — one corrupt line must not cost a whole file). A session's **delegated** work is a real transcript of its own, written beside the parent under `/subagents/`. Discovery is that one nested shape and no more — not a recursive walk — so a directory that is not a session-id directory with a `subagents` child contributes nothing rather than being crawled. Each such record takes a **namespaced** id, `/` -(`usage-index.mjs:354`), because Claude Code names every subagent file +(`usage-index.mjs:440`), because Claude Code names every subagent file `agent-.jsonl` and that stem is not unique across two parent sessions; an unnamespaced id would silently collide two unrelated records into one. §4.1 covers how a namespaced id is validated and resolved back to its file. @@ -68,22 +76,22 @@ evidence when their native records name it; Claude history normally leaves it un ### 1.1 Claude entry vocabulary Each line has a top-level `type`. The parser (`parseClaude`, -`usage-parsers.mjs:747-777`) reads: +`usage-parsers.mjs:822-871`) reads: | `type` | What the parser takes from it | |---|---| -| "ai-title" | The model-written session title (`usage-parsers.mjs:768`) — preferred over the first-prompt fallback | -| `user` | A user-**role** turn — which is *not* the same as "the human"; see §3. On a turn that passes isHumanPrompt, also its `permissionMode` — the session's permission posture, read on the person's own turn only (`usage-parsers.mjs:593-609`) — and the opening of the response-latency window | -| `assistant` | One content block of a model message: `model` id, the message's `usage` token counts (repeated on every block's line, so counted once per `message.id`, else `requestId`, last line winning), `tool_use` blocks (`usage-parsers.mjs:661-743`) | -| any | Side-band fields read regardless of type: `attributionSkill`/`attributionPlugin` (`usage-parsers.mjs:770-771`), `isSidechain` (`usage-parsers.mjs:772-773`), `cwd` for project derivation | +| "ai-title" | The model-written session title (`usage-parsers.mjs:852`) — preferred over the first-prompt fallback | +| `user` | A user-**role** turn — which is *not* the same as "the human"; see §3. On a turn that passes isHumanPrompt, also its `permissionMode` — the session's permission posture, read on the person's own turn only (`usage-parsers.mjs:628-644`) — and the opening of the response-latency window | +| `assistant` | One content block of a model message: `model` id, the message's `usage` token counts (repeated on every block's line, so counted once per `message.id`, else `requestId`, last line winning), `tool_use` blocks (`usage-parsers.mjs:708-818`) | +| any | Side-band fields read regardless of type: `attributionSkill`/`attributionPlugin` (`usage-parsers.mjs:854-855`), `isSidechain` (`usage-parsers.mjs:856-862`), `cwd` for project derivation | A real assistant completion also closes two pieces of per-entry evidence the transcript does not state outright. It **closes the latency window** the preceding human prompt opened, into one `noteLatencySample` call over the gap -between them (`usage-parsers.mjs:722-725`); and it **conditionally sets `ctxLastTokens`** to the +between them (`usage-parsers.mjs:779-782`); and it **conditionally sets `ctxLastTokens`** to the tokens actually in the model's window for that turn — fresh input plus what was served from cache — so the field always describes the last completion rather -than a running total (`usage-parsers.mjs:405-410`; the call site is at lines 658–668). That write is +than a running total (`usage-parsers.mjs:430-435`; the call site is at lines 658–668). That write is evidence-gated: an entry whose `message.usage` is absent decodes to all-zeros, and a zero is not a measurement of an empty context, so it must not overwrite a real prior value. Neither is a field Claude Code writes; both are derived, per @@ -93,7 +101,7 @@ An assistant entry with `isApiErrorMessage: true` is a **local placeholder** Claude Code writes when a request dies before a real completion (connection drop, rate limit, auth failure — `model: ""`, all-zero usage). It is real engaged time but not a model attempt: counted as an *exception*, never -pushed into `models`, priced, or counted as a response (`usage-parsers.mjs:700-735`; the full story is +pushed into `models`, priced, or counted as a response (`usage-parsers.mjs:757-792`; the full story is [`usage-scorecard-metrics.md`](usage-scorecard-metrics.md) §10). It is not a latency sample either — the pending window is deliberately left open, so the first *real* completion that eventually follows is what gets timed. @@ -101,22 +109,22 @@ first *real* completion that eventually follows is what gets timed. ### 1.2 Codex entry vocabulary Codex rollout lines carry `type` + `payload`. The parser (`parseCodex`, -`usage-parsers.mjs:1183-1240`) reads: +`usage-parsers.mjs:1355-1442`) reads: | `type` / `payload.type` | What the parser takes from it | |---|---| -| `session_meta` | Authoritative session id, `cwd`, and `thread_source` — the FIRST such line in the file wins for all three, AND for `inferenceProvider`/`providerProvenance` too (`usage-parsers.mjs:831-841`, gate; `:767-784`, why); a subagent rollout replays its PARENT thread's own session_meta line later in the same file, and a later-wins rule let that relabel the record `subagent`→`user` and re-key its id to the parent's — `"subagent"` marks a delegated thread whose rollout may open with its parent's replayed history; events before the replay boundary (`codex-replay.mjs`) count for nothing, so the subagent's own tokens are counted and the replay is not, and the record stays flagged `subagent` (`usage-parsers.mjs:1188-1195`; `usage-scorecard-metrics.md` Appendix A, Bug B) | -| `turn_context` | The model id in effect from this point on, plus `approval_policy` (a string) and `sandbox_policy` (an **object** keyed `.type`, e.g. `{"type":"danger-full-access"}`) — the permission posture, last evidence winning, since a session may renegotiate mid-run (`usage-parsers.mjs:843-864`) | -| `event_msg` → `token_count` | A **cumulative** usage snapshot, turned into its increase over the previous one and booked on the event's local day under the model of the turn in effect; a snapshot lower than its predecessor starts a new segment (the host's counter restarted), so every segment counts. A replayed snapshot only advances the running total (`usage-parsers.mjs:898-909`) | -| `event_msg` → `task_started` | `model_context_window` — denominator-only compatibility evidence, not proof of a paired input/window sample — and the turn's start time (`usage-parsers.mjs:915-934`) | -| `event_msg` → `task_complete` | The host's own `duration_ms` for the turn, taken as a latency sample only when no prompt-to-response gap already covered it; a non-null `error` counts as an exception (`usage-parsers.mjs:936-954`) | -| `event_msg` → `turn_aborted` | An explicit interrupt: counted in `aborts`, and it clears both latency states so an unanswered prompt is never timed against a later, unrelated response (`usage-parsers.mjs:1048-1089`) | +| `session_meta` | Authoritative session id, `cwd`, and `thread_source` — the FIRST such line in the file wins for all three, AND for `inferenceProvider`/`providerProvenance` too (`usage-parsers.mjs:925-941`, gate; `:1343-1373`, why); a subagent rollout replays its PARENT thread's own session_meta line later in the same file, and a later-wins rule let that relabel the record `subagent`→`user` and re-key its id to the parent's — `"subagent"` marks a delegated thread whose rollout may open with its parent's replayed history; events before the replay boundary (`codex-replay.mjs`) count for nothing, so the subagent's own tokens are counted and the replay is not, and the record stays flagged `subagent` (`usage-parsers.mjs:1360-1367`; `usage-scorecard-metrics.md` Appendix A, Bug B) | +| `turn_context` | The model id in effect from this point on, plus `approval_policy` (a string) and `sandbox_policy` (an **object** keyed `.type`, e.g. `{"type":"danger-full-access"}`) — the permission posture, last evidence winning, since a session may renegotiate mid-run (`usage-parsers.mjs:943-964`) | +| `event_msg` → `token_count` | A **cumulative** usage snapshot, turned into its increase over the previous one and booked on the event's local day under the model of the turn in effect; a snapshot lower than its predecessor starts a new segment (the host's counter restarted), so every segment counts. A replayed snapshot only advances the running total (`usage-parsers.mjs:1003-1022`) | +| `event_msg` → `task_started` | `model_context_window` — denominator-only compatibility evidence, not proof of a paired input/window sample — and the turn's start time (`usage-parsers.mjs:1028-1047`) | +| `event_msg` → `task_complete` | The host's own `duration_ms` for the turn, taken as a latency sample only when no prompt-to-response gap already covered it; a non-null `error` counts as an exception (`usage-parsers.mjs:1049-1076`) | +| `event_msg` → `turn_aborted` | An explicit interrupt: counted in `aborts`, and it clears both latency states so an unanswered prompt is never timed against a later, unrelated response (`usage-parsers.mjs:1170-1218`) | | `event_msg` → `user_message` | A legacy-format prompt CANDIDATE — Codex does not route tool output through this event, but the text still needs the human-prompt gate below before it counts | | `event_msg` → `agent_message` | A legacy-format model response | | `event_msg` → `item_completed` → `UserMessage` | A current-format prompt candidate; text blocks use the observed lowercase `text` discriminator; also gated below | | `event_msg` → `item_completed` → `AgentMessage` | A current-format model response; text blocks use the observed uppercase `Text` discriminator | -| Human-prompt gate (`isCodexHumanMessage`, `usage-parsers.mjs:970-974`) | Codex carries no discipline of its own for telling a typed prompt apart from harness output or a mirrored cross-host envelope replayed into the rollout rather than typed there. Reuses "HARNESS_OUTPUT_RE" verbatim (Claude's own envelope markers reproduce byte-for-byte inside a mirrored rollout) plus two Codex-specific machine markers (`CODEX_MACHINE_ENVELOPE_RE`, `usage-parsers.mjs:956-973`): a `/` with one slash, where the parent half reuses `VALID_ID`'s own charset and the child half must match the real on-disk `agent-…` shape. The namespaced grammar is a **narrowing** of the plain one, never a loosening: both are the same path-traversal guard, and a traversal shape is rejected at either tier. -2. **Locate by id** across both roots (`locate`, `usage-index.mjs:1031-1050`), +2. **Locate by id** across both roots (`locate`, `usage-index.mjs:1227-1246`), consulting the scan cache when present but never requiring it — `readSession` works with no prior buildIndex. A namespaced id resolves through this call: `locateSubagent(nested.parentId, nested.stem, r.claude, id)` - (`usage-index.mjs:1048`), which builds the + (`usage-index.mjs:1244`), which builds the nested path from the two **already-validated capture groups** rather than from raw request text. -3. **Realpath containment** (`usage-index.mjs:1122-1135`) — the resolved file +3. **Realpath containment** (`usage-index.mjs:1318-1331`) — the resolved file must live under a transcript root *after* `realpathSync` collapses symlinks; a symlink planted inside a root pointing at `/etc/anything` passes a lexical `startsWith` but fails this. Roots are realpath'd too so a symlinked dotfiles setup still works. -4. **Size cap** — `MAX_SESSION_BYTES` (64 MB, `usage-index.mjs:195`): a +4. **Size cap** — `MAX_SESSION_BYTES` (64 MB, `usage-index.mjs:241`): a transcript is read whole and JSON-expands ~5×, so an unbounded read is a memory-amplification primitive. Oversized reads as unavailable, not risky. ### 4.2 Parse and price The file is parsed with `withTurns: true` by the provider's parser -(`usage-index.mjs:1157-1164`), and `meta` is assembled by `sessionPayload` -(`usage-aggregate.mjs:1324-1361`). Its call builds a narrower subset of the Sessions +(`usage-index.mjs:1330-1339`), and `meta` is assembled by `sessionPayload` +(`usage-aggregate.mjs:1479-1518`). Its call builds a narrower subset of the Sessions view fields: `prompts`, `responses`, `exceptions`, `sidechain`, `threadSource`, `models`, `tools`, `skill`/`plugin`, worktree — plus a `cost` priced from the same per-model usage rows `aggregate()` uses. OpenCode recorded row cost wins over @@ -421,11 +429,11 @@ never renames a retained session model, changes historical token pricing, or rew ### 4.3 Mask, then truncate — both marked, differently -Every turn body is passed through `maskSecrets` (`usage-aggregate.mjs:168-173` — the +Every turn body is passed through `maskSecrets` (`usage-aggregate.mjs:171-176` — the configured secret shapes) **server-side, before serialization**, then length-capped at `MAX_TURN_CHARS` (40,000, -`usage-aggregate.mjs:73`) with the marker appended at the truncation call -("originalChars is measured", `usage-aggregate.mjs:1346-1355`). Two invariants: +`usage-aggregate.mjs:76`) with the marker appended at the truncation call +("originalChars is measured", `usage-aggregate.mjs:1503-1512`). Two invariants: * **Presence is the signal.** `truncated`/`originalChars` are emitted only when the slice fired, so a complete turn cannot be misread as abridged. @@ -604,7 +612,7 @@ was wrong before, for the curious. `isHumanPrompt` once counted `harness-output` envelopes as human prompts — 32 claimed vs 20 real on the reference session. Cached session records carried the inflated counts, hence the wholesale `SCHEMA_VERSION` 5 cache - invalidation ("no longer count as human prompts", `usage-index.mjs:70-73`). + invalidation ("no longer count as human prompts", `usage-index.mjs:78-81`). * **Session expander fields shipped but unrendered.** The per-session fields §6.1's expander now renders (classification `basis` + confidence, the token split, flags) once travelled on the wire and rendered nowhere. @@ -612,7 +620,7 @@ was wrong before, for the curious. assembled `meta` left cost undefined, and `fmtUsd(undefined)` renders the truthy string `"$0.00"` — a fixed-looking zero on a panel whose whole subject is cost. `meta.cost` is now priced via this call: `sessionCost(rec, deps)` - (`usage-aggregate.mjs:1318`) — over the same per-model usage rows aggregate() reads. + (`usage-aggregate.mjs:1473`) — over the same per-model usage rows aggregate() reads. * **Aggregate-side incidents** (the v4/v5 cache bumps, the Codex parsing defects) are recorded in `usage-scorecard-metrics.md` Appendix A. diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index 576d9f40..58d06d53 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -151,15 +151,18 @@ up to six minutes each: ak status --refresh=live --only learning # trains a cycle in an isolated dir; asserts patterns persist to disk ak status --refresh=live --only aqe # agentic-qe genuinely on ruvector (no FsyncFailed) ak status --refresh=live --only harvest # Ruflo's learning-write path (post-task + distill) in an isolated store -ak status --refresh=live --only memory-routes # CLI/MCP routing observation, remembered as the memory check +ak status --refresh=live --only memory-routes # CLI round trip remembered as memory; routing observation remembered separately ak status --refresh=live --only learning,harvest,aqe,memory-routes,security,deja-vu,providers,mcp,aqe-embedding ``` If `ak status --refresh=live --only aqe` warns that RVF is held by another live process, another AQE process (usually the AQE MCP server in an open Claude Code session) owns the store. -That is contention, not a storage failure, even though agentic-qe 3.14.3 also prints -`FsyncFailed` in this case ([#240](https://github.com/pacphi/agentic-kit/issues/240)). -A `FsyncFailed` without the live-owner lines still fails verification. +Released agentic-qe 3.14.4 passed native macOS and Linux live-owner probes with `LockHeld` +and no `FsyncFailed` ([#240](https://github.com/pacphi/agentic-kit/issues/240)). +The old 3.14.3 sequence also printed `FsyncFailed`; ak now treats any `FsyncFailed` or +`0x0303` as an RVF failure, even alongside live-owner lines. An ordinary live lock stays +busy with SQLite fallback observed; owner health and RVF integrity remain unverified. +Native Windows AQE live-lock conformance has not been run. ## Known upstream gaps (not fixable by sync) diff --git a/docs/upgrading.md b/docs/upgrading.md index c6e61c49..0b93b93b 100644 --- a/docs/upgrading.md +++ b/docs/upgrading.md @@ -150,7 +150,7 @@ When the ChatGPT desktop app imports a Claude Code transcript, it saves a copy a Project discovery no longer counts these copies: they give a folder no Codex host and no Desktop origin, and System says how many it set aside. A snapshot taken before this change still holds the old hosts and origins, so the Footprint snapshot schema advances to v8. This build reports a v7 -snapshot as unreadable until you run **Full scan** in System or `ak system --refresh=machine`. It +snapshot as unreadable until you run **Refresh machine** in System or `ak system --refresh=machine`. It is never shown under the new rule. See [ADR-0060](adr/0060-session-surface-initiator-and-product-names.md) §3. ## 2026-09-27: Ruflo support window @@ -389,7 +389,7 @@ working context for the already-ranked top-N rows; it does not read prompts, tit System > Sessions renders the identity as one two-line transcript link: localized date/time first, then a shortened opaque native ID. Focus or hover discloses the original filename, full native ID, and detailed localized time with timezone. If an older snapshot or host has no declared opening -instant, the measured mtime is explicitly labeled **Last active**. Run **Full scan** or +instant, the measured mtime is explicitly labeled **Last active**. Run **Refresh machine** or `ak system --refresh=machine` to populate native identity for an existing snapshot; no configuration or payload migration is required. @@ -403,7 +403,7 @@ such as the user home from triggering several hundred thousand unrelated filesys Because that population is narrower than the v6 measurement contract, the Footprint snapshot schema advances to v7. A v6 snapshot is reported as unreadable by this build until the next explicit -**Full scan** or `ak system --refresh=machine`; it is never silently reinterpreted. +**Refresh machine** or `ak system --refresh=machine`; it is never silently reinterpreted. ## 2026-09-03: System Catalog snapshot v6 @@ -452,9 +452,10 @@ user's agentic-kit state directory. Existing System snapshot files remain read-o Catalog schema v4 is still refreshed with `ak system --refresh=machine`. `ak sync` neither selects nor executes Maintenance findings. -Browser refresh now reads the saved Maintenance report without polling providers. Use **Refresh -evidence** or `ak maintain --refresh` for current provider/version evidence. A successful System -deep rescan also chains one Maintenance scan after the snapshot is persisted. +Browser **Reload** reads the saved Maintenance report without polling providers. Use the header's +**Refresh** control with **Refresh** selected, or `ak maintain --refresh`, for current provider/version +evidence. A successful **Refresh machine** operation persists a new System snapshot before +updating Maintenance. The first provider set is intentionally narrower than the inventory. Claude plugin disable, update, and remove; exact Codex plugin/MCP removal; exact receipt-owned skill archive; one bounded diff --git a/docs/upstream-watch.md b/docs/upstream-watch.md index 5857fc88..39a49d13 100644 --- a/docs/upstream-watch.md +++ b/docs/upstream-watch.md @@ -101,18 +101,23 @@ Every command also takes `--concurrency <1-16>` (default 4) and `--registry //subagents/*.jsonl`. A directory that is not a session-id dir with a `subagents` child — Claude Code's own `memory` dir, say — contributes nothing rather than being crawled. Each subagent record takes a -**namespaced** id, `/` (`usage-index.mjs:354`), because Claude +**namespaced** id, `/` (`usage-index.mjs:440`), because Claude Code names every such file `agent-.jsonl` and that stem is not guaranteed unique across two parents; an unnamespaced id would silently collide two unrelated subagent records into one. `locateSubagent` -(`usage-index.mjs:1015-1034`) resolves that id back to the nested path when a +(`usage-index.mjs:1211-1230`) resolves that id back to the nested path when a reader opens the session, building the candidate path from the two validated capture groups rather than from raw request input. -The parsers are `parseClaude` (`usage-parsers.mjs:742-777`) and `parseCodex` -(`usage-parsers.mjs:1183-1240`). They normalize raw JSONL bytes; project evidence also consults the local filesystem. +The parsers are `parseClaude` (`usage-parsers.mjs:817-871`) and `parseCodex` +(`usage-parsers.mjs:1355-1442`). They normalize raw JSONL bytes; project evidence also consults the local filesystem. Missing-time fallback paths can consult the clock. Their output is local evidence, not a network response or an invoice. Nothing in this transcript pipeline calls a provider API or a billing endpoint; **no transcript @@ -196,22 +289,27 @@ turns`. **Formula:** ```text -sessions = count of session records with responses > 0 AND end >= cutoff -responses = Σ over included sessions of session.responses +eligible = responses > 0 OR positive Codex component usage OR retained OpenCode observations +current.sessions = count of eligible records with non-null end >= cutoff (no upper bound) +windowStart = now - days × DAY_MS +previous.sessions = count of eligible records with windowStart - days × DAY_MS <= end < windowStart +responses = Σ over included sessions of accountedResponses (else responses) ``` **Source:** -- Filter: a parsed record with zero assistant turns is dropped entirely — "no - assistant turn → not a session" (`usage-aggregate.mjs:888-899`) — and a record whose - last activity falls outside the requested window is dropped too - (`usage-aggregate.mjs:898-899`). +- Filter: `buildSessionRows` (`usage-aggregate.mjs:970-982`) accepts response-bearing records, + positive Codex component usage or retained OpenCode observations. An unknown end or an end + outside the requested window is excluded. Refused OpenCode acquisitions retain uncertainty + separately without becoming ordinary session rows. The current projection supplies no + upper bound; `previousWindow` (`usage-aggregate.mjs:1274-1281`) supplies the exclusive + `windowStart` upper bound and derives both bounds from displayed `days` and `now`. - `responses` accumulation: Claude increments once per API message id — every transcript line of one message counts once, the last line's usage winning - (`usage-parsers.mjs:661-696`); Codex increments per `agent_message` event - (`usage-parsers.mjs:1021-1025`). -- Totals: `totals.responses += s.responses` per included session -(`usage-aggregate.mjs:928`). + (`usage-parsers.mjs:708-753`); Codex increments per `agent_message` event + (`usage-parsers.mjs:1143-1147`). +- Totals: `totals.responses += s._accountedResponses` per included session +(`usage-aggregate.mjs:1050`). - Render: `kpi("sessions", fmtNum(t.sessions), fmtNum(t.responses)+" assistant turns", "")` (`dashboard/client.mjs`). @@ -332,22 +430,22 @@ same sentence fingerprints identically whichever host recorded it: | Symbol | Location | Notes | |---|---|---| -| `normalizePromptText` | `src/lib/usage-parsers.mjs:261` | lowercased, whitespace-collapsed, trailing punctuation stripped | -| `promptFingerprint` | `src/lib/usage-parsers.mjs:296` | the `{h, t, th}` hash/count/token-hash triple | -| `promptShape` | `src/lib/usage-parsers.mjs:342` | the `q`/`o` flags, anchored on the question and persona-opener rules below | -| `QUESTION_WH_RE` | `src/lib/usage-parsers.mjs:311` | one of two rules the `q` flag checks | -| `QUESTION_AUX_RE` | `src/lib/usage-parsers.mjs:312` | the other | -| `PERSONA_OPENER_RE` | `src/lib/usage-parsers.mjs:318` | what the `o` flag checks | -| `notePromptFingerprint` | `src/lib/usage-parsers.mjs:359` | records one fingerprint, or counts overflow past the caps below | -| `MAX_PROMPT_FPS` | `src/lib/usage-parsers.mjs:240` | the per-session fingerprint cap | -| `MAX_TOKEN_HASHES` | `src/lib/usage-parsers.mjs:255` | the per-fingerprint token-hash cap | +| `normalizePromptText` | `src/lib/usage-parsers.mjs:286` | lowercased, whitespace-collapsed, trailing punctuation stripped | +| `promptFingerprint` | `src/lib/usage-parsers.mjs:321` | the `{h, t, th}` hash/count/token-hash triple | +| `promptShape` | `src/lib/usage-parsers.mjs:367` | the `q`/`o` flags, anchored on the question and persona-opener rules below | +| `QUESTION_WH_RE` | `src/lib/usage-parsers.mjs:336` | one of two rules the `q` flag checks | +| `QUESTION_AUX_RE` | `src/lib/usage-parsers.mjs:337` | the other | +| `PERSONA_OPENER_RE` | `src/lib/usage-parsers.mjs:343` | what the `o` flag checks | +| `notePromptFingerprint` | `src/lib/usage-parsers.mjs:384` | records one fingerprint, or counts overflow past the caps below | +| `MAX_PROMPT_FPS` | `src/lib/usage-parsers.mjs:265` | the per-session fingerprint cap | +| `MAX_TOKEN_HASHES` | `src/lib/usage-parsers.mjs:280` | the per-fingerprint token-hash cap | | `PROVENANCE_TAGS` | `src/lib/usage-provenance.mjs:21` | the closed four-tag vocabulary | | the ordered provenance rules | `src/lib/usage-provenance.mjs:33-77` | matched against, in order, to resolve a tag | | `provenanceOf` | `src/lib/usage-provenance.mjs:93` | resolves one turn's provenance tag | Wired on the Claude path where userTurnKind is called — -`src/lib/usage-parsers.mjs:600-621`; on the Codex path inside -`handleCodexUserMessage` — `src/lib/usage-parsers.mjs:985-1006`; on the +`src/lib/usage-parsers.mjs:635-656`; on the Codex path inside +`handleCodexUserMessage` — `src/lib/usage-parsers.mjs:1107-1128`; on the opencode path inside `recordUserMessage` — `src/lib/usage-opencode.mjs:184-197` **What this does not model:** @@ -455,12 +553,12 @@ this day carried the fingerprint layer", so a zero here is *measured*. | Symbol | Location | |---|---| -| `TAP_MAX_TOKENS` | `src/lib/usage-aggregate.mjs:287` | -| the baseline window and floor | `src/lib/usage-aggregate.mjs:294` | -| `v16Projection` (per session) | `src/lib/usage-aggregate.mjs:330` | -| `foldSessionPrompts` | `src/lib/usage-aggregate.mjs:369` | -| `sealPromptHosts` | `src/lib/usage-aggregate.mjs:387` | -| `buildPromptBaselines` | `src/lib/usage-aggregate.mjs:417` | +| `TAP_MAX_TOKENS` | `src/lib/usage-aggregate.mjs:290` | +| the baseline window and floor | `src/lib/usage-aggregate.mjs:297` | +| `v16Projection` (per session) | `src/lib/usage-aggregate.mjs:333` | +| `foldSessionPrompts` | `src/lib/usage-aggregate.mjs:374` | +| `sealPromptHosts` | `src/lib/usage-aggregate.mjs:392` | +| `buildPromptBaselines` | `src/lib/usage-aggregate.mjs:422` | | `detectSupervisionTapShare` | `src/lib/usage-insights.mjs:780` | | `detectHeadlessShare` | `src/lib/usage-insights.mjs:808` | | `detectHostPromptAsymmetry` | `src/lib/usage-insights.mjs:845` | @@ -557,7 +655,8 @@ call has no per-token rate, and the unknown-model fallback would invent one (1M input + 1M output tokens would read $18). Those messages count toward `unpricedMessages`, add no dollars, and are left out of the cache-saving estimate; the gap is coverage the reader can see, not a silent $0. A local -provider that reports a cost, including 0, stays an observed figure. A +provider that reports a cost, including 0, stays an observed figure. For nonlocal or unknown +providers, reported zero with positive tokens is unpriced. A custom-named local provider cannot be recognised from its id and keeps the fallback rate. @@ -598,7 +697,7 @@ lexicographically so no `Date` parsing is involved and the module stays clock-free. `foldSessionUsageRow` passes each usage row's own `day` to `costOf` -(`usage-aggregate.mjs:739-741`), which +(`usage-aggregate.mjs:746-748`), which it already has because rows are keyed by `(day, model)`. **This is the whole point:** tokens metered in August must still read as August's rate when the panel is opened in December. Pricing by *today's* date instead would restate a @@ -668,9 +767,9 @@ tokens = input + output + cacheRead + cacheWrite (summed across all rows in wi ``` **Source:** `t.tokens` from `totals`, accumulated per row at -`usage-aggregate.mjs:765` (`rowTokens = row.input + row.output + row.cacheRead + +`usage-aggregate.mjs:775` (`rowTokens = row.input + row.output + row.cacheRead + row.cacheWrite`) and rolled into `totals.tokens` via `addTo` -(`usage-aggregate.mjs:656-665`). Rendered with `fmtTok()` +(`usage-aggregate.mjs:663-672`). Rendered with `fmtTok()` (`dashboard/client.mjs`): `≥1e9` → `"X.XB"`, `≥1e6` → `"X.XM"`, `≥1e3` → `"X.XK"`, else the rounded integer. @@ -684,7 +783,7 @@ per row is **gross input minus cached input** — Claude's parser reads `cache_read_input_tokens` and `cache_creation_input_tokens` as separate fields the provider already reports separately (`telemetry-records.mjs:216-224`); Codex's parser subtracts `cached_input_tokens` from `input_tokens` explicitly -at this call site (`usage-parsers.mjs:1132-1183`, `input: Math.max(0, gross - cacheRead)`) because +at this call site (`usage-parsers.mjs:1296-1355`, `input: Math.max(0, gross - cacheRead)`) because Codex's own `input_tokens` field **includes** cached tokens and would double-count them against the separately-reported `cacheRead` figure if left as-is. This is asserted by test: @@ -775,25 +874,25 @@ session data, and each needs its own fix: human, or genuinely idle) donates its *entire* idle stretch to the span, even though no work happened during it. Fix: split each session into active sub-intervals wherever the gap between two consecutive timestamps - exceeds `IDLE_GAP_MS` (15 minutes, `usage-parsers.mjs:29`), then union + exceeds `IDLE_GAP_MS` (15 minutes, `usage-parsers.mjs:32`), then union *those* sub-intervals — this is `engagedSeconds`. **Source:** -- `mergeIntervals()` (`usage-aggregate.mjs:40-65`) — the pure union primitive, +- `mergeIntervals()` (`usage-aggregate.mjs:43-68`) — the pure union primitive, sorts intervals and merges any two that are "overlapping OR exactly touching" - (`s <= curEnd`, `usage-aggregate.mjs:56`), returning total covered seconds + (`s <= curEnd`, `usage-aggregate.mjs:59`), returning total covered seconds rounded to the nearest second. -- `activeIntervals()` (`usage-parsers.mjs:469-487`) — splits one session's +- `activeIntervals()` (`usage-parsers.mjs:494-512`) — splits one session's sorted timestamp list into sub-intervals wherever a gap exceeds IDLE_GAP_MS; "a run of one timestamp yields a zero-length interval and so - contributes nothing" (comment, `usage-parsers.mjs:469-475`). + contributes nothing" (comment, `usage-parsers.mjs:494-500`). - Aggregation, each its own call to `mergeIntervals`: `engagedSeconds` over - every session's active sub-intervals (`usage-aggregate.mjs:1064-1069`); + every session's active sub-intervals (`usage-aggregate.mjs:1174-1179`); `spanUnionSeconds` over whole spans instead - (`usage-aggregate.mjs:1044-1068`); spanMs is a running sum of - "s._span[1] - s._span[0]" across the loop (`usage-aggregate.mjs:963-981`), - finalized into `spanMinutes` (`usage-aggregate.mjs:1064-1068`). + (`usage-aggregate.mjs:1154-1178`); spanMs is a running sum of + "s._span[1] - s._span[0]" across the loop (`usage-aggregate.mjs:1045-1076`), + finalized into `spanMinutes` (`usage-aggregate.mjs:1174-1178`). - Render: `fmtHours()` (`dashboard/client.mjs`, `≥10h` rounds to the nearest hour, else one decimal place) and `fmtMins()` (`dashboard/client.mjs`, `≥60min` rounds to hours, else whole @@ -846,12 +945,12 @@ byDay[day].sessionsActive = count of distinct sessions with any usage row that d **Source:** the day key is the row's own `row.day`, computed once at parse time as **local calendar day**, not UTC -(`usage-parsers.mjs:35`/`usage-parsers.mjs:1174` call `localDay(at)`) — so a +(`usage-parsers.mjs:38`/`usage-parsers.mjs:1332` call `localDay(at)`) — so a session that runs from 23:58 local to 00:05 local has its session count attributed to the day its *first* usage row landed on (test: `tests/kit/usage-index.test.mjs:738`, "a session that opens before midnight is counted on its first billed day"). Accumulation, at this call: `dayBucket(byDay, -row.day)` then `d.cost = round(d.cost + rowCost)` (`usage-aggregate.mjs:764-771`). Bar height: +row.day)` then `d.cost = round(d.cost + rowCost)` (`usage-aggregate.mjs:774-781`). Bar height: `h = maxDay ? max(2, cost/maxDay*100) : 2` (`dashboard/client.mjs`) — every non-empty day gets a visually nonzero bar (floor of 2%), so a very cheap day is never rendered as invisible. @@ -879,11 +978,11 @@ renders "no sessions in window" instead of zeroed figures (`dashboard/client.mjs`). **Formula:** identical aggregation to every other bucket -(`byHost[s.host]`, populated via `addTo()` (`usage-aggregate.mjs:667-676`), - called once per session at this call: `usage-aggregate.mjs:968-993`), keyed by the literal string +(`byHost[s.host]`, populated via `addTo()` (`usage-aggregate.mjs:674-683`), + called once per session at this call: `usage-aggregate.mjs:1050-1103`), keyed by the literal string `"claude"` or `"codex"` assigned at parse time (this call: `blankSession(id, 'claude')` / `blankSession(id, 'codex')`, -`usage-parsers.mjs:197-225`, `:1220`, `parseClaude`/`parseCodex` entry points). +`usage-parsers.mjs:216-250`, `:1383`, `parseClaude`/`parseCodex` entry points). OpenCode's SQLite reader builds the same record shape and contributes a third host key. @@ -895,22 +994,27 @@ The historical Codex incidents in Appendix A are examples, not an exhaustive dia **Two identity maps, with separate evidence.** The aggregate buckets window spend by two identities, and reading one as the other is the -mistake this split exists to prevent (`usage-aggregate.mjs:913`, -`usage-aggregate.mjs:941-942`): +mistake this split exists to prevent (`usage-aggregate.mjs:994`, +`usage-aggregate.mjs:1022-1023`): - **`byHost`** — the execution host: which CLI wrote the transcript (`claude`, `codex`, `opencode`). This is what the host cards render. It is a fact about the file's provenance on disk, and it proves nothing about which vendor served the tokens. -- **`byProvider`** — the inference-provider string **as recorded**, ungated: - `s.provider ?? 'unknown'`. This map keeps its historical name and its - historical shape for callers that want the raw string, whatever its - evidence. A session that recorded no provider keys to `'unknown'`. - -`byProvider` uses the served session row's recorded inference provider, or `unknown`. -Codex session metadata/turn context and OpenCode assistant `providerID` can establish -that value with observed provenance; Claude history normally leaves it absent. The -Scorecard UI does not rank this map, but session details expose provider/provenance. +- **`byProvider`** — OpenCode response, token and cost totals are split by the provider on each + assistant usage row, with missing row providers in `unknown`. `foldSessionUsageRows` + (`usage-aggregate.mjs:791-815`) accumulates these per-provider shares; the second pass applies + them at this call: `addTo(bucket(byProvider, provider), { ...s, ...usage })` + (`usage-aggregate.mjs:1083-1092`). Each response/token/cost share lands once, but one session + counts once under **each** provider it used. Provider session counts therefore need not sum + to the overall session count; they are not disjoint session populations. + +When there are no per-row provider shares, the fallback uses the session's recorded provider, +`s.provider ?? 'unknown'`, with its accounted response count at this call: `addTo` +(`usage-aggregate.mjs:1091`). +Codex session metadata/turn context and OpenCode assistant `providerID` supply observed metadata; +Claude history normally leaves the session provider absent. These facts are not network +attestation. The Scorecard UI does not rank this map, but session details expose provider/provenance. The source's former parser field is retained separately as `transcriptProvider`. **What this does not model:** a workflow that hands off between Claude and @@ -935,9 +1039,9 @@ punchcard[dow + "-" + hour] += 1 per assistant/agent_message response, at its **Source:** incremented once per Claude API message (all of a message's transcript lines are one hit) -(`usage-parsers.mjs:42`, keyed by this call: `punchKey(at)`) and once per Codex -`agent_message` (`usage-parsers.mjs:1021-1025`), merged into the window-level -`punchcard` object per session (`usage-aggregate.mjs:943-1007`). Cell intensity is +(`usage-parsers.mjs:45`, keyed by this call: `punchKey(at)`) and once per Codex +`agent_message` (`usage-parsers.mjs:1143-1147`), merged into the window-level +`punchcard` object per session (`usage-aggregate.mjs:1024-1117`). Cell intensity is linear against the single busiest cell in the window: `v = pcMax ? n/pcMax : 0` (`dashboard/client.mjs`) — this is a **relative**, not absolute, scale, so the heatmap's brightest cell is always @@ -979,9 +1083,9 @@ byModel[model].sessions = count of DISTINCT sessions whose s.models includes th ``` **Source:** cost/tokens/responses accumulate inside the usage-row loop -(`usage-aggregate.mjs:752-779`); the per-model session count is deliberately computed +(`usage-aggregate.mjs:762-790`); the per-model session count is deliberately computed **separately**, once per session over its `s.models` array -(`usage-aggregate.mjs:910-919`) rather than inside the cost loop, precisely +(`usage-aggregate.mjs:991-1000`) rather than inside the cost loop, precisely **so that a model can appear in `byModel` — with a nonzero session count — even in a session that contributed zero cost/tokens/responses for that model.** This is not an edge case invented for this document: it is the @@ -991,11 +1095,11 @@ excluded subagent-replay session still shows up as "used," at zero cost, rather than vanishing. `byModel[...].responses` is populated from each usage row's response field at -this call (`usage-aggregate.mjs:775-778`). The shared usage-row accumulator is -defined at `usage-parsers.mjs:499-515`; Claude passes one response per API -message at its call site (`usage-parsers.mjs:677`). Codex passes the session's +this call (`usage-aggregate.mjs:786-789`). The shared usage-row accumulator is +defined at `usage-parsers.mjs:524-550`; Claude passes one response per API +message at its call site (`usage-parsers.mjs:723`). Codex passes the session's whole response count once, where finalizeCodexUsage makes the corresponding -call (`usage-parsers.mjs:1136-1179`). +call (`usage-parsers.mjs:1303-1337`). The two parsers therefore hand the aggregate the same response-bearing row shape, despite their different per-turn and cumulative transcript formats. @@ -1014,12 +1118,12 @@ split `server_error` 27, `authentication_failed` 3, `rate_limit` 3 — three distinct underlying causes, one placeholder shape). The parser recognizes the decoded API-error placeholder and returns before model -or usage attribution (`usage-parsers.mjs:701-720`). The turn does **not** increment -the response count or punchcard — it is not a model response (`usage-parsers.mjs:661-696`) — it *is* real +or usage attribution (`usage-parsers.mjs:758-777`). The turn does **not** increment +the response count or punchcard — it is not a model response (`usage-parsers.mjs:708-753`) — it *is* real engaged time (its timestamp still extends the session span), someone was genuinely waiting on it — and increments the record's exception count instead. Aggregation rolls that count into the window -total (`usage-aggregate.mjs:968-979`) and keeps it on the session row beside the -delegation-source fields (`usage-aggregate.mjs:838-867`), so it remains +total (`usage-aggregate.mjs:1050-1074`) and keeps it on the session row beside the +delegation-source fields (`usage-aggregate.mjs:909-944`), so it remains inspectable in Sessions without creating a fake model row. When `totals.exceptions > 0`, the panel header shows a small `"· N dropped/errored turns excluded"` note (`dashboard/client.mjs`); @@ -1378,7 +1482,7 @@ both credential-free for ak: `windowDurationMins: 10080` (the weekly). Windows are therefore keyed and labelled by duration (`windowLabel`, `quota.mjs:52`), never by slot name. The same rule applies to the historical snapshots parsed out of rollouts: the -normalizer at `usage-parsers.mjs:866-884` keeps a flat `windows` list keyed by +normalizer at `usage-parsers.mjs:971-989` keeps a flat `windows` list keyed by `window_minutes`. **Freshness is part of the number.** Both sides carry `fetchedAt`; the view @@ -1424,7 +1528,7 @@ Codex ≥0.140 maintains its own SQLite thread ledger (`~/.codex/state_N.sqlite` globs and takes the newest). `readCodexState` (`:62`, whose own delegate call reads the db file) reads per-thread `thread_source` (`user` vs `subagent`) plus `thread_spawn_edges`, and -`applyCodexLedger` (`usage-aggregate.mjs:1280-1310`) overlays that onto parsed +`applyCodexLedger` (`usage-aggregate.mjs:1392-1465`) overlays that onto parsed sessions: a thread that ONLY the ledger identifies as a subagent has its token usage stripped — with no `thread_source` in its own rollout its parsed usage is the unsubtracted cumulative total, which replays the parent's entire token history @@ -1432,7 +1536,7 @@ unsubtracted cumulative total, which replays the parent's entire token history visible. A rollout that says `thread_source: subagent` itself is left untouched: the parser already reduced it to the subagent's own usage (§16.2). **The parser is primary, the ledger is the fallback**: `rec.threadSource ?? -t?.threadSource ?? fromEdges` (`usage-aggregate.mjs:1280`) reads the rollout's +t?.threadSource ?? fromEdges` (`usage-aggregate.mjs:1392`) reads the rollout's own `session_meta.thread_source` first, and only consults the ledger when that line is missing entirely. This is sound because `thread_source` is now the FIRST session_meta line's value (§1.2) rather than whichever meta @@ -1443,7 +1547,7 @@ migration generation its `N` reflects; a rollout the ledger cannot resolve (an older Codex build, a migrated-beyond-recognition state file) still gets a correct `threadSource` straight from its own transcript rather than falling through unclassified. Codex sessions also carry -`reasoningOutput` (`usage-parsers.mjs:1136-1183`) — reasoning tokens are a **subset** +`reasoningOutput` (`usage-parsers.mjs:1303-1355`) — reasoning tokens are a **subset** of output tokens and are annotation only, never added to any sum. ## 14. Known limitations, restated as a single checklist @@ -1468,7 +1572,7 @@ the same list: - [x] A percentile taken from the overflow bucket of a histogram is printed with `≥`, and an unmeasured one is `null` rather than `0` (§15). - [x] A latency figure is never called TTFT: it is a prompt-to-answer gap or - a host-measured turn duration, and neither transcript records TTFT (§15). + a host-measured turn duration. Codex host-reported first-token evidence is a separate metric (§15). - [x] Permission posture keeps `not-recorded` as a first-class bucket — unmapped evidence is never folded into a real posture — and the inference provider is separately reported when native evidence exists (§8, §16). @@ -1509,23 +1613,23 @@ p(q), over N samples, landing in bucket i (count n_i, running total `cum` before **Source:** -- Edges: `LAT_BUCKET_EDGES` and `LEN_BUCKET_EDGES` (`usage-aggregate.mjs:212`, `:215`). - The parsers carry their own copies (`usage-parsers.mjs:365-370`) and the +- Edges: `LAT_BUCKET_EDGES` and `LEN_BUCKET_EDGES` (`usage-aggregate.mjs:215`, `:215`). + The parsers carry their own copies (`usage-parsers.mjs:390-395`) and the browser bundle a third pair (`LAT_EDGES`/`LEN_EDGES`), because the payload ships bucket *counts* and never the edges they were binned on. -- Slotting: `bucketIndex` (`usage-parsers.mjs:379-381`) — one definition of a +- Slotting: `bucketIndex` (`usage-parsers.mjs:404-406`) — one definition of a boundary, shared by every histogram built on these edges. -- Sampling: `noteLatencySample` (`usage-parsers.mjs:372-377`) allocates +- Sampling: `noteLatencySample` (`usage-parsers.mjs:397-402`) allocates `latHist` lazily, so a session that never observed a latency keeps `latHist: null` — absent, not a fabricated row of zeroes. - Session length: `seal` derives each session's `lenSeconds` from its own - active intervals (`usage-parsers.mjs:489-496`) — the §6 engaged figure for + active intervals (`usage-parsers.mjs:514-521`) — the §6 engaged figure for one session, never its first-to-last span. This is a per-session parse result, kept distinct from the next window-level fold. -- Window merge: `buildRhythm` (`usage-aggregate.mjs:1084-1109`) adds the +- Window merge: `buildRhythm` (`usage-aggregate.mjs:1194-1219`) adds the per-session `latHist` slot-wise and buckets each session's `lenSeconds`. -- Percentiles: `percentileFromBuckets` (`usage-aggregate.mjs:241-258`). The +- Percentiles: `percentileFromBuckets` (`usage-aggregate.mjs:244-261`). The browser re-implementation `bucketPercentile` (`usage-rhythm.mjs:106-126`) is pinned to byte-identical output, and the browser's edge copies to the server constants, by `tests/kit/dashboard-usage-telemetry.test.mjs:924-942` and @@ -1558,14 +1662,14 @@ median target 50 lands in bucket 1 (running total 40, n = 25), giving **Overflow floors, and why `≥` is not decoration.** The last bucket of either histogram has no upper edge to interpolate towards, so a percentile landing in it reports that bucket's **floor** and nothing more — -`if (i >= edges.length) return round(lo, 2)` (`usage-aggregate.mjs:252`). +`if (i >= edges.length) return round(lo, 2)` (`usage-aggregate.mjs:255`). A p95 printed as `≥60s` therefore means *at least 60 seconds* — the counts cannot say whether the real figure is 61 seconds or 61 minutes, and printing a bare `60s` would state a precision they do not carry. Both renderers apply the prefix by the same rule (`v >= lastEdge`), so a value that reaches the last edge by interpolation and one that came from the overflow slot print identically — the two are the same claim. An empty histogram is `null`, never -`0` (`usage-aggregate.mjs:244`): "nothing was measured" and "measured zero" are +`0` (`usage-aggregate.mjs:247`): "nothing was measured" and "measured zero" are different statements and only the first is true, so the cards print `not measured` and the CLI prints `no samples`. @@ -1575,13 +1679,13 @@ same way, and the panel says so rather than implying a single clock: | Host | How a latency sample is produced | |---|---| -| codex | **Host-measured.** `task_started` remembers the turn's start (`usage-parsers.mjs:915-934`) and `task_complete` samples Codex's own `duration_ms` (`usage-parsers.mjs:936-954`) — but only if no prompt-gap already covered that turn (so a turn is never sampled twice) and only within the same 3600 s cap the derived paths apply. | -| codex | Also derives a prompt-gap when one is available: `handleCodexUserMessage` opens the window (`usage-parsers.mjs:976-1006`) and the next agent message closes it, clearing `turnStartedAt` so the `duration_ms` fallback cannot double-fire (`usage-parsers.mjs:1008-1019`). | -| claude | **Derived from event gaps.** A human prompt sets `latState.pendingMs` (`pendingMs`, `usage-parsers.mjs:593-608`); the first real assistant turn closes that gap into a `noteLatencySample` call (`usage-parsers.mjs:722-725`). | +| codex | **Host-measured.** `task_started` remembers the turn's start (`usage-parsers.mjs:1028-1047`) and `task_complete` samples Codex's own `duration_ms` (`usage-parsers.mjs:1049-1076`) — but only if no prompt-gap already covered that turn (so a turn is never sampled twice) and only within the same 3600 s cap the derived paths apply. | +| codex | Also derives a prompt-gap when one is available: `handleCodexUserMessage` opens the window (`usage-parsers.mjs:1098-1128`) and the next agent message closes it, clearing `turnStartedAt` so the `duration_ms` fallback cannot double-fire (`usage-parsers.mjs:1130-1141`). | +| claude | **Derived from event gaps.** A human prompt sets `latState.pendingMs` (`pendingMs`, `usage-parsers.mjs:628-643`); the first real assistant turn closes that gap into a `noteLatencySample` call (`usage-parsers.mjs:779-782`). | | opencode | Derived from its message stream, measured to **completion**: `rec.pendingPromptMs` is the user message's `time.created`, and the first assistant row closes it at that row's `time.completed` (`closeLatencyWindow`, `usage-opencode.mjs:318-324`) — OpenCode inserts the assistant row ~15 ms after the prompt and fills it in as it generates, so its own `time.created` is not a response time. A row with no completed stamp yields no sample. | **Every** path is capped: a sample above `MAX_LATENCY_SAMPLE_SECONDS` -(3600 s, `usage-parsers.mjs:445-460`) is an idle resume — the person walked away +(3600 s, `usage-parsers.mjs:470-485`) is an idle resume — the person walked away and came back — not a wait for a reply, so it is dropped from sampling entirely rather than parked in the overflow bucket beside genuinely slow turns. That includes Codex's host-measured `duration_ms`. An earlier ruling exempted @@ -1594,11 +1698,11 @@ reference corpus before the fix: 12 of 835 durations exceeded the cap, the largest 94,079,450 ms ≈ 26.1 hours, all of them landing in the `≥60s` overflow bucket and dragging `latP95` into it. An interrupted turn contributes nothing at all — `turn_aborted` clears both -pending states (`usage-parsers.mjs:1048-1089`), so a prompt that was never +pending states (`usage-parsers.mjs:1170-1218`), so a prompt that was never answered can never be timed against a later, unrelated reply. A dropped API turn is likewise never a sample: the error branch returns before the latency block and deliberately leaves `pendingMs` set, so the first real completion -that eventually follows is what gets timed (`usage-parsers.mjs:601-610`). +that eventually follows is what gets timed (`usage-parsers.mjs:636-645`). **This figure is never labeled TTFT, in any surface.** Time-to-first-token measures when a stream *starts*; every figure here measures when a turn @@ -1611,7 +1715,8 @@ TTFT" beside the per-host note it qualifies. A true TTFT exists for Claude Code, but only as a span in its opt-in OpenTelemetry beta ([monitoring](https://code.claude.com/docs/en/monitoring-usage)) — a different, non-transcript evidence class that this scorecard does not read. -Neither transcript store records it, so no panel here may borrow the name. +Codex can separately record host-reported first-token timing. That evidence is retained +independently and does not rename or replace the completion-latency metric. **What this does not model:** @@ -1639,8 +1744,8 @@ Neither transcript store records it, so no panel here may borrow the name. means a window dominated by one host is really reporting that host's instrument; - the bucketing *function* is implemented twice: the parsers export - `bucketIndex` (`usage-parsers.mjs:379-381`) and the aggregate keeps a private - copy of the same loop (`usage-aggregate.mjs:222-225`), because the dependency + `bucketIndex` (`usage-parsers.mjs:404-406`) and the aggregate keeps a private + copy of the same loop (`usage-aggregate.mjs:225-228`), because the dependency between the two modules is deliberately one-way. The *edges* they run on are pinned equal by test — `AGG_LAT_EDGES` against `LAT_BUCKET_EDGES` (`tests/kit/usage-index.test.mjs:15-19`) — but the two function bodies @@ -1670,13 +1775,13 @@ mode = normalizeMode(host, raw evidence) or 'not-recorded' **Source:** `normalizeMode` (`usage-modes.mjs:23-35`) is the whole taxonomy; `MODES` (`usage-modes.mjs:4`) is the closed four-value vocabulary. Per-day folding is "addCost(d.byMode, rec.mode ?? 'not-recorded', rowCost)" -(`usage-aggregate.mjs:752-772`) — in the usage-row pass, because only a row knows +(`usage-aggregate.mjs:762-782`) — in the usage-row pass, because only a row knows which day its dollars landed on. The window bucket is -this call: `addTo(bucket(byMode, s.mode ?? 'not-recorded'), s)` (`usage-aggregate.mjs:983-993`). +this call: `addTo(bucket(byMode, s.mode ?? 'not-recorded'), s)` (`usage-aggregate.mjs:1093-1103`). The evidence each parser reads: Claude's `permissionMode`, off the human prompt -only (`usage-parsers.mjs:593-608`); Codex's `approval_policy`/`sandbox_policy` +only (`usage-parsers.mjs:628-643`); Codex's `approval_policy`/`sandbox_policy` off each `turn_context`, last one wins since a session may renegotiate mid-run -(`usage-parsers.mjs:843-864`); OpenCode's `mode` off each assistant message +(`usage-parsers.mjs:943-964`); OpenCode's `mode` off each assistant message (`usage-opencode.mjs:344-345`). Render is `modeChart` in `src/lib/dashboard/client/usage.mjs`; the CLI table is `printScoreModeTable` (`src/commands/usage.mjs:270-272`). @@ -1708,7 +1813,7 @@ or `{"type":"workspace-write", …}` with sibling fields such as `network_access` — never the bare string the taxonomy is written against. A survey of this machine's rollouts (400 files, 2026-08-28) found 1,110 object occurrences and **zero** string ones. `handleCodexTurnContext` -(`usage-parsers.mjs:843-864`) therefore reads `sandbox_policy.type` and passes +(`usage-parsers.mjs:943-964`) therefore reads `sandbox_policy.type` and passes that to `normalizeMode`, which is unchanged and still accepts the string form. Before this extraction the object reached `normalizeMode` intact, matched no rule, and stringified into `modeRaw` as `"never/[object Object]"`: the `plan`, @@ -1731,11 +1836,11 @@ second field. (`usage-modes.mjs:25`, `:32`), so an unrecognised raw value — a future `permissionMode`, a policy this taxonomy has not been taught — yields no mode. The raw string is kept beside the normalized one as `modeRaw` -(`usage-parsers.mjs:208`) precisely because the mapping is a judgement call and +(`usage-parsers.mjs:240`) precisely because the mapping is a judgement call and a reader checking it needs the evidence it was made from. `not-recorded` is a first-class bucket key rather than a display fallback, folded at this call: `addTo(bucket(byMode, s.mode ?? 'not-recorded'), s)` -(`usage-aggregate.mjs:983-993`), it is always offered as a row by the CLI table +(`usage-aggregate.mjs:1093-1103`), it is always offered as a row by the CLI table even at zero (`printBucketTable`, `src/commands/usage.mjs:257-264`), and `segColor` (`src/lib/dashboard/client/usage-rhythm.mjs:191-192`) forces it to the de-emphasis ink rather than letting a palette give @@ -1747,17 +1852,21 @@ evidence must never read as a posture. **Formula:** ```text -source = (session.sidechain || session.threadSource == 'subagent') - ? 'subagent' : 'main' +delegated = isSubagentSession(session) + OR session.threadSource IN ['guardian_review', 'agent_created_thread'] +source = delegated ? 'subagent' : 'main' bySource[k].cost = Σ over sessions with that source of session.cost centre of the donut = round(main / (main + subagent) × 100) % ``` -**Source:** `sourceKey` (`usage-aggregate.mjs:932-936`). Both rows are created, +**Source:** `sourceKey` (`usage-aggregate.mjs:1013-1017`) uses the shared +`isSubagentSession` predicate (`usage-context.mjs:20-23`) and explicitly includes guardian reviews +and agent-created threads. The same source classification gates `humanPrompts`: only main-session +prompts enter the autonomy denominator (`usage-aggregate.mjs:1051-1054`). Both rows are created, at this call to it, before the fold -("Both source rows always exist", `usage-aggregate.mjs:962-966`) so "no subagent sessions" renders +("Both source rows always exist", `usage-aggregate.mjs:1044-1048`) so "no subagent sessions" renders as a zero rather than a row the UI silently drops. Claude's evidence is the -`isSidechain` flag on any entry in the file (`usage-parsers.mjs:758-763`, decoded at +`isSidechain` flag on any entry in the file (`usage-parsers.mjs:842-847`, decoded at `telemetry-records.mjs:267`); Codex's is the ledger-backed `thread_source` (§13c). Render is `sourceDonut` in `src/lib/dashboard/client/usage.mjs`. @@ -1768,7 +1877,7 @@ replay cannot be separated reports none. A zero cannot establish whether actual **Claude — real, priced, included.** A session's delegated work is written to its own transcript under `//subagents/`, and those files are -discovered by `listClaudeSubagents` (`usage-index.mjs:349-354`, §1) and parsed +discovered by `listClaudeSubagents` (`usage-index.mjs:435-440`, §1) and parsed like any other. `parseClaude` already prices those bytes and marks the record `sidechain` from its own `isSidechain` entries, so the cost is real, is included in `totals.cost`, and the session opens in the Sessions tab like a main-thread @@ -1856,10 +1965,10 @@ costPerSessionP90 = nearest-rank P90 of the same set cacheSavedUsd = Σ rows (costOf(1M as input) - costOf(1M as cacheRead)) × cacheRead / 1e6 ``` -**Source:** the derived block is `finishTotals` (`usage-aggregate.mjs:1042-1082`), +**Source:** the derived block is `finishTotals` (`usage-aggregate.mjs:1152-1192`), which the previous-window projection calls too so a baseline is never derived a second, drifting way. `median` and `percentile` are exact over the values -(`usage-aggregate.mjs:1029-1041`), unlike §15's bucketed percentiles. +(`usage-aggregate.mjs:1125-1151`), unlike §15's bucketed percentiles. Active days come from `byDay`'s key count and the streak from `activeStreak` in `src/lib/dashboard/client/usage.mjs`; the tiles are `cadenceCells` there, and `printScoreCadence` (`src/commands/usage.mjs:219-242`) in the CLI. @@ -1868,13 +1977,13 @@ Active days come from `byDay`'s key count and the streak from `activeStreak` in `totals.humanPrompts` is not the fingerprint-based `typedPrompts` count. It can include control records and unrecognized machine-authored prompts. It is accumulated under an explicit main-thread guard -(`usage-aggregate.mjs:953`): a subagent's prompts are written by the harness, +(`usage-aggregate.mjs:1035`): a subagent's prompts are written by the harness, so counting them would report a person as having typed work nobody asked for by hand — and would grow the denominator exactly in the windows where delegation was heaviest, making autonomy fall as automation rose. `totals.prompts` still -records every prompt beside it (`usage-aggregate.mjs:928`); the two are +records every prompt beside it (`usage-aggregate.mjs:1009`); the two are different questions and both are on the wire. Touch rate is those same human -prompts per engaged hour (`usage-aggregate.mjs:1030`), so both per-prompt figures +prompts per engaged hour (`usage-aggregate.mjs:1140`), so both per-prompt figures share one denominator. A rate whose denominator is zero is `null`, never `0` — no engaged time means the rate was never measured, which is not what "zero per hour" claims. @@ -1900,8 +2009,8 @@ presence, not verified billing. A session with no usage rows contributes to neit that map nor its per-day session count. **Cost per session is a median over priced sessions only.** A session carries -`_priced` when it had any usage rows at all (`usage-aggregate.mjs:873-879`), and -only those costs enter the distribution (`usage-aggregate.mjs:962-980`). A session +`_priced` when it had any usage rows at all (`usage-aggregate.mjs:950-956`), and +only those costs enter the distribution (`usage-aggregate.mjs:1044-1075`). A session with no usage rows costs `$0` *structurally* — nothing was ever measured for it, the common case being a Codex subagent that only its ledger row identifies, whose tokens are stripped as a double-count (§16.2, §13c) — and letting those in would report "the typical session @@ -1915,9 +2024,9 @@ positive figure that rounds away at two decimals prints `<$0.01`, never "nothing" are different claims. **What the cache saved, asked as a difference.** `cacheSavingPerMillion` -(`usage-aggregate.mjs:732-746`) prices one million tokens twice through the +(`usage-aggregate.mjs:739-756`) prices one million tokens twice through the *injected* pricer — once as fresh input, once as cache reads — and takes the -gap; `cacheSavedFor` (`usage-aggregate.mjs:732-753`) scales that to the tokens +gap; `cacheSavedFor` (`usage-aggregate.mjs:739-763`) scales that to the tokens a row actually read from cache. Nothing in that path knows what the cache multiplier is, so the saving cannot drift out of step with §3's table the way a hard-coded "0.9 × input" would the day the multiplier changed. Both probes @@ -1936,48 +2045,48 @@ this row = $4.50 × 2,000,000 / 1e6 = $9.00 ``` The window total is the sum of those per-row figures -(`usage-aggregate.mjs:943-979`), carried on each session row as `cacheSavedUsd` -(`usage-aggregate.mjs:836-857`) so it is auditable a row at a time rather than only +(`usage-aggregate.mjs:1024-1074`), carried on each session row as `cacheSavedUsd` +(`usage-aggregate.mjs:907-933`) so it is auditable a row at a time rather than only in aggregate, and rendered in the cache tile's subtitle as `saved ≈ $X vs uncached`. **Deltas: what "the previous window" is, exactly.** For a displayed window of `d` days ending at `now`, the baseline is the equal-length window immediately before it — the half-open interval `[now − 2d, now − d)` -(`previousWindow`, `usage-aggregate.mjs:1157-1181`). Both bounds are derived from +(`previousWindow`, `usage-aggregate.mjs:1267-1292`). Both bounds are derived from `now` and `d`, the window the UI is *showing*, and never from the parse cutoff: the caller widens that cutoff on purpose so older records survive to be aggregated here, and deriving the baseline from a widened bound would silently stretch it to whatever lookback the caller happened to pass. The dashboard route widens it to the depth the personal tap-share baseline needs rather than to the previous window alone: `days + BASELINE_TRAILING_DAYS` (`lookbackDays`, -`src/lib/dashboard-server.mjs:1860`); `ak usage score` applies the same rule +`src/lib/dashboard-server.mjs:1872`); `ak usage score` applies the same rule (`src/commands/usage.mjs:316`). One extra window would be a strict subset — too shallow for `promptBaselines`, which needs BASELINE_MIN_ACTIVE_DAYS of history BEFORE the displayed window and returns null without it — while this depth is a strict superset of the previous window at every supported width. A delta against an unknown-length window is not a delta. The upper bound is exclusive so a session ending exactly at the boundary belongs to the current -window and is not counted in both (`endMs`, `usage-aggregate.mjs:888-899`). Asking for +window and is not counted in both (`endMs`, `usage-aggregate.mjs:966-980`). Asking for `previous` without widening the lookback yields an all-zero baseline — the older records were never read off disk — and every chip self-suppresses against it rather than claiming a change it cannot measure. Leaving `previous` off entirely leaves `agg.previous` as `null` — "not requested", which a zeroed totals object would misreport as "measured nothing" -(`usage-aggregate.mjs:1229`). A chip self-suppresses when the baseline is null +(`usage-aggregate.mjs:1313`). A chip self-suppresses when the baseline is null or zero, and a magnitude that rounds to zero prints flat rather than drawing an arrow the printed number does not support (`deltaChip`, `usage-rhythm.mjs:36-53`; `fmtDelta`, `src/commands/usage.mjs:184-195`). **Engaged time by day is a sibling map, not a `byDay` field.** `byDay`'s presence contract is **days with retained usage rows** — a key exists exactly when tokens -landed on that day (`dayBucket`, `usage-aggregate.mjs:698-705`) — and that is +landed on that day (`dayBucket`, `usage-aggregate.mjs:705-712`) — and that is what the active-day count and the streak above are counted from. Engaged time does not share that key set: a session that runs past midnight, or a day spent reading, produces worked time on a day that billed nothing. So -`buildEngagedByDay` (`usage-aggregate.mjs:1133-1153`) keys its own map, cutting +`buildEngagedByDay` (`usage-aggregate.mjs:1243-1263`) keys its own map, cutting each active interval at every local midnight it crosses -(`splitAtLocalMidnight`, `usage-aggregate.mjs:1119-1130`) and unioning the pieces +(`splitAtLocalMidnight`, `usage-aggregate.mjs:1229-1240`) and unioning the pieces per day, which makes the map sum exactly to `totals.engagedSeconds`. Folding it into `byDay` would have forced one of two lies: inventing zero-token `byDay` rows, or dropping real worked time. The consequence is visible on the tiles — @@ -2013,8 +2122,8 @@ byDay[day].exceptions += session.exceptions attributed to the session's FIRST ``` **Source:** `exceptions` and `aborts` accumulate together onto totals -(`totals.exceptions += s.exceptions`, `usage-aggregate.mjs:954`); the per-day series lands on `byDay` itself — -`byDay[s._day].exceptions` (`usage-aggregate.mjs:977`). Render is +(`totals.exceptions += s.exceptions`, `usage-aggregate.mjs:1055`); the per-day series lands on `byDay` itself — +`byDay[s._day].exceptions` (`usage-aggregate.mjs:1106`). Render is `relRate`/`relStat`/`relTrend` in `src/lib/dashboard/client/usage.mjs` (that bundle shares a basename with the CLI command module, so cited here by name, no line); `printScoreReliability` (`src/commands/usage.mjs:270-295`) prints @@ -2025,9 +2134,9 @@ model, and each host signals that differently: | Host | What is counted, and where | |---|---| -| claude | The API-error placeholder — Claude Code synthesizes a local turn with no completion behind it when a connection drops, a rate limit rejects, or auth fails. The decoder sets `isApiError` from either `isApiErrorMessage` or the literal `` model marker (`telemetry-records.mjs:269`), because the flag is not set on every build that emits the placeholder; the parser counts it as an exception, not a response, and returns before any model or usage attribution (`usage-parsers.mjs:701-720`). | -| codex | A `task_complete` event carrying a non-null `error` (`usage-parsers.mjs:936-954`). | -| codex | `turn_aborted` is counted **separately**, into `rec.aborts` (`usage-parsers.mjs:1048-1089`) — not into exceptions. | +| claude | The API-error placeholder — Claude Code synthesizes a local turn with no completion behind it when a connection drops, a rate limit rejects, or auth fails. The decoder sets `isApiError` from either `isApiErrorMessage` or the literal `` model marker (`telemetry-records.mjs:269`), because the flag is not set on every build that emits the placeholder; the parser counts it as an exception, not a response, and returns before any model or usage attribution (`usage-parsers.mjs:758-777`). | +| codex | A `task_complete` event carrying a non-null `error` (`usage-parsers.mjs:1049-1076`). | +| codex | `turn_aborted` is counted **separately**, into `rec.aborts` (`usage-parsers.mjs:1170-1218`) — not into exceptions. | | opencode | An assistant message carrying a non-null `error` other than `MessageAbortedError` (`usage-opencode.mjs:344-348`). | | opencode | `MessageAbortedError` — how OpenCode records a turn the user stopped — is counted **separately**, into `rec.aborts` (name at `usage-opencode.mjs:307-308`), and keeps the row's tokens and cost. | @@ -2036,7 +2145,7 @@ recorded interruption; it does not independently prove who initiated it. An exce is the turn failing. Summing them would report a deliberate interruption as a reliability problem and move a number that is supposed to mean "how often did this break". They are counted, carried -(`aborts`, `usage-aggregate.mjs:809-833`) and displayed side by side, with the +(`aborts`, `usage-aggregate.mjs:836-904`) and displayed side by side, with the distinction stated on the tile rather than left to the label. **Aborts are CODEX-AND-OPENCODE normalized evidence.** This counter consumes Codex @@ -2056,7 +2165,7 @@ treatment `latHist` (§15) and the context chip (§16) already get. **Exceptions ride the session's first-billed day.** The per-day series uses the same attribution as the session count — `byDay[s._day].exceptions += s.exceptions` -(`usage-aggregate.mjs:954-957`) — which is *not* the moment a turn dropped: a session spanning midnight lands all of its +(`usage-aggregate.mjs:1106`) — which is *not* the moment a turn dropped: a session spanning midnight lands all of its exceptions on the day its tokens first billed. That keeps the reliability trend and the session trend drawn on one convention — the alternative, attributing each exception to its own timestamp, would have made the two lines disagree @@ -2073,15 +2182,12 @@ measurement behind §10 breaks 33 such placeholder turns down as `server_error` placeholder shape, which is why the panel counts them together and §10 excludes them from the model ranking rather than showing a `$0` model row. -**What this does not model:** the rate's denominator is *responses*, which -includes the exception turns themselves (they increment `rec.responses` before -the error branch returns, `usage-parsers.mjs:568`) — they were real engaged -time, someone was genuinely waiting on them. A retry that eventually succeeded -appears as one exception plus one successful response, not as a single -recovered turn; nothing in either transcript links the two. And the worst-day -flag names the day with the most exceptions without inventing a threshold for -what counts as a spike, because any constant chosen here would be a judgement -the data never made. +**What this does not model:** the rate's denominator is accounted responses. Claude API-error +placeholders increment `rec.exceptions` and return before response, usage or punchcard accounting +(`recordClaudeAssistantTurn`, `usage-parsers.mjs:749-777`); they do not enter that denominator. +A later successful retry can contribute one response alongside the earlier exception, but the +metric does not pair them into a single recovered turn. The worst-day flag names the day with +the most exceptions without inventing a threshold for what counts as a spike. --- @@ -2101,7 +2207,7 @@ byTool[name] += session.tools[name] summed across sessions byDay[day].byModelFamily[fam] += rowCost fam = modelFamily(row.model) ``` -**Source:** the tool tally is folded into `byTool` at `usage-aggregate.mjs:943-1007`; +**Source:** the tool tally is folded into `byTool` at `usage-aggregate.mjs:1024-1117`; the per-day family split is this call: `addCost(d.byModelFamily, modelFamily(row.model), rowCost)` (`usage-aggregate.mjs:776`), inside the usage-row pass because only a row knows its day. Render is `toolRows`/`modelMix` in @@ -2109,10 +2215,10 @@ row knows its day. Render is `toolRows`/`modelMix` in **Tool names are the host's own, never renamed.** Claude's tally is keyed by the `tool_use` block's own `name` (`collectClaudeToolNames`, -`usage-parsers.mjs:627-638`). Codex's five tallied item types — +`usage-parsers.mjs:662-673`). Codex's five tallied item types — `CommandExecution`, `McpToolCall`, `FileChange`, `CollabAgentToolCall`, -`DynamicToolCall` (`CODEX_TOOL_ITEM_TYPES`, `usage-parsers.mjs:1040-1046`, tallied at this -call: `CODEX_TOOL_ITEM_TYPES.has(decoded.unknownItemType)` — `usage-parsers.mjs:1092-1103`) — +`DynamicToolCall` (`CODEX_TOOL_ITEM_TYPES`, `usage-parsers.mjs:1162-1168`, tallied at this +call: `CODEX_TOOL_ITEM_TYPES.has(decoded.unknownItemType)` — `usage-parsers.mjs:1221-1235`) — keep those exact spellings in the ranking. Mapping `CommandExecution` onto `Bash`, or `FileChange` onto `Edit`, would be a claim about equivalence that neither host makes: the vocabularies are host-specific, the semantics do not @@ -2126,12 +2232,12 @@ above it is read against — and the fold row is dimmed because `Other` is a residue, not a tool. **Model-family folding, with the rules pinned.** `modelFamily` -(`usage-aggregate.mjs:272-279`) lowercases the id, keeps only the segment after +(`usage-aggregate.mjs:275-282`) lowercases the id, keeps only the segment after the last `/` so a namespaced id still ends on the same tokens, then: | Rule | Example id | Family | |---|---|---| -| contains an Anthropic family name (`CLAUDE_FAMILIES`, `usage-aggregate.mjs:264`) | `claude-opus-5-20260401` | `opus` | +| contains an Anthropic family name (`CLAUDE_FAMILIES`, `usage-aggregate.mjs:267`) | `claude-opus-5-20260401` | `opus` | | — matched by containment, not position, since the id shape has moved | `claude-3-5-sonnet-20241022` | `sonnet` | | — and after the last slash, so a namespaced id still folds | `openrouter/anthropic/claude-haiku-4-5` | `haiku` | | otherwise matches `gpt-(\d+)` | `gpt-5.6-sol` | `gpt-5` | @@ -2360,14 +2466,14 @@ commit `540be18` in the historical fix. `parseCodex`'s single `addUsage()` call never included a `responses` field — Claude's parser passes `responses: 1` per API message at this call: -`usage-parsers.mjs:677` (the current equivalent), but Codex's call +`usage-parsers.mjs:723` (the current equivalent), but Codex's call passed no such field at all. Because `byModel[model].responses` is summed -directly from each usage row's `responses` field (`usage-aggregate.mjs:775-778`, +directly from each usage row's `responses` field (`usage-aggregate.mjs:786-789`, `m.responses += row.responses`), **every** Codex model in §10's Models-in-Play list displayed `0 resp` regardless of real token/cost volume or actual `agent_message` count. **Fix:** parseCodex now passes `responses: rec.responses` (the session's own tallied response count, inside -`finalizeCodexUsage`, `usage-parsers.mjs:1136-1183`) on its `addUsage()` call. +`finalizeCodexUsage`, `usage-parsers.mjs:1303-1355`) on its `addUsage()` call. #### Bug B — subagent thread-replay could double-bill tokens @@ -2396,10 +2502,10 @@ confirmed as a real Codex rollout field by **[C7]**) and skips the `addUsage()` call entirely when its value is `'subagent'` — `finalizeCodexUsage` returns early at this call: `if (!lastUsage || rec.threadSource === 'subagent')` -(`usage-parsers.mjs:1136-1173`). The session record itself is **not** +(`usage-parsers.mjs:1303-1331`). The session record itself is **not** dropped — it remains visible in the Sessions tab with `threadSource` surfaced (mirroring the existing `sidechain` flag Claude sessions already -carry, `usage-parsers.mjs:760-763`), so a maintainer auditing the raw data can +carry, `usage-parsers.mjs:844-847`), so a maintainer auditing the raw data can still see it; it simply contributes zero tokens/cost, exactly as intended by the "models still shows up in §10's list, with zero cost" mechanism §10 describes. @@ -2461,8 +2567,8 @@ parity: it does not apply ledger fallback or allocate costs per turn/day/model. `byModel` on the first run after the change, purely because the cache predated it; every unit test still passed, since tests only exercise a fresh parse. `SCHEMA_VERSION` went to `4` specifically to force the one-time - re-parse; the constant now reads `24` (`usage-index.mjs:177`), each bump since - having forced its own re-parse the same way. + re-parse. That is the historical v4 migration; the current `SCHEMA_VERSION` is `26` + (`usage-index.mjs:198`), with the present compatibility contract stated above. Re-querying the same live server after the bump returned `totals.exceptions: 20` with `` absent from `byModel` — measured, not projected. diff --git a/package.json b/package.json index 6eb0778b..95e6c309 100644 --- a/package.json +++ b/package.json @@ -54,6 +54,7 @@ "scripts": { "test": "node scripts/run-tests.mjs unit", "test:ui": "node scripts/run-tests.mjs ui", + "test:quality": "node scripts/run-tests.mjs focus tests/quality/comment-label-guard.test.mjs", "test:surface": "node --test tests/kit/dispatch-surface.test.mjs", "test:aqe-external-provider-live": "node --test tests/live/aqe-external-provider-transport.test.mjs", "test:aqe-stop-hook-live": "AK_AQE_CONFORMANCE=1 node --test tests/live/aqe-stop-hook-conformance.test.mjs", diff --git a/scripts/real-state-tripwire.mjs b/scripts/real-state-tripwire.mjs index e74198d2..09b23794 100644 --- a/scripts/real-state-tripwire.mjs +++ b/scripts/real-state-tripwire.mjs @@ -34,7 +34,8 @@ export const CONCURRENT_WRITERS = [ { kind: 'state', pattern: /^statusline-debug\.log$/, writer: 'statusline debug log (src/templates/statusline-footer.cjs:2-15)' }, { kind: 'repo', pattern: /^\.swarm(\/|$)/, writer: 'Ruflo hooks and daemon of a live session' }, { kind: 'repo', pattern: /^\.agentic-qe\/(?!llm-config\.json)/, writer: 'AQE hooks of a live session' }, - { kind: 'repo', pattern: /^\.claude-flow\/(?!config\.json$)/, writer: 'Ruflo hooks and statusline caches of a live session' }, + { kind: 'repo', pattern: /^\.claude-flow(?:\/(?!config\.json$)|$)/, writer: 'Ruflo hooks and statusline caches of a live session' }, + { kind: 'repo', pattern: /^\.claude\/(?:proven-config\.json|\.proven-config-version)$/, writer: 'Ruflo proven-config adoption on CLI startup (@claude-flow/cli 3.48.0 dist/src/config/proven-config-refresh.js:24,30,94,120-125; dist/src/index.js:173-175)' }, { kind: 'user-file', pattern: /^\.claude\.json$/, writer: 'Claude Code session state in ~/.claude.json (https://code.claude.com/docs/en/settings)' }, ]; diff --git a/scripts/run-roots.mjs b/scripts/run-roots.mjs new file mode 100644 index 00000000..ff4621c2 --- /dev/null +++ b/scripts/run-roots.mjs @@ -0,0 +1,229 @@ +// Builtin-only attribution and conservative run-root handling. Native probes +// deliberately cannot authorize sibling deletion on any supported platform. +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { randomUUID } from 'node:crypto'; + +export const RUN_ROOT_NAME = /^ak-suite-[A-Za-z0-9]{6}$/; +export const OWNER_FILE = '.ak-suite-owner.json'; +export const HOLD_DIR = '.ak-suite-holds'; +export const IGNORED_IN_ROOT = new Set(['node-compile-cache', OWNER_FILE, HOLD_DIR]); +const MAX_OWNER_BYTES = 8192; +const currentUid = () => process.getuid?.() ?? null; + +/** Attribution timestamp only: startedAt is NOT an observed OS process start. + * @param {{pid?:number, now?:number, hostname?:string, uid?:number|null, platform?:string}} [options] + */ +export function ownerRecord({ pid = process.pid, now = Date.now(), hostname = os.hostname(), + uid = currentUid(), platform = process.platform } = {}) { + return { schema: 1, runId: randomUUID(), pid, startedAt: now, hostname, uid, platform, proofMode: 'list-only' }; +} + +/** Private, exclusive staging file followed by atomic publication. */ +export function writeOwner(root, record) { + const canonical = fs.realpathSync(root); + if (canonical !== root || !fs.lstatSync(root).isDirectory()) throw Error('noncanonical run root'); + const bound = { ...record, root, tmpdir: path.dirname(root) }; + if (!validRecord(bound, root)) throw Error('invalid owner record'); + const staging = path.join(root, `${OWNER_FILE}.tmp`); + fs.writeFileSync(staging, JSON.stringify(bound), { flag: 'wx', mode: 0o600 }); + fs.renameSync(staging, path.join(root, OWNER_FILE)); +} + +function validRecord(r, root) { + return r !== null && typeof r === 'object' && !Array.isArray(r) + && r.schema === 1 && typeof r.runId === 'string' && /^[a-f0-9-]{36}$/.test(r.runId) + && Number.isSafeInteger(r.pid) && r.pid > 0 + && Number.isSafeInteger(r.startedAt) && r.startedAt > 0 + && r.hostname === os.hostname() && r.uid === currentUid() && r.platform === process.platform + && r.proofMode === 'list-only' && r.root === root && r.tmpdir === path.dirname(root); +} + +/** Bounded, no-follow metadata read; missing, changed or invalid means unknown. */ +export function readOwner(root) { + let fd; + let record = null; + try { + const file = path.join(root, OWNER_FILE); + const before = fs.lstatSync(file, { bigint: true }); + if (!before.isFile() || before.isSymbolicLink() || before.nlink !== 1n + || before.size > BigInt(MAX_OWNER_BYTES) || (currentUid() !== null && before.uid !== BigInt(currentUid()))) { + throw Error('unsafe owner file'); + } + // Bitwise flags treat an unavailable platform constant as zero. + fd = fs.openSync(file, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | fs.constants.O_NONBLOCK); + const opened = fs.fstatSync(fd, { bigint: true }); + if (!opened.isFile() || !sameIdentity(before, opened) || opened.size > BigInt(MAX_OWNER_BYTES)) { + throw Error('owner file changed at open'); + } + const bytes = Buffer.alloc(MAX_OWNER_BYTES + 1); + const count = fs.readSync(fd, bytes, 0, bytes.length, 0); + if (count > MAX_OWNER_BYTES || !sameIdentity(opened, fs.lstatSync(file, { bigint: true }))) throw Error('owner file changed at read'); + const parsed = JSON.parse(bytes.subarray(0, count).toString('utf8')); + if (validRecord(parsed, root)) record = parsed; + } catch { /* Unknown metadata never grants ownership. */ } + finally { if (fd !== undefined) fs.closeSync(fd); } + return record; +} + +/** Create the private hold directory before launching any suite command. */ +export function prepareRunRootHolds(root, runId) { + if (readOwner(root)?.runId !== runId || fs.realpathSync(root) !== root) throw Error('run root owner mismatch'); + fs.mkdirSync(path.join(root, HOLD_DIR), { mode: 0o700 }); +} + +/** A command acquires this hold before launching children that use its run root. + * Outside a guarded run it returns null; local sandbox retention still applies. + * @param {{env?:NodeJS.ProcessEnv}} [options] + */ +export function acquireRunRootHold({ env = process.env } = {}) { + const root = env.AK_SUITE_ROOT; + const runId = env.AK_SUITE_RUN_ID; + if (root === undefined && runId === undefined) return null; + if (!root || !runId || readOwner(root)?.runId !== runId || fs.realpathSync(root) !== root) { + throw Error('cannot establish run-root hold: owner mismatch'); + } + const dir = path.join(root, HOLD_DIR); + const stat = fs.lstatSync(dir); + if (!stat.isDirectory() || stat.isSymbolicLink() || fs.realpathSync(dir) !== dir) { + throw Error('cannot establish run-root hold: unsafe hold directory'); + } + const id = randomUUID(); + const token = randomUUID(); + const file = path.join(dir, id); + fs.writeFileSync(file, token, { flag: 'wx', mode: 0o600 }); + return { root, runId, file, token, pid: process.pid }; +} + +/** Remove only the marker returned to this process by acquireRunRootHold. */ +export function releaseRunRootHold(hold) { + if (hold === null) return; + if (!hold || hold.pid !== process.pid || readOwner(hold.root)?.runId !== hold.runId + || fs.realpathSync(hold.root) !== hold.root + || path.dirname(hold.file) !== path.join(hold.root, HOLD_DIR)) throw Error('run-root hold owner mismatch'); + const dir = path.join(hold.root, HOLD_DIR); + const parent = fs.lstatSync(dir); + if (!parent.isDirectory() || parent.isSymbolicLink() || fs.realpathSync(dir) !== dir) { + throw Error('run-root hold directory changed'); + } + const stat = fs.lstatSync(hold.file); + if (!stat.isFile() || stat.isSymbolicLink() || stat.nlink !== 1 + || fs.readFileSync(hold.file, 'utf8') !== hold.token) throw Error('run-root hold changed'); + fs.unlinkSync(hold.file); +} + +/** Missing or unreadable hold state is uncertainty, never permission to remove. */ +export function inspectRunRootHolds(root, runId) { + try { + if (readOwner(root)?.runId !== runId) throw Error('owner changed'); + const dir = path.join(root, HOLD_DIR); + const stat = fs.lstatSync(dir); + if (!stat.isDirectory() || stat.isSymbolicLink() || fs.realpathSync(dir) !== dir) throw Error('unsafe hold directory'); + return { unresolved: fs.readdirSync(dir).length > 0, reason: 'unresolved child hold' }; + } catch { return { unresolved: true, reason: 'hold inspection uncertain' }; } +} + +/** Pure path check also accepts Windows paths in cross-platform unit fixtures. */ +export function unsafeTempBase(tmpdir, homedir) { + const windows = path.win32.isAbsolute(tmpdir) && !path.posix.isAbsolute(tmpdir); + const flavor = windows ? path.win32 : path.posix; + const normalize = (p) => windows ? flavor.resolve(p).toLowerCase() : flavor.resolve(p); + const tmp = normalize(tmpdir); + if (tmp === normalize(flavor.parse(tmp).root)) return 'filesystem root'; + if (tmp === normalize(homedir)) return 'home directory'; + return null; +} + +/** @param {string} dir + * @param {{tmpdir:string, homedir:string, uid?:number|null, requireOwner?:boolean}} options + * @returns {{ok:boolean, reason?:string}} + */ +export function removableRunRoot(dir, { tmpdir, homedir, uid = currentUid(), requireOwner = true }) { + try { + if (!path.isAbsolute(dir) || path.resolve(dir) !== dir || !path.isAbsolute(tmpdir) + || fs.realpathSync(tmpdir) !== tmpdir) return { ok: false, reason: 'noncanonical absolute path required' }; + const unsafe = unsafeTempBase(tmpdir, fs.realpathSync(homedir)); + if (unsafe) return { ok: false, reason: `unsafe temp base: ${unsafe}` }; + if (!RUN_ROOT_NAME.test(path.basename(dir)) || path.dirname(dir) !== tmpdir) { + return { ok: false, reason: 'not an exact direct run-root child' }; + } + const stat = fs.lstatSync(dir); + if (!stat.isDirectory() || stat.isSymbolicLink() || fs.realpathSync(dir) !== dir) { + return { ok: false, reason: 'not a canonical nonsymlink directory' }; + } + if (uid !== currentUid() || (uid !== null && stat.uid !== uid)) return { ok: false, reason: 'foreign filesystem owner' }; + if (requireOwner && !readOwner(dir)) return { ok: false, reason: 'no valid owner record (missing, malformed or foreign)' }; + return { ok: true }; + } catch { return { ok: false, reason: 'path inspection failed' }; } +} + +/** Probes are injected code, never derived from metadata or CLI input. + * completeExit must prove ALL users/descendants gone, not a snapshot/handle scan. + * @typedef {{alive:(pid:number)=>boolean|null, startedAfter:(pid:number,ms:number)=>boolean|null, + * completeExit:(root:string,owner:object)=>boolean|null, listOnly?:boolean}} Probes + * @param {string} root + * @param {ReturnType} owner + * @param {Probes} probes + */ +export function proveAbandoned(root, owner, probes) { + try { + if (!validRecord(owner, root)) return { abandoned: false, reason: 'invalid owner metadata' }; + if (probes.listOnly) return { abandoned: false, reason: 'cannot prove complete descendant exit (list-only)' }; + const alive = probes.alive(owner.pid); + if (alive !== false && (alive !== true || probes.startedAfter(owner.pid, owner.startedAt) !== true)) { + return { abandoned: false, reason: 'owner alive or identity uncertain' }; + } + if (probes.completeExit(root, owner) !== true) return { abandoned: false, reason: 'descendant exit uncertain or live user' }; + return { abandoned: true, reason: 'injected complete exit proof' }; + } catch { return { abandoned: false, reason: 'probe failed; exit uncertain' }; } +} + +/** @returns {Probes} No process scans: none could authorize removal. */ +export function defaultProbes(_platform = process.platform) { + return { listOnly: true, alive: () => null, startedAfter: () => null, completeExit: () => null }; +} + +// Keep inode/device IDs and change times exact; Number stats can alias distinct files. +function sameIdentity(a, b) { return a.dev === b.dev && a.ino === b.ino && a.ctimeNs === b.ctimeNs; } + +/** Revalidate after injected probes; recursive rm can still fail partway through. + * The fixture seam is not an installed platform containment implementation. + * @param {{tmpdir:string, selfRoot?:string, homedir:string, uid?:number|null, probes?:Probes, + * log?:(s:string)=>void, remove?:(root:string)=>void}} options + */ +export function collectAbandonedRoots({ tmpdir, selfRoot, homedir, uid = currentUid(), + probes = defaultProbes(), log = console.error, remove = (root) => fs.rmSync(root, { recursive: true }) }) { + /** @type {{removed:string[], kept:Array<{path:string,reason:string}>}} */ + const result = { removed: [], kept: [] }; + const report = (message) => { try { log(message); } catch { /* Reporting cannot alter cleanup outcomes. */ } }; + const keep = (root, reason) => { result.kept.push({ path: root, reason }); report(`kept run root ${root}: ${reason}`); }; + let names; + try { names = fs.readdirSync(tmpdir); } + catch { keep(tmpdir, 'could not list run roots'); return result; } + for (const name of names) { + if (!RUN_ROOT_NAME.test(name)) continue; + const root = path.join(tmpdir, name); + if (root === selfRoot) continue; + const options = { tmpdir, homedir, uid, requireOwner: true }; + const safe = removableRunRoot(root, options); + if (!safe.ok) { keep(root, safe.reason); continue; } + try { + const identity = fs.lstatSync(root, { bigint: true }); + const owner = readOwner(root); + if (!owner) { keep(root, 'owner changed during inspection'); continue; } + const proof = proveAbandoned(root, owner, probes); + if (!proof.abandoned) { keep(root, proof.reason); continue; } + const boundary = removableRunRoot(root, options); + if (!boundary.ok || !sameIdentity(identity, fs.lstatSync(root, { bigint: true })) + || JSON.stringify(owner) !== JSON.stringify(readOwner(root))) { + keep(root, 'root or owner changed before removal'); continue; + } + try { remove(root); } + catch (error) { keep(root, `removal failed; root may be partially removed: ${error.message}`); continue; } + result.removed.push(root); + report(`removed abandoned run root ${root}`); + } catch { keep(root, 'inspection changed or failed; removal not attempted'); } + } + return result; +} diff --git a/scripts/run-tests.mjs b/scripts/run-tests.mjs index d579b881..2335a1d5 100644 --- a/scripts/run-tests.mjs +++ b/scripts/run-tests.mjs @@ -9,6 +9,8 @@ import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; +import { ownerRecord, writeOwner, prepareRunRootHolds, inspectRunRootHolds, + unsafeTempBase, removableRunRoot, collectAbandonedRoots, IGNORED_IN_ROOT } from './run-roots.mjs'; import { realStateRoots, snapshotRoots, compareSnapshots, isStrict, formatReport } from './real-state-tripwire.mjs'; const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); @@ -26,7 +28,8 @@ export const SUITES = { ['--test', 'tests/ui/dashboard-project-context.mjs', 'tests/ui/maintenance-projects.mjs', 'tests/ui/maintenance-host-alignment.mjs', 'tests/ui/intelligence-picker.mjs', 'tests/ui/usage-project-groups.mjs', 'tests/ui/context-coverage.mjs', 'tests/ui/host-readiness.mjs', - 'tests/ui/maintenance-focus.mjs', 'tests/ui/maintenance-guidance.mjs'], + 'tests/ui/maintenance-focus.mjs', 'tests/ui/maintenance-guidance.mjs', + 'tests/ui/session-surfaces.mjs'], ], }; @@ -47,23 +50,50 @@ export function commandsFor(mode, env = process.env) { * @param {{ env?: NodeJS.ProcessEnv, repoRoot?: string, platform?: string, homedir?: string, log?: (s: string) => void }} [o] * @returns {number} exit code: 2 when the suite temp root sits inside a git repository, * else the first failing command's, else 3 on a real-state change, else 4 on leftover - * temp folders, else 0 + * temp folders or failed own-root inspection/cleanup, else 0 */ export function runGuarded(commands, { env = process.env, repoRoot = REPO, platform = process.platform, homedir = os.homedir(), log = console.error, } = {}) { // Every command runs with this run's own templated temp root: leftovers are then // attributable to the run, and they fail it. - const tempRoot = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'ak-suite-'))); + const tmpdir = fs.realpathSync(os.tmpdir()); + const unsafe = unsafeTempBase(tmpdir, fs.realpathSync(homedir)); + if (unsafe) { log(`unsafe temp base ${tmpdir}: ${unsafe}`); return 2; } + const tempRoot = fs.mkdtempSync(path.join(tmpdir, 'ak-suite-')); + const owner = ownerRecord(); + try { writeOwner(tempRoot, owner); } + catch (error) { log(`could not record run owner; kept run root ${tempRoot}: ${error.message}`); return 2; } + const identity = fs.lstatSync(tempRoot, { bigint: true }); + const removeOwnRoot = () => { + const safe = removableRunRoot(tempRoot, { tmpdir, homedir, requireOwner: false }); + if (!safe.ok) { log(`kept own run root ${tempRoot}: ${safe.reason}`); return false; } + try { + const current = fs.lstatSync(tempRoot, { bigint: true }); + if (current.dev !== identity.dev || current.ino !== identity.ino || current.birthtimeNs !== identity.birthtimeNs) { + log(`kept own run root ${tempRoot}: directory identity changed`); return false; + } + fs.rmSync(tempRoot, { recursive: true, force: true, maxRetries: 3 }); + try { log(`removed own run root ${tempRoot}`); } + catch { /* Reporting cannot change a completed removal into a failure. */ } + return true; + } catch (error) { + log(`own run root removal failed; may be partially removed ${tempRoot}: ${error.message}`); + return false; + } + }; const enclosing = enclosingRepository(tempRoot); if (enclosing) { - fs.rmSync(tempRoot, { recursive: true, force: true }); + removeOwnRoot(); log(`the suite temp root ${tempRoot} is inside the git repository ${enclosing}; tests that probe "outside a ` + 'git repository" would write into it. Point TMPDIR outside any repository.'); return 2; } + try { prepareRunRootHolds(tempRoot, owner.runId); } + catch (error) { log(`could not prepare run-root holds; kept ${tempRoot}: ${error.message}`); return 2; } /** @type {NodeJS.ProcessEnv} */ - const childEnv = { ...env, TMPDIR: tempRoot, TEMP: tempRoot, TMP: tempRoot }; + const childEnv = { ...env, TMPDIR: tempRoot, TEMP: tempRoot, TMP: tempRoot, + AK_SUITE_ROOT: tempRoot, AK_SUITE_RUN_ID: owner.runId }; // Tests assert on plain text; a shell's FORCE_COLOR (Claude Code sets 3) // colours console.log into pipes and, beside NO_COLOR, adds a Node warning. delete childEnv.FORCE_COLOR; @@ -74,16 +104,33 @@ export function runGuarded(commands, { let code = 0; for (const args of commands) { const r = spawnSync(process.execPath, args, { cwd: repoRoot, env: childEnv, stdio: 'inherit' }); + if (r.signal) { log(`interrupted run; kept run root ${tempRoot}: ${r.signal}`); return 1; } if (r.error) { log(`could not run node ${args.join(' ')}: ${r.error.message}`); code = 1; break; } if (r.status !== 0) { code = r.status ?? 1; break; } } - const leftovers = fs.readdirSync(tempRoot).filter((name) => name !== 'node-compile-cache'); - fs.rmSync(tempRoot, { recursive: true, force: true, maxRetries: 3 }); + let leftovers = []; + let ownHygieneFailed = false; + try { leftovers = fs.readdirSync(tempRoot).filter((name) => !IGNORED_IN_ROOT.has(name)); } + catch (error) { + log(`could not list own run root; kept ${tempRoot}: ${error.message}`); + ownHygieneFailed = true; + } + // A child may have explicitly declared unresolved ownership before launch. + // Ordinary test failures still remove their own roots when all holds clear. + if (!ownHygieneFailed) { + const holds = inspectRunRootHolds(tempRoot, owner.runId); + if (holds.unresolved) { + log(`kept own run root ${tempRoot}: ${holds.reason}`); + ownHygieneFailed = true; + } else if (!removeOwnRoot()) ownHygieneFailed = true; + } + try { collectAbandonedRoots({ tmpdir, selfRoot: tempRoot, homedir, log }); } + catch (error) { log(`could not list sibling run roots: ${error.message}`); } if (leftovers.length) log(`temp folders left behind by the run (${leftovers.length}):\n ${leftovers.join('\n ')}`); const result = compareSnapshots(before, snapshotRoots(roots), { strict: isStrict(env) }); const report = formatReport(result); if (report) log(report); - return code || (result.failing.length ? 3 : 0) || (leftovers.length ? 4 : 0); + return code || (result.failing.length ? 3 : 0) || (leftovers.length || ownHygieneFailed ? 4 : 0); } /** The nearest folder at or above `dir` that holds a `.git` entry, or null. */ @@ -100,6 +147,15 @@ function enclosingRepository(dir) { function main(argv) { const [mode] = argv; if (mode === 'unit' || mode === 'ui') return runGuarded(commandsFor(mode)); + if (mode === 'focus') { + const files = argv.slice(1); + // Reject Node options disguised as filenames before constructing an argv vector. + if (!files.length || files.some((file) => !file || file.startsWith('-') || !isTestFile(file))) { + console.error('usage: run-tests.mjs focus (each file must exist)'); + return 2; + } + return runGuarded([['--test', ...files]]); + } if (mode === 'exec') { const sep = argv.indexOf('--'); const repoAt = argv.indexOf('--repo'); @@ -107,10 +163,15 @@ function main(argv) { const repoRoot = repoAt >= 0 && repoAt < sep ? path.resolve(argv[repoAt + 1]) : REPO; return runGuarded([argv.slice(sep + 1)], { repoRoot }); } - console.error('usage: run-tests.mjs unit|ui|exec'); + console.error('usage: run-tests.mjs unit|ui|exec|focus'); return 2; } +function isTestFile(file) { + try { return fs.statSync(path.resolve(REPO, file)).isFile(); } + catch { return false; } +} + // Compare real paths (drive-letter case differs on Windows): a missed match would // make `pnpm test` exit 0 having run nothing. const isMain = () => { diff --git a/scripts/trace-ort.mjs b/scripts/trace-ort.mjs new file mode 100644 index 00000000..7c5afcac --- /dev/null +++ b/scripts/trace-ort.mjs @@ -0,0 +1,77 @@ +// Passive adaptation of vidaunited's Node resolution hook: +// https://github.com/ruvnet/ruflo/issues/2885#issuecomment-5867331087 +// Enable only with an explicit absolute TRACE_ORT_LOG and NODE_OPTIONS=--import=. +import * as moduleApi from 'node:module'; +import fs from 'node:fs'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const SOURCE = 'ruvnet/ruflo#2885:issuecomment-5867331087'; +const log = process.env.TRACE_ORT_LOG; +const seen = new Set(); +let warned = false; + +function warn(message) { + if (warned) return; + warned = true; + try { process.stderr.write(`[trace-ort] ${message}\n`); } catch { /* observation is best effort */ } +} + +function append(record) { + try { + fs.appendFileSync(log, `${JSON.stringify({ schema: 1, pid: process.pid, ...record })}\n`, { flag: 'a' }); + } catch { + warn('log unavailable; trace artifact is incomplete'); + } +} + +function packageAt(file) { + const parts = path.normalize(file).split(path.sep); + for (let i = parts.length - 2; i >= 0; i--) { + if (parts[i] !== 'node_modules') continue; + const first = parts[i + 1]; + const scoped = first === '@huggingface' || first === '@xenova'; + const name = scoped ? `${first}/${parts[i + 2]}` : first; + if (!['@huggingface/transformers', '@xenova/transformers', 'onnxruntime-node'].includes(name)) continue; + const end = i + (scoped ? 3 : 2); + if (parts.length <= end) continue; + return { name, root: parts.slice(0, end).join(path.sep) || path.sep }; + } + return null; +} + +function versionAt(root) { + try { + const metadata = JSON.parse(fs.readFileSync(path.join(root, 'package.json'), 'utf8')); + return typeof metadata.version === 'string' && metadata.version.length <= 128 && metadata.version.length > 0 + ? metadata.version : 'unknown'; + } catch { return 'unknown'; } +} + +function note(url) { + if (!url?.startsWith('file:')) return; + const found = packageAt(fileURLToPath(url)); + if (!found || seen.has(found.root)) return; + seen.add(found.root); + const rootTruncated = found.root.length > 4096; + append({ type: 'package', name: found.name, version: versionAt(found.root), + root: found.root.slice(0, 4096), ...(rootTruncated ? { rootTruncated: true } : {}) }); +} + +if (typeof log !== 'string' || !path.isAbsolute(log)) { + warn('log target missing or relative; set absolute TRACE_ORT_LOG'); +} else if (typeof moduleApi.registerHooks !== 'function') { + warn('Node module.registerHooks unavailable; requires Node 22.15 or newer'); +} else { + append({ type: 'start', hook: 'trace-ort/1', source: SOURCE, node: process.version, + platform: process.platform, arch: process.arch }); + try { + moduleApi.registerHooks({ + resolve(specifier, context, nextResolve) { + const result = nextResolve(specifier, context); + try { note(result.url); } catch { warn('resolution observation failed; trace artifact is incomplete'); } + return result; + }, + }); + } catch { warn('hook registration failed; trace artifact is incomplete'); } +} diff --git a/scripts/upstream-watch.mjs b/scripts/upstream-watch.mjs index a2ce70f2..a92b3e21 100644 --- a/scripts/upstream-watch.mjs +++ b/scripts/upstream-watch.mjs @@ -17,7 +17,7 @@ import { import { createDispatcher, dispatch } from './upstream-watch/dispatch.mjs'; import { createFetcher, mapLimit, retrying } from './upstream-watch/fetch.mjs'; import { createLedgerStore, toRecord } from './upstream-watch/ledger-branch.mjs'; -import { commitSafe, isoSeconds, renderNotice, sentence } from './upstream-watch/ledger.mjs'; +import { LEDGER_EVENTS, commitSafe, isoSeconds, renderNotice, sentence } from './upstream-watch/ledger.mjs'; import { renderEvents, renderReport } from './upstream-watch/render.mjs'; const USAGE = `usage: node scripts/upstream-watch.mjs report [--json] [--concurrency <1-16>] [--registry ] @@ -28,7 +28,6 @@ const USAGE = `usage: node scripts/upstream-watch.mjs report [--json] [--concurr const PENDING = new Set(['watching', 'fixed-unreleased']); // record exits BLIND when gh, the registry, the ledger branch or every upstream thread is unreadable. const BLIND = 3; -const LEDGER_EVENTS = ['reply', 'acknowledged', 'closed', 'merged', 'released', 'reopened', 'stale', 'retire-proposed', 'retest-due', 'idle', 'fired', 'dispatch-pr']; class UsageError extends Error {} @@ -248,8 +247,8 @@ async function ledgerQuery(registry, options, { stdout, stderr, now, ledgerStore const WEEK = 7 * 86_400_000; -function blindRecord(error, { stdout, stderr, json }, extra = {}) { - stderr.write(`${error}\n`); +function blindRecord(error, { stdout, stderr, json }, extra = {}, { writeError = true } = {}) { + if (writeError) stderr.write(`${error}\n`); const result = { blind: true, error, records: [], fetchErrors: [], dispatchErrors: [], wouldFire: [], deferred: [], fired: [], parent: null, commit: null, notice: { post: false, body: '' }, ...extra, }; @@ -286,7 +285,8 @@ async function record(registry, fetcher, options, { stdout, stderr, now, ledgerS const recorded = ledger.records.map((item) => item.line).join('\n'); const all = ledgerEvents(report, registry, { since }); const released = all.filter((event) => event.event === 'released' && event.fields.branch); - const fired = await dispatch({ released, records: ledger.records, dispatcher, repo, sentinel, now, recordedAt: runAt, dryRun: options.dryRun, pause: sleep }); + const eligibleIds = new Set(registry.watch.filter((entry) => PENDING.has(entry.status)).map((entry) => entry.id)); + const fired = await dispatch({ released, records: ledger.records, dispatcher, repo, sentinel, now, recordedAt: runAt, dryRun: options.dryRun, eligibleIds, pause: sleep }); const records = [...withoutRecorded(all, recorded).map((event) => toRecord(event, runAt)), ...fired.records]; const checkedAt = fetchErrors.length ? (ledger.checkedAt ?? since) : runAt; let commit = null; @@ -298,10 +298,10 @@ async function record(registry, fetcher, options, { stdout, stderr, now, ledgerS sentences: records.map((item) => commitSafe(sentence(item))), }); } catch (error) { - // The routine already ran for these; without the commit the next run fires again. + // These sessions were observed; without the commit, later runs may fire again. const sessions = fired.records.filter((item) => item.event === 'fired'); for (const item of sessions) stderr.write(`Fired ${item.id} before the ledger commit failed: session ${item.fields.session}\n`); - return blindRecord(`Could not build the ledger commit: ${error.message}`, io, { fetchErrors, dispatchErrors: fired.errors, fired: sessions }); + return blindRecord(`Could not build the ledger commit: ${error.message}`, io, { fetchErrors, dispatchErrors: fired.errors, fired: sessions, deferred: fired.deferred }); } } const body = renderNotice({ records, mention, date: runAt.slice(0, 10), recordedAt: runAt }); @@ -312,7 +312,7 @@ async function record(registry, fetcher, options, { stdout, stderr, now, ledgerS for (const item of fetchErrors) stderr.write(`Could not check ${item.id}: ${item.error}\n`); for (const item of fired.errors) stderr.write(`Dispatch ${item.id}: ${item.error}\n`); const lines = [...records.map((item) => item.line), ...fired.wouldFire.map((item) => `Would fire ${item.id} ${item.version} ${item.branch}`), - ...fired.deferred.map((item) => `Deferred to the next run: ${item.id} ${item.version} ${item.branch}`)]; + ...fired.deferred.map((item) => `Deferred for a later check: ${item.id} ${item.version} ${item.branch}`)]; stdout.write(options.json ? `${JSON.stringify(result, null, 2)}\n` : lines.length ? `${lines.join('\n')}\n` : 'No new records.\n'); return 0; } @@ -334,7 +334,11 @@ export async function main(argv, { stderr.write(`upstream registry is ${registry.registryStatus ?? registry.status}:\n${registry.errors.map((error) => ` ${error}`).join('\n')}\n`); const status = { status: registry.registryStatus ?? registry.status, errors: registry.errors }; if (options.command === 'record') { - return blindRecord(`upstream registry is ${status.status}`, { stdout, stderr, json: options.json }, { registry: status }); + return blindRecord(`upstream registry is ${status.status}`, { stdout, stderr, json: options.json }, { registry: status }, { writeError: false }); + } + if (options.command === 'ledger') { + if (options.json) stdout.write(`${JSON.stringify({ registry: status }, null, 2)}\n`); + return BLIND; } stdout.write(options.json ? `${JSON.stringify({ registry: status }, null, 2)}\n` : 'No report: the upstream registry is not valid.\n'); return 0; diff --git a/scripts/upstream-watch/dispatch.mjs b/scripts/upstream-watch/dispatch.mjs index a2029eb2..d4b1e76b 100644 --- a/scripts/upstream-watch/dispatch.mjs +++ b/scripts/upstream-watch/dispatch.mjs @@ -3,7 +3,7 @@ // at most MAX_FIRES times and not again within REFIRE_AFTER_DAYS, and record // the draft pull request once the branch has one. A run fires at most // MAX_FIRES_PER_RUN fixes, spaced apart, and stops at the first failed call; -// the rest are deferred to the next run. A dry run lists what would fire +// the rest stay eligible for later checks. A dry run lists what would fire // instead of firing. Injectable exec, fetch and pause. import { eventLine } from './classify.mjs'; import { run } from './fetch.mjs'; @@ -12,12 +12,14 @@ import { toRecord } from './ledger-branch.mjs'; export const FIRE_URL = (routine) => `https://api.anthropic.com/v1/claude_code/routines/${routine}/fire`; export const FIRE_HEADERS = { 'anthropic-beta': 'experimental-cc-routine-2026-04-01', 'anthropic-version': '2023-06-01', 'content-type': 'application/json' }; export const REFIRE_AFTER_DAYS = 3; +export const PR_OBSERVE_DAYS = 7; export const MAX_FIRES = 2; export const FIRE_TIMEOUT_MS = 30_000; -// Sessions start gradually: a few per run, spaced apart; the rest wait for the next run. +// Sessions start gradually: a few per run, spaced apart; later checks revisit the rest. export const MAX_FIRES_PER_RUN = 3; export const FIRE_SPACING_MS = 15_000; -// A 5xx answer means no session was started, so the call is safe to repeat. +// The endpoint documents retries for HTTP 500/503, but has no idempotency key. +// These bounded retries do not guarantee that only one server-side session exists. export const FIRE_ATTEMPTS = 3; export const FIRE_BACKOFF_MS = 2_000; // `gh pr list --head` matches the branch name in any fork; only a pull request @@ -27,6 +29,17 @@ const ROUTINE = /^trig_[A-Za-z0-9]+$/; const DISPATCH_BRANCH = /^upstream\/[a-z0-9._-]+$/; const DAY = 86_400_000; +function diagnostic(value, token) { + if (typeof value !== 'string') return ''; + // Redact before bounding, so truncation cannot expose the start of a token. + return value.split(token).join('[REDACTED]').replace(/[\p{Cc}\p{Cf}]/gu, ' ').slice(0, 200); +} + +export function sessionList(sessions) { + if (sessions.length < 3) return sessions.join(' and '); + return `${sessions.slice(0, -1).join(', ')}, and ${sessions.at(-1)}`; +} + export function createDispatcher({ exec = run, fetchImpl = globalThis.fetch, env = process.env, sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms)) } = {}) { return { async branchExists(branch) { @@ -56,14 +69,19 @@ export function createDispatcher({ exec = run, fetchImpl = globalThis.fetch, env method: 'POST', headers: { ...FIRE_HEADERS, authorization: `Bearer ${token}` }, body: JSON.stringify({ text }), signal: AbortSignal.timeout(FIRE_TIMEOUT_MS), }); - const body = await response.json().catch(() => null); + const body = await response.json().catch((error) => { + // Only a complete non-JSON response uses the status-only fallback. + // A body-stream failure is ambiguous and must not cause another POST. + if (error instanceof SyntaxError) return null; + throw error; + }); const session = body?.claude_code_session_url; if (response.status === 200 && typeof session === 'string') return session; - const requestId = response.headers?.get?.('request-id'); - const detail = body?.error?.message ?? (typeof body?.error === 'string' ? body.error : ''); + const requestId = diagnostic(response.headers?.get?.('request-id'), token); + const detail = diagnostic(body?.error?.message ?? (typeof body?.error === 'string' ? body.error : ''), token); failure = `the routine trigger answered HTTP ${response.status}${typeof session === 'string' ? '' : ' without a session'}` - + `${requestId ? ` (request-id ${requestId})` : ''}${detail ? `: ${String(detail).slice(0, 200)}` : ''}`; - if (response.status < 500) break; + + `${requestId ? ` (request-id ${requestId})` : ''}${detail ? `: ${detail}` : ''}`; + if (typeof session === 'string' || ![500, 503].includes(response.status)) break; } throw new Error(failure); }, @@ -71,7 +89,7 @@ export function createDispatcher({ exec = run, fetchImpl = globalThis.fetch, env } /** Fire (or, in a dry run, list in `wouldFire`) each released fix; the branch and pull request lookups only read. */ -export async function dispatch({ released, records, dispatcher, repo, sentinel, now, recordedAt, dryRun = false, pause = (ms) => new Promise((resolve) => setTimeout(resolve, ms)) }) { +export async function dispatch({ released, records, dispatcher, repo, sentinel, now, recordedAt, dryRun = false, eligibleIds = null, pause = (ms) => new Promise((resolve) => setTimeout(resolve, ms)) }) { const out = []; const errors = []; const wouldFire = []; @@ -86,19 +104,20 @@ export async function dispatch({ released, records, dispatcher, repo, sentinel, if (await dispatcher.branchExists(branch)) continue; const firings = recordsOf(event.id, 'fired'); if (firings.length >= MAX_FIRES) { - errors.push({ id: event.id, error: `dispatch did not complete after ${firings.length} firings; see ${firings.map((item) => item.fields.session).join(' and ')}` }); + errors.push({ id: event.id, error: `dispatch did not complete after ${firings.length} firings; see ${sessionList(firings.map((item) => item.fields.session))}` }); continue; } const newest = Math.max(...firings.map((item) => Date.parse(item.recordedAt)), 0); if (newest && now.getTime() - newest < REFIRE_AFTER_DAYS * DAY) continue; - if (dryRun) { - wouldFire.push({ id: event.id, version, branch }); - continue; - } if (triggerFailed || attempted >= MAX_FIRES_PER_RUN) { deferred.push({ id: event.id, version, branch }); continue; } + if (dryRun) { + attempted++; + wouldFire.push({ id: event.id, version, branch }); + continue; + } if (attempted++) await pause(FIRE_SPACING_MS); try { const session = await dispatcher.fire(`${event.id} ${version} ${branch}`); @@ -113,6 +132,9 @@ export async function dispatch({ released, records, dispatcher, repo, sentinel, } for (const id of new Set(records.filter((item) => item.event === 'fired').map((item) => item.id))) { if (recordsOf(id, 'dispatch-pr').length) continue; + if (eligibleIds && !eligibleIds.has(id)) continue; + const latestFiring = Math.max(...recordsOf(id, 'fired').map((item) => Date.parse(item.recordedAt))); + if (!Number.isFinite(latestFiring) || now.getTime() - latestFiring >= PR_OBSERVE_DAYS * DAY) continue; try { const branch = recordsOf(id, 'fired').at(-1).fields.branch; const pr = await dispatcher.openPullRequest(repo, branch); diff --git a/scripts/upstream-watch/fetch.mjs b/scripts/upstream-watch/fetch.mjs index 198fad9a..5077d683 100644 --- a/scripts/upstream-watch/fetch.mjs +++ b/scripts/upstream-watch/fetch.mjs @@ -14,6 +14,10 @@ const VERSION = /^\d+\.\d+\.\d+(?:-[\w.]+)?$/; const NOT_FOUND = /HTTP 404|Not Found/i; const NO_MATCH = /No match found for version/; +/** An invalid local argument or fixture cannot recover by waiting for the network. */ +export class PermanentFetchError extends Error {} +const invalid = (message) => new PermanentFetchError(message); + // What closed a thread: its closing pull requests, else the ClosedEvent's closer. export const FIXING_CHANGES_QUERY = `query($owner:String!,$name:String!,$number:Int!){repository(owner:$owner,name:$name){defaultBranchRef{name} issueOrPullRequest(number:$number){__typename ... on Issue{closedByPullRequestsReferences(first:10,includeClosedPrs:true){nodes{number merged mergedAt baseRefName mergeCommit{oid} repository{nameWithOwner}}} timelineItems(last:1,itemTypes:[CLOSED_EVENT]){nodes{... on ClosedEvent{closer{__typename ... on Commit{oid} ... on PullRequest{number merged mergedAt baseRefName mergeCommit{oid} repository{nameWithOwner}}}}}}} ... on PullRequest{number merged mergedAt baseRefName mergeCommit{oid} repository{nameWithOwner}}}}}`; @@ -80,7 +84,7 @@ export function createFetcher({ exec = run } = {}) { }, async thread(id) { const [, repo, number] = ID.exec(id) ?? []; - if (!repo) throw new Error(`not an owner/repo#number id: ${id}`); + if (!repo) throw invalid(`not an owner/repo#number id: ${id}`); const issue = await json('gh', ['api', `repos/${repo}/issues/${number}`]); return { issue, comments: await this.comments(repo, number) }; }, @@ -89,7 +93,7 @@ export function createFetcher({ exec = run } = {}) { * per line; `--slurp` would need gh 2.48, newer than apt's gh on Ubuntu 24.04. */ async comments(repo, number) { - if (!OWNER_REPO.test(repo ?? '') || !/^[1-9]\d*$/.test(String(number))) throw new Error(`not an issue: ${repo}#${number}`); + if (!OWNER_REPO.test(repo ?? '') || !/^[1-9]\d*$/.test(String(number))) throw invalid(`not an issue: ${repo}#${number}`); const args = ['api', '--paginate', '--jq', '.[]', `repos/${repo}/issues/${number}/comments?per_page=100`]; const result = await exec('gh', args); if (result.status !== 0) throw new Error(`gh ${args.join(' ')} failed: ${(result.stderr || result.error?.message || 'no output').trim()}`); @@ -98,7 +102,7 @@ export function createFetcher({ exec = run } = {}) { /** Merged pull requests (or the closing commit) that fixed a thread; empty when none qualifies. */ async fixingChanges(id) { const [, repo, number] = ID.exec(id) ?? []; - if (!repo) throw new Error(`not an owner/repo#number id: ${id}`); + if (!repo) throw invalid(`not an owner/repo#number id: ${id}`); const [owner, name] = repo.split('/'); const answer = await json('gh', ['api', 'graphql', '-f', `query=${FIXING_CHANGES_QUERY}`, '-F', `owner=${owner}`, '-F', `name=${name}`, '-F', `number=${number}`]); return changesOf(repo, answer?.data?.repository); @@ -109,10 +113,10 @@ export function createFetcher({ exec = run } = {}) { * a rate limit never reads as "not contained". */ async contains(repo, refs, sha) { - if (!SHA.test(sha ?? '')) throw new Error(`not a commit: ${sha}`); - if (!OWNER_REPO.test(repo ?? '')) throw new Error(`not an owner/repo: ${repo}`); + if (!SHA.test(sha ?? '')) throw invalid(`not a commit: ${sha}`); + if (!OWNER_REPO.test(repo ?? '')) throw invalid(`not an owner/repo: ${repo}`); for (const ref of refs) { - if (!REF.test(ref ?? '')) throw new Error(`not a tag name: ${ref}`); + if (!REF.test(ref ?? '')) throw invalid(`not a tag name: ${ref}`); const args = ['api', `repos/${repo}/compare/${ref}...${sha}`, '--jq', '{status:.status}']; const result = await exec('gh', args); if (result.status !== 0) { @@ -120,7 +124,7 @@ export function createFetcher({ exec = run } = {}) { throw new Error(`gh ${args.join(' ')} failed: ${(result.stderr || result.error?.message || 'no output').trim()}`); } const { status } = JSON.parse(result.stdout); - if (!['behind', 'identical', 'ahead', 'diverged'].includes(status)) throw new Error(`gh ${args.join(' ')} returned status ${status}`); + if (!['behind', 'identical', 'ahead', 'diverged'].includes(status)) throw invalid(`gh ${args.join(' ')} returned status ${status}`); return { ref, contained: status === 'behind' || status === 'identical' }; } return { ref: null, contained: null }; @@ -132,8 +136,8 @@ export function createFetcher({ exec = run } = {}) { * every range to its highest published match with npm. */ async bundled(chain, name, at = null) { - for (const pkg of [...chain, name]) if (!PACKAGE_NAME.test(pkg ?? '')) throw new Error(`not a package name: ${pkg}`); - if (at !== null && !VERSION.test(at)) throw new Error(`not a version: ${at}`); + for (const pkg of [...chain, name]) if (!PACKAGE_NAME.test(pkg ?? '')) throw invalid(`not a package name: ${pkg}`); + if (at !== null && !VERSION.test(at)) throw invalid(`not a version: ${at}`); const carrierVersion = at ?? await json('npm', ['view', chain[0], 'version', '--json']); let [pkg, version] = [chain[0], carrierVersion]; const trail = [`${pkg} ${version}`]; @@ -159,13 +163,13 @@ export function createFetcher({ exec = run } = {}) { * only scheduled runs show that the watch is alive. */ async lastRun(repo) { - if (!OWNER_REPO.test(repo ?? '')) throw new Error(`not an owner/repo: ${repo}`); + if (!OWNER_REPO.test(repo ?? '')) throw invalid(`not an owner/repo: ${repo}`); const answer = await json('gh', ['api', `repos/${repo}/actions/workflows/upstream-watch.yml/runs?status=success&event=schedule&per_page=1`]); const run = answer?.workflow_runs?.[0]; return run ? { at: run.run_started_at, url: run.html_url } : null; }, async release({ channel, name }) { - if (!PACKAGE_NAME.test(name)) throw new Error(`not a package or repository name: ${name}`); + if (!PACKAGE_NAME.test(name)) throw invalid(`not a package or repository name: ${name}`); if (channel === 'npm') return releaseFacts('npm', await json('npm', ['view', name, 'time', 'dist-tags', '--json'])); return releaseFacts('github-release', await json('gh', ['api', `repos/${name}/releases?per_page=100`])); }, @@ -186,7 +190,7 @@ export function retrying(fetcher, { delays = [2000, 10_000], sleep = (ms) => new try { return await method.apply(fetcher, args); } catch (error) { - if (attempt >= delays.length) throw error; + if (error instanceof PermanentFetchError || attempt >= delays.length) throw error; await sleep(delays[attempt]); } } diff --git a/scripts/upstream-watch/ledger.mjs b/scripts/upstream-watch/ledger.mjs index 88a27467..12e75be5 100644 --- a/scripts/upstream-watch/ledger.mjs +++ b/scripts/upstream-watch/ledger.mjs @@ -5,6 +5,9 @@ // autolinks to this repository and `owner/repo#n` mentions the upstream thread. const code = (value) => `\`${value}\``; +/** Event names accepted by the ledger query and rendered below. */ +export const LEDGER_EVENTS = ['reply', 'acknowledged', 'closed', 'merged', 'released', 'reopened', 'stale', 'retire-proposed', 'retest-due', 'idle', 'fired', 'dispatch-pr']; + export const isoSeconds = (date) => new Date(date).toISOString().replace(/\.\d{3}Z$/, 'Z'); /** @@ -80,10 +83,13 @@ export function renderNotice({ records, mention, date, recordedAt }) { const bullets = items.map((item) => `- ${sentence(item)}${item.event === 'released' && sessions.has(item.id) ? ` Routine session: ${sessions.get(item.id)}` : ''}`); const head = `@${mention} upstream watch: ${items.length} ${items.length === 1 ? 'item needs' : 'items need'} you (${date}).`; const foot = `The full record: \`node scripts/upstream-watch.mjs ledger --recorded-since ${recordedAt}\``; + const lengths = [0]; + for (const bullet of bullets) lengths.push(lengths.at(-1) + bullet.length); for (let count = bullets.length; count > 0; count--) { const more = bullets.length - count; - const body = `${[head, '', ...bullets.slice(0, count), ...(more ? ['', `${more} more; see the ledger.`] : []), '', foot].join('\n')}\n`; - if (body.length <= NOTICE_MAX) return body; + const suffix = more ? `${more} more; see the ledger.` : ''; + const length = head.length + foot.length + lengths[count] + suffix.length + count + 4 + (more ? 2 : 0); + if (length <= NOTICE_MAX) return `${[head, '', ...bullets.slice(0, count), ...(more ? ['', suffix] : []), '', foot].join('\n')}\n`; } return `${[head, '', `${bullets.length} items; see the ledger.`, '', foot].join('\n')}\n`; } diff --git a/src/commands/audit.mjs b/src/commands/audit.mjs index 90a1141f..69535fcb 100644 --- a/src/commands/audit.mjs +++ b/src/commands/audit.mjs @@ -9,6 +9,7 @@ import { collectContextEvidence } from '../lib/context-audit-sources.mjs'; import { loadKitConfig } from '../lib/config.mjs'; import { projectCensus, projectsInScope } from '../lib/project-census.mjs'; import { installedVersion } from '../lib/versions.mjs'; +import { reportFailure } from '../lib/output.mjs'; export const options = { json: { type: 'boolean', default: false }, @@ -152,8 +153,11 @@ export async function run({ contextCollectorFn = collectContextAudit, }) { if (positionals.length !== 1 || !['hooks', 'context'].includes(positionals[0])) { - console.error('ak audit requires the hooks or context subcommand'); - console.log(help); + const error = 'ak audit requires the hooks or context subcommand'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => { + console.error(error); + console.log(help); + } }); return 2; } if (positionals[0] === 'context') { @@ -161,7 +165,8 @@ export async function run({ try { report = await contextCollectorFn({ flags, pkgRoot, loadConfigFn }); } catch (error) { - console.error(`context audit failed: ${error.message}`); + const message = `context audit failed: ${error.message}`; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => console.error(message) }); return 2; } if (flags.json) console.log(JSON.stringify(report, null, 2)); @@ -172,7 +177,8 @@ export async function run({ try { report = collectHookAudit({ flags, detectVersionFn, loadConfigFn }); } catch (error) { - console.error(`hook audit failed: ${error.message}`); + const message = `hook audit failed: ${error.message}`; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => console.error(message) }); return 2; } if (flags.json) { diff --git a/src/commands/heal.mjs b/src/commands/heal.mjs index cd68085e..8b30d965 100644 --- a/src/commands/heal.mjs +++ b/src/commands/heal.mjs @@ -1,4 +1,5 @@ import path from 'node:path'; +import { reportFailure } from '../lib/output.mjs'; import { collectHookAudit } from './audit.mjs'; import { @@ -99,25 +100,28 @@ function validateMode(flags) { if (flags.apply && !flags.yes) throw new TypeError('--apply requires --yes'); } +function usageError(flags, message, showHelp = false) { + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => { + console.error(message); + if (showHelp) console.log(help); + } }); + return 2; +} + export async function run({ flags, positionals, detectVersionFn, loadConfigFn }) { if (positionals.length !== 1 || positionals[0] !== 'hooks') { - console.error('ak heal requires the hooks subcommand'); - console.log(help); - return 2; + return usageError(flags, 'ak heal requires the hooks subcommand', true); } try { validateMode(flags); } catch (error) { - console.error(`hook healing refused: ${error.message}`); - return 2; + return usageError(flags, `hook healing refused: ${error.message}`); } const transactionsRoot = transactionRoot(flags); if (flags.undo && flags.recover) { - console.error('hook healing refused: --undo and --recover are mutually exclusive'); - return 2; + return usageError(flags, 'hook healing refused: --undo and --recover are mutually exclusive'); } if (flags.undo || flags.recover) { if ((flags.action?.length ?? 0) || flags['plan-digest']) { - console.error('hook healing refused: rollback/recovery cannot be combined with plan action flags'); - return 2; + return usageError(flags, 'hook healing refused: rollback/recovery cannot be combined with plan action flags'); } const result = flags.recover ? (flags.apply @@ -145,8 +149,7 @@ export async function run({ flags, positionals, detectVersionFn, loadConfigFn }) return audit.summary.invalidSources || audit.summary.configurationIssues ? 1 : 0; } if (!flags.action?.length || !flags['plan-digest']) { - console.error('hook healing refused: apply requires --action and --plan-digest from a preview'); - return 2; + return usageError(flags, 'hook healing refused: apply requires --action and --plan-digest from a preview'); } if (unfinishedTransactions.length) { throw new Error(`unfinished hook transaction(s) require --recover first: ${unfinishedTransactions.map((item) => item.id).join(', ')}`); @@ -160,7 +163,6 @@ export async function run({ flags, positionals, detectVersionFn, loadConfigFn }) }); return printResult(result, flags.json); } catch (error) { - console.error(`hook healing failed: ${error.message}`); - return 2; + return usageError(flags, `hook healing failed: ${error.message}`); } } diff --git a/src/commands/models.mjs b/src/commands/models.mjs index 89026435..b9562567 100644 --- a/src/commands/models.mjs +++ b/src/commands/models.mjs @@ -1,4 +1,4 @@ -import { heading, info, ok, warn, dim } from '../lib/output.mjs'; +import { heading, info, ok, warn, dim, reportFailure } from '../lib/output.mjs'; import { loadKitConfig } from '../lib/config.mjs'; import { aqeRouterFile } from '../lib/providers.mjs'; import { readJson } from '../lib/settings.mjs'; @@ -129,10 +129,6 @@ async function runRefresh(ctx) { function runStatus(ctx) { const { flags, cacheFile, store, latest } = ctx; - if (flags.host && !ALL_OWNERS.includes(flags.host)) { - warn(`unsupported model host: ${flags.host}`); - return 2; - } const snapshot = visibleSnapshot(latest, flags.host); const since = flags.since ? Date.parse(flags.since) : null; const history = store.snapshots.filter((entry) => entry.scope.fingerprint === latest.scope.fingerprint @@ -174,7 +170,7 @@ function runDiff(ctx) { function runExplain(ctx) { const { positionals, flags, latest } = ctx; const selector = positionals[1] ?? flags.to; - if (!selector) { warn('usage: ak models explain HOST:MODEL'); return 2; } + if (!selector) { modelUsageError(flags, 'usage: ak models explain HOST:MODEL'); return 2; } const result = explainModel(latest, selector); if (flags.json) printJson(result); else if (!result.found) warn(`Model not found: ${selector}`); @@ -193,7 +189,7 @@ function runPlan(ctx) { const { flags, positionals, latest } = ctx; const activity = flags.activity; const to = flags.to ?? positionals[1]; - if (!activity || !to) { warn('usage: ak models plan --activity ACTIVITY [--from HOST:MODEL] --to HOST:MODEL'); return 2; } + if (!activity || !to) { modelUsageError(flags, 'usage: ak models plan --activity ACTIVITY [--from HOST:MODEL] --to HOST:MODEL'); return 2; } const result = planModelChange(latest, { activity, from: flags.from, to }); if (flags.json) printJson(result); else { @@ -211,9 +207,34 @@ function runPlan(ctx) { // dispatched by name once that store/latest snapshot is in hand (below). const READ_ACTIONS = { status: runStatus, diff: runDiff, explain: runExplain, plan: runPlan }; +function modelUsageError(flags, message) { + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => warn(message) }); +} + /** @param {{flags: Record, positionals: string[], deps?: Record}} input */ export async function run({ flags, positionals, deps = {} }) { const action = positionals[0] ?? 'status'; + if (action !== 'refresh' && !Object.hasOwn(READ_ACTIONS, action)) { + modelUsageError(flags, 'usage: ak models status|refresh|diff|explain|plan'); + return 2; + } + const maxPositionals = action === 'diff' ? 3 : action === 'explain' || action === 'plan' ? 2 : 1; + if (positionals.length > maxPositionals) { + modelUsageError(flags, `unexpected argument '${positionals[maxPositionals]}'`); + return 2; + } + if (action === 'explain' && !positionals[1] && !flags.to) { + modelUsageError(flags, 'usage: ak models explain HOST:MODEL'); + return 2; + } + if (action === 'plan' && (!flags.activity || !(flags.to ?? positionals[1]))) { + modelUsageError(flags, 'usage: ak models plan --activity ACTIVITY [--from HOST:MODEL] --to HOST:MODEL'); + return 2; + } + if (action === 'status' && flags.host && !ALL_OWNERS.includes(flags.host)) { + modelUsageError(flags, `unsupported model host: ${flags.host}`); + return 2; + } const cacheFile = deps.cacheFile ?? modelInventoryPath(); const readStore = deps.readStore ?? readModelStore; const append = deps.append ?? appendModelSnapshot; @@ -229,6 +250,5 @@ export async function run({ flags, positionals, deps = {} }) { if (!latest) return noSnapshot(flags, cacheFile); const handler = READ_ACTIONS[action]; - if (!handler) { warn('usage: ak models status|refresh|diff|explain|plan'); return 2; } return handler({ ...ctx, store, latest }); } diff --git a/src/commands/setup.mjs b/src/commands/setup.mjs index 772294d5..fc649db6 100644 --- a/src/commands/setup.mjs +++ b/src/commands/setup.mjs @@ -5,6 +5,7 @@ // Project scope (when run inside a git repo / --project): the port of // ruflo-setup-project — init, sanitize, pin, activate, verify, daemon. import fs from 'node:fs'; +import os from 'node:os'; import path from 'node:path'; import readline from 'node:readline/promises'; import { run as runCmd, have } from '../lib/exec.mjs'; @@ -381,21 +382,24 @@ function deployTokenAuditSkill(pkgRoot) { * left alone. Shares HOSTS/hostInstallState/installHost with `ak sync`'s * and `ak host pick`'s own host-install loops; the interactive confirmation * here (vs. their unconditional install) is this command's own UX. */ -async function installEnabledAbsentHosts(cfg, flags) { +export async function installEnabledAbsentHosts(cfg, flags, lifecycle = {}) { + const { installState, install, collectFacts } = { + installState: hostInstallState, install: installHost, collectFacts: collectIntegrationFacts, ...lifecycle, + }; let installed = false; for (const h of HOSTS) { if (!cfg.integrations?.hosts?.[h.id]) continue; - const st = await hostInstallState(h); + const st = await installState(h); if (st.method === 'absent') { if (await ask(`${h.id} CLI not found — install ${h.pkg} globally?`, true, flags.yes)) { - const r = await installHost(h.id); + const r = await install(h.id); (r.ok ? ok : warn)(`${h.id}: ${r.detail}`); if (r.ok) { installed = true; // hostInstallState() above already recorded the pre-install // 'absent' evidence; re-probe now so a subsequent `ak status` // doesn't read that stale row back. - await hostInstallState(h, { refresh: true, record: true, source: 'setup' }); + await installState(h, { refresh: true, record: true, source: 'setup' }); } } else warn(`${h.id} not installed — enable/install later with: ak host pick`); } else { @@ -404,7 +408,7 @@ async function installEnabledAbsentHosts(cfg, flags) { } // host-setup covers every host in one call; refresh it once after the // loop, not per host, once anything actually changed. - if (installed) await collectIntegrationFacts({ cfg, refresh: true, record: true, source: 'setup' }); + if (installed) await collectFacts({ cfg, refresh: true, record: true, source: 'setup' }); } /** Step 6b: host lifecycle wiring — connected MCPs, compact lazy gateway, @@ -467,7 +471,7 @@ async function printUndetectedHostHints(cfg) { } } -export async function run_machine({ flags, pkgRoot, cfg }) { +export async function run_machine({ flags, pkgRoot, cfg, deps = { hostLifecycle: undefined } }) { heading('machine setup'); if (flags['dry-run']) { info('dry-run: would ensure packages (incl. agent-browser and ruvnet-brain), deploy skill (blocks + MCP land in the final pass)'); return true; } @@ -479,7 +483,7 @@ export async function run_machine({ flags, pkgRoot, cfg }) { // key on `codex` being on PATH / dual-mode enablement). Running them here // warned + drifted on genuinely bare machines. deployTokenAuditSkill(pkgRoot); - await installEnabledAbsentHosts(cfg, flags); + await installEnabledAbsentHosts(cfg, flags, deps.hostLifecycle); if (!(await applyMachineHostLifecycles(cfg, pkgRoot))) return false; if (cfg.codexContext && cfg.integrations?.hosts?.codex) { try { await manageCodexContext(cfg, { persist: saveKitConfig }); } @@ -602,25 +606,52 @@ export async function startProjectDaemon(root, { } else warn('daemon failed to start — try: ruflo daemon start'); } -/** Step 7: write-verification (store → actual on-disk row, then clean up). - * The CLI mirrors the write into agentdb-memory.db under the memory root - * (CLAUDE_FLOW_MEMORY_PATH, else a config persistPath, else /.swarm). - * The probe pins that root beside the pinned memory.db, so both copies land - * in the stores cleanup checks even when the project's root is redirected; - * this is the user's real corpus, so nothing of the probe may be left behind. */ +/** Step 7: verify a real primary write. Ruflo also writes a native mirror + * under CLAUDE_FLOW_MEMORY_PATH; keep that disposable mirror in a private + * directory so setup cannot create a spare project store. */ export async function verifyProjectMemoryWrite(root, env, { runner = runCmd } = {}) { const probeKey = `_setup/verify-${process.pid}-${Date.now()}`; - const probeEnv = { ...env, CLAUDE_FLOW_MEMORY_PATH: path.dirname(env?.CLAUDE_FLOW_DB_PATH ?? paths.projectMemoryDb(root)) }; - const stored = (await runner('ruflo', ['memory', 'store', '-k', probeKey, '--value', 'setup-verify', '-n', '_setup'], { cwd: root, env: probeEnv })).code === 0; - const landed = stored ? findMemoryEntry(root, '_setup', probeKey) : null; - if (!landed) { + let mirrorDir; + let stored = false; + let landed = null; + let cleanup; + let runnerError = false; + let mirrorCleanupError = false; + try { + mirrorDir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-setup-memory-probe-')); + const probeEnv = { + ...env, + CLAUDE_FLOW_DB_PATH: env?.CLAUDE_FLOW_DB_PATH ?? paths.projectMemoryDb(root), + CLAUDE_FLOW_MEMORY_PATH: mirrorDir, + RUFLO_DAEMON_AUTOSTART: '0', + }; + stored = (await runner('ruflo', ['memory', 'store', '-k', probeKey, '--value', 'setup-verify', '-n', '_setup'], { cwd: root, env: probeEnv })).code === 0; + } catch { + runnerError = true; + } finally { + // A failed command may still have written its primary row. Keep the + // existing refusal behavior for unreadable or busy project stores. + if (mirrorDir) { + try { + landed = findMemoryEntry(root, '_setup', probeKey); + cleanup = removeMemoryProbe(root, '_setup', probeKey); + } catch { + runnerError = true; + } finally { + try { fs.rmSync(mirrorDir, { recursive: true, maxRetries: 3 }); } + catch { mirrorCleanupError = true; } + } + } + } + if (cleanup?.failed.length) { + warn(`memory probe cleanup failed in ${cleanup.failed.map((f) => `${path.basename(f.file)} (${f.kind})`).join(', ')} — remove ${probeKey} from _setup manually`); + } + if (mirrorCleanupError) warn(`memory probe temporary mirror cleanup failed at ${mirrorDir} — inspect it manually`); + if (!stored || !landed || runnerError || cleanup?.failed.length || mirrorCleanupError) { fail('memory write verification FAILED — run: ak status / ruflo doctor -c memory'); return; } - const cleanup = removeMemoryProbe(root, '_setup', probeKey); - if (cleanup.failed.length) { - warn(`memory write verified, but probe cleanup failed in ${cleanup.failed.map((f) => `${path.basename(f.file)} (${f.kind})`).join(', ')} — remove ${probeKey} from _setup manually`); - } else ok(`memory write VERIFIED (store → ${path.basename(landed.file)} row confirmed)`); + ok(`memory write VERIFIED (store → ${path.basename(landed.file)} row confirmed)`); } function reportProjectGuidance(result) { @@ -926,6 +957,7 @@ async function applySetupCodexRepairs(flags, repairPlan, cwd, repairTopology) { export async function run({ flags, pkgRoot, confirm = ask, dejaVuLifecycle = DEFAULT_DEJA_VU_LIFECYCLE, + deps = { hostLifecycle: undefined }, ...runtimeOverrides }) { const runtime = { ...DEFAULT_SETUP_RUNTIME, ...runtimeOverrides }; @@ -966,7 +998,7 @@ export async function run({ flags, cfg, hostFlags, companionPreflight, dejaVuFlagsResult, }); - if (!(await runtime.machineSetup({ flags, pkgRoot, cfg }))) return 1; + if (!(await runtime.machineSetup({ flags, pkgRoot, cfg, deps }))) return 1; if (!flags['dry-run'] && cfg.aqe !== false) { // Persist the selected choice even on failure, so retry/sync has an exact plan. saveKitConfig(cfg); diff --git a/src/commands/status.mjs b/src/commands/status.mjs index 0020ec5f..f7147d31 100644 --- a/src/commands/status.mjs +++ b/src/commands/status.mjs @@ -89,7 +89,7 @@ Slow proofs run only when named with --only, up to six minutes each: aqe storage, embedding configuration and provenance, and the browser payload memory-routes the memory round trip, plus whether CLI and MCP see each - other's writes (remembered as the memory check) + other's writes (CLI result remembered as memory; routing separately) A named check runs even when it would not apply; its result is remembered only when it applies. learning and harvest are never remembered. @@ -281,9 +281,9 @@ function strayArgumentError(positionals) { export async function run({ flags, positionals = [], pkgRoot, deps = {} }) { const request = refreshRequestFromFlags(flags); if ('error' in request) return usageError(flags, request.error); - if (!request.strength) return report(flags, { rows: await collect({ pkgRoot, refresh: false }) }); const stray = strayArgumentError(positionals); if (stray) return usageError(flags, stray); + if (!request.strength) return report(flags, { rows: await collect({ pkgRoot, refresh: false }) }); return runRefreshed({ flags, pkgRoot, request, deps }); } diff --git a/src/commands/status/sections/codex-mcp.mjs b/src/commands/status/sections/codex-mcp.mjs index 620db669..2e46c1c1 100644 --- a/src/commands/status/sections/codex-mcp.mjs +++ b/src/commands/status/sections/codex-mcp.mjs @@ -87,7 +87,7 @@ function topologyRows(cwd, cfg) { if (!topology.agenticQeRegistrations.length) { // Agentic-QE owns its Codex registration (ADR-0033); sync never writes it. rows.push(row('codex-mcp', 'warn', 'agentic-qe MCP is not concretely registered in Codex', - 'run: aqe platform setup codex --overwrite --with-ruflo', { repair: 'manual' })); + 'run: aqe init --auto --with-codex --codex-guidance compact in this project, then recheck; AQE 3.14.4 may still omit Codex assets (agentic-qe#755)', { repair: 'manual' })); } else { rows.push(row('codex-mcp', 'ok', 'agentic-qe MCP concretely registered in Codex')); } diff --git a/src/commands/status/sections/daemons.mjs b/src/commands/status/sections/daemons.mjs index 152eb9c5..e156cd14 100644 --- a/src/commands/status/sections/daemons.mjs +++ b/src/commands/status/sections/daemons.mjs @@ -87,6 +87,15 @@ const flatKeys = (entries, pick) => entries.map((e) => `"${e.key}": ${JSON.strin export function heldRow(held) { const file = DAEMON_CONFIG_RELATIVE.split(path.sep).join('/'); const want = flatKeys(held.entries, (e) => e.want); + if (held.reason === 'yaml-shadow') return row('daemons', 'warn', + `${file} is not ak-managed: creating it would hide existing .claude-flow/config.yaml or config.yml daemon values`, + `review the YAML daemon values and set ${want} in the active config yourself, ${RESTART}`, { repair: 'manual' }); + if (held.reason === 'higher-priority-json') return row('daemons', 'warn', + `${file} is not ak-managed: Ruflo reads claude-flow.config.json first`, + `review claude-flow.config.json and set ${want} there yourself, ${RESTART}`, { repair: 'manual' }); + if (held.reason === 'explicit-config') return row('daemons', 'warn', + `${file} is not ak-managed: Ruflo currently reads CLAUDE_FLOW_CONFIG before YAML`, + `review the CLAUDE_FLOW_CONFIG file and set ${want} there yourself, ${RESTART}`, { repair: 'manual' }); return held.invalid ? row('daemons', 'warn', `${file} is not ak-managed: it is unreadable or not a JSON object, so ak leaves it untouched`, `fix ${file} so it is a JSON object holding ${want} (flat keys), ${RESTART}`, { repair: 'manual' }) @@ -96,16 +105,16 @@ export function heldRow(held) { /** The Ruflo repository around `cwd`, kit.json, and the keys its config.json * keeps from ak (read once for the deferral and drift rows). */ -function rufloContext(cwd, { loadConfig, rufloVersion, platform }) { +function rufloContext(cwd, { loadConfig, rufloVersion, platform, env }) { const root = rufloDaemonProjectRoot(cwd); if (!root) return { root: null, cfg: null, held: null }; const cfg = loadConfig(); - return { root, cfg, held: daemonConfigHeld(root, { cfg, rufloVersion, platform }) }; + return { root, cfg, held: daemonConfigHeld(root, { cfg, rufloVersion, platform, env }) }; } -function driftRows({ root, cfg, held }, { rufloVersion, platform }) { +function driftRows({ root, cfg, held }, { rufloVersion, platform, env }) { if (!root) return []; - const parts = daemonDrift(root, { cfg, rufloVersion, platform }); + const parts = daemonDrift(root, { cfg, rufloVersion, platform, env }); return [held && heldRow(held), parts && row('daemons', 'warn', `ak-managed daemon settings differ from what Ruflo ${rufloVersion ?? '(version unknown)'} ` + `needs: ${parts.join('; ')}`, "sync applies ak's Ruflo daemon settings")].filter(Boolean); } @@ -134,10 +143,10 @@ export default { rows.push(row('daemons', 'ok', daemons.length ? `${daemons.length} running (one per active project is expected)` : 'none running')); } - const ruflo = rufloContext(cwd, { loadConfig, rufloVersion, platform }); + const ruflo = rufloContext(cwd, { loadConfig, rufloVersion, platform, env }); const deferral = deferralRow(root, { now, platform, ruflo }); if (deferral) rows.push(deferral); - rows.push(...driftRows(ruflo, { rufloVersion, platform })); + rows.push(...driftRows(ruflo, { rufloVersion, platform, env })); } catch (e) { rows.push(row('daemons', 'warn', `daemon check unavailable: ${e.message}`)); } diff --git a/src/commands/status/sections/project-memory.mjs b/src/commands/status/sections/project-memory.mjs index a76c2f57..e9055539 100644 --- a/src/commands/status/sections/project-memory.mjs +++ b/src/commands/status/sections/project-memory.mjs @@ -26,7 +26,7 @@ import fs from 'node:fs'; import path from 'node:path'; import { projectDaemonAlive } from '../../../lib/daemons.mjs'; -import { formatLiveCheckAge as ago } from '../../../lib/live-check-evidence.mjs'; +import { formatLiveCheckAge as ago, liveCheckInputsKey, readLiveCheck } from '../../../lib/live-check-evidence.mjs'; import { memoryMaintenanceStatus } from '../../../lib/memory-maintenance.mjs'; import { findStrayMemoryStores, projectMemoryStatus } from '../../../lib/project-memory.mjs'; import { findProbeRows } from '../../../lib/memory-probe-cleanup.mjs'; @@ -206,7 +206,19 @@ export default { ? storeMessage(store) : `${path.basename(store.file)} store is unreadable (${store.file}); existing-corpus access unverified`)); } - if (memory.secondary) rows.push(row('memory', 'warn', twoStoreMessage(rufloVersion, platform))); + if (memory.secondary) { + const routing = readLiveCheck('memory-routes', { + inputsKey: liveCheckInputsKey('memory-routes', { routingVersion: rufloVersion, platform }), now, + }); + const observed = typeof rufloVersion === 'string' && rufloVersion.length > 0 && + routing?.status === 'passed' && !routing.invalidated; + const message = observed + ? twoStoreMessage(rufloVersion, platform).replace('MCP routing needs separate verification.', 'Isolated CLI/MCP routing was observed.') + : twoStoreMessage(rufloVersion, platform); + rows.push(row('memory', observed ? 'info' : 'warn', observed + ? `${message}; isolated CLI/MCP routing observed (${ago(routing.ageMs)}) for this installed CLI version and platform; existing-corpus access unverified` + : message)); + } const orphaned = orphanedStoreRow(root, memory); if (orphaned) rows.push(orphaned); rows.push(...maintenanceRows(root, memory, now)); diff --git a/src/commands/status/sections/ruflo-components.mjs b/src/commands/status/sections/ruflo-components.mjs index 9bbdebad..c0d2c50a 100644 --- a/src/commands/status/sections/ruflo-components.mjs +++ b/src/commands/status/sections/ruflo-components.mjs @@ -1,7 +1,7 @@ // ADR-0058 §7: every row carries state + meaning + action, never a bare label. // Evidence comes from the cache; `ak status --refresh` re-probes first. The // projection itself (Claude env conflicts, policy state, missing hosts) is -// shared with Task 10's dashboard through rufloComponentsPayload — this +// shared with the dashboard through rufloComponentsPayload — this // section never re-implements it (controller ruling 3). import { installedVersion } from '../../../lib/versions.mjs'; import { row } from '../row.mjs'; @@ -29,7 +29,10 @@ export function rufloComponentRows(snapshot) { const rows = [row('ruflo-components', snapshot.summary.active === snapshot.summary.total ? 'ok' : 'info', `ruflo components: ${snapshot.summary.active} of ${snapshot.summary.total} active (ruflo ${snapshot.rufloVersion ?? 'not installed'})`)]; for (const c of snapshot.components) { - const text = `${c.label} — ${c.state.label}: ${c.state.meaning}${c.state.action ? ` ${c.state.action}` : ''}`; + // The applied-unverified action is carried by its manual fix below; repeating it + // in the message renders the same host restart instruction twice. + const action = c.state.id === 'applied-unverified' ? '' : c.state.action; + const text = `${c.label} — ${c.state.label}: ${c.state.meaning}${action ? ` ${action}` : ''}`; rows.push({ ...(FIXABLE.has(c.state.id) ? row('ruflo-components', LEVEL(c.state.id), text, `sync applies ${c.label} (${c.state.action || 'reconcile'})`) diff --git a/src/commands/status/sections/user-memory.mjs b/src/commands/status/sections/user-memory.mjs index 2759b8e3..d2306d6c 100644 --- a/src/commands/status/sections/user-memory.mjs +++ b/src/commands/status/sections/user-memory.mjs @@ -5,6 +5,8 @@ // that store when it exists, and the stray stores such sessions left before: // `~/.swarm` and `~/.codex/.chatgpt-projects/*/.swarm`. Everything here is // information only: ak never moves, merges or deletes a store. +import fs from 'node:fs'; +import path from 'node:path'; import * as paths from '../../../lib/paths.mjs'; import { findUserStrayStores, memoryDirStatus } from '../../../lib/project-memory.mjs'; import { homeRelative } from '../../../lib/ruflo-memory.mjs'; @@ -30,6 +32,23 @@ function strayRow(found, home, userDir) { + (found.complete ? '' : '; only the first 500 Codex project folders were checked')); } +function homeAqeRow(home) { + const dir = path.join(home, '.agentic-qe'); + let folder; + try { folder = fs.lstatSync(dir); } catch (e) { if (e.code === 'ENOENT') return null; throw e; } + if (!folder.isDirectory()) return row('aqe', 'info', `AQE home path ${dir} is not a directory; contents unverified`); + const db = path.join(dir, 'memory.db'); + let file; + try { file = fs.lstatSync(db); } catch (e) { if (e.code !== 'ENOENT') throw e; } + if (!file?.isFile()) return row('aqe', 'info', `AQE home directory ${dir}: memory.db absent; contents and runtime health unverified`); + let walBytes = 0; + try { + const wal = fs.lstatSync(`${db}-wal`); + if (wal.isFile()) walBytes = wal.size; + } catch (e) { if (e.code !== 'ENOENT') throw e; } + return row('aqe', 'info', `AQE home directory ${dir}: memory.db present (${formatBytes(file.size + walBytes)} with WAL); contents and runtime health unverified`); +} + export default { id: 'user-memory', /** @param {{ home?: string, env?: NodeJS.ProcessEnv, cfg?: any }} [ctx] */ @@ -47,6 +66,8 @@ export default { const found = findUserStrayStores({ home, codexHome: env.CODEX_HOME || undefined }); const stray = strayRow(found, home, dir); if (stray) rows.push(stray); + const aqe = homeAqeRow(home); + if (aqe) rows.push(aqe); } catch (e) { rows.push(row('memory', 'warn', `user-level memory check unavailable: ${e.message}`)); } diff --git a/src/commands/sync.mjs b/src/commands/sync.mjs index 3e6ef56c..48ca68ba 100644 --- a/src/commands/sync.mjs +++ b/src/commands/sync.mjs @@ -22,7 +22,7 @@ import { hostsWithLifecycle, lifecycleAdapterFor, lifecycleExecutionEnabled, det import { companionLifecycleFor } from '../lib/adapters/companion-lifecycle-registry.mjs'; import { renderApplyReport } from '../lib/adapters/lifecycle-render.mjs'; import { listDaemons, staleDaemons, reap } from '../lib/daemons.mjs'; -import { applyRufloDaemon } from '../lib/ruflo-daemon-config.mjs'; +import { applyRufloDaemon, rufloDaemonProjectRoot } from '../lib/ruflo-daemon-config.mjs'; import { cleanupProbeRows } from '../lib/memory-probe-cleanup.mjs'; import { rufloMemoryLocation } from '../lib/ruflo-memory.mjs'; import { installedRoutingVersion } from '../lib/ruflo-memory-contract.mjs'; @@ -484,16 +484,20 @@ export const SYNC_STEPS = [ // that are actually still alive, not the ones just killed. if (reaped.some((r) => r.killed)) await list({ cwd: ctx.cwd, refresh: true, record: true, source: 'sync' }); // Read the version now: the versions step may have just upgraded Ruflo. - const applied = await applyRufloDaemon(ctx.cwd, { - cfg: ctx.cfg, rufloVersion: installedRoutingVersion() ?? installedVersion('ruflo'), + const applied = await (ctx.daemonApply ?? applyRufloDaemon)(ctx.cwd, { + cfg: ctx.cfg, rufloVersion: (ctx.daemonVersion ?? (() => installedRoutingVersion() ?? installedVersion('ruflo')))(), }); if (!applied) return; - saveKitConfig(ctx.cfg); + (ctx.saveConfig ?? saveKitConfig)(ctx.cfg); const { config, autostart } = applied.result; if (applied.result.changed) ok(`ruflo daemon settings: config ${config}, start-on-use ${autostart}`); const { held } = applied.result; if (held) { - warn(`.claude-flow/config.json is not ak-managed here (${held.invalid ? 'unreadable or not a JSON object' : 'a key holds your own value'}); ` + const reason = held.reason === 'yaml-shadow' ? 'creating JSON would hide existing YAML daemon values' + : held.reason === 'higher-priority-json' ? 'Ruflo reads root claude-flow.config.json first' + : held.reason === 'explicit-config' ? 'Ruflo reads CLAUDE_FLOW_CONFIG before YAML' + : held.invalid ? 'unreadable or not a JSON object' : 'a key holds your own value'; + warn(`.claude-flow/config.json is not ak-managed here (${reason}); ` + `left as is, and the daemon is not restarted for ${held.entries.map((e) => e.key).join(', ')}`); } if (applied.restarted) ok('ruflo daemon restarted so it reads its settings'); @@ -1123,6 +1127,18 @@ async function converge({ // (not-applied/drifted/blocked) don't need an upgrade and stay in the plan. .filter((r) => !(flags['no-upgrade'] && r.subsystem === 'ruflo-components' && r.state === 'needs-ruflo')); + // The daemon step also runs for a versions item, even when the collector + // reports no current daemon drift. Its settings are decided after upgrades + // from the installed Ruflo version, so the preview can promise only this + // recheck, not exact keys or a restart. + if (!skip.has('versions') && candidates.some((r) => r.subsystem === 'versions') + && !candidates.some((r) => r.subsystem === 'daemons') + && rufloDaemonProjectRoot(cwd)) { + candidates.push(row('daemons', 'info', + 'package changes may change the daemon settings needed by the installed Ruflo version', + 'recheck daemon settings against the installed Ruflo version; write receipted keys or restart a running daemon only if needed')); + } + const cfg = loadKitConfig(); if (cfg.aqe !== false && cfg.aqeEmbedding && cfg.aqeEmbedding.mode !== 'unmanaged') { candidates.push(row('aqe-embedding', 'info', 'selected semantic backend requires live verification', diff --git a/src/commands/system.mjs b/src/commands/system.mjs index c417951d..7b583540 100644 --- a/src/commands/system.mjs +++ b/src/commands/system.mjs @@ -1,3 +1,4 @@ +import { censusDisclosure } from '../lib/census-presentation.mjs'; // ak system — the machine footprint in the terminal (ADR-0025). // // The CLI twin of the dashboard's System area, driving the SAME composed @@ -231,24 +232,25 @@ function renderRuntime(runtime) { } const rows = census?.value ?? []; if (!rows.length) { - console.log(` ${dim('no agent processes are running')}`); + console.log(` ${dim('no coding-agent or desktop-application processes are running')}`); return; } const sink = reasonSink(); console.log(''); - table(['HOST', 'PID', 'CPU', 'RSS', 'UPTIME', 'PROJECT'], rows.map((row) => [ - row.host, + table(['CODING-AGENT HOST / DESKTOP APPLICATION', 'PID', 'CPU', 'RSS', 'UPTIME', 'WORKING CONTEXT'], rows.map((row) => [ + row.application ?? row.host ?? 'Unknown process', String(row.pid), sink.cell(row.cpuPercent, fmtPercent), sink.cell(row.rssBytes, fmtBytes), sink.cell(row.uptimeMs, fmtDuration), - row.project?.status === UNKNOWN ? 'unattributed' : (row.project?.value?.label ?? 'unattributed'), + row.source?.status === UNKNOWN ? 'unattributed' + : (row.source?.value?.label ?? row.project?.value?.label ?? 'unattributed'), ])); sink.report(); - // `project` degrades per process (a cwd the platform will not disclose); its + // `source` degrades per process (a cwd the platform will not disclose); its // reason lives on the row, not in the numeric sink above. - for (const reason of new Set(rows.filter((row) => row.project?.status === UNKNOWN) - .map((row) => row.project.reason))) { + for (const reason of new Set(rows.filter((row) => (row.source ?? row.project)?.status === UNKNOWN) + .map((row) => (row.source ?? row.project).reason))) { console.log(` ${dim(`unattributed: ${reason}`)}`); } } @@ -348,6 +350,7 @@ function renderProjects(projects, now) { info(dim('not measured yet — run: ak system --refresh=machine')); return; } + info(dim(censusDisclosure(projects))); field('discovered', `${meas(projects.count)}${projects.truncated ? dim(' · list truncated') : ''}`); if (!projects.locMeasured) field('lines of code', dim('not measured in this scan')); diff --git a/src/commands/telemetry.mjs b/src/commands/telemetry.mjs index 77d3c503..142eaa66 100644 --- a/src/commands/telemetry.mjs +++ b/src/commands/telemetry.mjs @@ -98,7 +98,9 @@ export async function run({ flags, positionals, pkgRoot, deps = {} }) { return 0; } catch { // Source failures and hostile input must not echo filenames or parser details. - console.error('Telemetry failed: check command options, schema/digest, compatible snapshots, private identity, and readable input/new output files.'); + const error = 'Telemetry failed: check command options, schema/digest, compatible snapshots, private identity, and readable input/new output files.'; + console.error(error); + console.log(JSON.stringify({ error, exitCode: 2 })); return 2; } } diff --git a/src/commands/uninstall.mjs b/src/commands/uninstall.mjs index c27a1c2a..1a9fe198 100644 --- a/src/commands/uninstall.mjs +++ b/src/commands/uninstall.mjs @@ -101,9 +101,9 @@ function hasDejaVuOwnership(cfg) { function protectedDejaVuRoots(homeDir, env) { const absolute = (value) => typeof value === 'string' && path.isAbsolute(value); - const configBases = [path.join(homeDir, '.config'), env.XDG_CONFIG_HOME, env.APPDATA] + const configBases = [path.join(homeDir, '.config'), paths.xdgBase('XDG_CONFIG_HOME', null, { env }), env.APPDATA] .filter(absolute); - const dataBases = [path.join(homeDir, '.local', 'share'), env.XDG_DATA_HOME] + const dataBases = [path.join(homeDir, '.local', 'share'), paths.xdgBase('XDG_DATA_HOME', null, { env })] .filter(absolute); return { sourceRoots: [ diff --git a/src/commands/usage.mjs b/src/commands/usage.mjs index 2d786207..e2c578f7 100644 --- a/src/commands/usage.mjs +++ b/src/commands/usage.mjs @@ -2,7 +2,7 @@ // offline text scorecard (`score`) rendered from the SAME local-transcript // aggregate `ak dashboard`'s Usage tab reads — no cost/token/percentile // arithmetic is redone here; see the score section below for the boundary. -import { heading, info, ok, warn, dim } from '../lib/output.mjs'; +import { heading, info, ok, warn, dim, reportFailure } from '../lib/output.mjs'; import { stripUnsafeChars } from '../lib/text-safety.mjs'; import { readIndex } from '../lib/usage-index.mjs'; import { @@ -321,7 +321,8 @@ function scoreProjection(agg, windowDays) { async function runScore({ flags, deps }) { const windowDays = parseScoreWindow(flags.window ?? '14'); if (windowDays == null) { - warn(`ak usage score: --window must be 7, 14, or 30 (got ${JSON.stringify(flags.window)})`); + const message = `ak usage score: --window must be 7, 14, or 30 (got ${JSON.stringify(flags.window)})`; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => warn(message) }); return 2; } const readAgg = deps.readIndex ?? readIndex; @@ -953,7 +954,8 @@ function printDeepPass(deep) { async function runPrompts({ flags, deps }) { const win = parsePromptWindow(flags.window); if (win == null) { - warn(`ak usage prompts: --window must be 7, 14, 30, or all (got ${JSON.stringify(flags.window)})`); + const message = `ak usage prompts: --window must be 7, 14, 30, or all (got ${JSON.stringify(flags.window)})`; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => warn(message) }); return 2; } const readAgg = deps.readIndex ?? readIndex; @@ -1087,6 +1089,12 @@ export async function run({ flags, positionals, deps = {} }) { const provider = positionals[1]; const cacheFile = deps.cacheFile ?? openRouterActivityFile(); + if (['score', 'prompts'].includes(action) && positionals.length > 1) { + const message = `unexpected argument '${positionals[1]}'`; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => warn(message) }); + return 2; + } + if (action === 'status' && provider === undefined) { return runOpenRouterStatus({ flags, cacheFile, read: deps.read ?? readOpenRouterActivity }); } @@ -1096,6 +1104,7 @@ export async function run({ flags, positionals, deps = {} }) { return runOpenRouterRefresh({ flags, cacheFile, refresh: deps.refresh ?? refreshOpenRouterActivity }); } - warn('usage: ak usage status | ak usage refresh openrouter | ak usage score | ak usage prompts'); + const message = 'usage: ak usage status | ak usage refresh openrouter | ak usage score | ak usage prompts'; + reportFailure({ json: flags.json, payload: { error: message, exitCode: 2 }, human: () => warn(message) }); return 2; } diff --git a/src/commands/x/aqe-embedding.mjs b/src/commands/x/aqe-embedding.mjs index 25601712..fad9d06e 100644 --- a/src/commands/x/aqe-embedding.mjs +++ b/src/commands/x/aqe-embedding.mjs @@ -3,6 +3,7 @@ import { embeddingIntentFromFlags, embeddingSetupDisclosure } from '../../lib/aq import { prepareAqeEmbedding, AQE_EMBEDDING_COACHING } from '../../lib/aqe-embedding-lifecycle.mjs'; import { inspectAqeEmbeddingProjections, reconcileAqeEmbeddingProjections } from '../../lib/aqe-embedding-projection.mjs'; import { reconcileOpencodeAqeEmbedding } from '../../lib/opencode-core.mjs'; +import { reportFailure } from '../../lib/output.mjs'; export const options = { 'aqe-embedding-mode': { type: 'string' }, 'aqe-embedding-endpoint': { type: 'string' }, @@ -43,11 +44,19 @@ export async function run({ flags = {}, positionals = [], reconcileOpenCode = reconcileOpencodeAqeEmbedding, }) { const action = positionals[0] ?? 'status'; - if (!['status', 'configure', 'prepare', 'verify'].includes(action) || positionals.length > 1) return 2; + if (!['status', 'configure', 'prepare', 'verify'].includes(action) || positionals.length > 1) { + const error = 'usage: ak x aqe-embedding [status|configure|prepare|verify]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => console.error(error) }); + return 2; + } const cfg = load(); if (action === 'configure') { try { cfg.aqeEmbedding = embeddingIntentFromFlags(cfg, flags); } - catch { console.error('Invalid embedding selection; run ak x aqe-embedding --help.'); return 2; } + catch { + const error = 'Invalid embedding selection; run ak x aqe-embedding --help.'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => console.error(error) }); + return 2; + } } const emit = value => console.log(flags.json ? JSON.stringify(value) : value.detail); const openCodeReady = (dryRun) => { diff --git a/src/commands/x/aqe-store.mjs b/src/commands/x/aqe-store.mjs index fb26b5b7..9d4e5158 100644 --- a/src/commands/x/aqe-store.mjs +++ b/src/commands/x/aqe-store.mjs @@ -3,7 +3,7 @@ // while any AQE writer is open. The work is in src/lib/aqe-store-merge.mjs. import { mergeAqeStores, restoreSteps } from '../../lib/aqe-store-merge.mjs'; import { repoRoot } from '../../lib/paths.mjs'; -import { ok, warn, fail, info } from '../../lib/output.mjs'; +import { ok, warn, fail, info, reportFailure } from '../../lib/output.mjs'; export const options = { 'dry-run': { type: 'boolean', default: false }, @@ -139,7 +139,8 @@ function printInterrupted(result) { export async function run({ flags = {}, positionals = [], cwd = process.cwd(), merge = mergeAqeStores }) { const action = positionals[0] ?? 'status'; if (!['status', 'merge'].includes(action) || positionals.length > 1) { - fail('usage: ak x aqe-store [status|merge] [--yes] [--dry-run] [--json]'); + const error = 'usage: ak x aqe-store [status|merge] [--yes] [--dry-run] [--json]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); return 2; } const root = repoRoot(cwd); diff --git a/src/commands/x/codex-context.mjs b/src/commands/x/codex-context.mjs index 3502beca..16866127 100644 --- a/src/commands/x/codex-context.mjs +++ b/src/commands/x/codex-context.mjs @@ -1,6 +1,6 @@ import { loadKitConfig, saveKitConfig } from '../../lib/config.mjs'; import { inspectCodexContext, manageCodexContext, releaseCodexContext } from '../../lib/codex-context.mjs'; -import { info, ok, warn } from '../../lib/output.mjs'; +import { info, ok, warn, reportFailure } from '../../lib/output.mjs'; export const options = { 'dry-run': { type: 'boolean', default: false }, json: { type: 'boolean', default: false } }; export const help = `ak x codex-context — manage Codex's native per-model context capacities @@ -21,7 +21,11 @@ Examples: export async function run({ flags, positionals, contextOptions = {} }) { const choice = positionals[0] ?? 'status'; - if (!['status', 'max', 'off'].includes(choice) || positionals.length > 1) { warn(help); return 2; } + if (!['status', 'max', 'off'].includes(choice) || positionals.length > 1) { + const error = 'usage: ak x codex-context [status|max|off] [--dry-run] [--json]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(help) }); + return 2; + } const cfg = loadKitConfig(); const status = inspectCodexContext(cfg, contextOptions); if (choice === 'status' || flags['dry-run']) { diff --git a/src/commands/x/daemon-gc.mjs b/src/commands/x/daemon-gc.mjs index d6d6947d..8a1e691a 100644 --- a/src/commands/x/daemon-gc.mjs +++ b/src/commands/x/daemon-gc.mjs @@ -4,7 +4,7 @@ import { listDaemons, staleDaemons, reap, listMcpTransports, orphanedMcpTransports, reapMcpTransports, } from '../../lib/daemons.mjs'; -import { ok, warn, dim } from '../../lib/output.mjs'; +import { ok, warn, dim, reportFailure } from '../../lib/output.mjs'; export const options = { kill: { type: 'boolean', default: false }, @@ -32,10 +32,19 @@ Examples: ak x daemon-gc --kill reap stale background daemons ak x daemon-gc --mcp --kill also reap same-user PPID-1 MCP orphans`; -export async function run({ flags }) { - const daemons = await listDaemons(); +export async function run({ flags, positionals = [], deps = { daemonLifecycle: undefined, mcpLifecycle: undefined } }) { + if (positionals.length) { + const error = `unexpected argument '${positionals[0]}'`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); + return 2; + } + const { list, reap: reapFn } = { list: listDaemons, reap, ...deps.daemonLifecycle }; + const { list: listMcp, reap: reapMcp } = { + list: listMcpTransports, reap: reapMcpTransports, ...deps.mcpLifecycle, + }; + const daemons = await list(); const stale = staleDaemons(daemons); - const mcpTransports = await listMcpTransports(); + const mcpTransports = await listMcp(); const mcpOrphans = orphanedMcpTransports(mcpTransports); if (flags.json) { console.log(JSON.stringify({ @@ -47,14 +56,14 @@ export async function run({ flags }) { return 0; } if (stale.length && flags.kill) { - const reaped = reap(stale); + const reaped = reapFn(stale); for (const r of reaped) { if (r.killed) ok(`stopped stale daemon pid=${r.pid} ${dim(r.workspace ?? '')}`); else warn(`could not stop pid=${r.pid} (already exited?)`); } // Refresh daemon-sweep evidence so a later read sees the daemons that are // actually still alive, not the pre-reap list. - if (reaped.some((r) => r.killed)) await listDaemons({ refresh: true, record: true, source: 'daemon-gc' }); + if (reaped.some((r) => r.killed)) await list({ refresh: true, record: true, source: 'daemon-gc' }); } else if (stale.length) { for (const d of stale) { warn(`stale daemon pid=${d.pid} ${dim(d.workspace ?? '(unknown workspace)')} ${dim(d.workspaceExists ? `age ${d.ageSecs}s > TTL` : 'workspace gone')}`); @@ -63,7 +72,7 @@ export async function run({ flags }) { } if (flags.mcp && flags.kill) { - for (const result of reapMcpTransports(mcpOrphans)) { + for (const result of reapMcp(mcpOrphans)) { if (result.killed) ok(`stopped orphaned Ruflo MCP pid=${result.pid}`); else warn(`could not stop MCP pid=${result.pid} (identity or orphan proof changed)`); } diff --git a/src/commands/x/harvest.mjs b/src/commands/x/harvest.mjs index 86f2dcbc..7b690dc6 100644 --- a/src/commands/x/harvest.mjs +++ b/src/commands/x/harvest.mjs @@ -7,7 +7,7 @@ // memory root. It NEVER starts a daemon and NEVER backgrounds anything. import { loadKitConfig } from '../../lib/config.mjs'; import { planHarvest, runHarvest } from '../../lib/harvest.mjs'; -import { ok, fail, warn, info, dim, heading } from '../../lib/output.mjs'; +import { ok, fail, warn, info, dim, heading, reportFailure } from '../../lib/output.mjs'; export const options = { 'dry-run': { type: 'boolean', default: false }, @@ -45,7 +45,12 @@ Examples: ak x harvest record the outcome (only when opted in) ak x harvest --distill record, then distill the project store`; -export async function run({ flags }) { +export async function run({ flags, positionals = [] }) { + if (positionals.length) { + const error = `unexpected argument '${positionals[0]}'`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); + return 2; + } const cwd = process.cwd(); const cfg = loadKitConfig(); const enabled = cfg.harvest === true; diff --git a/src/commands/x/host-adapters-grants.mjs b/src/commands/x/host-adapters-grants.mjs index 8e3dbe8f..12c2ddd3 100644 --- a/src/commands/x/host-adapters-grants.mjs +++ b/src/commands/x/host-adapters-grants.mjs @@ -19,7 +19,7 @@ import { import { bootstrapHostAdapters } from '../../lib/adapters/admission.mjs'; import { loadKitConfig, saveKitConfig } from '../../lib/config.mjs'; import { applyAqeRouter } from '../../lib/providers.mjs'; -import { ok, warn, fail, info, bold } from '../../lib/output.mjs'; +import { ok, warn, fail, info, bold, reportFailure } from '../../lib/output.mjs'; import { findEntry, loadAndHash, stripControl, hookCommandsFor, } from './host-adapters.mjs'; @@ -396,12 +396,16 @@ async function reconcileRevokedAqeProvider({ } export async function revokeGrant({ - name, capability, grantsFile, cfg, env, cwd = process.cwd(), + name, capability, grantsFile, cfg, env, cwd = process.cwd(), flags = /** @type {{json?: boolean}} */ ({}), saveConfig = saveKitConfig, bootstrapAdapters = bootstrapHostAdapters, applyRouter = applyAqeRouter, }) { - if (typeof name !== 'string' || !name) { fail('usage: ak host adapters revoke-grant [capability]'); return 2; } + if (typeof name !== 'string' || !name) { + const error = 'usage: ak host adapters revoke-grant [capability]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); + return 2; + } const safeName = stripControl(name); if (typeof capability === 'string' && capability) { diff --git a/src/commands/x/host-adapters.mjs b/src/commands/x/host-adapters.mjs index 0264c2fe..9d2091db 100644 --- a/src/commands/x/host-adapters.mjs +++ b/src/commands/x/host-adapters.mjs @@ -25,7 +25,7 @@ import { HOST_REGISTRY } from '../../lib/adapters/registries.mjs'; import * as consentStore from '../../lib/adapters/consent.mjs'; import { runTieredConformance as defaultRunTieredConformance } from '../../lib/adapters/conformance.mjs'; import { loadKitConfig } from '../../lib/config.mjs'; -import { ok, warn, fail, info, dim, bold } from '../../lib/output.mjs'; +import { ok, warn, fail, info, dim, bold, reportFailure } from '../../lib/output.mjs'; import { grant as grantCap, gate as gateTier, status as statusReport, revokeGrant, } from './host-adapters-grants.mjs'; @@ -298,8 +298,12 @@ async function trust({ name, cfg, consent, reader, ask, isTTY, yes, expectHash } return 0; } -function revoke({ name, consent }) { - if (typeof name !== 'string' || !name) { fail('usage: ak host adapters revoke '); return 2; } +function revoke({ name, consent, flags = /** @type {{json?: boolean}} */ ({}) }) { + if (typeof name !== 'string' || !name) { + const error = 'usage: ak host adapters revoke '; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); + return 2; + } const existed = consent.revokeConsent(name); if (existed) { ok(`revoked consent for '${name}'`); return 0; } info(`no recorded consent for '${name}'`); @@ -437,9 +441,9 @@ async function conformance({ // standing consent or grant record, or it silently reactivates the next time // the flag is turned back on. const FAIL_SAFE_HANDLERS = { - revoke: (ctx) => revoke({ name: ctx.name, consent: ctx.consent }), + revoke: (ctx) => revoke({ name: ctx.name, consent: ctx.consent, flags: ctx.flags }), 'revoke-grant': (ctx) => revokeGrant({ - name: ctx.name, capability: ctx.positionals[2], grantsFile: ctx.grantsFile, cfg: ctx.cfg, env: ctx.env, cwd: ctx.cwd, + name: ctx.name, capability: ctx.positionals[2], grantsFile: ctx.grantsFile, cfg: ctx.cfg, env: ctx.env, cwd: ctx.cwd, flags: ctx.flags, ...(ctx.saveConfig ? { saveConfig: ctx.saveConfig } : {}), ...(ctx.bootstrapAdapters ? { bootstrapAdapters: ctx.bootstrapAdapters } : {}), ...(ctx.applyRouter ? { applyRouter: ctx.applyRouter } : {}), @@ -505,7 +509,8 @@ export async function run({ if (failSafe) return failSafe(ctx); if (!flagEnabled(env)) { - fail(`experimental host-adapter surface is disabled — set ${FLAG_ENV_VAR}=1`); + const error = `experimental host-adapter surface is disabled — set ${FLAG_ENV_VAR}=1`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); return 2; } @@ -514,6 +519,7 @@ export async function run({ const gated = GATED_HANDLERS[sub]; if (gated) return gated(ctx); - fail(`unknown host adapters subcommand: ${sub} (list|trust|revoke|conformance|grant|bless|gate|status|revoke-grant)`); + const error = `unknown host adapters subcommand: ${sub} (list|trust|revoke|conformance|grant|bless|gate|status|revoke-grant)`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); return 2; } diff --git a/src/commands/x/host.mjs b/src/commands/x/host.mjs index 73231e87..8f4ec772 100644 --- a/src/commands/x/host.mjs +++ b/src/commands/x/host.mjs @@ -32,7 +32,7 @@ import { hostManagement, hostEnableCommand, HOST_MANAGEMENT_LABELS, NOT_PARTICIPATING, } from '../../lib/host-management.mjs'; import { - ok, warn, fail, info, dim, bold, yellow, humanOutputToStderr, + ok, warn, fail, info, dim, bold, yellow, humanOutputToStderr, reportFailure, } from '../../lib/output.mjs'; import { repoRoot } from '../../lib/paths.mjs'; import { writeJsonWithBackup } from '../../lib/settings.mjs'; @@ -181,13 +181,19 @@ export const parseFallback = (str) => str.split(';').map((s) => s.trim()).filter return { provider: provider.trim().toLowerCase(), models: models.split(',').map((m) => m.trim()).filter(Boolean) }; }); -export async function run({ flags, positionals, pkgRoot }) { +export async function run({ flags, positionals, pkgRoot, deps = { hostLifecycle: undefined } }) { const sub = positionals[0] ?? 'status'; const cwd = process.cwd(); + if (['status', 'off', 'pick', 'reset-routes', 'align'].includes(sub) && positionals.length > 1) { + const error = `unexpected argument '${positionals[1]}'`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); + return 2; + } + if (sub === 'status') return status({ flags, cwd }); if (sub === 'off') return off({ cwd, pkgRoot, flags }); - if (sub === 'pick') return pick({ flags, cwd, pkgRoot }); + if (sub === 'pick') return pick({ flags, cwd, pkgRoot, deps }); if (sub === 'reset-routes') return resetRoutes({ flags, cwd }); if (sub === 'align') return (await import('./host-align.mjs')).run({ flags }); if (sub === 'adapters') { @@ -196,12 +202,17 @@ export async function run({ flags, positionals, pkgRoot }) { // read-only preview to give --dry-run, so it is refused outright // instead of silently behaving like a real run: a flag we declare is a // flag we honor, or refuse. - if (flags['dry-run']) { fail('ak host adapters has no preview; run it without --dry-run'); return 2; } + if (flags['dry-run']) { + const error = 'ak host adapters has no preview; run it without --dry-run'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); + return 2; + } return (await import('./host-adapters.mjs')).run({ flags, positionals: positionals.slice(1) }); } if (sub === 'check-connection') return (await import('./host-connection.mjs')).run({ flags, positionals: positionals.slice(1) }); - fail(`unknown host subcommand: ${sub} (status|pick|reset-routes|off|check-connection|adapters|align)`); + const error = `unknown host subcommand: ${sub} (status|pick|reset-routes|off|check-connection|adapters|align)`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => fail(error) }); return 2; } @@ -322,7 +333,7 @@ const STATUS_SECTIONS = [ async function status({ flags, cwd }) { const cfg = loadKitConfig(); - const facts = await collectIntegrationFacts({ cwd, cfg }); + const facts = await collectIntegrationFacts({ cwd, cfg, record: !flags['dry-run'] }); const hosts = facts.hosts; const providers = facts.providers; const { scope } = settingsTarget(cwd); @@ -378,27 +389,38 @@ async function resetRoutes({ flags, cwd }) { const cfg = loadKitConfig(); const policy = cfg.routing?.routes ?? {}; const diverged = divergedRoutes(policy); - if (!diverged.length) { ok('no seeded routes diverge from the current defaults'); return 0; } - - console.log(bold('seeded routes that diverge from current defaults')); - for (const d of diverged) { - const head = d.modelDiverged ? `${d.model} → ${d.defaultModel}` : d.model; - console.log(` ${d.activity.padEnd(18)} ${d.host.padEnd(7)} ${head}`); - for (const e of d.escalation) console.log(` ${dim('escalation:')} ${e.model} → ${e.defaultModel}`); - // The trade, not just the ids: choosing on a price axis while paying on a - // turns axis is the misreading this whole surface exists to prevent. - if (d.modelDiverged && d.currentNote) console.log(dim(` now: ${d.model} — ${d.currentNote}`)); - if (d.modelDiverged && d.defaultNote) console.log(dim(` default: ${d.defaultModel} — ${d.defaultNote}`)); - for (const e of d.escalation) { - const note = modelNote(e.defaultModel); - if (note) console.log(dim(` default: ${e.defaultModel} — ${note}`)); + const jsonDryRun = flags['dry-run'] && flags.json; + const previewJson = (activities) => console.log(JSON.stringify({ dryRun: true, activities }, null, 2)); + if (!diverged.length) { + if (jsonDryRun) previewJson([]); + else ok('no seeded routes diverge from the current defaults'); + return 0; + } + + if (!jsonDryRun) { + console.log(bold('seeded routes that diverge from current defaults')); + for (const d of diverged) { + const head = d.modelDiverged ? `${d.model} → ${d.defaultModel}` : d.model; + console.log(` ${d.activity.padEnd(18)} ${d.host.padEnd(7)} ${head}`); + for (const e of d.escalation) console.log(` ${dim('escalation:')} ${e.model} → ${e.defaultModel}`); + // The trade, not just the ids: choosing on a price axis while paying on a + // turns axis is the misreading this whole surface exists to prevent. + if (d.modelDiverged && d.currentNote) console.log(dim(` now: ${d.model} — ${d.currentNote}`)); + if (d.modelDiverged && d.defaultNote) console.log(dim(` default: ${d.defaultModel} — ${d.defaultNote}`)); + for (const e of d.escalation) { + const note = modelNote(e.defaultModel); + if (note) console.log(dim(` default: ${e.defaultModel} — ${note}`)); + } } } let picked; if (flags.activity !== undefined) { const want = flags.activity.split(',').map((s) => s.trim()).filter(Boolean); - for (const a of want.filter((a) => !ACTIVITIES.includes(a))) warn(`unknown activity '${a}' — ignored`); + for (const a of want.filter((a) => !ACTIVITIES.includes(a))) { + if (jsonDryRun) await humanOutputToStderr(() => warn(`unknown activity '${a}' — ignored`)); + else warn(`unknown activity '${a}' — ignored`); + } picked = want.filter((a) => diverged.some((d) => d.activity === a)); } else if (flags.yes || flags['dry-run']) { // --dry-run never prompts: with nothing else naming a subset, preview the @@ -413,11 +435,18 @@ async function resetRoutes({ flags, cwd }) { ? diverged.map((d) => d.activity) : ans.split(',').map((s) => s.trim()).filter((a) => diverged.some((d) => d.activity === a)); } - if (!picked.length) { info('no routes reset — routes left as they are'); return 0; } + if (!picked.length) { + if (jsonDryRun) previewJson([]); + else info('no routes reset — routes left as they are'); + return 0; + } if (flags['dry-run']) { - info(`would reset ${picked.length} route(s) to the current defaults: ${picked.join(', ')}`); - info('dry run — nothing changed'); + if (jsonDryRun) previewJson(picked); + else { + info(`would reset ${picked.length} route(s) to the current defaults: ${picked.join(', ')}`); + info('dry run — nothing changed'); + } return 0; } @@ -457,11 +486,30 @@ function printOffDryRunSummary(cfg, opts) { if (codexOwned) console.log(' would remove ak-managed Codex MCP wiring'); } -async function off({ cwd, pkgRoot, flags = {} }) { +async function off({ cwd, pkgRoot, flags = /** @type {{ json?: boolean, 'dry-run'?: boolean }} */ ({}) }) { const cfg = loadKitConfig(); if (flags['dry-run']) { - printOffDryRunSummary(cfg, { pkgRoot }); - info('dry run — nothing changed'); + if (flags.json) { + await humanOutputToStderr(() => { + printOffDryRunSummary(cfg, { pkgRoot }); + info('dry run — nothing changed'); + }); + console.log(JSON.stringify({ + dryRun: true, + wouldDisable: HOSTS.filter((h) => cfg.integrations.hosts[h.id]).map((h) => h.id), + primaryHost: DEFAULT_PRIMARY_HOST, + wouldClear: ['aqe provider/fallback', 'ruflo providers', 'activity routing'], + wouldStripManagedProviderEnv: true, + wouldRestoreOrRemoveManagedAqeConfig: true, + wouldReconcileOpencodeGuidance: !!pkgRoot, + wouldTeardownOpencode: !!(cfg.integrations.hosts.opencode || cfg.integrations?.ownership?.opencode), + wouldRemoveManagedCodexMcp: cfg.integrations?.ownership?.codex?.mcp === 'ak' + || cfg.integrations?.ownership?.codex?.reverseMcp === 'ak', + }, null, 2)); + } else { + printOffDryRunSummary(cfg, { pkgRoot }); + info('dry run — nothing changed'); + } return 0; } const codexMcpManaged = cfg.integrations?.ownership?.codex?.mcp === 'ak'; @@ -715,8 +763,9 @@ async function resolvePickDecision(cfg, { const known = new Set([...MANAGED_HOSTS, ...EFFECTIVE_ROUTING]); const unknown = enabled.filter((h) => !known.has(h)); if (unknown.length) { - fail(`unknown host(s): ${unknown.join(', ')} (valid: ${[...known].join(', ')}) — nothing changed`); - return { code: 2 }; + const error = `unknown host(s): ${unknown.join(', ')} (valid: ${[...known].join(', ')}) — nothing changed`; + fail(error); + return { code: 2, error }; } // The routing set needs at least one primary-capable member; OpenCode remains // routable but cannot satisfy that primary-host invariant on its own. @@ -869,25 +918,28 @@ async function retireCodexOnDisable(cfg, cwd, { codexMcpManaged, rufloCodexManag /** Install any enabled host that is entirely absent (external installs * untouched). Unlike setup's install loop, pick never prompts first — the * user already confirmed the trust manifest for this exact enable. */ -async function installPickAbsentHosts(cfg, cwd) { +export async function installPickAbsentHosts(cfg, cwd, lifecycle = {}) { + const { installState, install, collectFacts } = { + installState: hostInstallState, install: installHost, collectFacts: collectIntegrationFacts, ...lifecycle, + }; let installed = false; for (const h of HOSTS) { if (!cfg.integrations.hosts[h.id]) continue; - if ((await hostInstallState(h)).method !== 'absent') continue; + if ((await installState(h)).method !== 'absent') continue; info(`${h.id} not installed — installing ${h.pkg}…`); - const r = await installHost(h.id); + const r = await install(h.id); (r.ok ? ok : warn)(`${h.id}: ${r.detail}`); if (r.ok) { installed = true; // hostInstallState() above already recorded the pre-install 'absent' // evidence; re-probe now so a subsequent `ak status` doesn't read that // stale row back. - await hostInstallState(h, { refresh: true, record: true, source: 'host-pick' }); + await installState(h, { refresh: true, record: true, source: 'host-pick' }); } } // host-setup covers every host in one call; refresh it once after the // loop, not per host, once anything actually changed. - if (installed) await collectIntegrationFacts({ cwd, cfg, refresh: true, record: true, source: 'host-pick' }); + if (installed) await collectFacts({ cwd, cfg, refresh: true, record: true, source: 'host-pick' }); } /** opencode enable half: apply the same owner-module stack setup/sync use — @@ -1058,7 +1110,7 @@ async function resolvePickIntentAndPreview({ aqeProviderTypes, aqeChainProviderTypes, }); - if (decision.code !== undefined) return { code: decision.code }; + if (decision.code !== undefined) return decision; const { enabled, routing, primaryHost, aqeProvider, seed, } = decision; @@ -1095,7 +1147,7 @@ async function resolvePickIntentAndPreview({ }; } -export async function pick({ flags, cwd, pkgRoot, migrateRoutes = migrateRetiredRoutesInConfig }) { +export async function pick({ flags, cwd, pkgRoot, deps = { hostLifecycle: undefined }, migrateRoutes = migrateRetiredRoutesInConfig }) { const aqeProviderTypes = aqeSelectableProviderTypes(); const aqeChainProviderTypes = aqeSelectableChainProviderTypes(); const cfg = loadKitConfig(); @@ -1133,6 +1185,7 @@ export async function pick({ flags, cwd, pkgRoot, migrateRoutes = migrateRetired const outcome = jsonDryRun ? await humanOutputToStderr(resolve) : await resolve(); if (outcome.code !== undefined) { if (outcome.json) console.log(JSON.stringify(outcome.json, null, 2)); + else if (jsonDryRun) console.log(JSON.stringify({ error: outcome.error ?? 'host pick refused', exitCode: outcome.code }, null, 2)); return outcome.code; } const { @@ -1145,7 +1198,7 @@ export async function pick({ flags, cwd, pkgRoot, migrateRoutes = migrateRetired } saveKitConfig(cfg); - await installPickAbsentHosts(cfg, cwd); + await installPickAbsentHosts(cfg, cwd, deps.hostLifecycle); const { incompleteTeardown } = await applyPickOpencodeLifecycle(cfg, { pkgRoot, cwd, prevOpencode }); const { router } = await applyPickProviderStack(cfg, cwd, { diff --git a/src/commands/x/reference.mjs b/src/commands/x/reference.mjs index 8c42cd83..d37b40ff 100644 --- a/src/commands/x/reference.mjs +++ b/src/commands/x/reference.mjs @@ -1,7 +1,7 @@ // x reference — inspect (diff) or reconcile (sync) every managed host-guidance target. import { reconcileGuidance } from '../../lib/blocks.mjs'; import { loadKitConfig } from '../../lib/config.mjs'; -import { ok, warn, dim } from '../../lib/output.mjs'; +import { ok, warn, dim, reportFailure } from '../../lib/output.mjs'; export const options = { json: { type: 'boolean', default: false } }; @@ -20,6 +20,11 @@ Examples: export async function run({ flags, positionals, pkgRoot }) { const sub = positionals[0] ?? 'diff'; + if (!['diff', 'sync'].includes(sub) || positionals.length > 1) { + const error = 'usage: ak x reference [diff|sync] [--json]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); + return 2; + } const cfg = loadKitConfig(); const dryRun = sub !== 'sync'; const res = await reconcileGuidance({ diff --git a/src/commands/x/skills.mjs b/src/commands/x/skills.mjs index 08b62044..5dc7f6ac 100644 --- a/src/commands/x/skills.mjs +++ b/src/commands/x/skills.mjs @@ -3,7 +3,7 @@ import path from 'node:path'; import { collectCatalog } from '../../lib/footprint/catalog.mjs'; import { buildSkillMaintenancePlan } from '../../lib/skill-maintenance-plan.mjs'; import { loadKitConfig } from '../../lib/config.mjs'; -import { heading, info, warn, dim } from '../../lib/output.mjs'; +import { heading, info, warn, dim, reportFailure } from '../../lib/output.mjs'; export const options = { project: { type: 'string' }, @@ -27,7 +27,8 @@ Examples: export async function run({ flags, positionals }) { if ((positionals[0] ?? 'plan') !== 'plan' || positionals.length > 1) { - warn('usage: ak x skills plan [--project PATH] [--json]'); + const error = 'usage: ak x skills plan [--project PATH] [--json]'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); return 2; } const project = path.resolve(flags.project ?? process.cwd()); diff --git a/src/commands/x/statusline.mjs b/src/commands/x/statusline.mjs index 13465758..92567c16 100644 --- a/src/commands/x/statusline.mjs +++ b/src/commands/x/statusline.mjs @@ -3,7 +3,7 @@ import { PRESETS, applyCodexStatusline, inspectCodexStatusline, projectionFor, removeCodexStatusline, statuslineDrift, } from '../../lib/codex-statusline.mjs'; -import { ok, info, warn } from '../../lib/output.mjs'; +import { ok, info, warn, reportFailure } from '../../lib/output.mjs'; export const options = { 'dry-run': { type: 'boolean', default: false }, @@ -33,7 +33,7 @@ Examples: export async function run({ flags, positionals }) { const cfg = loadKitConfig(); const [target = 'status', choice] = positionals; - if (target === 'status') { + if (target === 'status' && positionals.length <= 1) { const current = inspectCodexStatusline(); const drift = statuslineDrift(cfg); const result = { ownership: cfg.statusline?.codex ?? null, current, drifted: drift.drifted }; @@ -45,8 +45,9 @@ export async function run({ flags, positionals }) { } return 0; } - if (target !== 'codex' || !choice) { - warn('usage: ak x statusline status | codex native | codex extended | codex off'); + if (target !== 'codex' || !choice || positionals.length !== 2) { + const error = 'usage: ak x statusline status | codex native | codex extended | codex off'; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); return 2; } if (choice === 'off') { @@ -63,7 +64,11 @@ export async function run({ flags, positionals }) { ok(`codex status-line management disabled${result.changed ? '; unchanged managed keys removed' : '; user-modified keys preserved'}`); return 0; } - if (!PRESETS[choice]) { warn(`unknown preset '${choice}' (expected native, extended, or off)`); return 2; } + if (!PRESETS[choice]) { + const error = `unknown preset '${choice}' (expected native, extended, or off)`; + reportFailure({ json: flags.json, payload: { error, exitCode: 2 }, human: () => warn(error) }); + return 2; + } if (flags['dry-run']) { info(`[dry-run] apply Codex ${choice} preset at user scope`); return 0; } // Persist ownership first. If the TOML merge then fails, sync retains enough // intent to report/retry it; the inverse ordering could mutate config.toml diff --git a/src/lib/aqe-readiness.mjs b/src/lib/aqe-readiness.mjs index 6c8c204f..91f819c4 100644 --- a/src/lib/aqe-readiness.mjs +++ b/src/lib/aqe-readiness.mjs @@ -24,33 +24,13 @@ export function aqeEmbeddingConfiguration({ packageRoot = aqeRoot(), env = proce } catch { return { status: 'missing-backend', backend: null }; } } -// TEMPORARY (remove once fixed upstream; tracked by pacphi/agentic-kit#240): on a live -// RVF lock, agentic-qe 3.14.3 logs the busy warning, then falls through to a create -// attempt that fails with FsyncFailed; store and lock are untouched (agentic-qe#574). -// 3.14.4 carries agentic-qe#719, a partial fix (it rethrows LockHeld); whether 3.14.4 -// still emits this sequence is unverified, so the rule stays for it. Only that exact -// sequence is contention. -// The middle line is emitted solely by AQE's live-owner quarantine refusal, so a bare -// FsyncFailed, or a lock warning plus FsyncFailed without it, still fails below. The rule -// has no version gate. Remove it and its test when a released agentic-qe fixes -// agentic-qe#574 and that release is the kit floor; #719 alone does not remove it. -const LIVE_OWNER_CONTENTION = [ - /is locked by a live process/, - /is unusable but its lock is held by a live process/, - /FsyncFailed|0x0303/, -]; - export function classifyAqeStartup(result) { if (result.code !== 0) return { status: 'failed', reason: 'AQE command failed' }; const output = `${result.stdout ?? ''}\n${result.stderr ?? ''}`; - const contention = LIVE_OWNER_CONTENTION.every(line => line.test(output)); - if (!contention && /FsyncFailed|0x0303/.test(output)) return { status: 'failed', reason: 'RVF backend failed' }; + if (/FsyncFailed|0x0303/.test(output)) return { status: 'failed', reason: 'RVF backend failed' }; if (/prewarm failed|Transformer initialization previously failed|Embedding model failed|semantic embedding unavailable/i.test(output)) { return { status: 'degraded', reason: 'Embedding initialization failed' }; } - if (contention) { - return { status: 'busy', reason: 'RVF is held by another live process; its FsyncFailed came from an AQE create attempt during contention (agentic-qe#574), not a storage error; SQLite fallback observed; owner health and RVF integrity unverified' }; - } if (/locked by a live process|lock is held by a live process/.test(output)) { return { status: 'busy', reason: 'RVF is held by another live process; SQLite fallback observed; owner health and RVF integrity unverified' }; } diff --git a/src/lib/aqe-store-merge.mjs b/src/lib/aqe-store-merge.mjs index 22fc12af..35ea1a4c 100644 --- a/src/lib/aqe-store-merge.mjs +++ b/src/lib/aqe-store-merge.mjs @@ -27,7 +27,7 @@ // 3 backup VACUUM INTO /backup/root-memory.db. // 4 rehearse on a copy of that backup: per stray copy, delete its // witness_chain rows (appended unlinked they break the root's -// audit chain; Branch 5 Task 0.2, agentic-qe#759) and the starter patterns the +// audit chain; agentic-qe#759) and the starter patterns the // root does not hold, with the rows that must reference them (a // *pattern_id column that is NOT NULL or a foreign key to // qe_patterns: embeddings, usage, null results, lineage, @@ -39,7 +39,7 @@ // must equal the preview's union by (name, qe_domain, // pattern_type) and experience id; integrity and foreign keys // clean. AQE 3.14.4 skips shared patterns as conflicts without -// pruning (Task 0.2), so no agentic-qe#736 prune step runs. +// pruning, so no agentic-qe#736 prune step runs. // The archived strays keep their starter patterns. // 5 apply writers again, and every store must look as it did when the // preview copied it (fingerprint: size and mtime of each file; the diff --git a/src/lib/census-presentation.mjs b/src/lib/census-presentation.mjs new file mode 100644 index 00000000..dbb5c752 --- /dev/null +++ b/src/lib/census-presentation.mjs @@ -0,0 +1,8 @@ +/** An absent legacy field is unknown, never a measured zero. Shared by CLI/UI. */ +export function censusDisclosure(census = {}) { + const count = (field) => Number.isInteger(census[field]) && census[field] >= 0 ? String(census[field]) : 'Unknown number of'; + return `${count('importedExcluded')} confirmed pure imported copies excluded (no project, host or origin contribution); ` + + `${count('importedMixed')} mixed files retain proven native activity; ` + + `${count('importedUnresolved')} files have unresolved bounded ownership (not confirmed exclusions). ` + + 'The dedicated Cowork transcript source is not covered; Cowork declarations in covered transcripts remain valid observations.'; +} diff --git a/src/lib/codex-import-marker.mjs b/src/lib/codex-import-marker.mjs index 1be22e44..afa1ab0f 100644 --- a/src/lib/codex-import-marker.mjs +++ b/src/lib/codex-import-marker.mjs @@ -12,14 +12,15 @@ export const CODEX_IMPORT_TURN_PREFIX = 'external-import-turn'; -/** One decoded rollout record: is it a turn of an imported thread? Only a +/** One decoded rollout record: does it explicitly belong to an imported turn? Only a * string `payload.turn_id` counts; the marker text inside a message does not. */ export function isCodexImportedLine(e) { const turnId = e?.payload?.turn_id; return typeof turnId === 'string' && turnId.startsWith(CODEX_IMPORT_TURN_PREFIX); } -/** A rollout's bounded head (raw JSON lines): is the rollout an imported copy? +/** Does a bounded head contain an imported turn? This does not prove the + * whole rollout is imported-only; later native turns require a separate scan. * A line without the marker text is not parsed, which keeps this cheap on the * large native heads; unparseable and non-string lines are skipped. */ export function isImportedCodexRollout(headLines) { @@ -31,3 +32,69 @@ export function isImportedCodexRollout(headLines) { } return false; } + +/** Stateful per-turn ownership. Explicit turn metadata outranks adjacency; + * a foreign completion never closes the active turn. Missing IDs in a mixed + * file cannot open a native turn. No IDs or payloads escape the state. */ +export function newCodexTurnOwnership({ hasImports = true } = {}) { + return { hasImports, ownershipComplete: true, activeId: null, activeOwner: hasImports ? 'ambiguous' : 'native', + importedIds: new Set(), importedTurnCountComplete: true, importedRecords: 0, ambiguousRecords: 0, nativeRecords: 0 }; +} + +const validTurnId = (id) => typeof id === 'string' && id.length > 0 + && id.length <= 256 && !/\s/u.test(id); + +function noteBoundary(state, p, imported) { + // Absence can mean an enriching context. An explicitly invalid declaration + // cannot preserve adjacency to the prior native turn in a mixed source. + if (state.hasImports && Object.hasOwn(p, 'turn_id') && !validTurnId(p.turn_id)) { + state.activeId = null; + state.activeOwner = 'ambiguous'; + state.ownershipComplete = false; + return; + } + if (!validTurnId(p.turn_id) && p.type !== 'task_started') return; + state.activeId = validTurnId(p.turn_id) ? p.turn_id : null; + state.activeOwner = imported ? 'imported' : state.activeId ? 'native' + : state.hasImports ? 'ambiguous' : 'native'; +} + +function endsActiveTurn(record, p, activeId) { + return record?.type === 'event_msg' && ['task_complete', 'turn_aborted'].includes(p.type) + && validTurnId(p.turn_id) && p.turn_id === activeId; +} + +function noteImportedId(state, id) { + if (state.importedIds.has(id)) return; + if (validTurnId(id) && state.importedIds.size < 4096) state.importedIds.add(id); + else state.importedTurnCountComplete = false; +} + +export function codexTurnOwner(state, record) { + const p = record?.payload ?? {}; + const id = p.turn_id; + const imported = isCodexImportedLine(record); + const boundary = record?.type === 'turn_context' + || (record?.type === 'event_msg' && p.type === 'task_started'); + // An ID-less context enriches an identified turn but cannot open one. + if (boundary) noteBoundary(state, p, imported); + let owner = imported ? 'imported' : state.activeOwner; + if (state.hasImports && id !== undefined && (!validTurnId(id) || id !== state.activeId) && !imported) owner = 'ambiguous'; + if (imported) { noteImportedId(state, id); state.importedRecords++; } + else if (owner === 'imported') state.importedRecords++; + else if (owner === 'ambiguous' && record?.type !== 'session_meta') state.ambiguousRecords++; + if (owner === 'native' && record?.type === 'event_msg' + && ['user_message', 'agent_message', 'item_completed', 'token_count'].includes(p.type)) state.nativeRecords++; + if (endsActiveTurn(record, p, state.activeId)) { + state.activeId = null; + state.activeOwner = state.hasImports ? 'ambiguous' : 'native'; + } + return owner; +} + +/** Only enumerated counts are persisted; turn IDs remain local to the walk. */ +export function codexImportEvidence(state) { + return { importedTurns: state.importedIds.size, importedTurnCountComplete: state.importedTurnCountComplete, + importedRecords: state.importedRecords, + ambiguousRecords: state.ambiguousRecords, nativeRecords: state.nativeRecords }; +} diff --git a/src/lib/codex-rollout-reader.mjs b/src/lib/codex-rollout-reader.mjs index fba25d8d..989cb113 100644 --- a/src/lib/codex-rollout-reader.mjs +++ b/src/lib/codex-rollout-reader.mjs @@ -57,12 +57,16 @@ function clippedStub(head) { } /** Parse one whole line, or `null` for anything that is not a JSON object. */ -function parseLine(buf) { - if (!buf.length || buf[0] !== OPEN_BRACE) return null; +function parseLine(buf, stats) { + if (!buf.length) return null; + if (buf[0] !== OPEN_BRACE) { + if (buf.toString('utf8').trim()) stats.malformedRecords++; + return null; + } try { const obj = JSON.parse(buf.toString('utf8')); return obj && typeof obj === 'object' ? obj : null; - } catch { return null; } + } catch { stats.malformedRecords++; return null; } } /** @@ -75,6 +79,7 @@ function parseLine(buf) { */ function* passOver(file, { chunkBytes, maxLineBytes, stats }) { stats.clippedLines = 0; + stats.malformedRecords = 0; const fd = fs.openSync(file, 'r'); try { const limit = fs.fstatSync(fd).size; @@ -102,7 +107,7 @@ function* passOver(file, { chunkBytes, maxLineBytes, stats }) { stats.clippedLines++; out = clippedStub(head); } else if (partsLen) { - out = parseLine(parts.length === 1 ? parts[0] : Buffer.concat(parts, partsLen)); + out = parseLine(parts.length === 1 ? parts[0] : Buffer.concat(parts, partsLen), stats); } parts = []; partsLen = 0; oversize = false; head = null; return out; @@ -119,7 +124,7 @@ function* passOver(file, { chunkBytes, maxLineBytes, stats }) { if (nl < 0) break; let obj; if (partsLen === 0 && !oversize && nl - start <= maxLineBytes) { - obj = parseLine(view.subarray(start, nl)); // whole line inside this chunk: no copy + obj = parseLine(view.subarray(start, nl), stats); // whole line inside this chunk: no copy } else { take(Buffer.from(view.subarray(start, nl))); // `chunk` is reused, so keep a copy obj = finish(); @@ -140,7 +145,7 @@ function* passOver(file, { chunkBytes, maxLineBytes, stats }) { * Open a rollout for parsing. Returns a source `parseCodex` accepts in place of * a string: `head` (the first 256 KiB, for session-origin detection), `lines` * (re-iterable — each iteration re-reads the file, which the subagent replay - * pre-pass needs) and `stats` (`clippedLines` of the LAST pass). + * pre-pass needs) and `stats` (`clippedLines` and `malformedRecords` of the LAST pass). * * Throws (ENOENT, EACCES, …) when the file cannot be opened or its head read; * the caller decides how to report that. @@ -151,7 +156,7 @@ function* passOver(file, { chunkBytes, maxLineBytes, stats }) { export function openCodexRollout(file, limits = {}) { const chunkBytes = limits.chunkBytes ?? DEFAULT_CHUNK_BYTES; const maxLineBytes = limits.maxLineBytes ?? DEFAULT_MAX_LINE_BYTES; - const stats = { clippedLines: 0 }; + const stats = { clippedLines: 0, malformedRecords: 0 }; const fd = fs.openSync(file, 'r'); let head; try { diff --git a/src/lib/codex-usage-walk.mjs b/src/lib/codex-usage-walk.mjs index da7e8a2e..92a35a06 100644 --- a/src/lib/codex-usage-walk.mjs +++ b/src/lib/codex-usage-walk.mjs @@ -33,9 +33,12 @@ const ZERO = Object.freeze({ input_tokens: 0, cached_input_tokens: 0, output_tok const MONOTONIC = ['input_tokens', 'cached_input_tokens', 'output_tokens', 'total_tokens']; const snapshotOf = (t) => Object.fromEntries(FIELDS.map((f) => [f, Number(t?.[f]) || 0])); +const isTotalOnly = (t) => t.total_tokens > 0 && !t.input_tokens && !t.cached_input_tokens && !t.output_tokens; /** * @typedef {object} CodexUsageWalk + * @property {boolean} excludedBaseline + * @property {boolean} unknownBaseline * @property {boolean} unattributable * @property {Record|null} prev * @property {string|null} model @@ -51,6 +54,8 @@ const snapshotOf = (t) => Object.fromEntries(FIELDS.map((f) => [f, Number(t?.[f] export function newCodexUsageWalk({ unattributable = false } = {}) { return { unattributable, + excludedBaseline: false, + unknownBaseline: false, // a total-only counter hid the component baseline prev: null, // the previous cumulative snapshot, replayed ones included model: null, // the model of the turn_context in effect lastMs: null, // the last finite event time seen on a token_count @@ -94,15 +99,31 @@ export function noteCodexWalkResponse(walk, ms, dayOf) { * the running total; an own one books its delta on `ms`'s day under the current * model. */ -export function walkCodexTokenCount(walk, total, ms, replay, dayOf) { +export function walkCodexTokenCount(walk, total, ms, replay, dayOf, last = null) { if (!total) return; const cur = snapshotOf(total); const prev = walk.prev; - walk.prev = cur; if (Number.isFinite(ms)) walk.lastMs = ms; - const restarted = prev !== null && MONOTONIC.some((f) => cur[f] < prev[f]); + if (isTotalOnly(cur)) { + // A total-only snapshot has no component baseline. Treat the next full + // snapshot as a new baseline unless that call itself proves a reset. + walk.unknownBaseline = true; + walk.excludedBaseline = replay; + return; + } + // A first native call can reset ABOVE the copied baseline. Matching + // last/total counters prove that reset; an identical re-emission does not. + const lastSnapshot = last ? snapshotOf(last) : null; + const explicitReset = (walk.excludedBaseline || walk.unknownBaseline) && lastSnapshot + && FIELDS.every((f) => cur[f] === lastSnapshot[f]) + && (prev === null || FIELDS.some((f) => cur[f] !== prev[f])); + const restarted = explicitReset || (prev !== null && MONOTONIC.some((f) => cur[f] < prev[f])); + const unknownBaseline = walk.unknownBaseline; + walk.unknownBaseline = false; + walk.prev = cur; + walk.excludedBaseline = replay; if (restarted) walk.segments++; - if (replay || walk.unattributable) return; + if (replay || walk.unattributable || (unknownBaseline && !restarted)) return; const base = prev === null || restarted ? ZERO : prev; const d = Object.fromEntries(FIELDS.map((f) => [f, Math.max(0, cur[f] - base[f])])); if (!d.input_tokens && !d.output_tokens && !d.cached_input_tokens) return; diff --git a/src/lib/dashboard-server.mjs b/src/lib/dashboard-server.mjs index 5b86f267..119daac9 100644 --- a/src/lib/dashboard-server.mjs +++ b/src/lib/dashboard-server.mjs @@ -1,4 +1,6 @@ import { HOST_HEALTH_POST_ROUTES, handleHostHealthPost } from './dashboard/host-health-api.mjs'; +import { REFRESH_POST_ROUTES, REFRESH_LOCAL_TIMEOUT_MS, createRefreshOperation, + dashboardRefreshStages, handleRefreshPost, handleRefreshGet } from './dashboard/refresh-api.mjs'; import { createHostReadinessReader } from './host-readiness.mjs'; // dashboard-server.mjs — a read-only, localhost-only web dashboard for the kit. // @@ -37,14 +39,13 @@ import { createHostReadinessReader } from './host-readiness.mjs'; // GET /api/system → the machine-footprint payload (ADR-0025): the cheap // tier (runtime census + known-file stats, TTL-cached // ~60s) merged with the last persisted deep snapshot, -// carried forward with ITS asOf. `?refresh=deep` starts -// or attaches to the single-flight deep scan and returns -// immediately with progress state; `&trees=1|0` sets -// whether that scan walks project working trees. -// GET /api/system/summary → the same read (same `?refresh=deep&trees=`) +// carried forward with ITS asOf. This GET is read-only. +// GET /api/system/summary → the same read // with the catalog projected to what the System page // draws (dashboard/system-summary.mjs). The page and its // Runtime poll read this; /api/system stays complete. +// POST /api/refresh → starts one explicit staged refresh operation. +// GET /api/refresh → reads progress for that operation. // // The status rows are gathered by calling status.mjs's own collect() IN // PROCESS — safe only because the evidence store (ADR-0063) made a warm-cache @@ -148,16 +149,16 @@ const STATUS_TIMEOUT_MS = 30_000; * settles within STATUS_TIMEOUT_MS, resolves to an honest empty payload rather than * rejecting or hanging the server, so /api/status always answers with valid * JSON. */ -function inProcessStatus(cwd) { +function inProcessStatus(cwd, { refresh = false, timeoutMs = STATUS_TIMEOUT_MS } = {}) { return () => new Promise((resolve) => { let settled = false; const timer = setTimeout(() => { if (settled) return; settled = true; resolve({ overall: 'unknown', rows: [], error: 'status collection timed out' }); - }, STATUS_TIMEOUT_MS); + }, timeoutMs); timer.unref?.(); - statusCollect({ pkgRoot: PKG_ROOT, cwd, refresh: false }) + statusCollect({ pkgRoot: PKG_ROOT, cwd, refresh }) .then((rows) => ({ overall: worstLevel(rows), rows })) .catch((e) => ({ overall: 'unknown', rows: [], error: String(e?.message ?? e) })) .then((result) => { @@ -286,7 +287,8 @@ function censusBackedDiscovery() { readCensus: () => (last ? { everSeen: last.everSeen, onDisk: last.onDisk, gitRepos: last.gitRepos, learning: last.learning, - complete: last.complete, + complete: last.complete, importedExcluded: last.importedExcluded, + importedMixed: last.importedMixed, importedUnresolved: last.importedUnresolved, } : null), }; } @@ -377,8 +379,8 @@ async function collectData({ cwd, fetchStatus, projectParam, getProjectSnapshot, intel: { selectedProjectKey: selected?.key ?? null, selectedProjectLabel: selected?.label ?? null, - projects: projects.map(({ key, label, path: projectPath, source, learningScope, learningScopeEvidence, learningOrigins, learningObservedAt }) => ( - { key, label, path: projectPath, source, + projects: projects.map(({ key, label, path: projectPath, source, hosts, sessionOrigins, sessionSurfaces, learningScope, learningScopeEvidence, learningOrigins, learningObservedAt }) => ( + { key, label, path: projectPath, source, hosts, sessionOrigins, sessionSurfaces, learningScope: ['repository', 'worktree', 'user'].includes(learningScope) ? learningScope : 'unknown', learningScopeEvidence: learningScopeEvidence ?? 'unclassified', learningObservedAt: learningObservedAt ?? null, learningOrigins: ['claude-desktop', 'codex-desktop'].filter((origin) => learningOrigins?.includes(origin)) } @@ -1086,7 +1088,7 @@ function lazyLive(liveOptions = {}) { * machineWideIntel?: (projects: Array) => any, * models?: any, modelScopeKey?: string, system?: any, systemOptions?: any, * maintenance?: any, maintenanceOptions?: any, hostReadiness?: any, - * management?: any, managementOptions?: any }} [opts] + * management?: any, managementOptions?: any, refreshStages?: Record Promise> }} [opts] * @returns {Promise<{ url: string, urlWithToken: string, port: number, token: string, close: () => Promise }>} */ export function startDashboard({ @@ -1097,7 +1099,7 @@ export function startDashboard({ transcriptClientBuffer = 64, transcriptMaxClients = 16, intelWatch, intelClientBuffer = 256, intelMaxClients = 32, discoverProjects, machineWideIntel, models, modelScopeKey, system, systemOptions = {}, - maintenance, maintenanceOptions = {}, management, managementOptions = {}, hostReadiness, + maintenance, maintenanceOptions = {}, management, managementOptions = {}, hostReadiness, refreshStages, } = {}) { const refused = refusesDefaultState({ fetchStatus, usage, limits, hooks, live, transcripts, intelWatch, discoverProjects, machineWideIntel, @@ -1166,7 +1168,7 @@ export function startDashboard({ const provideMaintenance = typeof maintenance === 'function' ? maintenance : maintenance ? async () => maintenance : async () => { if (refused) { - // Logged here because refreshMaintenanceAfterSystem swallows errors. + // Log once when the default Maintenance service is refused. if (!refusalLogged) { refusalLogged = true; console.error(HERMETIC_REFUSAL); } throw new TypeError(HERMETIC_REFUSAL); } @@ -1216,24 +1218,11 @@ export function startDashboard({ .finally(() => { inventoryRefreshPromise = null; }); return inventoryRefreshPromise; } - let maintenanceRefreshSource = null; - let maintenanceRefreshPromise = null; - function refreshMaintenanceAfterSystem(deepScan) { - if (maintenanceRefreshSource === deepScan) return maintenanceRefreshPromise; - maintenanceRefreshSource = deepScan; - maintenanceRefreshPromise = Promise.resolve(deepScan).then(async (result) => { - if (result?.ok !== true || result?.persisted?.ok === false) return null; - const model = await (await getMaintenance()).scan({ deep: false }); - refreshInventoryAfterProviderScan({ measured: true }); - return model; - }).catch(() => null).finally(() => { - if (maintenanceRefreshSource === deepScan) { - maintenanceRefreshSource = null; - maintenanceRefreshPromise = null; - } - }); - return maintenanceRefreshPromise; - } + const refreshOperation = createRefreshOperation({ stages: refreshStages ?? dashboardRefreshStages({ + cwd, pkgRoot: PKG_ROOT, getSystem, getMaintenance, refreshInventoryAfterProviderScan, + getHostReadiness, statusCollect: inProcessStatus(cwd, { refresh: true, timeoutMs: REFRESH_LOCAL_TIMEOUT_MS }), + loadConfig: loadKitConfig, + }) }); let transcriptServicePromise; const provideTranscripts = typeof transcripts === 'function' ? transcripts : transcripts ? async () => transcripts : async () => { @@ -1403,7 +1392,7 @@ export function startDashboard({ let maintenanceApiPromise; const getMaintenanceApi = async () => (maintenanceApiPromise ||= getMaintenance() .then((service) => createMaintenanceDashboardApi({ - service, management: getManagement, sessionToken: token, afterScan: refreshInventoryAfterProviderScan, + service, management: getManagement, sessionToken: token, }))); const server = http.createServer(async (req, res) => { @@ -1417,7 +1406,8 @@ export function startDashboard({ const maintenanceMutation = req.method === 'POST' && (MAINTENANCE_MUTATION_ROUTES.has(url) || MAINTENANCE_V2_MUTATION_ROUTES.has(url)); const healthMutation = req.method === 'POST' && HOST_HEALTH_POST_ROUTES.has(url); - if (req.method !== 'GET' && !maintenanceMutation && !healthMutation) { + const refreshMutation = req.method === 'POST' && REFRESH_POST_ROUTES.has(url); + if (req.method !== 'GET' && !maintenanceMutation && !healthMutation && !refreshMutation) { res.writeHead(405).end('method not allowed'); return; } @@ -1445,20 +1435,21 @@ export function startDashboard({ // Query tokens remain an SSE compatibility exception for GET. Mutation // capability can only be reached with the explicit header; it never rides // in a URL, browser history, referrer or server log. - const authorized = (maintenanceMutation || healthMutation) + const authorized = (maintenanceMutation || healthMutation || refreshMutation) ? tokenMatches(req.headers['x-dash-token'], token) : checkToken(req, query); if (url.startsWith('/api/') && !authorized) { sendUnauthorized(res, 'Wrong or missing dashboard token.'); return; } - if (maintenanceMutation || healthMutation) { + if (maintenanceMutation || healthMutation || refreshMutation) { const mutationRejection = maintenanceMutationRejection(req.headers); if (mutationRejection) { res.writeHead(403, { 'content-type': 'text/plain; charset=utf-8' }); res.end(mutationRejection); return; } - if (healthMutation) { await handleHostHealthPost(url, req, res, getHostReadiness); return; } + if (healthMutation) { await handleHostHealthPost(req, res, getHostReadiness); return; } + if (refreshMutation) { await handleRefreshPost(req, res, refreshOperation); return; } try { await (await getMaintenanceApi()).mutate(url, req, res); } catch { sendJson(res, 503, { error: 'maintenance operation unavailable' }); } return; @@ -1987,35 +1978,13 @@ export function startDashboard({ // `project` shapes the answer for a route: identity for /api/system (the // documented `ak system --json` shape), systemSummaryPayload for the page. async function handleSystem(req, res, query, project = (payload) => payload) { + if (query.has('refresh') || query.has('trees')) { + sendJson(res, 400, { error: 'start a refresh with POST /api/refresh' }); + return; + } try { const collector = await getSystem(); - // ORDER IS LOAD-BEARING: assemble the payload BEFORE starting a scan. - // The deep collectors are synchronous, so the first phase occupies the - // event loop the moment it gets a turn — and `read()` awaits, which - // hands it that turn. Starting first therefore made the *initiating* - // request wait out the phase it had just kicked off (measured: 9s), - // which is precisely the hang the progress state exists to avoid. const payload = await collector.read(); - if (query.get('refresh') === 'deep') { - // Start-or-attach and answer NOW. The collector's single flight means - // a second refresh joins the running scan rather than racing it, and - // it never rejects — the catch guards an injected collector that does - // not honour that contract, so a bad one cannot take the process down - // with an unhandled rejection. - // `trees` is a MEASUREMENT parameter, not a view filter: project - // working trees are only walked when it is set, and one large - // repository outweighs every shared cache combined — so the ranking - // has to be re-measured, not re-sorted. Absent means "keep whatever - // the collector already defaults to". - const trees = query.get('trees'); - const deepScan = Promise.resolve(collector.refreshDeep( - trees == null ? undefined : { includeProjectTrees: trees === '1' }, - )); - refreshMaintenanceAfterSystem(deepScan); - // The payload predates the start by microseconds; re-stamp the live - // scan block so this response reads "running", not "idle". - if (typeof collector.scanState === 'function') payload.scan = collector.scanState(); - } sendJson(res, 200, project(payload)); } catch (e) { sendJson(res, 503, { error: 'system footprint unavailable', reason: String(e && e.message || e) }); @@ -2024,13 +1993,11 @@ export function startDashboard({ } async function handleMaintenance(req, res, query) { - const refresh = query.getAll('refresh'); - if ([...query.keys()].some((key) => key !== 'refresh') - || refresh.length > 1 || (refresh.length === 1 && refresh[0] !== 'scan')) { - sendJson(res, 400, { error: 'invalid maintenance scan request' }); + if (query.size > 0) { + sendJson(res, 400, { error: 'start a refresh with POST /api/refresh' }); return; } - try { await (await getMaintenanceApi()).report(req, res, { refresh: query.get('refresh') === 'scan' }); } + try { await (await getMaintenanceApi()).report(req, res); } catch { sendJson(res, 503, { error: 'maintenance evidence unavailable' }); } return; } @@ -2065,7 +2032,7 @@ export function startDashboard({ // filesystem call, so the 400 happens here and nowhere deeper. Two // shapes are accepted: a plain id (unchanged — parseSessionId/ // resolvesInsideRoot, exactly as before), or a namespaced - // `/` Claude subagent id (Task 5 round 2), gated by + // `/` Claude subagent id, gated by // its own dedicated parse+containment pair rather than loosening // parseSessionId/resolvesInsideRoot — those two also gate the // live-playback/SSE routes, which this fix does not touch. @@ -2113,6 +2080,7 @@ export function startDashboard({ // out separately into sse.mjs's sseRoute(). const ROUTES = { '/api/status': handleStatus, + '/api/refresh': (_req, res) => handleRefreshGet(res, refreshOperation), '/api/host-health': async (_req, res) => { try { sendJson(res, 200, await getHostReadiness()); } catch { sendJson(res, 503, { error: 'Host health checks unavailable.' }); } diff --git a/src/lib/dashboard/client.mjs b/src/lib/dashboard/client.mjs index 90d166ba..167e835e 100644 --- a/src/lib/dashboard/client.mjs +++ b/src/lib/dashboard/client.mjs @@ -1,3 +1,5 @@ +import { censusDisclosure } from '../census-presentation.mjs'; +import { SESSION_SURFACE_LABELS, SESSION_HOST_LABELS, SESSION_INITIATOR_LABELS, SESSION_PROVIDER_LABELS, sessionPresentation, sessionProviderPresentation } from '../session-surface.mjs'; import { repositoryTree } from './project-groups.mjs'; import { contextCard } from './context-card.mjs'; import { contextHostCard } from './context-host-card.mjs'; @@ -92,8 +94,11 @@ aboutSrc = inject(aboutSrc, 'var ABOUT = []; // PLACEHOLDER:ABOUT_JS', `var ABOU const datetimeSrc = readSplit('datetime.mjs'); const hostReadinessSrc = readSplit('host-readiness.mjs'); +const sessionPresentationSrc = readSplit('session-presentation.mjs'); +const sessionVocabularySrc = `const SESSION_SURFACE_LABELS=${JSON.stringify(SESSION_SURFACE_LABELS)},SESSION_HOST_LABELS=${JSON.stringify(SESSION_HOST_LABELS)},SESSION_INITIATOR_LABELS=${JSON.stringify(SESSION_INITIATOR_LABELS)},SESSION_PROVIDER_LABELS=${JSON.stringify(SESSION_PROVIDER_LABELS)};${sessionPresentation.toString()}${sessionProviderPresentation.toString()}`; const intelligenceSrc = readSplit('intelligence.mjs'); const pollSrc = readSplit('poll.mjs'); +const refreshControlSrc = readSplit('refresh-control.mjs'); // usage-rhythm.mjs declares its OWN `esc` on disk, and its comment says why: // the tests import it as real ESM, where bootstrap.mjs's `esc` is still the // build-time stub. In the concatenated bundle every file shares ONE scope, so @@ -164,5 +169,5 @@ const bootSrc = readSplit('boot.mjs'); export const JS = ` (function(){ ${bootstrapSrc}${contextCard.toString()}${contextHostCard.toString()}${repositoryTree.toString()}${overviewSrc}${datetimeSrc}${hostReadinessSrc} -${intelligenceSrc}${pollSrc}${usageRhythmSrc}${usagePromptsSrc}${usageContextHooksSrc}${usageSrc}${modelLifecycleSrc}${usageOrchestratorsSrc}${rufloComponentsSrc}${aboutSrc}${systemReadoutSrc}${systemProjectsSrc}${maintenanceWorkspaceSrc}${maintenanceFiltersSrc}${maintenanceCardsSrc}${maintenanceOperationSrc}${maintenanceLanguageLogosSrc}${maintenanceFocusSrc}${maintenanceInventorySrc}${maintenanceRelationshipsSrc}${maintenanceInspectorSrc}${maintenanceGuidanceSrc}${maintenanceDiscoverySrc}${maintenanceActivitySrc}${systemMaintenanceActionsSrc}${systemMaintenanceSrc}${bootSrc}})(); +${censusDisclosure.toString()}${sessionVocabularySrc}${sessionPresentationSrc}${intelligenceSrc}${pollSrc}${refreshControlSrc}${usageRhythmSrc}${usagePromptsSrc}${usageContextHooksSrc}${usageSrc}${modelLifecycleSrc}${usageOrchestratorsSrc}${rufloComponentsSrc}${aboutSrc}${systemReadoutSrc}${systemProjectsSrc}${maintenanceWorkspaceSrc}${maintenanceFiltersSrc}${maintenanceCardsSrc}${maintenanceOperationSrc}${maintenanceLanguageLogosSrc}${maintenanceFocusSrc}${maintenanceInventorySrc}${maintenanceRelationshipsSrc}${maintenanceInspectorSrc}${maintenanceGuidanceSrc}${maintenanceDiscoverySrc}${maintenanceActivitySrc}${systemMaintenanceActionsSrc}${systemMaintenanceSrc}${bootSrc}})(); `; diff --git a/src/lib/dashboard/client/boot.mjs b/src/lib/dashboard/client/boot.mjs index a825ccaa..7b9f6cda 100644 --- a/src/lib/dashboard/client/boot.mjs +++ b/src/lib/dashboard/client/boot.mjs @@ -6,6 +6,7 @@ import { renderAbout, wireAboutNudge } from './about.mjs'; import { activeTab, initialLiveScope, setSystemView, setTab, syncHash, systemView } from './bootstrap.mjs'; import { tickClock, wireIntelPicker } from './intelligence.mjs'; import { pollStatus, schedulePoll, wirePoll, wireStripCollapse } from './poll.mjs'; +import { wireRefresh } from './refresh-control.mjs'; import { renderSystemFreshness, wireCatalogFilters, wireSystem } from './system-projects.mjs'; import { wireMaintenance } from './system-maintenance.mjs'; import { wireUsage } from './usage-orchestrators.mjs'; @@ -22,6 +23,7 @@ import { loadUsage, setUsageView } from './usage.mjs'; renderSystemFreshness(); wireHostHealth(); wirePoll(); + wireRefresh(); wireUsage(); wireIntelPicker(); wireAboutNudge(); diff --git a/src/lib/dashboard/client/host-readiness.mjs b/src/lib/dashboard/client/host-readiness.mjs index c05aef0b..8fd5dc47 100644 --- a/src/lib/dashboard/client/host-readiness.mjs +++ b/src/lib/dashboard/client/host-readiness.mjs @@ -1,6 +1,7 @@ // @ts-nocheck — browser bundle source; assembled by ../client.mjs. import { esc, authHeaders } from './bootstrap.mjs'; import { sourceHostIcon } from './usage.mjs'; +import { startRefresh, refreshRunning } from './refresh-control.mjs'; var HEALTH_REPORT=null, HEALTH_HOST=null, HEALTH_BUSY=false, HEALTH_BUSY_HOST=null, HEALTH_ACK=null; var HEALTH_NAMES={claude:'Claude Code',codex:'Codex',opencode:'OpenCode'}; @@ -30,7 +31,9 @@ export function renderHostReadiness(report,checking){ el.hidden=false; el.innerHTML=['claude','codex','opencode'].map(function(host){ var row=report&&report.hosts&&report.hosts[host]; - var state=checking||(HEALTH_BUSY&&HEALTH_BUSY_HOST===host)?'checking':row&&HEALTH_LABELS[row.status]?row.status:'unknown'; + var configuration=row&&row.checks&&row.checks.configuration; + var unassessed=host==='claude'&&configuration&&['unknown','not-run','not-checked'].includes(configuration.state); + var state=checking||(HEALTH_BUSY&&HEALTH_BUSY_HOST===host)?'checking':unassessed?'unknown':row&&HEALTH_LABELS[row.status]?row.status:'unknown'; var managed=!row||hostIsManaged(row); // A managed host's badge is its health; any other host's badge is its // management word, in a neutral colour: its problems are information. @@ -123,7 +126,7 @@ function renderHealthDialog(){ if(!row||HEALTH_ACK!==row.evidenceKey){consent.checked=false;HEALTH_ACK=null;} consent.disabled=HEALTH_BUSY||!row||!row.canCheckConnection; document.getElementById('host-health-connect').disabled=HEALTH_BUSY||!row||!row.canCheckConnection||!consent.checked; - document.getElementById('host-health-refresh').disabled=HEALTH_BUSY; + document.getElementById('host-health-run-refresh').disabled=HEALTH_BUSY||refreshRunning(); } async function runHealthCheck(connected){ @@ -175,7 +178,7 @@ export function wireHostHealth(){ var button=region.querySelector('[data-health-host="'+HEALTH_HOST+'"]');if(button)button.focus(); }); document.getElementById('host-health-consent').addEventListener('change',function(event){HEALTH_ACK=event.target.checked&&healthRow()?healthRow().evidenceKey:null;renderHealthDialog();}); - document.getElementById('host-health-refresh').addEventListener('click',function(){runHealthCheck(false);}); + document.getElementById('host-health-run-refresh').addEventListener('click',function(){startRefresh('local');}); document.getElementById('host-health-connect').addEventListener('click',function(){runHealthCheck(true);}); dialog.addEventListener('click',function(event){ var button=event.target.closest('[data-copy]'); diff --git a/src/lib/dashboard/client/intelligence.mjs b/src/lib/dashboard/client/intelligence.mjs index 239641be..5a3e2308 100644 --- a/src/lib/dashboard/client/intelligence.mjs +++ b/src/lib/dashboard/client/intelligence.mjs @@ -1,6 +1,8 @@ // @ts-nocheck — browser bundle source (never node-imported; client.mjs // reads it as text). See src/lib/dashboard/client/**'s eslint.config.mjs // override comment for why this directory isn't run through the node lib. +import { censusDisclosure } from '../../census-presentation.mjs'; +import { projectSurfacesHtml, surfaceNames } from './session-presentation.mjs'; import { renderHostReadiness } from './host-readiness.mjs'; import { renderAbout } from './about.mjs'; import { DASH_TOKEN, activeTab, esc, overviewView, positionThumb } from './bootstrap.mjs'; @@ -54,20 +56,18 @@ import { fmtNum, kpi } from './usage.mjs'; html+='

At least one transcript could not be read, ' +"so every figure above is a lower bound.

"; } + html+='

'+esc(censusDisclosure(c))+'

'; body.innerHTML=html; box.hidden=false; } - var INTEL_SCOPE_GROUPS=[['repository','Git repositories'],['worktree','Git worktrees'],['user','User-level learning'],['unknown','Other / unclassified']]; - var machineWideDesignationFilter='all'; + var INTEL_SCOPE_GROUPS=[['repository','Git repositories'],['worktree','Git worktrees'],['user','User-level learning'],['unknown','Unknown']]; + var machineWideDesignationFilter='all',machineWideSurfaceFilter='all'; function machineWideDesignation(p){ if(p.learningScope==='repository')return 'Git repository'; if(p.learningScope==='worktree')return 'Git worktree'; - var origins=Array.isArray(p.learningOrigins)?p.learningOrigins:[]; - if(origins.includes('codex-desktop'))return 'ChatGPT Desktop'; - if(origins.includes('claude-desktop'))return 'Claude Desktop'; - if(Array.isArray(p.hosts)&&p.hosts.includes('opencode'))return 'OpenCode'; - return 'Directory'; + if(p.learningScope==='user')return 'User-level learning'; + return 'Unknown'; } function intelScopeRows(rows,scope){ return rows.filter(function(p){ @@ -86,7 +86,7 @@ import { fmtNum, kpi } from './usage.mjs'; var label=p.label||'(unlabeled)'; return '
' +''+esc(label)+storeHtml+'' - +''+esc(machineWideDesignation(p))+'' + +''+esc(machineWideDesignation(p))+''+projectSurfacesHtml(p)+'' +''+esc(fmtNum(p.patternsLearned))+'' +''+esc(fmtNum(p.patternStoreCount))+'' +''+esc(lastTxt)+'
'; @@ -110,12 +110,15 @@ import { fmtNum, kpi } from './usage.mjs'; +kpi("most active project",totals.mostActiveProject||"—","by most recent learning adaptation","accent"); var table=document.getElementById("mw-table"); if(!table)return; - if(!perProject.length){table.innerHTML='
no projects discovered on this machine.
';return;} + if(!perProject.length){machineWideSurfaceFilter='all';table.innerHTML='
no projects discovered on this machine.
';return;} var designations=['all'].concat(Array.from(new Set(perProject.map(machineWideDesignation))).sort()); - var visible=(machineWideDesignationFilter==='all'?perProject:perProject.filter(function(row){return machineWideDesignation(row)===machineWideDesignationFilter;})).sort(function(a,b){return String(a.label||'').localeCompare(String(b.label||''),undefined,{sensitivity:'base',numeric:true})||String(a.key||a.path||'').localeCompare(String(b.key||b.path||''));}); + var surfaceChoices=Array.from(new Set(perProject.flatMap(surfaceNames))).sort(); + if(machineWideSurfaceFilter!=='all'&&!surfaceChoices.includes(machineWideSurfaceFilter))machineWideSurfaceFilter='all'; + var visible=(machineWideDesignationFilter==='all'?perProject:perProject.filter(function(row){return machineWideDesignation(row)===machineWideDesignationFilter;})).filter(function(row){return machineWideSurfaceFilter==='all'||surfaceNames(row).includes(machineWideSurfaceFilter);}).sort(function(a,b){return String(a.label||'').localeCompare(String(b.label||''),undefined,{sensitivity:'base',numeric:true})||String(a.key||a.path||'').localeCompare(String(b.key||b.path||''));}); table.innerHTML='
' +designations.map(function(designation){var label=designation==='all'?'All':designation;return '';}).join('') - +'
'+machineWideTable(visible); + +''+machineWideTable(visible); + var surfaceSelect=document.getElementById('mw-surface-filter');if(surfaceSelect)surfaceSelect.onchange=function(){machineWideSurfaceFilter=surfaceSelect.value;renderMachineWide(mw);}; if(table.querySelectorAll)Array.from(table.querySelectorAll('.mw-filter-pill')).forEach(function(button){button.addEventListener('click',function(){machineWideDesignationFilter=button.getAttribute('data-designation')||'all';renderMachineWide(mw);});}); } diff --git a/src/lib/dashboard/client/maintenance-activity.mjs b/src/lib/dashboard/client/maintenance-activity.mjs index d075d107..611351e7 100644 --- a/src/lib/dashboard/client/maintenance-activity.mjs +++ b/src/lib/dashboard/client/maintenance-activity.mjs @@ -49,19 +49,23 @@ import { beginMaintUndo } from './system-maintenance-actions.mjs'; +(entry.at?" — "+esc(mntAge(entry.at)):"")+""; } var mntExpandedHistoryDays=new Set(); + function mntScanTime(entry){ + var recorded=entry&&entry.recordedAt,completed=entry&&entry.completedAt; + return recorded&&Number.isFinite(Date.parse(recorded))?recorded:(completed&&Number.isFinite(Date.parse(completed))?completed:null); + } function renderMntScanHistory(){ var el=document.getElementById("mnt-scan-history");if(!el)return; var history=(MNT.activity&&(MNT.activity.scanHistory||MNT.activity.scans))||[]; - if(!history.length){el.innerHTML="

Scan history

No scans have completed yet.

";return;} + if(!history.length){el.innerHTML="

Scan history

No scan records yet.

";return;} var groups=new Map(); - history.slice().sort(function(a,b){return (Date.parse(b.completedAt)||0)-(Date.parse(a.completedAt)||0);}).forEach(function(entry){ - var day=formatLocalDay(entry.completedAt)||'Date not recorded'; + history.slice().sort(function(a,b){return (Date.parse(mntScanTime(b))||0)-(Date.parse(mntScanTime(a))||0);}).forEach(function(entry){ + var day=formatLocalDay(mntScanTime(entry))||'Date not recorded'; if(!groups.has(day))groups.set(day,[]); groups.get(day).push(entry); }); - el.innerHTML='

Scan history

' + el.innerHTML='

Scan history

Scanned sources grouped by local date, newest first
Date / timeSource scannedStatusEntries
' +Array.from(groups,function(group,index){var expanded=mntExpandedHistoryDays.has(group[0]);return '' - +group[1].map(function(entry){return '';}).join('')+'';}).join('')+'
Scan records grouped by local date, newest first
Date / timeSourceStatusEntries
'+esc(formatLocalTime(entry.completedAt)||'Time not recorded')+''+esc(entry.label||'Source no longer configured')+''+esc(MNT_SOURCE_COVERAGE_LABELS[entry.state]||(entry.state==='published'?'Complete':entry.state)) + +group[1].map(function(entry){return ''+esc(formatLocalTime(mntScanTime(entry))||'Time not recorded')+''+esc(entry.label||'Source no longer configured')+''+esc(MNT_SOURCE_COVERAGE_LABELS[entry.state]||(entry.state==='published'?'Complete':entry.state)) +(entry.limitingReason?''+esc(entry.limitingReason)+'':'')+''+(Number.isFinite(entry.visited)?esc(entry.visited.toLocaleString()):'—')+'
'; el.onclick=function(event){ var button=event.target.closest&&event.target.closest('[data-mnt-history-day]'); diff --git a/src/lib/dashboard/client/maintenance-discovery.mjs b/src/lib/dashboard/client/maintenance-discovery.mjs index cf6effe7..8774f6f4 100644 --- a/src/lib/dashboard/client/maintenance-discovery.mjs +++ b/src/lib/dashboard/client/maintenance-discovery.mjs @@ -144,7 +144,7 @@ import { MNT, mntAge, MNT_SOURCE_COVERAGE_LABELS, mntGet, mntPost, mntRegisterDe // optional structured per-source detail, not text — this workspace has // no use for it beyond what `coverage` already gives, so it stays unread. var narrative=(MNT.discovery&&(MNT.discovery.narrative||MNT.discovery.progress))||""; - // Automatic sources are covered by Re-measure machine (System Full scan) + // Automatic sources are covered by Refresh machine (machine measurement) // and carry no per-source controls; only roots the user added expose // Pause/Stop while running, Resume/Stop while paused, and Retry after a // failure. A user root starts scanning when it is saved (no "scan now"). @@ -170,7 +170,7 @@ import { MNT, mntAge, MNT_SOURCE_COVERAGE_LABELS, mntGet, mntPost, mntRegisterDe +(entry.limitingReason?''+esc(entry.limitingReason)+'':'') +(entry.lastCompletedAt?''+esc(mntAge(entry.lastCompletedAt))+'':'')+' '+controls+''; }).join('')+'' - +'

Evidence checks

Runtimes, package managers, Ollama, and providers are checked by Refresh evidence, separately from filesystem coverage. The toolbar reports the latest operation outcome.

' + +'

Evidence checks

Runtimes, package managers, Ollama, and providers are checked by Refresh, separately from filesystem coverage. The toolbar reports the latest operation outcome.

' +'

Project coverage

'+esc(MNT.discovery&&MNT.discovery.projectCoverageNote||'Configured project roots contribute to filesystem coverage. Machine-discovered projects appear in Inventory.')+'

'; } export function renderMntDiscovery(){ diff --git a/src/lib/dashboard/client/maintenance-filters.mjs b/src/lib/dashboard/client/maintenance-filters.mjs index 4f854743..54a4d9aa 100644 --- a/src/lib/dashboard/client/maintenance-filters.mjs +++ b/src/lib/dashboard/client/maintenance-filters.mjs @@ -1,4 +1,5 @@ // @ts-nocheck — dashboard browser bundle. +import { surfaceFacetLabel } from './session-presentation.mjs'; import { mntIcon, mntProjectDesignation } from './maintenance-cards.mjs'; import { esc } from './bootstrap.mjs'; import { MNT, MNT_GUIDANCE_LANE_LABELS, MNT_CONFLICT_EXPLANATIONS, MNT_CREDENTIAL_READINESS_LABELS, MNT_CURATED_VIEW_LABELS, MNT_SCOPE_LABELS, mntHumanize, mntKindLabel } from './maintenance-workspace.mjs'; @@ -9,13 +10,13 @@ import { MNT, MNT_GUIDANCE_LANE_LABELS, MNT_CONFLICT_EXPLANATIONS, MNT_CREDENTIA "evidenceFields","recentlyChanged", ]; var MNT_FACET_LABEL={ - family:"Resource",scope:"Scope",environment:"Environment",project:"Project",sessionOrigin:"Session origin",projectType:"Project type",kind:"Type",adapter:"Adapters",consumer:"Hosts", + family:"Resource",scope:"Scope",environment:"Environment",project:"Project",sessionOrigin:"Session surface",projectType:"Project type",kind:"Type",adapter:"Adapters",consumer:"Hosts", carrier:"Carrier",provenance:"Source",packageManager:"Package manager",versionState:"Version state", guidance:"Guidance",dependencyRole:"Dependency role",conflict:"Conflict", credentialReadiness:"Credential",channel:"Channel",evidenceFields:"Evidence available", recentlyChanged:"Recently changed", }; - var MNT_ADAPTER_LABELS={claude:'Claude',codex:'Codex',opencode:'OpenCode',hermes:'Hermes'}; + var MNT_ADAPTER_LABELS={claude:'Claude Code',codex:'Codex',opencode:'OpenCode',hermes:'Hermes Agent'}; var MNT_DEPENDENCY_ROLE_LABEL={"depends-on":"Depends on","depended-on-by":"Depended on by","none":"No dependency role"}; function mntFacetLabelsFor(facet){ @@ -24,8 +25,8 @@ import { MNT, MNT_GUIDANCE_LANE_LABELS, MNT_CONFLICT_EXPLANATIONS, MNT_CREDENTIA } export function mntFacetValueLabel(facet,value){ if(facet==="adapter"||facet==="consumer")return MNT_ADAPTER_LABELS[value]||mntHumanize(value); - if(facet==="sessionOrigin")return ({"claude-desktop":"Claude Desktop","codex-desktop":"ChatGPT Desktop",unknown:"Unclassified"})[value]||"Unclassified"; - if(facet==="projectType")return ({git:'Git',folder:'Folder',worktree:'Worktree',unknown:'Not checked'})[value]||'Not checked'; + if(facet==="sessionOrigin")return surfaceFacetLabel(value); + if(facet==="projectType")return ({git:'Git',folder:'Folder',worktree:'Worktree',unknown:'Unknown'})[value]||'Unknown'; if(facet==="scope")return MNT_SCOPE_LABELS[value]||mntHumanize(value); if(facet==="kind")return mntKindLabel(value); if(facet==="guidance")return MNT_GUIDANCE_LANE_LABELS[value]||mntHumanize(value); diff --git a/src/lib/dashboard/client/maintenance-focus.mjs b/src/lib/dashboard/client/maintenance-focus.mjs index d4518f30..567089f2 100644 --- a/src/lib/dashboard/client/maintenance-focus.mjs +++ b/src/lib/dashboard/client/maintenance-focus.mjs @@ -1,4 +1,5 @@ // @ts-nocheck — classic browser bundle source. +import { projectSurfacesHtml } from './session-presentation.mjs'; import { mntLanguageLogo } from './maintenance-language-logos.mjs'; import { esc } from './bootstrap.mjs'; import { MNT, MNT_SCOPE_LABELS, mntKindLabel } from './maintenance-workspace.mjs'; @@ -68,7 +69,7 @@ import { mntFacetValueLabel } from './maintenance-filters.mjs'; +(level==='project'?mntProjectKindBadge(node.projectKind):'') +(level==='project'&&node.languages&&node.languages.length?mntLanguageBadges(node.languages):'') +(node.description?''+esc(node.description)+'':'') - +(note?''+esc(note)+'':'')+''+esc(node.count)+' installation'+(node.count===1?'':'s')+''+mntIcon('chevron')+''; + +(note?''+esc(note)+'':'')+''+esc(node.count)+' installation'+(node.count===1?'':'s')+''+mntIcon('chevron')+''+(level==='project'?projectSurfacesHtml(node):'')+''; } function mntFocusInstallation(row,index){ var crumbs=(row.breadcrumb||[]).slice(),scope=row.scope||{}; diff --git a/src/lib/dashboard/client/maintenance-guidance.mjs b/src/lib/dashboard/client/maintenance-guidance.mjs index eef6576a..5ece2538 100644 --- a/src/lib/dashboard/client/maintenance-guidance.mjs +++ b/src/lib/dashboard/client/maintenance-guidance.mjs @@ -205,7 +205,7 @@ import { beginMaintPreview, beginMaintReconcile } from './system-maintenance-act var el=document.getElementById('mnt-guidance-coverage');if(!el)return; var coverage=MNT.guidance&&MNT.guidance.coverage||[]; el.innerHTML='

Guidance shows evidence-backed issues, updates, and recovery work. Optional management actions are available in Inventory.

' - +(coverage.length?'
Host coverage

Counts reflect saved evidence, not a health assessment. Action checks cover the listed resource types in host-specific adapters. Refresh evidence to update these checks.

    '+coverage.map(function(row){ + +(coverage.length?'
    Host coverage

    Counts reflect saved evidence, not a health assessment. Action checks cover the listed resource types in host-specific adapters. Refresh to update these checks.

      '+coverage.map(function(row){ return '
    • '+esc(row.label)+' · '+esc(row.placements)+' installations · '+esc(row.recommendations)+' guidance items · '+esc(row.optionalActions)+' optional actions
      '+esc(row.actionStatusLabel) +((row.actionKinds||[]).length?' ('+esc(row.actionKinds.map(mntKindLabel).join(', '))+')':'')+'
    • '; }).join('')+'
    ':''); @@ -282,7 +282,7 @@ import { beginMaintPreview, beginMaintReconcile } from './system-maintenance-act el.querySelector("#mnt-procedure-close").focus(); }).catch(function(){ if(seq!==mntProcedureSeq||!el.open)return; - el.innerHTML='

    This procedure could not be loaded. Refresh evidence and try again.

    '; + el.innerHTML='

    This procedure could not be loaded. Refresh and try again.

    '; el.querySelector("#mnt-procedure-close").focus(); }); } diff --git a/src/lib/dashboard/client/maintenance-inspector.mjs b/src/lib/dashboard/client/maintenance-inspector.mjs index a8eac097..bfd3797f 100644 --- a/src/lib/dashboard/client/maintenance-inspector.mjs +++ b/src/lib/dashboard/client/maintenance-inspector.mjs @@ -134,7 +134,7 @@ import { mntRenderGuidanceEntry, mntWireGuidanceActions } from './maintenance-gu if(MNT.inspector&&MNT.inspector.scanRequired){ el.hidden=false; el.innerHTML=mntInspectorCloseButton()+'

    No inventory has been built yet. ' - +"Use Refresh evidence, above, to build it.

    "; + +"Use Refresh, above, to build it.

    "; return; } if(mntInspectorError||!MNT.inspector){ @@ -194,7 +194,7 @@ import { mntRenderGuidanceEntry, mntWireGuidanceActions } from './maintenance-gu mntRevealed=result;mntRevealError=null;renderMntInspector(); }).catch(function(){ if(seq!==mntInspectorSeq||placementId!==MNT.plc)return; - mntRevealError="The exact path could not be revealed. Refresh evidence and try again.";renderMntInspector(); + mntRevealError="The exact path could not be revealed. Refresh and try again.";renderMntInspector(); }); } diff --git a/src/lib/dashboard/client/maintenance-operation.mjs b/src/lib/dashboard/client/maintenance-operation.mjs index 0c9d6cb2..1ca3c0ba 100644 --- a/src/lib/dashboard/client/maintenance-operation.mjs +++ b/src/lib/dashboard/client/maintenance-operation.mjs @@ -1,11 +1,11 @@ // @ts-nocheck — dashboard browser bundle. import { MNT, mntGet, mntRefreshActiveDestination } from './maintenance-workspace.mjs'; -import { loadSystem } from './system-projects.mjs'; -import { SYSTEM, systemBusy } from './system-readout.mjs'; +import { SYSTEM } from './system-readout.mjs'; +import { refreshRunning } from './refresh-control.mjs'; var mntOperationTimer=null,mntOperationWired=false; var MNT_SCAN_PHASES={install:'Reading installed tools',storage:'Measuring retained data',catalog:'Comparing skills, plugins, and MCP servers',projects:'Measuring projects',consumers:'Ranking disk use',persist:'Saving report'}; -export function mntWritesBlocked(){return !!(MNT.providersBusy||MNT.remeasureBusy||MNT.externalScanBusy);} +export function mntWritesBlocked(){return !!(refreshRunning()||MNT.externalScanBusy);} export function mntOperationText(){return MNT.operation&&MNT.operation.message||'No measurement in progress.';} function mntElapsed(started){ var seconds=Math.max(0,Math.floor((Date.now()-started)/1000)); @@ -13,7 +13,6 @@ function mntElapsed(started){ } export function renderMntOperation(){ var op=MNT.operation,busy=mntWritesBlocked(); - ['mnt-check-providers','mnt-remeasure'].forEach(function(id){var button=document.getElementById(id);if(button)button.disabled=busy;}); var el=document.getElementById('mnt-check-providers-status'); if(el){ var message=op?op.message:''; @@ -31,7 +30,7 @@ function mntSetOperation(message){ MNT.operation.message=message;renderMntOperation(); } function mntBeginOperation(kind){ - MNT.operation={kind:kind,startedAt:Date.now(),message:kind==='measure'?'Preparing measurement…':'Refreshing evidence…',failed:false}; + MNT.operation={kind:kind,startedAt:Date.now(),message:kind==='measure'?'Preparing measurement…':'Refreshing…',failed:false}; if(mntOperationTimer)clearInterval(mntOperationTimer); mntOperationTimer=setInterval(renderMntOperation,1000); renderMntOperation();mntRefreshActiveDestination(); @@ -41,16 +40,6 @@ function mntSetMeasurement(scan){ mntSetOperation(phase+(scan.total?' · '+scan.scanned+' of '+scan.total:'')); } function mntDelay(ms){return new Promise(function(resolve){setTimeout(resolve,ms);});} -function mntPollSystemMeasurement(){ - // Publish completion through System so its scheduled poll cannot replay stale running state. - return loadSystem().then(function(){ - var data=SYSTEM; - if(!data||data.error||!data.scan)throw new Error('The measurement status could not be read.'); - if(data.scan.running){mntSetMeasurement(data.scan);return mntDelay(3000).then(mntPollSystemMeasurement);} - if(data.scan.error)throw new Error('The measurement reported a problem.'); - mntSetOperation('Machine measured · refreshing evidence…'); - }); -} export function mntBuildStatusOf(page){ var refresh=page&&page.lastRefresh; if(refresh&&(refresh.status==='running'||refresh.status==='failed'))return refresh.status; @@ -81,7 +70,7 @@ function mntPollProviders(previousCheck){ if(activity&&activity.status==='failed')throw new Error('The evidence check failed.'); var fresh=previousCheck===undefined||(scan&&scan.checkedAt!==previousCheck); if(activity&&activity.status==='running'||!fresh){ - mntSetOperation('Refreshing evidence…'); + mntSetOperation('Refreshing…'); if(Date.now()-started>300000)throw new Error('The evidence check is still pending.'); return mntDelay(2000).then(tick); } @@ -92,27 +81,6 @@ function mntPollProviders(previousCheck){ });} return tick(); } -function mntRunOperation(measure){ - if(mntWritesBlocked()||(measure&&systemBusy))return Promise.resolve(); - MNT.remeasureBusy=measure;MNT.providersBusy=!measure;mntBeginOperation(measure?'measure':'evidence'); - var previousAt,previousCheck; - return Promise.all([mntGet('/api/maintenance/v2/inventory?limit=1'),mntGet('/api/maintenance')]).then(function(before){ - previousAt=before[0].lastRefresh&&before[0].lastRefresh.at; - previousCheck=before[1].scan&&before[1].scan.checkedAt; - if(measure){if(systemBusy)throw new Error('Another system request is in progress. Please retry.');return loadSystem(true).then(function(){if(!SYSTEM||SYSTEM.error)throw new Error('The measurement could not be started.');return mntPollSystemMeasurement();});} - return mntGet('/api/maintenance?refresh=scan'); - }).then(function(){return mntPollProviders(previousCheck);}) - .then(function(){return mntAwaitInventoryBuild(previousAt);}) - .then(function(){mntSetOperation(MNT.operation.staleMeasurement?'Evidence refreshed · machine measurement is stale. Re-measure machine.':MNT.operation.gaps?'Finished with coverage gaps · see Discovery.':'Inventory updated · checks complete.');}) - .catch(function(error){MNT.operation.failed=true;mntSetOperation(error.message+' Previous inventory remains available.');}) - .then(function(){ - MNT.remeasureBusy=false;MNT.providersBusy=false; - if(mntOperationTimer){clearInterval(mntOperationTimer);mntOperationTimer=null;} - renderMntOperation();mntRefreshActiveDestination(); - }); -} -export function mntRemeasureMachine(){return mntRunOperation(true);} -export function mntCheckProviders(){return mntRunOperation(false);} function mntObserveSystemScan(scan){ if(MNT.remeasureBusy||MNT.providersBusy){ MNT.externalScanBusy=!!(scan&&scan.running); @@ -133,7 +101,7 @@ function mntObserveSystemScan(scan){ Promise.resolve(op.baseline).then(function(before){ if(scan&&scan.error)throw new Error('The measurement reported a problem.'); return mntPollProviders(before&&before.check).then(function(){return mntAwaitInventoryBuild(before&&before.at);}); - }).then(function(){mntSetOperation(op.staleMeasurement?'Evidence refreshed · machine measurement is stale. Re-measure machine.':op.gaps?'Finished with coverage gaps · see Discovery.':'Inventory updated · checks complete.');}) + }).then(function(){mntSetOperation(op.staleMeasurement?'Evidence refreshed · machine measurement is stale. Refresh machine.':op.gaps?'Finished with coverage gaps · see Discovery.':'Inventory updated · checks complete.');}) .catch(function(error){op.failed=true;mntSetOperation(error.message+' Previous inventory remains available.');}) .then(function(){MNT.externalScanBusy=false;clearInterval(mntOperationTimer);mntOperationTimer=null;renderMntOperation();mntRefreshActiveDestination();}); } diff --git a/src/lib/dashboard/client/maintenance-workspace.mjs b/src/lib/dashboard/client/maintenance-workspace.mjs index 1ccfa42b..830f7905 100644 --- a/src/lib/dashboard/client/maintenance-workspace.mjs +++ b/src/lib/dashboard/client/maintenance-workspace.mjs @@ -10,7 +10,7 @@ // a node module — tests/kit/maintenance-dashboard-client-labels.test.mjs // asserts these literal maps stay byte-identical to that contract). import { authHeaders, esc } from './bootstrap.mjs'; -import { mntCheckProviders, mntRemeasureMachine, mntWireOperation } from './maintenance-operation.mjs'; +import { mntWireOperation } from './maintenance-operation.mjs'; import { ago } from './intelligence.mjs'; // ── Label vocabulary (copied from src/lib/maintenance/management/model.mjs) ─ @@ -186,7 +186,7 @@ import { ago } from './intelligence.mjs'; if(lastRefresh&&lastRefresh.status==="running"){ return '
    Building the inventory…
    '; } - return '
    No inventory has been built yet. Use Refresh evidence, above, to build it.
    '; + return '
    No inventory has been built yet. Use Refresh, above, to build it.
    '; } export function mntScanRequiredAnnouncement(lastRefresh){ if(lastRefresh&&lastRefresh.status==="failed"){ @@ -241,8 +241,16 @@ import { ago } from './intelligence.mjs'; }; } + var mntLastSyncedHash=null; export function mntSyncHash(){ - try{if(history.replaceState)history.replaceState(null,"",mntHash());}catch(e){} + var current=String(location.hash||""); + if(current&&!/^#system\/(?:maintenance|catalog)(?:\/|$)/.test(current))return; + if(mntLastSyncedHash!==null&¤t!==mntLastSyncedHash){ + var state=mntApplyHashState(); + if(state&&state.hasState)mntApplyState(state); + } + var next=mntHash(); + try{if(history.replaceState)history.replaceState(null,"",next);mntLastSyncedHash=next;}catch(e){} } // ── Preferences (owner-private; URL state overrides it on load, MNT-PRV-006) ─ @@ -434,8 +442,4 @@ import { ago } from './intelligence.mjs'; MNT.wired=true; mntWireTabs();mntWireOperation(); document.addEventListener("keydown",mntHandleEscape); - var checkProviders=document.getElementById("mnt-check-providers"); - if(checkProviders)checkProviders.addEventListener("click",mntCheckProviders); - var remeasure=document.getElementById("mnt-remeasure"); - if(remeasure)remeasure.addEventListener("click",mntRemeasureMachine); } diff --git a/src/lib/dashboard/client/poll.mjs b/src/lib/dashboard/client/poll.mjs index 9aed4904..1c161dec 100644 --- a/src/lib/dashboard/client/poll.mjs +++ b/src/lib/dashboard/client/poll.mjs @@ -13,7 +13,7 @@ import { loadModelLifecycle, loadUsage } from './usage.mjs'; // Governs EVERY tab, not just Usage (ADR-0009 §7). The old hardcoded 5 s poll // predated any expensive view; 30 s is the default now, and the whole range is // user-chosen and persisted. Every refresh path — automatic or manual — funnels - // through refreshAll(), so the single-flight guard and the cooldown are + // through reloadView(), so the single-flight guard and the cooldown are // impossible to route around. var LS_POLL="ak-dash-poll"; var POLL_DEFAULT_MS=30000; @@ -21,6 +21,7 @@ import { loadModelLifecycle, loadUsage } from './usage.mjs'; var POLL_LABEL={15000:"15s",30000:"30s",60000:"1m",300000:"5m",900000:"15m", 1800000:"30m",3600000:"1h",21600000:"6h",43200000:"12h",86400000:"24h"}; export var pollOn=true, pollMs=POLL_DEFAULT_MS, pollTimer=null, inflight=false, lastAttempt=0; + var lastManualAttempt=0; try{ var savedPoll=JSON.parse(localStorage.getItem(LS_POLL)||"null"); @@ -128,11 +129,17 @@ import { loadModelLifecycle, loadUsage } from './usage.mjs'; }); } - function refreshAll(){ + export function reloadView(force){ + // A deliberate Reload should not be lost to a recent background tick. + // Still coalesce a double-click before joining an in-flight read. + if(force==="manual"){ + if(Date.now()-lastManualAttempt0 + &&Number.isFinite(Date.parse(state.startedAt)) + &&typeof state.running==='boolean'&&Array.isArray(state.stages) + &&(state.running||typeof state.ok==='boolean'); +} +function refreshText(state){ + var stages=state&&state.stages||[]; + var active=stages.find(function(stage){return stage.state==='running';}); + var latest=active||stages[stages.length-1]; + var elapsed=Math.max(0,Math.floor((Date.now()-refreshStartedAt)/1000)); + if(state&&state.running)return (latest&&latest.label||'Preparing refresh')+' · '+elapsed+'s'; + return state&&state.ok?'Refresh complete.':state?'Refresh did not complete.':''; +} +function renderRefresh(state,message){ + var button=document.getElementById('refresh-run'),status=document.getElementById('refresh-status'); + if(button)button.disabled=refreshBusy; + if(status)status.textContent=message||refreshText(state); + document.querySelectorAll('[data-mnt-plan-plc], [data-mnt-undo-receipt], [data-mnt-reconcile-receipt]').forEach(function(action){ + if(refreshBusy){ + if(action.dataset.refreshWasDisabled===undefined)action.dataset.refreshWasDisabled=action.disabled?'1':'0'; + action.disabled=true; + }else if(action.dataset.refreshWasDisabled!==undefined){ + action.disabled=action.dataset.refreshWasDisabled==='1'; + delete action.dataset.refreshWasDisabled; + } + }); + if(refreshRenderedBusy!==refreshBusy){refreshRenderedBusy=refreshBusy;if(refreshBusy)mntRefreshActiveDestination();} +} +function refreshPoll(){ + if(!refreshBusy)return; + fetch('/api/refresh',{cache:'no-store',headers:authHeaders()}).then(function(response){ + if(!response.ok)throw new Error('Refresh status unavailable.'); + return response.json(); + }).then(function(state){ + if(!validRefreshState(state))throw new Error('Refresh status was incomplete.'); + if(refreshOperationId===null&&refreshRequireNewer&&Date.parse(state.startedAt)0;}); + return (project.sessionOrigins||[]).filter(function(row){return row.sessions>0;}); +} +export function surfaceNames(project){ + var entries=surfaceEntries(project); + return Array.from(new Set(entries.map(function(row){return sessionPresentation(row).label;}))).sort(); +} +export function surfaceRawText(origin){ + var raw=origin.rawEvidence||{},parts=[]; + ['entrypoint','originator','source','threadSource','sessionKind'].forEach(function(key){ + var values=Array.isArray(raw[key])?raw[key]:[raw[key]]; + values.slice(0,16).forEach(function(value){ + if(typeof value==='string'&&value.length<=80&&(value==='Codex Desktop'||/^[A-Za-z][A-Za-z0-9_.-]*$/.test(value)))parts.push(key+': '+value); + }); + }); + if(origin.rawEvidenceComplete===false)parts.push('additional raw declarations omitted by bound'); + return parts.join(' · '); +} +export function surfaceDetailHtml(origin){ + var p=sessionPresentation(origin),raw=surfaceRawText(origin); + return esc(p.label)+' · initiator: '+esc(p.initiator)+(p.note?' · '+esc(p.note):'')+(raw?' · '+esc(raw):'')+(Array.isArray(origin.attributes)?' · '+origin.attributes.filter(function(value){return ['on 3P','started from Claude Desktop','started from mobile','started from a project','started from web'].includes(value);}).map(esc).join(', '):''); +} +export function projectSurfacesHtml(project){ + var entries=surfaceEntries(project); + return '
    Session surfaces: '+esc(surfaceNames(project).join(', ')||'Unknown')+'' + +(entries.length?entries.map(function(row){return '
    Host: '+esc(Object.hasOwn(SESSION_HOST_LABELS,row.host)?SESSION_HOST_LABELS[row.host]:'Unknown')+' · '+surfaceDetailHtml(row)+' · provider: '+esc(sessionPresentation(row).provider)+' ('+esc(sessionPresentation(row).providerBasis)+')' + +' · '+esc(row.sessions)+' sessions ('+esc(row.countBasis||'legacy count basis unknown')+')
    ';}).join(''):'
    Host, initiator and provider: Unknown
    ')+'
    '; +} +export function surfaceFacetLabel(value){ + if(value==='unknown')return 'Legacy origin: no declared desktop origin'; + if(value==='surface-unknown')return 'Unknown session surface'; + if(value==='codex-desktop')return sessionPresentation({origin:value}).note; + return Object.hasOwn(SESSION_SURFACE_LABELS,value)?SESSION_SURFACE_LABELS[value]:'Unknown';} diff --git a/src/lib/dashboard/client/system-maintenance-actions.mjs b/src/lib/dashboard/client/system-maintenance-actions.mjs index 24bbbeee..c704dd6d 100644 --- a/src/lib/dashboard/client/system-maintenance-actions.mjs +++ b/src/lib/dashboard/client/system-maintenance-actions.mjs @@ -97,7 +97,7 @@ import { mntAge, mntFact, mntRefreshActiveDestination, mntText, mntValue } from var code=mntText(error&&error.code).toUpperCase(),status=Number(error&&error.status); var effect=mntText(error&&error.effect); if(code==="MAINTENANCE_SCAN_IN_PROGRESS"||code==="SYSTEM_SCAN_IN_PROGRESS")return { - title:code==="SYSTEM_SCAN_IN_PROGRESS"?"Full scan in progress":"Provider check in progress", + title:code==="SYSTEM_SCAN_IN_PROGRESS"?"machine measurement in progress":"Provider check in progress", message:"The server did not start this change. Wait for current evidence to finish, then preview this row again.", status:"The server verified that no mutation started.", }; diff --git a/src/lib/dashboard/client/system-maintenance.mjs b/src/lib/dashboard/client/system-maintenance.mjs index e0ab3073..15752750 100644 --- a/src/lib/dashboard/client/system-maintenance.mjs +++ b/src/lib/dashboard/client/system-maintenance.mjs @@ -16,7 +16,7 @@ import { wireMntDiscovery } from './maintenance-discovery.mjs'; import { wireMntActivity } from './maintenance-activity.mjs'; import { wireMaintActions } from './system-maintenance-actions.mjs'; - // poll.mjs's refreshAll() single-flight-guards its background maintenance + // poll.mjs's reloadView() single-flight-guards its background maintenance // refresh with `!maintenanceBusy` before ever calling loadMaintenance(true) // — the same shape the pre-ADR-0048 single-list workbench used. Keep the // export alive across the ADR-0048 rewrite: true for the duration of this diff --git a/src/lib/dashboard/client/system-projects.mjs b/src/lib/dashboard/client/system-projects.mjs index 5d6206a0..ec77eba4 100644 --- a/src/lib/dashboard/client/system-projects.mjs +++ b/src/lib/dashboard/client/system-projects.mjs @@ -2,6 +2,8 @@ // @ts-nocheck — browser bundle source (never node-imported; client.mjs // reads it as text). See src/lib/dashboard/client/**'s eslint.config.mjs // override comment for why this directory isn't run through the node lib. +import { censusDisclosure } from '../../census-presentation.mjs'; +import { projectSurfacesHtml } from './session-presentation.mjs'; import { authHeaders, esc } from './bootstrap.mjs'; import { formatLocalDateTime, formatLocalDateTimeLong, shortSessionId } from './datetime.mjs'; import { ago } from './intelligence.mjs'; @@ -256,7 +258,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; if(!pm||pm.status==="unknown"||!Array.isArray(pm.value)){ procs.innerHTML=sysEmpty((pm&&pm.reason)||"the process census is unavailable."); }else if(!pm.value.length){ - procs.innerHTML=sysEmpty("no host process is running right now \u2014 a measured zero."); + procs.innerHTML=sysEmpty("no coding-agent or desktop-application process is running right now \u2014 a measured zero."); }else{ var rows=pm.value,maxRss=0,body=""; for(i=0;imaxRss)maxRss=rv;} @@ -271,7 +273,8 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; +esc(source.value.label||source.value.path)+"" : '' +esc(String((source&&source.reason)||"not attributable").split("\u2014")[0].trim())+""; - body+=''+esc(p.host)+"" + body+='' + +esc(p.application||p.host||"Unknown process")+"" +''+esc(String(p.pid))+"" +""+proj+"" +''+mhtml(p.uptimeMs,fmtDur)+"" @@ -282,7 +285,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; } // pid is right-aligned in the body, so its header is too — a numeric // column whose header hangs off the far side reads as a different column. - procs.innerHTML='
    ' + procs.innerHTML='
    Host
    ' +'' +'' +""+body+"
    Coding-agent host / desktop applicationpidWorking contextUptimeCPURSS
    "; @@ -783,7 +786,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; var control=opts.expandable?'':''; var worktreeMark=opts.worktree?'':''; var display=opts.worktree?''+esc(pr.label||'worktree')+'':name; - return ''+worktreeMark+display+''+esc(pr.path||'not measured yet')+'' + return ''+worktreeMark+display+''+esc(pr.path||'not measured yet')+''+projectSurfacesHtml(pr)+'' +''+mhtml(pr.loc&&pr.loc.total,function(v){return "~"+fmtTok(v);})+"" +""+langCell(pr.loc)+"" +''+mhtml(pr.totalBytes,fmtBytes)+"" @@ -801,7 +804,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; +". This view shows "+esc(fmtNum(tree.repositories.length))+" verified repositor"+(tree.repositories.length===1?"y":"ies") +" with "+esc(fmtNum(worktrees))+" nested worktree"+(worktrees===1?"":"s")+"; "+esc(fmtNum(tree.excludedDirectories))+" non-repository directories are excluded." +" Line counts are approximate: extension-bucketed, with node_modules and vendored " - +"trees excluded. Disk is the whole project directory, .git and node_modules included."; + +"trees excluded. Disk is the whole project directory, .git and node_modules included. "+esc(censusDisclosure(p))+""; } export function renderSysProjects(d){ @@ -810,7 +813,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; var p=d.projects; if(!p){el.innerHTML=sysEmpty(NOT_SCANNED);return;} var all=p.projects||[]; - if(!all.length&&!(p.discoveryProjects||[]).length){el.innerHTML=sysEmpty("no repository was discovered on this machine.");return;} + if(!all.length&&!(p.discoveryProjects||[]).length){el.innerHTML=sysEmpty("no repository was discovered on this machine.")+'
    '+esc(censusDisclosure(p))+"
    ";return;} var tree=repositoryTree({projects:all,discoveryProjects:p.discoveryProjects}); var byPath={};tree.repositories.forEach(function(group){byPath[group.repository.path]=group;}); var repositories=sortProjects(tree.repositories.map(function(group){return group.repository;}),projSort.key,projSort.dir); @@ -839,7 +842,7 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; } export function renderSystemFreshness(){ - var el=document.getElementById("sys-asof"),btn=document.getElementById("sys-rescan"), + var el=document.getElementById("sys-asof"), freshness=document.getElementById("system-freshness"); if(!el)return; var scan=(SYSTEM&&SYSTEM.scan)||null,snap=(SYSTEM&&SYSTEM.snapshot)||null; @@ -852,22 +855,20 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; var started=Number(scan.startedAt),seconds=Number.isFinite(started)?Math.max(0,Math.floor((Date.now()-started)/1000)):null; var elapsed=seconds==null?"":seconds<60?seconds+"s":Math.floor(seconds/60)+"m "+seconds%60+"s"; el.classList.add("sy-scan"); - el.textContent="Full scan running \u00b7 "+phase+(scan.total?" "+fmtNum(scan.scanned)+" of "+fmtNum(scan.total):""); + el.textContent="Machine measurement running \u00b7 "+phase+(scan.total?" "+fmtNum(scan.scanned)+" of "+fmtNum(scan.total):""); if(elapsed){var clock=document.createElement("span");clock.setAttribute("aria-hidden","true");clock.textContent=" \u00b7 "+elapsed;el.appendChild(clock);} // The status line already says what is running, how far it has progressed, // and for how long. Repeating that sentence inside a disabled button made // the System rail wider than the viewport precisely when the scan was // active. There is no available action until it settles, so remove the // button from both the visual and accessibility layouts for that state. - if(btn){btn.disabled=true;btn.hidden=true;btn.title="the full scan is already running";} return; } if(freshness)freshness.removeAttribute("data-running"); el.classList.remove("sy-scan"); - if(btn){btn.hidden=false;btn.disabled=false;btn.textContent="\u21bb Full scan";btn.title="re-measure installs, storage, catalog, and projects";} - if(!SYSTEM){el.textContent="full scan \u2014 not loaded";return;} + if(!SYSTEM){el.textContent="machine measurement \u2014 not loaded";return;} if(!snap||!snap.measured||snap.asOf==null){ - el.textContent="full scan \u2014 never run on this machine"; + el.textContent="machine measurement \u2014 never run on this machine"; el.title=(snap&&snap.reason)||"no snapshot has been written yet"; el.setAttribute("data-stale","1"); return; @@ -878,10 +879,10 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; // formatter keeps one vocabulary for "how old is this figure". var age=limAge(Date.now()-Math.max(0,Number(snap.ageMs)||0)); var drift=snap.catalogDrift,changed=drift&&drift.status==="changed"; - el.textContent="full scan \u00b7 "+age+(changed?" \u00b7 catalog changed, scan again":(snap.stale?" \u00b7 stale, scan again":"")) + el.textContent="machine measurement \u00b7 "+age+(changed?" \u00b7 catalog changed, refresh machine":(snap.stale?" \u00b7 stale, refresh machine":"")) +(scan&&scan.error?" \u00b7 last scan reported a problem":""); el.title=(scan&&scan.error?scan.error+" \u2014 ":"") - +"full-scan figures were measured "+age+"; browser refresh does not start a scan"; + +"machine figures were measured "+age+"; Reload does not start a measurement"; if(snap.stale||changed)el.setAttribute("data-stale","1"); } @@ -920,16 +921,14 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; // The trees flag is a SCAN parameter, not a view filter: project trees are // only walked when it is set, so changing it means re-measuring. Undefined // keeps whatever the running configuration already had. - export function loadSystem(deep,trees){ + export function loadSystem(){ if(systemBusy)return Promise.resolve(); systemBusy=true; - if(deep&&SYSTEM&&SYSTEM.scan)SYSTEM.scan.running=true; renderSystemFreshness(); - var q=deep?("?refresh=deep"+(trees==null?"":"&trees="+(trees?"1":"0"))):""; // The slim page read (#237 M4): the same payload with the catalog cut to // what these views draw. /api/system stays the complete `ak system --json` // shape for scripts; the page never needed its repeated presence copies. - return fetch("/api/system/summary"+q,{cache:"no-store",headers:authHeaders()}) + return fetch("/api/system/summary",{cache:"no-store",headers:authHeaders()}) .then(function(r){return r.json();}) .then(function(d){SYSTEM=d;}) .catch(function(){SYSTEM={error:"the system footprint could not be read",scan:null,snapshot:null};}) @@ -1032,11 +1031,6 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; document.addEventListener("keydown",function(e){ if(e.key==="Escape"&&!sessionTooltip.hidden)hideSessionTooltip(); }); - var btn=document.getElementById("sys-rescan"); - if(btn)btn.addEventListener("click",function(){ - if(btn.disabled)return; - loadSystem(true); - }); var ctl=document.getElementById("sys-cons-ctl"); if(ctl)ctl.addEventListener("click",function(e){ var m=e.target.closest?e.target.closest("[data-cons-mode]"):null; @@ -1048,11 +1042,6 @@ import { fmtNum, fmtTok, limAge, pct } from './usage.mjs'; if(SYSTEM)renderSysConsumers(SYSTEM); return; } - var t=e.target.closest?e.target.closest("#sys-cons-trees"):null; - if(!t||t.disabled)return; - // Flipping the scope re-measures; the panel keeps showing the previous - // scan's figures, correctly labelled, until the new one lands. - loadSystem(true,t.getAttribute("aria-pressed")!=="true"); }); var pressure=document.getElementById("sys-pressure"); if(pressure)pressure.addEventListener("click",function(e){ diff --git a/src/lib/dashboard/client/system-readout.mjs b/src/lib/dashboard/client/system-readout.mjs index 3afaf472..e481d01a 100644 --- a/src/lib/dashboard/client/system-readout.mjs +++ b/src/lib/dashboard/client/system-readout.mjs @@ -601,14 +601,10 @@ import { fmtNum, fmtTok } from './usage.mjs'; if(trees){ var on=!!(c&&c.includeProjectTrees); trees.classList.toggle("on",on); - trees.setAttribute("aria-pressed",on?"true":"false"); - // Not a filter: whether trees were walked is decided at scan time, so the - // chip advertises the rescan it will start rather than pretending the - // answer is already on the client. + trees.textContent=on?'Project trees included':'Project trees excluded'; trees.title=on - ?"project working trees are in this ranking \u2014 click to rescan without them" - :"project working trees are excluded \u2014 click to rescan with them (a deep scan takes a while)"; - trees.disabled=!!(d.scan&&d.scan.running); + ?"The last machine measurement included project working trees." + :"The last machine measurement excluded project working trees."; } if(!el)return; if(!c&&!d.storage&&!d.install){ diff --git a/src/lib/dashboard/client/usage-rhythm.mjs b/src/lib/dashboard/client/usage-rhythm.mjs index 49288220..310ff429 100644 --- a/src/lib/dashboard/client/usage-rhythm.mjs +++ b/src/lib/dashboard/client/usage-rhythm.mjs @@ -3,7 +3,7 @@ // override comment for why this directory isn't run through the node lib. // Pure string-building chart primitives for the Usage tab's rhythm/mode -// panels (Task 9 wires these exports into usage.mjs). No DOM access, no +// panels (usage.mjs wires these exports). No DOM access, no // fetch, no module-level state — every function takes plain data in and // returns markup out. `esc` is copied from ./groups.mjs's real implementation // rather than imported: bootstrap.mjs's own `esc` is only a build-time diff --git a/src/lib/dashboard/client/usage.mjs b/src/lib/dashboard/client/usage.mjs index 09d34e48..7c5f2359 100644 --- a/src/lib/dashboard/client/usage.mjs +++ b/src/lib/dashboard/client/usage.mjs @@ -1,6 +1,8 @@ // @ts-nocheck — browser bundle source (never node-imported; client.mjs // reads it as text). See src/lib/dashboard/client/**'s eslint.config.mjs // override comment for why this directory isn't run through the node lib. +import { SESSION_HOST_LABELS, sessionProviderPresentation } from '../../session-surface.mjs'; +import { surfaceDetailHtml } from './session-presentation.mjs'; import { formatLocalDateTime } from './datetime.mjs'; import { VIEWS, authHeaders, esc, setTab, syncHash } from './bootstrap.mjs'; import { ago } from './intelligence.mjs'; @@ -1103,23 +1105,18 @@ import { renderUsage } from './usage-orchestrators.mjs'; +''+esc(sub||resetTxt(resetSec))+""; } - // An empty Claude panel is explained by WHICH statusLine a session runs - // (#238 M3): the tee lives only in the kit footer. The server sends the - // user-level statusLine's class (claudeChannel, never its path); the copy - // states Claude Code's precedence rule, because a project's own statusLine - // overrides the user-level one — which is how a footer-carrying project - // still fills this panel when the user-level script cannot. An unknown or - // missing class (an older server) gets the generic sentence. - var CLAUDE_PRECEDENCE="a project’s own statusLine takes precedence over your user-level one"; + // The channel describes the resolved settings context; local and managed + // settings can override the project and user settings. + var CLAUDE_PRECEDENCE="local or managed settings may override the project and user settings; the effective statusLine takes precedence"; var CLAUDE_SETUP="Set a project up with ak setup --project (ak sync keeps its footer current), then run a Pro/Max session there."; var CLAUDE_EMPTY={ - "kit-footer":"your user-level statusLine carries the kit footer, so limits arrive after the first response " + "kit-footer":"the effective statusLine carries the kit footer, so limits arrive after the first response " +"of a Claude Code session on a Pro/Max plan. Run one session, then revisit.", - "custom":"your user-level statusLine runs a custom script without the kit footer, so it does not report limits " - +"to ak. They arrive only from sessions in projects whose own statusLine carries the footer: "+CLAUDE_PRECEDENCE+". "+CLAUDE_SETUP, - "none":"you have no user-level statusLine, so only sessions in projects whose own statusLine carries the kit footer " + "custom":"the effective statusLine runs a custom script without the kit footer, so it does not report limits " + +"to ak. They arrive from sessions whose effective statusLine carries the footer: "+CLAUDE_PRECEDENCE+". "+CLAUDE_SETUP, + "none":"there is no effective statusLine, so only sessions whose effective statusLine carries the kit footer " +"report limits. "+CLAUDE_SETUP, - "project-helper":"your user-level statusLine runs each project’s own ruflo helper, so limits arrive from sessions " + "project-helper":"the effective statusLine runs each project’s own ruflo helper, so limits arrive from sessions " +"in projects where that helper carries the kit footer. Run ak sync in such a project to re-inject it, " +"then run a Pro/Max session there." }; @@ -1328,7 +1325,7 @@ import { renderUsage } from './usage-orchestrators.mjs'; // not exist, when in fact it was measured and found absent (ADR-0009 §5). function dash(v){return (v==null||v==="")?"—":String(v);} function reportedIdentity(v){v=String(v==null?"":v).trim();return v&&!/^unknown$/i.test(v)?v:null;} - function identityName(v){var raw=reportedIdentity(v);if(!raw)return"Not recorded";return{claude:"Claude Code",codex:"Codex",opencode:"OpenCode",anthropic:"Anthropic",openai:"OpenAI",openrouter:"OpenRouter",bedrock:"AWS Bedrock",vertex:"Google Vertex AI",foundry:"Microsoft Foundry",gateway:"Custom gateway",ollama:"Ollama",lmstudio:"LM Studio"}[raw.toLowerCase()]||raw;} + function identityName(v){return Object.hasOwn(SESSION_HOST_LABELS,v)?SESSION_HOST_LABELS[v]:'Unknown';} // ── per-session chips ───────────────────────────────────────────────────── // Evidence the row already carries, shown only where the transcript @@ -1403,7 +1400,7 @@ import { renderUsage } from './usage-orchestrators.mjs'; ? ' (conf '+esc(sx.confidence.toFixed(2))+")" : ""; var modelList=(Array.isArray(sx.models)?sx.models:[]).filter(function(model){return reportedIdentity(model);}); var models=modelList.length?modelList.join(", "):"Not recorded"; - var providerRaw=reportedIdentity(sx.provider),provider=identityName(providerRaw),provenance=reportedIdentity(sx.providerProvenance)||"unknown",providerContext=providerRaw?provenance+" evidence":"not established by source"; + var providerPresentation=sessionProviderPresentation(sx),provider=providerPresentation.label,providerContext=providerPresentation.basis; var toks="in "+fmtTok(sx.input)+" · out "+fmtTok(sx.output) +" · cache r "+fmtTok(sx.cacheRead)+" / w "+fmtTok(sx.cacheWrite) // Codex-only detail: reasoning tokens are a SUBSET of output (they bill @@ -1431,7 +1428,7 @@ import { renderUsage } from './usage-orchestrators.mjs'; // codex or opencode transcript can record an interrupt, so a claude row // reads "not recorded" rather than a measured-looking 0. +" · aborts "+(sx.host==="codex"||sx.host==="opencode"?fmtNum(Number(sx.aborts)||0):"not recorded for this host"); - var rows=[["execution host",esc(identityName(sx.host))],["inference provider",esc(provider)+" ("+esc(providerContext)+")"],["models",esc(models)],["posture",posture],["rhythm",esc(rhythm)],["basis",esc(basis)+conf],["tokens",esc(toks)], + var rows=[["session surface",surfaceDetailHtml(sx.sessionOrigin||{})],["execution host",esc(identityName(sx.host))],["inference provider",esc(provider)+" ("+esc(providerContext)+")"],["models",esc(models)],["posture",posture],["rhythm",esc(rhythm)],["basis",esc(basis)+conf],["tokens",esc(toks)], ["tools",esc(tools)],["flags",esc(flags)]]; return ' +

    @@ -219,8 +227,7 @@ export function renderPage({ name, version }) {
    - full scan — not run yet - + machine measurement — not run yet
    @@ -916,10 +923,6 @@ ${LIVE_HTML}

    - - Runs provider probes on the saved measurement and rebuilds the inventory. Seconds. - - Walks the filesystem, then refreshes evidence. Minutes.
    diff --git a/src/lib/dashboard/refresh-api.mjs b/src/lib/dashboard/refresh-api.mjs new file mode 100644 index 00000000..f588cf74 --- /dev/null +++ b/src/lib/dashboard/refresh-api.mjs @@ -0,0 +1,97 @@ +import { randomUUID } from 'node:crypto'; +import { REFRESH_STRENGTHS, runRefresh as sharedRunRefresh } from '../refresh.mjs'; +import { sendJson } from '../loopback-server.mjs'; +import { readMaintenanceJson } from './maintenance-security.mjs'; + +export const REFRESH_POST_ROUTES = new Set(['/api/refresh']); +export const REFRESH_LOCAL_TIMEOUT_MS = 5 * 60_000; + +/** One operation per dashboard server. Its public state never exposes stage results. */ +export function createRefreshOperation({ stages, runRefresh = sharedRunRefresh }) { + let current = null; + const state = () => current ? { ...current, stages: current.stages.map(stage => ({ ...stage })) } + : { running: false, lastRun: null }; + function start({ strength, projectTrees = false }) { + if (current?.running) return null; + current = { operationId: randomUUID(), running: true, strength, projectTrees, startedAt: new Date().toISOString(), + finishedAt: null, ok: null, stages: [] }; + Promise.resolve().then(() => runRefresh({ strength, projectTrees, stages, + onStage(event) { + const index = current.stages.findIndex(stage => stage.id === event.id); + if (index < 0) current.stages.push(event); + else current.stages[index] = event; + }, + })).then(outcome => { + current.ok = outcome.ok; + current.stages = outcome.stages.map(({ id, label, state: stageState, detail, elapsedMs }) => + ({ id, label, state: stageState, detail, elapsedMs })); + }).catch(error => { + current.ok = false; + current.stages.push({ id: 'refresh', label: 'Refresh', state: 'failed', + detail: error?.message ?? String(error), elapsedMs: 0 }); + }).finally(() => { current.running = false; current.finishedAt = new Date().toISOString(); }); + return state(); + } + return { start, state }; +} + +/** Dashboard collaborators are already memoized by startDashboard. */ +/** @param {{ cwd: string, pkgRoot?: string, getSystem: Function, getMaintenance: Function, + * refreshInventoryAfterProviderScan: Function, getHostReadiness: Function, + * statusCollect: Function, loadConfig: Function }} options */ +export function dashboardRefreshStages({ cwd, getSystem, getMaintenance, refreshInventoryAfterProviderScan, + getHostReadiness, statusCollect, loadConfig }) { + return { + async machine({ projectTrees }) { + const result = await (await getSystem()).refreshDeep({ includeProjectTrees: projectTrees === true }); + const ok = result?.ok === true && result.persisted?.ok !== false; + return { ok, detail: ok ? null : result?.error ?? 'the measurement did not finish' }; + }, + async maintenance() { + const model = await (await getMaintenance()).scan({ deep: false }); + const { providersChecked, providersTotal } = model?.scan ?? {}; + const detail = Number.isInteger(providersChecked) && Number.isInteger(providersTotal) + ? `checked ${providersChecked} of ${providersTotal} providers` : null; + return { ok: true, detail }; + }, + async inventory({ strength }) { + const result = await refreshInventoryAfterProviderScan({ measured: strength === 'machine' }); + return { ok: result != null, detail: result == null ? 'the inventory rebuild did not finish' : null }; + }, + async live() { + const { runLiveChecks } = await import('../live-checks.mjs'); + const results = await runLiveChecks({ cfg: loadConfig(), cwd }); + const counts = new Map(); + for (const { status } of results) counts.set(status, (counts.get(status) ?? 0) + 1); + return { ok: true, detail: results.length ? [...counts].map(([status, n]) => `${n} ${status}`).join(', ') + : 'no live check applies' }; + }, + async local() { + const status = await statusCollect(); + await getHostReadiness({ force: true }); + return { ok: !status?.error, detail: status?.error ?? null }; + }, + }; +} + +/** Authentication and same-origin policy are enforced by the server gate. */ +export async function handleRefreshPost(req, res, operation) { + let body; + try { body = await readMaintenanceJson(req, { maxBytes: 4096 }); } + catch (error) { + const code = error.status ?? error.statusCode; + sendJson(res, [400, 413, 415].includes(code) ? code : 400, { error: 'invalid refresh request' }); + return; + } + if (!body || Object.keys(body).some(key => !['strength', 'projectTrees'].includes(key)) + || !REFRESH_STRENGTHS.includes(body.strength) + || (body.projectTrees !== undefined && typeof body.projectTrees !== 'boolean') + || (body.projectTrees !== undefined && body.strength !== 'machine')) { + sendJson(res, 400, { error: 'invalid refresh request' }); return; + } + const state = operation.start(body); + if (!state) sendJson(res, 409, { error: 'a refresh is already running', state: operation.state() }); + else sendJson(res, 202, { started: true, state }); +} + +export function handleRefreshGet(res, operation) { sendJson(res, 200, operation.state()); } diff --git a/src/lib/dashboard/session-security.mjs b/src/lib/dashboard/session-security.mjs index 8145c150..299e1844 100644 --- a/src/lib/dashboard/session-security.mjs +++ b/src/lib/dashboard/session-security.mjs @@ -35,7 +35,7 @@ export function resolvesInsideRoot(root, id) { * Claude Code writes every one as `agent-`, or `agent--` * when the Task tool call carried a name (the Agent tool's own `name` * parameter, charset [A-Za-z0-9_-]). A survey of this machine's real - * corpus (404 files, Task 5 round 2) found most stems carry a name — a + * corpus (404 files) found most stems carry a name — a * hex-only pattern would leave most subagent sessions still unopenable. * '/', '.', and '\' are simply not in this charset, so no separate * traversal check is needed for this segment beyond the charset itself. */ diff --git a/src/lib/dashboard/styles/base.mjs b/src/lib/dashboard/styles/base.mjs index b9127c46..71bd47b3 100644 --- a/src/lib/dashboard/styles/base.mjs +++ b/src/lib/dashboard/styles/base.mjs @@ -115,7 +115,7 @@ body.gated .band,body.gated .tabbar,body.gated main{display:none} background:var(--panel); } .verdict-text{font-size:13px; font-weight:500; letter-spacing:-.006em} -.band-tools{display:flex; align-items:center; gap:10px} +.band-tools{display:flex; align-items:center; gap:10px;flex-wrap:wrap;max-width:100%} .pulse{ width:8px; height:8px; border-radius:50%; background:var(--accent); flex:none; animation:pulse 2.4s ease-out infinite; @@ -148,7 +148,15 @@ body.gated .band,body.gated .tabbar,body.gated main{display:none} .poll .play.on{color:var(--accent)} .poll .ivl{min-width:56px; justify-content:space-between} .poll .caret{opacity:.5} -.poll .refresh{width:28px; padding:0; font-size:14px} +.poll .refresh{padding:0 8px; font-size:12px} +.refresh-control{display:flex;align-items:center;gap:5px;min-width:0;flex-wrap:wrap} +.refresh-control button,.refresh-control select{border:1px solid var(--line);border-radius:8px;background:var(--panel);color:var(--ink);font:inherit;font-size:11px;padding:5px 8px} +.refresh-control button{cursor:pointer;font-weight:700} +.refresh-control button:disabled{opacity:.5;cursor:wait} +.refresh-control :focus-visible{outline:2px solid var(--accent);outline-offset:2px} +.refresh-trees{display:flex;align-items:center;gap:3px;font-size:10px;white-space:nowrap} +#refresh-status{font-size:10px;color:var(--ink-2);min-width:0} +@media(max-width:560px){.band-tools{width:100%;gap:6px}.refresh-control{width:100%}.refresh-trees{white-space:normal}} .poll .refresh.spin{animation:spin .6s linear} @keyframes spin{to{transform:rotate(360deg)}} .menu{ diff --git a/src/lib/dashboard/styles/maintenance.mjs b/src/lib/dashboard/styles/maintenance.mjs index 4c5e1c69..8976c3dd 100644 --- a/src/lib/dashboard/styles/maintenance.mjs +++ b/src/lib/dashboard/styles/maintenance.mjs @@ -349,7 +349,6 @@ export const MAINTENANCE_CSS = ` } /* The Maintenance toolbar owns measurement actions in this destination. */ -body:has(#panel-sys-maintenance:not([hidden])) #sys-rescan{display:none} .mnt-project-section{list-style:none;margin:0 0 24px} .mnt-project-section>h4{display:flex;align-items:center;gap:8px;font-size:14px;font-weight:600;margin:8px 0 12px;color:var(--ink)} .mnt-inspector:not([hidden]){animation:mnt-inspector-enter .16s ease-out} diff --git a/src/lib/dashboard/styles/usage.mjs b/src/lib/dashboard/styles/usage.mjs index f0b96990..47e0d22d 100644 --- a/src/lib/dashboard/styles/usage.mjs +++ b/src/lib/dashboard/styles/usage.mjs @@ -173,6 +173,7 @@ export const USAGE_CSS = ` width:32px; height:32px; display:grid; place-items:center; border-radius:50%; background:var(--bg); border:1px solid var(--line); } +.tabbar .source-pill .live-host[data-host=codex]{color:var(--ink)} .tabbar .source-pill .live-host-icon{width:20px; height:20px; stroke-width:1.8} .source-pill .sp-status{ display:flex; align-items:center; padding:5px 14px 5px 10px; font-size:13px; diff --git a/src/lib/dashboard/system-summary.mjs b/src/lib/dashboard/system-summary.mjs index 21c51a53..fcf05584 100644 --- a/src/lib/dashboard/system-summary.mjs +++ b/src/lib/dashboard/system-summary.mjs @@ -1,3 +1,4 @@ +import { mergeSessionSurfaces } from '../footprint/session-surfaces.mjs'; // The System page's slim read model (#237 M4, decision 8). GET /api/system // serves the footprint collector's payload // verbatim, the same shape as `ak system --json`, and that stays the @@ -204,7 +205,7 @@ function summaryInstall(install) { // A measured project row also carries `stack` (framework/manifest/dependency // detection — explicitly not rendered, per system-projects.mjs's langCell // comment), `nodeModulesRoots`, `treeBytes`/`gitBytes`/`nodeModulesBytes`, -// `footprintMtime`, `sessionOrigins`, `source` and `treeExclusions`; the +// `footprintMtime`, `source` and `treeExclusions`; the // Projects table (renderSysProjects) and repository grouping // (project-groups.mjs's repositoryTree) read only what is listed below. const SUMMARY_PROJECT_LANG_KEYS = Object.freeze(['id', 'name', 'lines']); @@ -249,6 +250,12 @@ function summaryRemote(remote) { return out; } +function summarySessionOrigins(entries) { + return Array.isArray(entries) ? entries.filter((row) => row && ['claude-desktop', 'codex-desktop', 'unknown'].includes(row.origin) + && Number.isInteger(row.sessions) && row.sessions >= 0).slice(0, 3).map((row) => ({ origin: row.origin, sessions: row.sessions, + ...(['declared-session-ids', 'transcript-files', 'database-sessions', 'recovered-project-sighting', 'mixed-observations'].includes(row.countBasis) ? { countBasis: row.countBasis } : {}) })) : null; +} + const SUMMARY_PROJECT_ROW_KEYS = Object.freeze(['path', 'label', 'hosts', 'totalBytes', 'lastActivity']); function summaryProjectRow(row) { @@ -258,12 +265,14 @@ function summaryProjectRow(row) { if ('loc' in row) out.loc = summaryProjectLoc(row.loc); if ('remote' in row) out.remote = summaryRemote(row.remote); if ('repository' in row) out.repository = summaryRepository(row.repository); + if ('sessionOrigins' in row) out.sessionOrigins = summarySessionOrigins(row.sessionOrigins); + if (Array.isArray(row.sessionSurfaces)) out.sessionSurfaces = mergeSessionSurfaces(row.sessionSurfaces); return out; } /** A discovery-only row (project-sources.mjs): `origins`, `exists`, - * `isGitRepo`, `lastSeenMs`, `sessions` and `sessionOrigins` never render — - * repositoryTree reads only path/label/hosts/repository off it. */ + * `isGitRepo`, `lastSeenMs` and total `sessions` never render. + * Surface and legacy origin evidence are retained for the local disclosure. */ const SUMMARY_DISCOVERY_PROJECT_KEYS = Object.freeze(['path', 'label', 'hosts']); function summaryDiscoveryProject(row) { @@ -271,11 +280,13 @@ function summaryDiscoveryProject(row) { const out = {}; for (const key of SUMMARY_DISCOVERY_PROJECT_KEYS) if (key in row) out[key] = row[key]; if ('repository' in row) out.repository = summaryRepository(row.repository); + if ('sessionOrigins' in row) out.sessionOrigins = summarySessionOrigins(row.sessionOrigins); + if (Array.isArray(row.sessionSurfaces)) out.sessionSurfaces = mergeSessionSurfaces(row.sessionSurfaces); return out; } const SUMMARY_PROJECTS_KEYS = Object.freeze([ - 'everSeen', 'onDisk', 'count', 'gitRepos', 'unresolved', 'importedExcluded', 'method', 'truncated', + 'everSeen', 'onDisk', 'count', 'gitRepos', 'unresolved', 'importedExcluded', 'importedMixed', 'importedUnresolved', 'method', 'truncated', ]); /** `sources`, `locMeasured`, `registryVersion`, `unrecognized`, `population`, diff --git a/src/lib/exec.mjs b/src/lib/exec.mjs index 6726a0f5..9a30aa6a 100644 --- a/src/lib/exec.mjs +++ b/src/lib/exec.mjs @@ -1,5 +1,5 @@ // Subprocess helpers. Rule (binding, from the plan): NOTHING goes through a -// shell string — execFile with argv arrays only, shell ALWAYS false. +// shell string — spawn with argv arrays only, shell ALWAYS false. // // npm/npx/claude/deja/ruflo/aqe/claude-flow are .cmd shims on Windows, and // Windows' CreateProcess cannot launch a .cmd directly — that historically @@ -7,26 +7,23 @@ // command line to cmd.exe as ONE string (CVE-class: any arg with `&`/`|`/`^` // breaks out into a second command). The actual fix is resolving the shim to // its real file on PATH. Native .com/.exe files run directly. A .cmd shim is -// never passed to execFile: Node does not execute batch files without a shell. -// Instead, its sibling .ps1 shim runs through Windows PowerShell's `-File` -// interface, preserving every caller argument as a separate argv element. -import { execFile, spawn } from 'node:child_process'; +// never passed to spawn: Node does not execute batch files without a shell. +// An exact recognized npm wrapper pair launches its declared public bin with +// Node directly, preserving interactive stdio and literal argv. Other wrappers +// keep their sibling .ps1 through Windows PowerShell's `-File` interface. +import { spawn } from 'node:child_process'; import { AsyncLocalStorage } from 'node:async_hooks'; -import { promisify } from 'node:util'; import fs from 'node:fs'; import path from 'node:path'; import { isWindows } from './paths.mjs'; +import { npmShimInvocation, windowsEnvValue, mergeWindowsEnv } from './windows-npm-shim.mjs'; -const pexecFile = promisify(execFile); const MAX_EXEC_BUFFER = 16 * 1024 * 1024; // A caller that bounds work it does not own (a live check under `ak status // --refresh=live` running the full live-check suite) scopes an AbortSignal // here; every run() inside the scope that passes no signal of its own uses it, -// so a timed-out check's direct child processes stop and its own cleanup still -// runs. The abort signals only that direct child: a process it started keeps -// running (on Windows a .cmd shim's child is PowerShell, so the ruflo or aqe -// node process behind it survives the abort). +// so a timed-out check's entire owned process tree stops and cleanup still runs. const abortScope = new AsyncLocalStorage(); /** Run `fn` with `signal` as the default abort signal for run() calls in it. @@ -41,18 +38,20 @@ const CMD_SHIMS = new Set([ /** Build a shell-free invocation for `cmd`, trying Windows' shim extensions in * PATHEXT order. A native executable is launched directly; a .cmd shim is - * accepted only when its sibling .ps1 and system PowerShell both exist. + * mapped to its own public Node bin only for an exact recognized npm wrapper + * pair; other .cmd wrappers require a sibling .ps1 and system PowerShell. * Falls back to the bare name with resolved:false when no safe target exists. * Exported: the execution adapters spawn these CLIs directly (subprocess.mjs * for claude/codex, opencode.mjs for the serve child, x/ruflo-mcp.mjs for the * Ruflo MCP launcher) and must share the same * resolution `run()`/`have()` use, or readiness passes but launch ENOENTs on - * Windows (swarm review, #88). */ -export function resolveShim(cmd, args = [], { windows = isWindows, env = process.env } = {}) { + * Windows (swarm review, #88). `npmBin:false` retains the PowerShell boundary + * for native diagnostics. */ +export function resolveShim(cmd, args = [], { windows = isWindows, env = process.env, npmBin = true } = {}) { const direct = { command: cmd, args: [...args], resolved: !windows }; if (!windows) return direct; - const systemRoot = env.SystemRoot || env.WINDIR; + const systemRoot = windowsEnvValue(env, 'SystemRoot') || windowsEnvValue(env, 'WINDIR'); const powershell = systemRoot ? path.join(systemRoot, 'System32', 'WindowsPowerShell', 'v1.0', 'powershell.exe') : null; @@ -64,7 +63,10 @@ export function resolveShim(cmd, args = [], { windows = isWindows, env = process if (ext === '.com' || ext === '.exe' || !ext) { return { command: candidate, args: [...args], resolved: true }; } - if (ext !== '.cmd' || !powershell) return null; + if (ext !== '.cmd') return null; + const npm = npmBin ? npmShimInvocation(candidate, args, env) : null; + if (npm) return npm; + if (!powershell) return null; const script = `${candidate.slice(0, -ext.length)}.ps1`; try { if (!fs.statSync(script).isFile() || !fs.statSync(powershell).isFile()) return null; @@ -80,15 +82,21 @@ export function resolveShim(cmd, args = [], { windows = isWindows, env = process }; if (path.isAbsolute(cmd)) return invocationFor(cmd) ?? direct; - const exts = (env.PATHEXT || '.COM;.EXE;.BAT;.CMD') + const exts = (windowsEnvValue(env, 'PATHEXT') || '.COM;.EXE;.BAT;.CMD') .split(';') .map((ext) => ext.trim()) .filter(Boolean); - for (const dir of (env.PATH || env.Path || '').split(path.delimiter)) { + for (const dir of (windowsEnvValue(env, 'PATH') || '').split(path.delimiter)) { if (!dir) continue; for (const ext of exts) { - const invocation = invocationFor(path.join(dir, cmd + ext.toLowerCase())); + const candidate = path.join(dir, cmd + ext.toLowerCase()); + const invocation = invocationFor(candidate); if (invocation) return invocation; + // An earlier custom/unusable .cmd still selects this installation. Do + // not silently switch to a later package because its shim is recognized. + if (ext.toLowerCase() === '.cmd') { + try { if (fs.statSync(candidate).isFile()) return direct; } catch { /* absent */ } + } } } return direct; @@ -142,13 +150,18 @@ export function killProcessTree(child, { platform = process.platform, spawnFn = /** Accumulate one child stream, capped at `maxBuffer` — `execFile` applies its * own cap internally, so the spawn-based path below has to reimplement it. */ function captureStream(stream, encoding, maxBuffer, onOverflow) { - const state = { text: '', overflowed: false }; - stream?.setEncoding?.(encoding); + const chunks = []; + const state = { + bytes: 0, + overflowed: false, + get text() { return Buffer.concat(chunks).toString(encoding); }, + }; stream?.on('data', (chunk) => { if (state.overflowed) return; - state.text += chunk; - if (state.text.length > maxBuffer) { - state.text = state.text.slice(0, maxBuffer); + const remaining = maxBuffer - state.bytes; + chunks.push(chunk.subarray(0, remaining)); + state.bytes += chunk.length; + if (state.bytes > maxBuffer) { state.overflowed = true; onOverflow(); } @@ -156,63 +169,121 @@ function captureStream(stream, encoding, maxBuffer, onOverflow) { return state; } -/** The stdin-feeding, process-group variant of `run()`, used whenever a caller - * passes `opts.input`. Two reasons it exists, both from the security review: - * the payload stays out of argv (SEC-7 — `ps -ww` shows argv to the same user - * here, and `/proc//cmdline` shows it to ANY local user on Linux), and - * the child leads its own process group so the timeout reaps its subprocesses - * (SEC-8). - * - * WHY `spawn` AND NOT `execFile`: execFile forwards only a fixed whitelist of - * options through to spawn, and `detached` is not on it — passing it there is - * silently ignored, and the child stays in the PARENT's process group, where - * `process.kill(-pid)` fails ESRCH and the grandchild survives. Measured, not - * assumed. The timeout is likewise managed here rather than handed to the - * child process API, whose own `timeout` signals only the direct child. */ -function runWithInput(command, args, execOpts, { windows, input }) { +/** `spawn` is required for both paths: execFile silently drops `detached`, so + * its AbortSignal and timeout can only stop the direct child. Input stays on + * stdin and never enters argv. The Windows taskkill completion is awaited + * before returning, even if the direct child closes first. */ +function runOwned(command, args, execOpts, { windows, input }) { const { timeout, encoding, maxBuffer, cwd, env, signal } = execOpts; + if (signal?.aborted) return Promise.resolve({ code: 1, stdout: '', stderr: 'The operation was aborted' }); return new Promise((resolve) => { let child; try { - child = spawn(command, args, { cwd, env, signal, shell: false, detached: !windows }); + child = spawn(command, args, { cwd, env, shell: false, detached: !windows }); } catch (err) { resolve(failureResult(err)); return; } let failure = null; - const abort = (reason) => { failure ??= reason; killGroup(child); }; + let closed = false; + let stopping = Promise.resolve(); + const closePipes = () => { + child.stdout?.destroy(); + child.stderr?.destroy(); + child.stdin?.destroy(); + }; + const incomplete = (reason) => { + failure = `${reason}; incomplete process-tree cleanup`; + closePipes(); + }; + const abort = (reason) => { + if (closed) return; + if (failure) return; + failure = reason; + if (!windows) { killGroup(child); return; } + // A reaped root PID may already name a different process. The owned + // descendant may still hold our pipes, but cannot be safely found by PID. + if (child.exitCode !== null || child.signalCode !== null) { + incomplete(reason); + return; + } + if (!Number.isInteger(child.pid) || child.pid < 1) { + incomplete(reason); + return; + } + stopping = new Promise((done) => { + let killer; + const fallback = (detail) => { + incomplete(`${reason} (${detail})`); + if (child.exitCode === null && child.signalCode === null) { + try { child.kill('SIGKILL'); } catch { /* exited */ } + } + done(); + }; + try { + killer = spawn('taskkill.exe', ['/PID', String(child.pid), '/T', '/F'], { + stdio: 'ignore', shell: false, + }); + } catch { fallback('taskkill could not start'); return; } + const finish = (code, detail) => { + clearTimeout(deadline); + killer.removeListener('error', onError); + killer.removeListener('close', onClose); + if (code === 0) done(); + else fallback(detail); + }; + const onError = () => finish(null, 'taskkill could not start'); + const onClose = (code) => finish(code, `taskkill exited ${code}`); + const deadline = setTimeout(() => { + try { killer.kill('SIGKILL'); } catch { /* already exited */ } + killer.unref?.(); + finish(null, 'taskkill exceeded cleanup deadline'); + }, 1_000); + killer.once('error', onError); + killer.once('close', onClose); + }); + }; + const onSignal = () => abort('The operation was aborted'); + signal?.addEventListener('abort', onSignal, { once: true }); + if (signal?.aborted) onSignal(); const out = captureStream(child.stdout, encoding, maxBuffer, () => abort('stdout maxBuffer length exceeded')); const errOut = captureStream(child.stderr, encoding, maxBuffer, () => abort('stderr maxBuffer length exceeded')); - const timer = setTimeout(() => abort(`timed out after ${timeout}ms`), timeout); - timer.unref?.(); + const timer = timeout > 0 ? setTimeout(() => abort(`timed out after ${timeout}ms`), timeout) : null; + timer?.unref?.(); child.on('error', (err) => { clearTimeout(timer); - resolve(failureResult(err, out.text, errOut.text)); + signal?.removeEventListener('abort', onSignal); + stopping.then(() => resolve(failureResult(err, out.text, errOut.text))); }); - child.on('close', (code) => { + child.on('close', (code, exitSignal) => { + closed = true; clearTimeout(timer); + signal?.removeEventListener('abort', onSignal); const exitCode = typeof code === 'number' ? code : 1; - resolve({ + stopping.then(() => resolve({ code: failure ? exitCode || 1 : exitCode, stdout: out.text, - stderr: failure ? errOut.text || failure : errOut.text, - }); + stderr: failure ? [errOut.text, failure].filter(Boolean).join('\n') : errOut.text, + ...(exitSignal ? { signal: exitSignal } : {}), + })); }); // A child that exits without reading its input makes this write fail with // EPIPE. That is the child's own non-zero exit to report, not a crash here. child.stdin?.on('error', () => {}); - child.stdin?.end(input); + if (typeof input === 'string') child.stdin?.end(input); + else child.stdin?.end(); }); } /** Run a command; never throws. Returns {code, stdout, stderr}. * `opts.input` (a string) is delivered on the child's stdin instead of argv; - * see `runWithInput` for the two guarantees that carries. */ + * `runOwned` keeps that payload out of the process table. */ export async function run(cmd, args = [], opts = {}) { try { - const env = opts.env ? { ...process.env, ...opts.env } : process.env; const windows = opts.windows ?? isWindows; + const env = opts.env + ? (windows ? mergeWindowsEnv(process.env, opts.env) : { ...process.env, ...opts.env }) : process.env; const invocation = CMD_SHIMS.has(cmd) ? resolveShim(cmd, args, { windows, env }) : { command: cmd, args }; @@ -229,13 +300,9 @@ export async function run(cmd, args = [], opts = {}) { env, shell: false, }; - if (typeof opts.input === 'string') { - return await runWithInput(invocation.command, invocation.args, execOpts, { - windows, input: opts.input, - }); - } - const { stdout, stderr } = await pexecFile(invocation.command, invocation.args, execOpts); - return { code: 0, stdout, stderr }; + return await runOwned(invocation.command, invocation.args, execOpts, { + windows, input: opts.input, + }); } catch (err) { return failureResult(err); } diff --git a/src/lib/file-identity.mjs b/src/lib/file-identity.mjs new file mode 100644 index 00000000..b2f764c3 --- /dev/null +++ b/src/lib/file-identity.mjs @@ -0,0 +1,18 @@ +// Persisted file IDs are decimal strings. A legacy Number is comparable only +// when it was a safe integer; an unsafe Number has already lost information. +export function fileId(value) { + if (typeof value === 'bigint') return value >= 0n ? value.toString() : null; + if (typeof value === 'number') return Number.isSafeInteger(value) && value >= 0 ? String(value) : null; + if (typeof value === 'string' && /^(?:0|[1-9]\d*)$/.test(value)) return value; + return null; +} + +export function sameFileId(left, right) { + const a = fileId(left); + return a !== null && a === fileId(right); +} + +export function statMtimeMs(stat) { + if (typeof stat.mtimeNs !== 'bigint') return stat.mtimeMs; + return Number(stat.mtimeNs / 1_000_000n) + Number(stat.mtimeNs % 1_000_000n) / 1_000_000; +} diff --git a/src/lib/footprint/codex-import-discovery.mjs b/src/lib/footprint/codex-import-discovery.mjs new file mode 100644 index 00000000..41b46566 --- /dev/null +++ b/src/lib/footprint/codex-import-discovery.mjs @@ -0,0 +1,80 @@ +// Bounded metadata inspection for import-marked heads. No content is retained. +import { newCodexTurnOwnership, codexTurnOwner } from '../codex-import-marker.mjs'; +import { codexReplayPlan, isCodexReplayLine } from '../codex-replay.mjs'; + +export const IMPORT_TAIL_BYTES = 2 * 1024 * 1024; +export const IMPORT_SCAN_BYTES = 512 * 1024 * 1024; +const MAX_RECORDS = 20_000; + +function recordsIn(buffer, { dropFirst, dropLast }) { + const lines = buffer.toString('utf8').split('\n'); + if (dropFirst) lines.shift(); + if (dropLast) lines.pop(); + const records = []; + let complete = true; + for (const line of lines) { + if (!line.trim()) continue; + if (records.length >= MAX_RECORDS) { complete = false; break; } + try { + const record = JSON.parse(line); + if (!record || typeof record !== 'object' || Array.isArray(record)) throw new Error('invalid envelope'); + records.push(record); + } catch { complete = false; records.push(null); } + } + return { records, complete }; +} + +/** Read at most headBytes + 2 MiB per candidate, within a shared scan budget. + * Missing middle bytes reset ownership; adjacency never spans an unread gap. + * A mixed result proves some native activity, not that every turn was seen. */ +export function inspectCodexImport(file, { fsImpl, headBytes, budget }) { + let fd; + try { + fd = fsImpl.openSync(file, 'r'); + const size = fsImpl.fstatSync(fd).size; + const total = Math.min(size, headBytes + IMPORT_TAIL_BYTES); + if (budget.remaining < total) return { kind: 'unresolved' }; + budget.remaining -= total; + const contiguous = size <= total; + const windows = contiguous ? [[0, size]] : [[0, headBytes], [size - IMPORT_TAIL_BYTES, IMPORT_TAIL_BYTES]]; + const groups = []; + let complete = contiguous; + for (const [offset, length] of windows) { + const b = Buffer.alloc(length); + const n = fsImpl.readSync(fd, b, 0, length, offset); + const group = recordsIn(b.subarray(0, n), { dropFirst: offset > 0, dropLast: offset + n < size }); + complete = complete && n === length && group.complete; + groups.push(group.records); + } + return classifyWindows(groups, complete); + } catch { return { kind: 'unresolved' }; } + finally { if (fd !== undefined) { try { fsImpl.closeSync(fd); } catch { /* failed observation */ } } } +} + +function classifyWindows(groups, complete) { + const first = groups[0].find((r) => r?.type === 'session_meta'); + // A sampled subagent replay boundary cannot safely use the fallback that + // depends on absence of an addressed parent message in the entire file. + if (!complete && first?.payload?.thread_source === 'subagent') return { kind: 'unresolved' }; + const replay = codexReplayPlan(groups.flat()); + if (replay.unprovable) return { kind: 'unresolved' }; + let state = newCodexTurnOwnership(); + let genuine = false; + let ambiguous = false; + let cwd = null; + for (const [index, records] of groups.entries()) { + if (index) state = newCodexTurnOwnership(); + for (const record of records) { + if (record === null) return { kind: 'unresolved', complete: false }; + const before = state.nativeRecords; + const owner = codexTurnOwner(state, record); + if (!state.ownershipComplete) return { kind: 'unresolved', complete: false }; + if (owner !== 'native' || isCodexReplayLine(replay.boundary, record)) continue; + if (record?.type === 'turn_context' && !cwd && typeof record.payload?.cwd === 'string') cwd = record.payload.cwd; + if (state.nativeRecords > before) genuine = true; + } + ambiguous ||= state.ambiguousRecords > 0; + } + return genuine ? { kind: 'mixed', cwd, complete } + : { kind: complete && !ambiguous ? 'imported' : 'unresolved', complete }; +} diff --git a/src/lib/footprint/consumers.mjs b/src/lib/footprint/consumers.mjs index 41345a2a..8b5c0081 100644 --- a/src/lib/footprint/consumers.mjs +++ b/src/lib/footprint/consumers.mjs @@ -55,7 +55,7 @@ import fs from 'node:fs'; import path from 'node:path'; import { - claudeDir, codexDir, configDir, globalRoot, home, isWindows, npxCacheDir, + claudeDir, codexDir, configDir, globalRoot, home, isWindows, npxCacheDir, xdgBase, } from '../paths.mjs'; import { hasValue, measured, rootMeasurements, sumMeasurements, unknown, walkTree, @@ -167,8 +167,8 @@ export const CONSUMER_WALK_LIMITS = Object.freeze({ // to audit for the sake of a read-only ranking. Kit and host paths still come // from paths.mjs — nothing home-relative that the kit itself owns is spelled out // below. -const xdgCache = (env) => env.XDG_CACHE_HOME || path.join(home, '.cache'); -const xdgData = (env) => env.XDG_DATA_HOME || path.join(home, '.local', 'share'); +const xdgCache = (env) => xdgBase('XDG_CACHE_HOME', path.join(home, '.cache'), { env }); +const xdgData = (env) => xdgBase('XDG_DATA_HOME', path.join(home, '.local', 'share'), { env }); const macCache = () => path.join(home, 'Library', 'Caches'); const winLocalAppData = (env) => env.LOCALAPPDATA || path.join(home, 'AppData', 'Local'); diff --git a/src/lib/footprint/index.mjs b/src/lib/footprint/index.mjs index 863a2a1d..35daee7e 100644 --- a/src/lib/footprint/index.mjs +++ b/src/lib/footprint/index.mjs @@ -28,7 +28,7 @@ import fs from 'node:fs'; import path from 'node:path'; import { - claudeDir, claudeSettingsPath, claudeUserMcpPath, codexConfigPath, codexDir, configDir, home, + claudeDir, claudeSettingsPath, claudeUserMcpPath, codexConfigPath, codexDir, configDir, stateBase, } from '../paths.mjs'; import { loadKitConfig } from '../config.mjs'; import { defaultOpencodeDbPath } from '../usage-opencode.mjs'; @@ -65,8 +65,8 @@ export const INCLUDE_PROJECT_TREES_DEFAULT = false; * point: these are the files that grow fastest between deep scans (ledgers, * tee files, index caches), and a user watching one grow should not have to * run a deep scan to see it move. */ -function knownFileSpecs() { - const stateRoot = process.env.XDG_STATE_HOME || path.join(home, '.local', 'state'); +export function knownFileSpecs() { + const stateRoot = stateBase(); const kit = (name) => path.join(configDir(), name); // [id, host, category, label, path] — the categories are STORAGE_CATEGORIES' // vocabulary so a known file and its deep-tier node land in the same bucket. diff --git a/src/lib/footprint/install.mjs b/src/lib/footprint/install.mjs index a83414c1..f4c96e03 100644 --- a/src/lib/footprint/install.mjs +++ b/src/lib/footprint/install.mjs @@ -26,7 +26,7 @@ import path from 'node:path'; import { MANAGED_COMPANION_REGISTRY } from '../adapters/companion-registry.mjs'; import { HOST_REGISTRY } from '../adapters/registries.mjs'; import { - home, isWindows, globalRoot, npxCacheDir, claudeDir, codexPluginCacheDir, + home, isWindows, globalRoot, npxCacheDir, claudeDir, codexPluginCacheDir, xdgBase, } from '../paths.mjs'; import { installedVersion, KIT_PKG } from '../versions.mjs'; import { kbDir, present as brainPresent, installedVersion as brainVersion } from '../ruvnet-brain.mjs'; @@ -580,7 +580,7 @@ function vibiumCachePath({ env, platform }) { if (platform === 'win32') { return path.join(env.LOCALAPPDATA || path.join(home, 'AppData', 'Local'), 'vibium'); } - return path.join(env.XDG_CACHE_HOME || path.join(home, '.cache'), 'vibium'); + return path.join(xdgBase('XDG_CACHE_HOME', path.join(home, '.cache'), { env }), 'vibium'); } const presentFile = (file, fsImpl) => { diff --git a/src/lib/footprint/project-sources.mjs b/src/lib/footprint/project-sources.mjs index be8e1421..60ba198c 100644 --- a/src/lib/footprint/project-sources.mjs +++ b/src/lib/footprint/project-sources.mjs @@ -18,28 +18,27 @@ // // Content boundary. This is DISCOVERY, invariant 9's candidate-path source, not // a measurement: it reads a session's cwd and explicit launch-origin declaration -// from the bounded head. The same read `native-transcript-discovery.mjs` already +// from bounded head metadata (and bounded import continuation windows). The read `native-transcript-discovery.mjs` already // performs for Observability at the same trust boundary. No message, prompt or // tool payload is retained or emitted; every figure the System area // renders is measured downstream by walk.mjs-backed collectors from the paths // this module returns. // -// Cost. The corpus here is ~3,200 transcripts. Each file is opened once and -// only its HEAD is read (HEAD_BYTES, JSON-parsed up to HEAD_MAX_LINES non-blank -// lines) — a session's cwd is recorded in its opening records or not at all, so -// reading further would cost the whole corpus to learn nothing. A file that -// cannot be read or parsed is counted and skipped; one bad transcript never -// aborts the walk (invariant 6). +// Cost. Ordinary files use HEAD_BYTES and HEAD_MAX_LINES. Import-marked +// heads additionally use codex-import-discovery's bounded head/tail windows, +// record cap and shared byte budget. Missing ranges remain explicitly unknown. import fs from 'node:fs'; import path from 'node:path'; import { claudeDir, codexDir } from '../paths.mjs'; import { resolveProjectLabel } from '../live/index.mjs'; import { withDb } from '../sqlite.mjs'; -import { defaultOpencodeDbPath } from '../usage-opencode.mjs'; +import { selectOpencodeSource } from '../usage-opencode.mjs'; import { presenceOf, statNode, UNKNOWN, walkTree } from './walk.mjs'; import { inspectProjectIdentity } from './project-identity.mjs'; +import { mergeSessionSurfaces, sessionSurfaceSighting } from './session-surfaces.mjs'; import { transcriptSessionOrigin } from './session-origin.mjs'; import { isImportedCodexRollout } from '../codex-import-marker.mjs'; +import { inspectCodexImport, IMPORT_SCAN_BYTES } from './codex-import-discovery.mjs'; /** Hosts in the order every payload lists them. */ export const PROJECT_SOURCE_HOSTS = Object.freeze(['claude', 'codex', 'opencode']); @@ -54,6 +53,11 @@ export const HEAD_MAX_LINES = 40; * `sessions/YYYY/MM/DD/.jsonl`, 4 deep. 8 leaves room for either root * gaining a level without letting an unexpected tree run away. */ const TRANSCRIPT_MAX_DEPTH = 8; +const CLAUDE_NON_CONVERSATION_TYPES = new Set([ + 'bridge-session', 'cost-state', 'file-history-snapshot', 'queue-operation', + 'ai-title', 'atis-latch', 'last-prompt', 'attachment', 'mode', + 'permission-mode', 'agent-name', 'agent-setting', 'system', 'progress', 'summary', +]); /** lstat budget for one encoded-directory decode. The decode is a bounded * search (below), so it needs a ceiling of its own; 512 covers a deep path @@ -63,8 +67,8 @@ const DECODE_STAT_BUDGET = 512; /** The one-line statement of what was counted and how, so no surface can render * these numbers without being able to say where they came from. */ export const PROJECT_SOURCE_METHOD = - 'every cwd named by a Claude or Codex transcript head, plus every OpenCode session ' - + 'directory, de-duplicated by resolved real path — not only projects with ruflo state'; + 'eligible Claude and Codex bounded transcript cwd evidence, plus every OpenCode session ' + + 'directory, de-duplicated by resolved real path; Claude sessions use declared sessionId'; // ── transcript heads ────────────────────────────────────────────────────────── @@ -122,7 +126,7 @@ function firstClaudeSessionMetadata(lines) { let startedAt = null; for (const record of parsedHeadRecords(lines)) { if (!cwd && typeof record.cwd === 'string' && record.cwd) cwd = record.cwd; - if (!nativeId && typeof record.sessionId === 'string' && record.sessionId) nativeId = record.sessionId; + if (!nativeId && typeof record.sessionId === 'string' && /^\S+$/.test(record.sessionId)) nativeId = record.sessionId; if (!startedAt) startedAt = normalizedInstant(record.timestamp); if (cwd && nativeId && startedAt) break; } @@ -232,6 +236,12 @@ function rootStatus(walkResult) { return presence === 'present' ? 'ok' : presence; } +function completeSessionScan(result, unreadable, unknown, imports) { + return result.complete !== false && unreadable === 0 && unknown === 0 && imports === 0; +} + +function hasCodexImports(host, lines) { return host === 'codex' && isImportedCodexRollout(lines); } + /** * Every cwd named by the transcripts under `root`, plus the counts a liner note * needs to state what was and was not recoverable. @@ -269,6 +279,14 @@ export function scanTranscriptCwds(root, host, { unresolved: 0, recoveredFromDirName: 0, importedExcluded: 0, + importedMixed: 0, + importedUnresolved: 0, + sessions: 0, + duplicateSessionFiles: 0, + subagentExcluded: 0, + nonConversationExcluded: 0, + unknownSessionFiles: 0, + sessionCountComplete: status === 'absent' || (status === 'ok' && result.complete !== false), sightings: [], truncated: Boolean(result.truncated), truncatedBy: result.truncatedBy ?? null, @@ -287,29 +305,76 @@ export function scanTranscriptCwds(root, host, { let empty = 0; let unreadable = 0; let importedExcluded = 0; + let importedMixed = 0; + let importedUnresolved = 0; + const importBudget = { remaining: IMPORT_SCAN_BYTES }; + let duplicateSessionFiles = 0; + let subagentExcluded = 0; + let nonConversationExcluded = 0; + let unknownSessionFiles = 0; + const seenClaudeIds = new Set(); + // The walk's order is an implementation detail. File path order makes the + // first declaration the stable owner when an id appears in multiple files. + files.sort((a, b) => a.file.localeCompare(b.file)); for (const { file, mtimeMs } of files) { + const relativeParts = path.relative(root, file).split(path.sep); + if (host === 'claude' && relativeParts.includes('subagents')) { + subagentExcluded += 1; + continue; + } + const lines = readHead(file, { fsImpl, headBytes, maxLines }); + if (lines === null) { unreadable += 1; continue; } + if (lines.length === 0) { empty += 1; continue; } + // Imported turns do not establish project/origin evidence. Look for own + // activity within fixed read budgets before excluding a whole file. + let genuineCwd = null; + if (hasCodexImports(host, lines)) { + const imported = inspectCodexImport(file, { fsImpl, headBytes, budget: importBudget }); + if (imported.kind === 'imported') { importedExcluded++; continue; } + if (imported.kind === 'unresolved') { importedUnresolved++; continue; } + importedMixed++; + genuineCwd = imported.cwd; + } + let weight = 1; + if (host === 'claude') { + const records = [...parsedHeadRecords(lines)]; + const hasConversation = records.some((record) => record.type === 'user' || record.type === 'assistant'); + // A sideband-only head excludes a file only when the read reached EOF. + // A later conversation may exist beyond either bounded limit. + let fullyRead = false; + try { fullyRead = fsImpl.statSync(file).size <= headBytes && lines.length < maxLines; } + catch { /* the head alone does not establish an end-of-file */ } + if (!hasConversation && (fullyRead && records.length > 0 + && records.every((record) => CLAUDE_NON_CONVERSATION_TYPES.has(record.type)))) { + nonConversationExcluded += 1; + continue; + } + const { nativeId } = firstClaudeSessionMetadata(lines); + if (!nativeId || !hasConversation) { + unknownSessionFiles += 1; + weight = 0; + } else if (seenClaudeIds.has(nativeId)) { + duplicateSessionFiles += 1; + continue; + } else { + seenClaudeIds.add(nativeId); + } + } let group = null; if (decodeDir) { - const key = path.relative(root, file).split(path.sep)[0]; + const key = relativeParts[0]; group = groups.get(key); if (!group) { group = { key, withCwd: false, newestMtimeMs: null }; groups.set(key, group); } if (Number.isFinite(mtimeMs) && (group.newestMtimeMs === null || mtimeMs > group.newestMtimeMs)) { group.newestMtimeMs = mtimeMs; } } - const lines = readHead(file, { fsImpl, headBytes, maxLines }); - if (lines === null) { unreadable += 1; continue; } - if (lines.length === 0) { empty += 1; continue; } - // An imported copy of a Claude Code transcript is not a Codex session: it - // names the folder the Claude session ran in and declares the ChatGPT - // desktop app as originator. It gives no project, host or origin and is - // counted, never dropped silently (ADR-0052 §3, ADR-0060 §3). - if (host === 'codex' && isImportedCodexRollout(lines)) { importedExcluded += 1; continue; } - const cwd = firstCwd(lines, host); + const cwd = genuineCwd || firstCwd(lines, host); if (!cwd) { withoutCwd += 1; continue; } withCwd += 1; - sightings.push({ cwd, mtimeMs, origin: 'cwd', sessionOrigin: transcriptSessionOrigin(lines, host) }); + sightings.push({ cwd, mtimeMs, origin: 'cwd', weight, + sessionOrigin: transcriptSessionOrigin(lines, host) }); if (group) group.withCwd = true; } @@ -320,7 +385,7 @@ export function scanTranscriptCwds(root, host, { const decoded = decodeDir(group.key); if (decoded) { recoveredFromDirName += 1; - sightings.push({ cwd: decoded, mtimeMs: group.newestMtimeMs, origin: 'encoded-dir' }); + sightings.push({ cwd: decoded, mtimeMs: group.newestMtimeMs, origin: 'encoded-dir', weight: 0 }); } else { unresolved += 1; } @@ -336,10 +401,18 @@ export function scanTranscriptCwds(root, host, { unresolved, recoveredFromDirName, importedExcluded, + importedMixed, + importedUnresolved, + sessions: host === 'claude' ? seenClaudeIds.size : undefined, + duplicateSessionFiles, + subagentExcluded, + nonConversationExcluded, + unknownSessionFiles, + sessionCountComplete: completeSessionScan(result, unreadable, unknownSessionFiles, importedUnresolved), sightings, // Transcripts we could not read, and project directories whose path could // not be recovered, both mean the project list is a floor. - complete: result.complete !== false && unreadable === 0 && unresolved === 0, + complete: completeSessionScan(result, unreadable, unresolved, importedUnresolved), }; } @@ -350,12 +423,14 @@ export function scanTranscriptCwds(root, host, { * from there rather than guessed at. The store is opened READ-ONLY and only * that one column is selected. * - * An absent store means OpenCode was never used on this machine (a real zero); + * An absent selected store contributes no observed projects; * a store that will not open, or an older schema without `directory`, is * reported degraded with its reason rather than silently contributing nothing. */ -export function scanOpencodeDirectories({ dbFile = defaultOpencodeDbPath(), withDb: withDbImpl = withDb } = {}) { +export function scanOpencodeDirectories({ dbFile = undefined, selection = selectOpencodeSource(dbFile === undefined ? {} : { roots: { opencode: dbFile } }), withDb: withDbImpl = withDb } = {}) { + dbFile = selection.dbFile; const base = { host: 'opencode', root: dbFile, status: 'ok', reason: null, sessions: 0, sightings: [], complete: true }; + if (!dbFile) return { ...base, ...selection.health, complete: selection.health.status === 'absent' }; const result = withDbImpl(dbFile, (db) => db.prepare( 'SELECT directory, COUNT(*) AS sessions, MAX(COALESCE(time_updated, time_created)) AS lastMs' + " FROM session WHERE directory IS NOT NULL AND directory <> '' GROUP BY directory", @@ -365,7 +440,7 @@ export function scanOpencodeDirectories({ dbFile = defaultOpencodeDbPath(), with return { ...base, status: absent ? 'absent' : 'degraded', - reason: absent ? null : (result.error?.message ?? 'store unreadable'), + reason: absent ? null : (result.error?.kind ?? 'store-unreadable'), complete: absent, }; } @@ -428,7 +503,8 @@ function gitPresence(projectPath, fsImpl) { * exists: boolean, isGitRepo: boolean, lastSeenMs: number|null, * sessions: number }>, * everSeen: number, onDisk: number, gitRepos: number, unresolved: number, - * importedExcluded: number, complete: boolean, method: string, + * importedExcluded: number, importedMixed: number, importedUnresolved: number, + * complete: boolean, method: string, * sources: Record<'claude'|'codex'|'opencode', object>, * }} `everSeen` counts projects INCLUDING vanished ones; `onDisk` counts the * measurable subset. `complete: false` means at least one transcript or @@ -439,7 +515,7 @@ function gitPresence(projectPath, fsImpl) { export function discoverProjectSources({ claudeRoot = path.join(claudeDir(), 'projects'), codexRoot = path.join(codexDir(), 'sessions'), - opencodeDbFile = defaultOpencodeDbPath(), + opencodeDbFile, walk = walkTree, fsImpl = fs, now = Date.now, @@ -469,12 +545,13 @@ export function discoverProjectSources({ const resolved = resolvePath(cwd, fsImpl); let row = byPath.get(resolved); if (!row) { - row = { path: resolved, hosts: new Set(), origins: new Set(), sessionOrigins: new Map(), sessions: 0, lastSeenMs: null }; + row = { path: resolved, hosts: new Set(), origins: new Set(), sessionOrigins: new Map(), sessionSurfaces: [], sessions: 0, lastSeenMs: null }; byPath.set(resolved, row); } row.hosts.add(host); row.origins.add(sighting.origin ?? 'cwd'); - const weight = Number.isFinite(sighting.weight) ? sighting.weight : 1; + const weight = sighting.origin === 'encoded-dir' ? 0 + : Number.isFinite(sighting.weight) ? sighting.weight : 1; row.sessions += weight; const declared = sighting.sessionOrigin; const origin = ['claude-desktop', 'codex-desktop'].includes(declared?.origin) @@ -485,8 +562,9 @@ export function discoverProjectSources({ row.sessionOrigins.set(origin, membership); } membership.sessions += weight; + row.sessionSurfaces.push(sessionSurfaceSighting(sighting, host, weight)); membership.countBases.add(sighting.origin === 'encoded-dir' ? 'recovered-project-sighting' - : host === 'opencode' ? 'database-sessions' : 'transcript-files'); + : host === 'opencode' ? 'database-sessions' : host === 'claude' ? 'declared-session-ids' : 'transcript-files'); membership.evidence.add(origin === 'unknown' ? 'desktop-origin-not-declared' : declared.evidence); const at = sighting.mtimeMs; if (Number.isFinite(at) && (row.lastSeenMs === null || at > row.lastSeenMs)) row.lastSeenMs = at; @@ -508,6 +586,7 @@ export function discoverProjectSources({ isGitRepo: exists && gitPresence(row.path, fsImpl), lastSeenMs: row.lastSeenMs, sessions: row.sessions, + sessionSurfaces: mergeSessionSurfaces(row.sessionSurfaces), sessionOrigins: [...row.sessionOrigins.values()].map((entry) => ({ origin: entry.origin, sessions: entry.sessions, evidence: [...entry.evidence].filter(Boolean).sort(), countBasis: entry.countBases.size === 1 ? [...entry.countBases][0] : 'mixed-observations', @@ -531,6 +610,8 @@ export function discoverProjectSources({ gitRepos: projects.filter((project) => project.isGitRepo).length, unresolved, importedExcluded, + importedMixed: sources.codex?.importedMixed ?? 0, + importedUnresolved: sources.codex?.importedUnresolved ?? 0, complete: PROJECT_SOURCE_HOSTS.every((host) => sources[host]?.complete !== false), method: PROJECT_SOURCE_METHOD, sources, diff --git a/src/lib/footprint/projects.mjs b/src/lib/footprint/projects.mjs index 87213cc6..145c16d4 100644 --- a/src/lib/footprint/projects.mjs +++ b/src/lib/footprint/projects.mjs @@ -417,6 +417,7 @@ function missingProject(project, reason, presence = 'absent') { source: project.source ?? null, repository: project.repository ?? null, sessionOrigins: project.sessionOrigins ?? null, + sessionSurfaces: project.sessionSurfaces ?? null, hosts: Array.isArray(project.hosts) ? [...project.hosts] : null, remote: { status: 'unknown', name: null, raw: null, hostname: null, host: null, slug: null, webUrl: null, reason }, loc: locNotMeasured(reason), @@ -449,7 +450,7 @@ function notify(onProgress, payload) { * attribution (which hosts saw this project), never as a measurement. * * @param {{ path: string, label: string, source?: string, hosts?: string[], remote?: object, - * repository?: object, sessionOrigins?: Array<{origin:string,sessions:number,evidence?:string[]}> }} project + * repository?: object, sessionSurfaces?: object[], sessionOrigins?: Array<{origin:string,sessions:number,evidence?:string[]}> }} project * @param {{ walk?: Function, limits?: object, detect?: Function, loc?: boolean, * asOf?: number|null, fsImpl?: typeof fs }} [options] * `loc: false` skips the stack pass entirely — the expensive part of a project @@ -567,6 +568,7 @@ export function measureProject(project, { source: project.source ?? null, repository: project.repository ?? null, sessionOrigins: project.sessionOrigins ?? null, + sessionSurfaces: project.sessionSurfaces ?? null, hosts: Array.isArray(project.hosts) ? [...project.hosts] : null, // collectProjects preflights the remote to choose the stated hosted-repo // population. Reuse that exact evidence instead of opening .git/config a @@ -684,15 +686,15 @@ function aggregateUnrecognized(rows) { }; } -/** The KPI counts a discovery payload carries; absent fields read as zero, so - * an older payload (no `importedExcluded`) keeps rendering as it did. */ +/** Import fields absent from an old discovery payload remain unknown. */ function payloadCounts(payload) { return { everSeen: payload?.everSeen ?? 0, onDisk: payload?.onDisk ?? 0, gitRepos: payload?.gitRepos ?? 0, unresolved: payload?.unresolved ?? 0, - importedExcluded: payload?.importedExcluded ?? 0, + importedExcluded: payload?.importedExcluded ?? null, + importedMixed: payload?.importedMixed ?? null, importedUnresolved: payload?.importedUnresolved ?? null, complete: payload?.complete !== false, method: payload?.method ?? null, sources: payload?.sources ?? null, @@ -804,7 +806,8 @@ function buildProjectsSection({ asOf, out, eligible, selected, excluded, counts, onDisk: kpi(counts?.onDisk ?? 0), gitRepos: kpi(counts?.gitRepos ?? 0), unresolved: counts?.unresolved ?? 0, - importedExcluded: counts?.importedExcluded ?? 0, + importedExcluded: counts?.importedExcluded ?? null, + importedMixed: counts?.importedMixed ?? null, importedUnresolved: counts?.importedUnresolved ?? null, method: counts?.method ?? null, sources: counts?.sources ?? null, scanned: out.length, diff --git a/src/lib/footprint/runtime.mjs b/src/lib/footprint/runtime.mjs index 6747a198..85b0d840 100644 --- a/src/lib/footprint/runtime.mjs +++ b/src/lib/footprint/runtime.mjs @@ -47,7 +47,8 @@ function sourceOf(entry, { platform, classifyContext }) { return measured({ kind: 'host-service', label: `${hostTitle(entry.host)} app service` }); } if (entry.controllerKind === 'desktop-app') { - return measured({ kind: 'desktop-app', label: `${hostTitle(entry.host)} desktop app` }); + return measured({ kind: 'desktop-app', label: entry.application + ?? `${hostTitle(entry.host)} desktop app` }); } if (!entry.cwd) return unknown(cwdDetail(entry.cwdReason ?? 'cwd-unavailable')); const pathImpl = platform === 'win32' ? path.win32 : path; @@ -189,6 +190,7 @@ export async function collectRuntimeCensus({ const source = sourceOf(entry, { platform, classifyContext }); return { host: entry.host, + application: entry.application ?? null, pid: entry.pid, startedAt: entry.startedAt, // Kept alongside `project` so a renderer or a debug log can key on the diff --git a/src/lib/footprint/session-origin.mjs b/src/lib/footprint/session-origin.mjs index 064a9353..06128aa5 100644 --- a/src/lib/footprint/session-origin.mjs +++ b/src/lib/footprint/session-origin.mjs @@ -1,7 +1,20 @@ -// Explicit host declarations only. The same Desktop origin can belong to any -// Git repository/worktree/folder. SDK, app-server and vscode are ambiguous. -const CLAUDE_DESKTOP = new Set(['claude-desktop', 'claude-desktop-3p', 'remote_desktop']); -const CODEX_DESKTOP = new Set(['Codex Desktop', 'codex_work_desktop']); +import { classifySessionSurface } from '../session-surface.mjs'; + +// Compatibility origin is retained for existing project and usage consumers. +// It only names a local desktop application; remote_desktop is a cloud session. +const desktopOrigin = (surface) => surface === 'claude-desktop' ? 'claude-desktop' + : surface === 'chatgpt-desktop-codex' || surface === 'chatgpt-desktop-work' ? 'codex-desktop' : 'unknown'; + +function adapted(classification, evidence) { + const origin = desktopOrigin(classification.surface); + const legacy = { origin, evidence: origin === 'unknown' ? 'desktop-origin-not-declared' : evidence }; + // Accessors expose the new dimensions without changing the serialized legacy + // sessionOrigin shape (or the usage cache) before parser integration. + for (const [key, value] of Object.entries(classification)) { + Object.defineProperty(legacy, key, { value, enumerable: false }); + } + return legacy; +} /** Classify one bounded transcript head; never retain arbitrary metadata. */ export function transcriptSessionOrigin(lines, host) { @@ -10,16 +23,15 @@ export function transcriptSessionOrigin(lines, host) { try { record = JSON.parse(line); } catch { continue; } if (!record || typeof record !== 'object') continue; if (host === 'codex' && record.type === 'session_meta') { - const value = record.payload?.originator; - return CODEX_DESKTOP.has(value) - ? { origin: 'codex-desktop', evidence: `session_meta.originator:${value}` } - : { origin: 'unknown', evidence: 'desktop-origin-not-declared' }; + const payload = record.payload ?? {}; + return adapted(classifySessionSurface({ host, originator: payload.originator, + source: payload.source, threadSource: payload.thread_source }), + `session_meta.originator:${payload.originator}`); } if (host === 'claude' && typeof record.entrypoint === 'string') { - return CLAUDE_DESKTOP.has(record.entrypoint) - ? { origin: 'claude-desktop', evidence: `entrypoint:${record.entrypoint}` } - : { origin: 'unknown', evidence: 'desktop-origin-not-declared' }; + return adapted(classifySessionSurface({ host, entrypoint: record.entrypoint, + sessionKind: record.sessionKind }), `entrypoint:${record.entrypoint}`); } } - return { origin: 'unknown', evidence: 'desktop-origin-not-declared' }; + return adapted(classifySessionSurface({ host }), 'desktop-origin-not-declared'); } diff --git a/src/lib/footprint/session-surfaces.mjs b/src/lib/footprint/session-surfaces.mjs new file mode 100644 index 00000000..1be219ce --- /dev/null +++ b/src/lib/footprint/session-surfaces.mjs @@ -0,0 +1,46 @@ +// Additive project presentation evidence. Never replaces legacy origin counts. +import { sessionPresentation } from '../session-surface.mjs'; +const RAW_FIELDS = ['entrypoint', 'originator', 'source', 'threadSource', 'sessionKind']; +const COUNT_BASES = ['declared-session-ids', 'transcript-files', 'database-sessions', 'recovered-project-sighting', 'mixed-observations']; +const token = (value) => typeof value === 'string' && value.length <= 80 + && (value === 'Codex Desktop' || /^[A-Za-z][A-Za-z0-9_.-]*$/u.test(value)); + +/** Merge aggregates without letting the last transcript overwrite earlier evidence. */ +export function mergeSessionSurfaces(entries = []) { + const groups = new Map(); + for (const entry of entries) { + if (!entry || !Number.isInteger(entry.sessions) || entry.sessions < 0) continue; + const presentation = sessionPresentation(entry); + const host = ['claude', 'codex', 'opencode'].includes(entry.host) ? entry.host : 'unknown'; + const initiator = ['person', 'automation', 'agent', 'imported-copy'].includes(entry.initiator) ? entry.initiator : 'unknown'; + const provider = presentation.provider === 'Unknown' ? null : entry.thirdPartyProvider; + const key = `${host}:${presentation.surface}:${initiator}:${provider}`; + if (!groups.has(key)) groups.set(key, { host, surface: presentation.surface, initiator, + thirdPartyProvider: provider, thirdPartyProviderBasis: provider ? 'assistant-model-id' : null, attributes: [], + sessions: 0, countBasis: entry.countBasis, rawEvidence: {}, rawEvidenceComplete: true }); + const row = groups.get(key); + row.sessions += entry.sessions; + row.attributes = [...new Set([...row.attributes, ...(Array.isArray(entry.attributes) ? entry.attributes : []).filter((value) => + ['on 3P', 'started from Claude Desktop', 'started from mobile', 'started from a project', 'started from web'].includes(value))])].sort(); + if (row.countBasis !== entry.countBasis || !COUNT_BASES.includes(row.countBasis)) row.countBasis = 'mixed-observations'; + row.rawEvidenceComplete &&= entry.rawEvidenceComplete !== false; + for (const field of RAW_FIELDS) { + const raw = entry.rawEvidence?.[field]; + const values = (Array.isArray(raw) ? raw : [raw]).filter(token); + const union = [...new Set([...(row.rawEvidence[field] ?? []), ...values])].sort(); + if (union.length) row.rawEvidence[field] = union.slice(0, 16); + if (union.length > 16) row.rawEvidenceComplete = false; + } + } + return [...groups.values()].sort((a, b) => `${a.host}:${a.surface}:${a.initiator}:${a.thirdPartyProvider}`.localeCompare(`${b.host}:${b.surface}:${b.initiator}:${b.thirdPartyProvider}`)); +} + + +export function sessionSurfaceSighting(sighting, host, weight) { + const declared = sighting.sessionOrigin ?? {}; + return { host, surface: declared.surface, initiator: declared.initiator, rawEvidence: declared.rawEvidence, + attributes: declared.attributes, sessions: weight, thirdPartyProvider: declared.thirdPartyProvider, + thirdPartyProviderBasis: declared.thirdPartyProviderBasis, + countBasis: sighting.origin === 'encoded-dir' ? 'recovered-project-sighting' + : host === 'opencode' ? 'database-sessions' : host === 'claude' ? 'declared-session-ids' : 'transcript-files' }; +} diff --git a/src/lib/footprint/storage-reclaim-detectors.mjs b/src/lib/footprint/storage-reclaim-detectors.mjs index e162425a..97da3b89 100644 --- a/src/lib/footprint/storage-reclaim-detectors.mjs +++ b/src/lib/footprint/storage-reclaim-detectors.mjs @@ -16,15 +16,15 @@ // process.platform, so the wrong-platform root simply reads absent and a machine // carrying both (a tool that moved its cache) reports both. import path from 'node:path'; -import { home, isWindows } from '../paths.mjs'; +import { home, isWindows, xdgBase } from '../paths.mjs'; import { decodeClaudeProjectDir } from './project-sources.mjs'; import { rootMeasurements, measured, unknown, statNode, sumMeasurements, hasValue, } from './walk.mjs'; import { candidate } from './storage-reclaim.mjs'; -const xdgCache = (env) => env.XDG_CACHE_HOME || path.join(home, '.cache'); -const xdgData = (env) => env.XDG_DATA_HOME || path.join(home, '.local', 'share'); +const xdgCache = (env) => xdgBase('XDG_CACHE_HOME', path.join(home, '.cache'), { env }); +const xdgData = (env) => xdgBase('XDG_DATA_HOME', path.join(home, '.local', 'share'), { env }); const macCache = () => path.join(home, 'Library', 'Caches'); const winLocalAppData = (env) => env.LOCALAPPDATA || path.join(home, 'AppData', 'Local'); diff --git a/src/lib/footprint/storage.mjs b/src/lib/footprint/storage.mjs index 3b62e2c3..db519ecd 100644 --- a/src/lib/footprint/storage.mjs +++ b/src/lib/footprint/storage.mjs @@ -52,7 +52,7 @@ // `detectWorktrees` is false. import fs from 'node:fs'; import path from 'node:path'; -import { home, claudeDir, codexDir, configDir } from '../paths.mjs'; +import { home, claudeDir, codexDir, configDir, stateBase } from '../paths.mjs'; import { defaultOpencodeDbPath } from '../usage-opencode.mjs'; import { decodeClaudeProjectDir, transcriptMetadata } from './project-sources.mjs'; import { classifyWorkingContext } from './working-context.mjs'; @@ -118,8 +118,9 @@ const flatDir = () => true; * * @returns {StorageRoot[]} */ -export function defaultStorageRoots({ env = process.env, projects = null } = {}) { - const stateRoot = env.XDG_STATE_HOME || path.join(home, '.local', 'state'); +export function defaultStorageRoots({ env = process.env, projects = null, + home: h = home, platform = process.platform, p = path } = {}) { + const stateRoot = stateBase({ env, home: h, platform, p }); const opencodeData = path.dirname(defaultOpencodeDbPath()); const claude = (name) => path.join(claudeDir(), name); const codex = (name) => path.join(codexDir(), name); @@ -183,7 +184,7 @@ export function defaultStorageRoots({ env = process.env, projects = null } = {}) { id: 'ak-runtime-debug', category: 'ledgers-and-logs', host: 'agentic-kit', label: 'runtime-debug.log', - path: path.join(stateRoot, 'agentic-kit', 'runtime-debug.log'), layout: 'tree', + path: p.join(stateRoot, 'agentic-kit', 'runtime-debug.log'), layout: 'tree', }, { id: 'ak-config', category: 'kit-caches', host: 'agentic-kit', diff --git a/src/lib/hook-audit/agentic-dependency-constraints.json b/src/lib/hook-audit/agentic-dependency-constraints.json index 8f51ef94..d1310d2a 100644 --- a/src/lib/hook-audit/agentic-dependency-constraints.json +++ b/src/lib/hook-audit/agentic-dependency-constraints.json @@ -287,7 +287,7 @@ "refs": ["closed-upstream review 2026-09-26"], "files": ["docs/adr/0034-schema-native-handoffs-and-hermetic-seats.md", "src/lib/execution/codex.mjs"] }, - "adjustment": "None: closed 2026-04-03 as model behaviour without a fix; parseHandoffText is the validation wrapper the maintainer recommended. Re-probe --output-schema with MCP on Codex upgrades (last probed on 0.149.1).", + "adjustment": "None: closed 2026-04-03 as model behaviour without a fix; parseHandoffText is the validation wrapper the maintainer recommended.", "status": "retired", "constraintIds": [], "history": [ @@ -501,14 +501,15 @@ "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, "mapping": "mapped", "kitImpact": { "refs": ["lane C: live-lock busy rule", "pacphi/agentic-kit#240"], "files": ["docs/host-support.md"] }, - "adjustment": "agentic-qe 3.14.4 reports the real LockHeld instead of FsyncFailed after a live lock (Windows confirmation 2026-09-28). Re-run the live-owner fixture on macOS and Linux against 3.14.4; if no FsyncFailed follows the live-lock warning, remove the temporary busy rule in src/lib/aqe-readiness.mjs and close pacphi/agentic-kit#240.", - "status": "watching", + "adjustment": "Released agentic-qe 3.14.4 passed disposable live-owner conformance on macOS and Linux: status and the shipped adapter emitted LockHeld without FsyncFailed, and RVF/lock bytes were unchanged. ak retired the exact FsyncFailed-as-busy exception; old 3.14.3 output now fails closed. Ordinary live-lock SQLite fallback remains busy. This is a verified artifact baseline for this rule, not a universal AQE minimum or native Windows conformance. Upstream issue state remains separate from this kit retirement.", + "status": "retired", "constraintIds": [], "history": [ { "date": "2026-09-26", "event": "commented" }, { "date": "2026-09-26", "event": "registered" }, { "date": "2026-09-27", "event": "commented", "note": "#719 fixed FsyncFailed; a small fall-through remains (issuecomment-5863219357)" }, - { "date": "2026-09-28", "event": "commented", "note": "thanked fedor-drmanovic for the Windows 3.14.4 results; ak re-runs its live-owner fixture on macOS and Linux next (issuecomment-5878041570)" } + { "date": "2026-09-28", "event": "commented", "note": "thanked fedor-drmanovic for the Windows 3.14.4 results; ak re-runs its live-owner fixture on macOS and Linux next (issuecomment-5878041570)" }, + { "date": "2026-09-29", "event": "retired", "note": "ak retired its exact FsyncFailed-as-busy exception after released 3.14.4 passed native macOS and Linux live-owner probes; ordinary LockHeld fallback remains busy; no native Windows AQE proof or upstream closure claimed" } ] }, { @@ -681,7 +682,7 @@ "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": "3.14.1" } }, "mapping": "mapped", "kitImpact": { "refs": ["constraint agentic-qe-3.14.0-codex-guidance-policy"], "files": [] }, - "adjustment": "Keep constraint agentic-qe-3.14.0-codex-guidance-policy: the 2026-09-27 conformance on 3.14.4 failed. On each Agentic-QE release run AK_AQE_CONFORMANCE=1 node --test tests/live/aqe-codex-guidance-conformance.test.mjs; sunset the constraint when it passes without its todo.", + "adjustment": "Keep constraint agentic-qe-3.14.0-codex-guidance-policy. The released 3.14.5 artifact passed the selected compact owned-byte and foreign-content preservation check, but full/none modes, all receipt and platform checks, and the complete ak conformance remain unverified. Run the existing opt-in full conformance on a future released artifact before any sunset.", "status": "released", "constraintIds": ["agentic-qe-3.14.0-codex-guidance-policy"], "history": [ @@ -689,7 +690,8 @@ { "date": "2026-09-06", "event": "closed", "note": "completed" }, { "date": "2026-09-26", "event": "registered" }, { "date": "2026-09-26", "event": "released", "note": "fix first released in 3.14.1; the ak change is pending" }, - { "date": "2026-09-27", "event": "reviewed", "note": "Branch 5 conformance on ak's path (full aqe init --auto --with-codex --codex-guidance full|compact|none through the aqe command, agentic-qe 3.14.4, sandboxed): FAIL, constraint kept. full: no block when AGENTS.md exists and platform verify reports detected=none; compact and none select correctly and are idempotent, but compact adds 6 bytes outside its sentinel; through the aqe bin resolvePackageRoot() misses the package, so no Codex hooks or skills install; platform verify exits 0 on failed checks; platform setup codex: Module not found in bundle. Evidence: Branch 5 report b5-655-conformance; upstream issue drafted, not filed" } + { "date": "2026-09-27", "event": "reviewed", "note": "Branch 5 conformance on ak's path (full aqe init --auto --with-codex --codex-guidance full|compact|none through the aqe command, agentic-qe 3.14.4, sandboxed): FAIL, constraint kept. full: no block when AGENTS.md exists and platform verify reports detected=none; compact and none select correctly and are idempotent, but compact adds 6 bytes outside its sentinel; through the aqe bin resolvePackageRoot() misses the package, so no Codex hooks or skills install; platform verify exits 0 on failed checks; platform setup codex: Module not found in bundle. Evidence: Branch 5 report b5-655-conformance; upstream issue drafted, not filed" }, + { "date": "2026-09-29", "event": "reviewed", "note": "Private source-bound agentic-qe 3.14.5 tarball proof: selected compact AGENTS sentinel is 315 bytes; separate foreign prefix/suffix survives two same-option public init calls with bytes and mtime stable between calls 1/2. Full/none, complete receipts and all-platform conformance unverified; constraint retained. Evidence: docs/archive/2026-09-29-aqe-released-artifact-receipt.md." } ] }, { @@ -749,7 +751,7 @@ { "date": "2026-09-26", "event": "registered" }, { "date": "2026-09-27", "event": "retired", "note": "merged and released in 3.14.4; context only: agentic-qe#574, tracked by pacphi/agentic-kit#240, drives removing the busy rule" } ], - "note": "Context only: a partial fix for agentic-qe#574 (rethrows LockHeld). Its release alone does not prove a live lock no longer surfaces FsyncFailed, so agentic-qe#574 drives removing the busy rule and closing pacphi/agentic-kit#240." + "note": "Context only: a partial fix for agentic-qe#574 (rethrows LockHeld), released in 3.14.4. Native macOS and Linux conformance against that artifact later showed no FsyncFailed under a live lock; ak retired the exact exception on 2026-09-29. Native Windows AQE conformance and pacphi/agentic-kit#240 closure remain separate." }, { "id": "proffesor-for-testing/agentic-qe#734", @@ -1128,7 +1130,8 @@ { "date": "2026-09-26", "event": "commented", "note": "tested the single-threaded ONNX Runtime mitigation on Apple Silicon: inconclusive, the abort does not reproduce locally (issuecomment-5850382245)" }, { "date": "2026-09-27", "event": "commented", "note": "hosted probe run 36333572972 on Ruflo 3.46.1: 10/10 aborts by default and 10/10 with single-threaded ONNX Runtime sessions; mitigation disproven (issuecomment-5857781254)" }, { "date": "2026-09-27", "event": "commented", "note": "3.47.0 status, transformers 3.x / ONNX Runtime 1.21 lead, fix approaches; 11/11 nightly aborts in the continue-on-error step (issuecomment-5863246672)" }, - { "date": "2026-09-28", "event": "commented", "note": "thanked vidaunited for the trace hook; 12 of 12 macOS nightly aborts on 3.47.0, with a new WASM backend line before the abort (issuecomment-5878044230)" } + { "date": "2026-09-28", "event": "commented", "note": "thanked vidaunited for the trace hook; 12 of 12 macOS nightly aborts on 3.47.0, with a new WASM backend line before the abort (issuecomment-5878044230)" }, + { "date": "2026-09-29", "event": "commented", "note": "posted the approved source-bound native trace on #2885 (issuecomment-5891510508); the failure occurred after heal, the outer command exited 1, and its learning step remained nonblocking for the job; the trace does not establish a cause or released fix" } ] }, { @@ -1524,7 +1527,8 @@ "history": [ { "date": "2026-09-24", "event": "filed" }, { "date": "2026-09-26", "event": "registered" }, - { "date": "2026-09-27", "event": "commented", "note": "answered from the Codex CLI source (issuecomment-5863289657)" } + { "date": "2026-09-27", "event": "commented", "note": "answered from the Codex CLI source (issuecomment-5863289657)" }, + { "date": "2026-09-29", "event": "reviewed", "note": "still open with only ak's source-analysis comment; no maintainer-supported guidance or live hook proof, so the conditional Codex fix line remains deferred" } ] }, { @@ -2024,15 +2028,16 @@ "dependency": null, "doneWhen": { "state": "closed-completed", "release": null }, "mapping": "mapped", - "kitImpact": { "refs": ["decision 7", "lane C: live-lock busy rule"], "files": ["src/lib/aqe-readiness.mjs"] }, - "adjustment": "Close #240 when a released agentic-qe no longer falls through to create after a live RVF lock (agentic-qe#574) and ak removes the temporary busy rule. agentic-qe#719 (partial fix: rethrows LockHeld) was released in 3.14.4 on 2026-09-27; whether that release alone clears FsyncFailed under a live lock is not yet verified.", + "kitImpact": { "refs": ["decision 7", "lane C: live-lock busy rule"], "files": ["docs/troubleshooting.md"] }, + "adjustment": "After the final main PR merges, the controller can close #240 with the released 3.14.4 native macOS/Linux live-owner evidence and ak's exact exception retirement for agentic-qe#574. The old 3.14.3 FsyncFailed sequence now fails closed; ordinary LockHeld fallback remains busy. No native Windows AQE conformance or upstream issue closure is claimed.", "status": "watching", "constraintIds": [], "tracks": ["proffesor-for-testing/agentic-qe#574", "proffesor-for-testing/agentic-qe#719"], "history": [ { "date": "2026-09-26", "event": "filed" }, { "date": "2026-09-26", "event": "registered" }, - { "date": "2026-09-27", "event": "commented", "note": "moved the upstream part of this tracker to the watch registry; agentic-qe 3.14.4 carries PR 719, not yet verified to stop FsyncFailed under a live lock (issuecomment-5858567140)" } + { "date": "2026-09-27", "event": "commented", "note": "moved the upstream part of this tracker to the watch registry; agentic-qe 3.14.4 carries PR 719, not yet verified to stop FsyncFailed under a live lock (issuecomment-5858567140)" }, + { "date": "2026-09-29", "event": "reviewed", "note": "released 3.14.4 passed native macOS and Linux live-owner conformance and ak retired the exact exception; issue closure waits for final main PR" } ] }, { @@ -2129,10 +2134,10 @@ "kind": "issue", "title": "`agentic-qe mcp` double-spawns the server (stdio:'inherit') and drops stdin → premature shutdown", "dependency": "agentic-qe", - "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, + "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": "3.14.4" } }, "mapping": "mapped", "kitImpact": { "refs": [], "files": ["docs/host-support.md"] }, - "adjustment": "Once a released agentic-qe mcp stops double-spawning the server, drop the host-support.md risk link.", + "adjustment": "The 3.14.4 MCP entry runs in-process; host-support.md records adoption from that version. Keep watching the still-open upstream issue until its closed-completed condition is met.", "status": "watching", "constraintIds": [], "history": [ @@ -2147,10 +2152,10 @@ "kind": "issue", "title": "init: platform installs are additive, not exclusive — always installs Claude Code surface; need per-platform/only mode", "dependency": "agentic-qe", - "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, + "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": "3.14.4" } }, "mapping": "mapped", "kitImpact": { "refs": [], "files": ["docs/host-support.md"] }, - "adjustment": "Once a released agentic-qe init offers an exclusive per-platform mode, drop the host-support.md risk link.", + "adjustment": "The 3.14.4 aqe init --no-claude option supports exclusive platform initialization; host-support.md records adoption from that version. Keep watching the still-open upstream issue until its closed-completed condition is met.", "status": "watching", "constraintIds": [], "history": [ @@ -2168,7 +2173,7 @@ "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, "mapping": "mapped", "kitImpact": { "refs": [], "files": ["docs/host-support.md"] }, - "adjustment": "Once a released agentic-qe fixes the listed MCP tool bugs, drop the host-support.md risk link.", + "adjustment": "Keep the host-support.md risk link limited to the remaining 3.14.4 GOAP maxSteps/world-state, test-generation quality, and coherence recommendation-text gaps. memory_delete and cross-phase stats were fixed; goap_execute was not re-exercised.", "status": "watching", "constraintIds": [], "history": [ @@ -2620,18 +2625,19 @@ "kind": "issue", "title": "Witness chain forks when two processes append at once. `append()` reads the tail and inserts it without a transaction, so `aqe audit verify --chain audit` reports BROKEN.", "dependency": "agentic-qe", - "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, + "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": "3.14.5" } }, "mapping": "mapped", "kitImpact": {"refs": ["investigation aqe-audit-chain", "Branch 5: stray-store merge runs with no other AQE writer"], "files": ["src/lib/aqe-store-holders.mjs", "src/lib/aqe-store-merge.mjs"]}, - "adjustment": "Once a released agentic-qe appends to the witness chain atomically, ak no longer needs to stop other AQE writers before merging stray stores.", - "status": "fixed-unreleased", + "adjustment": "Released agentic-qe 3.14.5 passed a native macOS fresh WAL chain append check: one sequential and three two-process rounds each yielded 4001 valid rows, with only one concurrent round demonstrably interleaved. Keep ak's live-holder refusal around stray-store backup, import and archive; the probe does not establish old-fork repair, import splice safety, live host behavior, signatures, or Windows/Linux conformance.", + "status": "released", "constraintIds": [], "history": [ { "date": "2026-09-27", "event": "filed" }, { "date": "2026-09-27", "event": "registered" }, { "date": "2026-09-27", "event": "reviewed", "note": "Branch 5: ak x aqe-store merge finds the processes holding the root or a stray store by open file (lsof, /proc; the host-session census on Windows) before its backup, its real import and its archive, and refuses while any holds one; there is no force" }, { "date": "2026-09-27", "event": "commented", "note": "still reproduces on 3.14.4, with impact; refs agentic-qe#759 (issuecomment-5863202746)" }, - { "date": "2026-09-28", "event": "closed", "note": "completed by PR 764; no release contains it yet" } + { "date": "2026-09-28", "event": "closed", "note": "completed by PR 764; no release contains it yet" }, + { "date": "2026-09-29", "event": "released", "note": "npm-integrity-verified 3.14.5 tarball, source-bound native macOS fresh-chain append proof: immediate transaction; 4001 valid rows in sequential control and three synchronized two-process rounds; only round 1 demonstrably interleaved. No old-fork repair, signed-entry, import-splice or live-holder removal claim. Evidence: docs/archive/2026-09-29-aqe-released-artifact-receipt.md." } ] }, { @@ -2702,8 +2708,8 @@ "dependency": "agentic-qe", "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, "mapping": "mapped", - "kitImpact": {"refs": ["ak runs aqe init --with-codex, not platform setup", "the codex-mcp status row's manual fix names aqe platform setup codex (src/commands/status/sections/codex-mcp.mjs)"], "files": []}, - "adjustment": "none: ak does not run platform setup; once fixed, the codex-mcp status row's manual fix (aqe platform setup codex --overwrite --with-ruflo) works as written.", + "kitImpact": {"refs": ["ak runs aqe init --with-codex, not platform setup", "the codex-mcp status row formerly named the broken platform setup command (src/commands/status/sections/codex-mcp.mjs)"], "files": []}, + "adjustment": "The codex-mcp status row now points to AQE's supported aqe init --auto --with-codex --codex-guidance compact flow and notes the separate 3.14.4 Codex asset gap (agentic-qe#755). ak does not run platform setup or write AQE's Codex registration.", "status": "fixed-unreleased", "constraintIds": [], "history": [ @@ -2749,6 +2755,26 @@ { "date": "2026-09-27", "event": "registered" }, { "date": "2026-09-28", "event": "closed", "note": "completed by PR 766; no release contains it yet" } ] + }, + { + "id": "proffesor-for-testing/agentic-qe#778", + "url": "https://github.com/proffesor-for-testing/agentic-qe/issues/778", + "relation": "filed", + "kind": "issue", + "title": "fix: repeated aqe init changes generated settings on an unchanged project", + "dependency": "agentic-qe", + "doneWhen": { "state": "closed-completed", "release": { "channel": "npm", "name": "agentic-qe", "minVersion": null } }, + "mapping": "mapped", + "kitImpact": { "refs": ["ak setup invokes aqe init and can inherit its repeat-run settings churn"], "files": [] }, + "adjustment": "Released AQE 3.14.5 still rewrites .claude/settings.json on all three same-option public init runs of an unchanged minimal project; run 2 also creates a backup and changes domain/learning defaults. AGENTS.md and CLAUDE.md remain stable after run 1. PR #783 merged after the 3.14.5 publication; wait for a later released artifact and repeat exact conformance before claiming setup convergence or setting a minimum fixed version.", + "status": "fixed-unreleased", + "constraintIds": [], + "history": [ + { "date": "2026-09-29", "event": "filed", "note": "approved exact draft posted; remote title and body match receipt" }, + { "date": "2026-09-29", "event": "registered", "note": "3.14.4 disposable project repro: settings changed on all three runs; guidance hashes and mtimes did not" }, + { "date": "2026-09-29", "event": "closed", "note": "upstream completed via PR #783; merge commit 63d3deb dated 13:05:19 UTC, after npm 3.14.5 publication at 10:08:33 UTC" }, + { "date": "2026-09-29", "event": "reviewed", "note": "Source-bound released 3.14.5 disposable public init still churns settings on each of three same-option runs, with backup and default changes on run 2. Subsequent release conformance unverified; global 3.14.4 metadata unchanged. Evidence: docs/archive/2026-09-29-aqe-released-artifact-receipt.md." } + ] } ] } diff --git a/src/lib/hook-audit/providers/opencode.mjs b/src/lib/hook-audit/providers/opencode.mjs index 25cbc560..a7738968 100644 --- a/src/lib/hook-audit/providers/opencode.mjs +++ b/src/lib/hook-audit/providers/opencode.mjs @@ -1,6 +1,7 @@ import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; +import { xdgBase } from '../../paths.mjs'; import { normalizedOccurrence, publicSource, readBoundedFile, readJsonSource, @@ -90,7 +91,7 @@ function moduleRecords(source) { } export function auditOpenCodeHooks({ - opencodeRoot = path.join(process.env.XDG_CONFIG_HOME || path.join(os.homedir(), '.config'), 'opencode'), + opencodeRoot = path.join(xdgBase('XDG_CONFIG_HOME', path.join(os.homedir(), '.config')), 'opencode'), projectRoots = [process.cwd()], opencodeVersion = 'unknown', ownership = null, diff --git a/src/lib/hook-remediation/engine.mjs b/src/lib/hook-remediation/engine.mjs index 767b449e..2bbfa475 100644 --- a/src/lib/hook-remediation/engine.mjs +++ b/src/lib/hook-remediation/engine.mjs @@ -1,5 +1,6 @@ import fs from 'node:fs'; import path from 'node:path'; +import { sameFileId } from '../file-identity.mjs'; import { assertHookHealingPlanIntegrity, buildHookHealingPlan, @@ -27,8 +28,8 @@ function sameOwnerAndParent(snapshot, expected) { && snapshot.gid === (expected.gid ?? snapshot.gid) && snapshot.specialMode === (expected.specialMode ?? 0) && snapshot.parent.realPath === (expected.parent?.realPath ?? snapshot.parent.realPath) - && snapshot.parent.dev === (expected.parent?.dev ?? snapshot.parent.dev) - && snapshot.parent.ino === (expected.parent?.ino ?? snapshot.parent.ino); + && sameFileId(snapshot.parent.dev, expected.parent?.dev ?? snapshot.parent.dev) + && sameFileId(snapshot.parent.ino, expected.parent?.ino ?? snapshot.parent.ino); } function preflight(plan, actionIds, expectedPlanDigest, options) { diff --git a/src/lib/hook-remediation/fs-port.mjs b/src/lib/hook-remediation/fs-port.mjs index 90a6ad9f..42a9ae5c 100644 --- a/src/lib/hook-remediation/fs-port.mjs +++ b/src/lib/hook-remediation/fs-port.mjs @@ -3,6 +3,7 @@ import path from 'node:path'; import { randomBytes } from 'node:crypto'; import { MAX_AUDIT_SOURCE_BYTES, sha256 } from '../hook-audit/common.mjs'; +import { fileId, sameFileId, statMtimeMs } from '../file-identity.mjs'; export const MAX_HOOK_TARGET_BYTES = MAX_AUDIT_SOURCE_BYTES; @@ -44,29 +45,27 @@ export function inspectHookTarget(file, containmentRoot, { const realFile = fsImpl.realpathSync(file); if (!contained(realRoot, realFile, platform)) throw new Error('target escapes its containment root'); descriptor = fsImpl.openSync(file, fs.constants.O_RDONLY | (fs.constants.O_NOFOLLOW ?? 0)); - const opened = fsImpl.fstatSync(descriptor); + const opened = fsImpl.fstatSync(descriptor, { bigint: true }); if (!opened.isFile() || opened.size > maxBytes) throw new Error('opened target is not a bounded regular file'); - // BigInt identity: Windows file IDs can exceed 2^53; `opened` stays Number for the image. - const openedId = fsImpl.fstatSync(descriptor, { bigint: true }); - if (openedId.dev !== stat.dev || openedId.ino !== stat.ino) { + if (opened.dev !== stat.dev || opened.ino !== stat.ino) { throw new Error('target identity changed between inspection and open'); } const reopened = fsImpl.realpathSync(file); if (reopened !== realFile || !contained(realRoot, reopened, platform)) { throw new Error('target path changed between inspection and open'); } - const bytes = readDescriptor(fsImpl, descriptor, opened.size); + const bytes = readDescriptor(fsImpl, descriptor, Number(opened.size)); const parent = path.dirname(realFile); - const parentStat = fsImpl.statSync(parent); + const parentStat = fsImpl.statSync(parent, { bigint: true }); return { file: path.resolve(file), containmentRoot: path.resolve(containmentRoot), realFile, realRoot, bytes, sha256: sha256(bytes), size: bytes.length, - mode: platform === 'win32' ? null : opened.mode & 0o777, - modeSupported: platform !== 'win32', mtimeMs: opened.mtimeMs, - uid: typeof opened.uid === 'number' ? opened.uid : null, - gid: typeof opened.gid === 'number' ? opened.gid : null, - specialMode: platform === 'win32' ? 0 : opened.mode & 0o7000, - parent: { realPath: parent, dev: parentStat.dev, ino: parentStat.ino }, + mode: platform === 'win32' ? null : Number(opened.mode & 0o777n), + modeSupported: platform !== 'win32', mtimeMs: statMtimeMs(opened), + uid: typeof opened.uid === 'bigint' ? Number(opened.uid) : null, + gid: typeof opened.gid === 'bigint' ? Number(opened.gid) : null, + specialMode: platform === 'win32' ? 0 : Number(opened.mode & 0o7000n), + parent: { realPath: parent, dev: fileId(parentStat.dev), ino: fileId(parentStat.ino) }, }; } finally { if (descriptor !== undefined) fsImpl.closeSync(descriptor); @@ -134,9 +133,10 @@ export function atomicReplaceHookTarget(snapshot, bytes, desiredMode = snapshot. || current.specialMode !== snapshot.specialMode) { throw new Error(`target changed immediately before replacement: ${snapshot.file}`); } - const currentParent = fsImpl.statSync(path.dirname(current.realFile)); + const currentParent = fsImpl.statSync(path.dirname(current.realFile), { bigint: true }); if (current.parent.realPath !== snapshot.parent.realPath - || currentParent.dev !== snapshot.parent.dev || currentParent.ino !== snapshot.parent.ino) { + || !sameFileId(currentParent.dev, snapshot.parent.dev) + || !sameFileId(currentParent.ino, snapshot.parent.ino)) { throw new Error(`target parent changed immediately before replacement: ${snapshot.file}`); } const suffix = randomBytes(12).toString('hex'); diff --git a/src/lib/hook-remediation/store.mjs b/src/lib/hook-remediation/store.mjs index 32a91cf7..d4214eb4 100644 --- a/src/lib/hook-remediation/store.mjs +++ b/src/lib/hook-remediation/store.mjs @@ -3,6 +3,7 @@ import path from 'node:path'; import { randomBytes } from 'node:crypto'; import { MAX_AUDIT_SOURCE_BYTES, sha256, stableJson } from '../hook-audit/common.mjs'; +import { fileId } from '../file-identity.mjs'; export const HOOK_HEAL_RECEIPT_SCHEMA = 'hook-heal-receipt/v1'; const RECEIPT_ID = /^tx-[0-9TZ.-]+-[a-f0-9]{16}$/; @@ -202,7 +203,7 @@ function validImage(image, { preimage = false } = {}) { && (image.gid === null || Number.isInteger(image.gid)) && Number.isInteger(image.specialMode) && image.specialMode >= 0 && image.specialMode <= 0o7000 && path.isAbsolute(image.parent?.realPath ?? '') - && Number.isInteger(image.parent?.dev) && Number.isInteger(image.parent?.ino) + && fileId(image.parent?.dev) !== null && fileId(image.parent?.ino) !== null )); } diff --git a/src/lib/host-alignment.mjs b/src/lib/host-alignment.mjs index ab9867c9..b1757f5c 100644 --- a/src/lib/host-alignment.mjs +++ b/src/lib/host-alignment.mjs @@ -4,6 +4,7 @@ import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { createHash, randomUUID } from 'node:crypto'; +import { fileId } from './file-identity.mjs'; import { writeFileWithBackup } from './file-write.mjs'; import { inspectCodexTomlStructure, isTomlTableLine } from './codex-toml-safety.mjs'; import { enabledPluginRefs } from './codex-plugins.mjs'; @@ -58,13 +59,13 @@ function uniqueJson(source) { } function readSource(file) { - const stat = fs.lstatSync(file); + const stat = fs.lstatSync(file, { bigint: true }); if (!stat.isFile() || stat.isSymbolicLink() || stat.size > 2 * 1024 * 1024) throw new Error('configuration is not a bounded regular file'); const bytes = fs.readFileSync(file); const source = bytes.toString('utf8'); if (!bytes.equals(Buffer.from(source))) throw new Error('configuration is not UTF-8'); - return { file, source, digest: hash(bytes), mode: stat.mode & 0o777, - identity: { real: fs.realpathSync(file), device: stat.dev, inode: stat.ino, mode: stat.mode } }; + return { file, source, digest: hash(bytes), mode: Number(stat.mode & 0o777n), + identity: { real: fs.realpathSync(file), device: fileId(stat.dev), inode: fileId(stat.ino), mode: Number(stat.mode) } }; } function safeJsonTransport(entry) { diff --git a/src/lib/host-health-evidence.mjs b/src/lib/host-health-evidence.mjs index 600b17ed..2d2cd983 100644 --- a/src/lib/host-health-evidence.mjs +++ b/src/lib/host-health-evidence.mjs @@ -4,6 +4,7 @@ import fs from 'node:fs'; import path from 'node:path'; import { createHmac, randomBytes } from 'node:crypto'; import { hostHealthInputPaths } from './paths.mjs'; +import { fileId, statMtimeMs } from './file-identity.mjs'; export function createHostHealthSnapshot({ secret = randomBytes(32), env = process.env, inputPaths = hostHealthInputPaths } = {}) { return ({ cwd, cfg }) => { @@ -35,9 +36,9 @@ export function createHostHealthSnapshot({ secret = randomBytes(32), env = proce for (const dir of (env.PATH ?? '').split(path.delimiter).slice(0, 256)) { const file = path.resolve(dir, host + (process.platform === 'win32' ? '.cmd' : '')); try { - const st = fs.statSync(file); + const st = fs.statSync(file, { bigint: true }); if (!st.isFile()) continue; - hash.update(JSON.stringify([host, fs.realpathSync(file), st.size, st.mtimeMs, st.ino])); + hash.update(JSON.stringify([host, fs.realpathSync(file), Number(st.size), statMtimeMs(st), fileId(st.dev), fileId(st.ino)])); break; } catch { /* next PATH entry */ } } diff --git a/src/lib/host-readiness-local.mjs b/src/lib/host-readiness-local.mjs index ebd19eaa..2ea1bb7d 100644 --- a/src/lib/host-readiness-local.mjs +++ b/src/lib/host-readiness-local.mjs @@ -9,6 +9,7 @@ import path from 'node:path'; import { createHash } from 'node:crypto'; import { readContextConfig } from './codex-context-config.mjs'; import { withDb } from './sqlite.mjs'; +import { xdgBase } from './paths.mjs'; const LIMIT = 1024 * 1024; const plain = x => x !== null && typeof x === 'object' && !Array.isArray(x); @@ -199,8 +200,8 @@ function selectedOpenCodeAgent(config, agentDirs) { } function loadOpenCode({ cwd, home, env }, evidence) { - const global = path.join(env.XDG_CONFIG_HOME || path.join(home, '.config'), 'opencode'); - const data = path.join(env.XDG_DATA_HOME || path.join(home, '.local/share'), 'opencode'); + const global = path.join(xdgBase('XDG_CONFIG_HOME', path.join(home, '.config'), { env }), 'opencode'); + const data = path.join(xdgBase('XDG_DATA_HOME', path.join(home, '.local/share'), { env }), 'opencode'); const auth = document(path.join(data, 'auth.json'), evidence, env); if (Object.values(auth).some(value => value?.type === 'wellknown') || openCodeRemote(data, evidence)) throw new Error('unsupported'); if (fs.existsSync(path.join(global, 'config'))) throw new Error('unsupported'); // legacy TOML migration is native-owned @@ -289,7 +290,7 @@ function defaultOpenCode({ config, auth, home, env, allowed, credentialed }, evi // Native default tries recent available selections, then a configured // provider. Do not borrow credentials from an unrelated provider. - const recent = document(path.join(env.XDG_STATE_HOME || path.join(home, '.local/state'), 'opencode/model.json'), evidence, env).recent; + const recent = document(path.join(xdgBase('XDG_STATE_HOME', path.join(home, '.local/state'), { env }), 'opencode/model.json'), evidence, env).recent; if (Array.isArray(recent) && recent.length) throw new Error('unsupported'); // availability requires native provider catalog const configured = Object.keys(config.provider ?? {}).filter(allowed); if (configured.length === 1) provider = configured[0]; diff --git a/src/lib/live-check-evidence.mjs b/src/lib/live-check-evidence.mjs index 3ec1448f..c3143f03 100644 --- a/src/lib/live-check-evidence.mjs +++ b/src/lib/live-check-evidence.mjs @@ -30,8 +30,10 @@ import { warn } from './output.mjs'; // The checks whose results are remembered: the quick, free live checks. One // list, owned by the refresh vocabulary (a constants-only import). import { LIVE_CHECK_IDS } from './refresh.mjs'; +import { installedRoutingVersion } from './ruflo-memory-contract.mjs'; export { LIVE_CHECK_IDS }; +export const RECORDED_CHECK_IDS = Object.freeze([...LIVE_CHECK_IDS, 'memory-routes']); const STATUSES = new Set(['passed', 'failed', 'inconclusive']); /** The sources a record is written with, and the ones it may still be read with. */ const WRITE_SOURCES = new Set(['sync', 'status-refresh-live']); @@ -45,7 +47,7 @@ const REASON_MAX = 200; export const liveCheckDir = () => path.join(paths.evidenceDir(), 'live-check'); function assertKnownId(id) { - if (!LIVE_CHECK_IDS.includes(id)) throw new TypeError(`unknown live check id: ${String(id).slice(0, 40)}`); + if (!RECORDED_CHECK_IDS.includes(id)) throw new TypeError(`unknown live check id: ${String(id).slice(0, 40)}`); } /** Printable, single-line, bounded. Stored reasons come from the kit's own @@ -141,18 +143,22 @@ const INPUTS = { security: () => ({ ruflo: rufloVersion() }), 'deja-vu': ({ cfg }) => ({ dejaVu: cfg.integrations?.tools?.dejaVu ?? null }), memory: () => ({ ruflo: rufloVersion() }), + 'memory-routes': ({ routingVersion, platform }) => ({ + routingVersion: routingVersion === undefined ? installedRoutingVersion() : routingVersion, + platform: platform ?? process.platform, + }), }; /** * The inputs key a check ran against. Writers and readers MUST both call this * so the same configuration yields the same key. Never throws. * @param {string} id - * @param {{cfg?:any,env?:NodeJS.ProcessEnv,cwd?:string}} [facts] + * @param {{cfg?:any,env?:NodeJS.ProcessEnv,cwd?:string,routingVersion?:string|null,platform?:string}} [facts] */ -export function liveCheckInputsKey(id, { cfg = {}, env = process.env, cwd = process.cwd() } = {}) { +export function liveCheckInputsKey(id, { cfg = {}, env = process.env, cwd = process.cwd(), routingVersion, platform } = {}) { assertKnownId(id); let parts; - try { parts = INPUTS[id]({ cfg: cfg ?? {}, env, cwd }); } catch { parts = { unavailable: true }; } + try { parts = INPUTS[id]({ cfg: cfg ?? {}, env, cwd, routingVersion, platform }); } catch { parts = { unavailable: true }; } return digest({ id, ...parts }); } @@ -181,12 +187,12 @@ export function embeddingProbeOutcome(live) { * when it could not be remembered. Returns whether it was recorded. * @param {string} id * @param {{status:string,reason?:string|null}|null} outcome - * @param {{source:string,cfg?:any,env?:NodeJS.ProcessEnv,cwd?:string,now?:number}} context + * @param {{source:string,cfg?:any,env?:NodeJS.ProcessEnv,cwd?:string,now?:number,inputsKey?:string}} context */ -export function rememberLiveCheck(id, outcome, { source, cfg, env, cwd, now } = /** @type {any} */ ({})) { +export function rememberLiveCheck(id, outcome, { source, cfg, env, cwd, now, inputsKey } = /** @type {any} */ ({})) { if (!outcome) return false; const recorded = recordLiveCheck({ id, status: outcome.status, reason: outcome.reason ?? null, source, - inputsKey: liveCheckInputsKey(id, { cfg, env, cwd }) }, { now }); + inputsKey: inputsKey ?? liveCheckInputsKey(id, { cfg, env, cwd }) }, { now }); if (!recorded) warn(`${id}: could not remember this live check result; ak status will not show it`); return recorded; } diff --git a/src/lib/live-checks.mjs b/src/lib/live-checks.mjs index 2aaab40c..9553008a 100644 --- a/src/lib/live-checks.mjs +++ b/src/lib/live-checks.mjs @@ -31,7 +31,7 @@ import { runHarvest } from './harvest.mjs'; import { runLifecycle } from './adapters/lifecycle.mjs'; import { companionLifecycleFor } from './adapters/companion-lifecycle-registry.mjs'; import { ok, warn, fail, info, heading, captureOutput } from './output.mjs'; -import { rememberLiveCheck, embeddingProbeOutcome } from './live-check-evidence.mjs'; +import { rememberLiveCheck, liveCheckInputsKey, embeddingProbeOutcome } from './live-check-evidence.mjs'; import { LIVE_CHECK_IDS, SLOW_PROOF_IDS } from './refresh.mjs'; export { LIVE_CHECK_IDS, SLOW_PROOF_IDS }; @@ -107,9 +107,11 @@ export async function observeProjectMemoryRoutes(tmp, env, namespace, deps) { try { const observation = await probeProjectMemoryRoutes(tmp, env, namespace, deps); for (const { level, message } of describeMemoryRoutes(observation)) (level === 'ok' ? ok : warn)(message); + return observation; } catch (e) { // An observation problem must never turn a working CLI proof into a failure. warn(`cross-interface routing not observed: ${e.message}`); + return null; } } @@ -132,11 +134,12 @@ async function purgeProofNamespace(tmp, env, namespace, key, runner = runCmd) { /** The memory round trip in an isolated folder under `tmpRoot`, removed * whatever happens. `observeRoutes: false` keeps the quick `memory` check to * the CLI round trip: the route observation (the `memory-routes` proof) - * starts a real MCP server and can only add warnings, which a live-check - * record does not carry. - * @param {{ observeRoutes?: boolean, tmpRoot?: string, runner?: typeof runCmd, haveCmd?: typeof have }} [options] */ + * starts a real MCP server. Its observation has separate evidence, while + * the CLI round-trip evidence remains under `memory`. + * @param {{ observeRoutes?: boolean, routeVerdict?: boolean, onCliOutcome?: (outcome:{status:string,reason:null})=>void, + * tmpRoot?: string, runner?: typeof runCmd, haveCmd?: typeof have }} [options] */ export async function verifyMemory({ - observeRoutes = true, tmpRoot = os.tmpdir(), runner = runCmd, haveCmd = have, + observeRoutes = true, routeVerdict = false, onCliOutcome, tmpRoot = os.tmpdir(), runner = runCmd, haveCmd = have, } = {}) { heading('memory — store, retrieve, locate the on-disk row, purge, and observe CLI/MCP routing in an isolated dir'); if (!(await haveCmd('ruflo'))) { fail('ruflo CLI not installed — cannot prove project memory'); return false; } @@ -182,17 +185,29 @@ export async function verifyMemory({ purged = await purgeProofNamespace(tmp, env, namespace, key, runner); if (!purged) { fail('isolated namespace purge did not remove the proof row'); return false; } ok('isolated proof namespace purged'); - if (observeRoutes) await observeProjectMemoryRoutes(tmp, env, namespace); + onCliOutcome?.({ status: 'passed', reason: null }); + if (observeRoutes) { + const route = /** @type {{status?:string,cliToMcp?:string,mcpToCli?:string}|null} */ + (await observeProjectMemoryRoutes(tmp, env, namespace)); + if (routeVerdict) return route?.status === 'observed' && + ['visible', 'not-visible'].includes(route.cliToMcp) && + ['visible', 'not-visible'].includes(route.mcpToCli) + ? { status: 'passed', reason: null } + : { status: 'inconclusive', reason: 'cross-interface routing not observed completely' }; + } return true; } catch (e) { fail(`memory proof error: ${e.message}`); return false; } finally { - if (stored && !purged) { - await runner('ruflo', ['memory', 'purge', '--namespace', namespace, '--force'], - { cwd: tmp, env, timeout: 120_000 }); + try { + if (stored && !purged) { + await runner('ruflo', ['memory', 'purge', '--namespace', namespace, '--force'], + { cwd: tmp, env, timeout: 120_000 }); + } + } finally { + fs.rmSync(tmp, { recursive: true, force: true }); } - fs.rmSync(tmp, { recursive: true, force: true }); } } @@ -451,10 +466,10 @@ async function verifyProjectProviders(root, cfg, { runner, haveCmd }) { * bridge derives agentdb-memory.db from (CLAUDE_FLOW_MEMORY_PATH), and * AGENTDB_PATH. An inherited value of any of them would otherwise receive the * proof rows. Nothing is seeded: the proof is Ruflo's own verbs succeeding. */ -export async function verifyHarvest({ runner = runCmd, haveCmd = have } = {}) { +export async function verifyHarvest({ tmpRoot = os.tmpdir(), runner = runCmd, haveCmd = have } = {}) { heading('harvest — record an outcome and distill, in an isolated store'); if (!(await haveCmd('ruflo'))) { fail('ruflo CLI not installed — cannot prove the harvest write path'); return false; } - const tmp = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'agentic-kit-harvest-'))); + const tmp = fs.realpathSync(fs.mkdtempSync(path.join(tmpRoot, 'agentic-kit-harvest-'))); const swarm = path.join(tmp, '.swarm'); // Pinned explicitly: a temporary folder inside a Git checkout would make the // derived project root the enclosing repository. @@ -577,7 +592,7 @@ export async function verifyDejaVu({ const enabled = cfg?.integrations?.tools?.dejaVu?.enabled === true; if (!dejaVuProofApplies(cfg)) { warn('deja-vu disabled and unowned — skipped'); - return true; + return { status: 'skipped', reason: 'disabled and unowned' }; } if (!adapter) { fail('deja-vu lifecycle adapter unavailable'); @@ -637,9 +652,10 @@ const CHECKS = Object.freeze([ // The full AQE proof remembers only its live embedding request, itself, and // only for a backend the kit manages. { ...slow('aqe'), run: ({ cfg, cwd, onEvidence }) => verifyAqe({ cfg, cwd, onEvidence }) }, - // The memory round trip plus the CLI/MCP route observation; its verdict is - // the memory check's. - { ...slow('memory-routes', 'memory'), run: () => verifyMemory({ observeRoutes: true }) }, + // The memory round trip plus the CLI/MCP route observation has distinct + // evidence; a CLI-only pass cannot establish routing. + { ...slow('memory-routes', 'memory-routes'), run: ({ onCliOutcome }) => + verifyMemory({ observeRoutes: true, routeVerdict: true, onCliOutcome }) }, ].map((check) => Object.freeze(check))); /** @@ -673,7 +689,9 @@ const duration = (ms) => (ms < 1000 ? `${ms} ms` : `${Math.round(ms / 1000)} s`) async function runOneLiveCheck(check, ctx, { timeoutMs, graceMs }) { const controller = new AbortController(); const started = Date.now(); - const work = captureOutput(() => withAbortSignal(controller.signal, () => check.run(ctx))) + let cliOutcome = null; + const work = captureOutput(() => withAbortSignal(controller.signal, + () => check.run({ ...ctx, onCliOutcome: (outcome) => { cliOutcome = outcome; } }))) .then(({ result, entries }) => ({ outcome: checkOutcome(result, entries, check.id), entries }), () => ({ outcome: { status: 'inconclusive', reason: 'the check could not run' }, entries: [] })); const deadline = sleep(timeoutMs); @@ -687,7 +705,8 @@ async function runOneLiveCheck(check, ctx, { timeoutMs, graceMs }) { settled = { outcome: { status: 'inconclusive', reason: `no result within ${duration(timeoutMs)}` }, entries: late?.entries ?? [] }; } const { outcome, entries } = settled; - return { id: check.id, status: outcome.status, reason: outcome.reason ?? null, elapsedMs: Date.now() - started, entries }; + return { id: check.id, status: outcome.status, reason: outcome.reason ?? null, + elapsedMs: Date.now() - started, entries, cliOutcome }; } /** @@ -706,15 +725,28 @@ export async function runLiveChecks({ cfg = loadKitConfig(), cwd = process.cwd(), only = [], checks = liveChecksFor(cfg, only), timeoutMs, graceMs = LIVE_CHECK_GRACE_MS, source = 'status-refresh-live', } = {}) { - const remember = (id, outcome) => rememberLiveCheck(id, outcome, { source, cfg, cwd }); + const remember = (id, outcome, inputsKey) => rememberLiveCheck(id, outcome, { source, cfg, cwd, inputsKey }); const ctx = { cfg, cwd, onEvidence: remember }; + // Capture the installed implementation before any proof starts. A package + // upgrade while the slow check runs cannot turn an old observation into a + // pass for the newly installed CLI. + const routeKeys = checks.map((check) => check.id === 'memory-routes' + ? liveCheckInputsKey('memory-routes', { cfg, cwd }) : null); const results = await Promise.all(checks.map(async (check) => ({ ...(await runOneLiveCheck(check, ctx, { timeoutMs: timeoutMs ?? check.timeoutMs ?? QUICK_TIMEOUT_MS, graceMs })), applies: check.applies?.(cfg ?? {}) ?? true, }))); checks.forEach((check, i) => { const evidenceId = check.evidenceId === undefined ? check.id : check.evidenceId; - if (evidenceId && results[i].applies) remember(evidenceId, results[i]); + if (!results[i].applies || results[i].status === 'skipped') return; + if (check.id === 'memory-routes') { + remember('memory', results[i].cliOutcome ?? results[i]); + if (routeKeys[i] !== liveCheckInputsKey('memory-routes', { cfg, cwd })) { + results[i].status = 'inconclusive'; + results[i].reason = 'installed routing implementation changed during the check'; + } + remember('memory-routes', results[i], routeKeys[i]); + } else if (evidenceId) remember(evidenceId, results[i]); }); - return results; + return results.map(({ cliOutcome: _cliOutcome, ...result }) => result); } diff --git a/src/lib/live/jsonl-tailer.mjs b/src/lib/live/jsonl-tailer.mjs index 8a3c3663..610abc44 100644 --- a/src/lib/live/jsonl-tailer.mjs +++ b/src/lib/live/jsonl-tailer.mjs @@ -1,5 +1,6 @@ import fs from 'node:fs'; import { StringDecoder } from 'node:string_decoder'; +import { fileId } from '../file-identity.mjs'; const limit = (value, ceiling) => Number.isSafeInteger(value) && value > 0 ? Math.min(value, ceiling) : ceiling; @@ -70,7 +71,7 @@ export class JsonlTailer { reconcile() { if (!this.canRead(this.#file)) return; let stat; - try { stat = fs.statSync(this.#file); } catch (error) { + try { stat = fs.statSync(this.#file, { bigint: true }); } catch (error) { if (error.code === 'ENOENT') this.#absent(); else this.#unreadable(error); return; @@ -82,31 +83,32 @@ export class JsonlTailer { })); return; } - const identity = `${stat.dev}:${stat.ino}`; + const identity = `${fileId(stat.dev)}:${fileId(stat.ino)}`; + const size = Number(stat.size); if (this.#identity == null) { this.#identity = identity; // A file that appeared after tailing began holds only new records, so // neither a resume offset nor startAtEnd may skip any of it. if (this.#absentSeen) this.#offset = 0; - else if (this.startOffset != null) this.#offset = Math.min(this.startOffset, stat.size); - else if (this.startAtEnd) this.#offset = stat.size; - } else if (this.#identity !== identity || stat.size < this.#offset) { + else if (this.startOffset != null) this.#offset = Math.min(this.startOffset, size); + else if (this.startAtEnd) this.#offset = size; + } else if (this.#identity !== identity || size < this.#offset) { this.#identity = identity; this.#resetPosition(); } - if (stat.size <= this.#offset) { + if (size <= this.#offset) { // Nothing new to read. Prove readability anyway when it is not yet known // (or was lost), so a mode-000 file cannot pass as healthy by staying // the same size. if (this.#presence !== 'readable') this.#probeReadable(); - this.#coverage(stat.size); + this.#coverage(size); return; } try { const flags = fs.constants.O_RDONLY | (fs.constants.O_NOFOLLOW ?? 0); const fd = fs.openSync(this.#file, flags); try { - let remaining = Math.min(stat.size - this.#offset, this.maxReadBytes); + let remaining = Math.min(size - this.#offset, this.maxReadBytes); const bytes = Buffer.alloc(Math.min(remaining, this.maxChunkBytes)); while (remaining > 0) { const count = fs.readSync(fd, bytes, 0, Math.min(bytes.length, remaining), this.#offset); @@ -118,7 +120,7 @@ export class JsonlTailer { } finally { fs.closeSync(fd); } this.#presence = 'readable'; } catch (error) { this.#unreadable(error); } - this.#coverage(stat.size); + this.#coverage(size); } #resetPosition() { diff --git a/src/lib/live/live-sessions-service.mjs b/src/lib/live/live-sessions-service.mjs index 0e22fc24..470141d0 100644 --- a/src/lib/live/live-sessions-service.mjs +++ b/src/lib/live/live-sessions-service.mjs @@ -57,6 +57,10 @@ export class LiveSessionsService { #projection = emptyLiveProjection(); #tailers = new Map(); #contexts = new Map(); + // Keep a bounded set of displaced native readers with their byte offsets + // and partial-line state. Re-entry within this retention window must not + // replay already counted records. + #dormantTailers = new Map(); #timer = null; #started = false; #edgeKeys = new Set(); @@ -140,6 +144,7 @@ export class LiveSessionsService { if (this.#timer != null) this.#options.clearInterval(this.#timer); this.#timer = null; for (const tailer of this.#tailers.values()) tailer.close(); + for (const { tailer } of this.#dormantTailers.values()) tailer.close(); this.#runtimeBindings.clear(); this.#historyPages.clear(); this.#started = false; @@ -253,9 +258,14 @@ export class LiveSessionsService { const desiredNative = new Set([...claude, ...codex]); for (const [file, context] of this.#contexts) { if (!['claude', 'codex'].includes(context.adapter) || desiredNative.has(file)) continue; - this.#tailers.get(file)?.close(); + const tailer = this.#tailers.get(file); + tailer?.close(); + if (tailer) this.#dormantTailers.set(file, { tailer, context }); this.#tailers.delete(file); this.#contexts.delete(file); + while (this.#dormantTailers.size > Math.max(2, this.#options.maxFiles * 2)) { + this.#dormantTailers.delete(this.#dormantTailers.keys().next().value); + } } for (const file of claude) { this.#add(file, { @@ -272,6 +282,13 @@ export class LiveSessionsService { #add(file, context, initial) { if (this.#tailers.has(file) || this.#tailers.size >= this.#options.maxFiles) return; + const dormant = this.#dormantTailers.get(file); + if (dormant) { + this.#dormantTailers.delete(file); + this.#tailers.set(file, dormant.tailer); + this.#contexts.set(file, dormant.context); + return; + } if (initial && ['claude', 'codex'].includes(context.adapter)) { // One shared bootstrap context so metadata learned early (codex // session_meta id/meta, project, model, provider) persists across the diff --git a/src/lib/live/process-sessions.mjs b/src/lib/live/process-sessions.mjs index acf2eb80..7b778746 100644 --- a/src/lib/live/process-sessions.mjs +++ b/src/lib/live/process-sessions.mjs @@ -1,10 +1,10 @@ import { execFile } from 'node:child_process'; import fs from 'node:fs'; -import os from 'node:os'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; import { promisify } from 'node:util'; import { inspectGitWorkspace } from './git-workspace.mjs'; +import { stateBase } from '../paths.mjs'; const execFileAsync = promisify(execFile); @@ -31,7 +31,7 @@ const WIN32_SURVEY_SCRIPT = fileURLToPath( function runtimeDebug(stage, fields = {}) { if (!process?.env || process.env.AK_RUNTIME_DEBUG !== '1') return; try { - const root = process.env.XDG_STATE_HOME || path.join(os.homedir(), '.local', 'state'); + const root = stateBase(); const file = process.env.AK_RUNTIME_DEBUG_FILE || path.join(root, 'agentic-kit', 'runtime-debug.log'); const safeStage = String(stage || 'unknown').replace(/[^a-z0-9._-]/gi, '_').slice(0, 64); const kv = Object.entries(fields) @@ -52,8 +52,35 @@ const HOST_NAMES = new Map([ ['opencode', 'opencode'], ]); -const tokens = (command) => String(command ?? '').trim().split(/\s+/) - .map((token) => token.replace(/^['"]|['"]$/g, '')); +function tokens(command) { + const args = []; + let word = ''; + let quote = null; + const input = String(command ?? '').trim(); + for (let index = 0; index < input.length; index++) { + const char = input[index]; + if (quote && char === '\\' && input[index + 1] === quote) { + word += char + input[++index]; + } else if (char === quote) { + quote = null; + word += char; + } else if (!quote && (char === '"' || char === "'")) { + quote = char; + word += char; + } else if (!quote && /\s/.test(char)) { + if (word) args.push(word); + word = ''; + } else { + word += char; + } + } + // An unmatched quote in flattened `ps args=` cannot establish argument + // boundaries, so no token can prove a service or MCP subcommand. + if (quote) return []; + if (word) args.push(word); + return args.map((arg) => ((arg.startsWith('"') && arg.endsWith('"')) + || (arg.startsWith("'") && arg.endsWith("'"))) ? arg.slice(1, -1) : arg); +} // win32.basename deliberately, on every platform: it splits on BOTH separators, // so `C:\opt\bin\codex` and `/usr/local/bin/codex` both reduce to `codex`. The // POSIX result is unchanged (a POSIX path has no backslash to split on), and @@ -67,8 +94,9 @@ const executableName = (value) => path.win32.basename(String(value ?? '')).toLow * full executable path (macOS `comm=`) and argv as one space-joined string, so * `/Users/me/Library/Application Support/x/codex mcp-server` would otherwise * split inside the path and shift every argument by one. The prefix must end - * at whitespace or the end of the string. Spaces inside later arguments stay - * ambiguous, which is inherent to the args= format. + * at whitespace or the end of the string. Later balanced quotes preserve a + * value boundary where `ps args=` includes them; flattened unquoted values + * remain ambiguous. */ function argvOf(command, executable) { const text = String(command ?? '').trim(); @@ -80,6 +108,57 @@ function argvOf(command, executable) { return tokens(text); } +// Codex's root CLI accepts options before a subcommand. Consume only options +// with known arity; an unknown option, missing value, prompt, or `--` leaves +// the role unproven. In particular, a value equal to "app-server" is a value, +// not a subcommand. `ps args=` has no reliable quoting for spaced values, so +// ambiguous command lines degrade to an ordinary controller. +const CODEX_VALUE_OPTIONS = new Set([ + '-c', '--config', '-s', '--sandbox', '-a', '--ask-for-approval', + '-m', '--model', '-p', '--profile', '-C', '--cd', + '--enable', '--disable', '--local-provider', '--add-dir', + '--remote', '--remote-auth-token-env', +]); +const CODEX_SWITCH_OPTIONS = new Set([ + '--strict-config', '--oss', '--approve-for-me', + '--dangerously-bypass-approvals-and-sandbox', '--dangerously-bypass-hook-trust', + '--worktree', '--search', '--no-alt-screen', '--no-daemon', +]); +function validConfigOverride(value) { + const match = /^([a-zA-Z0-9_.-]+)=(.+)$/.exec(value); + if (!match) return false; + const literal = match[2]; + // Codex accepts raw-string fallback values. Once `ps` flattens an unquoted + // raw string, the following word could be part of that value. Only a + // self-delimiting TOML scalar can establish where a subcommand begins. + return /^"(?:\\.|[^"\\])*"$/.test(literal) + || /^'[^']*'$/.test(literal) + || /^(?:true|false|-?\d+(?:\.\d+)?)$/.test(literal); +} +function codexSubcommand(argv, offset) { + for (let index = offset; index < argv.length && index < offset + 32; index++) { + const token = argv[index]; + if (token === 'app-server' || token === 'mcp-server') return token; + if (token === '--') return null; + if (CODEX_SWITCH_OPTIONS.has(token)) continue; + if (CODEX_VALUE_OPTIONS.has(token)) { + if (index + 1 >= argv.length || argv[index + 1].startsWith('-')) return null; + if ((token === '-c' || token === '--config') + && !validConfigOverride(argv[index + 1])) return null; + index++; + continue; + } + const option = token?.split('=', 1)[0]; + if (CODEX_VALUE_OPTIONS.has(option) && token.length > option.length + 1) { + if ((option === '-c' || option === '--config') + && !validConfigOverride(token.slice(option.length + 1))) return null; + continue; + } + return null; + } + return null; +} + /** Identify only a controller executable or its supported Node launcher. */ export function hostFromCommand(command, executable = null) { const argv = argvOf(command, executable); @@ -91,23 +170,48 @@ export function hostFromCommand(command, executable = null) { host = HOST_NAMES.get(executableName(argv[wrapperIndex])) ?? null; argumentOffset = wrapperIndex + 1; } - if (host === 'codex' && argv[argumentOffset] === 'mcp-server') return null; + if (host === 'codex' && codexSubcommand(argv, argumentOffset) === 'mcp-server') return null; return host; } const DESKTOP_HOSTED_CLAUDE_CLI = /\/claude-code\/[^/]+\/claude\.app\/Contents\/MacOS\/claude$/i; +// Match the native executable inside the application bundle. A folder name in +// argv, cwd, or an unrelated executable never establishes an application. +const DESKTOP_APPLICATIONS = [ + { executable: /\/Claude\.app\/Contents\/MacOS\/Claude$/, name: 'Claude Desktop' }, + { executable: /\/ChatGPT\.app\/Contents\/MacOS\/ChatGPT$/, name: 'ChatGPT desktop app' }, +]; +const desktopApplication = (executable) => DESKTOP_APPLICATIONS + .find((app) => app.executable.test(String(executable ?? '').replaceAll('\\', '/')))?.name ?? null; +// The installed ChatGPT bundle carries a native CodexCLI.app executable. +// Resources/codex-cli/bin/codex is a shell wrapper, not a process image. +const DESKTOP_HOSTED_CODEX_CLI = /\/ChatGPT\.app\/Contents\/Resources\/codex-cli\/CodexCLI\.app\/Contents\/MacOS\/codex$/; +const APP_BUNDLE_PATH = /\/[^/]+\.app\/Contents\//i; +const KNOWN_APP_BUNDLE_PATH = /\/(?:Claude|ChatGPT)\.app\/Contents\//; +const supportedExecutable = (executable) => { + const nativePath = String(executable ?? '').replaceAll('\\', '/'); + return !APP_BUNDLE_PATH.test(nativePath) || KNOWN_APP_BUNDLE_PATH.test(nativePath) + || DESKTOP_HOSTED_CLAUDE_CLI.test(nativePath); +}; /** Reduce process argv to a non-sensitive controller role, then discard argv. */ function controllerKind(row) { + if (desktopApplication(row.executable)) return 'desktop-app'; const argv = argvOf(row.command, row.executable); - if (argv.includes('app-server')) return 'host-service'; + const program = executableName(row.executable ?? argv[0]); + const argumentOffset = ['node', 'nodejs'].includes(program) ? 2 : 1; + const codex = program === 'codex' + || (['node', 'nodejs'].includes(program) && executableName(argv[1]) === 'codex'); + if (codex ? codexSubcommand(argv, argumentOffset) === 'app-server' + : argv[argumentOffset] === 'app-server') return 'host-service'; const executable = String(row.executable ?? argv[0] ?? '').replaceAll('\\', '/'); // The Claude desktop app runs its own Claude Code CLI from a versioned // bundle (`…/Claude/claude-code//claude.app/Contents/MacOS/claude`). // It is a user's project session, not the desktop app, even though its path // has the `.app/Contents/` shape. - if (DESKTOP_HOSTED_CLAUDE_CLI.test(executable)) return 'project-session'; - if (/\/[^/]+\.app\/Contents\//i.test(executable)) return 'desktop-app'; + if (DESKTOP_HOSTED_CLAUDE_CLI.test(executable) + || DESKTOP_HOSTED_CODEX_CLI.test(executable)) return 'project-session'; + if (APP_BUNDLE_PATH.test(executable)) return 'desktop-app'; return 'project-session'; } @@ -157,7 +261,8 @@ function parseArgsByPid(output) { const isHostCandidate = (row) => { const name = executableName(row.executable); - return HOST_NAMES.has(name) || name === 'node' || name === 'nodejs'; + return supportedExecutable(row.executable) && (!!desktopApplication(row.executable) + || HOST_NAMES.has(name) || name === 'node' || name === 'nodejs'); }; /** @@ -244,8 +349,10 @@ function rootControllers(rows) { const byPid = new Map(rows.map((row) => [row.pid, row])); const candidates = new Map(); for (const row of rows) { - const host = hostFromCommand(row.command, row.executable); - if (host) candidates.set(row.pid, { ...row, host }); + const application = desktopApplication(row.executable); + const host = application || !supportedExecutable(row.executable) + ? null : hostFromCommand(row.command, row.executable); + if (application || host) candidates.set(row.pid, { ...row, host, application }); else if (row.command) runtimeDebug('classify', { pid: row.pid, exe: row.executable, host: 'none' }); } const roots = [...candidates.values()].filter((candidate) => { @@ -487,7 +594,7 @@ export async function listActiveHostSessions({ } = {}) { const rows = processRows ?? await collectRows({ platform, execFileImpl, uid, scriptPath, env }); - const controllers = rootControllers(rows); + const controllers = rootControllers(rows).filter((row) => controllerKind(row) === 'project-session'); const pids = controllers.map((row) => row.pid); let cwds = cwdByPid; if (!cwds) { @@ -628,6 +735,7 @@ export async function surveyHostProcesses({ const attributable = typeof cwd === 'string' && isAbsoluteFor(cwd); return { host: row.host, + application: row.application ?? null, controllerKind: controllerKind(row), pid: row.pid, ppid: row.ppid, diff --git a/src/lib/live/transcript-streams.mjs b/src/lib/live/transcript-streams.mjs index 32dee5a4..32a41201 100644 --- a/src/lib/live/transcript-streams.mjs +++ b/src/lib/live/transcript-streams.mjs @@ -1,5 +1,6 @@ import fs from 'node:fs'; import path from 'node:path'; +import { fileId } from '../file-identity.mjs'; import { JsonlTailer } from './jsonl-tailer.mjs'; import { LiveReplayStream } from './replay-stream.mjs'; import { @@ -155,12 +156,13 @@ class TranscriptStream { this.#host = host; this.#sessionId = sessionId; this.#options = options; - const epoch = fs.statSync(file).ino; + const stat = fs.statSync(file, { bigint: true }); + const epoch = fileId(stat.ino); this.#stream = new LiveReplayStream({ capacity: options.replayCapacity, prefix: `tx-${host}-${sessionId}-${epoch}`, }); - const offset = fs.statSync(file).size; + const offset = Number(stat.size); const history = tailLines(file, offset, options.maxHistoryBytes, options.maxHistoryRecords); const candidates = []; for (const raw of history.lines) { diff --git a/src/lib/maintenance/discovery/history.mjs b/src/lib/maintenance/discovery/history.mjs index 0d2c0691..0b6f5aff 100644 --- a/src/lib/maintenance/discovery/history.mjs +++ b/src/lib/maintenance/discovery/history.mjs @@ -1,9 +1,9 @@ // ADR-0048 scan history — bounded, retained scan summaries // (docs/maintenance.md "Retention"). -// This store holds ONLY terminal scan summaries. It structurally cannot reach +// This store holds terminal scan summaries and paused continuation boundaries. It cannot reach // receipts, dispositions, or recipe acceptance records — those live in other // agents' stores — so `clearHistory` cannot violate MNT-PRV-008 by scope -// alone; the `isProtected` guard below is defense in depth for a summary that +// alone; the `hasOpenContinuation` guard below is defense in depth for a summary that // still names an open continuation. import fs from 'node:fs'; import path from 'node:path'; @@ -30,14 +30,14 @@ function pruneSummaries(summaries, retention, now) { const cutoff = now - retention.maxAgeDays * 86_400_000; const bySource = new Map(); for (const summary of summaries - .filter((entry) => Date.parse(entry.completedAt) >= cutoff) - .sort((a, b) => Date.parse(a.completedAt) - Date.parse(b.completedAt))) { + .filter((entry) => Date.parse(entry.recordedAt ?? entry.completedAt) >= cutoff) + .sort((a, b) => Date.parse(a.recordedAt ?? a.completedAt) - Date.parse(b.recordedAt ?? b.completedAt))) { const key = JSON.stringify([summary.environmentId, summary.sourceId]); const list = bySource.get(key) ?? []; list.push(summary); bySource.set(key, list.slice(-retention.maxSummaries)); } - return [...bySource.values()].flat().sort((a, b) => Date.parse(a.completedAt) - Date.parse(b.completedAt)); + return [...bySource.values()].flat().sort((a, b) => Date.parse(a.recordedAt ?? a.completedAt) - Date.parse(b.recordedAt ?? b.completedAt)); } /** @@ -47,7 +47,8 @@ function pruneSummaries(summaries, retention, now) { */ export function createScanHistoryStore(dir, { fsImpl = fs, now = Date.now, retention = SCAN_HISTORY_RETENTION, - hasOpenContinuation = (/** @type {string} */ _scanId) => false, + hasOpenContinuation = (scanId) => typeof scanId === 'string' && /^[A-Za-z0-9_-]{1,80}$/u.test(scanId) + && fsImpl.existsSync(path.join(dir, 'checkpoints', `${scanId}.json`)), } = {}) { const file = path.join(dir, 'scan-history.json'); let effectiveRetention = clampRetention(retention); @@ -86,7 +87,7 @@ export function createScanHistoryStore(dir, { return environmentId ? summaries.filter((entry) => entry.environmentId === environmentId) : summaries; } - /** Remove only terminal, non-protected summaries; refuses (never throws) to + /** Remove only non-protected summaries; refuses (never throws) to * touch a summary whose scanId still names an open continuation. * @param {{environmentId?: string}} [options] */ function clearHistory({ environmentId } = {}) { diff --git a/src/lib/maintenance/discovery/orchestrator.mjs b/src/lib/maintenance/discovery/orchestrator.mjs index 38aa917d..4ad5ce95 100644 --- a/src/lib/maintenance/discovery/orchestrator.mjs +++ b/src/lib/maintenance/discovery/orchestrator.mjs @@ -118,11 +118,18 @@ function toCoverageRecord(record) { }; } -function toSummary(record, now) { +function toSummary(record, now, state = record.scanState) { + const recordedAt = new Date(now()).toISOString(); return { scanId: record.scanId, sourceId: record.sourceId, environmentId: record.environmentId, - state: record.scanState, startedAt: record.createdAt, completedAt: new Date(now()).toISOString(), + state, startedAt: record.createdAt, + completedAt: state === 'paused' ? null : recordedAt, + ...(state === 'paused' ? { recordedAt } : {}), visited: record.visited, limitingReason: record.limitingReason, ceiling: record.ceiling ?? null, + ...(state === 'paused' ? { + completedPartitions: record.completedPartitions.length, + pendingPartitions: record.pendingPartitions.length, + } : {}), }; } @@ -180,7 +187,8 @@ function remainingEntryBudget(ceilingEntries, visited) { * remove: (scanId: string) => void, list: () => string[] }} CheckpointStore * @typedef {{ recordSummary: (summary: object) => any, * list: (options?: {environmentId?: string}) => Array<{sourceId: string, environmentId: string, - * state: string, completedAt: string, visited?: number, limitingReason?: string|null, + * state: string, completedAt: string|null, recordedAt?: string, visited?: number, + * completedPartitions?: number, pendingPartitions?: number, limitingReason?: string|null, * ceiling?: string|null}> }} HistoryStore * @typedef {{ write: (entry: object) => boolean, * current: () => Array<{sourceId: string, environmentId: string}> }} LastGoodStore @@ -446,6 +454,11 @@ export function createScanOrchestrator({ error.code = 'SOURCE_NOT_PAUSABLE'; throw error; } + if (!historyStore) throw new Error('scan history unavailable; cannot persist pause'); + // Commit the continuation before claiming that pause is durable. If + // either store rejects the write, leave the live state unchanged. + persistCheckpoint(record); + historyStore.recordSummary(toSummary(record, now, 'paused')); transition(record, 'paused'); return coverageFor(toCoverageRecord(record)); } @@ -486,8 +499,8 @@ export function createScanOrchestrator({ /** A configured source with no live record — never started, paused, or * stopped in this process (e.g. right after a restart, when `records` is - * empty again) — first restores its latest terminal history summary when - * that summary is itself a failure, or a stop at a limit other than the + * empty again) — first restores its latest history summary when + * that summary is a pause, a failure, or a stop at a limit other than the * user's own request (audit finding M1b): a real failure must read the * same on the Discovery panel and the Inventory banner both before and * after a restart, since both read this same `coverage()`. A user stop is @@ -501,11 +514,13 @@ export function createScanOrchestrator({ const record = records.get(source.sourceId); if (record) return coverageFor(toCoverageRecord(record)); const summary = summaryBySource?.get(`${source.environmentId}:${source.sourceId}`); - if (summary && (summary.state === 'failed' || (summary.state === 'stopped' && summary.limitingReason !== 'stopped-by-user'))) { + if (summary && (summary.state === 'paused' || summary.state === 'failed' || (summary.state === 'stopped' && summary.limitingReason !== 'stopped-by-user'))) { return coverageFor({ ...toCoverageRecord(initRecord(source)), scanState: summary.state, visited: summary.visited ?? 0, + completedPartitions: summary.completedPartitions ?? 0, + pendingPartitions: summary.pendingPartitions ?? 0, limitingReason: summary.limitingReason ?? null, ceiling: summary.ceiling ?? null, }); @@ -516,10 +531,10 @@ export function createScanOrchestrator({ return coverageFor(toCoverageRecord(initRecord(source))); } - /** Index the newest terminal summary per `(environmentId, sourceId)` from + /** Index the newest summary per `(environmentId, sourceId)` from * the history store, read once per `coverage()` call rather than once per * source (M1b). `historyStore.list()` is already ascending by - * `completedAt` with ties in recording order, so `>=` (not `>`) keeps the + * record timestamp with ties in recording order, so `>=` (not `>`) keeps the * most-recently-recorded summary on a same-millisecond tie — a later * successful scan must win over an earlier failure even when both * complete within the same clock tick. */ @@ -528,7 +543,7 @@ export function createScanOrchestrator({ for (const summary of historyStore?.list() ?? []) { const key = `${summary.environmentId}:${summary.sourceId}`; const existing = bySource.get(key); - if (!existing || Date.parse(summary.completedAt) >= Date.parse(existing.completedAt)) bySource.set(key, summary); + if (!existing || Date.parse(summary.recordedAt ?? summary.completedAt) >= Date.parse(existing.recordedAt ?? existing.completedAt)) bySource.set(key, summary); } return bySource; } diff --git a/src/lib/maintenance/discovery/partitions.mjs b/src/lib/maintenance/discovery/partitions.mjs index d2b249a3..6066d605 100644 --- a/src/lib/maintenance/discovery/partitions.mjs +++ b/src/lib/maintenance/discovery/partitions.mjs @@ -5,14 +5,15 @@ // partition is executed as one or more `observeWalkForest` calls. import fs from 'node:fs'; import path from 'node:path'; +import { fileId, sameFileId, statMtimeMs } from '../../file-identity.mjs'; const DEFAULT_MAX_FANOUT = 64; function stampFor(target, fsImpl) { try { - const stat = fsImpl.lstatSync(target); + const stat = fsImpl.lstatSync(target, { bigint: true }); return { - target, mtimeMs: stat.mtimeMs, ino: Number(stat.ino) || null, size: stat.isFile() ? stat.size : null, + target, mtimeMs: statMtimeMs(stat), ino: fileId(stat.ino), size: stat.isFile() ? Number(stat.size) : null, }; } catch (error) { return { target, mtimeMs: null, ino: null, size: null, reason: error?.code ?? 'io' }; @@ -110,6 +111,6 @@ export function mergeWalkResults(results) { export function partitionDrifted(partition, { fsImpl = fs } = {}) { return partition.sourceStamps.some((recorded) => { const current = stampFor(recorded.target, fsImpl); - return current.mtimeMs !== recorded.mtimeMs || current.ino !== recorded.ino || current.size !== recorded.size; + return current.mtimeMs !== recorded.mtimeMs || !sameFileId(current.ino, recorded.ino) || current.size !== recorded.size; }); } diff --git a/src/lib/maintenance/management/activity.mjs b/src/lib/maintenance/management/activity.mjs index 7c4093e8..531a1287 100644 --- a/src/lib/maintenance/management/activity.mjs +++ b/src/lib/maintenance/management/activity.mjs @@ -71,14 +71,26 @@ function recipeEventSummary(event) { return { kind: event.kind, recipeId: event.recipeId ?? null, recipeVersion: event.recipeVersion ?? null, at: event.at, pendingIds: event.pendingIds ?? undefined }; } +export function validScanTime(value) { + if (typeof value !== 'string' || value.length > 40) return null; + const parts = /^(\d{4})-(\d{2})-(\d{2})T(\d{2}):(\d{2}):(\d{2})(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})$/.exec(value); + if (!parts) return null; + const [, yearText, monthText, dayText, hourText, minuteText, secondText] = parts; + const [year, month, day, hour, minute, second] = [yearText, monthText, dayText, hourText, minuteText, secondText].map(Number); + const leap = year % 4 === 0 && (year % 100 !== 0 || year % 400 === 0); + const monthDays = [31, leap ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31]; + return day >= 1 && day <= monthDays[month - 1] && hour <= 23 && minute <= 59 && second <= 59 + && Number.isFinite(Date.parse(value)) ? value : null; +} + function scanSummary(entry) { if (!SCAN_STATES.includes(entry.state)) throw new TypeError(`unknown scan state: ${entry.state}`); - return { sourceId: entry.sourceId, environmentId: entry.environmentId, state: entry.state, label: entry.label, visited: entry.visited, limitingReason: entry.limitingReason ?? null, completedAt: entry.completedAt ?? null }; + return { sourceId: entry.sourceId, environmentId: entry.environmentId, state: entry.state, label: entry.label, visited: entry.visited, limitingReason: entry.limitingReason ?? null, recordedAt: validScanTime(entry.recordedAt), completedAt: validScanTime(entry.completedAt) }; } function latestScans(scanHistory) { const latest = new Map(); - const timestamp = (entry) => Date.parse(entry.completedAt) || 0; + const timestamp = (entry) => Date.parse(entry.recordedAt ?? entry.completedAt) || 0; for (const entry of scanHistory.map(scanSummary)) { const key = JSON.stringify([entry.environmentId, entry.sourceId]); const previous = latest.get(key); diff --git a/src/lib/maintenance/management/focus-navigation.mjs b/src/lib/maintenance/management/focus-navigation.mjs index 5197f615..9c8fb858 100644 --- a/src/lib/maintenance/management/focus-navigation.mjs +++ b/src/lib/maintenance/management/focus-navigation.mjs @@ -15,7 +15,7 @@ function projectDescriptor(placement, kind) { return { projectKind: kind ?? 'unknown', languages: placement.projectLanguages ?? [], repositoryId: placement.repositoryId ?? null, repositoryLabel: placement.repositoryLabel ?? null, repositoryEvidence: placement.repositoryEvidence ?? null, repositoryObservedAt: placement.repositoryObservedAt ?? null, - sessionOrigins: placement.sessionOrigins ?? [] }; + sessionOrigins: placement.sessionOrigins ?? [], sessionSurfaces: placement.sessionSurfaces ?? null }; } function descriptor(level, placement, resource, projectLabels, projectKinds) { diff --git a/src/lib/maintenance/management/projection-projects.mjs b/src/lib/maintenance/management/projection-projects.mjs index 84d1aaec..6155b005 100644 --- a/src/lib/maintenance/management/projection-projects.mjs +++ b/src/lib/maintenance/management/projection-projects.mjs @@ -1,3 +1,4 @@ +import { mergeSessionSurfaces } from '../../footprint/session-surfaces.mjs'; import { projectLanguages } from './project-languages.mjs'; // ADR-0048 project identity, breadcrumb, worktree/repository grouping, and // instruction-context-file mapping — split out of projection.mjs purely to @@ -71,11 +72,12 @@ export function enrichProjectPresentation(builder, registry, rows, { installatio builder.upsertResource(entry.repositoryResourceId, { kind: 'related-storage', displayName: `${entry.repositoryLabel} repository` }); } if (['git', 'worktree', 'folder'].includes(row.repository?.kind)) entry.projectKind = row.repository.kind; + if (Array.isArray(row.sessionSurfaces)) entry.sessionSurfaces = mergeSessionSurfaces(row.sessionSurfaces); if (Array.isArray(row.sessionOrigins)) entry.sessionOrigins = row.sessionOrigins .filter((origin) => ['claude-desktop', 'codex-desktop', 'unknown'].includes(origin.origin) - && Number.isInteger(origin.sessions) && origin.sessions > 0) + && Number.isInteger(origin.sessions) && origin.sessions >= 0) .map(({ origin, sessions, countBasis }) => ({ origin, sessions, - ...(['transcript-files', 'database-sessions', 'recovered-project-sighting', 'mixed-observations'].includes(countBasis) ? { countBasis } : {}), + ...(['declared-session-ids', 'transcript-files', 'database-sessions', 'recovered-project-sighting', 'mixed-observations'].includes(countBasis) ? { countBasis } : {}), })); } } @@ -89,6 +91,7 @@ export function projectPresentation(entry) { repositoryEvidence: entry?.repositoryEvidence ?? (entry?.repositoryResourceId ? 'project-discovery' : null), repositoryObservedAt: entry?.repositoryObservedAt ?? null, sessionOrigins: entry?.sessionOrigins ?? [], + sessionSurfaces: entry?.sessionSurfaces ?? null, }; } diff --git a/src/lib/maintenance/management/query.mjs b/src/lib/maintenance/management/query.mjs index 84e7519a..cac14c61 100644 --- a/src/lib/maintenance/management/query.mjs +++ b/src/lib/maintenance/management/query.mjs @@ -267,14 +267,23 @@ function channelValues(placement, index) { return placement.versions?.channel ? [placement.versions.channel] : []; } +// Existing saved filters address legacy origin membership. Keep the coarse +// origin keys independently labelled even when a refresh adds precise surfaces. +function sessionOriginFacetValues(placement) { + if (!placement.projectId) return []; + const legacy = (placement.sessionOrigins ?? []).filter((entry) => entry.sessions > 0).map((entry) => entry.origin); + if (!Array.isArray(placement.sessionSurfaces)) return [...new Set(placement.sessionOrigins?.length ? legacy : ['unknown'])]; + return [...new Set([...placement.sessionSurfaces.filter((entry) => entry.sessions > 0).map((entry) => entry.surface === 'unknown' ? 'surface-unknown' : entry.surface), + ...(placement.sessionOrigins?.length ? legacy : ['unknown'])])]; +} + const FACET_EXTRACTORS = Object.freeze({ family: (placement, index) => [familyFor(placement, index)], scope: (placement) => [placement.administrativeScope], environment: (placement) => [placement.environmentId], project: (placement) => (placement.projectId ? [placement.projectId] : []), projectType: (placement) => placement.projectId ? [PROJECT_KINDS.includes(placement.projectKind) ? placement.projectKind : 'unknown'] : [], - sessionOrigin: (placement) => placement.projectId - ? [...new Set(placement.sessionOrigins?.length ? placement.sessionOrigins.map((entry) => entry.origin) : ['unknown'])] : [], + sessionOrigin: sessionOriginFacetValues, kind: (placement) => [placement.kind], consumer: (placement) => (placement.consumerHosts ?? []).filter((host) => ['claude', 'codex', 'opencode'].includes(host)), adapter: (placement) => [...new Set([...(placement.consumerHosts ?? []), ...(placement.kind === 'host-adapter' ? [placement.hostNamespace] : [])].filter((host) => typeof host === 'string' && host && !['claude', 'codex', 'opencode', 'agentic-kit'].includes(host)))], @@ -406,7 +415,7 @@ function buildPlacementRow(placement, index) { ...(placement.projectId ? { projectKind: PROJECT_KINDS.includes(placement.projectKind) ? placement.projectKind : 'unknown' } : {}), ...(placement.projectId ? { repositoryId: placement.repositoryId ?? null, repositoryLabel: placement.repositoryLabel ?? null, repositoryEvidence: placement.repositoryEvidence ?? null, repositoryObservedAt: placement.repositoryObservedAt ?? null, - sessionOrigins: placement.sessionOrigins ?? [] } : {}), + sessionOrigins: placement.sessionOrigins ?? [], sessionSurfaces: placement.sessionSurfaces ?? null } : {}), displayName: placement.displayName, ...(index.resourcesById.get(placement.resourceId)?.installationSource ? { installationSource: index.resourcesById.get(placement.resourceId).installationSource } : {}), ...(description ? { description } : {}), diff --git a/src/lib/maintenance/management/service-discovery.mjs b/src/lib/maintenance/management/service-discovery.mjs index 4339e33b..78e94538 100644 --- a/src/lib/maintenance/management/service-discovery.mjs +++ b/src/lib/maintenance/management/service-discovery.mjs @@ -391,7 +391,7 @@ export function scanProgress(ctx) { coverage, progress: ctx.orchestrator().progress(), narrative: progressNarrative(filesystemCoverage, { totalSources: filesystemCoverage.length }), - evidenceChecks: coverage.filter((entry) => entry.filesystem === false).map((entry) => ({ sourceId: entry.sourceId, label: entry.label, method: 'Refresh evidence' })), + evidenceChecks: coverage.filter((entry) => entry.filesystem === false).map((entry) => ({ sourceId: entry.sourceId, label: entry.label, method: 'Refresh' })), forbiddenClaims: claimsAllowed(filesystemCoverage), }; }; diff --git a/src/lib/maintenance/service.mjs b/src/lib/maintenance/service.mjs index b0228f86..1c00ba59 100644 --- a/src/lib/maintenance/service.mjs +++ b/src/lib/maintenance/service.mjs @@ -58,7 +58,7 @@ function scanRequiredModel({ status = 'not-scanned', now = Date.now } = {}) { observedUsage: { status: 'not-measured', statement: 'Usage evidence is unavailable.' }, consumerHosts: { basis: 'not-measured', hosts: [], count: 0, truncated: false }, impact: { summary: 'Scanning changes no installed resource.', bytes: null, files: null, dependencies: 'unknown', preserved: ['All installed resources'] }, - nextAction: { operation: 'scan', label, providerId: 'system.deep-scan', providerVersion: '1', safetyClass: 'never-automatic', rollback: 'reversible', restart: 'not-required', executable: false, recommendation: label, steps: ['Run `ak maintain --refresh=machine` to measure the machine.', 'Return to Maintenance when the scan completes.'], preserved: ['All installed resources'], blockedReason: 'A current saved scan is required before Maintenance can recommend changes.' }, + nextAction: { operation: 'scan', label, providerId: 'system.deep-scan', providerVersion: '1', safetyClass: 'never-automatic', rollback: 'reversible', restart: 'not-required', executable: false, recommendation: label, steps: ['Run `ak maintain --refresh=machine` or Refresh › Machine in the dashboard to measure the machine.', 'Return to Maintenance when the scan completes.'], preserved: ['All installed resources'], blockedReason: 'A current saved scan is required before Maintenance can recommend changes.' }, }; return deepFreeze({ schemaVersion: 1, mode: 'control-plane', capabilities: NO_CONTROL_CAPABILITIES, diff --git a/src/lib/mcp-probe.mjs b/src/lib/mcp-probe.mjs index dcdc4416..d7302370 100644 --- a/src/lib/mcp-probe.mjs +++ b/src/lib/mcp-probe.mjs @@ -2,6 +2,7 @@ // repository content, environment values, or stderr are returned in receipts. import { spawn } from 'node:child_process'; import { resolveShim, killProcessTree } from './exec.mjs'; +import { mergeWindowsEnv } from './windows-npm-shim.mjs'; /** @param {{command:string,args?:string[],cwd?:string,env?:NodeJS.ProcessEnv,timeoutMs?:number}} options */ export function probeMcp({ command, args = [], cwd, env = {}, timeoutMs = 30_000 }) { @@ -12,7 +13,7 @@ export function probeMcp({ command, args = [], cwd, env = {}, timeoutMs = 30_000 } return new Promise((resolve) => { const started = performance.now(); - const mergedEnv = { ...process.env, ...env }; + const mergedEnv = process.platform === 'win32' ? mergeWindowsEnv(process.env, env) : { ...process.env, ...env }; const invocation = resolveShim(command, args, { env: mergedEnv }); const child = spawn(invocation.command, invocation.args, { cwd, env: mergedEnv, shell: false, detached: process.platform !== 'win32', diff --git a/src/lib/mcp-tool-call.mjs b/src/lib/mcp-tool-call.mjs index 7b83678d..26154fb9 100644 --- a/src/lib/mcp-tool-call.mjs +++ b/src/lib/mcp-tool-call.mjs @@ -14,6 +14,7 @@ // shim's node process). import { spawn } from 'node:child_process'; import { resolveShim, killProcessTree } from './exec.mjs'; +import { mergeWindowsEnv } from './windows-npm-shim.mjs'; const MAX_OUTPUT_BYTES = 2 * 1024 * 1024; /** How long the killed server may take to exit before the call reports @@ -44,7 +45,7 @@ export async function callMcpTools({ }) { validate({ command, args, calls, timeoutMs, exitGraceMs }); return new Promise((resolve) => { - const merged = { ...process.env, ...env }; + const merged = process.platform === 'win32' ? mergeWindowsEnv(process.env, env) : { ...process.env, ...env }; const invocation = resolveShim(command, args, { env: merged }); const child = spawn(invocation.command, invocation.args, { cwd, env: merged, shell: false, detached: process.platform !== 'win32', diff --git a/src/lib/paths.mjs b/src/lib/paths.mjs index 393fb420..134cd45b 100644 --- a/src/lib/paths.mjs +++ b/src/lib/paths.mjs @@ -10,14 +10,20 @@ import { writePrivateFileAtomic } from './file-write.mjs'; const home = os.homedir(); const isWindows = process.platform === 'win32'; +/** The XDG Base Directory spec ignores relative environment overrides. */ +export function xdgBase(name, fallback, { env = process.env, p = path } = {}) { + const value = env[name]; + return value && p.isAbsolute(value) ? value : fallback; +} + /** Kit config dir: XDG on POSIX, %APPDATA% on Windows. */ function configBase() { if (isWindows) return process.env.APPDATA || path.join(home, 'AppData', 'Roaming'); - return process.env.XDG_CONFIG_HOME || path.join(home, '.config'); + return xdgBase('XDG_CONFIG_HOME', path.join(home, '.config')); } -function stateBase() { - if (isWindows) return process.env.LOCALAPPDATA || path.join(home, 'AppData', 'Local'); - return process.env.XDG_STATE_HOME || path.join(home, '.local', 'state'); +export function stateBase({ env = process.env, home: h = home, platform = process.platform, p = path } = {}) { + if (platform === 'win32') return env.LOCALAPPDATA || p.join(h, 'AppData', 'Local'); + return xdgBase('XDG_STATE_HOME', p.join(h, '.local', 'state'), { env, p }); } export const configDir = () => path.join(configBase(), 'agentic-kit'); export const telemetryDir = () => path.join(configDir(), 'telemetry'); @@ -98,8 +104,10 @@ export function toolInternalDirs({ home: h = home, env = process.env, platform = p.join(h, '.claude'), env.CLAUDE_CONFIG_DIR, p.join(h, '.codex'), env.CODEX_HOME, p.join(h, '.claude-flow'), p.join(h, '.ruflo'), - p.join(h, '.config'), env.XDG_CONFIG_HOME, p.join(h, '.local'), env.XDG_DATA_HOME, - env.XDG_STATE_HOME, p.join(h, '.cache'), env.XDG_CACHE_HOME, + p.join(h, '.config'), xdgBase('XDG_CONFIG_HOME', null, { env, p }), + p.join(h, '.local'), xdgBase('XDG_DATA_HOME', null, { env, p }), + xdgBase('XDG_STATE_HOME', null, { env, p }), p.join(h, '.cache'), + xdgBase('XDG_CACHE_HOME', null, { env, p }), ]; if (platform === 'win32') dirs.push(p.join(h, 'AppData'), env.APPDATA, env.LOCALAPPDATA); if (platform === 'darwin') dirs.push(p.join(h, 'Library', 'Application Support'), p.join(h, 'Library', 'Caches')); @@ -349,8 +357,8 @@ export function hostHealthInputPaths(cwd, env = process.env) { path.join(codex, 'requirements.toml'), '/etc/codex/config.toml', '/etc/codex/requirements.toml', ...(process.platform === 'win32' ? [path.join(env.ProgramData || 'C:\\ProgramData', 'OpenAI', 'Codex', 'config.toml')] : []), path.join(opencode, 'config.json'), path.join(opencode, 'opencode.json'), path.join(opencode, 'opencode.jsonc'), - path.join(env.XDG_STATE_HOME || path.join(home, '.local', 'state'), 'opencode', 'model.json'), - path.join(env.XDG_DATA_HOME || path.join(home, '.local', 'share'), 'opencode', 'auth.json'), + path.join(xdgBase('XDG_STATE_HOME', path.join(home, '.local', 'state'), { env }), 'opencode', 'model.json'), + path.join(xdgBase('XDG_DATA_HOME', path.join(home, '.local', 'share'), { env }), 'opencode', 'auth.json'), env.OPENCODE_CONFIG, ].filter(Boolean); let root = path.resolve(cwd); diff --git a/src/lib/pricing.mjs b/src/lib/pricing.mjs index 7aad7833..7e175181 100644 --- a/src/lib/pricing.mjs +++ b/src/lib/pricing.mjs @@ -328,6 +328,9 @@ const tokens = (v) => (Number.isFinite(v) && v > 0 ? v : 0); */ export function costOf(usage) { const { model, provider, input, output, cacheRead, cacheWrite, cacheWrite1h, day } = usage ?? {}; + // The Auto-review alias selects a server-side model whose price is not + // published. A generic unknown-model fallback would manufacture dollars. + if (model === 'codex-auto-review') return 0; const { in: rin, out: rout, cacheReadMultiplier, cacheWriteMultiplier, cacheWrite1hMultiplier } = priceFor(model, provider, day); const writes = tokens(cacheWrite); const writes1h = Math.min(tokens(cacheWrite1h), writes); diff --git a/src/lib/project-census.mjs b/src/lib/project-census.mjs index 063209b2..42673ec2 100644 --- a/src/lib/project-census.mjs +++ b/src/lib/project-census.mjs @@ -38,6 +38,7 @@ // is the whole point of this module and no surface should render a project // count without one. import fs from 'node:fs'; +import { mergeSessionSurfaces } from './footprint/session-surfaces.mjs'; import path from 'node:path'; import os from 'node:os'; import { discoverProjectSources } from './footprint/project-sources.mjs'; @@ -172,6 +173,7 @@ function mergeByIdentity(rows) { existing.hosts = [...new Set([...(existing.hosts ?? []), ...(row.hosts ?? [])])]; existing.learningState = [...new Set([...(existing.learningState ?? []), ...(row.learningState ?? [])])]; existing.learningOrigins = [...new Set([...(existing.learningOrigins ?? []), ...(row.learningOrigins ?? [])])].sort(); + existing.sessionSurfaces = mergeSessionSurfaces([...(existing.sessionSurfaces ?? []), ...(row.sessionSurfaces ?? [])]); existing.sessions = (existing.sessions ?? 0) + (row.sessions ?? 0); if ((row.lastSeenMs ?? -1) > (existing.lastSeenMs ?? -1)) existing.lastSeenMs = row.lastSeenMs; // Prefer the shallowest path that carries learning state: a repo root over diff --git a/src/lib/project-memory.mjs b/src/lib/project-memory.mjs index e42dbe3b..07856e85 100644 --- a/src/lib/project-memory.mjs +++ b/src/lib/project-memory.mjs @@ -143,24 +143,34 @@ export function removeMemoryProbe(root, namespace, key) { // aqe a .agentic-qe/ folder below the project root (AQE resolves a // relative AQE_MEMORY_PATH against the folder it runs in) // Bounded: at most `maxDirs` folders listed and `maxDepth` levels deep; dot -// folders (other checkouts under .claude/worktrees, .git) and node_modules are -// never walked, only a root dot folder's own markers are checked. A folder that +// tool homes (including other checkouts under .claude/worktrees), .git and +// node_modules are never walked. Ordinary dot folders are walked. A folder that // holds `.git` (nested repository, submodule, worktree inside the checkout) is // another repository: neither it nor anything below it is searched; it is // listed in `nestedRepositories`. const RUFLO_STORE_FILES = Object.freeze(['memory.db', 'agentdb-memory.db']); const ROOT_STRAYS = Object.freeze([['agentdb.db', 'agentdb-cli'], ['agentdb.rvf', 'agentdb-rvf'], ['ruvector.db', 'ruvector']]); +const SCAN_EXCLUDED_DIRS = new Set(['.git', '.swarm', '.agentic-qe', '.claude', '.codex', '.claude-flow', '.agents', '.harness', 'node_modules']); -const isDirectory = (file) => { try { return fs.lstatSync(file).isDirectory(); } catch { return false; } }; const isFile = (file) => { try { return fs.lstatSync(file).isFile(); } catch { return false; } }; const storeBytes = (file) => (fileBytes(file) ?? 0) + (fileBytes(`${file}-wal`) ?? 0); export function findStrayMemoryStores(root, { maxDepth = 4, maxDirs = 2000 } = {}) { - if (!isDirectory(root)) return { strays: [], complete: true, visited: 0, nestedRepositories: [] }; + let rootStat; + try { rootStat = fs.lstatSync(root); } catch (e) { + return { strays: [], complete: e.code === 'ENOENT', visited: 0, nestedRepositories: [] }; + } + if (!rootStat.isDirectory()) return { strays: [], complete: true, visited: 0, nestedRepositories: [] }; const found = new Map(); const nestedRepositories = []; let visited = 0; let complete = true; + const stat = (file) => { + try { return fs.lstatSync(file); } catch (e) { + if (e.code !== 'ENOENT') complete = false; + return null; + } + }; const add = (kind, file, sizeBytes) => { const relative = path.relative(root, file).split(path.sep).join('/'); found.set(relative, { kind, path: relative, file, sizeBytes }); @@ -170,14 +180,14 @@ export function findStrayMemoryStores(root, { maxDepth = 4, maxDirs = 2000 } = { visited += 1; try { return fs.readdirSync(dir, { withFileTypes: true }).sort((a, b) => a.name.localeCompare(b.name)); - } catch { return []; } + } catch { complete = false; return []; } }; const checkMarkers = (dir, { ruflo = true } = {}) => { - if (isDirectory(path.join(dir, '.agentic-qe'))) add('aqe', path.join(dir, '.agentic-qe'), null); + if (stat(path.join(dir, '.agentic-qe'))?.isDirectory()) add('aqe', path.join(dir, '.agentic-qe'), null); if (!ruflo) return; for (const name of RUFLO_STORE_FILES) { const file = path.join(dir, '.swarm', name); - if (isFile(file)) add('ruflo', file, storeBytes(file)); + if (stat(file)?.isFile()) add('ruflo', file, storeBytes(file)); } }; const walkSwarm = (dir, depth) => { @@ -192,7 +202,7 @@ export function findStrayMemoryStores(root, { maxDepth = 4, maxDirs = 2000 } = { // a worktree inside the checkout) is another repository: its stores are its // own, and ak's pin makes its `.agentic-qe` that repository's store (review M3). const otherRepository = (full) => { - try { fs.lstatSync(path.join(full, '.git')); } catch { return false; } + if (!stat(path.join(full, '.git'))) return false; nestedRepositories.push(path.relative(root, full).split(path.sep).join('/')); return true; }; @@ -200,25 +210,26 @@ export function findStrayMemoryStores(root, { maxDepth = 4, maxDirs = 2000 } = { const entries = list(dir); for (const entry of entries ?? []) { if (!entry.isDirectory()) continue; + if (entry.name === 'node_modules') continue; const full = path.join(dir, entry.name); if (entry.name !== '.git' && otherRepository(full)) continue; - if (entry.name.startsWith('.')) { - // .swarm's own subtree is walked separately; only its AQE marker here. - if (depth === 0 && entry.name !== '.git') checkMarkers(full, { ruflo: entry.name !== '.swarm' }); - continue; - } - if (entry.name === 'node_modules') continue; - checkMarkers(full); + if (entry.name !== '.git') checkMarkers(full, { ruflo: entry.name !== '.swarm' }); + if (SCAN_EXCLUDED_DIRS.has(entry.name)) continue; if (depth + 1 < maxDepth) walk(full, depth + 1); + else { + // A marker at the depth boundary is visible, but descendants are not. + const children = list(full); + if (children?.some((child) => child.isDirectory() && !SCAN_EXCLUDED_DIRS.has(child.name))) complete = false; + } } }; for (const [name, kind] of ROOT_STRAYS) { const file = path.join(root, name); - if (isFile(file)) add(kind, file, storeBytes(file)); + if (stat(file)?.isFile()) add(kind, file, storeBytes(file)); } // .swarm first: it is small, and the most likely home of a stray Ruflo store. - if (isDirectory(path.join(root, '.swarm'))) walkSwarm(path.join(root, '.swarm'), 0); + if (stat(path.join(root, '.swarm'))?.isDirectory()) walkSwarm(path.join(root, '.swarm'), 0); walk(root, 0); const strays = [...found.values()].sort((a, b) => a.path.localeCompare(b.path)); return { strays, complete, visited, nestedRepositories: nestedRepositories.sort() }; diff --git a/src/lib/quota.mjs b/src/lib/quota.mjs index 731577db..b520245d 100644 --- a/src/lib/quota.mjs +++ b/src/lib/quota.mjs @@ -32,7 +32,7 @@ import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { spawn } from 'node:child_process'; -import { configDir, claudeSettingsPath } from './paths.mjs'; +import { configDir, claudeSettingsPath, claudeManagedSettingsPath } from './paths.mjs'; import { managedHostIds } from './adapters/registries.mjs'; import { recordedHostPresence } from './providers.mjs'; @@ -114,9 +114,9 @@ export function readClaudeLimits({ file = claudeLimitsFile() } = {}) { // command line, then the project's .claude/settings.local.json, then its // .claude/settings.json, then the user's ~/.claude/settings.json // (code.claude.com/docs/en/settings). The dashboard cannot know which project -// the next session starts in, so it classifies the USER-level statusLine, the -// one every project without its own inherits, and lets the panel state the -// precedence rule beside the class. +// the next session starts in, so it classifies the machine-managed file when +// present and otherwise the USER-level statusLine, the one every project +// without its own inherits. Other managed policy channels are not observed. // // Read-only and path-free: this reads the settings file and, at most, the // script files the command names (stat first; regular files under a size cap), @@ -127,22 +127,33 @@ export function readClaudeLimits({ file = claudeLimitsFile() } = {}) { export const CLAUDE_TEE_CHANNELS = Object.freeze(['none', 'kit-footer', 'project-helper', 'custom', 'unknown']); const KIT_FOOTER_MARKER = 'ruflo-seg:BEGIN'; const MAX_STATUSLINE_SCRIPT_BYTES = 4 * 1024 * 1024; -const MAX_STATUSLINE_SCRIPTS = 8; -// A script the command names: double-quoted, single-quoted (both may hold -// spaces), or a bare token. Only the JavaScript family, since the tee is JS. -const STATUSLINE_SCRIPT_TOKEN = /"([^"]*?\.[cm]?js)"|'([^']*?\.[cm]?js)'|([^\s"'`;|&()=,]+?\.[cm]?js)(?![\w.])/g; const HOME_PREFIX = /^(?:~|\$HOME|\$\{HOME\}|%USERPROFILE%|%HOME%)(?=[\\/]|$)/; // The one project-relative script the kit injects into (fixStatusline). const PROJECT_HELPER_SUFFIX = '.claude/helpers/statusline.cjs'; +// The two generated project-helper commands already supported by the classifier. +// They are known templates, not evidence that arbitrary shell text runs a helper. +const PROJECT_HELPER_COMMANDS = new Set([ + 'sh -c \'D="${CLAUDE_PROJECT_DIR:-.}"; [ -f "$D/.claude/helpers/statusline.cjs" ] || D="${HOME}"; exec node "$D/.claude/helpers/statusline.cjs"\'', + 'node -e "const fs=require(\'fs\'),p=require(\'path\');const d=process.env.CLAUDE_PROJECT_DIR||\'.\';const f=p.join(d,\'.claude/helpers/statusline.cjs\');const h=p.join(process.env.USERPROFILE||process.env.HOME||\'.\', \'.claude/helpers/statusline.cjs\');require(fs.existsSync(f)?f:h);"', +]); -function statusLineScripts(command) { - const out = []; - for (const m of command.matchAll(STATUSLINE_SCRIPT_TOKEN)) { - const token = (m[1] ?? m[2] ?? m[3] ?? '').trim(); - if (token && !out.includes(token)) out.push(token); - if (out.length >= MAX_STATUSLINE_SCRIPTS) break; - } - return out; +function directStatusLineScript(command) { + // Only a direct Node script (or a directly executable JS script) proves + // which file runs. Do not scan arguments, inline programs, shell chains, or + // wrapper bodies for plausible paths; none establishes the executed target. + const direct = command.trim().match(/^(?:(?:node|node\.exe)(?:\s+--no-warnings)?\s+)?("[^"]+"|'[^']+'|[^\s"']+\.([cm]?js))$/i); + if (!direct) return null; + const raw = direct[1]; + const quote = raw[0] === '"' || raw[0] === "'" ? raw[0] : null; + const token = quote ? raw.slice(1, -1) : raw; + if (!/\.[cm]?js$/i.test(token) || token.startsWith('-')) return null; + // A quoted path keeps literal punctuation. In bare command text, shell + // operators, globs, and a leading comment marker do not name a script. + if (!quote && (token.startsWith('#') + || token.includes('[') || token.includes(']') + || /[`;|&()<>{}*?]/.test(token.replace(/^\$\{HOME\}/, '$HOME')))) return null; + if (quote === '"' && (token.includes('`') || token.includes('$('))) return null; + return token; } function scriptCarriesFooter(file, fsImpl) { @@ -166,27 +177,54 @@ function scriptChannel(token, { fsImpl, home }) { return scriptCarriesFooter(expanded, fsImpl) ? 'kit-footer' : 'custom'; } +function managedStatusLine(file, fsImpl) { + if (!file) return { state: 'absent' }; + let managed; + try { managed = JSON.parse(fsImpl.readFileSync(file, 'utf8')); } + catch (error) { return { state: error?.code === 'ENOENT' ? 'absent' : 'unknown' }; } + if (!managed || typeof managed !== 'object' || Array.isArray(managed)) return { state: 'unknown' }; + if (!Object.hasOwn(managed, 'statusLine')) return { state: 'absent' }; + const line = managed.statusLine; + // A present null/false value is not a documented way to disable statusLine. + // Do not infer effective policy or fall through to a lower-precedence footer. + if (line == null || line === false) return { state: 'unknown' }; + if (!line || typeof line !== 'object' || Array.isArray(line) + || (line.type !== undefined && line.type !== 'command') + || typeof line.command !== 'string' || !line.command.trim()) return { state: 'unknown' }; + return { state: 'present', settings: managed }; +} + /** - * Classify the user-level Claude statusLine by whether it can feed the quota - * tee: 'none' (no user-level statusLine), 'kit-footer' (its script carries the + * Classify the effective local managed/user Claude statusLine by whether it can + * feed the quota tee: 'none' (no statusLine), 'kit-footer' (its script carries the * footer), 'project-helper' (it runs each project's ruflo helper, so it depends * on the project), 'custom' (anything else: another script, an inline command, - * a missing file), or 'unknown' (the settings file exists but cannot be read). + * a missing file), or 'unknown' (a relevant settings file cannot be read or + * interpreted). This does not observe managed-settings.d drop-ins, server, + * MDM, or SDK managed policy. * - * @param {{ settingsFile?: string, fsImpl?: any, home?: string }} [o] + * @param {{ settingsFile?: string, managedSettingsFile?: string|null, + * fsImpl?: any, home?: string, platform?: NodeJS.Platform }} [o] * @returns {'none'|'kit-footer'|'project-helper'|'custom'|'unknown'} */ export function classifyClaudeTeeChannel({ - settingsFile = claudeSettingsPath(), fsImpl = fs, home = os.homedir(), + settingsFile = claudeSettingsPath(), managedSettingsFile, + fsImpl = fs, home = os.homedir(), platform = process.platform, } = {}) { - let settings; - try { settings = JSON.parse(fsImpl.readFileSync(settingsFile, 'utf8')); } - catch (error) { return error?.code === 'ENOENT' ? 'none' : 'unknown'; } + const managedFile = managedSettingsFile === undefined + ? claudeManagedSettingsPath(platform) : managedSettingsFile; + const managed = managedStatusLine(managedFile, fsImpl); + if (managed.state === 'unknown') return 'unknown'; + let settings = managed.settings; + if (!settings) { + try { settings = JSON.parse(fsImpl.readFileSync(settingsFile, 'utf8')); } + catch (error) { return error?.code === 'ENOENT' ? 'none' : 'unknown'; } + } const command = settings?.statusLine?.command; if (typeof command !== 'string' || !command.trim()) return 'none'; - const classes = statusLineScripts(command).map((token) => scriptChannel(token, { fsImpl, home })); - if (classes.includes('kit-footer')) return 'kit-footer'; - return classes.includes('project-helper') ? 'project-helper' : 'custom'; + if (PROJECT_HELPER_COMMANDS.has(command.trim())) return 'project-helper'; + const script = directStatusLineScript(command); + return script ? scriptChannel(script, { fsImpl, home }) : 'custom'; } // ── Codex (app-server) ────────────────────────────────────────────────────── @@ -461,15 +499,19 @@ export function unsupportedQuotaHosts({ enabledHosts = {} } = {}) { * @param {{ now?: number, claudeFile?: string, codexCacheFile?: string, ttlMs?: number, * timeoutMs?: number, spawnImpl?: any, bin?: string, * enabledHosts?: Record, - * claudeSettingsFile?: string, home?: string, + * claudeSettingsFile?: string, claudeManagedSettingsFile?: string|null, + * home?: string, * codexPresence?: () => 'found'|'not-found'|'unconfirmed' }} [o] */ export async function readLimits({ now = Date.now(), claudeFile, codexCacheFile, ttlMs, timeoutMs, spawnImpl, bin, enabledHosts, - claudeSettingsFile, home, codexPresence, + claudeSettingsFile, claudeManagedSettingsFile, home, codexPresence, } = {}) { const claude = readClaudeLimits({ file: claudeFile ?? claudeLimitsFile() }); - const claudeChannel = classifyClaudeTeeChannel({ settingsFile: claudeSettingsFile ?? claudeSettingsPath(), home }); + const claudeChannel = classifyClaudeTeeChannel({ + settingsFile: claudeSettingsFile ?? claudeSettingsPath(), + managedSettingsFile: claudeManagedSettingsFile, home, + }); const presence = (codexPresence ?? (() => recordedHostPresence('codex', { now })))(); const { limits: codex, unavailable: codexUnavailable } = presence === 'found' ? await collectCodexLimitsDetailed({ diff --git a/src/lib/ruflo-components/snapshot.mjs b/src/lib/ruflo-components/snapshot.mjs index 2ab54c6d..69d92b09 100644 --- a/src/lib/ruflo-components/snapshot.mjs +++ b/src/lib/ruflo-components/snapshot.mjs @@ -230,7 +230,7 @@ export function componentSnapshot({ cfg, rufloVersion, evidence, projection, now } /** Controller ruling 3: the ONE read-only projection every surface (status's own - * section, and Task 10's dashboard) builds a snapshot from — so none of them + * section, and the dashboard) builds a snapshot from — so none of them * re-implements it. Never spawns anything and never writes: the Claude env read * is a dry run, and the evidence comes from whatever is already cached (a * caller wanting fresher evidence collects + writes it BEFORE calling this, diff --git a/src/lib/ruflo-daemon-config.mjs b/src/lib/ruflo-daemon-config.mjs index f517d557..81048959 100644 --- a/src/lib/ruflo-daemon-config.mjs +++ b/src/lib/ruflo-daemon-config.mjs @@ -100,10 +100,14 @@ function readDaemonConfig(file) { } } +const pathPresent = (candidate) => { + try { fs.lstatSync(candidate); return true; } catch (error) { return error?.code !== 'ENOENT'; } +}; + /** A desired key ak leaves to the user: the file is unreadable (`invalid`), * or it holds the user's own value `have` instead of `want`. * @typedef {{ key: string, want: number, have?: unknown }} HeldKey - * @typedef {{ invalid: boolean, entries: HeldKey[] } | null} HeldConfig */ + * @typedef {{ invalid: boolean, entries: HeldKey[], reason?: 'yaml-shadow'|'higher-priority-json'|'explicit-config' } | null} HeldConfig */ /** Plan the config.json edit: which owned keys to drop, which to set. */ function planConfig(current, owned, desired) { @@ -127,26 +131,64 @@ function planConfig(current, owned, desired) { return { next, nextOwned, removed, written, conflicts }; } -function reconcileConfig(root, receipt, desired, dryRun) { +/** @returns {{status: string, changed: boolean, activeChanged: boolean, held: HeldConfig}} */ +function reconcileConfig(root, receipt, desired, dryRun, env) { const file = path.join(root, DAEMON_CONFIG_RELATIVE); + // Never follow a user-controlled .claude-flow link (or special file) when + // reading, creating, replacing or removing config.json. + try { + const dir = fs.lstatSync(path.dirname(file)); + if (!dir.isDirectory() || dir.isSymbolicLink()) { + const entries = Object.entries(desired).map(([key, want]) => ({ key, want })); + return { status: 'user-managed', changed: false, activeChanged: false, held: entries.length ? { invalid: true, entries } : null }; + } + } catch (error) { + if (error?.code !== 'ENOENT') { + const entries = Object.entries(desired).map(([key, want]) => ({ key, want })); + return { status: 'user-managed', changed: false, activeChanged: false, held: entries.length ? { invalid: true, entries } : null }; + } + } + const higherPriority = pathPresent(path.join(root, 'claude-flow.config.json')); + // ConfigFileManager.findConfig checks the explicit path only after both + // JSON candidates. existsSync resolves a relative env path from process.cwd, + // exactly as Ruflo does; it does not resolve it against the project root. + const explicitConfig = !higherPriority && typeof env?.CLAUDE_FLOW_CONFIG === 'string' + && fs.existsSync(env.CLAUDE_FLOW_CONFIG); const owned = receipt.configKeys ?? {}; const current = readDaemonConfig(file); if (current.state === 'invalid') { const entries = Object.entries(desired).map(([key, want]) => ({ key, want })); - return { status: 'user-managed', changed: false, held: entries.length ? { invalid: true, entries } : null }; + return { status: 'user-managed', changed: false, activeChanged: false, held: entries.length ? { invalid: true, entries } : null }; } if (current.state === 'absent') { receipt.configKeys = {}; receipt.configCreated = false; - if (!Object.keys(desired).length) return { status: 'absent', changed: false, held: null }; + if (!Object.keys(desired).length) return { status: 'absent', changed: false, activeChanged: false, held: null }; + // Ruflo 3.48.0 chooses root JSON, then .claude-flow/config.json, then + // config.yaml/yml. Creating our JSON over YAML would hide all user YAML + // daemon values; under root JSON this file would have no effect at all. + const yaml = ['config.yaml', 'config.yml'].some((name) => pathPresent(path.join(root, '.claude-flow', name))); + if (higherPriority || explicitConfig || yaml) return { + status: 'user-managed', changed: false, activeChanged: false, + held: { + invalid: false, reason: higherPriority ? 'higher-priority-json' : explicitConfig ? 'explicit-config' : 'yaml-shadow', + entries: Object.entries(desired).map(([key, want]) => ({ key, want })), + }, + }; if (!dryRun) { writePrivateFileAtomic(file, `${JSON.stringify(desired, null, 2)}\n`); receipt.configKeys = { ...desired }; receipt.configCreated = true; } - return { status: 'written', changed: true, held: null }; + return { status: 'written', changed: true, activeChanged: true, held: null }; } - const plan = planConfig(current.value, owned, desired); + // A root JSON file wins even over existing .claude-flow/config.json. Keep + // receipted still-needed keys and clean obsolete ones, but add no new keys + // to a file the daemon does not read. + const effectiveDesired = higherPriority + ? Object.fromEntries(Object.entries(desired).filter(([key]) => Object.hasOwn(owned, key))) + : desired; + const plan = planConfig(current.value, owned, effectiveDesired); const changed = plan.written || plan.removed; const empty = Object.keys(plan.next).length === 0; if (!dryRun) { @@ -156,10 +198,13 @@ function reconcileConfig(root, receipt, desired, dryRun) { receipt.configCreated ??= false; if (changed && empty && receipt.configCreated === true) receipt.configCreated = false; } - const held = plan.conflicts.length ? { invalid: false, entries: plan.conflicts } : null; - if (plan.written) return { status: 'written', changed, held }; - if (plan.removed) return { status: 'removed', changed, held }; - return { status: held ? 'user-managed' : 'converged', changed: false, held }; + /** @type {HeldConfig} */ + const held = higherPriority && Object.keys(desired).length + ? { invalid: false, reason: 'higher-priority-json', entries: Object.entries(desired).map(([key, want]) => ({ key, want })) } + : plan.conflicts.length ? { invalid: false, entries: plan.conflicts } : null; + if (plan.written) return { status: 'written', changed, activeChanged: !higherPriority, held }; + if (plan.removed) return { status: 'removed', changed, activeChanged: !higherPriority, held }; + return { status: held ? 'user-managed' : 'converged', changed: false, activeChanged: false, held }; } function reconcileAutostart(root, receipt, wanted, dryRun) { @@ -196,25 +241,27 @@ const hasOwnership = (receipt) => Object.keys(receipt.configKeys ?? {}).length > * Converge one project's managed daemon settings. * @param {string} root the Ruflo project root * @param {{rufloVersion?: (string|null), platform?: string, receipts: Record, - * autoStart?: boolean, dryRun?: boolean, desired?: Record, runner?: unknown}} options + * autoStart?: boolean, dryRun?: boolean, desired?: Record, runner?: unknown, + * env?: NodeJS.ProcessEnv}} options * `autoStart` is kit.json rufloDaemon.autoStart !== false; `runner` is accepted and never used. - * @returns {{config: string, autostart: string, changed: boolean, held: HeldConfig}} + * @returns {{config: string, autostart: string, changed: boolean, configActiveChanged: boolean, held: HeldConfig}} * `held` names the desired keys ak cannot manage here (null when none). */ export function reconcileRufloDaemon(root, { rufloVersion = null, platform = process.platform, receipts, autoStart = true, dryRun = false, - desired = desiredDaemonKeys({ rufloVersion, platform }), + desired = desiredDaemonKeys({ rufloVersion, platform }), env = process.env, } = /** @type {any} */ ({})) { const key = path.resolve(root); const receipt = structuredClone(receipts[key] ?? {}); - const config = reconcileConfig(root, receipt, desired, dryRun); + const config = reconcileConfig(root, receipt, desired, dryRun, env); const autostart = reconcileAutostart(root, receipt, autoStart, dryRun); if (!dryRun) { if (hasOwnership(receipt)) receipts[key] = receipt; else delete receipts[key]; } return { - config: config.status, autostart: autostart.status, changed: config.changed || autostart.changed, held: config.held, + config: config.status, autostart: autostart.status, changed: config.changed || autostart.changed, + configActiveChanged: config.activeChanged, held: config.held, }; } @@ -254,16 +301,17 @@ export function daemonIntent(cfg) { * is left for Ruflo's start-on-use. Returns null outside a Ruflo project. * @param {string} cwd * @param {{cfg: any, rufloVersion?: (string|null), platform?: string, runner?: typeof run, - * alive?: (root: string) => boolean, dryRun?: boolean}} options + * alive?: (root: string) => boolean, dryRun?: boolean, env?: NodeJS.ProcessEnv}} options */ export async function applyRufloDaemon(cwd, { cfg, rufloVersion = null, platform = process.platform, runner = run, alive = projectDaemonAlive, dryRun = false, + env = process.env, }) { const root = rufloDaemonProjectRoot(cwd); if (!root) return null; const intent = daemonIntent(cfg); - const result = reconcileRufloDaemon(root, { rufloVersion, platform, receipts: intent.receipts, autoStart: intent.autoStart, dryRun }); - const configChanged = result.config === 'written' || result.config === 'removed'; + const result = reconcileRufloDaemon(root, { rufloVersion, platform, receipts: intent.receipts, autoStart: intent.autoStart, dryRun, env }); + const configChanged = result.configActiveChanged; // A config.json that keeps the floor from ak (unreadable, or the user's own // value) changed nothing a restart would pick up. const floorHeld = result.held?.entries.some((e) => e.key === MEMORY_FLOOR_KEY) ?? false; @@ -279,17 +327,17 @@ export async function applyRufloDaemon(cwd, { /** Status: the desired keys a user-managed config.json keeps from ak (a dry run). * @returns {HeldConfig} */ -export function daemonConfigHeld(root, { cfg, rufloVersion = null, platform = process.platform }) { +export function daemonConfigHeld(root, { cfg, rufloVersion = null, platform = process.platform, env = process.env }) { const intent = daemonIntent(structuredClone(cfg ?? {})); const receipts = structuredClone(intent.receipts); - return reconcileRufloDaemon(root, { rufloVersion, platform, receipts, autoStart: intent.autoStart, dryRun: true }).held; + return reconcileRufloDaemon(root, { rufloVersion, platform, receipts, autoStart: intent.autoStart, dryRun: true, env }).held; } /** Status: what a sync would change here, as phrases, or null when converged. */ -export function daemonDrift(root, { cfg, rufloVersion = null, platform = process.platform }) { +export function daemonDrift(root, { cfg, rufloVersion = null, platform = process.platform, env = process.env }) { const intent = daemonIntent(structuredClone(cfg ?? {})); const receipts = structuredClone(intent.receipts); - const r = reconcileRufloDaemon(root, { rufloVersion, platform, receipts, autoStart: intent.autoStart, dryRun: true }); + const r = reconcileRufloDaemon(root, { rufloVersion, platform, receipts, autoStart: intent.autoStart, dryRun: true, env }); if (!r.changed) return null; const parts = []; const desired = desiredDaemonKeys({ rufloVersion, platform }); diff --git a/src/lib/ruflo-memory.mjs b/src/lib/ruflo-memory.mjs index 358bdc9f..b7aa541e 100644 --- a/src/lib/ruflo-memory.mjs +++ b/src/lib/ruflo-memory.mjs @@ -36,9 +36,10 @@ import { componentById } from './ruflo-components/catalogue.mjs'; // are the same folder, as the filesystem treats them. const realOr = (p, file) => { try { return fs.realpathSync(file); } catch { return p.resolve(file); } }; const same = (p, a, b) => p.relative(a, b) === ''; +// Strict containment: equality cannot make a tool root a deeper temp boundary. const inside = (p, child, parent) => { const rel = p.relative(parent, child); - return rel === '' || (!rel.startsWith('..') && !p.isAbsolute(rel)); + return rel !== '' && !rel.startsWith('..') && !p.isAbsolute(rel); }; /** `~/…` for a folder under the home folder, else the absolute path. */ export function homeRelative(file, home = paths.home, p = path) { @@ -57,7 +58,8 @@ function unsuitableReason(dir, { home, env, platform, p }) { if (temps.some((temp) => same(p, dir, temp))) return 'a temporary folder'; const tool = paths.toolInternalDirs({ home, env, platform, p }).find((folder) => { const real = realOr(p, folder); - return inside(p, dir, real) && !temps.some((temp) => inside(p, temp, real) && inside(p, dir, temp)); + return (same(p, dir, real) || inside(p, dir, real)) + && !temps.some((temp) => inside(p, temp, real) && inside(p, dir, temp)); }); return tool ? `inside ${homeRelative(tool, home, p)}, a tool's own folder` : null; } @@ -75,14 +77,20 @@ export function rufloMemoryLocation(cwd = process.cwd(), { home = paths.home, env = process.env, platform = process.platform, p = path, } = {}) { const options = { home, env, platform, p }; - let reason = null; - for (const [kind, candidate] of [['project', paths.repoRoot(cwd, p)], ['folder', cwd]]) { - if (!candidate) continue; - const root = realOr(p, candidate); - const why = unsuitableReason(root, options); - if (!why) return { kind, root, dir: p.join(root, '.swarm'), db: paths.projectMemoryDb(root, p), reason: null }; - reason ??= why; + const repository = paths.repoRoot(cwd, p); + const projectRoot = repository && realOr(p, repository); + const projectReason = projectRoot && unsuitableReason(projectRoot, options); + if (projectRoot && !projectReason) { + return { kind: 'project', root: projectRoot, dir: p.join(projectRoot, '.swarm'), db: paths.projectMemoryDb(projectRoot, p), reason: null }; } + const folderRoot = realOr(p, cwd); + const folderReason = unsuitableReason(folderRoot, options); + if (!folderReason) { + return { kind: 'folder', root: folderRoot, dir: p.join(folderRoot, '.swarm'), db: paths.projectMemoryDb(folderRoot, p), reason: null }; + } + const reason = projectReason && !same(p, projectRoot, folderRoot) && projectReason !== folderReason + ? `${folderReason}, in a repository whose root is ${projectReason}` + : folderReason; const dir = paths.userMemoryDir(home, p); return { kind: 'user', root: dir, dir, db: p.join(dir, 'memory.db'), reason }; } diff --git a/src/lib/session-surface.mjs b/src/lib/session-surface.mjs new file mode 100644 index 00000000..bba57dd5 --- /dev/null +++ b/src/lib/session-surface.mjs @@ -0,0 +1,186 @@ +// ADR-0060 vocabulary. This pure classifier uses declarations, never a folder name. +// Keep labels here so later CLI and dashboard views can share one vocabulary. +const LABELS = Object.freeze({ + unknown: 'Unknown', + 'claude-code-cli': 'Claude Code CLI', + 'claude-code-vscode': 'Claude Code for VS Code', + 'claude-desktop': 'Claude Desktop', + cowork: 'Cowork', + 'cloud-session': 'Cloud session', + 'claude-agent-sdk': 'Claude Agent SDK', + 'claude-noninteractive': 'Non-interactive mode (claude -p)', + 'github-actions': 'GitHub Actions', + 'claude-tag': 'Claude Tag', + 'other-claude': 'Other Claude surface', + 'chatgpt-desktop-codex': 'ChatGPT desktop app · Codex', + 'chatgpt-desktop-work': 'ChatGPT desktop app · ChatGPT Work (local)', + 'codex-cli': 'Codex CLI', + 'codex-cli-exec': 'Codex CLI · non-interactive (codex exec)', + 'codex-ide': 'Codex IDE extension', + 'codex-sdk': 'Codex SDK', + 'codex-mcp': 'Codex MCP server', + 'chatgpt-work-cloud': 'ChatGPT Work (cloud)', + 'other-openai': 'Other OpenAI client', +}); + +const CLAUDE = new Map([ + ['cli', ['claude-code-cli', 'person']], + ['claude-vscode', ['claude-code-vscode', 'person']], + ['claude-desktop', ['claude-desktop', 'person']], + ['claude-desktop-3p', ['claude-desktop', 'person']], + ['local-agent', ['cowork', 'person']], ['local_agent', ['cowork', 'person']], + ['remote_cowork', ['cowork', 'person']], + ['remote', ['cloud-session', 'person']], ['remote_desktop', ['cloud-session', 'person']], + ['remote_mobile', ['cloud-session', 'person']], ['remote_projects', ['cloud-session', 'person']], + ['remote_trigger', ['cloud-session', 'automation']], + ['remote_cowork_trigger', ['cloud-session', 'automation']], + ['sdk-py', ['claude-agent-sdk', 'automation']], ['sdk-ts', ['claude-agent-sdk', 'automation']], + ['sdk-cli', ['claude-noninteractive', 'automation']], + ['claude-code-github-action', ['github-actions', 'automation']], + ['claude_in_slack', ['claude-tag', 'person']], + ['claude-in-slack', ['claude-tag', 'person']], ['claude-in-teams', ['claude-tag', 'person']], + ['mcp', ['other-claude', 'automation']], ['ssh-remote', ['other-claude', 'person']], +]); + +const CODEX = new Map([ + ['Codex Desktop', ['chatgpt-desktop-codex', 'person']], + ['codex_work_desktop', ['chatgpt-desktop-work', 'person']], + ['codex-tui', ['codex-cli', 'person']], + ['codex_exec', ['codex-cli-exec', 'automation']], + ['codex_vscode', ['codex-ide', 'person']], + ['codex_sdk_ts', ['codex-sdk', 'automation']], + ['codex_python_sdk', ['codex-sdk', 'automation']], + ['codex_work_web', ['chatgpt-work-cloud', 'person']], + ['codex_work_mobile', ['chatgpt-work-cloud', 'person']], + ['codex_work_cca', ['chatgpt-work-cloud', 'person']], + ['chatgpt_cca', ['chatgpt-work-cloud', 'person']], +]); +const CLAUDE_ATTRIBUTES = new Map([ + ['claude-desktop-3p', 'on 3P'], ['remote_desktop', 'started from Claude Desktop'], + ['remote_mobile', 'started from mobile'], ['remote_projects', 'started from a project'], + ['remote', 'started from web'], +]); +// Declared origin fields are internal enum-like tokens. Retain unfamiliar tokens +// for local detail, but drop malformed or oversized prompt-like content. +function bounded(value) { + return typeof value === 'string' && value.length <= 80 + && (value === 'Codex Desktop' || /^[A-Za-z][A-Za-z0-9_.-]*$/u.test(value)) ? value : null; +} + +function retain(rawEvidence, field, value) { + if (value !== null) rawEvidence[field] = value; +} + +export function sessionSurfaceLabel(surface) { + return Object.hasOwn(LABELS, surface) ? LABELS[surface] : LABELS.unknown; +} + +// Only provider-specific IDs recorded by this session qualify. Plain Claude +// IDs occur across hosts; an OpenAI-compatible gateway may reuse any model +// name. Never retain the raw ID here (it may contain account or route data). +// https://code.claude.com/docs/en/amazon-bedrock#pin-model-versions +// https://code.claude.com/docs/en/google-vertex-ai#pin-model-versions +const BEDROCK_MODEL = /^(?:(?:us|eu|apac|jp|au|ca|sa|us-gov|global)\.)?anthropic\.claude-(?:(?:opus|sonnet|haiku|fable)-[1-9][0-9]?(?:-[1-9][0-9]?)?|[1-9][0-9]?(?:-[1-9][0-9]?)?-(?:opus|sonnet|haiku))(?:(?:-(20[0-9]{6})-v[1-9][0-9]*(?::[0-9]+)?)|(?:-v[1-9][0-9]*(?::[0-9]+)?))?$/u; +const VERTEX_MODEL = /^claude-(?:(?:opus|sonnet|haiku|fable)-[1-9][0-9]?(?:-[1-9][0-9]?)?|[1-9][0-9]?(?:-[1-9][0-9]?)?-(?:opus|sonnet|haiku))@(20[0-9]{6})$/u; + +function validCalendarDate(value) { + if (!value) return true; + const year = Number(value.slice(0, 4)), month = Number(value.slice(4, 6)), day = Number(value.slice(6)); + const date = new Date(Date.UTC(year, month - 1, day)); + return date.getUTCFullYear() === year && date.getUTCMonth() === month - 1 && date.getUTCDate() === day; +} + +export function claudeProviderFromModelId(model) { + if (typeof model !== 'string' || model.length > 100) return null; + const bedrock = BEDROCK_MODEL.exec(model); + if (bedrock) return validCalendarDate(bedrock[1]) ? 'amazon-bedrock' : null; + const vertex = VERTEX_MODEL.exec(model); + if (vertex) return validCalendarDate(vertex[1]) ? 'google-vertex-ai' : null; + return null; +} + +/** @returns {[string, string, string[]]} */ +function claudeClassification(entrypoint, sessionKind) { + if (entrypoint === null) return ['unknown', 'unknown', []]; + const [surface, defaultInitiator] = CLAUDE.get(entrypoint) ?? ['other-claude', 'unknown']; + const initiator = ['bg', 'daemon', 'daemon-worker'].includes(sessionKind) ? 'automation' : defaultInitiator; + const attribute = CLAUDE_ATTRIBUTES.get(entrypoint); + return [surface, initiator, attribute ? [attribute] : []]; +} + +/** @returns {[string, string, string[]]} */ +function codexClassification(originator, source, threadSource, rejectedThreadSource) { + let [surface, initiator] = originator === null ? ['unknown', 'unknown'] + : CODEX.get(originator) ?? ['other-openai', 'unknown']; + if (originator === 'codex_cli_rs' && source === 'mcp') [surface, initiator] = ['codex-mcp', 'agent']; + if (surface === 'codex-mcp' || ['subagent', 'guardian_review', 'agent_created_thread'].includes(threadSource)) initiator = 'agent'; + else if (['codex-cli-exec', 'codex-sdk'].includes(surface) || threadSource === 'automation') initiator = 'automation'; + else if (['user', 'chatgpt_handoff'].includes(threadSource)) initiator = 'person'; + else if (threadSource !== null || rejectedThreadSource) initiator = 'unknown'; + return [surface, initiator, []]; +} + +/** @param {{host?:string,entrypoint?:unknown,originator?:unknown,source?:unknown, + * threadSource?:unknown,sessionKind?:unknown,importedCopy?:boolean}} declaration */ +export function classifySessionSurface(declaration = {}) { + const { host } = declaration; + const entrypoint = bounded(declaration.entrypoint); + const originator = bounded(declaration.originator); + const source = bounded(declaration.source); + const threadSource = bounded(declaration.threadSource); + // Absence does not contradict a known surface's default initiator. A value + // that was declared but rejected by the bounded token parser does. + const rejectedThreadSource = declaration.threadSource != null && threadSource === null; + const sessionKind = bounded(declaration.sessionKind); + const rawEvidence = {}; + let surface = 'unknown', initiator = 'unknown', attributes = []; + + if (host === 'claude') { + retain(rawEvidence, 'entrypoint', entrypoint); + retain(rawEvidence, 'sessionKind', sessionKind); + [surface, initiator, attributes] = claudeClassification(entrypoint, sessionKind); + } else if (host === 'codex') { + retain(rawEvidence, 'originator', originator); + retain(rawEvidence, 'source', source); + retain(rawEvidence, 'threadSource', threadSource); + [surface, initiator, attributes] = codexClassification(originator, source, threadSource, rejectedThreadSource); + } + if (declaration.importedCopy === true) initiator = 'imported-copy'; + return { surface, initiator, label: sessionSurfaceLabel(surface), rawEvidence, attributes, + thirdPartyProvider: null }; +} + +/** Shared display table: classification remains declaration-owned. */ +export const SESSION_SURFACE_LABELS = LABELS; +export const SESSION_HOST_LABELS = Object.freeze({ claude: 'Claude Code', codex: 'Codex', opencode: 'OpenCode' }); +export const SESSION_INITIATOR_LABELS = Object.freeze({ person: 'Person', automation: 'Automation', agent: 'Agent', 'imported-copy': 'Imported copy', unknown: 'Unknown' }); +export const SESSION_PROVIDER_LABELS = Object.freeze({ 'amazon-bedrock': 'Amazon Bedrock', 'google-vertex-ai': 'Google Vertex AI', + anthropic: 'Anthropic', openai: 'OpenAI', openrouter: 'OpenRouter', bedrock: 'Amazon Bedrock', vertex: 'Google Vertex AI', + foundry: 'Microsoft Foundry', gateway: 'Custom gateway', ollama: 'Ollama', lmstudio: 'LM Studio', 'local-openai': 'Local OpenAI-compatible provider' }); + +/** Presentation accepts old snapshots without inventing a precise app mode. */ +export function sessionPresentation(origin = {}) { + const surface = Object.hasOwn(SESSION_SURFACE_LABELS, origin.surface) ? origin.surface + : origin.surface == null && origin.origin === 'claude-desktop' ? 'claude-desktop' : 'unknown'; + const provider = origin.thirdPartyProviderBasis === 'assistant-model-id' + && ['amazon-bedrock', 'google-vertex-ai'].includes(origin.thirdPartyProvider) + ? SESSION_PROVIDER_LABELS[origin.thirdPartyProvider] : 'Unknown'; + return { surface, label: SESSION_SURFACE_LABELS[surface], + initiator: Object.hasOwn(SESSION_INITIATOR_LABELS, origin.initiator) ? SESSION_INITIATOR_LABELS[origin.initiator] : 'Unknown', provider, + providerBasis: provider === 'Unknown' ? 'not established' : 'assistant-model-id; not network attestation', + note: !origin.surface && origin.origin === 'codex-desktop' + ? 'ChatGPT desktop app observed; mode not recorded in this legacy snapshot' + : !origin.surface && origin.origin === 'claude-desktop' ? 'Claude Desktop observed in this legacy snapshot' : '', + }; +} + + +/** A recorded provider ID is source evidence, not network attestation. */ +export function sessionProviderPresentation(session = {}) { + const surface = sessionPresentation(session.sessionOrigin ?? {}); + if (surface.provider !== 'Unknown') return { label: surface.provider, basis: surface.providerBasis }; + const provider = typeof session.provider === 'string' ? session.provider.toLowerCase() : ''; + return session.providerProvenance === 'observed' && Object.hasOwn(SESSION_PROVIDER_LABELS, provider) + ? { label: SESSION_PROVIDER_LABELS[provider], basis: 'recorded provider ID; not network attestation' } + : { label: 'Unknown', basis: 'not established' }; +} diff --git a/src/lib/usage-aggregate.mjs b/src/lib/usage-aggregate.mjs index 28a96720..8a810d57 100644 --- a/src/lib/usage-aggregate.mjs +++ b/src/lib/usage-aggregate.mjs @@ -1,5 +1,7 @@ -import { rowCostEvidence, sessionCostEvidence, acquisitionSummary } from './usage-cost.mjs'; +import { foldIncompleteOpencodeObservations, hasOpencodeObservations, opencodeObservationProjection } from './usage-opencode-observations.mjs'; +import { rowCostEvidence, sessionCostEvidence, acquisitionSummary, reconcileClaudeCostState } from './usage-cost.mjs'; import { isLocalInferenceProvider } from './usage-local-provider.mjs'; +import { classifySessionSurface } from './session-surface.mjs'; // usage-aggregate.mjs — pure arithmetic over ALREADY-PARSED session records: // interval math, secret masking, and the two shapes usage-index.mjs hands its // consumers (the batch Aggregate from `aggregate()`, and the single-session @@ -337,7 +339,9 @@ function firstBilledDay(rec) { * already a managed artifact. */ function v16Projection(rec) { - const fps = rec.promptFPs; + // Schema-26 caches can still contain fingerprints written before OpenCode + // child sessions were excluded. Treat those as measured zero on consumption. + const fps = rec.host === 'opencode' && isSubagentSession(rec) ? [] : rec.promptFPs; if (!Array.isArray(fps)) { return { typedPrompts: null, tapPrompts: null, _typedTokens: [], _questions: 0, _personas: 0 }; } @@ -428,6 +432,7 @@ function buildPromptBaselines(records, { days, now }) { const endDay = localDay(now - days * DAY_MS); const perHost = Object.create(null); for (const rec of records) { + if (rec?.host === 'opencode' && isSubagentSession(rec)) continue; if (!rec || !Array.isArray(rec.promptFPs)) continue; const day = firstBilledDay(rec); if (day === null || day < startDay || day >= endDay) continue; @@ -520,6 +525,7 @@ function decoratePromptFP(fp, rec, day) { function windowFingerprints(records, cutoff) { const out = []; for (const rec of records) { + if (rec?.host === 'opencode' && isSubagentSession(rec)) continue; if (!rec?.responses || !Array.isArray(rec.promptFPs)) continue; if (rec.end == null || rec.end < cutoff) continue; const day = firstBilledDay(rec); @@ -774,6 +780,7 @@ function foldSessionUsageRow(row, rec, deps, acc, byDay, byModel, activeDays) { m.responses += row.responses; m.input += row.input; m.output += row.output; m.cacheRead += row.cacheRead; m.cacheWrite += row.cacheWrite; m.tokens += rowTokens; m.cost = round(m.cost + rowCost); + return rowCost; } /** Fold every usage row of one record; returns `{input, output, cacheRead, @@ -783,13 +790,29 @@ function foldSessionUsageRow(row, rec, deps, acc, byDay, byModel, activeDays) { * usable: it can open at 23:58 and only bill after midnight). */ function foldSessionUsageRows(rec, deps, byDay, byModel, rates) { const acc = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, cost: 0, cacheSaved: 0 }; + const providerUsage = rec.host === 'opencode' ? Object.create(null) : null; const activeDays = new Set(); for (const row of rec.usage) { acc.cacheSaved += cacheSavedFor(row, rec, deps, rates); - foldSessionUsageRow(row, rec, deps, acc, byDay, byModel, activeDays); + const rowCost = foldSessionUsageRow(row, rec, deps, acc, byDay, byModel, activeDays); + if (providerUsage) { + // Only per-row accounting belongs here. A full bucket also has session + // minutes/confidence initialized to zero; spreading it over the session + // row below would erase those measured session fields. + const provider = row.provider ?? 'unknown'; + const p = providerUsage[provider] ??= { + responses: 0, input: 0, output: 0, cacheRead: 0, cacheWrite: 0, + tokens: 0, cost: 0, + }; + p.responses += row.responses; + p.input += row.input; p.output += row.output; + p.cacheRead += row.cacheRead; p.cacheWrite += row.cacheWrite; + p.tokens += row.input + row.output + row.cacheRead + row.cacheWrite; + p.cost += rowCost; + } } for (const day of activeDays) byDay[day].sessionsActive++; - return { ...acc, costEvidence: sessionCostEvidence(rec, deps), firstDay: firstBilledDay(rec) }; + return { ...acc, providerUsage, costEvidence: sessionCostEvidence(rec, deps), firstDay: firstBilledDay(rec) }; } /** @@ -831,28 +854,77 @@ function v11Projection(rec) { }; } +const CODEX_EFFORT_VALUES = new Set(['none', 'minimal', 'low', 'medium', 'high', 'xhigh']); + +/** Normalize cached v26 records that predate these optional Codex details. */ +function codexObservationProjection(rec) { + if (rec.host !== 'codex') return { codexEffort: null, firstTokenMs: null, compactions: 0, + compactionEvidence: { lowerBound: 0, upperBound: 0 } }; + const rawEffort = rec.codexEffort; + const counts = Object.fromEntries(Object.entries(rawEffort?.counts ?? {}) + .filter(([key, value]) => CODEX_EFFORT_VALUES.has(key) && Number.isSafeInteger(value) && value > 0)); + const codexEffort = CODEX_EFFORT_VALUES.has(rawEffort?.last) && Object.keys(counts).length + ? { last: rawEffort.last, counts } : null; + const rawTiming = rec.firstTokenMs; + const firstTokenMs = rawTiming?.provenance === 'host-observed' + && Number.isSafeInteger(rawTiming.count) && rawTiming.count > 0 + && Number.isFinite(rawTiming.total) && rawTiming.total >= 0 + && Number.isFinite(rawTiming.min) && rawTiming.min >= 0 + && Number.isFinite(rawTiming.max) && rawTiming.max >= rawTiming.min + ? { count: rawTiming.count, total: rawTiming.total, min: rawTiming.min, + max: rawTiming.max, provenance: 'host-observed' } : null; + const compactions = Number.isSafeInteger(rec.compactions) && rec.compactions > 0 ? rec.compactions : 0; + const rawCompaction = rec.compactionEvidence; + const validBounds = Number.isSafeInteger(rawCompaction?.lowerBound) + && rawCompaction.lowerBound === compactions + && Number.isSafeInteger(rawCompaction.upperBound) + && rawCompaction.upperBound >= compactions; + const compactionEvidence = validBounds + ? { lowerBound: compactions, upperBound: rawCompaction.upperBound } + : { lowerBound: compactions, upperBound: compactions ? null : 0 }; + return { codexEffort, firstTokenMs, compactions, compactionEvidence }; +} + /** One aggregate session row from a parsed record, its folded usage sums, * and its classifier verdict. */ +function sessionProviderIdentity(rec) { + // Cached v26 OpenCode records can still carry the old last-wins session + // field. Derive it from the row identities so no additional schema bump is + // needed for this unreleased format. + if (rec.host !== 'opencode') return { + provider: rec.inferenceProvider ?? null, + providerProvenance: rec.providerProvenance ?? 'unknown', + }; + const providers = new Set((rec.usage ?? []).map((row) => row.provider ?? null)); + const provider = providers.size === 1 ? [...providers][0] : null; + return { provider, providerProvenance: provider ? 'observed' : 'unknown' }; +} + function buildSessionRow(rec, usage, verdict) { const { input, output, cacheRead, cacheWrite, cost, cacheSaved, firstDay } = usage; return { id: rec.id, host: rec.host ?? rec.provider, - provider: rec.inferenceProvider ?? null, + ...sessionProviderIdentity(rec), transcriptProvider: rec.provider, - providerProvenance: rec.providerProvenance ?? 'unknown', title: rec.title, project: rec.project, projectEvidence: rec.projectEvidence ? { ...rec.projectEvidence } : null, sessionOrigin: rec.sessionOrigin ? { ...rec.sessionOrigin } : null, + ...(rec.importEvidence ? { importEvidence: { ...rec.importEvidence } } : {}), worktree: rec.worktree ?? null, start: new Date(rec.start ?? rec.end).toISOString(), minutes: Math.round(((rec.end - (rec.start ?? rec.end)) / 60_000) * 10) / 10, prompts: rec.prompts, responses: rec.responses, exceptions: rec.exceptions, - sidechain: rec.sidechain, threadSource: rec.threadSource, + // Transcript response count stays per session; shared copies have a + // separate count for the global accounting folds. + ...(rec.accountedResponses === undefined ? {} : { accountedResponses: rec.accountedResponses }), + _accountedResponses: rec.accountedResponses ?? rec.responses, + sidechain: rec.sidechain, threadSource: rec.threadSource, parentSessionId: rec.parentSessionId ?? null, models: rec.models.slice(), input, output, cacheRead, cacheWrite, tokens: input + output + cacheRead + cacheWrite, cost: round(cost), costEvidence: usage.costEvidence, + claudeCostState: reconcileClaudeCostState(rec.originalUsage ? { ...rec, usage: rec.originalUsage } : rec), acquisitionCoverage: rec.acquisitionCoverage ?? null, // What the cache avoided for THIS session, so the window total is // auditable a row at a time rather than only in aggregate. @@ -866,6 +938,7 @@ function buildSessionRow(rec, usage, verdict) { // records (the schema bump re-derives those). reasoningOutput: rec.reasoningOutput ?? 0, rateLimits: rec.rateLimits ?? null, + ...(rec.host === 'opencode' ? opencodeObservationProjection(rec) : codexObservationProjection(rec)), ...v11Projection(rec), // v16: what this session's operator actually typed. The underscore-prefixed // members are working material for the window fold (per-host lengths and @@ -884,6 +957,7 @@ function buildSessionRow(rec, usage, verdict) { _active: Array.isArray(rec.active) && rec.active.length ? rec.active : [[rec.start ?? rec.end, rec.end]], _punchcard: rec.punchcard, _day: firstDay, + _providerUsage: usage.providerUsage, }; } @@ -896,7 +970,10 @@ function buildSessionRow(rec, usage, verdict) { function buildSessionRows(records, { cutoff, endMs = null, deps, byDay, byModel, rates }) { const sessions = []; for (const rec of records) { - if (!rec || !rec.responses) continue; // no assistant turn → not a session + // Codex can spend measured tokens in an aborted/tool-only turn. A positive + // component row is enough evidence to include it even without a message. + if (!rec || (!rec.responses && !hasOpencodeObservations(rec) && !(rec.host === 'codex' && rec.usage?.some((row) => + row.input > 0 || row.output > 0 || row.cacheRead > 0 || row.cacheWrite > 0)))) continue; if (rec.end === null || rec.end < cutoff) continue; // outside the window if (endMs != null && rec.end >= endMs) continue; // ... or after the window asked for const usage = foldSessionUsageRows(rec, deps, byDay, byModel, rates); @@ -934,7 +1011,7 @@ function foldSessionIntoTree(tree, s) { /** Subagent work is either Claude's sidechain flag or Codex's ledger-backed * thread source; both mean "not a session a human was driving". */ function sourceKey(s) { - return isSubagentSession(s) ? 'subagent' : 'main'; + return isSubagentSession(s) || ['guardian_review', 'agent_created_thread'].includes(s.threadSource) ? 'subagent' : 'main'; } /** Second pass over the (now sorted) session rows: totals, the by-host/ @@ -946,6 +1023,7 @@ function foldSessionTotals(sessions, byDay, byModel) { const totals = { sessions: sessions.length, prompts: 0, humanPrompts: 0, responses: 0, exceptions: 0, aborts: 0, + compactions: 0, compactionEvidence: { lowerBound: 0, upperBound: 0 }, firstTokenMs: null, input: 0, output: 0, cacheRead: 0, cacheWrite: 0, tokens: 0, cost: 0, cacheSavedUsd: 0, spanMinutes: 0, spanUnionSeconds: 0, engagedSeconds: 0, // v16. `humanPrompts` above is main-thread PROMPT COUNTS; these two are the @@ -969,12 +1047,25 @@ function foldSessionTotals(sessions, byDay, byModel) { for (const s of sessions) { const source = sourceKey(s); - totals.prompts += s.prompts; totals.responses += s.responses; + totals.prompts += s.prompts; totals.responses += s._accountedResponses; // Only a MAIN-thread prompt is a human typing. A subagent's prompts are // written by the harness, so counting them would inflate every // per-prompt denominator with work nobody asked for by hand. if (source === 'main') totals.humanPrompts += s.prompts; totals.exceptions += s.exceptions; totals.aborts += Number(s.aborts) || 0; + totals.compactions += s.compactions; + totals.compactionEvidence.lowerBound += s.compactionEvidence.lowerBound; + totals.compactionEvidence.upperBound = totals.compactionEvidence.upperBound === null + || s.compactionEvidence.upperBound === null ? null + : totals.compactionEvidence.upperBound + s.compactionEvidence.upperBound; + if (s.firstTokenMs) { + totals.firstTokenMs ??= { count: 0, total: 0, min: s.firstTokenMs.min, + max: s.firstTokenMs.max, provenance: 'host-observed' }; + totals.firstTokenMs.count += s.firstTokenMs.count; + totals.firstTokenMs.total += s.firstTokenMs.total; + totals.firstTokenMs.min = Math.min(totals.firstTokenMs.min, s.firstTokenMs.min); + totals.firstTokenMs.max = Math.max(totals.firstTokenMs.max, s.firstTokenMs.max); + } totals.input += s.input; totals.output += s.output; totals.cacheRead += s.cacheRead; totals.cacheWrite += s.cacheWrite; totals.tokens += s.tokens; totals.cost += s.cost; @@ -983,18 +1074,28 @@ function foldSessionTotals(sessions, byDay, byModel) { spanMs += s._span[1] - s._span[0]; const hostBucket = bucket(byHost, s.host ?? 'unknown'); - addTo(hostBucket, s); + const accounted = { ...s, responses: s._accountedResponses }; + addTo(hostBucket, accounted); // Aborts are host-capability evidence (codex and opencode record a user // stop; claude does not), so they are kept per host — a reader dividing // by "responses that could have recorded one" needs them apart. hostBucket.aborts = (hostBucket.aborts ?? 0) + (Number(s.aborts) || 0); - addTo(bucket(byProvider, s.provider ?? 'unknown'), s); + if (s._providerUsage && Object.keys(s._providerUsage).length) { + // OpenCode's provider is observed on each assistant row. A session may + // count under several providers, but each response/token/dollar lands + // exactly once under the row's own provider (or unknown). + for (const [provider, usage] of Object.entries(s._providerUsage)) { + addTo(bucket(byProvider, provider), { ...s, ...usage }); + } + } else { + addTo(bucket(byProvider, s.provider ?? 'unknown'), accounted); + } // 'not-recorded' is a first-class key, not a display fallback: a transcript // that carried no mode evidence must not be folded into a real posture. - addTo(bucket(byMode, s.mode ?? 'not-recorded'), s); - addTo(bucket(bySource, source), s); - addTo(bucket(byProject, s.project), s); - addTo(bucket(byCategory, s.category), s); + addTo(bucket(byMode, s.mode ?? 'not-recorded'), accounted); + addTo(bucket(bySource, source), accounted); + addTo(bucket(byProject, s.project), accounted); + addTo(bucket(byCategory, s.category), accounted); foldSessionByModel(byModel, s); // Exceptions ride the SAME first-billed-day attribution as the session // count, so the reliability trend and the session trend are drawn from one @@ -1178,6 +1279,7 @@ function previousWindow(records, { days, now, deps, rates }) { cutoff: windowStart - days * DAY_MS, endMs: windowStart, deps, byDay, byModel, rates, }); const folded = foldSessionTotals(sessions, byDay, byModel); + foldIncompleteOpencodeObservations(records, folded.totals, windowStart - days * DAY_MS, windowStart); finishTotals(folded.totals, sessions, folded); return { totals: folded.totals, rhythm: buildRhythm(sessions) }; } @@ -1234,6 +1336,7 @@ export function aggregate(records, { days, now, cutoff, deps, previous = false, sessions.sort((a, b) => b.cost - a.cost || Date.parse(b.start) - Date.parse(a.start)); const folded = foldSessionTotals(sessions, byDay, byModel); + foldIncompleteOpencodeObservations(records, folded.totals, cutoff); const { totals, byHost, byProvider, byMode, bySource, byTool, byProject, byCategory, punchcard, promptsByHost, promptStatsByDay, tree } = folded; @@ -1247,8 +1350,8 @@ export function aggregate(records, { days, now, cutoff, deps, previous = false, const projectTree = buildProjectTree(tree); for (const s of sessions) { - delete s._span; delete s._active; delete s._punchcard; delete s._day; delete s._priced; - delete s._typedTokens; delete s._questions; delete s._personas; + delete s._span; delete s._active; delete s._punchcard; delete s._day; delete s._priced; delete s._providerUsage; + delete s._typedTokens; delete s._questions; delete s._personas; delete s._accountedResponses; } const codexRateLimits = buildCodexRateLimits(sessions); @@ -1300,20 +1403,64 @@ export function aggregate(records, { days, now, cutoff, deps, previous = false, * stays visible/auditable. * Exported for test. */ +function ledgerParent(rec, ledger, byId) { + const parentId = rec.parentSessionId ?? (ledger.parents instanceof Map ? ledger.parents.get(rec.id) : null); + if (typeof parentId !== 'string' || parentId === rec.id) return null; + const parent = byId.get(parentId); + // A chain or cycle supplies no reliable top-level surface. Walk only IDs + // already present in this bounded record set; never infer a missing link. + if (!parent || ['subagent', 'guardian_review', 'agent_created_thread'].includes( + parent.threadSource ?? ledger.threads.get(parentId)?.threadSource)) return null; + const seen = new Set([rec.id]); + let cursor = parentId; + while (cursor) { + if (seen.has(cursor)) return null; + seen.add(cursor); + const node = byId.get(cursor); + if (!node) return null; + cursor = node.parentSessionId ?? (ledger.parents instanceof Map ? ledger.parents.get(cursor) : null); + } + return { parentId, parent }; +} + +function ledgerOrigin(rec, source, parent) { + let origin = rec.sessionOrigin; + // An imported transcript is a copy, regardless of what the optional ledger + // says about its thread. Preserve the parser's explicit override. + if (origin?.evidence === 'imported-copy' || origin?.initiator === 'imported-copy') return origin; + if (origin && source !== rec.threadSource) { + const raw = origin.rawEvidence ?? {}; + origin = { ...origin, ...classifySessionSurface({ host: 'codex', originator: raw.originator, + source: raw.source, threadSource: source }) }; + } + if (origin && parent?.sessionOrigin && ['subagent', 'guardian_review', 'agent_created_thread'].includes(source)) { + origin = { ...origin, surface: parent.sessionOrigin.surface, label: parent.sessionOrigin.label, + attributes: [...(origin.attributes ?? []), source === 'guardian_review' ? 'Auto-review' : 'Subagents'] }; + } + return origin; +} + export function applyCodexLedger(records, ledger) { - if (!ledger || !(ledger.threads instanceof Map)) return records; - return records.map((rec) => { + // The rollout's own source can carry a verified parent even when Codex's + // optional SQLite ledger is unavailable. + if (!ledger || !(ledger.threads instanceof Map)) ledger = { threads: new Map(), parents: new Map() }; + const byId = new Map(records.filter((rec) => rec?.provider === 'codex').map((rec) => [rec.id, rec])); + const overlaid = records.map((rec) => { if (!rec || rec.provider !== 'codex') return rec; const t = ledger.threads.get(rec.id); const fromEdges = ledger.parents instanceof Map && ledger.parents.has(rec.id) ? 'subagent' : null; const source = rec.threadSource ?? t?.threadSource ?? fromEdges; + const parentLink = ledgerParent(rec, ledger, byId); // A rollout that classified ITSELF (any source, subagent included) already // carries the right usage: a subagent's is its own, replay subtracted. - if (source === rec.threadSource) return rec; - const out = { ...rec, threadSource: source }; - if (source === 'subagent') { out.usage = []; out.reasoningOutput = 0; } + if (source === rec.threadSource && !parentLink && !rec.parentSessionId) return rec; + const out = { ...rec, threadSource: source, + parentSessionId: parentLink?.parentId ?? null }; + if (source !== rec.threadSource && source === 'subagent') { out.usage = []; out.reasoningOutput = 0; } + out.sessionOrigin = ledgerOrigin(rec, source, parentLink?.parent); return out; }); + return overlaid.some((rec, index) => rec !== records[index]) ? overlaid : records; } /** The /api/session payload for any parsed record (claude, codex, opencode): @@ -1331,6 +1478,7 @@ export function sessionPayload(rec, turns, deps) { transcriptProvider: rec.provider, providerProvenance: rec.providerProvenance ?? 'unknown', title: rec.title, project: rec.project, + ...(rec.importEvidence ? { importEvidence: { ...rec.importEvidence } } : {}), worktree: rec.worktree ?? null, start: rec.start === null ? null : new Date(rec.start).toISOString(), end: rec.end === null ? null : new Date(rec.end).toISOString(), @@ -1345,6 +1493,7 @@ export function sessionPayload(rec, turns, deps) { // because fmtUsd(undefined) is the truthy string "$0.00". cost: sessionCost(rec, deps), costEvidence: sessionCostEvidence(rec, deps), + claudeCostState: reconcileClaudeCostState(rec), acquisitionCoverage: rec.acquisitionCoverage ?? null, ...usage, tokens: usage.input + usage.output + usage.cacheRead + usage.cacheWrite, }, diff --git a/src/lib/usage-claude-dedup.mjs b/src/lib/usage-claude-dedup.mjs new file mode 100644 index 00000000..edc2c867 --- /dev/null +++ b/src/lib/usage-claude-dedup.mjs @@ -0,0 +1,142 @@ +// Reconcile cached, per-file Claude message claims only after discovery. A +// cached session is always the transcript's own observation; no scan order or +// earlier cache write gets to decide which copied message owns the charge. +const COMPONENTS = ['input', 'output', 'cacheRead', 'cacheWrite', 'cacheWrite1h']; +const nonnegative = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0; +const total = (claim) => COMPONENTS.reduce((n, key) => n + (claim.usage[key] ?? 0), 0); +const compareText = (a, b) => a < b ? -1 : a > b ? 1 : 0; +const punchKey = (ms) => { + const d = new Date(ms); + return `${(d.getDay() + 6) % 7}-${d.getHours()}`; +}; + +function compareClaims(a, b) { + const attribution = (item) => JSON.stringify([item.rec.id, item.rec.project, + item.rec.worktree, item.rec.start, item.rec.end, item.rec.title, item.rec.sessionOrigin, + item.rec.sidechain, item.rec.threadSource, item.rec.parentSessionId, + item.rec.inferenceProvider, item.rec.providerProvenance, item.rec.projectEvidence, + item.rec.claudeSourceKey]); + return total(b.claim) - total(a.claim) + || (b.claim.at ?? 0) - (a.claim.at ?? 0) + || compareText(attribution(a), attribution(b)) + || compareText(a.claim.model, b.claim.model) + || compareText(a.claim.day, b.claim.day); +} + +function validRow(row) { + return !!row && typeof row.day === 'string' && typeof row.model === 'string' + && Number.isSafeInteger(row.responses) && row.responses >= 0 + && COMPONENTS.every((key) => nonnegative(row[key] ?? 0)) + && (row.cacheWrite1h ?? 0) <= row.cacheWrite; +} + +function validClaim(claim) { + return !!claim && typeof claim === 'object' + && typeof claim.identity === 'string' && /^[a-f0-9]{64}$/u.test(claim.identity) + && Number.isSafeInteger(claim.at) + && typeof claim.day === 'string' && /^\d{4}-\d{2}-\d{2}$/u.test(claim.day) + && typeof claim.model === 'string' && !!claim.model && claim.model.length <= 100 + && !!claim.usage && typeof claim.usage === 'object' + && COMPONENTS.every((key) => nonnegative(claim.usage[key])) + && claim.usage.cacheWrite1h <= claim.usage.cacheWrite; +} + +function sourceRows(rec) { + const rows = new Map(); + for (const row of rec.usage) { + if (!validRow(row)) return null; + const rowKey = JSON.stringify([row.day, row.model]); + if (rows.has(rowKey)) return null; + rows.set(rowKey, row); + } + return rows; +} + +function claimsFitRows(rec, rows) { + const sums = new Map(), punches = new Map(), identities = new Set(); + for (const claim of rec.claudeMessages) { + if (!validClaim(claim) || identities.has(claim.identity)) return false; + identities.add(claim.identity); + const rowKey = JSON.stringify([claim.day, claim.model]); + if (!rows.has(rowKey)) return false; + const sum = sums.get(rowKey) ?? Object.fromEntries([...COMPONENTS, 'responses'].map((key) => [key, 0])); + for (const key of COMPONENTS) sum[key] += claim.usage[key]; + sum.responses++; + sums.set(rowKey, sum); + const pk = punchKey(claim.at); + punches.set(pk, (punches.get(pk) ?? 0) + 1); + } + for (const [key, sum] of sums) { + const row = rows.get(key); + if (sum.responses > row.responses || COMPONENTS.some((field) => sum[field] > (row[field] ?? 0))) return false; + } + return [...punches].every(([key, count]) => Number.isSafeInteger(rec.punchcard[key]) && rec.punchcard[key] >= count); +} + +/** A compatible cache must have enough internally consistent detail to + * subtract its claims from the file's original usage rows. A bad cache is + * reparsed from the source rather than passed into accounting. */ +export function validClaudeMessageClaims(rec) { + if (!rec || !Array.isArray(rec.claudeMessages) || !Array.isArray(rec.usage) + || !Number.isSafeInteger(rec.responses) || rec.responses < rec.claudeMessages.length + || !rec.punchcard || typeof rec.punchcard !== 'object') return false; + const rows = sourceRows(rec); + return !!rows + && rec.usage.reduce((sum, row) => sum + row.responses, 0) === rec.responses + && Object.values(rec.punchcard).reduce((sum, count) => sum + count, 0) === rec.responses + && claimsFitRows(rec, rows); +} + +function alter(rec, claim, factor, usage = claim.usage) { + const row = rec.usage.find((r) => r.day === claim.day && r.model === claim.model); + if (!row) return; + for (const key of COMPONENTS) { + if (key === 'cacheWrite1h' && !row.cacheWrite1h && !usage.cacheWrite1h) continue; + row[key] = (row[key] ?? 0) + factor * (usage[key] ?? 0); + } + row.responses += factor; + rec.accountedResponses += factor; + const pk = punchKey(claim.at); + rec.punchcard[pk] = (rec.punchcard[pk] ?? 0) + factor; + if (rec.punchcard[pk] === 0) delete rec.punchcard[pk]; +} + +/** Return accounting copies. `responses` remains the file's original count; + * `accountedResponses` is the globally unique count used by aggregate folds. + * ID-less messages remain untouched, including equal-looking token vectors. */ +export function reconcileClaudeMessages(records) { + const copies = records.map((rec) => rec?.provider === 'claude' + ? { ...rec, usage: rec.usage.map((row) => ({ ...row })), + originalUsage: rec.usage, punchcard: { ...rec.punchcard }, accountedResponses: rec.responses } + : rec); + const groups = new Map(); + for (const rec of copies) { + if (rec?.provider !== 'claude') continue; + for (const claim of rec.claudeMessages ?? []) { + if (typeof claim.identity !== 'string' || !claim.identity) continue; + const list = groups.get(claim.identity) ?? []; + list.push({ rec, claim }); + groups.set(claim.identity, list); + } + } + for (const list of groups.values()) { + if (list.length < 2) continue; + // The fixed acquisition pool decides charges for the displayed and + // comparison windows. A deeper explicit lookback may reveal an older + // copy, but that copy cannot steal or enlarge an eligible charge. If no + // copy is eligible, reconcile the historical observations among themselves. + const eligible = list.filter(({ rec }) => rec.claudeIdentityEligible === true); + const owners = eligible.length ? eligible : list; + owners.sort(compareClaims); + const winner = owners[0]; + // Progressive snapshots can be split across copied files. Their + // component-wise maxima are the complete observed usage, charged once to + // the deterministic winner. The 1h tier cannot exceed total writes. + const richest = Object.fromEntries(COMPONENTS.map((key) => + [key, owners.reduce((max, { claim }) => Math.max(max, claim.usage[key] ?? 0), 0)])); + richest.cacheWrite1h = Math.min(richest.cacheWrite1h, richest.cacheWrite); + for (const item of list) alter(item.rec, item.claim, -1); + alter(winner.rec, winner.claim, 1, richest); + } + return copies; +} diff --git a/src/lib/usage-cost.mjs b/src/lib/usage-cost.mjs index e807205c..2310621e 100644 --- a/src/lib/usage-cost.mjs +++ b/src/lib/usage-cost.mjs @@ -1,7 +1,132 @@ import { isLocalInferenceProvider } from './usage-local-provider.mjs'; -// Price only the missing-cost portion of a coalesced row. Reported zero is -// evidence; missing coverage is an estimate even when that estimate is zero. +const amount = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0; +const count = (value) => Number.isSafeInteger(value) && value >= 0; +const MODEL_LIMIT = 32; +const MODEL_KEY_LIMIT = 100; +// Only a plausible epoch-millisecond value may be compared with transcript ISO +// timestamps. Other numeric units remain malformed diagnostics, not time proof. +const epochMs = (value) => count(value) + && value >= Date.UTC(2020, 0, 1) && value < Date.UTC(2100, 0, 1); + +function checkpointModels(models) { + if (!models || typeof models !== 'object' || Array.isArray(models) + || Object.keys(models).length > MODEL_LIMIT) return null; + const parsed = Object.create(null); + for (const [model, usage] of Object.entries(models)) { + if (model.length > MODEL_KEY_LIMIT || !/^[A-Za-z0-9._:/-]+$/u.test(model) + || !usage || typeof usage !== 'object' || Array.isArray(usage) + || !['inputTokens', 'outputTokens', 'cacheReadInputTokens', 'cacheCreationInputTokens'] + .every((key) => count(usage[key])) + || (usage.webSearchRequests !== undefined && !count(usage.webSearchRequests)) + || (usage.costUSD !== undefined && !amount(usage.costUSD))) return null; + parsed[model] = { + input: usage.inputTokens, output: usage.outputTokens, + cacheRead: usage.cacheReadInputTokens, cacheWrite: usage.cacheCreationInputTokens, + webSearchRequests: usage.webSearchRequests ?? null, + }; + } + return parsed; +} + +/** A cost-state line is a cumulative Claude Code checkpoint, never a message charge. */ +export function recordClaudeCostState(previous, record, sessionId) { + const state = previous ?? { + reportedUsd: null, currency: 'USD', provenance: 'claude-code-cost-state', + coverage: 'unknown', validSnapshots: 0, malformedSnapshots: 0, unsupportedSnapshots: 0, + modelUsage: null, hasUnknownModelCost: null, startMs: null, endMs: null, + }; + if (!Object.hasOwn(record, 'totalCostUSD')) { + state.unsupportedSnapshots++; + return state; + } + if (typeof record.sessionId === 'string' && record.sessionId !== sessionId) { + state.unsupportedSnapshots++; + return state; + } + const parsed = checkpointModels(record.modelUsage); + if (!amount(record.totalCostUSD) || typeof record.sessionId !== 'string' + || !epochMs(record.startTime) || !parsed + || (record.hasUnknownModelCost !== undefined && typeof record.hasUnknownModelCost !== 'boolean')) { + state.malformedSnapshots++; + return state; + } + state.validSnapshots++; + state.reportedUsd = record.totalCostUSD; + state.coverage = 'session-cumulative'; + state.modelUsage = parsed; + state.hasUnknownModelCost = record.hasUnknownModelCost ?? null; + state.startMs = record.startTime; + // Observed cost-state records carry no checkpoint-end timestamp. + state.endMs = null; + return state; +} + +function messageModelTotals(rec) { + const rows = Object.create(null); + for (const row of rec.usage ?? []) { + const summed = rows[row.model] ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }; + for (const key of ['input', 'output', 'cacheRead', 'cacheWrite']) summed[key] += row[key]; + rows[row.model] = summed; + } + return rows; +} + +function differentModelTotals(observed, rows) { + return !Object.keys(rows).length || Object.keys(rows).length !== Object.keys(observed).length + || Object.entries(observed).some(([model, item]) => !rows[model] + || ['input', 'output', 'cacheRead', 'cacheWrite'].some((key) => rows[model][key] !== item[key])); +} + +function servingProviderReason(rec) { + if (rec.sessionOrigin?.thirdPartyProvider) return 'serving-provider-different'; + if (rec.providerProvenance !== 'observed' || !rec.inferenceProvider) return 'serving-provider-unverified'; + return rec.inferenceProvider === 'anthropic' ? null : 'serving-provider-different'; +} + +/** Report provable scope differences without treating equal totals as scope proof. */ +export function reconcileClaudeCostState(rec) { + const state = rec.claudeCostState; + if (!state) return null; + const base = { + reportedUsd: state.reportedUsd, currency: state.currency, provenance: state.provenance, + coverage: state.coverage, validSnapshots: state.validSnapshots, + malformedSnapshots: state.malformedSnapshots, unsupportedSnapshots: state.unsupportedSnapshots, + checkpointStartMs: state.startMs ?? null, checkpointEndMs: state.endMs ?? null, + messageFirstAtMs: rec.claudeMessageCoverage?.firstAtMs ?? null, + messageLastAtMs: rec.claudeMessageCoverage?.lastAtMs ?? null, + missingMessageTimestamps: rec.claudeMessageCoverage?.missingTimestampMessages ?? 0, + }; + if (state.reportedUsd === null || state.coverage !== 'session-cumulative' || !state.modelUsage) { + return { ...base, status: 'scope-unknown', estimatedUsd: null, scopeReasons: ['valid-checkpoint-unavailable'] }; + } + const messages = rec.claudeMessageCoverage; + if (messages?.firstAtMs !== null && messages?.firstAtMs !== undefined + && state.startMs > messages.firstAtMs) { + return { ...base, status: 'scope-mismatch', estimatedUsd: null, + scopeReasons: ['checkpoint-start-after-message'] }; + } + const observed = state.modelUsage; + if (differentModelTotals(observed, messageModelTotals(rec))) { + return { ...base, status: 'scope-mismatch', estimatedUsd: null, + scopeReasons: ['model-token-totals-differ'] }; + } + // Equal counters cannot prove an equal interval or serving provider. The + // observed checkpoint has no end timestamp, and Claude transcript model IDs + // do not attest which provider served a completion. + return { ...base, status: 'scope-unknown', estimatedUsd: null, + scopeReasons: [ + 'checkpoint-end-unreported', + messages?.missingTimestampMessages || messages?.firstAtMs === null ? 'message-time-incomplete' : null, + servingProviderReason(rec), + state.hasUnknownModelCost !== false ? 'model-cost-unknown' : null, + Object.values(observed).some((item) => item.webSearchRequests !== 0) ? 'auxiliary-cost-scope-unknown' : null, + ].filter(Boolean) }; +} + +// Price only the missing-cost portion of a coalesced row. OpenCode's +// positive-token reported zero can mean an absent model rate; those responses +// are explicitly unpriced. Missing coverage is an estimate even when zero. // // A missing-cost portion served by a LOCAL provider is neither: the pricing // table has no rate for a model that costs nothing per call, and the @@ -9,8 +134,9 @@ import { isLocalInferenceProvider } from './usage-local-provider.mjs'; // `unpricedMessages` — coverage the reader can see — and contribute no dollars. export function rowCostEvidence(row, rec, deps) { const observed = typeof row.costObserved === 'number' && Number.isFinite(row.costObserved) && row.costObserved >= 0; - const missing = row.costMissingUsage ?? (observed ? null : row); - const unpriced = !!missing && isLocalInferenceProvider(row.provider); + const untrustedMessages = row.costUntrustedMessages ?? 0; + const missing = row.costMissingUsage ?? (observed || untrustedMessages ? null : row); + const unpriced = !!missing && (isLocalInferenceProvider(row.provider) || row.model === 'codex-auto-review'); const estimatedUsd = missing && !unpriced ? (deps.costOf({ model: row.model, provider: row.provider ?? rec.provider, day: row.day, input: missing.input, output: missing.output, @@ -22,7 +148,7 @@ export function rowCostEvidence(row, rec, deps) { estimatedUsd, observedMessages: observed ? (row.costObservedMessages ?? row.responses ?? 0) : 0, estimatedMessages: unpriced ? 0 : missingMessages, - unpricedMessages: unpriced ? missingMessages : 0, + unpricedMessages: (unpriced ? missingMessages : 0) + untrustedMessages, }; } diff --git a/src/lib/usage-index.mjs b/src/lib/usage-index.mjs index f294cafb..e5d60f64 100644 --- a/src/lib/usage-index.mjs +++ b/src/lib/usage-index.mjs @@ -5,9 +5,11 @@ // ~/.codex/sessions///
    /rollout--.jsonl // // The corpus is large (1.3 GB on the reference machine) and a finished -// transcript never changes again, so every file is parsed AT MOST ONCE: the +// transcript never changes again, so parsing is reused within its local calendar +// context: the // derived per-session record is cached in ~/.config/agentic-kit/usage-index.json -// keyed by (path, mtime, size). A warm refresh only stats. +// keyed by (path, mtime, size, localTimeContext). A warm refresh stats source files and validates +// cached Claude message claims before using them for cross-file accounting. // // Three rules this module exists to enforce: // 1. Engaged time is the UNION of ACTIVE intervals, never the sum of spans. @@ -36,12 +38,16 @@ // consumer's existing import path expects them from here. import fs from 'node:fs'; import path from 'node:path'; +import { createHash } from 'node:crypto'; import { configDir, claudeDir, codexDir } from './paths.mjs'; import { readClaudeWindowLog, statClaudeWindowLedger } from './claude-window-ledger.mjs'; import { writePrivateFileAtomic } from './file-write.mjs'; import { readCodexStateResult } from './codex-state.mjs'; +import { selectOpencodeSource } from './usage-opencode-source.mjs'; +import { reusableOpencodeObservations } from './usage-opencode-cache.mjs'; +import { opencodeStorageHealth } from './usage-opencode-health.mjs'; import { - defaultOpencodeDbPath, listSessionsResult as listOpencodeSessionsResult, + listSessionsResult as listOpencodeSessionsResult, parseSession as parseOpencodeSession, sessionExistsResult as opencodeSessionExistsResult, usageNotReportedWarnings, } from './usage-opencode.mjs'; @@ -50,6 +56,7 @@ import { MAX_TELEMETRY_UNKNOWN_KINDS, recordTelemetryUnit, } from './usage-telemetry.mjs'; import { parseClaude, parseCodex } from './usage-parsers.mjs'; +import { reconcileClaudeMessages, validClaudeMessageClaims } from './usage-claude-dedup.mjs'; import { openCodexRollout } from './codex-rollout-reader.mjs'; import { maskSecrets, applyCodexLedger, aggregate, sessionPayload } from './usage-aggregate.mjs'; @@ -183,9 +190,38 @@ export { MAX_TURN_CHARS, mergeIntervals, maskSecrets, normalizeSessionIdentity, // across providers (rows now carry `provider`); and no mark on completed // responses whose provider reported no tokens (`tokensUnreported`). None can be // corrected in place, so every cached OpenCode record re-parses. -export const SCHEMA_VERSION = 25; +// The unreleased v26 migration also records per-turn imported exclusion evidence. +// A v26 Claude entry written before cost-state support lacks `claudeCostState`; +// reparse that entry in place rather than bumping the unreleased schema again. +// Claude record coverage is also added within v26: entries without its count-only +// parseStats reparse, so a legacy cache never manufactures a zero unknown count. +export const SCHEMA_VERSION = 26; // v26 adds parse-time session surface fields; v25 records reparse. +// OpenCode cost trust and compaction/reconciliation interpretation changed +// within schema 26. This entry marker requires both semantics, independently +// of source identity and the separately enforced local-calendar context. +const OPENCODE_PARSE_SEMANTICS = 'cost-trust-v2-observations-v1'; +const compatibleOpencodeCache = (candidate, entry) => candidate.provider !== 'opencode' + || (entry?.parseSemantics === OPENCODE_PARSE_SEMANTICS + && entry?.sourceIdentity === candidate.sourceIdentity && !!candidate.sourceIdentity); + +/** Local buckets depend on the full zone's historical rules, not today's offset. + * Include the runtime rule-data version; an upgrade can change past buckets. + * Unknown zones are deliberately non-reusable, including legacy v26 entries. */ +function localTimeContext() { + try { + const zone = Intl.DateTimeFormat().resolvedOptions().timeZone; + if (!zone) return null; + return JSON.stringify([zone, process.versions.tz ?? null, process.versions.icu ?? null]); + } catch { return null; } +} +function compatibleLocalTime(entry, context) { + return context !== null && entry?.localTimeContext === context; +} const DAY_MS = 86_400_000; +// Dashboard windows stop at 365 days. One displayed window plus its equal +// previous window is therefore bounded at 730 days of Claude identity reads. +const MAX_CLAUDE_IDENTITY_DAYS = 730; // One day of slack past dashboard-server.mjs's 365-day clampDays ceiling — // see the carry-forward pruning comment in scan() below. const KEEP_MS = 366 * DAY_MS; @@ -268,9 +304,10 @@ function rootHealth(dir) { function emptyCodexDiagnostics() { return { - files: 0, cachedFiles: 0, parsedFiles: 0, unparsedFiles: 0, unparsedReasons: {}, importedExcluded: 0, - filesWithTokens: 0, filesWithResponses: 0, - legacyEvents: 0, itemCompletedEvents: 0, tokenCountEvents: 0, + files: 0, cachedFiles: 0, parsedFiles: 0, unparsedFiles: 0, unparsedReasons: {}, importedExcluded: 0, importedMixed: 0, importedTurnsExcluded: 0, importAmbiguousRecords: 0, + importOwnershipIncompleteFiles: 0, importedTurnCountIncompleteFiles: 0, + filesWithTokens: 0, filesWithResponses: 0, zeroResponseUsageFiles: 0, zeroResponseUnsupportedFiles: 0, + legacyEvents: 0, itemCompletedEvents: 0, tokenCountEvents: 0, totalOnlyTokenCountEvents: 0, prompts: 0, responses: 0, unknownItemTypes: {}, unknownItemTypeOverflow: 0, clippedLines: 0, warnings: [], }; } @@ -283,6 +320,7 @@ function addCodexParseDiagnostics(target, stats) { target.legacyEvents += stats.legacyEvents; target.itemCompletedEvents += stats.itemCompletedEvents; target.tokenCountEvents += stats.tokenCountEvents; + target.totalOnlyTokenCountEvents += stats.totalOnlyTokenCountEvents ?? 0; target.prompts += stats.prompts; target.responses += stats.responses; target.clippedLines += stats.clippedLines ?? 0; @@ -300,10 +338,10 @@ function addCodexParseDiagnostics(target, stats) { function finalizeCodexHealth(root, diagnostics) { const warnings = []; - const tokenFiles = diagnostics.filesWithTokens; const responseFiles = diagnostics.filesWithResponses; - if (tokenFiles > 0 && responseFiles === 0) warnings.push('zero-response-yield'); - else if (tokenFiles > responseFiles) warnings.push('partial-response-yield'); + if (diagnostics.zeroResponseUnsupportedFiles > 0 && responseFiles === 0) warnings.push('zero-response-yield'); + else if (diagnostics.zeroResponseUnsupportedFiles > 0) warnings.push('partial-response-yield'); + if (diagnostics.totalOnlyTokenCountEvents > 0) warnings.push('total-only-token-count'); if (Object.keys(diagnostics.unknownItemTypes).length || diagnostics.unknownItemTypeOverflow > 0) { warnings.push('unknown-item-types'); } @@ -312,11 +350,13 @@ function finalizeCodexHealth(root, diagnostics) { if (diagnostics.unparsedFiles > 0) warnings.push('unparsed-rollouts'); if (diagnostics.clippedLines > 0) warnings.push('oversized-lines-clipped'); diagnostics.warnings = warnings; - const hasYieldWarning = warnings.includes('zero-response-yield') || warnings.includes('partial-response-yield'); - const status = root.status === 'ok' && hasYieldWarning + const hasUsageWarning = warnings.includes('zero-response-yield') || warnings.includes('partial-response-yield') + || warnings.includes('total-only-token-count'); + const status = root.status === 'ok' && hasUsageWarning ? 'degraded' : root.status; const reason = status === 'degraded' && root.status === 'ok' - ? (warnings.includes('zero-response-yield') ? 'parse-yield-zero' : 'parse-yield-partial') : root.reason; + ? (warnings.includes('zero-response-yield') ? 'parse-yield-zero' + : warnings.includes('partial-response-yield') ? 'parse-yield-partial' : 'usage-total-only') : root.reason; return { ...root, status, reason, diagnostics }; } @@ -334,6 +374,42 @@ function attachTelemetryHealth(health, common, diagnostics = health.diagnostics) }; } +function claudeParseHealth(root, common) { + if (root.status !== 'ok' || common.unitsSeen === common.unitsParsed) return root; + return { ...root, status: 'degraded', reason: 'transcript-parse-incomplete' }; +} + +const CLAUDE_RECORD_COUNTERS = ['knownHandledRecords', 'knownIgnoredRecords', + 'unknownRecords', 'invalidTypeRecords', 'malformedRecords']; +function validClaudeRecordStats(stats) { + return stats && typeof stats === 'object' && !Array.isArray(stats) + && Object.keys(stats).length === CLAUDE_RECORD_COUNTERS.length + && CLAUDE_RECORD_COUNTERS.every((key) => Number.isSafeInteger(stats[key]) && stats[key] >= 0); +} +function emptyClaudeRecordDiagnostics() { + return { knownHandledRecords: 0, knownIgnoredRecords: 0, unknownRecords: 0, + invalidTypeRecords: 0, malformedRecords: 0 }; +} +function addClaudeRecordDiagnostics(target, stats, provider) { + if (provider !== 'claude' || !validClaudeRecordStats(stats)) return; + for (const key of CLAUDE_RECORD_COUNTERS) target[key] += stats[key]; +} +function claudeRecordCoverage(root, common, counts) { + const incomplete = root.status !== 'ok' || common.unitsSeen !== common.unitsParsed + || counts.unknownRecords > 0 || counts.invalidTypeRecords > 0 || counts.malformedRecords > 0; + return { ...counts, coverage: root.status === 'absent' || (root.status === 'ok' && common.unitsSeen === 0) + ? 'not-observed' : common.unitsParsed === 0 ? 'unknown' : incomplete ? 'incomplete' : 'complete' }; +} +function finalizeClaudeRecordHealth(root, common, counts) { + const base = attachTelemetryHealth(claudeParseHealth(root, common), common); + const records = claudeRecordCoverage(root, common, counts); + const incompleteRecords = root.status === 'ok' && common.unitsSeen === common.unitsParsed + && records.coverage === 'incomplete'; + return { ...base, + ...(incompleteRecords ? { status: 'degraded', reason: 'transcript-record-coverage-incomplete' } : {}), + diagnostics: { ...base.diagnostics, records } }; +} + function defaultRoots() { return { claude: path.join(claudeDir(), 'projects'), @@ -467,7 +543,8 @@ function parseCodexFile(entry, sink, limits) { function parseFile(entry, sink = {}, limits = {}) { if (entry.provider === 'opencode') { try { - const parsed = parseOpencodeSession({ dbFile: entry.dbFile, id: entry.id }); + const parsed = parseOpencodeSession({ dbFile: entry.dbFile, id: entry.id, + maxSessionBytes: limits.maxSessionBytes, maxSessionRows: limits.maxSessionRows }); // Title hygiene matches the JSONL parsers: the cached index lands on // disk, so the same secrets mask applies here. if (parsed?.session) parsed.session.title = maskSecrets(parsed.session.title); @@ -589,6 +666,7 @@ function scanKey(o = {}) { return JSON.stringify([ Number(o.days) || 14, Number(o.lookbackDays) || 0, !!o.previous, !!o.prompts, !!o.force, roots, o.cachePath || '', o.claudeWindowConfigDir || '', + localTimeContext(), selectOpencodeSource({ roots: o.roots }), ]); } @@ -612,11 +690,14 @@ function notify(onProgress, payload) { * pair this with `previous: true` (below) to actually get the * older records back out, via `previous.totals`/`previous.rhythm`, * rather than by hand-splitting a widened `sessions[]`. Undefined - * (default) behaves exactly as `days` alone: no widening. + * (default) still discovers Claude files through the equal-length + * preceding identity window; other hosts retain the `days` cutoff. * @property {boolean} [previous] also have `aggregate` project the * equal-length window immediately before the displayed one (see - * usage-aggregate.mjs's `previousWindow`); needs `lookbackDays` set - * wide enough for those older records to have been read at all. + * usage-aggregate.mjs's `previousWindow`); Claude's equal-length + * predecessor is acquired for identity accounting even without an + * explicit lookback, while other hosts need `lookbackDays` set wide + * enough for their older records to have been read. * Forwarded to `aggregate`'s own `previous` option unchanged. * @property {boolean} [prompts] also have `aggregate` build the prompt * repetition projection (`agg.promptPatterns` — recurring clusters, @@ -634,9 +715,10 @@ function notify(onProgress, payload) { * statusline's `claude-context-windows/` ledger (tests). Unset reads * the real config dir only for default-root scans; overridden `roots` * read no ledger. `null` disables ledger pairing. - * @property {{streamAboveBytes?: number, chunkBytes?: number, maxLineBytes?: number}} [readLimits] + * @property {{streamAboveBytes?: number, chunkBytes?: number, maxLineBytes?: number, maxSessionBytes?: number, maxSessionRows?: number}} [readLimits] * override where a Codex rollout switches from a whole-string read to - * the bounded streaming reader, and that reader's chunk/line limits (tests) + * the bounded streaming reader, its chunk/line limits, and OpenCode session + * acquisition byte/row limits (tests) * @property {number} [now] override "now" (tests) * @property {number} [maxAgeMs] readIndex only: memo TTL * @property {object|null} [codexState] override the Codex SQLite thread ledger @@ -671,19 +753,23 @@ export async function buildIndex(o = {}) { * with defaults — its mere presence (vs `undefined`) is what makes an * override hermetic; see the codex ledger comment below. */ function discoverOpencodeSource(rawRoots, cutoff) { - const ocDb = rawRoots === undefined ? defaultOpencodeDbPath() : (rawRoots?.opencode ?? null); - if (!ocDb || !fs.existsSync(ocDb)) return { health: { status: 'absent', reason: null }, candidates: [], ocDb }; + const selection = selectOpencodeSource({ roots: rawRoots }); + const ocDb = selection.dbFile; + const storageCoverage = opencodeStorageHealth(selection); + const sourceHealth = (health) => ({ ...health, storageCoverage }); + if (!ocDb) return { health: sourceHealth(selection.health), candidates: [], ocDb }; + if (!fs.existsSync(ocDb)) return { health: sourceHealth({ status: 'absent', reason: null }), candidates: [], ocDb }; const listed = listOpencodeSessionsResult({ dbFile: ocDb, cutoffMs: cutoff }); if (!listed.ok) { const health = listed.error.kind === 'absent' ? { status: 'absent', reason: 'absent' } : { status: 'degraded', reason: listed.error.kind }; - return { health, candidates: [], ocDb }; + return { health: sourceHealth(health), candidates: [], ocDb }; } const candidates = listed.value.map((e) => ({ - file: `opencode://${e.id}`, provider: 'opencode', id: e.id, dbFile: ocDb, + file: `opencode://${e.id}`, provider: 'opencode', id: e.id, dbFile: ocDb, sourceIdentity: selection.sourceIdentity, stat: { mtimeMs: e.mtimeMs, size: e.size, updatedMs: e.updatedMs }, })); - return { health: { status: 'ok', reason: null }, candidates, ocDb }; + return { health: sourceHealth(selection.health), candidates, ocDb }; } /** Codex's own per-file bookkeeping for one scan candidate: file counts, the @@ -695,8 +781,26 @@ function discoverOpencodeSource(rawRoots, cutoff) { function recordCodexCandidate(codexDiagnostics, { session, parseStats, cacheHit, failure }) { codexDiagnostics.files++; if (cacheHit) codexDiagnostics.cachedFiles++; - if (session?.imported === true) { codexDiagnostics.importedExcluded++; return false; } - if (session) { addCodexParseDiagnostics(codexDiagnostics, parseStats); return true; } + const imports = session?.importEvidence; + codexDiagnostics.importedTurnsExcluded += imports?.importedTurns ?? 0; + if (imports?.ownershipComplete === false) codexDiagnostics.importOwnershipIncompleteFiles++; + if (imports?.importedTurnCountComplete === false) codexDiagnostics.importedTurnCountIncompleteFiles++; + codexDiagnostics.importAmbiguousRecords += imports?.ambiguousRecords ?? 0; + if (session?.imported === true) { + codexDiagnostics.importedExcluded++; + codexDiagnostics.clippedLines += parseStats?.clippedLines ?? 0; + return false; + } + if (imports) codexDiagnostics.importedMixed++; + if (session) { + addCodexParseDiagnostics(codexDiagnostics, parseStats); + if (!session.responses && parseStats?.tokenCountEvents > 0) { + if (session.usage?.some((row) => row.input > 0 || row.output > 0 || row.cacheRead > 0 || row.cacheWrite > 0)) + codexDiagnostics.zeroResponseUsageFiles++; + else codexDiagnostics.zeroResponseUnsupportedFiles++; + } + return true; + } codexDiagnostics.unparsedFiles++; const reason = failure.reason ?? 'parse-error'; codexDiagnostics.unparsedReasons[reason] = (codexDiagnostics.unparsedReasons[reason] ?? 0) + 1; @@ -739,35 +843,60 @@ function withWindowLedger(entry, windowConfigDir) { return { ...entry, windowConfigDir, windowStat: statClaudeWindowLedger(windowConfigDir, entry.id) }; } +function compatibleCostStateCache(c, hit) { + return c.provider !== 'claude' || (Object.hasOwn(hit.session ?? {}, 'claudeCostState') + && Object.hasOwn(hit.session ?? {}, 'claudeMessageCoverage') + && validClaudeRecordStats(hit.parseStats) + && validClaudeMessageClaims(hit.session) + && (!hit.session.claudeCostState || Object.hasOwn(hit.session.claudeCostState, 'startMs'))); +} + +function parseValidatedCandidate(c, failure, readLimits) { + const parsed = parseFile(c, failure, readLimits); + const session = parsed?.session ?? null; + // A source with malformed accounting cannot enter the cache or global + // reconciliation, even when the parser could salvage other fields. + return { session: c.provider === 'claude' && session && !validClaudeMessageClaims(session) ? null : session, + parseStats: parsed?.parseStats ?? null, observationFingerprint: parsed?.observationFingerprint ?? null }; +} + +function withClaudeIdentityEligibility(session, candidate, cutoff) { + if (candidate.provider !== 'claude') return session; + return { ...session, + claudeSourceKey: createHash('sha256').update(candidate.file).digest('hex'), + claudeIdentityEligible: Number.isFinite(candidate.stat.mtimeMs) && candidate.stat.mtimeMs >= cutoff + && Number.isFinite(session.end) && session.end >= cutoff }; +} + /** Parse (or reuse the cached parse of) one scan candidate, updating the * common cross-host telemetry diagnostics and codex's extra per-file * diagnostics as side effects. Pulled out of scan()'s loop so the per-file * bookkeeping — which is genuinely provider-specific (codex tracks file * counts and yield diagnostics no other source has) — is not inlined into * the generic scan loop's own complexity. */ -function processCandidate(c, cache, commonDiagnostics, codexDiagnostics, readLimits = {}) { +function processCandidate(c, cache, commonDiagnostics, codexDiagnostics, claudeRecordDiagnostics, readLimits = {}, timeContext = localTimeContext()) { const hit = cache?.entries?.[c.file]; // `updatedMs` exists only on OpenCode candidates (its rows are rewritten in // place, so created-time and count cannot see a finished turn); file-backed - // sources key on mtime/size alone. + // sources key on mtime/size plus the local calendar context. const updated = c.stat.updatedMs === undefined ? {} : { upd: c.stat.updatedMs }; - const cacheHit = !!(hit && hit.mtime === c.stat.mtimeMs && hit.size === c.stat.size + const cacheHit = !!(compatibleLocalTime(hit, timeContext) && hit.mtime === c.stat.mtimeMs && hit.size === c.stat.size && hit.upd === updated.upd && ledgerStillValid(hit, c.windowStat) + && compatibleCostStateCache(c, hit) + && compatibleOpencodeCache(c, hit) + && reusableOpencodeObservations(c, hit, readLimits) && (c.provider !== 'codex' || hit.parseStats)); - const key = { mtime: c.stat.mtimeMs, size: c.stat.size, ...updated, ...windowKey(c.windowStat, cacheHit ? hit : null) }; - let session = cacheHit ? hit.session : null; - let parseStats = cacheHit ? hit.parseStats : null; + const key = { localTimeContext: timeContext, mtime: c.stat.mtimeMs, size: c.stat.size, ...updated, ...windowKey(c.windowStat, cacheHit ? hit : null), + ...(c.provider === 'opencode' ? { parseSemantics: OPENCODE_PARSE_SEMANTICS, sourceIdentity: c.sourceIdentity } : {}) }; const failure = {}; - if (!session) { - const parsed = parseFile(c, failure, readLimits); - session = parsed ? parsed.session : null; - parseStats = parsed?.parseStats ?? null; - } + const { session, parseStats, observationFingerprint } = cacheHit && hit.session + ? hit : parseValidatedCandidate(c, failure, readLimits); const counted = c.provider !== 'codex' || recordCodexCandidate(codexDiagnostics, { session, parseStats, cacheHit, failure }); if (counted && commonDiagnostics[c.provider]) { recordTelemetryUnit(commonDiagnostics[c.provider], session); + addClaudeRecordDiagnostics(claudeRecordDiagnostics, parseStats, c.provider); if (c.provider === 'codex') { addTelemetryDiagnostics(commonDiagnostics.codex, { unknownKinds: parseStats?.unknownItemTypes, @@ -775,7 +904,11 @@ function processCandidate(c, cache, commonDiagnostics, codexDiagnostics, readLim }); } } - return { key, session, parseStats }; + return { key: observationCacheKey(c, key, observationFingerprint), session, parseStats }; +} + +function observationCacheKey(candidate, key, observationFingerprint) { + return candidate.provider === 'opencode' ? { ...key, observationFingerprint } : key; } /** Carry forward cached entries outside the window whose source still exists, @@ -796,10 +929,14 @@ function processCandidate(c, cache, commonDiagnostics, codexDiagnostics, readLim * Mutates `entries` and `records` in place; returns the possibly-updated * opencode health (a degraded existence check discovered mid-loop must * still be visible to the NEXT entry's check and to the final report). */ -function carryForwardCachedEntries(cache, entries, records, { now, cutoff, ocDb, opencodeHealth }) { +function carryForwardCachedEntries(cache, entries, records, { now, cutoff, ocDb, opencodeHealth, attemptedFiles, timeContext }) { if (!cache?.entries) return opencodeHealth; + let legacyCacheEntriesExcluded = 0; + let timezoneCacheEntriesExcluded = 0; for (const [file, e] of Object.entries(cache.entries)) { - if (entries[file] || !e?.session) continue; + // A candidate that failed reparsing must not be revived just because its + // path still stats (it may now be unreadable or no longer be a file). + if (entries[file] || attemptedFiles.has(file) || !e?.session) continue; const lastActivity = e.session.end ?? e.session.start; // No timestamp at all → can't judge age; keep it rather than guess. if (lastActivity != null && now - lastActivity > KEEP_MS) continue; @@ -811,9 +948,15 @@ function carryForwardCachedEntries(cache, entries, records, { now, cutoff, ocDb, if (!result) continue; entries[file] = result.entry; opencodeHealth = result.health; - if (result.pushRecord && (lastActivity == null || lastActivity >= cutoff)) records.push(e.session); + if (!result.pushRecord && (lastActivity == null || lastActivity >= cutoff)) legacyCacheEntriesExcluded++; + if (result.pushRecord && (lastActivity == null || lastActivity >= cutoff)) { + // Retain the original marker so a degraded source cannot launder + // old buckets into the new timezone on the following refresh. + if (compatibleLocalTime(e, timeContext)) records.push(e.session); + else timezoneCacheEntriesExcluded++; + } } - return opencodeHealth; + return { ...opencodeHealth, legacyCacheEntriesExcluded, timezoneCacheEntriesExcluded }; } /** The opencode half of carryForwardCachedEntries — split out to keep both @@ -822,18 +965,22 @@ function carryForwardCachedEntries(cache, entries, records, { now, cutoff, ocDb, * caller to apply; see carryForwardCachedEntries for why a kept entry is * pushed back into `records` (unlike claude/codex's carry-forward). */ function carryForwardOpencodeEntry(file, e, opencodeHealth, ocDb) { - const dbFile = e.dbFile ?? ocDb; + if (!ocDb) return null; + const selection = selectOpencodeSource({ roots: { opencode: ocDb } }); + if (!selection.sourceIdentity || e.sourceIdentity !== selection.sourceIdentity) return null; + const dbFile = ocDb; const exists = opencodeHealth.status === 'degraded' ? null : (dbFile ? opencodeSessionExistsResult({ dbFile, id: file.slice('opencode://'.length) }) : null); if (opencodeHealth.status === 'degraded' || (exists?.ok && exists.value)) { - return { entry: { ...e, dbFile }, health: opencodeHealth, pushRecord: true }; + return { entry: { ...e, dbFile }, health: opencodeHealth, + pushRecord: compatibleOpencodeCache({ provider: 'opencode', sourceIdentity: selection.sourceIdentity }, e) }; } if (exists && !exists.ok && exists.error.kind !== 'absent') { return { entry: { ...e, dbFile }, - health: { status: 'degraded', reason: exists.error.kind }, - pushRecord: true, + health: { ...opencodeHealth, status: 'degraded', reason: exists.error.kind }, + pushRecord: compatibleOpencodeCache({ provider: 'opencode', sourceIdentity: selection.sourceIdentity }, e), }; } return null; @@ -871,35 +1018,44 @@ async function scan(o = {}) { const deps = await loadDeps(injected); const r = { ...defaultRoots(), ...(roots ?? {}) }; const cacheFile = cachePath ?? defaultCachePath(); - // Widened when the caller passes lookbackDays (a server wanting a - // `previous`-window projection, e.g.) — every DISCOVERY/parse use below - // (candidates, opencode listing, carry-forward) shares this ONE value, so - // widening it here is the entire discovery-side effect. It does NOT reach - // `aggregate`'s own cutoff below — see displayCutoff — so the CURRENT - // window's `sessions`/`totals` never silently widen with it; only - // `aggregate`'s `previous` projection (when requested) reads the extra - // records this pulls in. Unset, `lookbackDays ?? days` is exactly `days` — - // today's behavior, unchanged. + // The display and its equal-length comparison need ONE Claude identity + // pool even if the caller does not request the comparison. This fixed pool + // cannot depend on `previous` or an explicit, deeper `lookbackDays`, or a + // toggle would elect a different owner for the same message. Only Claude + // discovery pays the additional read; other hosts retain their original + // cutoff. The 730-day ceiling covers the dashboard's supported 365-day + // display+comparison maximum and is reported as a cap for wider callers. const cutoff = now - (lookbackDays ?? days) * DAY_MS; + const requestedDays = Number(days); + const identityDays = Math.min(MAX_CLAUDE_IDENTITY_DAYS, + 2 * Math.max(1, Number.isFinite(requestedDays) ? requestedDays : 14)); + const identityCutoff = now - identityDays * DAY_MS; + const claudeCutoff = Math.min(cutoff, identityCutoff); // Primary transcript roots: read once at root level (cheap — not the // recursive per-file walk listClaude/listCodex still do below). const claudeHealth = rootHealth(r.claude); const codexHealth = rootHealth(r.codex); const windowConfigDir = resolveWindowConfigDir(o, roots); - const candidates = [...listClaude(r.claude), ...listCodex(r.codex)] + const claudeCandidates = listClaude(r.claude) .map((e) => withWindowLedger({ ...e, stat: statSafe(e.file) }, windowConfigDir)) + .filter((e) => e.stat && e.stat.mtimeMs >= claudeCutoff); + const codexCandidates = listCodex(r.codex) + .map((e) => ({ ...e, stat: statSafe(e.file) })) .filter((e) => e.stat && e.stat.mtimeMs >= cutoff); + const candidates = [...claudeCandidates, ...codexCandidates]; const opencodeSource = discoverOpencodeSource(roots, cutoff); let opencodeHealth = opencodeSource.health; const ocDb = opencodeSource.ocDb; candidates.push(...opencodeSource.candidates); + const timeContext = localTimeContext(); const cache = force ? null : readCache(cacheFile); const entries = {}; const records = []; const codexDiagnostics = emptyCodexDiagnostics(); + const claudeRecordDiagnostics = emptyClaudeRecordDiagnostics(); const commonDiagnostics = { claude: emptyTelemetryDiagnostics(), codex: emptyTelemetryDiagnostics(), @@ -910,27 +1066,28 @@ async function scan(o = {}) { notify(onProgress, { scanned: 0, total, phase: 'scan' }); for (const c of candidates) { - const { key, session, parseStats } = processCandidate(c, cache, commonDiagnostics, codexDiagnostics, readLimits); + const { key, session, parseStats } = processCandidate(c, cache, commonDiagnostics, codexDiagnostics, claudeRecordDiagnostics, readLimits, timeContext); if (session) { entries[c.file] = { ...key, session, ...(parseStats ? { parseStats } : {}), ...(c.dbFile ? { dbFile: c.dbFile } : {}), }; - if (!session.imported) records.push(session); + if (!session.imported) records.push(withClaudeIdentityEligibility(session, c, identityCutoff)); } scanned++; if (scanned % 100 === 0) notify(onProgress, { scanned, total, phase: 'scan' }); } opencodeHealth = carryForwardCachedEntries(cache, entries, records, { - now, cutoff, ocDb, opencodeHealth, + now, cutoff, ocDb, opencodeHealth, timeContext, attemptedFiles: new Set(candidates.map((candidate) => candidate.file)), }); writeCache(cacheFile, { schemaVersion: SCHEMA_VERSION, updatedAt: new Date(now).toISOString(), entries }); // Completed OpenCode responses whose provider reported no token counts: one // informational health warning (the sessions themselves are still counted). addTelemetryDiagnostics(commonDiagnostics.opencode, { - warnings: usageNotReportedWarnings(records.filter((rec) => rec.host === 'opencode')), + warnings: [...opencodeSource.health.storageCoverage.warnings, + ...usageNotReportedWarnings(records.filter((rec) => rec.host === 'opencode'))], }); notify(onProgress, { scanned: total, total, phase: 'aggregate' }); @@ -946,7 +1103,17 @@ async function scan(o = {}) { // `previous: true` caller would find its "current" totals silently // absorbing what should have been the previous window (the bug this fixes). const displayCutoff = now - days * DAY_MS; - const result = aggregate(applyCodexLedger(records, ledger), { + const unknownEligibility = records.filter((rec) => rec?.provider === 'claude' + && typeof rec.claudeIdentityEligible !== 'boolean').length; + // A deeper historical request can reveal a Claude file whose mtime/end was + // outside the fixed identity pool. Keep its prior history, but do not let a + // newly discovered outside-pool session enter the displayed current window. + const outsideCurrent = (rec) => rec?.provider === 'claude' + && rec.claudeIdentityEligible !== true && rec.end >= displayCutoff; + const observed = applyCodexLedger(records, ledger); + const currentExcluded = observed.filter(outsideCurrent).length; + const reconciled = reconcileClaudeMessages(observed.filter((rec) => !outsideCurrent(rec))); + const result = aggregate(reconciled, { days, now, cutoff: displayCutoff, deps, previous, prompts, }); const codexSourceHealth = finalizeCodexHealth(codexHealth, codexDiagnostics); @@ -954,7 +1121,16 @@ async function scan(o = {}) { warnings: codexSourceHealth.diagnostics.warnings, }); result.sourceHealth = { - claude: attachTelemetryHealth(claudeHealth, commonDiagnostics.claude), + claude: { + ...finalizeClaudeRecordHealth(claudeHealth, commonDiagnostics.claude, claudeRecordDiagnostics), + identityCoverage: { horizonDays: identityDays, + horizonCoversComparison: unknownEligibility === 0 && 2 * requestedDays <= MAX_CLAUDE_IDENTITY_DAYS, + horizonCoversRequestedHistory: unknownEligibility === 0 && Number(lookbackDays ?? days) <= identityDays, + outOfPoolRecords: records.filter((rec) => rec?.provider === 'claude' && rec.claudeIdentityEligible !== true).length, + unknownEligibilityRecords: unknownEligibility, + outsideCurrentExcluded: currentExcluded, + basis: 'file-mtime-and-session-end' }, + }, codex: attachTelemetryHealth(codexSourceHealth, commonDiagnostics.codex), opencode: attachTelemetryHealth(opencodeHealth, commonDiagnostics.opencode), codexLedger: codexLedgerHealth, @@ -988,7 +1164,7 @@ export async function readIndex(o = {}) { // `cachePath` per test so their keys already differ. Ruled parked with that // reason rather than left implied. const key = scanKey({ ...o, days }); - if (_memo && _memo.key === key && now - _memo.at < maxAgeMs) return _memo.agg; + if (localTimeContext() !== null && _memo && _memo.key === key && now - _memo.at < maxAgeMs) return _memo.agg; const agg = await buildIndex({ ...o, days }); _memo = { key, at: now, agg }; return agg; @@ -1113,7 +1289,7 @@ export async function readSession(id, o = {}) { // opencode sessions live in the SQLite store, not a JSONL file — resolve // them before the file-locating path (pseudo-key opencode://). - const ocDb = o.roots === undefined ? defaultOpencodeDbPath() : (o.roots?.opencode ?? null); + const ocDb = selectOpencodeSource({ roots: o.roots }).dbFile; const ocExists = ocDb && fs.existsSync(ocDb) ? opencodeSessionExistsResult({ dbFile: ocDb, id }) : null; if (ocExists?.ok && ocExists.value) { diff --git a/src/lib/usage-opencode-bounds.mjs b/src/lib/usage-opencode-bounds.mjs index d672f76e..d76e14f8 100644 --- a/src/lib/usage-opencode-bounds.mjs +++ b/src/lib/usage-opencode-bounds.mjs @@ -1,3 +1,4 @@ +import { availableOpencodeMetadata } from './usage-opencode-observations.mjs'; // Bound native-to-JS materialization before any JSON bodies are selected. // SQLite octet_length can inspect stored byte lengths without materializing // long TEXT/BLOB bodies. Include ids and metadata too, not only JSON payloads. @@ -9,10 +10,11 @@ const limit = (value, ceiling) => Number.isSafeInteger(value) && value > 0 export function sessionAcquisitionCoverage(db, id, { maxSessionBytes, maxSessionRows }) { const byteLimit = limit(maxSessionBytes, MAX_BYTES); const rowLimit = limit(maxSessionRows, MAX_ROWS); + const metadataBytes = availableOpencodeMetadata(db).map(name => ` + COALESCE(octet_length(${name}), 0)`).join(''); const totals = db.prepare(` SELECT COALESCE(SUM(bytes), 0) AS bytes, COALESCE(SUM(rows), 0) AS rows, MAX(latest) AS latest FROM ( SELECT COALESCE(SUM(octet_length(id) + COALESCE(octet_length(parent_id), 0) - + COALESCE(octet_length(directory), 0) + COALESCE(octet_length(title), 0)), 0) AS bytes, + + COALESCE(octet_length(directory), 0) + COALESCE(octet_length(title), 0) ${metadataBytes}), 0) AS bytes, COUNT(*) AS rows, MAX(CASE WHEN typeof(time_created) IN ('integer', 'real') THEN time_created END) AS latest FROM session WHERE id = ? UNION ALL diff --git a/src/lib/usage-opencode-cache.mjs b/src/lib/usage-opencode-cache.mjs new file mode 100644 index 00000000..2d6154d8 --- /dev/null +++ b/src/lib/usage-opencode-cache.mjs @@ -0,0 +1,52 @@ +import { createHash } from 'node:crypto'; +import { withDb } from './sqlite.mjs'; +import { sessionAcquisitionCoverage } from './usage-opencode-bounds.mjs'; +import { availableOpencodeMetadata, hasOpencodeV2Rows } from './usage-opencode-observations.mjs'; + +/** Called only inside the parser/read-probe snapshot after the same session's + * acquisition budget passes. Hash only observation inputs, never the entire DB + * or user/assistant text. Each framed row goes straight into the hash; only its + * digest survives. Part removal/rewrite can preserve every upstream timestamp. */ +export function opencodeObservationFingerprint(db, id) { + const hash = createHash('sha256'); + const add = row => { const json = JSON.stringify(row); hash.update(`${Buffer.byteLength(json)}:`).update(json); }; + const columns = ['id', ...availableOpencodeMetadata(db)]; + add(db.prepare(`SELECT ${columns.join(', ')} FROM session WHERE id = ?`).get(id) ?? null); + add(hasOpencodeV2Rows(db, id)); + // Multi-path extraction keeps JSON type distinctions (true vs 1, null vs + // strings) and excludes prompt bodies. Malformed rows retain a validity + // marker so the parser can still salvage other rows in the same session. + for (const row of db.prepare(`SELECT id, json_valid(data) AS valid, + CASE WHEN json_valid(data) THEN json_extract(data, + '$.role', '$.parentID', '$.summary', '$.finish', '$.error', '$.time.completed', + '$.tokens', '$.cost', '$.providerID') END AS observation + FROM message WHERE session_id = ? ORDER BY id`).iterate(id)) add(row); + add('parts'); + for (const row of db.prepare(`SELECT p.message_id, p.data + FROM part p JOIN message m ON m.id = p.message_id + WHERE m.session_id = ? AND CASE WHEN json_valid(p.data) + THEN json_extract(p.data, '$.type') IN ('compaction', 'step-finish') ELSE 0 END + ORDER BY p.rowid`).iterate(id)) add(row); + return hash.digest('hex'); +} + +/** A warm-cache probe has its own read transaction and the parser's per-session + * byte/row ceilings. Failure or insufficient coverage cannot authorize reuse. */ +export function readOpencodeObservationFingerprint({ dbFile, id, maxSessionBytes, maxSessionRows }) { + const result = withDb(dbFile, db => { + db.exec('BEGIN'); + if (!sessionAcquisitionCoverage(db, id, { maxSessionBytes, maxSessionRows }).complete) return null; + return opencodeObservationFingerprint(db, id); + }); + return result.ok ? result.value : null; +} + +export function reusableOpencodeObservations(candidate, entry, limits) { + if (candidate.provider !== 'opencode') return true; + return typeof entry.observationFingerprint === 'string' + && /^[a-f0-9]{64}$/.test(entry.observationFingerprint) + && entry.observationFingerprint === readOpencodeObservationFingerprint({ + dbFile: candidate.dbFile, id: candidate.id, + maxSessionBytes: limits.maxSessionBytes, maxSessionRows: limits.maxSessionRows, + }); +} diff --git a/src/lib/usage-opencode-health.mjs b/src/lib/usage-opencode-health.mjs new file mode 100644 index 00000000..5e04bb77 --- /dev/null +++ b/src/lib/usage-opencode-health.mjs @@ -0,0 +1,18 @@ +// Source coverage is independent of V1 listing, parsing and the selected window. +import { withDb } from './sqlite.mjs'; +import { observeOpencodeStorageCoverage } from './usage-opencode-storage-coverage.mjs'; + +/** Observe only the selected store and its explicitly resolved legacy root. + * Failed database access is unknown, never evidence that V2 storage is empty. + * @param {{dbFile: string | null, legacyRoot: string | null}} selection */ +export function opencodeStorageHealth({ dbFile, legacyRoot }) { + const legacy = observeOpencodeStorageCoverage({ legacyRoot }); + if (!dbFile) return legacy; + const observed = withDb(dbFile, (db) => { + db.exec('PRAGMA query_only = ON'); + return observeOpencodeStorageCoverage({ db }); + }); + const v2 = observed.ok ? observed.value.v2 : { status: 'unknown' }; + const warnings = observed.ok ? observed.value.warnings : ['opencode-v2-observation-incomplete']; + return { v2, legacy: legacy.legacy, warnings: [...warnings, ...legacy.warnings] }; +} diff --git a/src/lib/usage-opencode-observations.mjs b/src/lib/usage-opencode-observations.mjs new file mode 100644 index 00000000..43e16ff9 --- /dev/null +++ b/src/lib/usage-opencode-observations.mjs @@ -0,0 +1,142 @@ +import { isLocalInferenceProvider } from './usage-local-provider.mjs'; +// OpenCode v1.18.33: core/session/projector.ts accumulates step-finish usage, +// while opencode/session/processor.ts retains only the last step's message +// tokens. These counters are diagnostics, never an additional billing source. +export const OPENCODE_METADATA = ['version', 'time_compacting', 'cost', 'tokens_input', + 'tokens_output', 'tokens_reasoning', 'tokens_cache_read', 'tokens_cache_write']; + +export function availableOpencodeMetadata(db) { + const columns = new Set(db.prepare('PRAGMA table_info(session)').all().map(row => row.name)); + return OPENCODE_METADATA.filter(name => columns.has(name)); +} + +const finite = v => typeof v === 'number' && Number.isFinite(v) && v >= 0; +const integer = v => Number.isSafeInteger(v) && v >= 0; +const parse = raw => { try { return JSON.parse(raw); } catch { return null; } }; +const finished = data => typeof data?.finish === 'string' && data.finish.length > 0 && data.error == null; + +function compactions(srow, messages, parts) { + const requests = new Set(messages.filter(row => row.data?.role === 'user' + && (parts.get(row.id) ?? []).some(part => part.type === 'compaction')).map(row => row.id)); + const completed = new Set(); + const failed = new Set(); + const pending = new Set(); + let orphaned = 0; + for (const { data } of messages) { + if (data?.role !== 'assistant' || data.summary !== true) continue; + if (requests.has(data.parentID)) { + if (finished(data)) completed.add(data.parentID); + else if (data.error != null) failed.add(data.parentID); + else pending.add(data.parentID); + } else if (finished(data)) orphaned++; + } + const unresolved = [...requests].filter(id => !completed.has(id) && (!failed.has(id) || pending.has(id))).length; + const lowerBound = completed.size; + const incomplete = messages.some(row => !row.data || !['user', 'assistant'].includes(row.data.role)); + const inFlight = srow.time_compacting != null; + return { + compactions: lowerBound, + compactionEvidence: { lowerBound, upperBound: incomplete || (inFlight && !unresolved) + ? null : lowerBound + unresolved + orphaned }, + opencodeCompaction: { requests: requests.size, failed: [...failed].filter(id => !completed.has(id)).length, + unresolved, orphaned, inFlight }, + }; +} + +function counters(data) { + const t = data?.tokens; + const values = [t?.input, t?.output, t?.reasoning, t?.cache?.read, t?.cache?.write]; + if (!values.every(integer) || !finite(data?.cost)) return null; + // A nonzero total inconsistent with the verified additive convention is an + // unsupported older provider basis, rather than a forced apparent mismatch. + if (t.total != null && t.total !== 0 && (!integer(t.total) || t.total !== values.reduce((a, b) => a + b, 0))) return null; + return [data.cost, ...values]; +} +const same = (a, b) => a.every((n, i) => i === 0 + ? Math.abs(n - b[i]) <= 1e-9 * Math.max(1, n, b[i]) : n === b[i]); +const unknown = reason => ({ state: 'unknown', reason, basis: 'opencode-v1.18.33-single-step' }); + +function incompleteMessages(srow, messages, assistants) { + return !assistants.length || srow.time_compacting != null || messages.some(row => !row.data + || !['user', 'assistant'].includes(row.data.role)) || assistants.some(({ data }) => !finished(data) + || !finite(data.time?.completed) || data.time.completed <= 0); +} + +function reconciliation(srow, messages, parts, hasV2) { + if (srow.version !== '1.18.33') return unknown('unsupported-version'); + if (hasV2 !== false) return unknown('unsupported-v2-scope'); + const session = [srow.cost, srow.tokens_input, srow.tokens_output, srow.tokens_reasoning, + srow.tokens_cache_read, srow.tokens_cache_write]; + if (!finite(session[0]) || !session.slice(1).every(integer)) return unknown('invalid-session-counters'); + if (session.every(n => n === 0)) return unknown('unpopulated-session-counters'); + const assistants = messages.filter(row => row.data?.role === 'assistant'); + if (incompleteMessages(srow, messages, assistants)) return unknown('incomplete-messages'); + const sums = [0, 0, 0, 0, 0, 0]; + for (const { id, data } of assistants) { + const usage = counters(data); + if (!usage) return unknown('invalid-message-counters'); + if (usage.slice(1).every(n => n === 0)) return unknown('unreported-message-usage'); + if (data.cost === 0 && !isLocalInferenceProvider(data.providerID)) return unknown('untrusted-message-cost'); + const steps = (parts.get(id) ?? []).filter(part => part.type === 'step-finish'); + const step = steps.length === 1 ? counters(steps[0]) : null; + if (!step || !same(usage, step)) return unknown('unproved-step-scope'); + for (let i = 0; i < sums.length; i++) sums[i] += usage[i]; + } + // A user-owned step or unreadable part cannot be assigned to these messages. + const assistantIds = new Set(assistants.map(row => row.id)); + for (const [id, rows] of parts) if (!assistantIds.has(id) + && rows.some(row => row.type === 'step-finish')) return unknown('unproved-step-scope'); + if (!sums.every(finite) || !sums.slice(1).every(integer)) return unknown('invalid-message-counters'); + return { state: same(session, sums) ? 'matched' : 'mismatch', reason: null, + basis: 'opencode-v1.18.33-single-step' }; +} + +export function hasOpencodeV2Rows(db, id) { + const v2Table = db.prepare("SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = 'session_message'").get(); + if (!v2Table) return false; + try { return !!db.prepare('SELECT 1 FROM session_message WHERE session_id = ? LIMIT 1').get(id); } + catch { return null; } // unreadable scope is unknown, not an absent V2 stream +} + +/** Metadata only: no raw message/parent identities, text, or charge values persist. */ +export function opencodeObservations(db, srow, rows, parts) { + const messages = rows.map(row => ({ id: row.id, data: parse(row.data) })); + const hasV2 = hasOpencodeV2Rows(db, srow.id); + return { ...compactions(srow, messages, parts), + opencodeReconciliation: reconciliation(srow, messages, parts, hasV2) }; +} + +export function opencodeObservationProjection(rec) { + const lower = Number.isSafeInteger(rec.compactions) && rec.compactions >= 0 ? rec.compactions : 0; + const upper = rec.compactionEvidence?.upperBound; + const valid = rec.compactionEvidence?.lowerBound === lower && (upper === null + || (Number.isSafeInteger(upper) && upper >= lower)); + return { codexEffort: null, firstTokenMs: null, compactions: lower, + compactionEvidence: { lowerBound: lower, upperBound: valid && rec.acquisitionCoverage?.complete !== false ? upper : null }, + opencodeReconciliation: rec.opencodeReconciliation ?? unknown('missing-observation'), + opencodeCompaction: rec.opencodeCompaction ?? null }; +} + + +/** Evidence can exist without a completed/billable response. Empty, fully + * observed OpenCode sessions still have no compaction evidence to retain. */ +export function hasOpencodeObservations(rec) { + if (rec.host !== 'opencode' || rec.acquisitionCoverage?.complete === false) return false; + const bounds = opencodeObservationProjection(rec).compactionEvidence; + return bounds.lowerBound > 0 + || bounds.upperBound === null || bounds.upperBound > 0; +} + + +/** Refused acquisitions retain uncertainty without becoming ordinary zero-cost + * session rows. Fold their bounds alone into the selected current/previous window. */ +export function foldIncompleteOpencodeObservations(records, totals, cutoff, endMs = Infinity) { + for (const rec of records) { + if (rec?.host !== 'opencode' || rec.responses || rec.acquisitionCoverage?.complete !== false + || !Number.isFinite(rec.end) || rec.end < cutoff || rec.end >= endMs) continue; + const bounds = opencodeObservationProjection(rec).compactionEvidence; + totals.compactions += bounds.lowerBound; + totals.compactionEvidence.lowerBound += bounds.lowerBound; + totals.compactionEvidence.upperBound = null; + } +} diff --git a/src/lib/usage-opencode-source.mjs b/src/lib/usage-opencode-source.mjs new file mode 100644 index 00000000..af3740c1 --- /dev/null +++ b/src/lib/usage-opencode-source.mjs @@ -0,0 +1,77 @@ +// Persisted OpenCode source selection. Never infer an installation channel from +// a filename, version, or mtime, and never combine potentially copied stores. +import fs from 'node:fs'; +import path from 'node:path'; +import os from 'node:os'; +import { createHash } from 'node:crypto'; +import { xdgBase } from './paths.mjs'; + +// Reject filesystem control bytes rather than letting path normalization hide them. +// eslint-disable-next-line no-control-regex +const validPath = (value) => typeof value === 'string' && value.length > 0 && !/[\x00-\x1f\x7f]/u.test(value); +const unavailable = (reason, status = 'degraded') => ({ dbFile: null, legacyRoot: null, sourceIdentity: null, health: { status, reason } }); + +function selected(file, selection, fsImpl) { + if (!validPath(file) || !path.isAbsolute(file)) return unavailable('database-path-invalid'); + let dbFile = path.resolve(file); + try { dbFile = fsImpl.realpathSync(dbFile); } + catch (error) { + if (error.code !== 'ENOENT') return unavailable('database-path-unreadable'); + } + return { dbFile, legacyRoot: path.join(path.dirname(dbFile), 'storage'), sourceIdentity: createHash('sha256').update(dbFile).digest('hex'), + health: { status: 'ok', reason: null, selection } }; +} + +function discover(dataRoot, fsImpl) { + let dir; + const candidates = new Set(); + try { dir = fsImpl.opendirSync(dataRoot); } + catch (error) { + return error.code === 'ENOENT' + ? selected(path.join(dataRoot, 'opencode.db'), 'discovered', fsImpl) + : unavailable('database-discovery-unreadable'); + } + try { + try { + for (let count = 0; ; count++) { + const entry = dir.readSync(); + if (!entry) break; + if (count >= 256) return unavailable('database-discovery-limit'); + if (!/^opencode(?:-[A-Za-z0-9_-]+)?\.db$/.test(entry.name)) continue; + const candidate = selected(path.join(dataRoot, entry.name), 'discovered', fsImpl); + if (!candidate.dbFile) return candidate; + if (!fsImpl.statSync(candidate.dbFile).isFile()) return unavailable('database-path-invalid'); + candidates.add(candidate.dbFile); + } + } finally { dir.closeSync(); } + } catch { + // Once the root opened, any read/stat/close failure leaves enumeration + // incomplete. A missing candidate cannot establish a unique source. + return unavailable('database-discovery-unreadable'); + } + if (candidates.size > 1) return unavailable('database-selection-ambiguous'); + return selected([...candidates][0] ?? path.join(dataRoot, 'opencode.db'), 'discovered', fsImpl); +} + +/** @param {{roots?: {opencode?: string}, env?: NodeJS.ProcessEnv, fsImpl?: typeof fs}} [options] */ +export function selectOpencodeSource({ roots, env = process.env, fsImpl = fs } = {}) { + if (roots !== undefined) return roots?.opencode === undefined + ? unavailable(null, 'absent') : selected(roots.opencode, 'explicit-root', fsImpl); + const override = env.OPENCODE_DB; + if (override && (!validPath(override) || override.split(/[\\/]/).includes('..'))) return unavailable('database-path-invalid'); + const direct = override === ':memory:' ? unavailable('database-in-memory') + : override && path.isAbsolute(override) ? selected(override, 'environment', fsImpl) : null; + const home = env.HOME ?? env.USERPROFILE ?? os.homedir(); + const base = xdgBase('XDG_DATA_HOME', null, { env }) ?? path.join(home, '.local', 'share'); + if (!validPath(base) || !path.isAbsolute(base)) return direct + ? { ...direct, legacyRoot: null } : unavailable('database-path-invalid'); + const dataRoot = path.join(base, 'opencode'); + // Upstream legacy storage stays under Path.data even when OPENCODE_DB moves. + const withLegacyRoot = (result) => ({ ...result, legacyRoot: path.join(dataRoot, 'storage') }); + if (direct) return withLegacyRoot(direct); + if (override) return withLegacyRoot(selected(path.join(dataRoot, override), 'environment', fsImpl)); + if (['1', 'true'].includes(env.OPENCODE_DISABLE_CHANNEL_DB)) { + return withLegacyRoot(selected(path.join(dataRoot, 'opencode.db'), 'channel-disabled', fsImpl)); + } + return withLegacyRoot(discover(dataRoot, fsImpl)); +} diff --git a/src/lib/usage-opencode-storage-coverage.mjs b/src/lib/usage-opencode-storage-coverage.mjs new file mode 100644 index 00000000..405205eb --- /dev/null +++ b/src/lib/usage-opencode-storage-coverage.mjs @@ -0,0 +1,87 @@ +// Read-only metadata observations for OpenCode stores that the V1 reader does +// not consume. The caller owns database selection, handle lifetime and root. +import fs from 'node:fs'; +import path from 'node:path'; + +const MAX_ENTRIES = 256; +const MAX_DEPTH = 5; +const status = (value) => ({ status: value }); + +/** @param {import('node:sqlite').DatabaseSync | null} db */ +function observeV2(db) { + if (db == null) return status('not-observed'); + try { + const schema = db.prepare("SELECT type FROM sqlite_master WHERE name = 'session_message' LIMIT 1").get(); + if (schema == null) return status('missing'); + if (schema.type !== 'table') return status('unknown'); + return status(db.prepare('SELECT 1 FROM session_message LIMIT 1').get() == null ? 'empty' : 'present'); + } catch { + return status('unknown'); + } +} + +/** @param {string} child @param {string} name @param {number} depth @param {number} maxDepth @param {Array<{directory: string, depth: number}>} pending */ +function inspectLegacyEntry(child, name, depth, maxDepth, pending) { + const info = fs.lstatSync(child); + if (info.isSymbolicLink()) return 'incomplete'; + if (info.isFile() && name.toLowerCase().endsWith('.json')) return 'present'; + if (!info.isDirectory()) return 'continue'; + if (depth >= maxDepth) return 'incomplete'; + pending.push({ directory: child, depth: depth + 1 }); + return 'continue'; +} + +/** @param {string | null} root @param {number} maxEntries @param {number} maxDepth */ +function observeLegacy(root, maxEntries, maxDepth) { + if (root == null) return status('not-observed'); + if (typeof root !== 'string' || !path.isAbsolute(root)) return status('unknown'); + try { + let info; + try { info = fs.lstatSync(root); } + catch (error) { + if (/** @type {NodeJS.ErrnoException} */ (error).code === 'ENOENT') return status('absent'); + return status('unknown'); + } + if (!info.isDirectory()) return status('unknown'); + const pending = [{ directory: root, depth: 0 }]; + let entries = 0; + let incomplete = false; + while (pending.length) { + const { directory, depth } = pending.shift(); + const handle = fs.opendirSync(directory); + try { + let entry; + while ((entry = handle.readSync()) !== null) { + if (++entries > maxEntries) return status('unknown'); + const child = path.join(directory, entry.name); + const finding = inspectLegacyEntry(child, entry.name, depth, maxDepth, pending); + if (finding === 'present') return status('present'); + if (finding === 'incomplete') incomplete = true; + } + } finally { handle.closeSync(); } + } + return status(incomplete ? 'unknown' : 'absent'); + } catch { + return status('unknown'); + } +} + +/** + * Observe unsupported storage without reading message bodies or JSON content. + * `db` must already be opened read-only by the caller. Null means unobserved. + * Limits are clamped so caller mistakes cannot turn this into an unbounded walk. + * @param {{db?: import('node:sqlite').DatabaseSync | null, legacyRoot?: string | null, maxEntries?: number, maxDepth?: number}} [options] + * @returns {{v2: {status: string}, legacy: {status: string}, warnings: string[]}} + */ +export function observeOpencodeStorageCoverage({ db = null, legacyRoot = null, maxEntries = MAX_ENTRIES, maxDepth = MAX_DEPTH } = {}) { + const entryLimit = Number.isInteger(maxEntries) && maxEntries > 0 ? Math.min(maxEntries, MAX_ENTRIES) : MAX_ENTRIES; + const depthLimit = Number.isInteger(maxDepth) && maxDepth >= 0 ? Math.min(maxDepth, MAX_DEPTH) : MAX_DEPTH; + const v2 = observeV2(db); + const legacy = observeLegacy(legacyRoot, entryLimit, depthLimit); + const warnings = []; + if (v2.status === 'present') warnings.push('opencode-v2-session-message-present'); + if (v2.status === 'unknown') warnings.push('opencode-v2-observation-incomplete'); + if (legacy.status === 'present') warnings.push('opencode-legacy-json-present'); + if (legacy.status === 'unknown') warnings.push('opencode-legacy-observation-incomplete'); + return { v2, legacy, warnings }; +} diff --git a/src/lib/usage-opencode.mjs b/src/lib/usage-opencode.mjs index 47c2cb5d..89db25f3 100644 --- a/src/lib/usage-opencode.mjs +++ b/src/lib/usage-opencode.mjs @@ -13,17 +13,18 @@ // on bad input — an absent/corrupt db simply reads as "no opencode source". // // Two attribution rules, grounded in the store itself: -// - COST is opencode's own metered figure on each assistant message -// (data.cost). That is OBSERVED truth, so usage rows carry it as -// `costObserved` and the aggregate prefers it over the pricing table — -// never re-priced from a guessed rate (kimi/openrouter/local rates are -// exactly what ak does not know and must not invent). +// - COST is opencode's own figure on each assistant message (data.cost). +// A positive recorded cost is observed. A zero with positive tokens from +// an unverified non-local provider is unpriced: OpenCode may default a +// missing model rate to zero. Never re-price that row from a guessed rate. // - INFERENCE PROVIDER is the assistant row's providerID when observed // (provenance 'observed'), never the host. A bare `opencode` host id says // nothing about who served the model. // Subagent sessions (parent_id set) keep their tokens: opencode child sessions // record their OWN messages, not a replay of the parent's — the codex // double-count rule does not apply (different storage semantics). +import { availableOpencodeMetadata, opencodeObservations } from './usage-opencode-observations.mjs'; +import { opencodeObservationFingerprint } from './usage-opencode-cache.mjs'; import { withDb } from './sqlite.mjs'; import { sessionAcquisitionCoverage } from './usage-opencode-bounds.mjs'; // Shared record shape/accumulator with parseClaude/parseCodex — see their @@ -34,11 +35,15 @@ import { addUsage, blankSession, noteContextSample, noteLatencySample, notePromptFingerprint, } from './usage-parsers.mjs'; import { normalizeMode } from './usage-modes.mjs'; +import { xdgBase } from './paths.mjs'; +export { selectOpencodeSource } from './usage-opencode-source.mjs'; import { observeUsageProject } from './usage-project-evidence.mjs'; +import { isLocalInferenceProvider } from './usage-local-provider.mjs'; -/** The live opencode store. Overridable via roots in tests. */ +/** Conventional footprint census location, not an authoritative transcript source. + * Usage and project discovery must use selectOpencodeSource instead. */ export function defaultOpencodeDbPath() { - const home = process.env.XDG_DATA_HOME ?? null; + const home = xdgBase('XDG_DATA_HOME', null); return home ? `${home}/opencode/opencode.db` : `${process.env.HOME ?? process.env.USERPROFILE}/.local/share/opencode/opencode.db`; @@ -190,15 +195,16 @@ function recordUserMessage(rec, turns, { rowId, at, withTurns, partsByMessage }) // Opens the prompt→assistant-message latency window; closed by the next // recordAssistantMessage (mirrors parseClaude/parseCodex's latState). rec.pendingPromptMs = at; - // Every opencode user message IS a prompt-kind turn (this source carries no - // harness-injected user rows), so it always fingerprints — on BOTH paths, - // which is why the scan path now loads user text parts (see loadTextParts). - // I1: that makes this the WIDEST of the three fingerprinted populations — + // Main-session user messages are prompt-kind turns. A child session's user + // messages are agent-written, so they do not enter the prompt fingerprint + // layer, though its prompts and usage remain accounted for on BOTH paths. + // The scan path loads user text parts for main-session fingerprints. + // I1: main sessions are the widest of the three fingerprinted populations — // claude gates on userTurnKind, codex additionally on - // CODEX_MACHINE_ENVELOPE_RE, opencode on nothing. Compare per provenance tag, + // CODEX_MACHINE_ENVELOPE_RE, opencode on no further turn kind. Compare per provenance tag, // never in total. const text = messagePartsText(partsByMessage, rowId, ['text']); - notePromptFingerprint(rec, text, 'prompt'); + if (!rec.sidechain) notePromptFingerprint(rec, text, 'prompt'); if (!withTurns) return; turns.push({ role: 'user', at: new Date(at).toISOString(), text, prompt: true, kind: 'prompt' }); } @@ -250,10 +256,6 @@ function recordAssistantUsage(rec, data, at) { const model = typeof data.modelID === 'string' && data.modelID ? data.modelID : 'unknown'; if (!rec.models.includes(model)) rec.models.push(model); const provider = typeof data.providerID === 'string' && data.providerID ? data.providerID : null; - if (provider) { - rec.inferenceProvider = provider; - rec.providerProvenance = 'observed'; - } const t = data.tokens ?? {}; const cache = t.cache ?? {}; const day = localDay(at || Date.now()); @@ -271,7 +273,11 @@ function recordAssistantUsage(rec, data, at) { if (isUnreportedUsage(data, t, cache)) usageRow.tokensUnreported = (usageRow.tokensUnreported ?? 0) + 1; // Retain missing-cost tokens separately before coalescing by day/model. usageRow.costObserved ??= null; - if (typeof data.cost === 'number' && Number.isFinite(data.cost) && data.cost >= 0) { + const hasMeasuredTokens = [t.input, t.output, t.reasoning, cache.read, cache.write] + .some((value) => typeof value === 'number' && Number.isFinite(value) && value > 0); + if (data.cost === 0 && hasMeasuredTokens && !isLocalInferenceProvider(provider)) { + usageRow.costUntrustedMessages = (usageRow.costUntrustedMessages ?? 0) + 1; + } else if (typeof data.cost === 'number' && Number.isFinite(data.cost) && data.cost >= 0) { usageRow.costObserved = (usageRow.costObserved ?? 0) + data.cost; usageRow.costObservedMessages = (usageRow.costObservedMessages ?? 0) + 1; } else { @@ -368,22 +374,9 @@ function processMessageRow(rec, turns, row, { withTurns, partsByMessage }) { recordAssistantMessage(rec, turns, { data, rowId: row.id, at, withTurns, partsByMessage }); } -/** The `part` rows a parse needs. `withTurns` wants every part (text, tool and - * reasoning, for both roles) to build turn rows; the scan path wants only the - * USER text parts, which is all a prompt fingerprint reads — the assistant - * bodies it would otherwise pull in are the bulk of the store and are never - * looked at there. - * - * This is the one place the scan path reads message BODIES at all, so its cost - * was measured rather than assumed. The live store on this machine is too - * small to time (2 sessions, 2 parts; ~5 µs/session, where the two - * `json_extract` predicates cannot pay for themselves because there is nothing - * to exclude). Benchmarked instead against a synthetic store at realistic scale - * — 300 sessions, 18k messages, 63k parts, 75 MB — the filtered query runs - * **45 µs/session and materializes 0.6 MB**, against 125 µs/session and 61 MB - * for the unfiltered join the reader path uses: 2.8x faster, and ~100x less - * text pulled into memory. Only the fingerprints are retained; the text itself - * is discarded with the row. */ +/** Scan reads user text for fingerprints and compaction/step-finish metadata + * for observations. Step-finish usage is never added to message usage. + * Detail additionally reads text/reasoning/tool bodies for transcript turns. */ function loadTextParts(db, id, withTurns) { if (withTurns) { return db.prepare(` @@ -395,9 +388,10 @@ function loadTextParts(db, id, withTurns) { return db.prepare(` SELECT p.message_id AS message_id, p.data AS data FROM part p JOIN message m ON m.id = p.message_id - WHERE m.session_id = ? - AND json_extract(m.data, '$.role') = 'user' - AND json_extract(p.data, '$.type') = 'text' + WHERE m.session_id = ? AND ( + (json_extract(m.data, '$.role') = 'user' + AND json_extract(p.data, '$.type') IN ('text', 'compaction')) + OR json_extract(p.data, '$.type') = 'step-finish') ORDER BY p.rowid ASC `).all(id); } @@ -455,15 +449,22 @@ export function parseSession({ dbFile, id, withTurns = false, maxSessionBytes, m delete rec.stamps; return { session: rec, turns: [] }; } - const srow = db.prepare('SELECT id, parent_id, directory, title FROM session WHERE id = ?').get(id); + const columns = ['id', 'parent_id', 'directory', 'title', ...availableOpencodeMetadata(db)]; + const srow = db.prepare(`SELECT ${columns.join(', ')} FROM session WHERE id = ?`).get(id); if (!srow) return null; const msgRows = db.prepare('SELECT id, time_created, data FROM message WHERE session_id = ? ORDER BY time_created ASC, id ASC').all(id); const partsByMessage = buildPartsIndex(loadTextParts(db, id, withTurns)); const rec = initSessionRecord(srow); - Object.assign(rec, { acquisitionCoverage }); + Object.assign(rec, { acquisitionCoverage }, opencodeObservations(db, srow, msgRows, partsByMessage)); const turns = []; for (const row of msgRows) processMessageRow(rec, turns, row, { withTurns, partsByMessage }); + // A session can switch providers, including to a row with no providerID. + // Its usage rows retain the observed identity; the session names a provider + // only when every assistant row agrees on one. + const providers = new Set(rec.usage.map((row) => row.provider ?? null)); + rec.inferenceProvider = providers.size === 1 ? [...providers][0] : null; + rec.providerProvenance = rec.inferenceProvider ? 'observed' : 'unknown'; if (!withTurns) collectScanToolCounts(db, id, rec); if (!rec.title) rec.title = '(untitled)'; @@ -472,7 +473,7 @@ export function parseSession({ dbFile, id, withTurns = false, maxSessionBytes, m delete rec.stamps; delete rec.pendingPromptMs; delete rec.spans; - return { session: rec, turns }; + return { session: rec, turns, observationFingerprint: opencodeObservationFingerprint(db, id) }; }); return result.ok ? result.value : null; } diff --git a/src/lib/usage-parsers.mjs b/src/lib/usage-parsers.mjs index 802b4824..9ce3614d 100644 --- a/src/lib/usage-parsers.mjs +++ b/src/lib/usage-parsers.mjs @@ -14,6 +14,7 @@ import { repoRoot } from './paths.mjs'; import { windowAt } from './claude-window-ledger.mjs'; import { MAX_TELEMETRY_UNKNOWN_KINDS } from './usage-telemetry.mjs'; import { decodeClaudeRecord, decodeCodexRecord } from './telemetry-records.mjs'; +import { recordClaudeCostState } from './usage-cost.mjs'; import { codexReplayPlan, isCodexReplayLine } from './codex-replay.mjs'; import { newCodexUsageWalk, noteCodexWalkModel, noteCodexWalkResponse, walkCodexTokenCount, codexWalkRows, @@ -22,8 +23,9 @@ import { toMs, maskSecrets } from './usage-aggregate.mjs'; import { normalizeMode } from './usage-modes.mjs'; import { provenanceOf } from './usage-provenance.mjs'; import { promptSemantics } from './usage-prompt-semantics.mjs'; -import { observeUsageProject, usageSessionOrigin } from './usage-project-evidence.mjs'; -import { isCodexImportedLine } from './codex-import-marker.mjs'; +import { observeUsageProject, usageRecordOrigin, importedUsageRecordOrigin } from './usage-project-evidence.mjs'; +import { isCodexImportedLine, newCodexTurnOwnership, codexTurnOwner, codexImportEvidence } from './codex-import-marker.mjs'; +import { claudeProviderFromModelId } from './session-surface.mjs'; export { promptSemantics } from './usage-prompt-semantics.mjs'; @@ -54,6 +56,12 @@ function clip(text, max = 100) { return t.length > max ? `${t.slice(0, max - 1)}…` : t; } +function boundedClaudeModel(model) { + if (typeof model !== 'string' || model.length > 100 || model.includes('//')) return 'unknown'; + return /^[A-Za-z0-9._:/-]+$/u.test(model) || /^claude-[a-z0-9-]+@20[0-9]{6}$/u.test(model) + ? model : 'unknown'; +} + /** * Directory names that mean "the thing below me is a WORKTREE of the repo above * me", not a project of its own. `path.basename(cwd)` on a worktree yields the @@ -176,8 +184,10 @@ function applyProject(rec, res) { // ── transcript parsing ────────────────────────────────────────────────────── -/** Split JSONL into parsed objects, skipping anything that will not parse. */ -function* jsonLines(raw) { +/** Split JSONL into parsed records. Codex retains its conservative object-only + * framing; Claude also observes valid non-object JSON for shape diagnostics. */ +function* jsonLines(raw, stats = null, includeNonObjects = false) { + if (stats) stats.malformedRecords = 0; // Scanned lazily, not split up front: a caller that needs only the first // line (the subagent replay pre-pass) must not pay for the whole file. let pos = 0; @@ -186,10 +196,18 @@ function* jsonLines(raw) { const end = found < 0 ? raw.length : found; const start = pos; pos = end + 1; - if (end === start || raw.charCodeAt(start) !== 123 /* '{' */) continue; + if (end === start) continue; + if (includeNonObjects && !raw.slice(start, end).trim()) continue; + if (!includeNonObjects && raw.charCodeAt(start) !== 123 /* '{' */) { + if (stats && raw.slice(start, end).trim()) stats.malformedRecords++; + continue; + } let obj; - try { obj = JSON.parse(raw.slice(start, end)); } catch { continue; } - if (obj && typeof obj === 'object') yield obj; + try { obj = JSON.parse(raw.slice(start, end)); } catch { + if (stats) stats.malformedRecords++; + continue; + } + if (includeNonObjects || (obj && typeof obj === 'object')) yield obj; } } @@ -202,12 +220,18 @@ export function blankSession(id, provider) { id, provider, host: provider, inferenceProvider: null, providerProvenance: 'unknown', title: '', project: 'unknown', start: null, end: null, projectEvidence: null, sessionOrigin: { origin: 'unknown', evidence: 'desktop-origin-not-declared' }, - prompts: 0, responses: 0, exceptions: 0, sidechain: false, threadSource: null, models: [], tools: {}, + prompts: 0, responses: 0, exceptions: 0, sidechain: false, threadSource: null, parentSessionId: null, models: [], tools: {}, skill: null, plugin: null, worktree: null, usage: [], punchcard: {}, active: [], stamps: [], // Codex-only detail (v6): reasoning tokens inside output, and the last // rate-limit snapshot the rollout carried. Claude sessions keep the zero // and the null — absent, not unknown. reasoningOutput: 0, rateLimits: null, + // Codex host observations. Missing telemetry remains null; compactions + // count completed context replacements, never extra token spend. + codexEffort: null, firstTokenMs: null, compactions: 0, + compactionEvidence: { lowerBound: 0, upperBound: 0 }, + claudeCostState: null, + claudeMessageCoverage: null, // v11: cross-host permission posture (usage-modes.normalizeMode), a // response-latency histogram, THIS session's own engaged seconds, model // context-window detail, and codex's explicit-abort count. Every field @@ -649,6 +673,18 @@ function collectClaudeToolNames(rec, toolUses) { /** Did this decoded usage carry any token evidence at all? */ const hasClaudeUsage = (u) => u.input + u.output + u.cacheRead + u.cacheWrite > 0; +function claudeCrossFileIdentity(e) { + const messageId = e.message.id; + const requestId = e.requestId; + // These are the two observed Claude API ID forms. An arbitrary/malformed + // string is not proof that two files hold one provider message. + const rawId = typeof messageId === 'string' && /^msg_[A-Za-z0-9_-]{1,252}$/u.test(messageId) + ? ['message', messageId] : typeof requestId === 'string' && /^req_[A-Za-z0-9_-]{1,252}$/u.test(requestId) + ? ['request', requestId] : null; + const model = boundedClaudeModel(e.message.model); + return rawId ? sha(JSON.stringify(['claude', claudeProviderFromModelId(model), model, ...rawId]), 64) : null; +} + /** * Stage one assistant transcript line under its API message id. Claude Code * writes ONE line per content block (thinking / text / each tool_use) and @@ -658,26 +694,36 @@ const hasClaudeUsage = (u) => u.input + u.output + u.cacheRead + u.cacheWrite > * A later line with no token evidence never displaces an earlier one that had * some (honest-absent, same rule as the context sample below). A line with no * id at all is its own message: nothing is dropped and nothing is merged with - * an unrelated line. Dedup is scoped to ONE transcript — the same id can - * reappear in a subagent's file, and that cross-file overlap is not attempted. + * an unrelated line. A validated API identity is also retained as a hash for + * scan-wide accounting after every file has been parsed or loaded from cache. */ -function stageClaudeMessage(msgState, decoded, at, model) { - const key = decoded.messageId ?? `line:${msgState.seq++}`; +function stageClaudeMessage(msgState, decoded, at, model, recordedAtMs, identity) { + const key = identity ?? `line:${msgState.seq++}`; const prior = msgState.groups.get(key); if (prior && hasClaudeUsage(prior.usage) && !hasClaudeUsage(decoded.usage)) return; // Re-set keeps the Map's first-seen insertion order, so flush order is stable. - msgState.groups.set(key, { at, model, usage: decoded.usage }); + msgState.groups.set(key, { at, model, usage: decoded.usage, recordedAtMs, identity }); } /** Account every staged message exactly once: response count, punchcard, * the per-day/model usage row and the context sample — all from the message's * last line. Runs after the whole transcript has been read. */ function flushClaudeMessages(rec, msgState, windowLog) { - for (const { at, model, usage } of msgState.groups.values()) { + for (const { at, model, usage, recordedAtMs, identity } of msgState.groups.values()) { + if (hasClaudeUsage(usage)) { + if (recordedAtMs === null) rec.claudeMessageCoverage.missingTimestampMessages++; + else { + rec.claudeMessageCoverage.firstAtMs = Math.min(rec.claudeMessageCoverage.firstAtMs ?? recordedAtMs, recordedAtMs); + rec.claudeMessageCoverage.lastAtMs = Math.max(rec.claudeMessageCoverage.lastAtMs ?? recordedAtMs, recordedAtMs); + } + } rec.responses++; const pk = punchKey(at); rec.punchcard[pk] = (rec.punchcard[pk] ?? 0) + 1; addUsage(rec, localDay(at), model, { ...usage, responses: 1 }); + if (identity) rec.claudeMessages.push({ identity, at, day: localDay(at), model, + usage: { input: usage.input, output: usage.output, cacheRead: usage.cacheRead, + cacheWrite: usage.cacheWrite, cacheWrite1h: usage.cacheWrite1h ?? 0 } }); // Context pressure: the tokens actually IN the model's window for this // message (fresh input plus what got served from cache) — the last message // wins so the field reflects the LAST completion, not a running total. @@ -700,7 +746,7 @@ function flushClaudeMessages(rec, msgState, windowLog) { * latency/model/tool accounting plus its turn row. Usage, response count, * punchcard and context sample are STAGED per message id here and accounted * once by flushClaudeMessages. */ -function recordClaudeAssistantTurn(rec, turns, latState, msgState, ms, decoded, withTurns) { +function recordClaudeAssistantTurn(rec, turns, latState, msgState, ms, decoded, withTurns, identity) { noteSpan(rec, ms); const at = Number.isFinite(ms) ? ms : (rec.start ?? Date.now()); @@ -734,10 +780,10 @@ function recordClaudeAssistantTurn(rec, turns, latState, msgState, ms, decoded, latState.pendingMs = null; } - const model = typeof decoded.model === 'string' ? decoded.model : 'unknown'; + const model = boundedClaudeModel(decoded.model); if (!rec.models.includes(model)) rec.models.push(model); - stageClaudeMessage(msgState, decoded, at, model); + stageClaudeMessage(msgState, decoded, at, model, Number.isFinite(ms) ? ms : null, identity); const tools = collectClaudeToolNames(rec, decoded.toolUses); if (withTurns) { @@ -748,14 +794,37 @@ function recordClaudeAssistantTurn(rec, turns, latState, msgState, ms, decoded, } } +// Established Claude Code bookkeeping records with no message usage. New +// record types are never added implicitly to this list. +const CLAUDE_IGNORED_RECORD_TYPES = new Set(['bridge-session', 'file-history-snapshot', 'queue-operation', + 'atis-latch', 'last-prompt', 'attachment', 'mode', 'permission-mode', + 'agent-name', 'agent-setting', 'system', 'progress', 'summary']); +function claudeRecordCounter(type) { + if (typeof type !== 'string' || !type) return 'invalidTypeRecords'; + if (type === 'user' || type === 'assistant' || type === 'cost-state' || type === 'ai-title') return 'knownHandledRecords'; + return CLAUDE_IGNORED_RECORD_TYPES.has(type) ? 'knownIgnoredRecords' : 'unknownRecords'; +} +function* knownClaudeLines(raw, stats) { + for (const e of jsonLines(raw, stats, true)) { + const counter = claudeRecordCounter(e?.type); + stats[counter]++; + if (counter !== 'invalidTypeRecords' && counter !== 'unknownRecords') yield e; + } +} + /** - * Parse one Claude transcript. Returns `{ session, turns }`; `turns` is only - * populated when `withTurns` (the reader path) — the scan path does not need - * message bodies and holding them would balloon memory over 3,000 files. + * Parse one Claude transcript. Returns `{ session, turns, parseStats }`; + * `turns` is only populated when `withTurns` (the reader path) — the scan path + * does not need message bodies and holding them would balloon memory. */ export function parseClaude(raw, { id, dirName, withTurns = false, windowLog = null }) { const rec = blankSession(id, 'claude'); - rec.sessionOrigin = usageSessionOrigin(raw, 'claude'); + const parseStats = { knownHandledRecords: 0, knownIgnoredRecords: 0, + unknownRecords: 0, invalidTypeRecords: 0, malformedRecords: 0 }; + rec.claudeMessages = []; + rec.claudeMessageCoverage = { firstAtMs: null, lastAtMs: null, missingTimestampMessages: 0 }; + rec.sessionOrigin = usageRecordOrigin(raw, 'claude'); + const observedProviders = new Set(); const turns = []; const titleState = { firstPrompt: '', aiTitle: '' }; // Open by the most recent human prompt, closed by the first real assistant @@ -764,7 +833,11 @@ export function parseClaude(raw, { id, dirName, withTurns = false, windowLog = n // Assistant lines staged per API message id — see stageClaudeMessage. const msgState = { groups: new Map(), seq: 0 }; - for (const e of jsonLines(raw)) { + for (const e of knownClaudeLines(raw, parseStats)) { + if (e.type === 'cost-state') { + rec.claudeCostState = recordClaudeCostState(rec.claudeCostState, e, id); + continue; + } const ms = toMs(e.timestamp); if (e.type === 'ai-title') { if (typeof e.aiTitle === 'string') titleState.aiTitle = e.aiTitle; continue; } if (typeof e.attributionSkill === 'string' && !rec.skill) rec.skill = e.attributionSkill; @@ -780,18 +853,28 @@ export function parseClaude(raw, { id, dirName, withTurns = false, windowLog = n } if (decoded.role !== 'assistant' || !e.message) continue; - recordClaudeAssistantTurn(rec, turns, latState, msgState, ms, decoded, withTurns); + // A transcript's assistant model is tied to this session. Current global + // settings and process.env are not historical session evidence. + if (!decoded.isApiError) observedProviders.add(claudeProviderFromModelId(boundedClaudeModel(e.message.model))); + // The hash preserves API identity across copied files without persisting + // a raw provider ID in the cache. Different ID kinds/providers cannot meet. + recordClaudeAssistantTurn(rec, turns, latState, msgState, ms, decoded, withTurns, claudeCrossFileIdentity(e)); } flushClaudeMessages(rec, msgState, windowLog); + if (observedProviders.size === 1 && !observedProviders.has(null)) { + rec.sessionOrigin.thirdPartyProvider = observedProviders.values().next().value; + rec.sessionOrigin.thirdPartyProviderBasis = 'assistant-model-id'; + } + rec.title = maskSecrets(titleState.aiTitle || clip(titleState.firstPrompt)) || '(untitled)'; if (rec.project === 'unknown') applyProject(rec, projectLabel(null, dirName)); - return { session: seal(rec), turns }; + return { session: seal(rec), turns, parseStats }; } function codexParseStats() { return { - legacyEvents: 0, itemCompletedEvents: 0, tokenCountEvents: 0, + legacyEvents: 0, itemCompletedEvents: 0, tokenCountEvents: 0, totalOnlyTokenCountEvents: 0, prompts: 0, responses: 0, unknownItemTypes: {}, unknownItemTypeOverflow: 0, // Oversized rollout lines the streaming reader clipped instead of parsing // (codex-rollout-reader.mjs); always 0 for a rollout read as a string. @@ -839,20 +922,26 @@ function recordCodexUnknownType(stats, type) { * `handleCodexTurnContext`'s own `rec.project === 'unknown'` check — a * DIFFERENT gate that coincides with this one in the common case but is * not "the same rule" as this latch. */ -function handleCodexMeta(rec, metaState, decoded) { +function handleCodexMeta(rec, metaState, decoded, payload) { if (metaState.seen) return; metaState.seen = true; if (typeof decoded.sessionId === 'string' && decoded.sessionId) rec.id = decoded.sessionId; if (typeof decoded.cwd === 'string') applyProject(rec, projectLabel(decoded.cwd, null, repoRootOf(decoded.cwd))); if (typeof decoded.cwd === 'string') rec.projectEvidence = observeUsageProject(decoded.cwd); if (typeof decoded.threadSource === 'string') rec.threadSource = decoded.threadSource; + // Observed Codex shape: source.subagent.thread_spawn.parent_thread_id. + // Accept only a UUID-shaped identifier; arbitrary source objects are never + // copied into the usage record or used to infer a parent. + const parentId = payload?.source?.subagent?.thread_spawn?.parent_thread_id; + if (typeof parentId === 'string' && /^[0-9a-f]{8}-[0-9a-f]{4}-[1-8][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/iu.test(parentId) + && parentId !== rec.id) rec.parentSessionId = parentId; if (decoded.provider) { rec.inferenceProvider = decoded.provider; rec.providerProvenance = 'observed'; } } -function handleCodexTurnContext(rec, decoded, payload) { +function handleCodexTurnContext(rec, decoded, payload, captureEffort = true) { if (!rec.projectEvidence && typeof decoded.cwd === 'string') rec.projectEvidence = observeUsageProject(decoded.cwd); if (typeof decoded.model === 'string' && !rec.models.includes(decoded.model)) rec.models.push(decoded.model); if (decoded.provider) { @@ -874,6 +963,11 @@ function handleCodexTurnContext(rec, decoded, payload) { : payload.sandbox_policy; const m = normalizeMode({ host: 'codex', approvalPolicy: payload.approval_policy, sandboxPolicy: sandbox }); if (m.raw) { rec.mode = m.mode; rec.modeRaw = m.raw; } + if (captureEffort && ['none', 'minimal', 'low', 'medium', 'high', 'xhigh'].includes(payload.effort)) { + rec.codexEffort ??= { last: null, counts: {} }; + rec.codexEffort.last = payload.effort; + rec.codexEffort.counts[payload.effort] = (rec.codexEffort.counts[payload.effort] ?? 0) + 1; + } } /** Normalize one token_count event's rate-limit windows (primary/secondary), @@ -916,7 +1010,15 @@ function applyCodexRateLimit(rec, rl, ms) { * thread's own usage nor a context or rate-limit observation of it. */ function handleCodexTokenCount(rec, stats, usageState, decoded, ms, replay) { stats.tokenCountEvents++; - walkCodexTokenCount(usageState.walk, decoded.usage.total, ms, replay, localDay); + const total = decoded.usage.total; + if (!replay && Number.isFinite(Number(total?.total_tokens)) && Number(total.total_tokens) > 0 + && !['input_tokens', 'cached_input_tokens', 'output_tokens'].some((field) => + Number.isFinite(Number(total[field])) && Number(total[field]) > 0)) { + // A total alone cannot establish input, cache or output, so it cannot be + // priced or folded into a component row. Preserve the observed gap. + stats.totalOnlyTokenCountEvents++; + } + walkCodexTokenCount(usageState.walk, decoded.usage.total, ms, replay, localDay, usageState.importOwnership ? decoded.usage.last : null); if (replay) return; // Codex re-emits an identical token_count (measured: ~2.8% of events) with // the SAME cumulative total when no new model call happened; that is a @@ -956,7 +1058,16 @@ function handleCodexTaskStarted(rec, latState, payload) { * awaiting approval overnight arrives as a multi-hour "response" that the * prompt-gap path would have discarded. A non-null `error` counts as an * exception regardless of whether the fallback sample fires. */ -function handleCodexTaskComplete(rec, latState, payload) { +function handleCodexTaskComplete(rec, latState, payload, captureFirstToken = true) { + const first = payload.time_to_first_token_ms; + if (captureFirstToken && typeof first === 'number' && Number.isFinite(first) && first >= 0 + && first <= MAX_LATENCY_SAMPLE_SECONDS * 1000) { + rec.firstTokenMs ??= { count: 0, total: 0, min: first, max: first, provenance: 'host-observed' }; + rec.firstTokenMs.count++; + rec.firstTokenMs.total += first; + rec.firstTokenMs.min = Math.min(rec.firstTokenMs.min, first); + rec.firstTokenMs.max = Math.max(rec.firstTokenMs.max, first); + } const duration = Number(payload.duration_ms); if (latState.turnStartedAt !== null && Number.isFinite(duration) && duration / 1000 <= MAX_LATENCY_SAMPLE_SECONDS) { @@ -1003,7 +1114,7 @@ function handleCodexUserMessage(rec, turns, titleState, latState, decoded, ms, w rec.prompts++; if (!titleState.firstPrompt) titleState.firstPrompt = text; // Opens the prompt→agent-message latency window; closed by the first - // following handleCodexAssistantMessage (mirrors Claude's latState, Task 3). + // following handleCodexAssistantMessage (mirrors Claude's latState). latState.pendingPromptMs = ms; // Codex's kind is exactly this gate's verdict (see the turn row below), so // a gated message contributes no fingerprint — the layer sits behind the @@ -1066,8 +1177,8 @@ const CODEX_TOOL_ITEM_TYPES = new Set([ /** `item_completed` item types the host emits that are UNDERSTOOD and are * neither a message nor a tool: model reasoning, sub-agent lifecycle notes, * image views, extension calls, web searches and context compaction. They - * carry no usage or turn evidence this parser needs, so they are recognised - * and dropped. Without this list every scan raised the `unknown-item-types` + * carry no token usage; a completed compaction is separately counted as + * context evidence. Without this list every scan raised the `unknown-item-types` * warning permanently (six kinds landed in the 32-kind cap), which taught * readers to ignore the one diagnostic meant to flag a genuinely new shape. * Only a type in NEITHER set is unknown. */ @@ -1087,8 +1198,15 @@ function handleCodexEventMsg(rec, turns, stats, titleState, usageState, latState // thread's: they must not open latency windows, sample a context window or // count the parent's aborts against the child. if (replay && ['task_started', 'task_complete', 'turn_aborted'].includes(payload.type)) return; - if (payload.type === 'task_started') { handleCodexTaskStarted(rec, latState, payload); return; } - if (payload.type === 'task_complete') { handleCodexTaskComplete(rec, latState, payload); return; } + if (payload.type === 'task_started') { + noteCodexCompactionTurn(usageState, payload.turn_id); + handleCodexTaskStarted(rec, latState, payload); + return; + } + if (payload.type === 'task_complete') { + handleCodexTaskComplete(rec, latState, payload, !usageState.unprovable); + return; + } if (payload.type === 'turn_aborted') { rec.aborts++; // An interrupted turn leaves no valid latency evidence behind it: a @@ -1103,6 +1221,9 @@ function handleCodexEventMsg(rec, turns, stats, titleState, usageState, latState if (decoded.generation === 'legacy') stats.legacyEvents++; else if (decoded.generation === 'item') stats.itemCompletedEvents++; if (decoded.unknownItemType) { + if (!replay && !usageState.unprovable && decoded.unknownItemType === 'ContextCompaction') { + noteCodexCompaction(usageState, 'item', payload.turn_id); + } // A type this parser tallies is a type it UNDERSTANDS. Recording it as an // unknown kind too made the four tool items simultaneously "tools" in the // scorecard and "unknown kinds" in sourceHealth — raising the @@ -1133,17 +1254,52 @@ function rawPayload(e) { return e?.payload && typeof e.payload === 'object' ? e.payload : {}; } +/** Keep IDs transient. Top-level `compacted` has no turn ID, so its nearest + * preceding task-start segment is only pairing evidence, not proof of a + * one-to-one relationship with ContextCompaction. */ +function noteCodexCompactionTurn(state, turnId) { + const key = `turn-${++state.turnSequence}`; + state.currentCompactionTurn = key; + if (typeof turnId === 'string' && turnId.length <= 256) state.compactionTurnIds.set(turnId, key); +} + +function noteCodexCompaction(state, shape, turnId = null) { + // An item ID without a matching observed task_start cannot establish a + // separate turn from a nearby ID-less top-level compacted envelope. + const explicitKey = typeof turnId === 'string' && turnId.length <= 256 + ? state.compactionTurnIds.get(turnId) : null; + const key = explicitKey ?? state.currentCompactionTurn ?? 'unscoped'; + const counts = state.compactionTurns.get(key) ?? { completed: 0, item: 0 }; + counts[shape]++; + state.compactionTurns.set(key, counts); +} + +function finalizeCodexCompactions(rec, state) { + let lowerBound = 0; + let upperBound = 0; + for (const { completed, item } of state.compactionTurns.values()) { + lowerBound += Math.max(completed, item); + upperBound += completed + item; + } + rec.compactions = lowerBound; + rec.compactionEvidence = { lowerBound, upperBound }; +} + /** One line of a Codex rollout, dispatched on its decoded type. */ function processCodexLine(rec, turns, stats, titleState, usageState, latState, metaState, e, ms, withTurns) { const decoded = decodeCodexRecord(e); - if (decoded.type === 'meta') { handleCodexMeta(rec, metaState, decoded); return; } + const replay = isCodexReplayLine(usageState.boundary, e); + if (decoded.type === 'meta') { handleCodexMeta(rec, metaState, decoded, rawPayload(e)); return; } if (decoded.type === 'turnContext') { - handleCodexTurnContext(rec, decoded, rawPayload(e)); + if (!replay) handleCodexTurnContext(rec, decoded, rawPayload(e), !usageState.unprovable); noteCodexWalkModel(usageState.walk, decoded.model); return; } + if (e.type === 'compacted') { + if (!replay && !usageState.unprovable) noteCodexCompaction(usageState, 'completed'); + return; + } if (e.type !== 'event_msg') return; - const replay = isCodexReplayLine(usageState.boundary, e); handleCodexEventMsg(rec, turns, stats, titleState, usageState, latState, decoded, rawPayload(e), ms, withTurns, replay); } @@ -1153,12 +1309,15 @@ function processCodexLine(rec, turns, stats, titleState, usageState, latState, m * Codex activity inflated Codex responses and prompts and diluted its * coverage figures. `imported` is set ONLY on these records, and the scan * reports how many it excluded (`importedExcluded`) rather than dropping them - * silently. Parsing stops at the first imported line — nothing after it is - * read. */ + * silently. The whole source is examined for genuine later turns. */ function importedCodexSession(rec, stats) { + rec = { ...blankSession(rec.id, 'codex'), threadSource: rec.threadSource, + ...(rec.parentSessionId ? { parentSessionId: rec.parentSessionId } : {}), importEvidence: rec.importEvidence }; rec.imported = true; + rec.sessionOrigin = importedUsageRecordOrigin(); rec.title = '(imported Claude session)'; - return { session: seal(rec), turns: [], parseStats: { ...stats, imported: true } }; + return { session: seal(rec), turns: [], parseStats: { ...codexParseStats(), + importEvidence: rec.importEvidence, clippedLines: stats.clippedLines, imported: true } }; } /** The session's usage rows, from the walk (codex-usage-walk.mjs): one per @@ -1180,6 +1339,20 @@ function finalizeCodexUsage(rec, walk) { rec.reasoningOutput = reasoningOutput; } +/** Bind a mixed session only to its first declaration and native cwd evidence. */ +function finalizeMixedCodexOrigin(rec, firstMeta) { + // The first session declaration is valid origin evidence only after own + // activity is established. Replayed/later metadata never replaces it. + const p = firstMeta?.payload ?? {}; + rec.sessionOrigin = usageRecordOrigin(JSON.stringify({ type: 'session_meta', payload: { + originator: p.originator, source: p.source, thread_source: p.thread_source, + } }), 'codex'); + if (!rec.projectEvidence && typeof p.cwd === 'string') { + rec.projectEvidence = observeUsageProject(p.cwd); + applyProject(rec, projectLabel(p.cwd, null, repoRootOf(p.cwd))); + } +} + /** * Parse one Codex rollout. `total_token_usage` is CUMULATIVE, so each event's * spend is its DELTA against the previous snapshot (summing the snapshots @@ -1204,24 +1377,31 @@ function finalizeCodexUsage(rec, walk) { * so its prompts stay out of human-prompt figures. A subagent whose replay * cannot be separated (no ordinals at all) reports no usage, as before. * - * A rollout Codex imported from a Claude Code transcript - * (`external-import-turn-N`) is not Codex activity at all — see - * importedCodexSession. + * Imported turns (`external-import-turn-N`) never count. Identified native + * turns in the same file can count; missing boundaries remain unattributable. */ export function parseCodex(raw, { id, withTurns = false }) { // `raw` is the rollout text, or a streaming source (openCodexRollout) for one // too large to hold as a string: `{ head, lines, stats }`. Both feed the SAME // walk below, so the two paths cannot drift. + const readStats = { malformedRecords: 0 }; const source = typeof raw === 'string' - ? { head: raw, lines: { [Symbol.iterator]: () => jsonLines(raw) } } + ? { head: raw, stats: readStats, lines: { [Symbol.iterator]: () => jsonLines(raw, readStats) } } : raw; const rec = blankSession(id, 'codex'); - rec.sessionOrigin = usageSessionOrigin(source.head, 'codex'); + rec.sessionOrigin = usageRecordOrigin(source.head, 'codex'); const turns = []; const stats = codexParseStats(); const lines = source.lines; const plan = codexReplayPlan(lines); - const usageState = { walk: newCodexUsageWalk({ unattributable: plan.unprovable }), boundary: plan.boundary }; + let hasImports = false; + for (const e of lines) { if (isCodexImportedLine(e)) { hasImports = true; break; } } + const ownership = newCodexTurnOwnership({ hasImports }); + let firstMeta = null; + let genuineActivity = false; + const usageState = { walk: newCodexUsageWalk({ unattributable: plan.unprovable }), boundary: plan.boundary, + unprovable: plan.unprovable, importOwnership: hasImports, compactionTurns: new Map(), + compactionTurnIds: new Map(), currentCompactionTurn: null, turnSequence: 0 }; const titleState = { firstPrompt: '' }; // Opened by task_started (turn start remembered), closed either by a // prompt→agent-message gap sample or by task_complete's own duration_ms @@ -1234,13 +1414,48 @@ export function parseCodex(raw, { id, withTurns = false }) { const metaState = { seen: false }; for (const e of lines) { - if (isCodexImportedLine(e)) return importedCodexSession(rec, stats); const ms = toMs(e.timestamp); + if (hasImports) { + const nativeBefore = ownership.nativeRecords; + const owner = codexTurnOwner(ownership, e); + const replay = plan.unprovable || isCodexReplayLine(plan.boundary, e); + if (replay) ownership.nativeRecords = nativeBefore; + if (e.type === 'session_meta') { + if (!firstMeta) firstMeta = e; + // Keep the first identity, but a copied cwd is not project evidence. + const payload = { ...rawPayload(e), cwd: undefined }; + handleCodexMeta(rec, metaState, decodeCodexRecord({ ...e, payload }), payload); + continue; + } + if (owner !== 'native' || replay) { + const decoded = decodeCodexRecord(e); + if (decoded.type === 'tokenCount') { + // Excluded snapshots still advance the cumulative baseline. + walkCodexTokenCount(usageState.walk, decoded.usage.total, ms, true, localDay); + } + latState.pendingPromptMs = null; + latState.turnStartedAt = null; + continue; + } + if (e.type === 'event_msg' && ['user_message', 'agent_message', 'item_completed', 'token_count'].includes(e.payload?.type)) genuineActivity = true; + } noteSpan(rec, ms); processCodexLine(rec, turns, stats, titleState, usageState, latState, metaState, e, ms, withTurns); } + if (hasImports) { + const malformedRecords = source.stats?.malformedRecords ?? 0; + rec.importEvidence = { ...codexImportEvidence(ownership), malformedRecords, + ownershipComplete: ownership.ownershipComplete && !plan.unprovable + && (source.stats?.clippedLines ?? 0) === 0 && malformedRecords === 0 }; + if (!rec.importEvidence.ownershipComplete) rec.importEvidence.nativeRecords = 0; + stats.importEvidence = rec.importEvidence; + stats.clippedLines = source.stats?.clippedLines ?? 0; + if (!genuineActivity || !rec.importEvidence.ownershipComplete) return importedCodexSession(rec, stats); + finalizeMixedCodexOrigin(rec, firstMeta); + } finalizeCodexUsage(rec, usageState.walk); + finalizeCodexCompactions(rec, usageState); stats.clippedLines = source.stats?.clippedLines ?? 0; rec.title = maskSecrets(clip(titleState.firstPrompt)) || '(untitled)'; return { session: seal(rec), turns, parseStats: stats }; diff --git a/src/lib/usage-project-evidence.mjs b/src/lib/usage-project-evidence.mjs index 49918afc..88891389 100644 --- a/src/lib/usage-project-evidence.mjs +++ b/src/lib/usage-project-evidence.mjs @@ -7,6 +7,7 @@ import { inspectProjectIdentity } from './footprint/project-identity.mjs'; import { transcriptSessionOrigin } from './footprint/session-origin.mjs'; import { isImportedCodexRollout } from './codex-import-marker.mjs'; import { safeProjectLabel } from './live/project-label.mjs'; +import { classifySessionSurface } from './session-surface.mjs'; import { claudeDir, codexDir, opencodeDir, configDir } from './paths.mjs'; const CACHE = new Map(); @@ -56,7 +57,7 @@ export function observeUsageProject(cwd, { observedAt = Date.now(), cache = CACH return value; } -/** Same bounded head and exact origin allowlists as footprint discovery. An +/** Same bounded head and declared-origin token validation as footprint discovery. An * imported Codex copy of a Claude Code transcript declares the ChatGPT desktop * app as its originator but is not a session from it (ADR-0060 §3). */ export function usageSessionOrigin(raw, host) { @@ -65,3 +66,23 @@ export function usageSessionOrigin(raw, host) { if (host === 'codex' && isImportedCodexRollout(lines)) return { origin: 'unknown', evidence: 'imported-copy' }; return transcriptSessionOrigin(lines, host); } + +/** Serialize the classifier dimensions for usage records. The footprint + * adapter keeps them non-enumerable to preserve its legacy origin contract. */ +export function usageRecordOrigin(raw, host) { + const legacy = usageSessionOrigin(raw, host); + if (legacy.evidence === 'imported-copy') return importedUsageRecordOrigin(); + return { + ...legacy, + surface: legacy.surface, initiator: legacy.initiator, label: legacy.label, + rawEvidence: legacy.rawEvidence, attributes: legacy.attributes, + thirdPartyProvider: legacy.thirdPartyProvider, + }; +} + +/** The parser also calls this when an import marker falls beyond the bounded + * head. A copied declaration never establishes a genuine session surface. */ +export function importedUsageRecordOrigin() { + return { origin: 'unknown', evidence: 'imported-copy', + ...classifySessionSurface({ host: 'codex', importedCopy: true }) }; +} diff --git a/src/lib/versions.mjs b/src/lib/versions.mjs index 7e541a9a..8617c98b 100644 --- a/src/lib/versions.mjs +++ b/src/lib/versions.mjs @@ -159,27 +159,26 @@ async function fetchSelfCandidate(tags, cachedBest, fetchLatest) { return { best, observed, answered }; } -/** The self record to save after a lookup, or null to save nothing. A live - * winner is an observation. When every lookup failed, the record stays and - * `last` is restamped, so the next lookup waits one TTL window; `observedAt` - * keeps when the recorded best was seen (a record written before it existed - * takes the previous `last`). A partial answer that leaves a cached candidate - * winning renews nothing. Neither does a failure when the recorded best is - * one this install cannot use (a `next` candidate on a stable install): such - * a record is never fresh, so a restamp would only rewrite kit.json on every - * call, and dropping the candidate would make an empty record look fresh and - * stop the lookup for a TTL window once the registry is back. */ -function selfRecord(cached, usable, { best, observed, answered }, now = Date.now()) { - if (observed) return { last: now, best, observedAt: now }; - if (answered || (cached?.best && !usable)) return null; - return { ...cached, last: now, observedAt: cached?.observedAt ?? cached?.last }; +/** `observedAt` describes the winning candidate; `last` and `lastTags` scope + * the latest completed check. `attempt` throttles a lookup that could not + * update that check. Neither stamp makes a cached winner newly observed. */ +function selfRecord(cached, usable, { best, observed, answered }, tags, now = Date.now()) { + if (observed) return { last: now, best, observedAt: now, lastTags: tags }; + if (answered || (cached?.best && !usable)) { + return { ...cached, attempt: { at: now, tags } }; + } + const { attempt: _priorAttempt, ...prior } = cached ?? {}; + return { ...prior, last: now, observedAt: cached?.observedAt ?? cached?.last, lastTags: tags }; } +const sameTags = (recorded, tags) => Array.isArray(recorded) && recorded.length === tags.length + && tags.every((tag, index) => recorded[index] === tag); + /** Look the kit up on its channels; with `record`, save what selfRecord keeps. * Returns the winning candidate. */ async function lookUpSelf(cfg, cached, { tags, cachedBest, fetchLatest, record }) { const candidate = await fetchSelfCandidate(tags, cachedBest, fetchLatest); - const entry = record ? selfRecord(cached, cachedBest, candidate) : null; + const entry = record ? selfRecord(cached, cachedBest, candidate, tags) : null; if (entry) { cfg.versionCheck = { ...cfg.versionCheck, self: entry }; try { saveKitConfig(cfg); } catch { /* read-only envs: next call re-fetches */ } @@ -191,8 +190,10 @@ async function lookUpSelf(cfg, cached, { tags, cachedBest, fetchLatest, record } * (pkgRoot). Prerelease installs also consult the `next` dist-tag — * prereleases publish there, so `latest` alone would never see them; the * higher of latest/next wins. Cached in kit.json alongside versionCheck. - * Failed lookups preserve eligible cached evidence (see selfRecord); `force` - * retries within the TTL. cacheOnly=true reports the recorded best with no + * Failed lookups preserve eligible cached evidence (see selfRecord); an + * `attempt` stamp limits partial/unusable retries to once per tag set and TTL. + * `lastTags` scopes completed checks. `force` retries within the TTL. + * cacheOnly=true reports the recorded best with no * network and no write (`ak sync --skip self`); record=false reports what the * lookup found without saving it (ADR-0063). * @param {{ pkgRoot?: string, force?: boolean, cacheOnly?: boolean, record?: boolean, @@ -208,8 +209,19 @@ export async function selfDrift({ pkgRoot, force = false, cacheOnly = false, rec const tags = installed?.includes('-') ? ['latest', 'next'] : ['latest']; const cachedBest = cached?.best && tags.includes(cached.best.tag) && isValidSemver(cached.best.version) ? cached.best : null; - const fresh = !force && cached?.last && Date.now() - cached.last < ttlMs - && (!cached.best || cachedBest); + const now = Date.now(); + const attempt = cached?.attempt; + const attemptFresh = Number.isSafeInteger(attempt?.at) && attempt.at > 0 && attempt.at <= now + && sameTags(attempt.tags, tags) + && now - attempt.at < ttlMs; + // Older records have no scope: a `next` winner proves both tags were tried; + // otherwise only the single latest channel is safe to reuse. + const lastScope = cached?.lastTags === undefined + ? (tags.length === 1 || cached?.best?.tag === 'next') + : sameTags(cached.lastTags, tags); + const lastFresh = Number.isSafeInteger(cached?.last) && cached.last > 0 && cached.last <= now + && now - cached.last < ttlMs && lastScope && (!cached.best || cachedBest); + const fresh = !force && (lastFresh || attemptFresh); const best = fresh || cacheOnly ? cachedBest : await lookUpSelf(cfg, cached, { tags, cachedBest, fetchLatest, record }); return { pkg: KIT_PKG, diff --git a/src/lib/windows-npm-shim.mjs b/src/lib/windows-npm-shim.mjs new file mode 100644 index 00000000..433ee81c --- /dev/null +++ b/src/lib/windows-npm-shim.mjs @@ -0,0 +1,83 @@ +// Recognize, never interpret, npm cmd-shim 8's plain `env node` wrapper pair. +// Fingerprints normalize only CRLF and the package entry path. Any flags, +// environment assignments, comments or custom behavior keep the PS fallback. +// Fixtures were emitted by cmd-shim 8.0.0; the ps1 hash matches the native +// Ruflo 3.48.0 CI artifact. No package manager or package code runs here. +import fs from 'node:fs'; +import path from 'node:path'; +import { createHash } from 'node:crypto'; + +const CMD_TEMPLATE = '46268d032014c41f0112ffc0f52a9b297a809c289587e984e5919c65f29dad79'; +const PS_TEMPLATE = '11a7c411407320ddc34a9ae9633d6372b1867994ff723693be5be16148dd45a8'; + +export function windowsEnvValue(env, key) { + return env[Object.keys(env).sort().find((name) => name.toUpperCase() === key.toUpperCase())]; +} + +/** Windows treats environment names case-insensitively. Remove a replaced + * spelling before applying the override, so Node's sorted env selection and + * the resolver agree about the caller's PATH and runtime. */ +export function mergeWindowsEnv(base, overrides) { + const merged = { ...base }; + for (const [key, value] of Object.entries(overrides)) { + for (const old of Object.keys(merged)) if (old.toUpperCase() === key.toUpperCase()) delete merged[old]; + merged[key] = value; + } + return merged; +} + +function readBounded(file) { + const stat = fs.statSync(file); + if (!stat.isFile() || stat.size > 65536) throw Error('not a bounded shim or manifest'); + return fs.readFileSync(file, 'utf8').replaceAll('\r\n', '\n'); +} +const fingerprint = (source, target) => createHash('sha256').update(source.replaceAll(target, '')).digest('hex'); +const isFile = (file) => { try { return fs.statSync(file).isFile(); } catch { return false; } }; +function inside(root, file) { + const relative = path.relative(root, file); + return relative && !relative.split(path.sep).includes('..') && !path.isAbsolute(relative); +} +function plainNodeShebang(file) { + const fd = fs.openSync(file, 'r'); + try { + const data = Buffer.alloc(128); + const size = fs.readSync(fd, data, 0, data.length, 0); + return /^#!\/usr\/bin\/env node\r?\n/.test(data.subarray(0, size).toString('utf8')); + } finally { fs.closeSync(fd); } +} + +/** Map ONLY the PATH-selected, unchanged npm wrapper pair to its own manifest's + * public bin. Never consult a global-root guess or bypass an internal bundle. + * Return null when ownership, containment, template or runtime is uncertain. */ +export function npmShimInvocation(candidate, args, env) { + try { + const cmd = readBounded(candidate); + const ps = readBounded(`${candidate.slice(0, -4)}.ps1`); + const target = ps.match(/"\$basedir\/(node_modules\/[A-Za-z0-9@_./-]+)"/)?.[1]; + if (!target || target.split('/').some((part) => !part || part === '.' || part === '..')) return null; + if (fingerprint(ps, target) !== PS_TEMPLATE + || fingerprint(cmd, target.replaceAll('/', '\\')) !== CMD_TEMPLATE) return null; + const parts = target.split('/'); + const packageName = parts[1].startsWith('@') ? `${parts[1]}/${parts[2]}` : parts[1]; + const base = path.resolve(path.dirname(candidate)); + const packageRoot = path.join(base, 'node_modules', ...packageName.split('/')); + const pkg = JSON.parse(readBounded(path.join(packageRoot, 'package.json'))); + if (pkg.name !== packageName) return null; + const name = path.basename(candidate).slice(0, -4); + const declared = typeof pkg.bin === 'string' + ? (packageName.split('/').at(-1) === name ? pkg.bin : null) : pkg.bin?.[name]; + if (typeof declared !== 'string' || !declared || path.win32.isAbsolute(declared) + || path.isAbsolute(declared) || declared.split(/[\\/]/).includes('..')) return null; + const entry = path.resolve(base, ...parts); + if (path.resolve(packageRoot, declared) !== entry || !isFile(entry)) return null; + const realPackage = fs.realpathSync(packageRoot); + if (!inside(fs.realpathSync(base), realPackage) + || !inside(realPackage, fs.realpathSync(entry)) || !plainNodeShebang(entry)) return null; + // npm chooses adjacent node.exe first, then node.exe on the caller's PATH. + // Do not substitute agentic-kit's current runtime or a different npm tree. + const adjacent = path.join(base, 'node.exe'); + const node = isFile(adjacent) ? adjacent : (windowsEnvValue(env, 'PATH') || '') + .split(path.delimiter).filter(Boolean).map((dir) => path.resolve(dir, 'node.exe')).find(isFile); + return node ? { command: node, args: [entry, ...args], resolved: true } : null; + } catch { return null; } +} diff --git a/tests/dashboard.test.cjs b/tests/dashboard.test.cjs index 016c3d18..0a063c01 100644 --- a/tests/dashboard.test.cjs +++ b/tests/dashboard.test.cjs @@ -503,7 +503,7 @@ async function main() { assert(resolvesInsideRoot(root, path.resolve(path.sep, 'etc', 'passwd')) === false, 'an absolute path escapes the root'); }); - // ── namespaced subagent session ids (Task 5 round 2) ── + // ── namespaced subagent session ids ── await test('parseNamespacedSessionId accepts /, decodes percent-encoding, and parses named subagents', async () => { const r = parseNamespacedSessionId('fc8c05e0-c311-456e-a226-6dac4279199b/agent-a2fc4593254cc01b9'); assert(r && r.parentId === 'fc8c05e0-c311-456e-a226-6dac4279199b' && r.stem === 'agent-a2fc4593254cc01b9', @@ -749,7 +749,7 @@ async function main() { 'readSession must NOT be called for any hostile id — got ' + (spy.calls.readSession.length - before) + ' call(s)'); }); - // ── namespaced subagent session ids reach readSession end-to-end (Task 5 round 2) ── + // ── namespaced subagent session ids reach readSession end-to-end ── await test('GET /api/session/:parentId/:stem → a namespaced subagent id reaches readSession end-to-end', async () => { const before = spy.calls.readSession.length; const r = await get(usageSrv.url + 'api/session/parent-uuid/agent-a2fc4593254cc01b9', usageSrv.token); @@ -1127,44 +1127,20 @@ async function main() { await sysSrv.close(); } - await test('?refresh=deep is single-flight — two concurrent refreshes share one scan', async () => { - let release; - const gate = new Promise((resolve) => { release = resolve; }); - // Gate the CHEAP tier, not the deep one. Both requests then resume from the - // same promise in one microtask drain, and runDeep's first act is a - // setImmediate — so the second request PROVABLY reaches refreshDeep() while - // the first still holds the slot. Racing two bare HTTP requests would be - // testing the scheduler, not the single-flight rule. - const fx = systemFixture({ collectors: { runtime: async () => { await gate; return runtimeCensus(); } } }); - let maintenanceScans = 0; - const maintenance = { - async report() { return {}; }, - async scan() { maintenanceScans += 1; return {}; }, - async plan() { return {}; }, - }; + await test('GET /api/system rejects scan queries before reading the collector', async () => { + const fx = systemFixture(); const srv = await startDashboard({ port: 0, cwd: fixture, fetchStatus: async () => STUB_STATUS, usage: spyUsage().api, - system: fx.collector, maintenance, + system: fx.collector, }); try { - const both = Promise.all([ - get(srv.url + 'api/system?refresh=deep', srv.token), - get(srv.url + 'api/system?refresh=deep', srv.token), - ]); - await eventually(() => fx.calls.runtime === 2, 'both refreshes must reach the collector'); - release(); - const [a, b] = await both; - assert(a.status === 200 && b.status === 200, 'both refreshes must answer 200'); - const scanA = JSON.parse(a.body).scan; - assert(scanA.running === true && scanA.phase !== 'idle', - 'a refresh must report the scan it started, got ' + JSON.stringify(scanA)); - await eventually(() => fx.calls.persist === 1, 'the shared scan must run to completion'); - await eventually(() => maintenanceScans === 1, 'the completed scan must refresh Maintenance evidence once'); - assert(fx.calls.install === 1 && fx.calls.storage === 1 - && fx.calls.catalog === 1 && fx.calls.projects === 1, - 'the deep collectors ran twice — the single-flight slot did not hold: ' + JSON.stringify(fx.calls)); - assert(fx.calls.persist === 1, 'a shared scan must write exactly one snapshot'); - assert(maintenanceScans === 1, 'two attached System requests must not double-run Maintenance providers'); + for (const query of ['refresh=deep', 'refresh=deep&refresh=scan', 'trees=1']) { + const response = await get(srv.url + 'api/system?' + query, srv.token); + assert(response.status === 400, 'scan query must be rejected: ' + query); + contains(response.body, 'start a refresh with POST /api/refresh'); + } + assert(fx.calls.runtime === 0 && fx.calls.persist === 0, + 'rejected GET queries must not run collectors or persist measurements'); } finally { await srv.close(); } @@ -1321,7 +1297,7 @@ async function main() { const r = await get(uiSrv.url); contains(r.body, 'var prov=d.byHost||{}'); contains(r.body, 'var host=reportedIdentity(sx.host)||"unknown"'); - contains(r.body, 'var provider=reportedIdentity(sx.provider)'); + contains(r.body, 'var provider=sessionProviderPresentation(sx).label'); contains(r.body, 'Execution host: '); contains(r.body, 'Inference provider: '); contains(r.body, '"inference provider"'); @@ -1411,13 +1387,14 @@ async function main() { contains(r.body, 'POLL_COOLDOWN_MS=3000'); }); - await test('every refresh path is single-flight + cooldown guarded', async () => { + await test('Refresh control guards duplicate POSTs and status polling retains its cooldown', async () => { const r = await get(uiSrv.url); - const fn = r.body.slice(r.body.indexOf('function refreshAll(')); - const body = fn.slice(0, fn.indexOf('\n function ')); - assert(/inflight/.test(body), 'refreshAll must consult the single-flight flag'); - assert(/POLL_COOLDOWN_MS/.test(body), 'refreshAll must consult the cooldown'); - assert(/setInterval\(refreshAll/.test(r.body), 'the automatic poll must go through the SAME guarded path'); + const fn = r.body.slice(r.body.indexOf('function startRefresh(')); + const body = fn.slice(0, fn.indexOf('function refreshProjectTrees')); + contains(body, 'if(refreshBusy)return Promise.resolve(false)'); + contains(body, "fetch('/api/refresh',{method:'POST'"); + contains(r.body, 'POLL_COOLDOWN_MS=3000'); + assert(!/setInterval\(refreshAll/.test(r.body), 'retired automatic refresh path remains'); }); await test('the Usage tab is lazy — the shared status poll never fetches /api/usage', async () => { diff --git a/tests/fixtures/dashboard-status-child.mjs b/tests/fixtures/dashboard-status-child.mjs index 435ff3ce..9d4a4b51 100644 --- a/tests/fixtures/dashboard-status-child.mjs +++ b/tests/fixtures/dashboard-status-child.mjs @@ -10,9 +10,12 @@ // real HTTP — this script only starts the server and reports where it is // listening; the parent marks ledger call boundaries itself by appending // directly to the same ndjson file between requests. -import { startDashboard } from '../../src/lib/dashboard-server.mjs'; +// An optional test-only global root makes the Ruflo component path deterministic. +import { _setGlobalRootForTest } from '../../src/lib/paths.mjs'; -const [, , cwd] = process.argv; +const [, , cwd, fakeGlobalRoot] = process.argv; +if (fakeGlobalRoot) _setGlobalRootForTest(fakeGlobalRoot); +const { startDashboard } = await import('../../src/lib/dashboard-server.mjs'); const { port, token } = await startDashboard({ port: 0, cwd }); // Unbuffered, single line, parsed by the parent — printed only once the diff --git a/tests/fixtures/npm-windows-shim/license.txt b/tests/fixtures/npm-windows-shim/license.txt new file mode 100644 index 00000000..20a47625 --- /dev/null +++ b/tests/fixtures/npm-windows-shim/license.txt @@ -0,0 +1,15 @@ +The ISC License + +Copyright (c) npm, Inc. and Contributors + +Permission to use, copy, modify, and/or distribute this software for any +purpose with or without fee is hereby granted, provided that the above +copyright notice and this permission notice appear in all copies. + +THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES +WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF +MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR +ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES +WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN +ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF OR +IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. diff --git a/tests/fixtures/npm-windows-shim/ruflo.cmd b/tests/fixtures/npm-windows-shim/ruflo.cmd new file mode 100644 index 00000000..ea9a8598 --- /dev/null +++ b/tests/fixtures/npm-windows-shim/ruflo.cmd @@ -0,0 +1,17 @@ +@ECHO off +GOTO start +:find_dp0 +SET dp0=%~dp0 +EXIT /b +:start +SETLOCAL +CALL :find_dp0 + +IF EXIST "%dp0%\node.exe" ( + SET "_prog=%dp0%\node.exe" +) ELSE ( + SET "_prog=node" + SET PATHEXT=%PATHEXT:;.JS;=;% +) + +endLocal & goto #_undefined_# 2>NUL || title %COMSPEC% & "%_prog%" "%dp0%\node_modules\ruflo\bin\ruflo.js" %* diff --git a/tests/fixtures/npm-windows-shim/ruflo.ps1 b/tests/fixtures/npm-windows-shim/ruflo.ps1 new file mode 100644 index 00000000..7de4de00 --- /dev/null +++ b/tests/fixtures/npm-windows-shim/ruflo.ps1 @@ -0,0 +1,28 @@ +#!/usr/bin/env pwsh +$basedir=Split-Path $MyInvocation.MyCommand.Definition -Parent + +$exe="" +if ($PSVersionTable.PSVersion -lt "6.0" -or $IsWindows) { + # Fix case when both the Windows and Linux builds of Node + # are installed in the same directory + $exe=".exe" +} +$ret=0 +if (Test-Path "$basedir/node$exe") { + # Support pipeline input + if ($MyInvocation.ExpectingInput) { + $input | & "$basedir/node$exe" "$basedir/node_modules/ruflo/bin/ruflo.js" $args + } else { + & "$basedir/node$exe" "$basedir/node_modules/ruflo/bin/ruflo.js" $args + } + $ret=$LASTEXITCODE +} else { + # Support pipeline input + if ($MyInvocation.ExpectingInput) { + $input | & "node$exe" "$basedir/node_modules/ruflo/bin/ruflo.js" $args + } else { + & "node$exe" "$basedir/node_modules/ruflo/bin/ruflo.js" $args + } + $ret=$LASTEXITCODE +} +exit $ret diff --git a/tests/helpers/comment-label-guard.mjs b/tests/helpers/comment-label-guard.mjs new file mode 100644 index 00000000..db51e6e1 --- /dev/null +++ b/tests/helpers/comment-label-guard.mjs @@ -0,0 +1,84 @@ +import { Linter } from 'eslint'; + +const labels = /\bTask \d+(?:\.\d+)?[a-z]?\b|\b[Ff]ix round \d|final-review fix|\bBranch \d+[a-z]?\b[^.\n]{0,20}\bTask\b/; +const testNames = new Set(['test', 'it', 'describe', 'suite']); + +const modifiers = new Set(['only', 'skip', 'todo', 'concurrent', 'sequential', 'failing']); +const memberName = expression => expression.computed ? expression.property.value : expression.property.name; + +// Static contract: known declaration names/modifiers, each(table)(title), and +// first callback parameters of direct declarations (never each table data). +// An unbound `t.test` +// is accepted as the conventional context form; bound receivers must resolve +// to a declaration callback parameter. Runtime aliases are not evaluated. +function isTestContext(receiver, sourceCode) { + if (receiver.type !== 'Identifier') return false; + let scope = sourceCode.getScope(receiver); + while (scope) { + const variable = scope.set.get(receiver.name); + if (variable) { + return variable.defs.some(definition => { + const fn = definition.node; + const call = fn.parent; + return definition.type === 'Parameter' && fn.params[0] === definition.name + && call?.type === 'CallExpression' && call.arguments.includes(fn) + && isTestCall(call.callee, sourceCode, false); + }); + } + scope = scope.upper; + } + return receiver.name === 't'; +} + +function isTestCall(expression, sourceCode, allowEach = true) { + if (expression.type === 'Identifier') return testNames.has(expression.name); + if (expression.type === 'MemberExpression') { + const name = memberName(expression); + if (name === 'test') return isTestContext(expression.object, sourceCode); + return modifiers.has(name) && isTestCall(expression.object, sourceCode, allowEach); + } + return allowEach && expression.type === 'CallExpression' + && expression.callee.type === 'MemberExpression' + && memberName(expression.callee) === 'each' + && isTestCall(expression.callee.object, sourceCode); +} + +/** Parse real comments and static test titles through the existing lint parser. + * Strings, regex literals and template raw text stay opaque; syntax errors fail + * closed rather than quietly exempting malformed source from the guard. + */ +export function inspectCommentLabels(source, file = 'fixture.mjs') { + const findings = []; + const add = (text, node, kind) => { + const match = labels.exec(text); + if (match) findings.push({ + line: node.loc.start.line + text.slice(0, match.index).split('\n').length - 1, + kind, text: text.trim(), start: node.range[0], end: node.range[1], + }); + }; + const rule = { + create(context) { + return { + Program() { + for (const comment of context.sourceCode.getAllComments()) add(comment.value, comment, 'comment'); + }, + CallExpression(node) { + if (!isTestCall(node.callee, context.sourceCode)) return; + const title = node.arguments[0]; + if (title?.type === 'Literal' && typeof title.value === 'string') add(title.value, title, 'test title'); + if (title?.type === 'TemplateLiteral') { + for (const part of title.quasis) add(part.value.cooked ?? part.value.raw, part, 'test title'); + } + }, + }; + }, + }; + const messages = new Linter().verify(source, [{ + files: ['**/*.{js,mjs,cjs}'], + languageOptions: { ecmaVersion: 'latest', sourceType: file.endsWith('.cjs') ? 'commonjs' : 'module' }, + plugins: { labels: { rules: { references: rule } } }, + rules: { 'labels/references': 'error' }, + }], { filename: file, allowInlineConfig: false, reportUnusedDisableDirectives: false }); + if (messages.length) throw new Error(`Cannot parse ${file}: ${messages.map(message => `${message.line}:${message.message}`).join('; ')}`); + return findings.sort((a, b) => a.start - b.start); +} diff --git a/tests/helpers/spawn-guard.mjs b/tests/helpers/spawn-guard.mjs index e783badd..e06cdfea 100644 --- a/tests/helpers/spawn-guard.mjs +++ b/tests/helpers/spawn-guard.mjs @@ -1,7 +1,7 @@ // Test-only spawn ledger, meant to be loaded via `node --import`. Side-effect // free unless AK_SPAWN_LEDGER_FILE is set, so it is always safe to preload. // -// Branch 6a Task 8a (Ruling C): an earlier plan wanted a spawn-ledger seam +// A spawn ledger must cover every spawn path: a proposed seam // inside src/lib/exec.mjs's run(), but a repo-wide `grep -rln "child_process" // src/` found ~30 files that spawn directly, not through exec.mjs. A ledger // there would under-count and let a zero-spawn test pass vacuously on any @@ -41,8 +41,8 @@ if (ledgerFile) { } }; - // The brief's four forms, plus execSync/fork: empirically (see the Task 8a - // report) `exec` already routes through `execFile` internally, so it needs + // Wrap spawn, execFile, execFileSync, spawnSync, execSync and fork. + // `exec` already routes through `execFile` internally, so it needs // no separate wrapper, but `execSync` and `fork` call module-internal // implementations that bypass the four-function wrap even after // syncBuiltinESMExports() — confirmed nothing under src/ calls either diff --git a/tests/kit/about-install-edit-render.test.mjs b/tests/kit/about-install-edit-render.test.mjs new file mode 100644 index 00000000..3af9b5c6 --- /dev/null +++ b/tests/kit/about-install-edit-render.test.mjs @@ -0,0 +1,47 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import vm from 'node:vm'; +import { RANK, esc } from '../../src/lib/dashboard/groups.mjs'; +import { RUFLO_PIN_NOTE } from '../../src/lib/install-edits.mjs'; +import { installEditRows } from '../../src/commands/status/sections/natives.mjs'; + +const ABOUT_SOURCE = fs.readFileSync(new URL('../../src/lib/dashboard/client/about.mjs', import.meta.url), 'utf8'); +const context = vm.createContext({ RANK, esc, sourceHostIcon: () => '', aboutHostChip: () => null }); +vm.runInContext(ABOUT_SOURCE.replace(/^import .*;$/gm, '').replace(/\bexport /g, ''), context); + +function renderCard(id, rows) { + context.__entry = { id, name: id, category: 'engine-memory', tagline: '', paragraph: '', icon: { ref: 'R' } }; + context.__rows = rows; + return vm.runInContext('aboutCard(__entry, { rows: __rows })', context); +} + +test('About renders one escaped Ruflo install-edit line from the real natives row', () => { + const rufloRoot = '/test/ruflo'; + const rows = installEditRows([{ + state: 'applied', + file: `${rufloRoot}/node_modules/@claude-flow/cli/package.json`, + section: 'optionalDependencies', + name: 'better-sqlite3', + from: '', + to: '^12.10.0', + }], { rufloRoot }); + assert.equal(rows.length, 1); + assert.match(rows[0].message, /^ak applied Ruflo's native SQLite pin \(ruvnet\/ruflo#2219\)/); + + const rufloCard = renderCard('ruflo', rows); + const editLines = rufloCard.match(/
    [^<]*<\/div>/g) || []; + assert.deepEqual(editLines, [`
    ${esc(rows[0].message)}
    `]); + assert.ok(editLines[0].startsWith('
    ak applied Ruflo's native SQLite pin (ruvnet/ruflo#2219)')); + assert.doesNotMatch(rufloCard, //); + assert.doesNotMatch(renderCard('agentdb', rows), /
    /); + assert.doesNotMatch(renderCard('ruflo', []), /
    /); +}); + +test('About edit-line matcher accepts the status row wording contract', () => { + const editLineSource = ABOUT_SOURCE.split('function aboutEditLine(')[1]?.split('function aboutCard(')[0]; + assert.ok(editLineSource, 'the shipped About edit-line function exists'); + const pattern = editLineSource.match(/\/\^([^/]+)\/\.test\(String\(er\.message/); + assert.ok(pattern, 'the shipped About edit-line matcher exists'); + assert.match(RUFLO_PIN_NOTE, new RegExp(`^${pattern[1]}`)); +}); diff --git a/tests/kit/ak-launcher-evidence.test.mjs b/tests/kit/ak-launcher-evidence.test.mjs index 440bbdb7..e5cfc7bb 100644 --- a/tests/kit/ak-launcher-evidence.test.mjs +++ b/tests/kit/ak-launcher-evidence.test.mjs @@ -1,4 +1,4 @@ -// Branch 6a Task 7: claudeLauncherUnavailable() (src/lib/mcp.mjs) stops +// claudeLauncherUnavailable() (src/lib/mcp.mjs) stops // spawning `which ak` (and, when ak is present, `ak x ruflo-mcp --help`) on // every plain `ak status` call where sync's register() would run — it reuses // fresh evidence instead (kind 'ak-launcher', id 'machine', 6h TTL, diff --git a/tests/kit/aqe-live-lock-process.test.mjs b/tests/kit/aqe-live-lock-process.test.mjs new file mode 100644 index 00000000..467f34b8 --- /dev/null +++ b/tests/kit/aqe-live-lock-process.test.mjs @@ -0,0 +1,138 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { mkdtempSync, existsSync, rmSync, mkdirSync } from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { createProcessScope } from '../live/aqe-live-lock-process.mjs'; +import { spawnEnv, envValue } from './helpers/home-sandbox.mjs'; +import { ownerRecord, writeOwner, readOwner, prepareRunRootHolds, + inspectRunRootHolds } from '../../scripts/run-roots.mjs'; + +function sandbox(root) { + const home = path.join(root, 'home'); + mkdirSync(home); + return spawnEnv(home); +} + +test('abort closes a call-owned child before its temporary root is removed', async () => { + const root = mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-abort-proof-')); + const controller = new AbortController(); + const scope = createProcessScope(controller.signal); + let closed = false; + try { + const marker = path.join(root, 'child-ready'); + const run = scope.launch(process.execPath, ['-e', `require('node:fs').writeFileSync(${JSON.stringify(marker)}, 'ready');setInterval(() => {}, 1000)`], { cwd: root, env: sandbox(root) }); + assert.ok(run.child.pid > 0); + const deadline = Date.now() + 2000; + while (!existsSync(marker) && Date.now() < deadline) await new Promise((resolve) => setTimeout(resolve, 10)); + assert.ok(existsSync(marker), 'the child must be running before cancellation'); + controller.abort(); + await scope.closeAll(); + closed = true; + assert.equal(run.closed, true); + assert.throws(() => scope.launch(process.execPath, [], { env: {} }), /cannot launch/); + assert.ok(existsSync(root), 'root must remain until closure is established'); + } finally { + if (!closed) await scope.closeAll(); + if (closed) rmSync(root, { recursive: true, force: true }); + } + assert.equal(existsSync(root), false); +}); + +test('spawn failure is retained without an unhandled rejection', async () => { + const scope = createProcessScope(new AbortController().signal); + const root = mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-spawn-failure-')); + const run = scope.launch(path.join(root, 'ak-missing-executable'), [], { env: sandbox(root) }); + await assert.rejects(scope.wait(run, 1000), /ENOENT/); + await scope.closeAll(); + rmSync(root, { recursive: true, force: true }); +}); + +test('requires explicit sandbox env and passes it to the child', async () => { + const root = mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-env-proof-')); + const scope = createProcessScope(new AbortController().signal); + try { + assert.throws(() => scope.launch(process.execPath, [], {}), /explicit sandbox env/); + assert.throws(() => scope.launch(process.execPath, [], { env: {} }), /requires sandbox home/); + const env = sandbox(root); + const run = scope.launch(process.execPath, + ['-e', 'console.log(JSON.stringify({home:process.env.HOME,tmp:process.env.TMPDIR,state:process.env.XDG_STATE_HOME}))'], + { cwd: root, env }); + const result = await scope.wait(run, 2000); + assert.equal(result.code, 0); + const observed = JSON.parse(result.stdout.trim()); + assert.equal(observed.home, env.HOME); + assert.equal(observed.tmp, env.TMPDIR); + assert.equal(observed.state, env.XDG_STATE_HOME); + } finally { + await scope.closeAll(); + rmSync(root, { recursive: true, force: true }); + } +}); + +test('Windows bootstrap validation accepts preserved mixed-case names without adding duplicates', async () => { + const root = mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-windows-env-')); + const home = path.join(root, 'home'); + mkdirSync(home); + const env = spawnEnv(home, {}, { platform: 'win32', env: { + SYSTEMROOT: process.env.SystemRoot ?? process.env.SYSTEMROOT ?? 'C:\\Windows', + COMSPEC: process.env.ComSpec ?? process.env.COMSPEC ?? 'C:\\Windows\\System32\\cmd.exe', + PaThExT: process.env.PATHEXT ?? '.COM;.EXE;.BAT;.CMD', + Path: process.env.PATH ?? '', + } }); + assert.equal(env.SystemRoot, undefined); + assert.equal(env.ComSpec, undefined); + assert.equal(envValue(env, 'SystemRoot', 'win32'), env.SYSTEMROOT); + const scope = createProcessScope(new AbortController().signal, { platform: 'win32' }); + try { + assert.throws(() => scope.launch(process.execPath, [], { env: { ...env, COMSPEC: '' } }), + /requires Windows process env/); + const run = scope.launch(process.execPath, ['-e', 'console.log("bootstrapped")'], { env }); + const result = await scope.wait(run, 2000); + assert.equal(result.code, 0); + assert.match(result.stdout, /bootstrapped/); + assert.equal(Object.keys(env).filter((key) => key.toUpperCase() === 'SYSTEMROOT').length, 1); + assert.equal(Object.keys(env).filter((key) => key.toUpperCase() === 'COMSPEC').length, 1); + } finally { + await scope.closeAll(); + rmSync(root, { recursive: true, force: true }); + } +}); + +test('a guarded scope holds the enclosing run root until owned children close', async () => { + const base = mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-hold-base-')); + const root = mkdtempSync(path.join(base, 'ak-suite-')); + const owner = ownerRecord(); + writeOwner(root, owner); + prepareRunRootHolds(root, owner.runId); + const beforeRoot = process.env.AK_SUITE_ROOT; + const beforeId = process.env.AK_SUITE_RUN_ID; + process.env.AK_SUITE_ROOT = root; + process.env.AK_SUITE_RUN_ID = owner.runId; + let scope; + try { + scope = createProcessScope(new AbortController().signal, { closeLimitMs: 25 }); + assert.equal(inspectRunRootHolds(root, readOwner(root).runId).unresolved, true); + const run = scope.launch(process.execPath, ['-e', 'setInterval(() => {}, 1000)'], + { env: sandbox(base) }); + assert.equal(run.closed, false); + const realKill = run.child.kill.bind(run.child); + run.child.kill = () => false; + try { + await assert.rejects(scope.closeAll(), /did not close/); + assert.equal(inspectRunRootHolds(root, owner.runId).unresolved, true); + } finally { + run.child.kill = realKill; + } + await scope.closeAll(); + assert.equal(run.closed, true); + assert.equal(inspectRunRootHolds(root, owner.runId).unresolved, false); + } finally { + if (scope) await scope.closeAll(); + if (beforeRoot === undefined) delete process.env.AK_SUITE_ROOT; + else process.env.AK_SUITE_ROOT = beforeRoot; + if (beforeId === undefined) delete process.env.AK_SUITE_RUN_ID; + else process.env.AK_SUITE_RUN_ID = beforeId; + rmSync(base, { recursive: true, force: true }); + } +}); diff --git a/tests/kit/aqe-verification.test.mjs b/tests/kit/aqe-verification.test.mjs index 1bad2acd..60e24c5b 100644 --- a/tests/kit/aqe-verification.test.mjs +++ b/tests/kit/aqe-verification.test.mjs @@ -6,39 +6,40 @@ import { aqeVerificationPassed } from '../../src/lib/aqe-verification.mjs'; test('a live lock does not hide an independent storage error', () => { assert.equal(classifyAqeStartup({ code: 0, stderr: 'locked by a live process; FsyncFailed' }).status, 'failed'); }); -// TEMPORARY (remove with the rule in classifyAqeStartup, pacphi/agentic-kit#240): the -// exact stderr agentic-qe 3.14.3 emits when a healthy patterns.rvf is held by a live -// owner (captured from a fixture; store and lock bytes were unchanged). The FsyncFailed -// comes from a create attempt AQE should not make (agentic-qe#574). Remove this test -// when a released agentic-qe fixes agentic-qe#574 and that release is the kit's floor; -// agentic-qe#719 (in 3.14.4) is a partial fix and does not remove it. -const LIVE_OWNER_CONTENTION = [ +// AQE 3.14.3 emitted this exact sequence under a live lock. The released 3.14.4 +// artifact omits FsyncFailed in macOS and Linux conformance probes; old output fails closed. +const OLD_LIVE_OWNER_CONTENTION = [ '[RVF] /p/.agentic-qe/patterns.rvf is locked by a live process (pid 70149) — not breaking the lock; degrading to SQLite for this run.', '[RVF] /p/.agentic-qe/patterns.rvf is unusable but its lock is held by a live process — leaving it alone and degrading to SQLite for this run.', '[RVF] Shared adapter init failed: RVF error 0x0303: FsyncFailed', ].join('\n'); -test('live-owner lock contention reads as busy, not a storage failure (agentic-qe#574)', () => { - const startup = classifyAqeStartup({ code: 0, stdout: '', stderr: LIVE_OWNER_CONTENTION }); - assert.equal(startup.status, 'busy'); - assert.match(startup.reason, /another live process/); - assert.match(startup.reason, /integrity unverified/); - assert.equal(aqeVerificationPassed(startup, { status: 'passed', corpus: { status: 'healthy' } }), true); +test('old live-owner FsyncFailed sequence fails closed and blocks verification', () => { + const startup = classifyAqeStartup({ code: 0, stdout: '', stderr: OLD_LIVE_OWNER_CONTENTION }); + assert.deepEqual(startup, { status: 'failed', reason: 'RVF backend failed' }); + assert.equal(aqeVerificationPassed(startup, { status: 'passed', corpus: { status: 'healthy' } }), false); }); -test('the contention rule needs all three lines; a partial match still fails', () => { - const [locked, unusable, fsync] = LIVE_OWNER_CONTENTION.split('\n'); +test('FsyncFailed fails with or without partial live-lock lines', () => { + const [locked, unusable, fsync] = OLD_LIVE_OWNER_CONTENTION.split('\n'); for (const stderr of [fsync, `${locked}\n${fsync}`, `${unusable}\n${fsync}`]) { assert.equal(classifyAqeStartup({ code: 0, stderr }).status, 'failed', stderr); } - assert.equal(classifyAqeStartup({ code: 1, stderr: LIVE_OWNER_CONTENTION }).status, 'failed'); + assert.equal(classifyAqeStartup({ code: 1, stderr: OLD_LIVE_OWNER_CONTENTION }).status, 'failed'); }); -test('live-owner contention does not hide failed embedding initialization', () => { - const stderr = `${LIVE_OWNER_CONTENTION}\nReasoningBank prewarm failed`; - assert.equal(classifyAqeStartup({ code: 0, stderr }).status, 'degraded'); +test('FsyncFailed takes precedence over failed embedding initialization', () => { + const stderr = `${OLD_LIVE_OWNER_CONTENTION}\nReasoningBank prewarm failed`; + assert.equal(classifyAqeStartup({ code: 0, stderr }).status, 'failed'); }); test('a live lock does not hide failed embedding initialization', () => { assert.equal(classifyAqeStartup({ code: 0, stderr: 'locked by a live process; prewarm failed' }).status, 'degraded'); }); +test('ordinary live lock remains busy with owner health and RVF integrity unknown', () => { + const startup = classifyAqeStartup({ code: 0, stderr: '[RVF] locked by a live process; 0x0300: LockHeld; degrading to SQLite' }); + assert.equal(startup.status, 'busy'); + assert.match(startup.reason, /SQLite fallback observed/); + assert.match(startup.reason, /owner health and RVF integrity unverified/); + assert.equal(aqeVerificationPassed(startup, { status: 'passed', corpus: { status: 'healthy' } }), true); +}); test('busy RVF permits qualified semantic proof when corpus is verified', () => { assert.equal(aqeVerificationPassed({ status: 'busy' }, { status: 'passed', corpus: { status: 'healthy' } }), true); }); diff --git a/tests/kit/cli-json-honesty.test.mjs b/tests/kit/cli-json-honesty.test.mjs index 1a4a0a34..f82b79bf 100644 --- a/tests/kit/cli-json-honesty.test.mjs +++ b/tests/kit/cli-json-honesty.test.mjs @@ -23,9 +23,9 @@ const BIN = path.join(PKG_ROOT, 'bin', 'agentic-kit.mjs'); const KIT_JSON = path.join(HOME, '.config', 'agentic-kit', 'kit.json'); /** Run `ak …args` in the sandbox. */ -function ak(args) { +function ak(args, envOverrides = {}) { return spawnSync(process.execPath, [BIN, ...args], { - cwd: PROJECT, env: spawnEnv(HOME), encoding: 'utf8', timeout: 120_000, + cwd: PROJECT, env: { ...spawnEnv(HOME), ...envOverrides }, encoding: 'utf8', timeout: 120_000, }); } @@ -124,6 +124,72 @@ for (const [args, message] of USAGE_ERRORS) { }); } +const COMMAND_USAGE_ERRORS = [ + [['usage', 'bogus', '--json'], /usage: ak usage/], + [['usage', 'score', '--window', '99', '--json'], /--window must be/], + [['usage', 'prompts', '--window', '99', '--json'], /--window must be/], + [['usage', 'score', 'extra', '--json'], /unexpected argument 'extra'/], + [['usage', 'prompts', 'extra', '--json'], /unexpected argument 'extra'/], + [['models', 'bogus', '--json'], /usage: ak models/], + [['models', 'explain', '--json'], /usage: ak models explain/], + [['models', 'plan', '--json'], /usage: ak models plan/], + [['models', 'status', 'extra', '--json'], /unexpected argument/], + [['models', 'status', '--host', 'bogus', '--json'], /unsupported model host/], + [['audit', 'bogus', '--json'], /requires the hooks or context subcommand/], + [['heal', 'bogus', '--json'], /requires the hooks subcommand/], + [['heal', 'hooks', '--yes', '--json'], /--yes requires --apply/], + [['x', 'aqe-store', 'bogus', '--json'], /usage: ak x aqe-store/], + [['x', 'aqe-embedding', 'bogus', '--json'], /aqe-embedding/], + [['x', 'codex-context', 'bogus', '--json'], /codex-context/], + [['x', 'skills', 'bogus', '--json'], /usage: ak x skills/], + [['x', 'reference', 'bogus', '--json'], /reference/], + [['x', 'statusline', 'bogus', '--json'], /usage: ak x statusline/], + [['x', 'daemon-gc', 'bogus', '--json'], /unexpected argument/], + [['x', 'harvest', 'bogus', '--json'], /unexpected argument/], + [['host', 'bogus', '--json'], /unknown host subcommand/], + [['host', 'status', 'extra', '--json'], /unexpected argument/], + [['host', 'adapters', '--dry-run', '--json'], /has no preview/], + [['host', 'adapters', 'unknown', '--json'], /experimental host-adapter surface is disabled/], +]; + +for (const [args, message] of COMMAND_USAGE_ERRORS) { + test(`ak ${args.join(' ')} reports one command-level JSON usage error`, () => { + const child = ak(args); + const out = oneJson(child); + assert.equal(child.status, 2, child.stderr); + assert.deepEqual(Object.keys(out), ['error', 'exitCode']); + assert.equal(out.exitCode, 2); + assert.match(out.error, message); + assert.match(child.stderr, message); + }); +} + +for (const enabled of ['0', '1']) { + for (const verb of ['revoke', 'revoke-grant']) { + test(`ak host adapters ${verb} --json without a name is JSON with feature flag ${enabled}`, () => { + const child = ak(['host', 'adapters', verb, '--json'], { AK_EXPERIMENTAL_HOST_ADAPTERS: enabled }); + const out = oneJson(child); + assert.equal(child.status, 2, child.stderr); + assert.deepEqual(Object.keys(out), ['error', 'exitCode']); + assert.equal(out.exitCode, 2); + assert.match(out.error, new RegExp(`usage: ak host adapters ${verb} `)); + assert.match(child.stderr, new RegExp(`usage: ak host adapters ${verb} `)); + }); + } +} + +test('models rejects an unknown verb even when no snapshot exists', () => { + const child = ak(['models', 'bogus', '--json']); + assert.equal(child.status, 2, child.stderr); + assert.match(oneJson(child).error, /usage: ak models/); +}); + +test('plain status rejects a stray positional with exit 2', () => { + const child = ak(['status', 'stray']); + assert.equal(child.status, 2, child.stderr); + assert.match(child.stdout, /unexpected argument 'stray'/); +}); + test('without --json a command-level usage error still prints on stdout', () => { const child = ak(['status', '--refresh=bogus']); assert.equal(child.status, 2); diff --git a/tests/kit/codex-mcp-convergence.test.mjs b/tests/kit/codex-mcp-convergence.test.mjs index e5134edd..ae5e4b3c 100644 --- a/tests/kit/codex-mcp-convergence.test.mjs +++ b/tests/kit/codex-mcp-convergence.test.mjs @@ -323,7 +323,7 @@ test('the kit-managed browser env child is folded into the placeholder and Codex assert.equal(fs.readFileSync(file, 'utf8'), source, 'a converged placeholder is left alone'); }); -// F4 (Branch 3 fix round 2): Codex's Claude import copies Claude Code's +// F4: Codex's Claude import copies Claude Code's // claude-flow entry by name. Since B3-D1 that entry is ak's launcher in // Claude mode, so a machine without the placeholder gains a second Ruflo // transport in Codex. ak recognizes its own launcher form under that name diff --git a/tests/kit/daemon-gc-rerecord.test.mjs b/tests/kit/daemon-gc-rerecord.test.mjs new file mode 100644 index 00000000..5adb05e5 --- /dev/null +++ b/tests/kit/daemon-gc-rerecord.test.mjs @@ -0,0 +1,62 @@ +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { sandboxHome, rmrf } from './helpers/home-sandbox.mjs'; + +const home = sandboxHome('ak-daemon-gc-rerecord'); +after(() => rmrf(home)); +const trapDir = path.join(home, 'no-such-bin'); +const processProbeMarker = path.join(home, 'process-probe-reached'); +fs.mkdirSync(trapDir, { recursive: true }); +for (const command of ['ps', 'powershell']) { + const trap = path.join(trapDir, command); + fs.writeFileSync(trap, `#!/bin/sh\n: > '${processProbeMarker}'\nexit 99\n`, { mode: 0o755 }); +} +const { run } = await import('../../src/commands/x/daemon-gc.mjs'); + +for (const [name, kill, killed, expected] of [ + ['successful reap', true, true, 2], + ['failed reap', true, false, 1], + ['list only', false, true, 1], +]) { + test(`${name} re-records only after a successful kill`, async () => { + const calls = []; + let reapCalls = 0; + const daemon = { pid: 4242, workspace: '/missing-ak-workspace', workspaceExists: false, ageSecs: 1 }; + const result = await run({ + flags: { kill, mcp: false, quiet: true }, + deps: { + daemonLifecycle: { + list: async opts => { calls.push(opts); return [daemon]; }, + reap: () => { reapCalls++; return [{ ...daemon, killed }]; }, + }, + mcpLifecycle: { list: async () => [], reap: () => { throw new Error('MCP reap forbidden'); } }, + }, + }); + assert.equal(result, 0); + assert.equal(reapCalls, kill ? 1 : 0); + assert.equal(calls.length, expected); + if (expected === 2) assert.deepEqual(calls[1], { refresh: true, record: true, source: 'daemon-gc' }); + assert.equal(fs.existsSync(processProbeMarker), false, 'real process discovery was reached'); + }); +} + +test('JSON listing observes MCP transports without reaping or re-recording', async () => { + const daemonCalls = []; + let mcpCalls = 0; + const oldLog = console.log; + console.log = () => {}; + try { + assert.equal(await run({ + flags: { json: true, kill: true, mcp: false }, + deps: { + daemonLifecycle: { list: async opts => { daemonCalls.push(opts); return []; }, reap: () => { throw new Error('reap forbidden'); } }, + mcpLifecycle: { list: async () => { mcpCalls++; return []; }, reap: () => { throw new Error('MCP reap forbidden'); } }, + }, + }), 0); + } finally { console.log = oldLog; } + assert.deepEqual(daemonCalls, [undefined]); + assert.equal(mcpCalls, 1); + assert.equal(fs.existsSync(processProbeMarker), false, 'real process discovery was reached'); +}); diff --git a/tests/kit/daemon-sweep-evidence.test.mjs b/tests/kit/daemon-sweep-evidence.test.mjs index 7f29d1ad..d55d0f7a 100644 --- a/tests/kit/daemon-sweep-evidence.test.mjs +++ b/tests/kit/daemon-sweep-evidence.test.mjs @@ -1,4 +1,4 @@ -// Branch 6a Task 7: listDaemons()'s processSweep() (src/lib/daemons.mjs) +// listDaemons()'s processSweep() (src/lib/daemons.mjs) // stops spawning `ps -eo pid=,args=` (or the Windows CIM query) on every // plain `ak status` call — it reuses fresh evidence instead (kind // 'daemon-sweep', id 'machine', 5-minute TTL: this is live process-table @@ -8,7 +8,7 @@ // collect() explicitly threads refresh:false from ctx. // // processSweep has no injectable runner, so — mirroring -// tests/kit/host-setup-evidence.test.mjs's Task 5 precedent — these tests +// tests/kit/host-setup-evidence.test.mjs's state-isolation precedent — these tests // break PATH so a real `ps` ENOENTs deterministically (run() never throws; // it degrades to an empty result), then seed cached evidence with a marker // entry no real sweep could ever produce. Getting the marker back proves the diff --git a/tests/kit/daemons-status.test.mjs b/tests/kit/daemons-status.test.mjs index 455c7253..fcbdc1dd 100644 --- a/tests/kit/daemons-status.test.mjs +++ b/tests/kit/daemons-status.test.mjs @@ -29,8 +29,8 @@ test('no project memory: none running stays ok', async (t) => { [{ subsystem: 'daemons', level: 'ok', message: 'none running', fix: null, repair: null }]); }); -// Task 7: collect() threads refresh/record/source into the listDaemons() -// call (gating processSweep's `ps` spawn), mirroring hosts.mjs's Task 5 +// collect() threads refresh/record/source into the listDaemons() +// call (gating processSweep's `ps` spawn), mirroring hosts.mjs's evidence-cache // precedent, so a plain `ak status` stays cache-first while `--refresh` // forces a fresh sweep. test('collect() threads refresh/record/source into listDaemons, defaulting to a plain-status-shaped call', async (t) => { @@ -143,7 +143,7 @@ test('no deferral row without a live daemon for this project', async (t) => { assert.equal(deferral(await collect(cwd, { now: NOW, platform: 'darwin' })), undefined); }); -// ── ak-managed daemon settings drift (Task 2.2) ──────────────────────────── +// ── ak-managed daemon settings drift ──────────────────────────── function rufloRepo(t, { autoStart } = {}) { const cwd = project(t, { autoStart }); fs.mkdirSync(path.join(cwd, '.git')); diff --git a/tests/kit/dashboard-get-is-read-only.test.mjs b/tests/kit/dashboard-get-is-read-only.test.mjs new file mode 100644 index 00000000..fa0f7769 --- /dev/null +++ b/tests/kit/dashboard-get-is-read-only.test.mjs @@ -0,0 +1,92 @@ +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import http from 'node:http'; +import { sandboxHome, sandboxProject, rmrf } from './helpers/home-sandbox.mjs'; + +const home = sandboxHome('ak-dashboard-read-only'); +const project = sandboxProject('ak-dashboard-read-only'); +after(() => rmrf(home, project)); +const { startDashboard } = await import('../../src/lib/dashboard-server.mjs'); + +const EXACT = [ + '/api/status', '/api/refresh', '/api/host-health', '/api/live', '/api/live/history', + '/api/live/events', '/api/live/intelligence', '/api/models', '/api/ruflo-components', + '/api/usage', '/api/hooks', '/api/limits', '/api/system', '/api/system/summary', + '/api/maintenance', '/api/sessions', +]; +const PARAMETERIZED = [ + '/api/maintenance/v2/inventory', '/api/hooks/source/bad', + '/api/live/playback/bad/bad', '/api/live/transcripts/bad/bad/events', '/api/session/bad', +]; +const ERROR = { error: 'start a refresh with POST /api/refresh' }; + +function request(server, route, method = 'GET') { + return new Promise((resolve, reject) => { + const req = http.request(new URL(route, server.url), { method, headers: { + 'x-dash-token': server.token, origin: server.url.replace(/\/$/, ''), + 'sec-fetch-site': 'same-origin', 'content-type': 'application/json', + } }, res => { + // SSE endpoints intentionally stay open. Their response headers are enough + // to prove that route dispatch has happened; close the reader promptly. + if (res.headers['content-type']?.includes('text/event-stream')) { + res.destroy(); resolve({ status: res.statusCode, body: '' }); return; + } + let body = ''; + res.on('data', chunk => { body += chunk; }); + res.on('end', () => resolve({ status: res.statusCode, body })); + }); + req.on('error', reject); + req.setTimeout(2000, () => req.destroy(new Error(`timed out: ${route}`))); + req.end(method === 'POST' ? '{}' : undefined); + }); +} + +test('GET route inventory stays aligned with the server dispatch table', () => { + const source = fs.readFileSync(new URL('../../src/lib/dashboard-server.mjs', import.meta.url), 'utf8'); + const exact = source.match(/const ROUTES = \{([\s\S]*?)\n {4}\};/)?.[1] ?? ''; + const parameterized = source.match(/const PARAM_ROUTES = \[([\s\S]*?)\n {4}\];/)?.[1] ?? ''; + assert.deepEqual([...exact.matchAll(/^ {6}'([^']+)':/gm)].map(match => match[1]), EXACT); + assert.deepEqual([...parameterized.matchAll(/^ {6}\[(\/.+?\/),/gm)].map(match => match[1]), [ + String.raw`/^\/api\/maintenance\/v2\/.*$/`, + String.raw`/^\/api\/hooks\/source\/([^/]+)$/`, + String.raw`/^\/api\/live\/playback\/([^/]+)\/([^/]+)$/`, + String.raw`/^\/api\/live\/transcripts\/([^/]+)\/([^/]+)\/events$/`, + String.raw`/^\/api\/session\/(.*)$/`, + ]); +}); + +test('all GET routes, including root, cannot start measurement or provider work through scan queries', async t => { + const calls = { system: 0, maintenance: 0, inventory: 0, rebuild: 0, provider: 0 }; + const hostReadiness = async () => ({ hosts: {} }); + hostReadiness.checkConnection = async () => { calls.provider++; return {}; }; + const server = await startDashboard({ port: 0, cwd: project, + fetchStatus: async () => ({ overall: 'ok', rows: [], drift: [] }), + hostReadiness, discoverProjects: () => [], + system: { read: async () => ({ runtime: {}, knownFiles: [], storage: {}, projects: [], scan: {} }), + refreshDeep: async () => { calls.system++; return { ok: true }; } }, + maintenance: { report: async () => ({}), scan: async () => { calls.maintenance++; return {}; }, + plan: async () => ({}) }, + management: { refreshInventory: async () => { calls.inventory++; }, + rebuildAfterMeasurement: async () => { calls.rebuild++; } }, + usage: { readIndex: async () => ({ sessions: [] }), readSession: async () => null, + masker: async () => value => value }, live: { snapshot: async () => ({}), replay: async () => ({ events: [] }), + subscribe: () => () => {} }, models: async () => ({ status: 'empty' }), + limits: async () => ({}), hooks: {}, transcripts: {}, + }); + t.after(() => server.close()); + for (const route of ['/', '/index.html', ...EXACT, ...PARAMETERIZED]) { + for (const suffix of ['?refresh=deep', '?refresh=scan&refresh=deep&trees=1', '?trees=1']) { + await request(server, route + suffix); + assert.deepEqual(calls, { system: 0, maintenance: 0, inventory: 0, rebuild: 0, provider: 0 }, route + suffix); + } + } + for (const route of ['/api/system', '/api/system/summary', '/api/maintenance']) { + for (const suffix of ['?refresh=deep', '?refresh=scan&refresh=deep&trees=1', '?trees=1']) { + const response = await request(server, route + suffix); + assert.equal(response.status, 400, route + suffix); + assert.deepEqual(JSON.parse(response.body), ERROR); + } + } + assert.equal((await request(server, '/api/host-health/local', 'POST')).status, 405); +}); diff --git a/tests/kit/dashboard-hermetic-defaults.test.mjs b/tests/kit/dashboard-hermetic-defaults.test.mjs index 20d8bd12..47b92d1d 100644 --- a/tests/kit/dashboard-hermetic-defaults.test.mjs +++ b/tests/kit/dashboard-hermetic-defaults.test.mjs @@ -10,7 +10,6 @@ import fs from 'node:fs'; import http from 'node:http'; import path from 'node:path'; import { sandboxHome, assertSandboxed, rmrf } from './helpers/home-sandbox.mjs'; -import { waitUntil } from './helpers/wait-until.mjs'; const home = sandboxHome('ak-dash-hermetic'); after(() => rmrf(home)); @@ -42,9 +41,7 @@ test('an injected System collector alone never builds the default maintenance se t.after(() => server.close()); const deep = await get(server, 'api/system?refresh=deep'); - assert.equal(deep.status, 200); - await waitUntil(() => errors.some((e) => /maintenanceOptions\.controlRoot/.test(e)), - 'the refused default maintenance service must be logged', { timeout: 5000 }); + assert.equal(deep.status, 400); const maintenance = await get(server, 'api/maintenance'); assert.equal(maintenance.status, 503); assert.equal(fs.existsSync(path.join(paths.maintenanceControlDir())), false, diff --git a/tests/kit/dashboard-project-identity.test.mjs b/tests/kit/dashboard-project-identity.test.mjs index f00d60da..be689ea1 100644 --- a/tests/kit/dashboard-project-identity.test.mjs +++ b/tests/kit/dashboard-project-identity.test.mjs @@ -80,9 +80,10 @@ test('should_canonicalize_symlink_aliases_without_merging_same_named_repositorie assert.notEqual(inspectProjectIdentity(first).repositoryId, inspectProjectIdentity(second).repositoryId); }); test('should_attribute_only_explicit_desktop_metadata_and_ignore_names_and_ambiguous_sources', () => { - for (const entrypoint of ['claude-desktop', 'claude-desktop-3p', 'remote_desktop']) { + for (const entrypoint of ['claude-desktop', 'claude-desktop-3p']) { assert.equal(transcriptSessionOrigin(lines({ entrypoint }), 'claude').origin, 'claude-desktop'); } + assert.equal(transcriptSessionOrigin(lines({ entrypoint: 'remote_desktop' }), 'claude').surface, 'cloud-session'); for (const originator of ['Codex Desktop', 'codex_work_desktop']) { assert.equal(transcriptSessionOrigin(lines({ type: 'session_meta', payload: { originator } }), 'codex').origin, 'codex-desktop'); } @@ -150,5 +151,5 @@ test('should_qualify_encoded_directory_recovery_as_a_sighting_instead_of_a_verif }), scanOpencode: () => ({ complete: true, sightings: [] }) }); assert.deepEqual({ sessions: result.projects[0].sessions, origin: result.projects[0].sessionOrigins[0].origin, countBasis: result.projects[0].sessionOrigins[0].countBasis }, - { sessions: 1, origin: 'unknown', countBasis: 'recovered-project-sighting' }); + { sessions: 0, origin: 'unknown', countBasis: 'recovered-project-sighting' }); }); diff --git a/tests/kit/dashboard-refresh-api.test.mjs b/tests/kit/dashboard-refresh-api.test.mjs new file mode 100644 index 00000000..3a45b2e9 --- /dev/null +++ b/tests/kit/dashboard-refresh-api.test.mjs @@ -0,0 +1,188 @@ +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import http from 'node:http'; +import fs from 'node:fs'; +import path from 'node:path'; +import { sandboxHome, sandboxProject, rmrf } from './helpers/home-sandbox.mjs'; + +const home = sandboxHome('ak-dashboard-refresh'); +const project = sandboxProject('ak-dashboard-refresh'); +const controlRoot = fs.mkdtempSync(path.join(home, 'control-')); +after(() => rmrf(home, project)); +const { startDashboard } = await import('../../src/lib/dashboard-server.mjs'); + +function request(server, method, route, body, headers = {}) { + const url = new URL(route, server.url); + return new Promise((resolve, reject) => { + const req = http.request(url, { method, headers: { + 'content-type': 'application/json', 'x-dash-token': server.token, + origin: url.origin, 'sec-fetch-site': 'same-origin', ...headers, + } }, res => { + let data = ''; res.on('data', chunk => { data += chunk; }); + res.on('end', () => { + let json; + try { json = JSON.parse(data); } catch { json = null; } + resolve({ status: res.statusCode, json, body: data }); + }); + }); + req.on('error', reject); + req.end(body === undefined ? undefined : JSON.stringify(body)); + }); +} + +function stages(overrides = {}) { + return Object.fromEntries(['machine', 'maintenance', 'inventory', 'live', 'local'].map(id => + [id, overrides[id] ?? (async () => ({ ok: true }))])); +} + +async function serverWith(stagesMap) { + return startDashboard({ port: 0, cwd: project, + fetchStatus: async () => ({ overall: 'ok', rows: [] }), + hostReadiness: async () => ({ hosts: {} }), + system: { read: async () => ({}), refreshDeep: async () => ({ ok: true }) }, + maintenance: { report: async () => ({}), scan: async () => ({}), plan: async () => ({}) }, + management: { refreshInventory: async () => ({ ok: true }) }, + maintenanceOptions: { controlRoot }, usage: {}, discoverProjects: () => [], + refreshStages: stagesMap, + }); +} + +async function finished(server) { + const deadline = Date.now() + 3000; + while (Date.now() < deadline) { + const latest = await request(server, 'GET', '/api/refresh'); + if (latest.json.running === false && latest.json.lastRun !== null) return latest.json; + await new Promise(resolve => setTimeout(resolve, 5)); + } + throw new Error('refresh did not finish'); +} + +test('refresh POST needs the header capability and same origin, and validates its bounded body', async t => { + const server = await serverWith(stages()); + t.after(() => server.close()); + assert.deepEqual((await request(server, 'GET', '/api/refresh')).json, { running: false, lastRun: null }); + const valid = { strength: 'local' }; + assert.equal((await request(server, 'POST', '/api/refresh', valid, { 'x-dash-token': '' })).status, 401); + assert.equal((await request(server, 'POST', `/api/refresh?token=${server.token}`, valid, { 'x-dash-token': '' })).status, 401); + assert.equal((await request(server, 'POST', '/api/refresh', valid, { origin: 'http://evil.test' })).status, 403); + for (const body of [{ strength: 'other' }, { strength: 'local', extra: true }, + { strength: 'local', projectTrees: true }, { strength: 'machine', projectTrees: 'yes' }, {}]) { + assert.equal((await request(server, 'POST', '/api/refresh', body)).status, 400, JSON.stringify(body)); + } +}); + +test('refresh POST starts ordered local work once and GET exposes progress and completion', async t => { + const calls = []; + const server = await serverWith(stages(Object.fromEntries(['maintenance', 'inventory', 'local'].map(id => + [id, async () => { calls.push(id); return { ok: true }; }])))); + t.after(() => server.close()); + const post = await request(server, 'POST', '/api/refresh', { strength: 'local' }); + assert.equal(post.status, 202); + assert.equal(post.json.started, true); + const state = await finished(server); + assert.equal(state.ok, true); + assert.equal(state.strength, 'local'); + assert.deepEqual(state.stages.map(({ id, state: stageState }) => [id, stageState]), + [['maintenance', 'done'], ['inventory', 'done'], ['local', 'done']]); + assert.deepEqual(calls, ['maintenance', 'inventory', 'local']); + assert.ok(state.startedAt && state.finishedAt); +}); + +test('successive refreshes expose distinct stable operation identities', async t => { + const server = await serverWith(stages()); + t.after(() => server.close()); + const firstPost = await request(server, 'POST', '/api/refresh', { strength: 'local' }); + const first = await finished(server); + const secondPost = await request(server, 'POST', '/api/refresh', { strength: 'local' }); + const second = await finished(server); + assert.equal(firstPost.status, 202); + assert.equal(secondPost.status, 202); + assert.match(first.operationId, /^[0-9a-f-]{36}$/); + assert.equal(firstPost.json.state.operationId, first.operationId); + assert.equal(secondPost.json.state.operationId, second.operationId); + assert.notEqual(first.operationId, second.operationId); +}); + +test('a second POST cannot start work while the operation is in flight', async t => { + let release; + const held = new Promise(resolve => { release = resolve; }); + const server = await serverWith(stages({ maintenance: async () => { await held; return { ok: true }; } })); + t.after(() => { release(); return server.close(); }); + assert.equal((await request(server, 'POST', '/api/refresh', { strength: 'local' })).status, 202); + const conflict = await request(server, 'POST', '/api/refresh', { strength: 'machine' }); + assert.equal(conflict.status, 409); + assert.equal(conflict.json.error, 'a refresh is already running'); + assert.equal(conflict.json.state.running, true); + release(); + await finished(server); +}); + +test('a failed machine measurement skips dependent stages and still performs local refresh', async t => { + const called = []; + const server = await serverWith(stages({ + machine: async () => ({ ok: false, detail: 'measurement failed' }), + maintenance: async () => { called.push('maintenance'); return { ok: true }; }, + inventory: async () => { called.push('inventory'); return { ok: true }; }, + local: async () => { called.push('local'); return { ok: true }; }, + })); + t.after(() => server.close()); + assert.equal((await request(server, 'POST', '/api/refresh', { strength: 'machine', projectTrees: true })).status, 202); + const state = await finished(server); + assert.deepEqual(state.stages.map(({ id, state: stageState }) => [id, stageState]), + [['machine', 'failed'], ['maintenance', 'skipped'], ['inventory', 'skipped'], ['local', 'done']]); + assert.deepEqual(called, ['local']); + assert.equal(state.ok, false); +}); + +test('dashboard local stage forces host readiness after collecting status', async () => { + const { dashboardRefreshStages } = await import('../../src/lib/dashboard/refresh-api.mjs'); + const calls = []; + const actual = dashboardRefreshStages({ cwd: project, pkgRoot: project, + getSystem: async () => ({ refreshDeep: async () => ({ ok: true }) }), + getMaintenance: async () => ({ scan: async () => ({}) }), + refreshInventoryAfterProviderScan: async () => ({}), + statusCollect: async () => { calls.push('status'); return { rows: [] }; }, + getHostReadiness: async options => { calls.push(options); return { hosts: {} }; }, + loadConfig: () => ({}), + }); + assert.equal((await actual.local()).ok, true); + assert.deepEqual(calls, ['status', { force: true }]); +}); + +test('an inventory rebuild that the server reports unavailable fails its stage', async () => { + const { dashboardRefreshStages } = await import('../../src/lib/dashboard/refresh-api.mjs'); + const actual = dashboardRefreshStages({ cwd: project, + getSystem: async () => ({}), getMaintenance: async () => ({}), + refreshInventoryAfterProviderScan: async () => null, + statusCollect: async () => ({}), getHostReadiness: async () => ({}), loadConfig: () => ({}), + }); + assert.equal((await actual.inventory({ strength: 'local' })).ok, false); +}); + +test('machine refresh honors each current project-tree choice through a shared persistent collector', async t => { + const { dashboardRefreshStages } = await import('../../src/lib/dashboard/refresh-api.mjs'); + const { createSystemCollector } = await import('../../src/lib/footprint/index.mjs'); + const measured = []; + const collector = createSystemCollector({ cwd: project, + snapshotFile: path.join(home, 'refresh-snapshot.json'), + runWorkerImpl: async ({ includeProjectTrees, startedAt }) => { + measured.push(includeProjectTrees); + return { ok: true, asOf: startedAt, sections: {}, completeness: { complete: true }, + persisted: { ok: true }, error: null, + terminal: { running: false, phase: 'done', finishedAt: startedAt, durationMs: 0, error: null } }; + }, + }); + const actual = dashboardRefreshStages({ cwd: project, getSystem: async () => collector, + getMaintenance: async () => ({}), refreshInventoryAfterProviderScan: async () => ({}), + statusCollect: async () => ({}), getHostReadiness: async () => ({}), loadConfig: () => ({}), + }); + const server = await serverWith(stages({ machine: actual.machine })); + t.after(() => server.close()); + // Separate HTTP callers (including another tab) share this server's collector. + // An omitted choice must also override a prior checked selection. + for (const projectTrees of [false, true, false, true, undefined]) { + assert.equal((await request(server, 'POST', '/api/refresh', { strength: 'machine', projectTrees })).status, 202); + assert.equal((await finished(server)).ok, true); + } + assert.deepEqual(measured, [false, true, false, true, false]); +}); diff --git a/tests/kit/dashboard-ruflo-components.test.mjs b/tests/kit/dashboard-ruflo-components.test.mjs index 0b9450f5..bb84fa9c 100644 --- a/tests/kit/dashboard-ruflo-components.test.mjs +++ b/tests/kit/dashboard-ruflo-components.test.mjs @@ -1,6 +1,6 @@ -// Task 10: dashboard panel + About chip for ruflo components (ADR-0058). +// dashboard panel + About chip for ruflo components (ADR-0058). // Payload logic (rufloComponentsPayload) is already covered by -// tests/kit/ruflo-components-snapshot.test.mjs (Task 9's shared projection — +// tests/kit/ruflo-components-snapshot.test.mjs (the shared projection — // controller ruling 1), so this file covers only the dashboard-specific // surface: the route registration and the client bundle renderer. import { test } from 'node:test'; diff --git a/tests/kit/dashboard-status-cost.test.mjs b/tests/kit/dashboard-status-cost.test.mjs index f731e30a..e902f40d 100644 --- a/tests/kit/dashboard-status-cost.test.mjs +++ b/tests/kit/dashboard-status-cost.test.mjs @@ -1,7 +1,7 @@ -// Branch 6a Task 9 / Task 8b — the Review Focus item this whole branch was -// scoped around: "Dashboard cost. The 30-second poll must not start more -// processes or transfer more data than before the branch." Task 7 made a -// warm-cache collect() spawn-free; this task made dashboard-server.mjs call +// The dashboard cost regression budget is the requirement this test is +// built around: "Dashboard cost. The 30-second poll must not start more +// processes or transfer more data than before the branch." Evidence caching made a +// warm-cache collect() spawn-free; dashboard-server.mjs calls // it in process instead of shelling out. This file is the end-to-end proof // that swap actually delivered the promised cost property, simulating two // 30s poll ticks against a running server (not a bare collect() call): @@ -16,13 +16,28 @@ import assert from 'node:assert/strict'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; +import http from 'node:http'; import { spawnEnv, sandboxProject, writeKitConfig, offlineKitConfig } from './helpers/home-sandbox.mjs'; import { startGuardedDashboard, stopGuardedDashboard, getJson, markLedgerBoundary, readLedger, sliceByCallBoundary, isVersionDriftLookup, } from './helpers/dashboard-child-server.mjs'; -test('two 30s-poll-tick /api/status requests: the second starts no processes and transfers no more data than the first', async () => { +function postRefresh(port, token) { + return new Promise((resolve, reject) => { + const req = http.request({ host: '127.0.0.1', port, path: '/api/refresh', method: 'POST', headers: { + 'x-dash-token': token, 'content-type': 'application/json', + origin: `http://127.0.0.1:${port}`, 'sec-fetch-site': 'same-origin', + } }, res => { + let body = ''; res.on('data', chunk => { body += chunk; }); + res.on('end', () => resolve({ status: res.statusCode, body })); + }); + req.on('error', reject); + req.end(JSON.stringify({ strength: 'local' })); + }); +} + +test('two 30s-poll-tick /api/status requests: the second starts no processes and transfers no more data than the first', async t => { const ledgerFile = path.join(os.tmpdir(), `ak-dash-cost-${process.pid}.ndjson`); const home = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-dash-cost-home-')); fs.mkdirSync(path.join(home, '.config'), { recursive: true }); @@ -41,7 +56,7 @@ test('two 30s-poll-tick /api/status requests: the second starts no processes and const { port, token } = started; // Tick 1: a cold-cache poll. Every real dashboard's FIRST poll after - // startup pays this cost — Ruling A (Task 7) says a cold cache probes at + // startup pays this cost — a cold cache probes at // least once per gated kind, so this is not itself under budget. const first = await getJson(port, '/api/status', token); assert.equal(first.status, 200); @@ -69,6 +84,29 @@ test('two 30s-poll-tick /api/status requests: the second starts no processes and assert.ok(second.bytes <= first.bytes + budget, `second /api/status response (${second.bytes} bytes) exceeds the first (${first.bytes} bytes) ` + `by more than the ${budget}-byte budget — possible in-process cache growth/leak across polls`); + + const startedAt = Date.now(); + const refreshStart = await postRefresh(port, token); + assert.equal(refreshStart.status, 202, refreshStart.body); + let refresh; + do { + refresh = await getJson(port, '/api/refresh', token); + if (refresh.json.running) await new Promise(resolve => setTimeout(resolve, 50)); + } while (refresh.json.running && Date.now() - startedAt < 60_000); + assert.equal(refresh.json.running, false, 'local refresh must finish within 60 s in the sandbox'); + assert.deepEqual(refresh.json.stages.map(({ id, state }) => [id, state]), + [['maintenance', 'done'], ['inventory', 'done'], ['local', 'done']]); + markLedgerBoundary(ledgerFile, 'refresh'); + const third = await getJson(port, '/api/status', token); + assert.equal(third.status, 200); + markLedgerBoundary(ledgerFile, 'third'); + const { third: thirdSlice } = sliceByCallBoundary(readLedger(ledgerFile), ['first', 'second', 'refresh', 'third']); + const thirdUnexplained = thirdSlice.filter((l) => !isVersionDriftLookup(l)); + assert.equal(thirdUnexplained.length, 0, JSON.stringify(thirdUnexplained.map((l) => [l.cmd, l.args]))); + assert.ok(third.bytes <= second.bytes + budget, `third /api/status: ${third.bytes} bytes; second: ${second.bytes}`); + assert.deepEqual(Object.keys(third.json).sort(), Object.keys(second.json).sort()); + t.diagnostic(`status bytes cold/warm/post-refresh=${first.bytes}/${second.bytes}/${third.bytes}; ` + + `local refresh=${Date.now() - startedAt}ms; third unexplained spawns=${thirdUnexplained.length}`); } finally { await stopGuardedDashboard(child); fs.rmSync(ledgerFile, { force: true }); @@ -76,3 +114,8 @@ test('two 30s-poll-tick /api/status requests: the second starts no processes and fs.rmSync(project, { recursive: true, force: true }); } }); + +test('the idle polling client never requests the refresh route', () => { + const poll = fs.readFileSync(new URL('../../src/lib/dashboard/client/poll.mjs', import.meta.url), 'utf8'); + assert.doesNotMatch(poll, /\/api\/refresh/); +}); diff --git a/tests/kit/dashboard-status-inprocess.test.mjs b/tests/kit/dashboard-status-inprocess.test.mjs index 5dcf8e9d..ae36458e 100644 --- a/tests/kit/dashboard-status-inprocess.test.mjs +++ b/tests/kit/dashboard-status-inprocess.test.mjs @@ -1,10 +1,10 @@ -// Branch 6a Task 9: dashboard-server.mjs's default /api/status provider now +// dashboard-server.mjs's default /api/status provider now // calls status.mjs's own collect() in process instead of shelling out to // `node bin/agentic-kit.mjs status --json`. This file proves the swap kept // both properties that mattered about the old subprocess boundary: // // 1. A warm-cache request spawns nothing (tests/fixtures/status-zero-spawn-child.mjs -// already proved this for a bare collect() call — Task 8a/Task 7; this +// already proved this for a bare collect() call; this harness // proves it for a REQUEST ARRIVING AT A RUNNING SERVER, the actual shape // the dashboard's 30s poll exercises). // 2. The JSON /api/status now COMPUTES has the same {overall, rows} content @@ -14,12 +14,12 @@ // return, which the in-process path would otherwise silently drop. import { test } from 'node:test'; import assert from 'node:assert/strict'; -import { spawnSync } from 'node:child_process'; +import { spawn, spawnSync } from 'node:child_process'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; -import { spawnEnv, sandboxProject, writeKitConfig, offlineKitConfig } from './helpers/home-sandbox.mjs'; +import { spawnEnv, sandboxProject, writeKitConfig, offlineKitConfig, fakeGlobalRoot } from './helpers/home-sandbox.mjs'; import { startGuardedDashboard, stopGuardedDashboard, getJson, markLedgerBoundary, readLedger, sliceByCallBoundary, isVersionDriftLookup, @@ -139,3 +139,47 @@ test('GET /api/status (in-process) carries the same {overall, rows} `ak status - cleanup(home, project); } }); + +test('GET /api/status passes the server cwd through ruflo-components project-root discovery', async () => { + const { home, project, env } = sandbox('ak-dash-ruflo-cwd'); + const decoy = sandboxProject('ak-dash-ruflo-decoy'); + const fakeRoot = fakeGlobalRoot(home, { ruflo: '9.9.9' }); + fs.mkdirSync(path.join(project, '.claude-flow')); + fs.mkdirSync(path.join(project, '.harness')); + fs.writeFileSync(path.join(project, '.harness', 'mcp-policy.json'), '{invalid'); + fs.mkdirSync(path.join(decoy, '.claude-flow')); + writeKitConfig(home, offlineKitConfig({ rufloComponents: { mcpGovernance: { maxCallsPerMinute: 60 } } })); + + let child; + try { + const ready = await new Promise((resolve, reject) => { + child = spawn(process.execPath, [path.join(PKG_ROOT, 'tests/fixtures/dashboard-status-child.mjs'), project, fakeRoot], { + cwd: decoy, env, stdio: ['ignore', 'pipe', 'pipe'], + }); + let out = '', err = ''; + child.stdout.on('data', chunk => { + out += chunk; + const match = out.match(/READY (\d+) (\S+)\n/); + if (match) resolve({ port: Number(match[1]), token: match[2] }); + }); + child.stderr.on('data', chunk => { err += chunk; }); + child.once('error', reject); + child.once('exit', code => reject(new Error(`dashboard child exited ${code}: ${err || out}`))); + }); + const response = await getJson(ready.port, '/api/status', ready.token); + assert.equal(response.status, 200); + assert.ok(response.json.rows.some(row => row.subsystem === 'ruflo-components' + && /ruflo components:.*ruflo 9\.9\.9/.test(row.message)), + 'the status request must use the disposable fake Ruflo package'); + const governance = response.json.rows.find(row => row.subsystem === 'ruflo-components' + && /MCP tool governance/.test(row.message)); + assert.ok(governance, 'fake installed Ruflo must reach its component projection'); + assert.equal(governance.state, 'blocked'); + assert.match(governance.message, /policy file is invalid/, + 'the server-supplied project cwd, not process.cwd(), must drive rufloProjectRoot'); + } finally { + await stopGuardedDashboard(child); + cleanup(home, project); + fs.rmSync(decoy, { recursive: true, force: true }); + } +}); diff --git a/tests/kit/dashboard-usage-telemetry.test.mjs b/tests/kit/dashboard-usage-telemetry.test.mjs index b1b147cb..b69a907e 100644 --- a/tests/kit/dashboard-usage-telemetry.test.mjs +++ b/tests/kit/dashboard-usage-telemetry.test.mjs @@ -159,8 +159,8 @@ function spyReadIndex(records, calls) { test('/api/usage payload carries rhythm, mode, provider and a previous-window projection', async () => { const calls = []; - // Mirrors usage-index.mjs's corrected buildIndex/scan contract (Task 7 - // ruling B): `aggregate` always sees the DISPLAY cutoff, and `previous` + // Mirrors usage-index.mjs's corrected buildIndex/scan contract: + // `aggregate` always sees the DISPLAY cutoff, and `previous` // rides through from the caller's own options unchanged. const usage = { readIndex: spyReadIndex(FIXTURE_RECORDS, calls) }; const srv = await startDashboard({ @@ -799,11 +799,11 @@ test('the whole-history chip ships hidden, for the one view that needs it', () = assert.match(JS, /usage-days-all/, 'and the client actually toggles it'); }); -// ── usage-rhythm.mjs: rhythm/mode chart primitives (Task 8) ──────────────── +// ── usage-rhythm.mjs: rhythm/mode chart primitives ──────────────── // // These are pure string builders, imported directly here — real ESM on disk, // not read through the concatenated client.mjs bundle (that concatenation is -// exercised separately, above, via the `JS` import). Task 9 wires these +// exercised separately, above, via the `JS` import). The panels wire these // exports into usage.mjs's panels. test('deltaChip renders an up arrow and a rounded percent for a positive change', () => { @@ -851,7 +851,7 @@ test('histogram renders a marker and its label, with the bars emitted after the // geometry test below for the CSS half of this: .hist-bars must ALSO be // positioned for "markers first" to actually put bars on top). This only // asserts the half a string test can honestly observe: markup order. - // Task 9's screenshot verification is what confirms the rendered result. + // Screenshot verification is what confirms the rendered result. const html = histogram({ counts: [1, 5, 2], labels: ['<1s', '1-3s', '3s+'], markers: [{ atPct: 50, label: 'p50' }] }); assert.match(html, /p50/); assert.match(html, /hist-marker/); @@ -917,7 +917,7 @@ test('usage styles append the rhythm/mode chart primitive classes with the mark- assert.match(CSS, /\.hist-bars\{[^}]*position:relative/, 'hist-bars must be positioned to paint over the positioned marker overlay'); }); -// ── Task 9: the panels that consume all of the above ────────────────────── +// ── the panels that consume all of the above ────────────────────── test('usage-rhythm.mjs reaches the served bundle, with exactly one escaper in it', () => { // Without this splice every panel below throws ReferenceError in the browser diff --git a/tests/kit/deja-vu-lifecycle.test.mjs b/tests/kit/deja-vu-lifecycle.test.mjs index 33ddc330..76e95d70 100644 --- a/tests/kit/deja-vu-lifecycle.test.mjs +++ b/tests/kit/deja-vu-lifecycle.test.mjs @@ -10,10 +10,10 @@ import { runLifecycle } from '../../src/lib/adapters/lifecycle.mjs'; import { evidenceDir, writeEvidence, stableInputsKey } from '../../src/lib/evidence.mjs'; import { tempDir } from './helpers/temp-dir.mjs'; -// Task 5 Part 2: detect() now reads/writes evidence under the kit state dir +// detect() now reads/writes evidence under the kit state dir // (evidenceDir()) when refresh:false. Redirect the state base for this whole // file so those reads/writes never touch this machine's real evidence store — -// mirrors tests/kit/natives-runtime.test.mjs's Task 4 redirect. +// mirrors tests/kit/natives-runtime.test.mjs's state redirect. process.env.XDG_STATE_HOME = tempDir('ak-deja-vu-lifecycle-state'); process.env.LOCALAPPDATA = process.env.XDG_STATE_HOME; const resetEvidence = () => fs.rmSync(evidenceDir(), { recursive: true, force: true }); @@ -406,7 +406,7 @@ test('undo removes verified target before exact owned npm package and never touc ]); }); -// ── detect() refresh caching (Branch 6a Task 5 Part 2) ────────────────────── +// ── detect() refresh caching ────────────────────── // Ruling A: 6h max age. Ruling B: `refresh` defaults to true, so every caller // other than status's plain-status path (collectDejaVuRows) keeps probing // unconditionally — including detect()'s own internal reuse inside plan() diff --git a/tests/kit/deja-vu-teardown-verify.test.mjs b/tests/kit/deja-vu-teardown-verify.test.mjs index 8f5041cf..c36e0ca2 100644 --- a/tests/kit/deja-vu-teardown-verify.test.mjs +++ b/tests/kit/deja-vu-teardown-verify.test.mjs @@ -80,7 +80,7 @@ test('deja-vu verify cleanly skips disabled, unowned integration without probing const { result, out } = await captureLog(() => verify.verifyDejaVu({ cfg: cfg({ enabled: false }), adapter, })); - assert.equal(result, true); + assert.deepEqual(result, { status: 'skipped', reason: 'disabled and unowned' }); assert.deepEqual(calls, []); assert.match(out, /disabled and unowned — skipped/); }); diff --git a/tests/kit/doc-citations.test.mjs b/tests/kit/doc-citations.test.mjs index 4581f57f..a479f3bf 100644 --- a/tests/kit/doc-citations.test.mjs +++ b/tests/kit/doc-citations.test.mjs @@ -22,7 +22,7 @@ // nearby falls back to a plain range check (the file has enough lines) — // there is nothing sharper to hold it to. // -// Fix round 1, C-1 — the DEFINITION-SITE rule. A named symbol's word-boundary +// C-1 — the DEFINITION-SITE rule. A named symbol's word-boundary // match above only proves the TEXT appears somewhere in the widened window — // a call site (`printScoreReliability(agg);`) contains the identifier just as // much as its definition does, so a citation moved off a function's real @@ -34,14 +34,14 @@ // multiple such declarations is too ambiguous to gate on and relies on the // plain word-boundary check above instead. // -// Fix round 1, I-2 — SAME-ROW fallback for table citations. Cell-scoping +// I-2 — SAME-ROW fallback for table citations. Cell-scoping // (below) stops a neighbouring column's identifier from anchoring a // DIFFERENT fact's citation, but a `| `symbol` | prose (`file.mjs:N`) |` row // legitimately names its subject in one cell and cites it in another. When a // table citation's OWN cell yields no anchor at all, it falls back to the // REST OF ITS ROW (every other cell on the same line) — never another row. // -// Fix round 2 — the CALL-SITE marker (an affordance, not an escape hatch). +// the CALL-SITE marker (an affordance, not an escape hatch). // Round 1 fixed genuine call-site citations by DE-ANCHORING them — stripping // or requoting the identifier so the definition-site rule had nothing to gate // on. That made the citation's own NUMBER invisible to the gate again, and @@ -97,7 +97,7 @@ function fileIndex() { const IDENTIFIER_RE = /^[A-Za-z_$][\w$]{3,}$/; const IDENTIFIER_PREFIX_RE = /^([A-Za-z_$][\w$]{3,})\s*(?:=|:(?!:)|\()/; -// Fix round 2 — the literal word "call" (never "calls"/"called"/"calling", +// the literal word "call" (never "calls"/"called"/"calling", // which \b already excludes), read from the citation's own sentence/cell as // a deliberate self-declaration that this citation is about a call site. const CALL_SITE_MARKER_RE = /\bcall\b/i; @@ -111,7 +111,7 @@ function identifierIn(tok) { return m ? m[1] : null; } -// Fix round 1, C-1 — the definition-site rule. `name` is always a validated +// C-1 — the definition-site rule. `name` is always a validated // identifier token (from identifierIn), never doc free text, so it is safe // to interpolate into a RegExp without escaping. function definitionRe(name) { @@ -194,7 +194,7 @@ function extractCitations(docText) { spans.push({ tok: m[1].replace(/\s+/g, ' '), line: startLine, col: m.index - lineStartOffset[startLine - 1] }); } for (const c of cites) { - // Fix round 1, I-2: mine the citation's OWN cell first; a table citation + // I-2: mine the citation's OWN cell first; a table citation // whose own cell names nothing falls back to the REST OF ITS ROW (never // another row) — see the file header. Non-table citations are unaffected // (ownCellOnly never restricts anything for them). @@ -234,7 +234,7 @@ function extractCitations(docText) { } c.idAnchors = [...idAnchors]; c.strAnchors = [...strAnchors]; - // Fix round 2 — the call-site marker, read from the SAME own-cell/prose + // the call-site marker, read from the SAME own-cell/prose // context as the quoted-string anchors above (not the row-fallback: a // marker is a deliberate per-citation declaration, never inherited from // a sibling cell). @@ -286,7 +286,7 @@ function checkDoc(docRel, index) { ); continue; } - // Fix round 1, C-1 — the definition-site rule: a word-boundary hit above + // C-1 — the definition-site rule: a word-boundary hit above // only proves the identifier's TEXT is somewhere in the window, which a // call site satisfies just as well as a definition. Scoped to the // anchors that ACTUALLY hit (idHitAnchors) — an identifier merely mined @@ -296,7 +296,7 @@ function checkDoc(docRel, index) { // only a symbol whose text-match is doing the work here has to prove // that text-match is a definition, not a call site. // - // Fix round 2 — a citation self-declared as a call site (see the + // a citation self-declared as a call site (see the // CALL_SITE_MARKER_RE header note) skips ONLY this rule; idHit/strHit // above still had to pass, so the citation is not exempt from having a // real anchor, only from that anchor having to be a declaration. diff --git a/tests/kit/exec.test.mjs b/tests/kit/exec.test.mjs index 0526bcda..770ff176 100644 --- a/tests/kit/exec.test.mjs +++ b/tests/kit/exec.test.mjs @@ -3,7 +3,259 @@ import assert from 'node:assert/strict'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; -import { run, have, resolveShim } from '../../src/lib/exec.mjs'; +import { run, have, resolveShim, withAbortSignal } from '../../src/lib/exec.mjs'; + +const isAlive = (pid) => { + try { process.kill(pid, 0); return true; } catch { return false; } +}; +async function waitUntil(predicate) { + for (let n = 0; n < 60; n += 1) { + if (predicate()) return true; + await new Promise((resolve) => setTimeout(resolve, 50)); + } + return predicate(); +} + +for (const [name, options] of [ + ['no input with explicit signal', { input: undefined, inherited: false }], + ['input with explicit signal', { input: '', inherited: false }], + ['no input with inherited signal', { input: undefined, inherited: true }], + ['input with inherited signal', { input: '', inherited: true }], +]) { + test(`run() abort reaps owned grandchild: ${name}`, async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-abort-tree-')); + const pidFile = path.join(dir, 'pids.json'); + const controller = new AbortController(); + const code = `const {spawn}=require('node:child_process'); + const gc=spawn(process.execPath,['-e','setInterval(()=>{},1000)'],{stdio:'ignore'}); + require('node:fs').writeFileSync(${JSON.stringify(pidFile)},JSON.stringify([process.pid,gc.pid])); + setInterval(()=>{},1000);`; + let pids = []; + let pending; + try { + const launch = () => run(process.execPath, ['-e', code], { + input: options.input, timeout: 5_000, + ...(options.inherited ? {} : { signal: controller.signal }), + }); + pending = options.inherited ? withAbortSignal(controller.signal, launch) : launch(); + assert.equal(await waitUntil(() => fs.existsSync(pidFile)), true, 'owned children started'); + pids = JSON.parse(fs.readFileSync(pidFile, 'utf8')); + assert.equal(isAlive(pids[1]), true, 'grandchild alive before abort'); + controller.abort(); + const result = await pending; + assert.notEqual(result.code, 0); + assert.equal(await waitUntil(() => !isAlive(pids[1])), true, + 'abort must terminate the grandchild, not only its parent'); + } finally { + controller.abort(); + for (const pid of pids) { + if (isAlive(pid)) { + try { process.kill(pid, 'SIGKILL'); } catch { /* already exited */ } + } + } + await pending; + assert.equal(await waitUntil(() => pids.every((pid) => !isAlive(pid))), true, 'fixture children exited'); + fs.rmSync(dir, { recursive: true, force: true }); + } + }); +} + +test('run() does not spawn for a pre-aborted signal, including inherited abort', async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-preabort-')); + const marker = path.join(dir, 'spawned'); + const aborted = new AbortController(); + aborted.abort(); + const args = ['-e', `require('node:fs').writeFileSync(${JSON.stringify(marker)},'yes')`]; + try { + for (const input of [undefined, '']) { + const explicit = await run(process.execPath, args, { signal: aborted.signal, input }); + const inherited = await withAbortSignal(aborted.signal, + () => run(process.execPath, args, { input })); + assert.notEqual(explicit.code, 0); + assert.notEqual(inherited.code, 0); + assert.equal(fs.existsSync(marker), false); + } + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } +}); + +test('an explicit live signal takes precedence over an aborted inherited signal', async () => { + const inherited = new AbortController(); + inherited.abort(); + const explicit = new AbortController(); + for (const input of [undefined, '']) { + const result = await withAbortSignal(inherited.signal, () => run( + process.execPath, ['-e', 'process.stdout.write("ok")'], + { signal: explicit.signal, input }, + )); + assert.equal(result.code, 0, result.stderr); + assert.equal(result.stdout, 'ok'); + } +}); + +for (const stop of ['abort', 'timeout']) { + test(`Windows ${stop} after direct-child exit reports incomplete tree cleanup`, async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-exited-root-')); + const pidFile = path.join(dir, 'pids.json'); + const exitFile = path.join(dir, 'parent-exit'); + const readyFile = path.join(dir, 'descendant-ready'); + const controller = new AbortController(); + // libuv's Windows Job Object kills non-detached children when their + // parent exits. unref() alone only releases the event-loop reference. + // Deliberately escape that job while retaining the real output handles. + const descendant = `require('node:fs').writeFileSync(${JSON.stringify(readyFile)},'ready'); + setTimeout(()=>{},10000);`; + const code = `const {spawn}=require('node:child_process'); + const fs=require('node:fs'); + const gc=spawn(process.execPath,['-e',${JSON.stringify(descendant)}],{ + detached:process.platform==='win32',stdio:['ignore','inherit','inherit']}); + gc.unref(); + fs.writeFileSync(${JSON.stringify(pidFile)},JSON.stringify([process.pid,gc.pid])); + const ready=setInterval(()=>{ + if(fs.existsSync(${JSON.stringify(readyFile)})) clearInterval(ready); + },10); + setTimeout(()=>process.exit(2),4000).unref(); + process.on('exit',()=>fs.writeFileSync(${JSON.stringify(exitFile)},'yes'));`; + let pids = []; + let pending; + try { + const started = Date.now(); + pending = run(process.execPath, ['-e', code], { + windows: true, signal: controller.signal, timeout: stop === 'timeout' ? 1_200 : 5_000, + }); + assert.equal(await waitUntil(() => fs.existsSync(pidFile)), true); + pids = JSON.parse(fs.readFileSync(pidFile, 'utf8')); + assert.equal(await waitUntil(() => fs.existsSync(readyFile)), true, 'descendant initialized'); + assert.equal(await waitUntil(() => fs.existsSync(exitFile)), true, 'direct child exited'); + await new Promise((resolve) => setTimeout(resolve, 100)); + assert.equal(isAlive(pids[1]), true, 'descendant still owns the output pipe'); + const stoppedAt = Date.now(); + if (stop === 'abort') controller.abort(); + const result = await pending; + assert.notEqual(result.code, 0, 'an incomplete run cannot report success'); + assert.match(result.stderr, /incomplete.*tree cleanup/i); + assert.ok(Date.now() - (stop === 'abort' ? stoppedAt : started) + < (stop === 'abort' ? 1_400 : 2_600), 'return is bounded by abort or timeout'); + } finally { + controller.abort(); + for (const pid of pids) { + if (isAlive(pid)) { + try { process.kill(pid, 'SIGKILL'); } catch { /* already exited */ } + } + } + await pending; + assert.equal(await waitUntil(() => pids.every((pid) => !isAlive(pid))), true, 'fixture children exited'); + fs.rmSync(dir, { recursive: true, force: true }); + } + }); +} + +test('Windows taskkill stall has a bounded cleanup wait and reports uncertainty', { + skip: process.platform === 'win32', +}, async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-taskkill-stall-')); + const killer = path.join(dir, 'taskkill.exe'); + const oldPath = process.env.PATH; + fs.writeFileSync(killer, `#!${process.execPath}\nsetTimeout(() => process.exit(1), 1500);\n`, { mode: 0o755 }); + process.env.PATH = `${dir}${path.delimiter}${oldPath}`; + try { + const started = Date.now(); + const result = await run(process.execPath, ['-e', 'setInterval(()=>{},1000)'], { + windows: true, timeout: 100, + }); + assert.notEqual(result.code, 0); + assert.match(result.stderr, /incomplete.*tree cleanup/i); + assert.ok(Date.now() - started < 1400, 'does not wait for the stalled taskkill process'); + } finally { + process.env.PATH = oldPath; + fs.rmSync(dir, { recursive: true, force: true }); + } +}); + +test('run() enforces maxBuffer in UTF-8 bytes', async () => { + const result = await run(process.execPath, ['-e', 'process.stdout.write("ééé")'], { + maxBuffer: 4, + }); + assert.notEqual(result.code, 0); + assert.match(result.stderr, /maxBuffer/i); +}); + +function recordShimChildren(reported, observedShimPid, cleanupPids) { + // The reported parent is assertion evidence, never kill authority. + cleanupPids.push(reported[0], reported[1], observedShimPid); + assert.equal(reported[2], observedShimPid, 'Node is a child of the observed PowerShell shim'); +} + +test('a mismatched reported parent never becomes a fixture cleanup kill target', () => { + const cleanupPids = []; + const killTargets = []; + const unexpectedParent = 990003; + // Synthetic PIDs and an injected recording function: no real process signal. + const kill = (pid) => { killTargets.push(pid); }; + try { + assert.throws(() => recordShimChildren([990001, 990002, unexpectedParent], 990004, cleanupPids), + /Node is a child of the observed PowerShell shim/); + } finally { + for (const pid of cleanupPids) kill(pid); + } + assert.equal(killTargets.includes(unexpectedParent), false, 'unowned reported parent must never be signalled'); + assert.deepEqual(killTargets, [990001, 990002, 990004]); +}); + +test('Windows abort reaps the Node child behind a PowerShell shim and its grandchild', { + skip: process.platform !== 'win32', +}, async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-abort-shim-')); + const pidFile = path.join(dir, 'pids.json'); + const shimEntry = path.join(dir, 'powershell-pid'); + const controller = new AbortController(); + const quotedNode = process.execPath.replaceAll("'", "''"); + const pids = []; + let pending; + let outcome; + try { + fs.writeFileSync(path.join(dir, 'codex.cmd'), '@echo off\r\n'); + const code = `const {spawn}=require('node:child_process'); + const gc=spawn(process.execPath,['-e','setInterval(()=>{},1000)'],{stdio:'ignore'}); + require('node:fs').writeFileSync(${JSON.stringify(pidFile)},JSON.stringify([process.pid,gc.pid,process.ppid])); + setInterval(()=>{},1000);`; + // npm shims forward a script filename; PowerShell 5.1 reserializes + // native arguments, so multiline node -e source is not that interface. + const script = path.join(dir, 'fixture.cjs'); + fs.writeFileSync(script, code); + fs.writeFileSync(path.join(dir, 'codex.ps1'), + `[System.IO.File]::WriteAllText('${shimEntry.replaceAll("'", "''")}',[string]$PID)\n` + + `& '${quotedNode}' '${script.replaceAll("'", "''")}' $args\nexit $LASTEXITCODE\n`); + pending = run('codex', [], { + // PowerShell checks PATHEXT even for the absolute Node.exe path. + // Excluding .EXE changes native execution into document activation. + env: { PATH: dir, PATHEXT: '.COM;.EXE;.BAT;.CMD' }, signal: controller.signal, timeout: 10_000, + }); + pending.then((result) => { outcome = result; }); + assert.equal(await waitUntil(() => fs.existsSync(pidFile) || outcome), true, 'PowerShell launch settled or ready'); + assert.equal(fs.existsSync(shimEntry), true, `PowerShell entered owned shim: ${JSON.stringify(outcome)}`); + assert.equal(fs.existsSync(pidFile), true, `PowerShell launched Node: ${JSON.stringify(outcome)}`); + const reported = JSON.parse(fs.readFileSync(pidFile, 'utf8')); + recordShimChildren(reported, Number(fs.readFileSync(shimEntry, 'utf8')), pids); + assert.equal(isAlive(pids[1]), true); + controller.abort(); + const result = await pending; + assert.notEqual(result.code, 0); + assert.equal(await waitUntil(() => pids.every((pid) => !isAlive(pid))), true, + 'taskkill must remove the Node process and its grandchild'); + } finally { + controller.abort(); + for (const pid of pids) { + if (isAlive(pid)) { + try { process.kill(pid, 'SIGKILL'); } catch { /* already exited */ } + } + } + await pending; + assert.equal(await waitUntil(() => pids.every((pid) => !isAlive(pid))), true, 'fixture children exited'); + fs.rmSync(dir, { recursive: true, force: true }); + } +}); // code-quality Finding 2: exec.mjs used to set shell:true for a fixed set of // Windows .cmd shims (npm/npx/claude/ruflo/aqe/claude-flow), which handed diff --git a/tests/kit/footprint-projects.test.mjs b/tests/kit/footprint-projects.test.mjs index d5466b97..3e53b6d0 100644 --- a/tests/kit/footprint-projects.test.mjs +++ b/tests/kit/footprint-projects.test.mjs @@ -55,7 +55,7 @@ function write(file, content) { /** One Claude transcript: flat `cwd` on its own records. */ const claudeTranscript = (root, dirName, file, cwd) => write( path.join(root, dirName, file), - `${JSON.stringify({ type: 'user', cwd })}\n${JSON.stringify({ type: 'assistant' })}\n`, + `${JSON.stringify({ type: 'user', cwd, sessionId: `${dirName}/${file}` })}\n${JSON.stringify({ type: 'assistant' })}\n`, ); /** One Codex rollout: `payload.cwd` on the record that opens it. */ @@ -325,7 +325,7 @@ test('OpenCode sessions come from the store, and a broken store degrades with it withDb: () => ({ ok: false, error: { kind: 'io', message: 'SQLITE_CORRUPT' } }), }); assert.equal(broken.status, 'degraded'); - assert.equal(broken.reason, 'SQLITE_CORRUPT'); + assert.equal(broken.reason, 'io', 'health exposes the error category, not raw database error text'); assert.equal(broken.complete, false); }); diff --git a/tests/kit/footprint-windows.test.mjs b/tests/kit/footprint-windows.test.mjs index 7d145449..d714ae22 100644 --- a/tests/kit/footprint-windows.test.mjs +++ b/tests/kit/footprint-windows.test.mjs @@ -460,6 +460,34 @@ test('runtime census names the process source instead of treating every cwd as a assert.match(rows[3].source.reason, /reported no working directory/); }); +test('runtime census exposes application identity separately from coding-agent host', async () => { + const census = await collectRuntimeCensus({ + platform: 'darwin', + surveyImpl: async () => ({ processes: [ + { pid: 1, host: null, application: 'Claude Desktop', controllerKind: 'desktop-app', + startedAt: new Date(WIN_NOW - 1000).toISOString(), uptimeMs: 1000, + cpuPercent: 0, rssBytes: 100, cwd: '/', cwdReason: null }, + { pid: 2, host: null, application: 'ChatGPT desktop app', controllerKind: 'desktop-app', + startedAt: new Date(WIN_NOW - 1000).toISOString(), uptimeMs: 1000, + cpuPercent: 0, rssBytes: 100, cwd: '/', cwdReason: null }, + { pid: 3, host: 'codex', application: null, controllerKind: 'project-session', + startedAt: new Date(WIN_NOW - 1000).toISOString(), uptimeMs: 1000, + cpuPercent: 0, rssBytes: 100, cwd: '/repos/work', cwdReason: null }, + ] }), + listDaemonsImpl: async () => [], + osImpl: { totalmem: () => 1000, freemem: () => 500, cpus: () => [1] }, + now: WIN_NOW, + classifyContext: () => ({ kind: 'repository', label: 'work', path: '/repos/work', projectKey: 'work' }), + }); + assert.deepEqual(census.processes.value.map(({ host, application, source }) => + ({ host, application, sourceKind: source.value.kind, sourceLabel: source.value.label })), [ + { host: null, application: 'Claude Desktop', sourceKind: 'desktop-app', sourceLabel: 'Claude Desktop' }, + { host: null, application: 'ChatGPT desktop app', sourceKind: 'desktop-app', sourceLabel: 'ChatGPT desktop app' }, + { host: 'codex', application: null, sourceKind: 'repository', sourceLabel: 'work' }, + ]); + assert.equal(census.ephemeral, true); +}); + test('a Windows survey that cannot run at all leaves the machine facts standing', async () => { const census = await collectRuntimeCensus({ platform: 'win32', diff --git a/tests/kit/heal-natives.test.mjs b/tests/kit/heal-natives.test.mjs index 53f826c7..cc63e52e 100644 --- a/tests/kit/heal-natives.test.mjs +++ b/tests/kit/heal-natives.test.mjs @@ -389,7 +389,7 @@ test('healNatives leaves a present binding that loads alone', async () => { } finally { cleanup(); } }); -// ── Task 4: a repair's evidence round-trips into a subsequent status read ─── +// ── a repair's evidence round-trips into a subsequent status read ─── test('healNatives\'s repair evidence round-trips through rufloRuntimeNatives({ refresh: false }) without a second spawn', async () => { const { pkg, cleanup } = presentBindingTree(); diff --git a/tests/kit/helpers/dashboard-child-server.mjs b/tests/kit/helpers/dashboard-child-server.mjs index b5c32c1c..b2ab9563 100644 --- a/tests/kit/helpers/dashboard-child-server.mjs +++ b/tests/kit/helpers/dashboard-child-server.mjs @@ -1,8 +1,8 @@ // Shared harness for tests that need a REAL dashboard-server.mjs HTTP server, // in a real child process with tests/helpers/spawn-guard.mjs preloaded (Branch -// 6a Task 9 — proving the in-process /api/status path spawns nothing on a warm +// Proving the in-process /api/status path spawns nothing on a warm // cache needs the same "every child_process spawn lands in a ledger" technique -// tests/fixtures/status-zero-spawn-child.mjs (Task 8a) uses for a bare +// tests/fixtures/status-zero-spawn-child.mjs uses for a bare // collect() call, applied here to a whole running server instead). // // Unlike status-zero-spawn-child.mjs (which drives collect() directly and diff --git a/tests/kit/helpers/home-sandbox.mjs b/tests/kit/helpers/home-sandbox.mjs index 65f8bad0..3227851e 100644 --- a/tests/kit/helpers/home-sandbox.mjs +++ b/tests/kit/helpers/home-sandbox.mjs @@ -218,7 +218,7 @@ export const snapshot = (dir) => walk(dir, dir, new Map()); * modified path — the load-bearing assertion behind every `--dry-run` test. * `opts.ignore` is a list of relative-path prefixes (as `snapshot()` keys, * e.g. from `path.relative(dir, someSubdir)`) excluded from the diff — for a - * deliberate, documented side-channel write (Branch 6a Task 5: the shared + * deliberate, documented side-channel write (the shared * evidence probe cache) a caller still wants every OTHER path covered for. * @param {Map} before * @param {string} dir @@ -287,7 +287,9 @@ export function offlineKitConfig(extra = {}) { ttlHours: 24, last: Date.now(), seen: { ruflo: '9.9.9', 'agentic-qe': '9.9.9' }, - self: { last: Date.now(), best: { version: '0.0.1', tag: 'latest' } }, + // This fixture runs against the prerelease kit, whose self check uses + // both channels. A legacy latest-only record must retry under A3. + self: { last: Date.now(), best: { version: '0.0.1', tag: 'latest' }, lastTags: ['latest', 'next'] }, }, ...extra, }; diff --git a/tests/kit/helpers/interruption-scope.mjs b/tests/kit/helpers/interruption-scope.mjs new file mode 100644 index 00000000..512fa0c6 --- /dev/null +++ b/tests/kit/helpers/interruption-scope.mjs @@ -0,0 +1,92 @@ +import fs from 'node:fs'; +import { tempDir } from './temp-dir.mjs'; +import { acquireRunRootHold, releaseRunRootHold } from '../../../scripts/run-roots.mjs'; + +export function childGone(pid) { + try { process.kill(pid, 0); return false; } + catch (error) { return error.code === 'ESRCH'; } +} + +export async function until(check, description, timeout = 5000, signal) { + const deadline = Date.now() + timeout; + while (Date.now() < deadline) { + signal?.throwIfAborted(); + const result = check(); + if (result) return result; + await new Promise((resolve) => setTimeout(resolve, 25)); + } + throw Error(`timed out waiting for ${description}`); +} + +async function stopPid(pid) { + if (childGone(pid)) return; + try { await until(() => childGone(pid), 'private child exit', 2000); return; } catch { /* Escalate exact PID only. */ } + for (const signal of ['SIGTERM', 'SIGKILL']) { + if (childGone(pid)) return; + try { process.kill(pid, signal); } catch (error) { if (error.code !== 'ESRCH') throw error; } + try { await until(() => childGone(pid), 'signaled child exit', 2000); return; } catch { /* Retain on uncertainty. */ } + } + throw Error(`cannot prove owned child ${pid} exited`); +} + +/** One hook owns process shutdown and conditional directory deletion. */ +export function interruptionScope(t, { beforeRemove = () => {} } = {}) { + const hold = acquireRunRootHold(); + const home = tempDir('ak-interrupt', undefined, { manual: true }); + const entries = []; + let cleaning; + let stopping = false; + const active = () => { + t.signal.throwIfAborted(); + if (stopping) throw Error('fixture cleanup has started'); + }; + const cleanup = () => { + stopping = true; + cleaning ??= (async () => { + const results = await Promise.allSettled(entries.map(async (entry) => { + if (entry.stop) fs.writeFileSync(entry.stop, 'exit'); + if (entry.handshake && !(entry.launchError && !entry.child.pid)) { + const data = await until(() => fs.existsSync(entry.handshake) + && JSON.parse(fs.readFileSync(entry.handshake, 'utf8')), 'owned child identity', 2000); + if (!Number.isSafeInteger(data.pid) || data.pid <= 0) throw Error('invalid owned child identity'); + await stopPid(data.pid); + } + if (!entry.closed) { + try { await until(() => entry.closed, 'runner close', 2000); } catch { + entry.child.kill('SIGTERM'); + try { await until(() => entry.closed, 'runner termination', 2000); } catch { + entry.child.kill('SIGKILL'); + await until(() => entry.closed, 'runner forced close', 2000); + } + } + } + if (entry.child.pid && !childGone(entry.child.pid)) throw Error('owned runner PID remains'); + })); + const errors = results.filter(r => r.status === 'rejected').map(r => r.reason); + if (errors.length) throw new AggregateError(errors, 'owned process exit uncertain; retaining fixture and run root'); + releaseRunRootHold(hold); + })(); + return cleaning; + }; + const abort = () => { void cleanup().catch(() => {}); }; + t.after(async () => { + t.signal.removeEventListener('abort', abort); + await cleanup(); + beforeRemove(home); + fs.rmSync(home, { recursive: true, force: true, maxRetries: 3 }); + }); + t.signal.addEventListener('abort', abort, { once: true }); + return { + home, cleanup, active, + wait: (check, description, timeout) => until(check, description, timeout, t.signal), + launch(start, { handshake, stop } = {}) { + active(); + const child = start(); + const entry = { child, handshake, stop, closed: false, launchError: null }; + child.on('error', error => { entry.launchError = error; }); + child.once('close', () => { entry.closed = true; }); + entries.push(entry); + return child; + }, + }; +} diff --git a/tests/kit/helpers/temp-dir.mjs b/tests/kit/helpers/temp-dir.mjs index 817b4a67..70b2df28 100644 --- a/tests/kit/helpers/temp-dir.mjs +++ b/tests/kit/helpers/temp-dir.mjs @@ -8,14 +8,18 @@ import path from 'node:path'; /** * Create `/-XXXXXX` and remove it when the test (or, without - * `t`, the file) finishes. A caller that chdir-ed into it must chdir out first. + * `t`, the file) finishes, unless manual cleanup is requested. A caller that + * chdir-ed into it must chdir out first. * @param {string} prefix * @param {import('node:test').TestContext} [t] + * @param {{manual?:boolean}} [options] Caller owns removal when manual is true. * @returns {string} the real path of the new folder */ -export function tempDir(prefix, t) { +export function tempDir(prefix, t, { manual = false } = {}) { const dir = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), `${prefix}-`))); const remove = () => fs.rmSync(dir, { recursive: true, force: true, maxRetries: 3 }); - if (t) t.after(remove); else after(remove); + if (!manual) { + if (t) t.after(remove); else after(remove); + } return dir; } diff --git a/tests/kit/hook-audit.test.mjs b/tests/kit/hook-audit.test.mjs index 27100ff3..98900965 100644 --- a/tests/kit/hook-audit.test.mjs +++ b/tests/kit/hook-audit.test.mjs @@ -25,6 +25,30 @@ function fixture() { return { root, codexHome, project, cache }; } +function cliFixture(t) { + const fx = fixture(); + // An after hook preserves the command assertion if bounded cleanup also fails. + t.after(() => fs.rmSync(fx.root, { recursive: true, force: true, maxRetries: 3 })); + const preload = path.join(fx.root, 'probes.cjs'); + const launches = path.join(fx.root, 'unexpected-launches.jsonl'); + fs.writeFileSync(launches, ''); + fs.writeFileSync(preload, `const cp = require('node:child_process'); + const fs = require('node:fs'); + for (const method of ['spawn', 'spawnSync', 'exec', 'execSync', 'execFile', 'execFileSync', 'fork']) { + cp[method] = (command, args) => { + if (method === 'spawnSync' && command === 'codex' && JSON.stringify(args) === '["--version"]') + return { status: 0, stdout: 'codex 0.151.0', stderr: '' }; + if (method === 'execFileSync' && command === 'npm' && JSON.stringify(args) === '["root","-g"]') + return ${JSON.stringify(path.join(fx.root, 'global-packages'))}; + fs.appendFileSync(${JSON.stringify(launches)}, JSON.stringify({ method, command, args }) + '\\n'); + throw new Error('unexpected child launch during hook audit'); + }; + } + require('node:module').syncBuiltinESMExports(); + `); + return { ...fx, preload, launches }; +} + test('audit keeps SessionEnd compatibility separate from trust and never proposes an automatic cache edit', () => { const fx = fixture(); try { @@ -411,36 +435,30 @@ test('audit reports malformed hook documents and remains read-only', () => { } }); -test('ak audit hooks exposes the read-only audit as a porcelain command', () => { - const fx = fixture(); - try { - const result = spawnSync(process.execPath, [path.join(repoRoot, 'bin', 'agentic-kit.mjs'), 'audit', 'hooks', '--json'], { - cwd: fx.project, - env: spawnEnv(path.join(fx.root, 'home'), { CODEX_HOME: fx.codexHome }), - encoding: 'utf8', - }); - assert.equal(result.status, 0, result.stderr || result.stdout); - const report = JSON.parse(result.stdout); - assert.equal(report.mode, 'read-only'); - assert.equal(report.summary.automaticActions, 0); - } finally { - fs.rmSync(fx.root, { recursive: true, force: true }); - } +test('ak audit hooks exposes the read-only audit as a porcelain command', (t) => { + const fx = cliFixture(t); + const result = spawnSync(process.execPath, ['--require', fx.preload, path.join(repoRoot, 'bin', 'agentic-kit.mjs'), 'audit', 'hooks', '--json'], { + cwd: fx.project, + env: spawnEnv(path.join(fx.root, 'home'), { CODEX_HOME: fx.codexHome }), + encoding: 'utf8', + }); + assert.equal(result.status, 0, result.stderr || result.stdout); + const report = JSON.parse(result.stdout); + assert.equal(report.mode, 'read-only'); + assert.equal(report.summary.automaticActions, 0); + assert.equal(fs.readFileSync(fx.launches, 'utf8'), ''); }); -test('ak audit hooks human output is not followed by the generic network drift nudge', () => { - const fx = fixture(); - try { - const result = spawnSync(process.execPath, [path.join(repoRoot, 'bin', 'agentic-kit.mjs'), 'audit', 'hooks'], { - cwd: fx.project, - env: spawnEnv(path.join(fx.root, 'home'), { CODEX_HOME: fx.codexHome }), - encoding: 'utf8', - timeout: 3_000, - }); - assert.equal(result.error, undefined, result.error?.message); - assert.equal(result.status, 0, result.stderr || result.stdout); - assert.match(result.stdout, /trust: unchanged/); - } finally { - fs.rmSync(fx.root, { recursive: true, force: true }); - } +test('ak audit hooks human output is not followed by the generic network drift nudge', (t) => { + const fx = cliFixture(t); + const result = spawnSync(process.execPath, ['--require', fx.preload, path.join(repoRoot, 'bin', 'agentic-kit.mjs'), 'audit', 'hooks'], { + cwd: fx.project, + env: spawnEnv(path.join(fx.root, 'home'), { CODEX_HOME: fx.codexHome }), + encoding: 'utf8', + timeout: 3_000, + }); + assert.equal(result.error, undefined, result.error?.message); + assert.equal(result.status, 0, result.stderr || result.stdout); + assert.match(result.stdout, /trust: unchanged/); + assert.equal(fs.readFileSync(fx.launches, 'utf8'), '', 'audit must not launch the network drift nudge'); }); diff --git a/tests/kit/host-dry-run.test.mjs b/tests/kit/host-dry-run.test.mjs index a7db8ed2..68fec135 100644 --- a/tests/kit/host-dry-run.test.mjs +++ b/tests/kit/host-dry-run.test.mjs @@ -59,6 +59,23 @@ function ak(sb, ...args) { const readKit = (home) => fs.readFileSync(path.join(home, '.config', 'agentic-kit', 'kit.json'), 'utf8'); +for (const args of [ + ['host', 'status'], ['host'], ['x', 'host'], +]) { + test(`ak ${args.join(' ')} --dry-run --json reports status without recording evidence`, (t) => { + const sb = sandbox(t); + const beforeHome = snapshot(sb.home); + const beforeProject = snapshot(sb.project); + const r = ak(sb, ...args, '--dry-run', '--json'); + assert.equal(r.status, 0, r.all); + const out = JSON.parse(r.stdout); + assert.deepEqual(Object.keys(out), ['scope', 'config', 'hosts', 'providers']); + assert.ok(out.hosts.claude); + assertUnchanged(beforeHome, sb.home, 'status preview must not write host evidence or config'); + assertUnchanged(beforeProject, sb.project, 'status preview must not write project files'); + }); +} + test('ak host pick --dry-run previews and writes nothing', (t) => { const sb = sandbox(t); const beforeHome = snapshot(sb.home); @@ -119,6 +136,74 @@ test('ak host pick --dry-run --json carries previewOfCurrent, true only with no assert.equal(explicitJson.previewOfCurrent, false); }); +test('host pick refusal and x host alias dry-run emit one JSON object without writes', (t) => { + const sb = sandbox(t); + const beforeHome = snapshot(sb.home); + const beforeProject = snapshot(sb.project); + for (const command of ['host', 'x']) { + const args = command === 'x' ? ['x', 'host'] : ['host']; + const r = ak(sb, ...args, 'pick', '--host', 'claude,opencdoe', '--dry-run', '--json'); + assert.equal(r.status, 2, r.all); + const out = JSON.parse(r.stdout); + assert.deepEqual(Object.keys(out), ['error', 'exitCode']); + assert.equal(out.exitCode, 2); + assert.match(out.error, /unknown host\(s\): opencdoe/); + assert.match(r.stderr, /unknown host\(s\): opencdoe/); + } + assertUnchanged(beforeHome, sb.home, 'refused picks must not touch HOME'); + assertUnchanged(beforeProject, sb.project, 'refused picks must not touch the project'); +}); + +test('host off dry-run JSON previews teardown and preserves configuration', (t) => { + const sb = sandbox(t, divergedConfig()); + const beforeHome = snapshot(sb.home); + const beforeProject = snapshot(sb.project); + const r = ak(sb, 'host', 'off', '--dry-run', '--json'); + assert.equal(r.status, 0, r.all); + const out = JSON.parse(r.stdout); + assert.equal(out.dryRun, true); + assert.deepEqual(out.wouldDisable, ['claude', 'codex']); + assert.equal(out.primaryHost, 'claude'); + assert.deepEqual(out.wouldClear, ['aqe provider/fallback', 'ruflo providers', 'activity routing']); + assert.equal(out.wouldStripManagedProviderEnv, true); + assert.equal(out.wouldRestoreOrRemoveManagedAqeConfig, true); + assert.equal(out.wouldReconcileOpencodeGuidance, true); + assert.equal(out.wouldTeardownOpencode, false); + assert.equal(out.wouldRemoveManagedCodexMcp, false); + assert.match(r.stderr, /dry run/i); + assertUnchanged(beforeHome, sb.home, 'off preview must not touch HOME'); + assertUnchanged(beforeProject, sb.project, 'off preview must not touch the project'); +}); + +test('host reset-routes dry-run JSON reports selected and empty routes without writes', (t) => { + const sb = sandbox(t, divergedConfig()); + const beforeHome = snapshot(sb.home); + const beforeProject = snapshot(sb.project); + const selected = ak(sb, 'host', 'reset-routes', '--activity', 'architecture', '--dry-run', '--json'); + assert.equal(selected.status, 0, selected.all); + assert.deepEqual(JSON.parse(selected.stdout), { dryRun: true, activities: ['architecture'] }); + const empty = ak(sb, 'host', 'reset-routes', '--activity', 'design', '--dry-run', '--json'); + assert.equal(empty.status, 0, empty.all); + assert.deepEqual(JSON.parse(empty.stdout), { dryRun: true, activities: [] }); + const invalid = ak(sb, 'host', 'reset-routes', '--activity', 'not-an-activity', '--dry-run', '--json'); + assert.equal(invalid.status, 0, invalid.all); + assert.deepEqual(JSON.parse(invalid.stdout), { dryRun: true, activities: [] }); + assert.match(invalid.stderr, /unknown activity 'not-an-activity'/); + assertUnchanged(beforeHome, sb.home, 'route previews must not touch HOME'); + assertUnchanged(beforeProject, sb.project, 'route previews must not touch the project'); +}); + +test('host reset-routes dry-run JSON reports no divergence as an empty preview', (t) => { + const sb = sandbox(t); + const beforeHome = snapshot(sb.home); + const beforeProject = snapshot(sb.project); + const r = ak(sb, 'host', 'reset-routes', '--dry-run', '--json'); + assert.equal(r.status, 0, r.all); + assert.deepEqual(JSON.parse(r.stdout), { dryRun: true, activities: [] }); + assertUnchanged(beforeHome, sb.home, 'no-op route preview must not touch HOME'); + assertUnchanged(beforeProject, sb.project, 'no-op route preview must not touch the project'); +}); + test('ak host off --dry-run previews and writes nothing', (t) => { const sb = sandbox(t, divergedConfig()); const beforeHome = snapshot(sb.home); diff --git a/tests/kit/host-executable.test.mjs b/tests/kit/host-executable.test.mjs index e657d140..bc9ff1ae 100644 --- a/tests/kit/host-executable.test.mjs +++ b/tests/kit/host-executable.test.mjs @@ -27,7 +27,7 @@ test('hostExecutable accepts a launcher that answers --version', async () => { assert.deepEqual(calls, [['codex', '--version']]); }); -// ── hostExecutable refresh caching (Branch 6a Task 5) ─────────────────────── +// ── hostExecutable refresh caching ─────────────────────── // `runner` is injectable, so these prove zero-spawn directly instead of // needing the broken-PATH trick tests/kit/host-setup-evidence.test.mjs uses // for the non-injectable have()/hostVersion() probes. diff --git a/tests/kit/host-health-api.test.mjs b/tests/kit/host-health-api.test.mjs index 4b43311e..ceae90cb 100644 --- a/tests/kit/host-health-api.test.mjs +++ b/tests/kit/host-health-api.test.mjs @@ -37,9 +37,9 @@ test('connection route requires same-origin header auth, explicit consent, and b assert.equal((await request(server, 'POST', '/api/host-health/connection', [])).status, 400); assert.equal((await request(server, 'POST', '/api/host-health/connection', body)).status, 200); assert.equal(connections, 1); - assert.equal((await request(server, 'POST', '/api/host-health/local', { host: 'codex' })).status, 200); + assert.equal((await request(server, 'POST', '/api/host-health/local', { host: 'codex' })).status, 405); assert.equal(connections, 1); - assert.ok(reads >= 3); + assert.equal(reads, 2, 'retired local POST does not run another host check'); await server.close(); assert.equal(closed, true); }); diff --git a/tests/kit/host-pick-rerecord.test.mjs b/tests/kit/host-pick-rerecord.test.mjs new file mode 100644 index 00000000..fe755f45 --- /dev/null +++ b/tests/kit/host-pick-rerecord.test.mjs @@ -0,0 +1,62 @@ +import { test, after, beforeEach } from 'node:test'; +import assert from 'node:assert/strict'; +import { format } from 'node:util'; +import { sandboxHome, rmrf, writeKitConfig, offlineKitConfig } from './helpers/home-sandbox.mjs'; + +const home = sandboxHome('ak-host-pick-rerecord'); +after(() => rmrf(home)); +const host = await import('../../src/commands/x/host.mjs'); +const cfg = { integrations: { hosts: { claude: false, codex: true, opencode: false } } }; +const cwd = '/disposable-project'; + +// Node 22 can misread Unicode stdout between binary test events (nodejs/node#65934). +// Keep the messages visible, but frame them as test diagnostics; restore per test. +beforeEach(t => t.mock.method(console, 'log', (...args) => t.diagnostic(format(...args)))); + +for (const [name, initial, ok, expected] of [ + ['successful install', 'absent', true, 2], + ['failed install', 'absent', false, 1], + ['present host', 'external', true, 1], +]) { + test(`host pick ${name} re-records only after success`, async () => { + const states = []; + const facts = []; + const installs = []; + await host.installPickAbsentHosts(cfg, cwd, { + installState: async (_host, opts) => { states.push(opts); return { method: states.length === 1 ? initial : 'npm', version: '1.0.0' }; }, + install: async id => { installs.push(id); return { ok, detail: ok ? 'installed' : 'failed' }; }, + collectFacts: async opts => { facts.push(opts); }, + }); + assert.equal(states.length, expected); + assert.deepEqual(installs, initial === 'absent' ? ['codex'] : []); + if (expected === 2) { + assert.deepEqual(states[1], { refresh: true, record: true, source: 'host-pick' }); + assert.deepEqual(facts, [{ cwd, cfg, refresh: true, record: true, source: 'host-pick' }]); + } else assert.deepEqual(facts, []); + }); +} + +test('disabled hosts do not probe, install, or record evidence', async () => { + await host.installPickAbsentHosts({ integrations: { hosts: {} } }, cwd, { + installState: async () => { throw new Error('disabled host probed'); }, + install: async () => { throw new Error('disabled host installed'); }, + collectFacts: async () => { throw new Error('disabled host recorded'); }, + }); +}); + +test('host command dispatch passes the lifecycle to pick before installation', async () => { + writeKitConfig(home, offlineKitConfig()); + const sentinel = new Error('injected host lifecycle reached'); + let calls = 0; + await assert.rejects(host.run({ + flags: { host: 'codex', yes: true, 'aqe-provider': 'none' }, + positionals: ['pick'], + pkgRoot: process.cwd(), + deps: { hostLifecycle: { + installState: async () => { calls++; throw sentinel; }, + install: async () => { throw new Error('installer reached'); }, + collectFacts: async () => { throw new Error('collector reached'); }, + } }, + }), error => error === sentinel); + assert.equal(calls, 1); +}); diff --git a/tests/kit/host-setup-evidence.test.mjs b/tests/kit/host-setup-evidence.test.mjs index 77634616..66c8f378 100644 --- a/tests/kit/host-setup-evidence.test.mjs +++ b/tests/kit/host-setup-evidence.test.mjs @@ -1,4 +1,4 @@ -// Branch 6a Task 5, Part 1: detectHosts()/hostInstallState() stop spawning +// detectHosts()/hostInstallState() stop spawning // `which`/` --version` on every plain `ak status` call — they reuse // fresh evidence instead, gated by `refresh` (Ruling A: 6h max age; Ruling B: // every new `refresh` param defaults to true, so every caller other than @@ -6,7 +6,7 @@ // // Neither `have()` nor `hostVersion()` (providers.mjs) nor `installedVersion()` // (versions.mjs, via globalRoot()) is injectable, so — mirroring -// tests/kit/natives-runtime.test.mjs's Task 4 precedent — these tests break +// tests/kit/natives-runtime.test.mjs's state-isolation precedent — these tests break // PATH and redirect the npm global root to an empty fixture dir so a REAL // probe deterministically returns "absent", then seed cached evidence with a // value ("...-cache-marker") a real probe could never produce. Getting the @@ -227,7 +227,7 @@ test('collectIntegrationFacts({ refresh: true }) (its default) threads through a resetEvidence(); }); -// ── Task 4/5 joint fix: `record` suppresses persistence, never the probe ──── +// ── `record` suppresses persistence, never the probe ──── // (`sync.mjs`'s plan-computation reads pass `record: false` so a cold-cache // probe never writes evidence as a side effect of merely building the plan.) diff --git a/tests/kit/intel-history.test.mjs b/tests/kit/intel-history.test.mjs index 4a7bc844..603cbcfa 100644 --- a/tests/kit/intel-history.test.mjs +++ b/tests/kit/intel-history.test.mjs @@ -291,12 +291,12 @@ test('readMachineWideIntel aggregates totals and perProject rows across multiple { path: cwdAlpha, label: 'Alpha', key: null, learningScope: 'repository', patternsLearned: 10, patternStoreCount: 2, trajectoriesRecorded: 4, graphLatest: { nodes: 5, edges: 8 }, lastAdaptation: 1000, - learningState: [], + learningState: [], hosts: [], sessionOrigins: [], sessionSurfaces: null, }, { path: cwdBeta, label: 'Beta', key: null, learningScope: 'repository', patternsLearned: 20, patternStoreCount: 3, trajectoriesRecorded: 6, graphLatest: null, lastAdaptation: 2000, - learningState: [], + learningState: [], hosts: [], sessionOrigins: [], sessionSurfaces: null, }, ]); }); @@ -357,10 +357,12 @@ test('readMachineWideIntel degrades a project with missing/malformed data to nul assert.deepEqual(result.perProject[1], { path: cwdEmpty, label: 'Empty', key: null, learningScope: 'unknown', patternsLearned: null, patternStoreCount: 0, trajectoriesRecorded: null, graphLatest: null, lastAdaptation: null, learningState: [], + hosts: [], sessionOrigins: [], sessionSurfaces: null, }); assert.deepEqual(result.perProject[2], { path: cwdMalformed, label: 'Malformed', key: null, learningScope: 'unknown', patternsLearned: null, patternStoreCount: 0, trajectoriesRecorded: null, graphLatest: null, lastAdaptation: null, learningState: [], + hosts: [], sessionOrigins: [], sessionSurfaces: null, }); }); @@ -439,3 +441,10 @@ test('a project row with no learningState degrades to [] rather than undefined', assert.deepEqual(readMachineWideIntel([row]).perProject[0].learningState, []); } }); + +test('readMachineWideIntel passes through declared session presentation evidence', () => { + const evidence = { hosts: ['codex'], sessionOrigins: [{ origin: 'unknown', sessions: 2 }], + sessionSurfaces: [{ host: 'codex', surface: 'codex-cli', sessions: 2 }] }; + const [row] = readMachineWideIntel([{ path: tmp(), label: 'CLI project', ...evidence }]).perProject; + assert.deepEqual({ hosts: row.hosts, sessionOrigins: row.sessionOrigins, sessionSurfaces: row.sessionSurfaces }, evidence); +}); diff --git a/tests/kit/intelligence-picker-groups.test.mjs b/tests/kit/intelligence-picker-groups.test.mjs index b0d31bd6..c02d4d54 100644 --- a/tests/kit/intelligence-picker-groups.test.mjs +++ b/tests/kit/intelligence-picker-groups.test.mjs @@ -28,15 +28,15 @@ test('should_segment_and-alphabetize picker options with the same designations a { key: 'd', label: 'g-p-opaque', learningScope: 'unknown', learningOrigins: ['codex-desktop'] }, ] }); const html = elements['intel-project-select'].innerHTML; - for (const label of ['Git repositories', 'Git worktrees', 'User-level learning', 'Other / unclassified']) { + for (const label of ['Git repositories', 'Git worktrees', 'User-level learning', 'Unknown']) { assert.ok(html.includes(``)); } assert.ok(html.indexOf('value="a"') < html.indexOf('value="z"')); assert.match(html, /value="z" selected>Zulu — Git repository { diff --git a/tests/kit/intelligence-table-groups.test.mjs b/tests/kit/intelligence-table-groups.test.mjs index 25b02883..02d7af7a 100644 --- a/tests/kit/intelligence-table-groups.test.mjs +++ b/tests/kit/intelligence-table-groups.test.mjs @@ -1,3 +1,4 @@ +import * as vocabulary from '../../src/lib/session-surface.mjs'; import { test } from 'node:test'; import assert from 'node:assert/strict'; import fs from 'node:fs'; @@ -23,8 +24,9 @@ test('machine-wide inventory alphabetizes every retained row and preserves KPI t const source = fs.readFileSync(new URL('../../src/lib/dashboard/client/intelligence.mjs', import.meta.url), 'utf8') .replace(/^import .*;$/gm, '').replace(/\bexport /g, ''); const esc = (value) => String(value).replaceAll('&', '&').replaceAll('<', '<').replaceAll('>', '>').replaceAll('"', '"'); - const context = vm.createContext({ document: { getElementById: (id) => elements[id] }, esc, + const context = vm.createContext({ ...vocabulary, document: { getElementById: (id) => elements[id] }, esc, fmtNum: (value) => String(value ?? 0), kpi: (label, value) => `${label}:${value};` }); + vm.runInContext(fs.readFileSync(new URL('../../src/lib/dashboard/client/session-presentation.mjs', import.meta.url), 'utf8').replace(/^import .*;$/gm, '').replace(/\bexport /g, ''), context); vm.runInContext(`${source}\nglobalThis.renderTable=renderMachineWide;`, context); const perProject = Array.from({ length: 8 }, (_, i) => ({ key: `key-${i}`, label: `Repository ${8 - i}`, learningScope: 'repository', patternsLearned: i, patternStoreCount: 1 })); @@ -46,8 +48,9 @@ test('machine-wide inventory renders one designation column and filter pills for const source = fs.readFileSync(new URL('../../src/lib/dashboard/client/intelligence.mjs', import.meta.url), 'utf8') .replace(/^import .*;$/gm, '').replace(/\bexport /g, ''); const esc = (value) => String(value).replaceAll('&', '&').replaceAll('<', '<').replaceAll('>', '>').replaceAll('"', '"'); - const context = vm.createContext({ document: { getElementById: (id) => elements[id] }, esc, + const context = vm.createContext({ ...vocabulary, document: { getElementById: (id) => elements[id] }, esc, fmtNum: (value) => String(value ?? 0), kpi: (label, value) => `${label}:${value};` }); + vm.runInContext(fs.readFileSync(new URL('../../src/lib/dashboard/client/session-presentation.mjs', import.meta.url), 'utf8').replace(/^import .*;$/gm, '').replace(/\bexport /g, ''), context); vm.runInContext(`${source}\nglobalThis.renderTable=renderMachineWide;`, context); context.renderTable({ totals: { patternsLearnedLifetime: 2, projectCount: 2, mostActiveProject: 'Repository' }, perProject: [ { key: 'repo', label: 'Repository', learningScope: 'repository', patternsLearned: 1, patternStoreCount: 1 }, @@ -56,6 +59,7 @@ test('machine-wide inventory renders one designation column and filter pills for const html = elements['mw-table'].innerHTML; assert.match(html, /Designation/); assert.match(html, /Git repository/); - assert.match(html, /ChatGPT Desktop/); + assert.match(html, /Unknown/); + assert.doesNotMatch(html, /ChatGPT Desktop/); assert.match(html, /mw-filter-pill/); }); diff --git a/tests/kit/live-check-evidence.test.mjs b/tests/kit/live-check-evidence.test.mjs index f23f385b..7b010b60 100644 --- a/tests/kit/live-check-evidence.test.mjs +++ b/tests/kit/live-check-evidence.test.mjs @@ -7,6 +7,7 @@ import { test } from 'node:test'; import assert from 'node:assert/strict'; import fs from 'node:fs'; import path from 'node:path'; +import { DatabaseSync } from 'node:sqlite'; import { sandboxHome, assertSandboxed, snapshot, assertUnchanged, captureLog, rmrf, sandboxProject, writeKitConfig, offlineKitConfig, fakeGlobalRoot, @@ -18,9 +19,11 @@ const paths = await import('../../src/lib/paths.mjs'); const evidence = await import('../../src/lib/live-check-evidence.mjs'); const { writeEvidence } = await import('../../src/lib/evidence.mjs'); const aqeSection = (await import('../../src/commands/status/sections/aqe.mjs')).default; +const projectMemorySection = (await import('../../src/commands/status/sections/project-memory.mjs')).default; const { SYNC_STEPS } = await import('../../src/commands/sync.mjs'); assertSandboxed(paths, HOME); -paths._setGlobalRootForTest(fakeGlobalRoot(HOME, { ruflo: '9.9.9', 'agentic-qe': '9.9.9' })); +const GLOBAL_ROOT = fakeGlobalRoot(HOME, { ruflo: '9.9.9', 'agentic-qe': '9.9.9' }); +paths._setGlobalRootForTest(GLOBAL_ROOT); const PROJECT = sandboxProject('ak-live-evidence'); const NOW = Date.parse('2026-09-26T12:00:00Z'); @@ -33,6 +36,70 @@ test('the store lives under the kit state directory, one file per check', () => assert.deepEqual([...evidence.LIVE_CHECK_IDS].sort(), ['aqe-embedding', 'deja-vu', 'mcp', 'memory', 'providers', 'security']); assert.equal(evidence.LIVE_CHECK_TTL_MS, 24 * 3600_000); + assert.deepEqual(evidence.RECORDED_CHECK_IDS, [...evidence.LIVE_CHECK_IDS, 'memory-routes']); +}); + +test('routing evidence is separate from the generic memory round trip and bound to CLI version and platform', () => { + reset(); + const key = (routingVersion, platform) => evidence.liveCheckInputsKey('memory-routes', { routingVersion, platform }); + assert.notEqual(key('3.45.0', 'darwin'), key('3.45.1', 'darwin')); + assert.notEqual(key('3.45.0', 'darwin'), key('3.45.0', 'linux')); + evidence.recordLiveCheck({ id: 'memory', status: 'passed', source: 'status-refresh-live', inputsKey: 'generic' }, { now: NOW }); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: key('3.45.0', 'darwin'), now: NOW }), null); + evidence.recordLiveCheck({ id: 'memory-routes', status: 'passed', source: 'status-refresh-live', + inputsKey: key('3.45.0', 'darwin') }, { now: NOW }); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: key('3.45.1', 'darwin'), now: NOW }).invalidated, true); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: key('3.45.0', 'linux'), now: NOW }).invalidated, true); +}); + +test('the routing key follows the installed CLI even when the wrapper version stays fixed', (t) => { + const root = fs.mkdtempSync(path.join(HOME, 'route-version-')); + t.after(() => { paths._setGlobalRootForTest(GLOBAL_ROOT); rmrf(root); }); + const wrapper = path.join(root, 'ruflo'); + const cli = path.join(wrapper, 'node_modules', '@claude-flow', 'cli'); + fs.mkdirSync(cli, { recursive: true }); + fs.writeFileSync(path.join(wrapper, 'package.json'), JSON.stringify({ version: '3.45.0' })); + const setCli = (version) => fs.writeFileSync(path.join(cli, 'package.json'), JSON.stringify({ version })); + paths._setGlobalRootForTest(root); + setCli('3.45.0'); + const before = evidence.liveCheckInputsKey('memory-routes', { platform: 'darwin' }); + setCli('3.45.1'); + assert.notEqual(evidence.liveCheckInputsKey('memory-routes', { platform: 'darwin' }), before); +}); + +test('two-store row lowers only for applicable routing proof and warns after failure or upgrade', async (t) => { + reset(); + const cwd = sandboxProject('ak-route-row'); + t.after(() => rmrf(cwd)); + const dir = path.join(cwd, '.swarm'); + fs.mkdirSync(dir); + for (const name of ['memory.db', 'agentdb-memory.db']) { + const db = new DatabaseSync(path.join(dir, name)); + db.exec('CREATE TABLE memory_entries (status TEXT); INSERT INTO memory_entries VALUES (NULL)'); + db.close(); + } + const row = async (routingVersion = '3.45.0', platform = 'darwin', now = NOW) => + (await projectMemorySection.collect({ cwd, rufloVersion: routingVersion, platform, now })) + .find((item) => /two project memory stores/.test(item.message)); + const key = evidence.liveCheckInputsKey('memory-routes', { routingVersion: '3.45.0', platform: 'darwin' }); + assert.equal((await row()).level, 'warn'); + evidence.recordLiveCheck({ id: 'memory', status: 'passed', source: 'status-refresh-live', inputsKey: 'generic' }, { now: NOW }); + assert.equal((await row()).level, 'warn'); + evidence.recordLiveCheck({ id: 'memory-routes', status: 'passed', source: 'status-refresh-live', inputsKey: key }, { now: NOW }); + const beforeRead = snapshot(HOME); + assert.equal((await row()).level, 'info'); + assert.match((await row()).message, /existing-corpus access unverified/); + assertUnchanged(beforeRead, HOME, 'ordinary project-memory status reads neither probe nor write'); + assert.equal((await row('3.45.1')).level, 'warn'); + assert.equal((await row('3.45.0', 'linux')).level, 'warn'); + assert.equal((await row(null)).level, 'warn'); + assert.equal((await row('3.45.0', 'darwin', NOW + 2 * evidence.LIVE_CHECK_TTL_MS)).level, 'info'); + evidence.recordLiveCheck({ id: 'memory-routes', status: 'inconclusive', source: 'status-refresh-live', inputsKey: key }, { now: NOW + 1000 }); + assert.equal((await row('3.45.0', 'darwin', NOW + 2000)).level, 'warn'); + evidence.recordLiveCheck({ id: 'memory', status: 'passed', source: 'status-refresh-live', inputsKey: 'generic' }, { now: NOW + 3000 }); + assert.equal((await row('3.45.0', 'darwin', NOW + 4000)).level, 'warn'); + fs.writeFileSync(path.join(evidence.liveCheckDir(), 'memory-routes.json'), '{corrupt'); + assert.equal((await row()).level, 'warn'); }); test('a recorded result reads back with its source and age', () => { diff --git a/tests/kit/live-checks.test.mjs b/tests/kit/live-checks.test.mjs index 75a2b1e0..ad5db1a5 100644 --- a/tests/kit/live-checks.test.mjs +++ b/tests/kit/live-checks.test.mjs @@ -12,6 +12,7 @@ import assert from 'node:assert/strict'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; +import { DatabaseSync } from 'node:sqlite'; import { fileURLToPath } from 'node:url'; import { sandboxHome, assertSandboxed, snapshot, assertUnchanged, captureLog, rmrf, @@ -22,13 +23,15 @@ const HOME = sandboxHome('ak-live-checks'); const paths = await import('../../src/lib/paths.mjs'); const live = await import('../../src/lib/live-checks.mjs'); const evidence = await import('../../src/lib/live-check-evidence.mjs'); +const projectMemorySection = (await import('../../src/commands/status/sections/project-memory.mjs')).default; const { refreshRequestFromFlags, cliRefreshStages } = await import('../../src/lib/refresh.mjs'); const status = await import('../../src/commands/status.mjs'); const { loadKitConfig } = await import('../../src/lib/config.mjs'); assertSandboxed(paths, HOME); const PROJECT = sandboxProject('ak-live-checks'); -paths._setGlobalRootForTest(fakeGlobalRoot(HOME, { ruflo: '9.9.9' })); +const GLOBAL_ROOT = fakeGlobalRoot(HOME, { ruflo: '9.9.9' }); +paths._setGlobalRootForTest(GLOBAL_ROOT); const PKG_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..'); const seedHome = (cfg = offlineKitConfig()) => { @@ -93,7 +96,7 @@ test('each check carries its timeout and the evidence id its result is remembere assert.equal(entry('learning').evidenceId, null); assert.equal(entry('harvest').evidenceId, null); assert.equal(entry('aqe').evidenceId, null, 'the aqe proof remembers only its embedding request, itself'); - assert.equal(entry('memory-routes').evidenceId, 'memory'); + assert.equal(entry('memory-routes').evidenceId, 'memory-routes'); }); test('without --only the quick checks that apply run; mcp runs whenever Codex is enabled', async () => { @@ -219,9 +222,11 @@ test('a named check that does not apply says on its line that its result is not assert.match(out, /^⚠ {2}mcp failed \(20 ms\) — effective Codex MCP inventory unavailable; not remembered: this check does not apply to your setup$/m); assert.match(out, /^✓ security passed \(30 ms\)$/m, 'a check that applies says nothing more'); const skipped = await runStatus({ refresh: 'live', only: ['deja-vu'] }, [ - { id: 'deja-vu', status: 'passed', reason: null, elapsedMs: 2, entries: [], applies: false }, + { id: 'deja-vu', status: 'skipped', reason: 'disabled and unowned', elapsedMs: 2, entries: [], applies: false }, ]); - assert.match(skipped.out, /^✓ deja-vu passed \(2 ms\) — not remembered: this check does not apply to your setup$/m); + assert.equal(skipped.code, 1, 'a named skipped proof did not pass'); + assert.match(skipped.out, /Running live checks \(\d+ ms\): 1 skipped/); + assert.match(skipped.out, /^⚠ {2}deja-vu skipped \(2 ms\) — disabled and unowned; not remembered: this check does not apply to your setup$/m); }); test('one line per check; with --only each check\'s own lines are indented under it', async () => { @@ -585,25 +590,203 @@ test('a failed check is remembered for status with its first failure as the reas assert.equal(got.reason, '@claude-flow/security missing'); }); -test('the memory-routes proof is remembered as the memory check; learning and harvest are not remembered', async () => { +test('the memory-routes proof is remembered separately; learning and harvest are not remembered', async () => { seedHome(); rmrf(evidence.liveCheckDir()); await runOnly(['memory-routes', 'learning', 'harvest']); - const got = evidence.readLiveCheck('memory', {}); + const got = evidence.readLiveCheck('memory-routes', {}); assert.equal(got.status, 'failed'); assert.equal(got.source, 'status-refresh-live'); - assert.deepEqual(fs.readdirSync(evidence.liveCheckDir()), ['memory.json']); + assert.equal(evidence.readLiveCheck('memory', {}).status, 'failed'); + assert.deepEqual(fs.readdirSync(evidence.liveCheckDir()), ['memory-routes.json', 'memory.json']); +}); + +test('a failed or timed-out routing run replaces a previous pass; a later generic memory pass cannot revive it', async () => { + seedHome(); + rmrf(evidence.liveCheckDir()); + const cfg = offlineKitConfig(); + const route = (run, timeoutMs = 50) => ({ id: 'memory-routes', evidenceId: 'memory-routes', + timeoutMs, applies: () => true, run }); + const generic = { id: 'memory', evidenceId: 'memory', timeoutMs: 50, applies: () => true, + run: async () => true }; + await live.runLiveChecks({ cfg, cwd: PROJECT, checks: [route(async () => ({ status: 'passed' }))] }); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'passed'); + await live.runLiveChecks({ cfg, cwd: PROJECT, checks: [route(async () => false)] }); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'failed'); + await live.runLiveChecks({ cfg, cwd: PROJECT, checks: [route(() => new Promise(() => {}), 10)], graceMs: 1 }); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'inconclusive'); + await live.runLiveChecks({ cfg, cwd: PROJECT, checks: [generic] }); + assert.equal(evidence.readLiveCheck('memory').status, 'passed'); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'inconclusive'); +}); + +test('a route timeout after the CLI proof keeps the generic memory pass', async () => { + seedHome(); + rmrf(evidence.liveCheckDir()); + const check = { id: 'memory-routes', evidenceId: 'memory-routes', timeoutMs: 10, applies: () => true, + run: ({ onCliOutcome }) => { + onCliOutcome({ status: 'passed', reason: null }); + return new Promise(() => {}); + } }; + const [result] = await live.runLiveChecks({ cfg: offlineKitConfig(), cwd: PROJECT, checks: [check], graceMs: 1 }); + assert.equal(result.status, 'inconclusive'); + assert.equal(evidence.readLiveCheck('memory').status, 'passed'); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'inconclusive'); +}); + +test('a CLI upgrade during the route check cannot attribute the old observation to the new version', async (t) => { + seedHome(); + rmrf(evidence.liveCheckDir()); + const root = fs.mkdtempSync(path.join(HOME, 'routing-upgrade-')); + const cli = path.join(root, 'ruflo', 'node_modules', '@claude-flow', 'cli'); + fs.mkdirSync(cli, { recursive: true }); + const setVersion = (version) => fs.writeFileSync(path.join(cli, 'package.json'), JSON.stringify({ version })); + t.after(() => { paths._setGlobalRootForTest(GLOBAL_ROOT); rmrf(root); }); + paths._setGlobalRootForTest(root); + setVersion('3.45.0'); + const oldKey = evidence.liveCheckInputsKey('memory-routes'); + const check = { id: 'memory-routes', evidenceId: 'memory-routes', timeoutMs: 50, applies: () => true, + run: async ({ onCliOutcome }) => { + onCliOutcome({ status: 'passed', reason: null }); + setVersion('3.45.1'); + return { status: 'passed', reason: null }; + } }; + const [result] = await live.runLiveChecks({ cfg: offlineKitConfig(), cwd: PROJECT, checks: [check] }); + const currentKey = evidence.liveCheckInputsKey('memory-routes'); + assert.notEqual(currentKey, oldKey); + assert.equal(result.status, 'inconclusive'); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: oldKey }).status, 'inconclusive'); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: oldKey }).invalidated, false); + assert.equal(evidence.readLiveCheck('memory-routes', { inputsKey: currentKey }).invalidated, true); + assert.equal(evidence.readLiveCheck('memory').status, 'passed'); + + const project = sandboxProject('ak-route-upgrade-row'); + t.after(() => rmrf(project)); + const swarm = path.join(project, '.swarm'); + fs.mkdirSync(swarm); + for (const name of ['memory.db', 'agentdb-memory.db']) { + const db = new DatabaseSync(path.join(swarm, name)); + db.exec('CREATE TABLE memory_entries (status TEXT); INSERT INTO memory_entries VALUES (NULL)'); + db.close(); + } + const row = async (version) => (await projectMemorySection.collect({ cwd: project, + rufloVersion: version })).find((item) => /two project memory stores/.test(item.message)); + assert.equal((await row('3.45.0')).level, 'warn'); + assert.equal((await row('3.45.1')).level, 'warn'); }); test('a skipped deja-vu proof is not remembered as a pass', async () => { seedHome(); rmrf(evidence.liveCheckDir()); const [r] = await runOnly(['deja-vu']); - assert.equal(r.status, 'passed'); + assert.equal(r.status, 'skipped'); + assert.equal(r.reason, 'disabled and unowned'); assert.match(texts(r), /deja-vu disabled and unowned — skipped/); assert.equal(evidence.readLiveCheck('deja-vu', {}), null); }); +test('a skipped applicable check never writes conformance evidence', async () => { + seedHome(); + rmrf(evidence.liveCheckDir()); + const [result] = await live.runLiveChecks({ cfg: offlineKitConfig(), cwd: PROJECT, + checks: [{ id: 'deja-vu', evidenceId: 'deja-vu', applies: () => true, + run: async () => ({ status: 'skipped', reason: 'not installed' }) }] }); + assert.equal(result.status, 'skipped'); + assert.equal(evidence.readLiveCheck('deja-vu', {}), null); +}); + +test('learning removes only its call-owned folder after success, missing artifacts, thrown runner, and abort', async (t) => { + const tmpRoot = privateRoot(t); + const unrelated = path.join(tmpRoot, 'unrelated'); + fs.mkdirSync(unrelated); + const cases = [ + async (_cmd, _args, { cwd }) => { + const neural = path.join(cwd, '.claude-flow', 'neural'); + fs.mkdirSync(neural, { recursive: true }); + fs.writeFileSync(path.join(neural, 'stats.json'), '{"patternsLearned":1}'); + fs.writeFileSync(path.join(neural, 'patterns.json'), '[{"id":"one"}]'); + return { code: 0, stdout: '', stderr: '' }; + }, + async () => ({ code: 0, stdout: '', stderr: '' }), + async () => { throw new Error('runner failed'); }, + async () => { throw new DOMException('aborted', 'AbortError'); }, + ]; + for (const [i, runner] of cases.entries()) { + const { result } = await captureLog(() => live.verifyLearning({ tmpRoot, runner })); + assert.equal(result, i === 0, `case ${i}`); + assert.deepEqual(fs.readdirSync(tmpRoot), ['unrelated'], `case ${i} left a call-owned folder`); + } +}); + +test('memory removes its call-owned folder even when final purge throws', async (t) => { + const tmpRoot = privateRoot(t); + fs.mkdirSync(path.join(tmpRoot, 'unrelated')); + const runner = async (_cmd, args) => { + if (args[1] === 'init') return { code: 0, stdout: '', stderr: '' }; + if (args[1] === 'store') return { code: 1, stdout: '', stderr: 'store failed' }; + throw new Error('purge failed'); + }; + await assert.rejects(captureLog(() => live.verifyMemory({ tmpRoot, runner, haveCmd: async () => true })), /purge failed/); + assert.deepEqual(fs.readdirSync(tmpRoot), ['unrelated']); +}); + +test('memory removes only its call-owned folder when init throws or is aborted', async (t) => { + const tmpRoot = privateRoot(t); + fs.mkdirSync(path.join(tmpRoot, 'unrelated')); + for (const error of [new Error('runner failed'), new DOMException('aborted', 'AbortError')]) { + const { result } = await captureLog(() => live.verifyMemory({ tmpRoot, + runner: async () => { throw error; }, haveCmd: async () => true })); + assert.equal(result, false); + assert.deepEqual(fs.readdirSync(tmpRoot), ['unrelated']); + } +}); + +test('memory removes its call-owned folder after a successful mocked CLI round trip', async (t) => { + const tmpRoot = privateRoot(t); + fs.mkdirSync(path.join(tmpRoot, 'unrelated')); + let db; + let storedValue; + const runner = async (_cmd, args, { cwd }) => { + if (args[1] === 'init') { + fs.mkdirSync(path.join(cwd, '.swarm')); + db = new DatabaseSync(path.join(cwd, '.swarm', 'memory.db')); + db.exec('CREATE TABLE memory_entries (namespace TEXT, key TEXT)'); + } else if (args[1] === 'store') { + db.prepare('INSERT INTO memory_entries VALUES (?, ?)').run(args[args.indexOf('-n') + 1], args[args.indexOf('-k') + 1]); + storedValue = args[args.indexOf('--value') + 1]; + } else if (args[1] === 'retrieve') { + return { code: 0, stdout: storedValue, stderr: '' }; + } else if (args[1] === 'purge') { + db.exec('DELETE FROM memory_entries'); + db.close(); + } + return { code: 0, stdout: '', stderr: '' }; + }; + const { result } = await captureLog(() => live.verifyMemory({ + tmpRoot, runner, haveCmd: async () => true, observeRoutes: false, + })); + assert.equal(result, true); + assert.deepEqual(fs.readdirSync(tmpRoot), ['unrelated']); +}); + +test('harvest removes only its call-owned folder on success, nonzero result, thrown runner, and abort', async (t) => { + const tmpRoot = privateRoot(t); + fs.mkdirSync(path.join(tmpRoot, 'unrelated')); + for (const mode of ['success', 'nonzero', 'throw', 'abort']) { + const dirs = []; + const runner = async (_cmd, _args, { cwd }) => { + dirs.push(cwd); + if (mode === 'throw') throw new Error('runner failed'); + if (mode === 'abort') throw new DOMException('aborted', 'AbortError'); + return { code: mode === 'nonzero' ? 1 : 0, stdout: '', stderr: '' }; + }; + const { result } = await captureLog(() => live.verifyHarvest({ tmpRoot, runner, haveCmd: async () => true })); + assert.equal(result, mode === 'success', mode); + assert.ok(dirs.length >= 1 && dirs.every((dir) => path.dirname(dir) === tmpRoot), mode); + assert.deepEqual(fs.readdirSync(tmpRoot), ['unrelated'], mode); + } +}); + test('an aqe proof stopped before the embedding request remembers no embedding result', async () => { seedHome(); rmrf(evidence.liveCheckDir()); diff --git a/tests/kit/live-process-sessions.test.mjs b/tests/kit/live-process-sessions.test.mjs index 9b1bf1e4..b3cf3bc1 100644 --- a/tests/kit/live-process-sessions.test.mjs +++ b/tests/kit/live-process-sessions.test.mjs @@ -13,6 +13,12 @@ test('host process detection recognizes controllers and rejects helpers', () => assert.equal(hostFromCommand('node /opt/bin/codex'), 'codex'); assert.equal(hostFromCommand('/usr/local/bin/opencode --continue'), 'opencode'); assert.equal(hostFromCommand('codex mcp-server'), null); + assert.equal(hostFromCommand('codex -s read-only mcp-server'), null); + assert.equal(hostFromCommand('node /opt/bin/codex -c model="test" mcp-server'), null); + assert.equal(hostFromCommand('codex -c developer_instructions="say mcp-server hello"'), 'codex'); + assert.equal(hostFromCommand('codex --config=developer_instructions="say mcp-server hello"'), 'codex'); + assert.equal(hostFromCommand('codex --config developer_instructions=say mcp-server hello'), 'codex', + 'a flattened unquoted config value cannot prove an MCP subcommand'); assert.equal(hostFromCommand('/opt/bin/codex-code-mode-host'), null); assert.equal(hostFromCommand('node app.mjs codex'), null); assert.equal(hostFromCommand('python worker.py claude'), null); @@ -60,7 +66,7 @@ test('runtime survey keeps top-level sessions and folds nested host workers into test('runtime survey classifies host services and desktop apps without retaining argv', async () => { const startedAt = 'Mon Aug 3 12:00:00 2026'; const processRows = parseProcessList([ - `100 1 ${startedAt} /Applications/ChatGPT.app/Contents/Resources/codex /Applications/ChatGPT.app/Contents/Resources/codex app-server`, + `100 1 ${startedAt} /Applications/ChatGPT.app/Contents/Resources/codex-cli/CodexCLI.app/Contents/MacOS/codex /Applications/ChatGPT.app/Contents/Resources/codex-cli/CodexCLI.app/Contents/MacOS/codex app-server`, `200 1 ${startedAt} /Users/me/.codex/plugins/.plugin-appserver/codex /Users/me/.codex/plugins/.plugin-appserver/codex app-server`, `300 1 ${startedAt} /Applications/Claude.app/Contents/MacOS/Claude /Applications/Claude.app/Contents/MacOS/Claude`, `400 1 ${startedAt} /usr/local/bin/claude claude`, @@ -81,6 +87,58 @@ test('runtime survey classifies host services and desktop apps without retaining 'classification emits an enum, never the potentially sensitive argv'); }); +test('Codex global options before app-server identify a service, never a session', async () => { + const startedAt = 'Mon Aug 3 12:00:00 2026'; + const app = '/Applications/ChatGPT.app/Contents/MacOS/ChatGPT'; + const bundled = '/Applications/ChatGPT.app/Contents/Resources/codex-cli/CodexCLI.app/Contents/MacOS/codex'; + const processRows = [ + { pid: 10, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex -s read-only -a never app-server' }, + { pid: 20, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex --config model="test" app-server' }, + { pid: 30, ppid: 1, startedAt, executable: app, command: `${app} app-server` }, + { pid: 31, ppid: 30, startedAt, executable: bundled, + command: `${bundled} --config=model="test" --strict-config app-server` }, + { pid: 40, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex --config app-server' }, + { pid: 50, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex --unknown value app-server' }, + { pid: 60, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex -- app-server' }, + { pid: 70, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex --model app-server' }, + { pid: 80, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex --config malformed app-server' }, + { pid: 90, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex -c developer_instructions="say app-server hello"' }, + { pid: 91, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex --config=developer_instructions="say app-server hello"' }, + { pid: 92, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex --config developer_instructions=say app-server hello' }, + { pid: 93, ppid: 1, startedAt, executable: '/usr/local/bin/codex', + command: 'codex --config developer_instructions="say app-server hello" app-server' }, + ]; + const cwdByPid = new Map(processRows.map((row) => [row.pid, `/repos/${row.pid}`])); + const survey = await surveyHostProcesses({ platform: 'darwin', processRows, cwdByPid, + metricsByPid: new Map() }); + assert.deepEqual(survey.processes.map(({ pid, controllerKind }) => ({ pid, controllerKind })), [ + { pid: 10, controllerKind: 'host-service' }, + { pid: 20, controllerKind: 'host-service' }, + { pid: 30, controllerKind: 'desktop-app' }, + { pid: 40, controllerKind: 'project-session' }, + { pid: 50, controllerKind: 'project-session' }, + { pid: 60, controllerKind: 'project-session' }, + { pid: 70, controllerKind: 'project-session' }, + { pid: 80, controllerKind: 'project-session' }, + { pid: 90, controllerKind: 'project-session' }, + { pid: 91, controllerKind: 'project-session' }, + { pid: 92, controllerKind: 'project-session' }, + { pid: 93, controllerKind: 'host-service' }, + ]); + assert.deepEqual((await listActiveHostSessions({ platform: 'darwin', processRows, cwdByPid, + inspectWorkspace: async () => null })).map(({ pid }) => pid), [40, 50, 60, 70, 80, 90, 91, 92]); +}); + // macOS `ps -o comm=` prints the executable's full path, and many real paths // contain spaces (`Application Support`, `Visual Studio Code.app`). The Claude // desktop app hosts its own Claude Code CLI under such a path (#238 item 3). @@ -127,6 +185,39 @@ test('a Claude Code CLI hosted by the Claude desktop app is its own project sess .map((session) => ({ pid: session.pid, host: session.host })), [{ pid: 200, host: 'claude' }]); }); +test('both desktop applications remain applications while their bundled CLIs are sessions', async () => { + const startedAt = 'Mon Aug 3 12:00:00 2026'; + const claudeApp = '/Applications/Claude.app/Contents/MacOS/Claude'; + const chatgptApp = '/Applications/ChatGPT.app/Contents/MacOS/ChatGPT'; + const codexCli = '/Applications/ChatGPT.app/Contents/Resources/codex-cli/CodexCLI.app/Contents/MacOS/codex'; + const rows = [ + { pid: 100, ppid: 1, startedAt, executable: claudeApp, command: claudeApp }, + { pid: 110, ppid: 100, startedAt, executable: DESKTOP_CLI, command: DESKTOP_CLI }, + { pid: 120, ppid: 110, startedAt, executable: '/usr/local/bin/codex', command: 'codex exec review' }, + { pid: 200, ppid: 1, startedAt, executable: chatgptApp, command: chatgptApp }, + { pid: 210, ppid: 200, startedAt, executable: codexCli, command: `${codexCli} --model test` }, + { pid: 220, ppid: 200, startedAt, executable: codexCli, command: `${codexCli} app-server` }, + { pid: 300, ppid: 1, startedAt, executable: '/usr/local/bin/codex', command: 'codex' }, + { pid: 400, ppid: 1, startedAt, executable: '/Applications/Other.app/Contents/MacOS/codex', command: 'codex' }, + { pid: 500, ppid: 1, startedAt, executable: '/usr/local/bin/claude', command: 'claude --prompt app-server /Applications/ChatGPT.app/Contents/Resources/codex-cli/bin/codex' }, + ]; + const cwdByPid = new Map([...rows.map((row) => [row.pid, `/repos/${row.pid}`])]); + const survey = await surveyHostProcesses({ platform: 'darwin', processRows: rows, + cwdByPid, metricsByPid: new Map() }); + assert.deepEqual(survey.processes.map(({ pid, host, application, controllerKind }) => + ({ pid, host, application, controllerKind })), [ + { pid: 100, host: null, application: 'Claude Desktop', controllerKind: 'desktop-app' }, + { pid: 110, host: 'claude', application: null, controllerKind: 'project-session' }, + { pid: 200, host: null, application: 'ChatGPT desktop app', controllerKind: 'desktop-app' }, + { pid: 210, host: 'codex', application: null, controllerKind: 'project-session' }, + { pid: 300, host: 'codex', application: null, controllerKind: 'project-session' }, + { pid: 500, host: 'claude', application: null, controllerKind: 'project-session' }, + ]); + const sessions = await listActiveHostSessions({ platform: 'darwin', processRows: rows, + cwdByPid, inspectWorkspace: async () => null }); + assert.deepEqual(sessions.map(({ pid }) => pid), [110, 210, 300, 500]); +}); + test('a host CLI nested under an ordinary controller still folds into it', async () => { const startedAt = 'Mon Aug 3 12:00:00 2026'; const processRows = [ @@ -177,11 +268,34 @@ test('the POSIX survey finds a desktop-hosted CLI end to end through ps output w }); assert.deepEqual(sessions.map((session) => ({ pid: session.pid, host: session.host })), [ { pid: 200, host: 'claude' }, - { pid: 300, host: 'claude' }, ]); assert.equal(calls[1].args[1], '300,200', 'argv is still fetched only for host candidates'); }); +test('the POSIX header and targeted argv passes discover the ChatGPT bundled Codex CLI', async () => { + const startedAt = 'Mon Aug 3 12:00:00 2026'; + const app = '/Applications/ChatGPT.app/Contents/MacOS/ChatGPT'; + const cli = '/Applications/ChatGPT.app/Contents/Resources/codex-cli/CodexCLI.app/Contents/MacOS/codex'; + const calls = []; + const execFileImpl = async (_command, args) => { + calls.push(args); + if (args.includes('pid=,ppid=,lstart=,comm=')) return { stdout: [ + `100 1 ${startedAt} ${app}`, + `110 100 ${startedAt} ${cli}`, + `200 1 ${startedAt} /Applications/Other.app/Contents/MacOS/Other`, + ].join('\n') }; + if (args.includes('pid=,args=')) return { stdout: `100 ${app}\n110 ${cli} exec review\n` }; + throw new Error('unexpected process probe'); + }; + const survey = await surveyHostProcesses({ platform: 'darwin', uid: 501, execFileImpl, + cwdByPid: new Map([[100, '/'], [110, '/repos/work']]), metricsByPid: new Map() }); + assert.deepEqual(survey.processes.map(({ pid, host, application }) => ({ pid, host, application })), [ + { pid: 100, host: null, application: 'ChatGPT desktop app' }, + { pid: 110, host: 'codex', application: null }, + ]); + assert.equal(calls[1][1], '100,110', 'unknown applications never reach the argv pass'); +}); + test('workspace inspection is shared consistently across Claude, Codex, and OpenCode', async () => { const startedAt = 'Mon Aug 3 12:00:00 2026'; const processRows = parseProcessList([ diff --git a/tests/kit/live-service.test.mjs b/tests/kit/live-service.test.mjs index b5cb6528..85491242 100644 --- a/tests/kit/live-service.test.mjs +++ b/tests/kit/live-service.test.mjs @@ -185,6 +185,51 @@ test('bounded native discovery rotates to a newer transcript created after start service.close(); }); +test('a native transcript re-entering the bounded window resumes without replaying accepted records', () => { + const sb = sandbox(); + const old = path.join(sb.claude, 'old.jsonl'); + const recent = path.join(sb.claude, 'recent.jsonl'); + fs.writeFileSync(old, line({ + type: 'user', sessionId: 'old', cwd: '/work/old-project', + timestamp: '2026-07-27T11:00:00Z', message: { content: 'fixture only' }, + })); + fs.utimesSync(old, new Date(1_000), new Date(1_000)); + let tick; + const service = new LiveSessionsService({ + roots: sb.roots, maxFiles: 1, readCodexState: () => null, + setInterval: (fn) => { tick = fn; return { unref() {} }; }, clearInterval: () => {}, + now: () => '2026-07-27T12:00:00Z', + }); + try { + service.start(); + const before = service.snapshot().health.claude.accepted; + assert.ok(before > 0); + fs.writeFileSync(recent, line({ + type: 'user', sessionId: 'recent', cwd: '/work/recent-project', + timestamp: '2026-07-27T11:01:00Z', message: { content: 'fixture only' }, + })); + fs.utimesSync(recent, new Date(2_000), new Date(2_000)); + tick(); + const rotated = service.snapshot(); + const afterRotation = rotated.health.claude.accepted; + assert.ok(afterRotation > before); + service.close(); + service.start(); + fs.utimesSync(old, new Date(3_000), new Date(3_000)); + tick(); + const reentered = service.snapshot(); + assert.equal(reentered.health.claude.accepted, afterRotation); + assert.equal(reentered.sessions.length, rotated.sessions.length); + assert.equal(reentered.projects.length, rotated.projects.length); + fs.appendFileSync(old, line({ + type: 'assistant', sessionId: 'old', cwd: '/work/old-project', + timestamp: '2026-07-27T12:00:01Z', message: { content: 'fixture only' }, + })); + tick(); + assert.equal(service.snapshot().health.claude.accepted, afterRotation + 1); + } finally { service.close(); } +}); + test('metadata bootstrap is adversarially privacy bounded', () => { const sb = sandbox(); fs.writeFileSync(path.join(sb.codex, 'rollout-2026-07-27T12-00-00-x1.jsonl'), [ diff --git a/tests/kit/maintenance-dashboard-api.test.mjs b/tests/kit/maintenance-dashboard-api.test.mjs index 9e3c1efe..56c56e77 100644 --- a/tests/kit/maintenance-dashboard-api.test.mjs +++ b/tests/kit/maintenance-dashboard-api.test.mjs @@ -303,14 +303,14 @@ test('dashboard Maintenance API keeps GET lazy and mutation paths exact', async assert.equal(unknownMutation.status, 405); const rescanned = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(rescanned.status, 200); - assert.equal(service.calls.scan, 1, 'only the explicit scan query invokes maintenance scanning'); + assert.equal(rescanned.status, 400); + assert.equal(service.calls.scan, 0, 'GET cannot invoke maintenance scanning'); const unknownRefresh = await request(server, '/api/maintenance?refresh=deep', { origin: false, fetchSite: null }); const duplicateRefresh = await request(server, '/api/maintenance?refresh=scan&refresh=scan', { origin: false, fetchSite: null }); const unknownQuery = await request(server, '/api/maintenance?extra=scan', { origin: false, fetchSite: null }); assert.deepEqual([unknownRefresh.status, duplicateRefresh.status, unknownQuery.status], [400, 400, 400]); - assert.equal(service.calls.scan, 1, 'ambiguous scan queries never invoke maintenance scanning'); + assert.equal(service.calls.scan, 0, 'scan queries never invoke maintenance scanning'); }); test('dashboard Maintenance reports provider activity and refuses action requests during a scan', async (t) => { @@ -465,18 +465,6 @@ test('dashboard Maintenance API distinguishes pre-mutation refusal from a receip // ── ADR-0048: provider scans chain the inventory rebuild (QE defect D5) ───── -function eventually(predicate, message, timeout = 1500) { - const started = Date.now(); - return new Promise((resolve, reject) => { - const check = () => { - if (predicate()) return resolve(undefined); - if (Date.now() - started >= timeout) return reject(new Error(message)); - setTimeout(check, 5); - }; - check(); - }); -} - function recordingManagement({ refreshInventory, measured = true } = {}) { const calls = []; const rebuilt = { inventoryId: 'inv_refreshed', capturedAt: '2026-09-05T12:00:00.000Z' }; @@ -493,84 +481,39 @@ function recordingManagement({ refreshInventory, measured = true } = {}) { return { calls, facade }; } -test('GET /api/maintenance?refresh=scan chains exactly one management.refreshInventory({ deep:false }) without awaiting it', async (t) => { - const service = fixtureService(); - let release; - const gate = new Promise((resolve) => { release = resolve; }); - const management = recordingManagement({ refreshInventory: () => gate }); - const server = await startDashboard({ port: 0, maintenance: service, management: management.facade, usage: {} }); - t.after(() => server.close()); - - const plain = await request(server, '/api/maintenance', { origin: false, fetchSite: null }); - assert.equal(plain.status, 200); - assert.deepEqual(management.calls, [], 'a plain read never rebuilds the inventory'); - - const first = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(first.status, 200, 'the provider-scan response never waits for the inventory rebuild'); - await eventually(() => management.calls.length === 1, 'the provider scan must chain one inventory rebuild'); - assert.deepEqual(management.calls, [{ deep: false }], 'a cheap provider check rebuilds only; it never walks discovery sources'); - - const second = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(second.status, 200); - assert.equal(service.calls.scan, 2); - assert.equal(management.calls.length, 1, 'a rebuild still in flight is joined, not duplicated'); - release(); - await eventually(() => JSON.parse(JSON.stringify(management.calls)).length === 1, 'settled'); - const third = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(third.status, 200); - await eventually(() => management.calls.length === 2, 'a later provider scan rebuilds again once the flight settled'); -}); - -test('an inventory rebuild failure is logged and never turns the provider-scan response into an error', async (t) => { - const service = fixtureService(); - const management = recordingManagement({ refreshInventory: async () => { throw new Error('inventory store unavailable'); } }); - const logged = []; - const original = console.error; - console.error = (...args) => { logged.push(args.map(String).join(' ')); }; - t.after(() => { console.error = original; }); - const server = await startDashboard({ port: 0, maintenance: service, management: management.facade, usage: {} }); - t.after(() => server.close()); - - const scanned = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(scanned.status, 200); - await eventually(() => logged.some((line) => /maintenance inventory refresh failed/.test(line)), 'the failure is logged'); - assert.deepEqual(management.calls, [{ deep: false }]); - const again = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(again.status, 200); - await eventually(() => management.calls.length === 2, 'a failed rebuild does not wedge the single-flight slot'); -}); - -test('a completed System deep scan chains one provider scan and then one inventory rebuild', async (t) => { +test('GET Maintenance scan queries never invoke provider checks or inventory rebuilds', async (t) => { const service = fixtureService(); const management = recordingManagement(); - const collector = { - async read() { return { scan: { running: false, phase: 'idle' }, generatedAt: '2026-09-05T12:00:00.000Z' }; }, - async refreshDeep() { return { ok: true, persisted: { ok: true } }; }, - scanState() { return { running: true, phase: 'system' }; }, - }; - const server = await startDashboard({ port: 0, system: collector, maintenance: service, management: management.facade, usage: {} }); + const server = await startDashboard({ port: 0, maintenance: service, management: management.facade, usage: {} }); t.after(() => server.close()); - - const started = await request(server, '/api/system?refresh=deep', { origin: false, fetchSite: null }); - assert.equal(started.status, 200); - await eventually(() => service.calls.scan === 1, 'the completed System scan refreshes Maintenance evidence once'); - await eventually(() => management.calls.length === 1, 'the provider scan then rebuilds the inventory once'); - assert.deepEqual(management.calls, ['rebuildAfterMeasurement'], - 'a machine measurement walks discovery sources and rebuilds, and never also runs the cheap rebuild'); + assert.equal((await request(server, '/api/maintenance', { origin: false, fetchSite: null })).status, 200); + for (const query of ['refresh=scan', 'refresh=deep', 'refresh=scan&refresh=scan', 'trees=1', 'extra=scan']) { + const response = await request(server, '/api/maintenance?' + query, { origin: false, fetchSite: null }); + assert.equal(response.status, 400, query); + assert.deepEqual(JSON.parse(response.body), { error: 'start a refresh with POST /api/refresh' }); + } + assert.equal(service.calls.scan, 0); + assert.deepEqual(management.calls, []); }); -test('a completed System deep scan falls back to the cheap rebuild when the facade has no measured rebuild', async (t) => { - const service = fixtureService(); - const management = recordingManagement({ measured: false }); - const collector = { - async read() { return { scan: { running: false, phase: 'idle' } }; }, - async refreshDeep() { return { ok: true, persisted: { ok: true } }; }, - }; - const server = await startDashboard({ port: 0, system: collector, maintenance: service, management: management.facade, usage: {} }); - t.after(() => server.close()); - assert.equal((await request(server, '/api/system?refresh=deep', { origin: false, fetchSite: null })).status, 200); - await eventually(() => management.calls.length === 1, 'the chain still rebuilds'); - assert.deepEqual(management.calls, [{ deep: false }]); +test('POST refresh stages retain machine measurement, provider scan and inventory rebuild order', async () => { + const { dashboardRefreshStages } = await import('../../src/lib/dashboard/refresh-api.mjs'); + const calls = []; + const stage = dashboardRefreshStages({ cwd: process.cwd(), + getSystem: async () => ({ refreshDeep: async options => { calls.push(['machine', options]); return { ok: true, persisted: { ok: true } }; } }), + getMaintenance: async () => ({ scan: async options => { calls.push(['maintenance', options]); return { scan: {} }; } }), + refreshInventoryAfterProviderScan: async options => { calls.push(['inventory', options]); return {}; }, + getHostReadiness: async () => ({}), statusCollect: async () => ({}), loadConfig: () => ({}), + }); + assert.equal((await stage.machine({ projectTrees: true })).ok, true); + assert.equal((await stage.maintenance()).ok, true); + assert.equal((await stage.inventory({ strength: 'machine' })).ok, true); + assert.deepEqual(calls, [['machine', { includeProjectTrees: true }], ['maintenance', { deep: false }], + ['inventory', { measured: true }]]); + calls.length = 0; + assert.equal((await stage.maintenance()).ok, true); + assert.equal((await stage.inventory({ strength: 'local' })).ok, true); + assert.deepEqual(calls, [['maintenance', { deep: false }], ['inventory', { measured: false }]]); }); test('an injected maintenance service without an injected facade never composes the default facade (hermetic)', async (t) => { @@ -578,8 +521,8 @@ test('an injected maintenance service without an injected facade never composes const server = await startDashboard({ port: 0, maintenance: service, usage: {} }); t.after(() => server.close()); const scanned = await request(server, '/api/maintenance?refresh=scan', { origin: false, fetchSite: null }); - assert.equal(scanned.status, 200); - assert.equal(service.calls.scan, 1); + assert.equal(scanned.status, 400); + assert.equal(service.calls.scan, 0); const inventory = await request(server, '/api/maintenance/v2/inventory', { origin: false, fetchSite: null }); assert.equal(inventory.status, 503); assert.deepEqual(JSON.parse(inventory.body), { error: 'maintenance management unavailable' }); diff --git a/tests/kit/maintenance-dashboard-client-labels.test.mjs b/tests/kit/maintenance-dashboard-client-labels.test.mjs index 1d1e7272..48a92331 100644 --- a/tests/kit/maintenance-dashboard-client-labels.test.mjs +++ b/tests/kit/maintenance-dashboard-client-labels.test.mjs @@ -100,17 +100,9 @@ test('MNT-EVD-006: extractor is not vacuous (finds real, allowed literals)', () assert.ok(literals.includes('Clear all')); }); -// The Refresh-evidence rename introduced two labels the extractor structurally -// cannot see: "Refresh evidence"/"Refreshing evidence…" are assigned via a -// plain JS ternary (`button.textContent=busy?...:...`), not a `label:` -// property or a static HTML text node, and "Re-measure machine" lives only in -// page.mjs's server-rendered markup, which this file does not scan at all. -// Assert them directly so a future rename cannot silently reintroduce a -// prohibited word here without any test noticing. -test('MNT-EVD-006: the Refresh evidence / Re-measure machine controls carry no prohibited label', () => { - for (const label of ['Refresh evidence', 'Refreshing evidence…', 'Re-measure machine']) { - assert.equal(isProhibitedLabel(label), false, label); - } +test('the retired Maintenance refresh controls are absent from the page', () => { + const page = fs.readFileSync(new URL('../../src/lib/dashboard/page.mjs', import.meta.url), 'utf8'); + for (const id of ['mnt-check-providers', 'mnt-remeasure']) assert.doesNotMatch(page, new RegExp('id="' + id + '"')); }); // (b) The copied label maps in maintenance-workspace.mjs must equal diff --git a/tests/kit/maintenance-dashboard-e2e.test.mjs b/tests/kit/maintenance-dashboard-e2e.test.mjs index 73e5e65a..a41d249d 100644 --- a/tests/kit/maintenance-dashboard-e2e.test.mjs +++ b/tests/kit/maintenance-dashboard-e2e.test.mjs @@ -186,10 +186,24 @@ test('dashboard HTTP executes and undoes one maintenance finding through the rea const server = await startDashboard({ port: 0, maintenance: service, usage: {}, maintenanceOptions: HERMETIC_MAINTENANCE, fetchStatus: async () => ({ overall: 'ok', rows: [] }), + refreshStages: { + maintenance: async () => { await service.scan({ deep: false }); return { ok: true }; }, + inventory: async () => ({ ok: true }), local: async () => ({ ok: true }), + }, }); t.after(() => server.close()); - const report = await request(server, '/api/maintenance?refresh=scan'); + const started = await request(server, '/api/refresh', { method: 'POST', body: { strength: 'local' } }); + assert.equal(started.status, 202); + let state; + for (let attempt = 0; attempt < 100; attempt++) { + state = (await request(server, '/api/refresh')).body; + if (!state.running) break; + await new Promise(resolve => setTimeout(resolve, 5)); + } + assert.equal(state?.running, false, 'the explicit refresh finishes'); + assert.equal(state?.ok, true); + const report = await request(server, '/api/maintenance'); assert.equal(report.status, 200); assert.equal(report.headers['cache-control'], 'no-store'); assertNoPrivateTransport(report.body); diff --git a/tests/kit/maintenance-dashboard-v2-api.test.mjs b/tests/kit/maintenance-dashboard-v2-api.test.mjs index 9f9ddede..9e96bcca 100644 --- a/tests/kit/maintenance-dashboard-v2-api.test.mjs +++ b/tests/kit/maintenance-dashboard-v2-api.test.mjs @@ -588,7 +588,7 @@ test('v2 inventory projection keeps the nine inspector sections and the page env assert.equal(page.groups.length, 1); assert.deepEqual(Object.keys(page.groups[0].placements[0]).sort(), [ 'breadcrumb', 'carrier', 'consumerHosts', 'displayName', 'guidanceLane', 'kind', 'placementId', 'projectId', 'projectKind', - 'repositoryEvidence', 'repositoryId', 'repositoryLabel', 'repositoryObservedAt', 'rowAction', 'scope', 'sessionOrigins', 'versions', + 'repositoryEvidence', 'repositoryId', 'repositoryLabel', 'repositoryObservedAt', 'rowAction', 'scope', 'sessionOrigins', 'sessionSurfaces', 'versions', ]); const inspector = publicInspector(inspectorFor(inventory, PLACEMENT)); assert.deepEqual(Object.keys(inspector).sort(), [ @@ -748,26 +748,18 @@ test('v2 inventory projection carries row kind and opaque-id facet labels over e assert.doesNotMatch(JSON.stringify(hostile), /Users\/alice|leak/); }); -test('report({ refresh:true }) fires afterScan once after a successful provider scan and never lets it fail the response', async () => { +test('report reads persisted evidence and ignores retired refresh arguments', async () => { const events = []; - const service = { async report() { return {}; }, async scan() { events.push('scan'); return {}; }, async plan() { return {}; } }; + const service = { async report() { events.push('report'); return {}; }, async scan() { events.push('scan'); return {}; }, async plan() { return {}; } }; const api = createMaintenanceDashboardApi({ service, sessionToken: SESSION, afterScan: () => { events.push('afterScan'); throw new Error('rebuild failed'); }, }); const plain = fakeRes(); await api.report({}, plain, { refresh: false }); - assert.deepEqual([plain.out.status, events], [200, []]); + assert.deepEqual([plain.out.status, events], [200, ['report']]); const refreshed = fakeRes(); await api.report({}, refreshed, { refresh: true }); - assert.deepEqual([refreshed.out.status, events], [200, ['scan', 'afterScan']]); - const failing = createMaintenanceDashboardApi({ - service: { ...service, async scan() { throw new Error('provider check failed'); } }, sessionToken: SESSION, - afterScan: () => { events.push('never'); }, - }); - const failed = fakeRes(); - await failing.report({}, failed, { refresh: true }); - assert.equal(failed.out.status, 503); - assert.equal(events.includes('never'), false, 'afterScan only follows a successful scan'); + assert.deepEqual([refreshed.out.status, events], [200, ['report', 'report']]); }); test('lastRefresh is allowlisted on the report, inventory, and guidance envelopes with a guarded, label-safe message (QE D6b)', async () => { @@ -895,6 +887,31 @@ test('public activity retains historical scans as well as latest source summarie assert.equal(payload.scanHistory.length, 2); }); +test('public activity allows a bounded pause time while omitting invalid timestamps and private metadata', () => { + const scan = { sourceId: SOURCE, state: 'paused', recordedAt: '2026-09-09T12:00:00.000Z', completedAt: null, privatePath: PRIVATE_PATH }; + const payload = publicActivity({ scans: [scan], scanHistory: [scan, { ...scan, recordedAt: PRIVATE_PATH }, { ...scan, recordedAt: '1' }] }); + assert.equal(payload.scans[0].recordedAt, scan.recordedAt); + assert.equal(payload.scans[0].completedAt, null); + assert.equal(payload.scanHistory[1].recordedAt, undefined); + assert.equal(payload.scanHistory[2].recordedAt, undefined); + assert.equal(JSON.stringify(payload).includes(PRIVATE_PATH), false); +}); + +test('public activity rejects impossible pause dates and retains valid leap-day offset stamps', () => { + const base = { sourceId: SOURCE, state: 'paused', completedAt: null }; + const valid = '2024-02-29T23:59:59.125+05:30'; + const payload = publicActivity({ scans: [{ ...base, recordedAt: valid }], scanHistory: [ + { ...base, recordedAt: '2026-09-31T12:00:00Z', completedAt: '2026-09-30T12:00:00Z' }, + { ...base, recordedAt: '2025-02-29T12:00:00Z' }, + { ...base, recordedAt: valid }, + ] }); + assert.equal(payload.scans[0].recordedAt, valid); + assert.equal(payload.scanHistory[0].recordedAt, undefined); + assert.equal(payload.scanHistory[0].completedAt, '2026-09-30T12:00:00Z'); + assert.equal(payload.scanHistory[1].recordedAt, undefined); + assert.equal(payload.scanHistory[2].recordedAt, valid); +}); + test('v2 reports native persistence refusal without suggesting an action started', async () => { const refusal = Object.assign(new Error('private adapter unavailable'), { code: 'MAINTENANCE_PERSISTENCE_UNAVAILABLE' }); const { post } = harness({ management: stubManagement({ planAction: async () => { throw refusal; } }).facade }); diff --git a/tests/kit/maintenance-discovery-orchestrator.test.mjs b/tests/kit/maintenance-discovery-orchestrator.test.mjs index 1d72f60c..c9d92ba2 100644 --- a/tests/kit/maintenance-discovery-orchestrator.test.mjs +++ b/tests/kit/maintenance-discovery-orchestrator.test.mjs @@ -145,6 +145,99 @@ test('pause and resume: a paused source keeps its checkpoint and completes once assert.equal(published.scanState, 'published'); }); +test('a paused scan survives restart without publishing partial evidence, then resumes', async (t) => { + const dir = fixture(t); + const root = fixture(t); + buildLargeTree(root, { dirs: 5, filesPerDir: 4 }); + const first = control(root, dir); + const [checkpointed] = await first.orchestrator.start({ sourceIds: [SOURCE.sourceId], maxSlices: 1 }); + assert.equal(checkpointed.scanState, 'checkpointed'); + const paused = first.orchestrator.pause({ sourceId: SOURCE.sourceId }); + assert.equal(paused.state, 'paused'); + assert.ok(paused.visited > 0); + const [summary] = first.historyStore.list(); + assert.equal(summary.state, 'paused'); + assert.equal(summary.visited, paused.visited); + assert.equal(summary.completedPartitions, paused.completedPartitions); + assert.equal(summary.pendingPartitions, paused.pendingPartitions); + assert.equal(summary.completedAt, null); + assert.ok(summary.recordedAt); + assert.equal(first.lastGoodStore.current().length, 0); + + const restarted = control(root, dir); + const [row] = restarted.orchestrator.coverage(); + assert.equal(row.state, 'paused'); + assert.equal(row.visited, paused.visited); + assert.equal(row.completedPartitions, paused.completedPartitions); + assert.equal(row.pendingPartitions, paused.pendingPartitions); + assert.equal(row.lastCompletedAt, null); + assert.equal(restarted.checkpointStore.list().length, 1); + const [published] = await restarted.orchestrator.resume({ sourceIds: [SOURCE.sourceId] }); + assert.equal(published.scanState, 'published'); + assert.equal(restarted.orchestrator.coverage()[0].state, 'complete'); + assert.equal(restarted.checkpointStore.list().length, 0); + assert.equal(restarted.lastGoodStore.current().length, 1); + assert.equal(control(root, dir).orchestrator.coverage()[0].state, 'complete'); +}); + +test('a confirmed stop after pause prevents old paused state from returning', async (t) => { + const dir = fixture(t); + const root = fixture(t); + buildLargeTree(root); + const first = control(root, dir); + await first.orchestrator.start({ sourceIds: [SOURCE.sourceId], maxSlices: 1 }); + first.orchestrator.pause({ sourceId: SOURCE.sourceId }); + first.orchestrator.stop({ sourceId: SOURCE.sourceId, confirmed: true }); + assert.equal(first.checkpointStore.list().length, 0); + assert.equal(control(root, dir).orchestrator.coverage()[0].state, 'not-scanned'); +}); + +test('pause does not claim durable state when its summary write fails', async (t) => { + const dir = fixture(t); + const root = fixture(t); + buildLargeTree(root); + const first = control(root, dir); + await first.orchestrator.start({ sourceIds: [SOURCE.sourceId], maxSlices: 1 }); + const historyStore = { ...first.historyStore, recordSummary: () => { throw new Error('history unavailable'); } }; + const second = control(root, dir, { historyStore }); + assert.throws(() => second.orchestrator.pause({ sourceId: SOURCE.sourceId }), /history unavailable/); + assert.equal(second.orchestrator.coverage()[0].state, 'scanning'); + assert.equal(second.historyStore.list().length, 0); + assert.equal(second.checkpointStore.list().length, 1); +}); + +test('pause refuses when no history store can record the boundary', async (t) => { + const dir = fixture(t); + const root = fixture(t); + buildLargeTree(root); + const first = control(root, dir); + await first.orchestrator.start({ sourceIds: [SOURCE.sourceId], maxSlices: 1 }); + const restarted = control(root, dir, { historyStore: null }); + assert.throws(() => restarted.orchestrator.pause({ sourceId: SOURCE.sourceId }), /history unavailable/); + assert.equal(restarted.orchestrator.coverage()[0].state, 'scanning'); +}); + +test('paused history restores only its matching environment and loses to a later failure at the same millisecond', (t) => { + const dir = fixture(t); + const root = fixture(t); + const fixed = Date.parse('2026-09-08T12:00:00.000Z'); + const first = control(root, dir, { now: () => fixed }); + first.historyStore.recordSummary({ scanId: 'one', sourceId: SOURCE.sourceId, + environmentId: 'other', state: 'paused', completedAt: null, + recordedAt: new Date(fixed).toISOString(), visited: 7 }); + assert.equal(control(root, dir).orchestrator.coverage()[0].state, 'not-scanned'); + first.historyStore.recordSummary({ scanId: 'two', sourceId: SOURCE.sourceId, + environmentId: SOURCE.environmentId, state: 'paused', completedAt: null, + recordedAt: new Date(fixed).toISOString(), visited: 8 }); + assert.equal(control(root, dir).orchestrator.coverage()[0].state, 'paused'); + first.historyStore.recordSummary({ scanId: 'two', sourceId: SOURCE.sourceId, + environmentId: SOURCE.environmentId, state: 'failed', completedAt: new Date(fixed).toISOString(), + visited: 9, limitingReason: 'io-failure' }); + const [row] = control(root, dir).orchestrator.coverage(); + assert.equal(row.state, 'failed'); + assert.equal(row.visited, 9); +}); + test('MNT-DSC-018: stop previews affected work before confirming, then removes the active scan and retains history', async (t) => { const controlDir = fixture(t); const root = fixture(t); @@ -662,3 +755,18 @@ test('scan history rolls over independently at ten records per source and surviv [2, 3, 4, 5, 6, 7, 8, 9, 10, 11]); } }); + +test('clearHistory keeps a paused summary while its continuation is open', (t) => { + const dir = fixture(t); + const checkpoints = createCheckpointStore(path.join(dir, 'checkpoints'), { fsImpl: fs }); + const history = createScanHistoryStore(dir, { fsImpl: fs }); + const scanId = 'paused-scan'; + checkpoints.write({ scanId, sourceId: SOURCE.sourceId, environmentId: SOURCE.environmentId, + scanEpoch: 1, completedPartitions: [], pendingPartitions: [], workCounts: { visited: 1 }, + sourceStamps: [], createdAt: new Date().toISOString() }); + history.recordSummary({ scanId, sourceId: SOURCE.sourceId, environmentId: SOURCE.environmentId, + state: 'paused', completedAt: null, recordedAt: new Date().toISOString(), visited: 1 }); + assert.deepEqual(history.clearHistory(), { removed: 0, kept: 1 }); + checkpoints.remove(scanId); + assert.deepEqual(history.clearHistory(), { removed: 1, kept: 0 }); +}); diff --git a/tests/kit/maintenance-focus-client.test.mjs b/tests/kit/maintenance-focus-client.test.mjs index 6f778dca..e04c1cc3 100644 --- a/tests/kit/maintenance-focus-client.test.mjs +++ b/tests/kit/maintenance-focus-client.test.mjs @@ -1,13 +1,15 @@ import { test } from 'node:test'; import assert from 'node:assert/strict'; import fs from 'node:fs'; +import * as vocabulary from '../../src/lib/session-surface.mjs'; import { mntLanguageLogo } from '../../src/lib/dashboard/client/maintenance-language-logos.mjs'; const esc=s=>String(s).replace(/&/g,'&').replace(/s,mntFacetValueLabel:(_,v)=>v,mntIcon:()=>'',mntProjectKindBadge:kind=>esc(kind),mntAvailableTo:()=>''},['mntFocusChoose','mntFocusBack','mntFocusCrumbs','renderMntFocusResults']);} +const { projectSurfacesHtml } = load('session-presentation', { ...vocabulary, esc }, ['projectSurfacesHtml']); +function focus(state){return load('maintenance-focus',{MNT:state,esc,projectSurfacesHtml,mntLanguageLogo,MNT_SCOPE_LABELS:{user:'User',across:'All scopes',project:'Projects'},mntKindLabel:s=>s,mntFacetValueLabel:(_,v)=>v,mntIcon:()=>'',mntProjectKindBadge:kind=>esc(kind),mntAvailableTo:()=>''},['mntFocusChoose','mntFocusBack','mntFocusCrumbs','renderMntFocusResults']);} test('navigation turns User and resource type into explicit filters while retaining host refinements',()=>{ const state={scope:'across',facets:{consumer:['claude']}};const api=focus(state); api.mntFocusChoose('scope','user');api.mntFocusChoose('kind','mcp-registration');api.mntFocusChoose('resource','res_1'); diff --git a/tests/kit/maintenance-management-activity.test.mjs b/tests/kit/maintenance-management-activity.test.mjs index e0a6de0f..b6fddc27 100644 --- a/tests/kit/maintenance-management-activity.test.mjs +++ b/tests/kit/maintenance-management-activity.test.mjs @@ -77,6 +77,51 @@ test('activity shows the latest scan per source and environment without changing assert.deepEqual(scanHistory, original); }); +test('a newer paused record is the latest state and retains its distinct recorded time', () => { + const base = { sourceId: 'claude', environmentId: 'local', label: 'Claude', visited: 12 }; + const complete = { ...base, state: 'complete', completedAt: '2026-09-08T12:00:00.000Z' }; + const paused = { ...base, state: 'paused', recordedAt: '2026-09-09T12:00:00.000Z', completedAt: null }; + const result = buildActivity({ scanHistory: [paused, complete] }); + assert.equal(result.scans[0].state, 'paused'); + assert.equal(result.scans[0].recordedAt, paused.recordedAt); + assert.equal(result.scans[0].completedAt, null); + assert.equal(result.scanHistory[1].completedAt, complete.completedAt); +}); + +test('invalid recorded time falls back to completion without passing metadata through', () => { + const base = { sourceId: 'claude', environmentId: 'local', state: 'complete' }; + const result = buildActivity({ scanHistory: [ + { ...base, completedAt: '2026-09-09T12:00:00.000Z', recordedAt: '/private/path', privatePath: '/private/path' }, + { ...base, completedAt: '2026-09-08T12:00:00.000Z' }, + ] }); + assert.equal(result.scans[0].completedAt, '2026-09-09T12:00:00.000Z'); + assert.equal(result.scans[0].recordedAt, null); + assert.equal(buildActivity({ scanHistory: [{ ...base, completedAt: '/private/path' }] }).scanHistory[0].completedAt, null); + assert.equal(JSON.stringify(result).includes('/private/path'), false); +}); + +test('an impossible pause date cannot supersede a real completion or invent a calendar day', () => { + const base = { sourceId: 'claude', environmentId: 'local' }; + const completed = { ...base, state: 'complete', completedAt: '2026-09-30T12:00:00.000Z' }; + const paused = { ...base, state: 'paused', recordedAt: '2026-09-31T12:00:00Z', completedAt: null }; + const result = buildActivity({ scanHistory: [paused, completed] }); + assert.equal(result.scans[0].state, 'complete'); + assert.equal(result.scanHistory[0].recordedAt, null); + assert.equal(result.scanHistory[0].completedAt, null); + const fallback = buildActivity({ scanHistory: [{ ...completed, recordedAt: paused.recordedAt }] }); + assert.equal(fallback.scans[0].recordedAt, null); + assert.equal(fallback.scans[0].completedAt, completed.completedAt); +}); + +test('scan timestamps accept leap days, offsets, and fractional seconds', () => { + const recordedAt = '2024-02-29T23:59:59.125+05:30'; + const completedAt = '2024-02-29T08:00:00Z'; + const result = buildActivity({ scanHistory: [{ sourceId: 'a', environmentId: 'local', state: 'complete', recordedAt, completedAt }] }); + assert.equal(result.scans[0].recordedAt, recordedAt); + assert.equal(result.scans[0].completedAt, completedAt); + assert.equal(buildActivity({ scanHistory: [{ sourceId: 'a', environmentId: 'local', state: 'complete', completedAt: '2025-02-29T08:00:00Z' }] }).scans[0].completedAt, null); +}); + test('no label anywhere in buildActivity output is prohibited', () => { const activity = buildActivity({ receipts: [INTERRUPTED_RECEIPT], diff --git a/tests/kit/maintenance-management-service.test.mjs b/tests/kit/maintenance-management-service.test.mjs index 0a9da99f..562e49b5 100644 --- a/tests/kit/maintenance-management-service.test.mjs +++ b/tests/kit/maintenance-management-service.test.mjs @@ -696,6 +696,35 @@ test('scan lifecycle: pauseScan and resumeScan accept a bare sourceId (symmetric } }); +test('paused collection root survives a service restart in Discovery and the Inventory partial banner', async (t) => { + const controlRoot = fixtureRoot(t); + const root = fixtureRoot(t); + fs.writeFileSync(path.join(root, 'root-file.txt'), 'x'); + for (let i = 0; i < 5; i += 1) { + const child = path.join(root, `project-${i}`); + fs.mkdirSync(path.join(child, '.git'), { recursive: true }); + fs.writeFileSync(path.join(child, 'file.txt'), 'x'); + } + const sourceId = opaqueId('src', { kind: 'collection-root', root }, INSTALLATION_KEY); + const discovery = { collectionRoots: [{ root, sourceId, maxDepth: null, includeNetwork: false }] }; + const first = buildHarness(t, { controlRoot, discovery }); + await first.service.startScan({ sourceId, maxSlices: 1 }); + const paused = first.service.pauseScan({ sourceId }); + const pausedRow = paused.coverage.find((entry) => entry.sourceId === sourceId); + assert.equal(pausedRow.state, 'paused'); + assert.ok(pausedRow.visited > 0); + + const restarted = buildHarness(t, { controlRoot, discovery }); + const row = restarted.service.discovery().coverage.find((entry) => entry.sourceId === sourceId); + assert.equal(row.state, 'paused'); + assert.equal(row.visited, pausedRow.visited); + assert.equal(row.label, 'Collection root'); + await restarted.service.refreshInventory(); + const page = restarted.service.inventory({}); + assert.equal(page.partialSources.total, 5, 'the paused root joins four unscanned automatic roots'); + assert.ok(page.partialSources.entries.some((entry) => entry.sourceId === sourceId && entry.state === 'paused')); +}); + test('DSC: setAutomaticSource, addExclusion, and removeExclusion round-trip through kit.json', async (t) => { const h = buildHarness(t); await h.service.setAutomaticSource({ sourceId: 'ollama', enabled: false }); diff --git a/tests/kit/maintenance-presentation.test.mjs b/tests/kit/maintenance-presentation.test.mjs index adc9ff12..9ef6fe36 100644 --- a/tests/kit/maintenance-presentation.test.mjs +++ b/tests/kit/maintenance-presentation.test.mjs @@ -63,59 +63,37 @@ test('a project filter never strips context from a user placement', () => { const html = cards(state).renderMntGroups([group('r1', [row('p1')])], { value: 0 }); assert.match(html, /User · Codex › Skills/); }); -function operation(get) { - const state = {}; - const nodes = Object.fromEntries(['mnt-check-providers', 'mnt-remeasure', 'mnt-check-providers-status', 'mnt-operation-elapsed'].map((id) => [id, { textContent: '', dataset: {} }])); - const api = client('maintenance-operation', { MNT: state, mntGet: get, mntRefreshActiveDestination() {}, loadSystem: async () => {}, SYSTEM: {}, systemBusy: false, document: { getElementById: (id) => nodes[id], addEventListener() {} }, setInterval: () => 1, clearInterval() {}, setTimeout: (fn) => queueMicrotask(fn) }, ['mntCheckProviders', 'mntBuildStatusOf', 'mntAwaitInventoryBuild']); - return { state, nodes, ...api }; -} +test('Maintenance writes are blocked during the shared Refresh operation', () => { + const api = client('maintenance-operation', { + MNT: { externalScanBusy: false }, refreshRunning: () => true, + }, ['mntWritesBlocked']); + assert.equal(api.mntWritesBlocked(), true); +}); +test('Maintenance hash synchronization adopts an externally changed destination', () => { + const location = { hash: '#system/maintenance/inventory?scope=across' }; + const history = { replaceState(_state, _title, hash) { location.hash = hash; } }; + const api = client('maintenance-workspace', { location, history, localStorage: { setItem() {} } }, ['MNT', 'mntSyncHash']); + api.mntSyncHash(); + location.hash = '#system/maintenance/guidance?scope=project'; + api.mntSyncHash(); + assert.equal(api.MNT.destination, 'guidance'); + assert.equal(api.MNT.scope, 'project'); + assert.match(location.hash, /^#system\/maintenance\/guidance\?scope=project/); + location.hash = '#usage/score'; + api.mntSyncHash(); + assert.equal(location.hash, '#usage/score'); +}); test('an existing inventory does not mask a running or failed refresh', () => { - const api = operation(() => {}); + const api = client('maintenance-operation', {}, ['mntBuildStatusOf']); assert.equal(api.mntBuildStatusOf({ scanRequired: false, lastRefresh: { status: 'running' } }), 'running'); assert.equal(api.mntBuildStatusOf({ scanRequired: false, lastRefresh: { status: 'failed' } }), 'failed'); }); -test('refresh keeps both buttons disabled until a fresh inventory is published', async () => { - let inventoryCalls = 0, providerCalls = 0, release; - const publication = new Promise((resolve) => { release = resolve; }); - let signalWaiting; - const waiting = new Promise((resolve) => { signalWaiting = resolve; }); - const api = operation(async (url) => { - if (url.includes('/v2/inventory')) { - inventoryCalls++; - if (inventoryCalls === 1) return { scanRequired: false, lastRefresh: { at: 'old', status: 'ok' } }; - signalWaiting();return publication; - } - if (url.includes('?refresh=scan')) return {}; - providerCalls++; - return { activity: { status: 'idle' }, scan: { status: 'complete', checkedAt: providerCalls === 1 ? 'old' : 'new', coverage: 'complete' } }; - }); - const run = api.mntCheckProviders(); - await waiting; - assert.equal(api.nodes['mnt-check-providers'].disabled, true); - assert.equal(api.nodes['mnt-remeasure'].disabled, true); - assert.match(api.state.operation.message, /Updating inventory/); - release({ scanRequired: false, lastRefresh: { at: 'new', status: 'ok' }, partialSources: { total: 1 } }); - await run; - assert.equal(api.nodes['mnt-remeasure'].disabled, false); - assert.match(api.state.operation.message, /coverage gaps/); -}); -test('provider failure is visible and releases the busy state', async () => { - const api = operation(async (url) => { - if (url.includes('/v2/inventory')) return { lastRefresh: { at: 'old' } }; - if (url.includes('?refresh=scan')) throw new Error('Evidence request failed.'); - return { scan: { checkedAt: 'old' } }; - }); - await api.mntCheckProviders(); - assert.equal(api.state.operation.failed, true); - assert.match(api.state.operation.message, /Evidence request failed/); - assert.equal(api.nodes['mnt-remeasure'].disabled, false); -}); test('filesystem completion excludes provider checks without inventing their success', () => { const ctx = { state: {}, orchestrator: () => ({ coverage: () => [{ sourceId: 'files', state: 'complete', visited: 4 }], progress: () => [] }), lastGoodDiscoveryStore: { current: () => [] }, listSources: () => [{ sourceId: 'files', filesystem: true }, { sourceId: 'providers', filesystem: false, label: 'Providers' }] }; const result = scanProgress(ctx)(); assert.match(result.narrative, /1 of 1/); assert.equal(result.coverage.find((c) => c.sourceId === 'providers').state, 'not-scanned'); - assert.equal(result.evidenceChecks[0].method, 'Refresh evidence'); + assert.equal(result.evidenceChecks[0].method, 'Refresh'); }); test('Hermes path honors HERMES_HOME and otherwise resolves the user configuration', () => { const previous = process.env.HERMES_HOME; @@ -306,20 +284,6 @@ test('public inventory uses the measured project location instead of guessing fr assert.equal(page.facetLabels.project[skill.projectId], 'ampel'); assert.doesNotMatch(JSON.stringify(page), /\/private\/repo/); }); -test('a completed refresh over stale machine evidence waits for publication and asks for remeasurement', async () => { - let inventoryCalls=0, providerCalls=0; - const api=operation(async url=>{ - if(url.includes('/v2/inventory'))return {scanRequired:false,lastRefresh:{at:++inventoryCalls===1?'old':'new',status:'ok'}}; - if(url.includes('?refresh=scan'))return {scan:{status:'complete',checkedAt:'new'}}; - return {activity:{status:'complete'},scan:{status:'stale',checkedAt:++providerCalls===1?'old':'new',coverage:'partial'}}; - }); - await api.mntCheckProviders(); - assert.equal(api.state.operation.failed,false); - assert.equal(inventoryCalls,2); - assert.match(api.state.operation.message,/Evidence refreshed.*Re-measure machine/); - assert.equal(api.nodes['mnt-check-providers'].disabled,false); -}); - test('version inspector localizes measured and checked instants without interpreting version identifiers', () => { const api = client('maintenance-inspector', { esc, mntKindLabel: (value) => value, diff --git a/tests/kit/maintenance-project-grouping.test.mjs b/tests/kit/maintenance-project-grouping.test.mjs index 4169a9c5..d8a30138 100644 --- a/tests/kit/maintenance-project-grouping.test.mjs +++ b/tests/kit/maintenance-project-grouping.test.mjs @@ -1,9 +1,14 @@ +import fs from 'node:fs'; +import path from 'node:path'; +import vm from 'node:vm'; import { test } from 'node:test'; import assert from 'node:assert/strict'; import { buildManagementInventory } from '../../src/lib/maintenance/management/projection.mjs'; import { runInventoryQuery } from '../../src/lib/maintenance/management/query.mjs'; import { publicInventoryPage } from '../../src/lib/dashboard/maintenance-api.mjs'; import { validateMaintenanceV2Query } from '../../src/lib/dashboard/maintenance-security.mjs'; +import { discoverProjectSources } from '../../src/lib/footprint/project-sources.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; const options = { installationKey: 'maintenance-grouping-fixture-key', environment: { platform: 'darwin' }, now: () => 1700000000000 }; const repository = { kind: 'git', repositoryId: 'repository:0123456789abcdef0123', root: '/work/repo', @@ -62,3 +67,103 @@ test('should_report_session_counts_once_per_project_despite_multiple_installed_r assert.equal(page.navigation.nodes[0].count, 2); assert.equal(page.navigation.nodes[0].sessionOrigins[0].sessions, 7); }); +test('project census count basis survives discovery, management, and API without changing legacy meanings', (t) => { + // Catalog project roots are native absolute paths, just like discovery output. + const project = path.join(tempDir('ak-census-grouping', t), 'census-project'); + const discovered = discoverProjectSources({ + scanTranscripts: (_root, host) => ({ complete: true, sightings: host === 'claude' + ? [{ cwd: project, weight: 2, sessionOrigin: { origin: 'claude-desktop', evidence: 'declared' } }] + : [{ cwd: project, weight: 3, sessionOrigin: { origin: 'codex-desktop', evidence: 'declared' } }] }), + scanOpencode: () => ({ complete: true, sightings: [] }), + }); + const source = discovered.projects[0]; + assert.equal(source.path, project); + assert.deepEqual(source.sessionOrigins.map(({ countBasis, sessions }) => [countBasis, sessions]), + [['declared-session-ids', 2], ['transcript-files', 3]]); + const page = publicInventoryPage(runInventoryQuery(build({ projects: [], discoveryProjects: [source] }, [project]), + { scope: 'project', presentation: 'focus' })); + assert.deepEqual(page.navigation.nodes[0].sessionOrigins, + [{ origin: 'claude-desktop', sessions: 2, countBasis: 'declared-session-ids' }, + { origin: 'codex-desktop', sessions: 3, countBasis: 'transcript-files' }]); +}); +test('legacy basis and zero recovery keep their exact meaning; invalid basis is omitted', () => { + const project = '/legacy-project'; + const origins = [ + { origin: 'claude-desktop', sessions: 4, countBasis: 'transcript-files' }, + { origin: 'codex-desktop', sessions: 2, countBasis: 'not-a-basis' }, + { origin: 'unknown', sessions: 0, countBasis: 'recovered-project-sighting' }, + ]; + const page = publicInventoryPage(runInventoryQuery(build({ projects: [], discoveryProjects: [ + { path: project, sessionOrigins: origins }, + ] }, [project]), { scope: 'project', presentation: 'focus' })); + assert.deepEqual(page.navigation.nodes[0].sessionOrigins, [ + origins[0], { origin: 'codex-desktop', sessions: 2 }, origins[2], + ]); +}); +test('surface evidence survives public focus and row DTOs and facets exclude zero recovery observations', () => { + const sessionSurfaces = [{ host: 'codex', surface: 'chatgpt-desktop-work', initiator: 'agent', sessions: 2, + countBasis: 'transcript-files', rawEvidence: { originator: ['codex_work_desktop'] } }, + { host: 'claude', surface: 'cloud-session', initiator: 'automation', sessions: 0, countBasis: 'recovered-project-sighting' }]; + const inventory = build({ projects: [], discoveryProjects: [{ path: '/project', repository, sessionSurfaces }] }, ['/project']); + const page = publicInventoryPage(runInventoryQuery(inventory, { scope: 'project', presentation: 'focus', facets: { sessionOrigin: ['chatgpt-desktop-work'] } })); + assert.equal(page.total, 1); + assert.equal(page.navigation.nodes[0].sessionSurfaces[0].host, 'claude'); + assert.deepEqual(page.navigation.nodes[0].sessionSurfaces[1].rawEvidence.originator, ['codex_work_desktop']); + assert.equal(runInventoryQuery(inventory, { scope: 'project', facets: { sessionOrigin: ['cloud-session'] } }).total, 0); +}); + +for (const stateSource of ['saved preference', 'bookmarked hash']) { + test(`legacy desktop filter from ${stateSource} retains its exact membership`, () => { + const context = vm.createContext({ URLSearchParams, localStorage: { getItem: () => null } }); + const source = fs.readFileSync(new URL('../../src/lib/dashboard/client/maintenance-workspace.mjs', import.meta.url), 'utf8') + .replace(/^import\s[\s\S]*?from ['"][^'"]+['"];\s*$/gm, '').replace(/\bexport (?=(?:function|var)\b)/g, ''); + vm.runInContext(source, context); + if (stateSource === 'saved preference') context.mntApplyPreferredState({ lastView: { scope: 'project', facets: { sessionOrigin: ['codex-desktop'] } } }); + else context.mntApplyState(context.mntParseHashParts(['system', 'maintenance', 'inventory?scope=project&facet.sessionOrigin=codex-desktop'])); + const restored = JSON.parse(JSON.stringify(context.MNT)); + assert.deepEqual(restored.facets.sessionOrigin, ['codex-desktop']); + const rows = [{ path: '/desktop', sessionOrigins: [{ origin: 'codex-desktop', sessions: 3 }] }, + { path: '/unknown', sessionOrigins: [{ origin: 'unknown', sessions: 1 }] }]; + const inventory = build({ projects: [], discoveryProjects: rows }, rows.map((row) => row.path)); + const result = publicInventoryPage(runInventoryQuery(inventory, { scope: restored.scope, facets: restored.facets, presentation: 'focus' })); + assert.equal(result.total, 1); + assert.equal(result.navigation.nodes[0].sessionOrigins[0].origin, 'codex-desktop'); + assert.equal(result.navigation.nodes[0].sessionSurfaces, null); + rows[0].sessionSurfaces = [{ host: 'codex', surface: 'chatgpt-desktop-work', sessions: 3 }]; + const refreshed = build({ projects: [], discoveryProjects: rows }, rows.map((row) => row.path)); + assert.equal(runInventoryQuery(refreshed, { scope: restored.scope, facets: restored.facets }).total, 1, + 'refreshing to the richer contract must not invalidate the saved legacy filter'); + }); +} + +for (const stateSource of ['saved preference', 'bookmarked hash']) { + test(`legacy Unknown filter from ${stateSource} retains membership after surface refresh`, () => { + const context = vm.createContext({ URLSearchParams, localStorage: { getItem: () => null } }); + const source = fs.readFileSync(new URL('../../src/lib/dashboard/client/maintenance-workspace.mjs', import.meta.url), 'utf8') + .replace(/^import\s[\s\S]*?from ['"][^'"]+['"];\s*$/gm, '').replace(/\bexport (?=(?:function|var)\b)/g, ''); + vm.runInContext(source, context); + if (stateSource === 'saved preference') context.mntApplyPreferredState({ lastView: { scope: 'project', facets: { sessionOrigin: ['unknown'] } } }); + else context.mntApplyState(context.mntParseHashParts(['system', 'maintenance', 'inventory?scope=project&facet.sessionOrigin=unknown'])); + const restored = JSON.parse(JSON.stringify(context.MNT)); + assert.deepEqual(restored.facets.sessionOrigin, ['unknown']); + const rows = ['codex-cli', 'codex-ide', 'unknown'].map((surface) => ({ path: '/' + surface, + sessionOrigins: [{ origin: 'unknown', sessions: 1 }] })); + rows.push({ path: '/desktop', sessionOrigins: [{ origin: 'codex-desktop', sessions: 1 }] }); + const query = validateMaintenanceV2Query('inventory', new URLSearchParams('scope=project&facet.sessionOrigin=unknown')); + const before = build({ projects: [], discoveryProjects: rows }, rows.map((row) => row.path)); + assert.equal(runInventoryQuery(before, query).total, 3); + rows.forEach((row, i) => { row.sessionSurfaces = [{ host: 'codex', surface: ['codex-cli', 'codex-ide', 'unknown', 'chatgpt-desktop-work'][i], sessions: 1 }]; }); + const after = build({ projects: [], discoveryProjects: rows }, rows.map((row) => row.path)); + assert.equal(runInventoryQuery(after, { scope: restored.scope, facets: restored.facets }).total, 3); + assert.equal(runInventoryQuery(after, query).total, 3); + assert.equal(runInventoryQuery(after, { scope: 'project', facets: { sessionOrigin: ['surface-unknown'] } }).total, 1); + }); +} + +test('implicit legacy Unknown membership also survives richer surface evidence', () => { + const row = { path: '/no-origin' }; + const query = { scope: 'project', facets: { sessionOrigin: ['unknown'] } }; + assert.equal(runInventoryQuery(build({ projects: [], discoveryProjects: [row] }, [row.path]), query).total, 1); + row.sessionSurfaces = [{ host: 'codex', surface: 'codex-cli', sessions: 1 }]; + assert.equal(runInventoryQuery(build({ projects: [], discoveryProjects: [row] }, [row.path]), query).total, 1); +}); diff --git a/tests/kit/maintenance-refresh-injection.test.mjs b/tests/kit/maintenance-refresh-injection.test.mjs new file mode 100644 index 00000000..03a363fe --- /dev/null +++ b/tests/kit/maintenance-refresh-injection.test.mjs @@ -0,0 +1,78 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { spawnSync } from 'node:child_process'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..'); + +// The Maintenance service statically imports footprint/index.mjs, so blocking +// that module's import would reject a valid injected run. Instrument its real +// constructor instead; the management facade is lazy and must never import. +const guard = ` +export async function resolve(specifier, context, nextResolve) { + if (specifier.endsWith('/maintenance/management/service.mjs')) { + throw new Error('real management facade imported'); + } + return nextResolve(specifier, context); +} +export async function load(url, context, nextLoad) { + const loaded = await nextLoad(url, context); + if (!url.endsWith('/footprint/index.mjs')) return loaded; + const needle = ' const collect = {'; + const source = String(loaded.source); + if (source.split(needle).length !== 2) throw new Error('collector constructor guard no longer matches'); + return { ...loaded, source: source.replace(needle, + " throw new Error('real footprint collector constructed');\\n" + needle) }; +} +`; + +const child = ` +import assert from 'node:assert/strict'; +import { register } from 'node:module'; +import { pathToFileURL } from 'node:url'; +register('data:text/javascript,' + encodeURIComponent(process.env.AK_REFRESH_GUARD)); +const { run } = await import(pathToFileURL(process.env.AK_MAINTAIN_MODULE).href); +const calls = []; +const refreshStages = { + maintenance: async () => { calls.push('maintenance'); return { ok: true }; }, + inventory: async () => { calls.push('inventory'); return { ok: true }; }, + local: async () => { calls.push('local'); return { ok: true }; }, +}; +const service = { async report() { calls.push('report'); return { mode: 'read-only' }; } }; +const lines = []; +const originalLog = console.log; +console.log = (...args) => lines.push(args.join(' ')); +let code; +try { + code = await run({ flags: { json: true, refresh: '' }, positionals: ['report'], + deps: { refreshStages, service } }); +} finally { + console.log = originalLog; +} +assert.equal(code, 0, lines.join('\\n')); +assert.deepEqual(calls, ['maintenance', 'inventory', 'local', 'report']); +const result = JSON.parse(lines.join('\\n')); +assert.equal(result.mode, 'read-only'); +assert.equal(result.refresh.ok, true); +assert.deepEqual(result.refresh.stages.map(({ id }) => id), ['maintenance', 'inventory', 'local']); +originalLog('injected refresh used no real constructors'); +`; + +test('injected maintain refresh constructs no real collector or management facade', (t) => { + const home = tempDir('ak-maintain-injected-home', t); + const project = tempDir('ak-maintain-injected-project', t); + const result = spawnSync(process.execPath, ['--input-type=module', '--eval', child], { + cwd: project, + env: spawnEnv(home, { + AK_MAINTAIN_MODULE: path.join(ROOT, 'src/commands/maintain.mjs'), + AK_REFRESH_GUARD: guard, + }), + encoding: 'utf8', + }); + assert.ifError(result.error); + assert.equal(result.status, 0, `stdout: ${result.stdout}\nstderr: ${result.stderr}`); + assert.match(result.stdout, /injected refresh used no real constructors/); +}); diff --git a/tests/kit/mcp-scopes.test.mjs b/tests/kit/mcp-scopes.test.mjs index 9acbaef8..b8720b9a 100644 --- a/tests/kit/mcp-scopes.test.mjs +++ b/tests/kit/mcp-scopes.test.mjs @@ -357,7 +357,7 @@ test('status names the store the launcher picks from this folder', (t) => { assert.match(outside.message, /from here it uses the user-level store .*\.claude-flow.memory \(this folder is the home folder\)/); }); -// Plan Task 4.2: register() refuses with 'ak-not-on-path' when it would write +// register() refuses with 'ak-not-on-path' when it would write // a registration that starts `ak`, so every row whose sync fix goes through // that write is the user's step until `ak` resolves on PATH. const AK_OFF_PATH_FIX = 'put `ak` on PATH, then run `ak sync`'; @@ -400,10 +400,10 @@ test('the mcp section looks ak up only when ak manages the registration', async assert.equal(rows.find((r) => /not registered/.test(r.message)).fix, AK_OFF_PATH_FIX); }); -// Task 7: the mcp section's collect() threads refresh/record into +// the mcp section's collect() threads refresh/record into // launcherCheck() (claudeLauncherUnavailable, evidence-gated) so a plain // `ak status` (refresh: false) stays cache-first while `--refresh` forces it, -// mirroring hosts.mjs's Task 5 precedent. +// mirroring hosts.mjs's evidence-cache precedent. test('the mcp section threads refresh/record into launcherCheck, defaulting to a plain-status-shaped call', async (t) => { const { home, cwd } = fixture(t); const seen = []; @@ -421,7 +421,7 @@ test('the mcp section threads refresh/record into launcherCheck, defaulting to a 'explicit refresh/record/source thread straight through'); }); -// F1 (Branch 3 fix round 2): an `ak` on PATH older than the launcher's +// F1: an `ak` on PATH older than the launcher's // `--host` option cannot start `ak x ruflo-mcp --host claude` (it exits 2 on // the unknown option), so being on PATH is not enough. The check asks the PATH // `ak` for its launcher help; this is ak 4.0.0-alpha.56's, which has no --host. diff --git a/tests/kit/models-command.test.mjs b/tests/kit/models-command.test.mjs index b4774fd4..fc815f6f 100644 --- a/tests/kit/models-command.test.mjs +++ b/tests/kit/models-command.test.mjs @@ -40,6 +40,15 @@ test('models status is a cache-only read', async () => { assert.equal(JSON.parse(result.output).inventory.snapshotId, 'models:test'); }); +test('models rejects an unknown verb with a populated snapshot store', async () => { + const result = await capture(() => run({ + flags: { json: true }, positionals: ['bogus'], + deps: { loadConfig: () => cfg, readStore: () => store }, + })); + assert.equal(result.code, 2); + assert.match(result.output, /usage: ak models status\|refresh\|diff\|explain\|plan/); +}); + test('models status host filter rejects unknown owners and narrows evidence', async () => { const invalid = await capture(() => run({ flags: { json: true, host: 'unknown' }, positionals: ['status'], diff --git a/tests/kit/natives-runtime.test.mjs b/tests/kit/natives-runtime.test.mjs index 9581d9f2..b191a1c2 100644 --- a/tests/kit/natives-runtime.test.mjs +++ b/tests/kit/natives-runtime.test.mjs @@ -16,7 +16,7 @@ import { _setGlobalRootForTest } from '../../src/lib/paths.mjs'; import { evidenceDir, readEvidence, writeEvidence, stableInputsKey } from '../../src/lib/evidence.mjs'; import { tempDir } from './helpers/temp-dir.mjs'; -// Task 4: rufloRuntimeNatives now reads/writes evidence under the kit state dir +// rufloRuntimeNatives now reads/writes evidence under the kit state dir // (evidenceDir()) when refresh:false. Redirect the state base for this whole // file so those reads/writes never touch this machine's real evidence store — // mirrors tests/kit/heal-natives.test.mjs's XDG_STATE_HOME/LOCALAPPDATA redirect. @@ -386,7 +386,7 @@ test('rufloRuntimeNatives reports not-installed and never spawns when ruflo is a _setGlobalRootForTest(null); rm(g); }); -// ── Task 4: evidence-backed refresh (Ruling A/B) ──────────────────────────── +// ── evidence-backed refresh (Ruling A/B) ──────────────────────────── /** Seed a fresh, matching evidence record for one context, as a real probe + * writeEvidence would leave it. */ diff --git a/tests/kit/paths-global-root-evidence.test.mjs b/tests/kit/paths-global-root-evidence.test.mjs index 57d2c321..7294d627 100644 --- a/tests/kit/paths-global-root-evidence.test.mjs +++ b/tests/kit/paths-global-root-evidence.test.mjs @@ -1,4 +1,4 @@ -// Branch 6a Task 7: globalRoot() (src/lib/paths.mjs) stops spawning +// globalRoot() (src/lib/paths.mjs) stops spawning // `npm root -g` on every plain `ak status` call — it reuses fresh evidence // instead (kind 'npm-global-root', id 'machine', 24h TTL). Unlike every other // evidence-gated check in this branch, BOTH `refresh` and `record` DEFAULT TO @@ -16,7 +16,7 @@ // byte-compatible, not just paths.mjs talking to itself. // // globalRoot() has no injectable runner (it calls execFileSync directly), so -// — mirroring tests/kit/host-setup-evidence.test.mjs's Task 5 precedent — +// — mirroring tests/kit/host-setup-evidence.test.mjs's state-isolation precedent — // these tests break PATH so a real `npm root -g` ENOENTs deterministically, // then seed cached evidence with a marker path no real probe/fallback could // ever produce. Getting the marker back proves the cache was used. diff --git a/tests/kit/persisted-file-identity.test.mjs b/tests/kit/persisted-file-identity.test.mjs new file mode 100644 index 00000000..087f80f8 --- /dev/null +++ b/tests/kit/persisted-file-identity.test.mjs @@ -0,0 +1,229 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; + +import { tempDir } from './helpers/temp-dir.mjs'; +import { planPartitions, partitionDrifted } from '../../src/lib/maintenance/discovery/partitions.mjs'; +import { inspectHookTarget, atomicReplaceHookTarget } from '../../src/lib/hook-remediation/fs-port.mjs'; +import { JsonlTailer } from '../../src/lib/live/jsonl-tailer.mjs'; +import { createHostHealthSnapshot } from '../../src/lib/host-health-evidence.mjs'; +import { HOOK_HEAL_RECEIPT_SCHEMA, readHookReceipt, sealHookReceipt } from '../../src/lib/hook-remediation/store.mjs'; +import { inspectHostAlignment } from '../../src/lib/host-alignment.mjs'; +import { TranscriptStreams } from '../../src/lib/live/transcript-streams.mjs'; + +const A = 9007199254740992n; +const B = 9007199254740993n; + +test('Discovery stamp survives JSON with adjacent 64-bit inodes and fractional mtime', (t) => { + const root = tempDir('ak-partition-id', t); + const lstatSync = fs.lstatSync; + let ino = A; + const fsImpl = { ...fs, lstatSync(target, options) { + const stat = lstatSync(target, options); + stat.ino = options?.bigint ? ino : Number(ino); + return stat; + } }; + const stamp = planPartitions(root, { fsImpl }).partitions[0].sourceStamps[0]; + assert.equal(stamp.ino, '9007199254740992'); + assert.equal(typeof stamp.mtimeMs, 'number'); + assert.equal(stamp.mtimeMs, lstatSync(root).mtimeMs); + const restored = JSON.parse(JSON.stringify(stamp)); + assert.equal(partitionDrifted({ sourceStamps: [{ ...restored, ino: Number(A) }] }, { fsImpl }), true, + 'unsafe legacy number cannot authorize a match'); + ino = B; + assert.equal(partitionDrifted({ sourceStamps: [restored] }, { fsImpl }), true); + for (const malformed of ['-1', '01', -1]) { + assert.equal(partitionDrifted({ sourceStamps: [{ ...restored, ino: malformed }] }, { fsImpl }), true); + } + ino = 42n; + const safeStamp = planPartitions(root, { fsImpl }).partitions[0].sourceStamps[0]; + assert.equal(partitionDrifted({ sourceStamps: [{ ...safeStamp, ino: 42 }] }, { fsImpl }), false); +}); + +test('hook target parent identity survives JSON and refuses adjacent replacement', (t) => { + const root = tempDir('ak-hook-parent-id', t); + const file = path.join(root, 'hook.json'); + fs.writeFileSync(file, '{}'); + const statSync = fs.statSync; + let ino = A; + const fsImpl = { ...fs, statSync(target, options) { + const stat = statSync(target, options); + if (target === root) stat.ino = options?.bigint ? ino : Number(ino); + return stat; + } }; + // Exercise the parent guard on every OS before any mutation is allowed. + // This injected branch is not evidence of native Windows atomic replacement. + fsImpl.openSync = (target, flags) => { + assert.equal(flags, fs.constants.O_RDONLY | (fs.constants.O_NOFOLLOW ?? 0)); + return fs.openSync(target, flags); + }; + fsImpl.renameSync = () => assert.fail('parent drift must refuse before rename'); + const options = { fsImpl, platform: 'linux' }; + const snapshot = inspectHookTarget(file, root, options); + assert.equal(snapshot.parent.ino, '9007199254740992'); + const savedParent = JSON.parse(JSON.stringify(snapshot.parent)); + ino = B; + assert.throws(() => atomicReplaceHookTarget({ ...snapshot, parent: savedParent }, Buffer.from('[]'), undefined, options), + /target parent changed/); + assert.equal(fs.readFileSync(file, 'utf8'), '{}'); +}); + +test('Windows hook mutation refuses before accessing the filesystem', (t) => { + const root = tempDir('ak-hook-windows-refusal', t); + const file = path.join(root, 'hook.json'); + fs.writeFileSync(file, '{}'); + const snapshot = inspectHookTarget(file, root, { platform: 'win32' }); + const fsImpl = { lstatSync: () => assert.fail('unsupported mutation must refuse before inspection') }; + assert.throws(() => atomicReplaceHookTarget(snapshot, Buffer.from('[]'), undefined, { fsImpl, platform: 'win32' }), + /hook mutation is unsupported on Windows until replace-existing atomicity is proven/); + assert.equal(fs.readFileSync(file, 'utf8'), '{}'); + assert.deepEqual(fs.readdirSync(root), ['hook.json']); +}); + +test('hook snapshot gets identity and fractional mtime from one descriptor stat', (t) => { + const root = tempDir('ak-hook-one-stat', t); + const file = path.join(root, 'hook.json'); + fs.writeFileSync(file, '{}'); + const fstatSync = fs.fstatSync; + let calls = 0; + const fsImpl = { ...fs, fstatSync(fd, options) { + calls++; + assert.equal(options?.bigint, true); + const stat = fstatSync(fd, options); + stat.ino = A; + return stat; + }, lstatSync(target, options) { + const stat = fs.lstatSync(target, options); + if (target === file) stat.ino = A; + return stat; + } }; + const snapshot = inspectHookTarget(file, root, { fsImpl }); + assert.equal(calls, 1); + assert.equal(snapshot.mtimeMs, fs.statSync(file).mtimeMs); +}); + +test('JSONL replacement with a colliding Number inode resets offset and emits new records', (t) => { + const root = tempDir('ak-tailer-id', t); + const file = path.join(root, 'events.jsonl'); + fs.writeFileSync(file, '{"a":1}\n'); + const statSync = fs.statSync; + let ino = A; + t.mock.method(fs, 'statSync', (target, options) => { + const stat = statSync(target, options); + if (target === file) stat.ino = options?.bigint ? ino : Number(ino); + return stat; + }); + const records = []; + const tailer = new JsonlTailer(file, { onRecord: record => records.push(record) }); + tailer.reconcile(); + fs.writeFileSync(file, '{"b":2}\n'); + ino = B; + tailer.reconcile(); + assert.deepEqual(records, [{ a: 1 }, { b: 2 }]); +}); + +test('host health evidence changes when adjacent 64-bit launcher inode changes', (t) => { + const root = tempDir('ak-health-id', t); + const launcher = path.join(root, process.platform === 'win32' ? 'codex.cmd' : 'codex'); + fs.writeFileSync(launcher, 'launcher'); + const observedIds = []; + const statSync = fs.statSync; + let ino = A; + t.mock.method(fs, 'statSync', (target, options) => { + const stat = statSync(target, options); + if (target === launcher) { + assert.equal(options?.bigint, true); + stat.ino = ino; + observedIds.push(ino); + } + return stat; + }); + const snapshot = createHostHealthSnapshot({ env: { PATH: root, PATHEXT: '.CMD' }, inputPaths: () => [] }); + const before = snapshot({ cwd: root, cfg: {} }).key; + assert.deepEqual(observedIds, [A], 'the fixture must be the launcher fingerprinted by this platform'); + ino = B; + assert.notEqual(snapshot({ cwd: root, cfg: {} }).key, before); + assert.deepEqual(observedIds, [A, B]); +}); + +test('hook receipt accepts exact parent IDs but refuses unsafe legacy Numbers', (t) => { + const root = tempDir('ak-receipt-id', t); + fs.chmodSync(root, 0o700); + const id = 'tx-2026-09-29T00-00-00.000Z-0123456789abcdef'; + const dir = path.join(root, id); + fs.mkdirSync(dir, { mode: 0o700 }); + const file = path.join(dir, 'receipt.json'); + const digest = 'a'.repeat(64); + const receipt = { + schemaVersion: HOOK_HEAL_RECEIPT_SCHEMA, id, createdAt: new Date().toISOString(), + status: 'prepared', planDigest: digest, auditId: 'audit', + authorization: { mechanism: 'explicit-action-selection', actionIds: ['one'], trustMutationAuthorized: false }, + actions: [{ + id: 'one', host: 'codex', hostVersion: '1', recipeId: 'recipe', profileId: 'profile', + target: path.join(root, 'hook'), containmentRoot: root, classification: 'safe-automatic', state: 'prepared', + preimage: { sha256: digest, size: 2, mode: 0o600, modeSupported: true, + uid: null, gid: null, specialMode: 0, parent: { realPath: root, dev: '1', ino: '9007199254740993' } }, + postimage: { sha256: digest, size: 2, mode: 0o600, modeSupported: true }, + backup: { relative: path.join('backups', '0000.bin'), sha256: digest, size: 2 }, + }], + }; + // Seed sealed on-disk inputs for reader validation. Durable receipt writes + // have a separate contract and require native directory fsync support. + const seedReceipt = () => fs.writeFileSync(file, JSON.stringify(sealHookReceipt(receipt))); + for (const ino of [String(A), String(B)]) { + receipt.actions[0].preimage.parent.ino = ino; + seedReceipt(); + assert.equal(readHookReceipt(root, id).receipt.actions[0].preimage.parent.ino, ino); + } + receipt.actions[0].preimage.parent.ino = Number(B); + seedReceipt(); + assert.throws(() => readHookReceipt(root, id), /receipt action image is invalid/); + for (const malformed of ['-1', '01', -1]) { + receipt.actions[0].preimage.parent.ino = malformed; + seedReceipt(); + assert.throws(() => readHookReceipt(root, id), /receipt action image is invalid/); + } + receipt.actions[0].preimage.parent.ino = 42; + seedReceipt(); + assert.equal(readHookReceipt(root, id).receipt.actions[0].preimage.parent.ino, 42); +}); + +test('host alignment snapshot serializes adjacent file IDs distinctly', (t) => { + const root = tempDir('ak-align-id', t); + const file = path.join(root, '.mcp.json'); + fs.writeFileSync(file, '{}'); + const lstatSync = fs.lstatSync; + let ino = A; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = lstatSync(target, options); + if (target === file) stat.ino = options?.bigint ? ino : Number(ino); + return stat; + }); + const options = { home: root, codexHome: path.join(root, '.codex'), projectRoots: [root] }; + const before = inspectHostAlignment(options); + assert.equal(before.snapshots.find(s => s.file === file).identity.inode, '9007199254740992'); + ino = B; + assert.notEqual(inspectHostAlignment(options).digest, before.digest); +}); + +test('transcript replay cursor distinguishes adjacent 64-bit file epochs', (t) => { + const root = tempDir('ak-transcript-id', t); + const file = path.join(root, 'session-1.jsonl'); + fs.writeFileSync(file, ''); + const statSync = fs.statSync; + let ino = A; + t.mock.method(fs, 'statSync', (target, options) => { + const stat = statSync(target, options); + if (target === file) stat.ino = options?.bigint ? ino : Number(ino); + return stat; + }); + const first = new TranscriptStreams({ roots: { claude: root }, mask: value => value }); + const before = first.open('claude', 'session-1').snapshot().cursor; + first.close(); + ino = B; + const second = new TranscriptStreams({ roots: { claude: root }, mask: value => value }); + const after = second.open('claude', 'session-1').snapshot().cursor; + second.close(); + assert.notEqual(before, after); +}); diff --git a/tests/kit/project-memory.test.mjs b/tests/kit/project-memory.test.mjs index 7579dbba..30b1cb80 100644 --- a/tests/kit/project-memory.test.mjs +++ b/tests/kit/project-memory.test.mjs @@ -274,7 +274,7 @@ test('the stray search is bounded and says when it stopped early', (t) => { assert.equal(capped.complete, false); assert.equal(capped.visited, 3); const deep = findStrayMemoryStores(root); - assert.equal(deep.complete, true); + assert.equal(deep.complete, false, 'a depth cutoff leaves descendants unchecked'); assert.deepEqual(deep.strays, [], 'folders deeper than the depth bound are not searched'); assert.deepEqual(findStrayMemoryStores(path.join(root, 'missing')), { strays: [], complete: true, visited: 0, nestedRepositories: [] }); }); @@ -300,3 +300,63 @@ test('the stray search stops at a nested repository or an in-checkout worktree: assert.deepEqual(strays.map((stray) => `${stray.kind} ${stray.path}`), ['aqe docs/.agentic-qe']); assert.deepEqual(nestedRepositories, ['.tools', 'packages/api', 'wt/feature']); }); + +test('the stray search finds AQE below ordinary dot folders without entering tool homes or linked folders', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-stray-dot-')); + const outside = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-stray-outside-')); + t.after(() => { fs.rmSync(root, { recursive: true, force: true }); fs.rmSync(outside, { recursive: true, force: true }); }); + fs.mkdirSync(path.join(root, '.superpowers', 'sdd', 'program', 'reports', '.agentic-qe'), { recursive: true }); + fs.mkdirSync(path.join(root, '.notes', 'drafts', '.agentic-qe'), { recursive: true }); + for (const dir of ['.git', '.claude', '.codex', '.agentic-qe', '.swarm', 'node_modules']) { + fs.mkdirSync(path.join(root, dir, 'nested', '.agentic-qe'), { recursive: true }); + } + fs.mkdirSync(path.join(outside, '.agentic-qe')); + fs.symlinkSync(outside, path.join(root, '.notes', 'linked')); + fs.symlinkSync(root, path.join(root, '.notes', 'loop')); + fs.mkdirSync(path.join(root, '.other-repo', '.agentic-qe'), { recursive: true }); + fs.writeFileSync(path.join(root, '.other-repo', '.git'), 'gitdir: elsewhere\n'); + + const result = findStrayMemoryStores(root); + assert.equal(result.complete, true); + assert.deepEqual(result.strays.map((stray) => stray.path), [ + '.notes/drafts/.agentic-qe', + '.superpowers/sdd/program/reports/.agentic-qe', + ]); + assert.deepEqual(result.nestedRepositories, ['.other-repo']); +}); + +test('depth-limited dot subtrees do not claim complete stray coverage', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-stray-limit-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + fs.mkdirSync(path.join(root, '.notes', 'a', 'b', 'c', 'd', '.agentic-qe'), { recursive: true }); + const result = findStrayMemoryStores(root); + assert.equal(result.complete, false); + assert.deepEqual(result.strays, []); +}); + +test('dependency root markers are excluded before stray inspection', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-stray-dependency-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + fs.mkdirSync(path.join(root, 'node_modules', '.agentic-qe'), { recursive: true }); + touch(root, 'node_modules/.swarm/memory.db'); + touch(root, 'node_modules/.swarm/agentdb-memory.db'); + const result = findStrayMemoryStores(root); + assert.equal(result.complete, true); + assert.deepEqual(result.strays, []); +}); + +test('a listed dot subtree with denied marker metadata reports incomplete coverage', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-stray-metadata-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + fs.mkdirSync(path.join(root, '.notes', '.agentic-qe'), { recursive: true }); + const denied = path.join(root, '.notes', '.agentic-qe'); + const original = fs.lstatSync; + fs.lstatSync = (file, ...args) => { + if (file === denied) throw Object.assign(new Error('permission denied'), { code: 'EACCES' }); + return original(file, ...args); + }; + let result; + try { result = findStrayMemoryStores(root); } finally { fs.lstatSync = original; } + assert.equal(result.complete, false); + assert.deepEqual(result.strays, []); +}); diff --git a/tests/kit/project-sources-claude-sessions.test.mjs b/tests/kit/project-sources-claude-sessions.test.mjs new file mode 100644 index 00000000..ea1cb034 --- /dev/null +++ b/tests/kit/project-sources-claude-sessions.test.mjs @@ -0,0 +1,107 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { scanTranscriptCwds, discoverProjectSources } from '../../src/lib/footprint/project-sources.mjs'; + +function write(root, group, name, records) { + const file = path.join(root, group, name); + fs.mkdirSync(path.dirname(file), { recursive: true }); + fs.writeFileSync(file, `${records.map((record) => JSON.stringify(record)).join('\n')}\n`); +} + +test('Claude project census counts declared sessions and excludes bridge and subagent files', () => { + const root = tempDir('ak-claude-census'); + const claudeRoot = path.join(root, 'claude'); + const a = path.join(root, 'a'); + const b = path.join(root, 'b'); + fs.mkdirSync(a); fs.mkdirSync(b); + write(claudeRoot, 'a', '1.jsonl', [{ type: 'user', sessionId: 'one', cwd: a }]); + write(claudeRoot, 'a', '2.jsonl', [{ type: 'assistant', sessionId: 'two', cwd: a }]); + write(claudeRoot, 'a', '3.jsonl', [{ type: 'user', sessionId: 'one', cwd: b }]); + write(claudeRoot, 'a', '4.jsonl', [{ type: 'user', cwd: a }]); + write(claudeRoot, 'a', '5.jsonl', [{ type: 'user', sessionId: 12, cwd: a }]); + write(claudeRoot, 'a', '6.jsonl', [{ type: 'bridge-session', sessionId: 'bridge', cwd: b }]); + write(claudeRoot, 'a', '7.jsonl', [{ type: 'cost-state', sessionId: 'cost', cwd: b }]); + write(claudeRoot, 'a/one/subagents', 'agent-a.jsonl', [ + { type: 'assistant', sessionId: 'child', cwd: b, isSidechain: true }, + ]); + const scan = scanTranscriptCwds(claudeRoot, 'claude'); + assert.deepEqual({ files: scan.files, sessions: scan.sessions, duplicate: scan.duplicateSessionFiles, + subagents: scan.subagentExcluded, nonConversation: scan.nonConversationExcluded, + unknown: scan.unknownSessionFiles }, + { files: 8, sessions: 2, duplicate: 1, subagents: 1, nonConversation: 2, unknown: 2 }); + const result = discoverProjectSources({ claudeRoot, codexRoot: path.join(root, 'no-codex'), + scanOpencode: () => ({ sightings: [], complete: true }) }); + assert.equal(result.projects.length, 1); + assert.equal(result.projects[0].sessions, 2); + assert.equal(result.projects[0].sessionOrigins[0].countBasis, 'declared-session-ids'); +}); + +test('an incomplete head with cwd preserves project evidence but does not invent a session', () => { + const root = tempDir('ak-claude-head'); + const claudeRoot = path.join(root, 'claude'); + write(claudeRoot, 'a', 'partial.jsonl', [ + { type: 'system', cwd: root, sessionId: 'later' }, + { type: 'user', cwd: root, sessionId: 'later' }, + ]); + const scan = scanTranscriptCwds(claudeRoot, 'claude', { maxLines: 1 }); + assert.equal(scan.sessions, 0); + assert.equal(scan.unknownSessionFiles, 1); + assert.equal(scan.sightings[0].weight, 0); + assert.equal(scan.withCwd, 1); +}); + +test('excluded-only folder cannot become a project through encoded directory recovery', () => { + const root = tempDir('ak-claude-excluded'); + const claudeRoot = path.join(root, 'claude'); + write(claudeRoot, 'bridge-only', 'bridge.jsonl', [{ type: 'bridge-session', sessionId: 'bridge' }]); + write(claudeRoot, 'bridge-only/parent/subagents', 'agent-a.jsonl', + [{ type: 'assistant', sessionId: 'child', isSidechain: true }]); + const scan = scanTranscriptCwds(claudeRoot, 'claude', { decodeDir: () => root }); + assert.equal(scan.recoveredFromDirName, 0); + assert.equal(scan.unresolved, 0); + assert.deepEqual(scan.sightings, []); +}); + +test('encoded directory recovery is project evidence with zero verified sessions', () => { + const root = tempDir('ak-claude-recovery'); + const claudeRoot = path.join(root, 'claude'); + write(claudeRoot, 'legacy', 'unknown.jsonl', [{ type: 'user' }]); + const scan = scanTranscriptCwds(claudeRoot, 'claude', { decodeDir: () => root }); + assert.equal(scan.sessions, 0); + assert.equal(scan.unknownSessionFiles, 1); + assert.deepEqual(scan.sightings.map(({ origin, weight }) => [origin, weight]), [['encoded-dir', 0]]); + const discovered = discoverProjectSources({ claudeRoot, codexRoot: path.join(root, 'none'), + scanTranscripts: (source, host, options) => scanTranscriptCwds(source, host, { + ...options, decodeDir: host === 'claude' ? () => root : null, + }), scanOpencode: () => ({ sightings: [], complete: true }) }); + assert.equal(discovered.everSeen, 1); + assert.equal(discovered.projects[0].sessions, 0); + assert.equal(discovered.projects[0].sessionOrigins[0].sessions, 0); +}); + +test('bridge marker in a bounded head cannot exclude a later conversation', () => { + const root = tempDir('ak-claude-bridge-head'); + const claudeRoot = path.join(root, 'claude'); + write(claudeRoot, 'a', 'bridge.jsonl', [ + { type: 'bridge-session', sessionId: 'one', cwd: root }, + { type: 'user', sessionId: 'one', cwd: root }, + ]); + const scan = scanTranscriptCwds(claudeRoot, 'claude', { maxLines: 1 }); + assert.equal(scan.nonConversationExcluded, 0); + assert.equal(scan.unknownSessionFiles, 1); + assert.equal(scan.sessionCountComplete, false); + assert.equal(scan.sightings[0].weight, 0); +}); + +test('unreadable root cannot claim a complete zero-session count', () => { + const scan = scanTranscriptCwds('/unreadable', 'claude', { + walk: () => ({ status: 'unknown', reason: 'EACCES', complete: false }), + }); + assert.equal(scan.status, 'degraded'); + assert.equal(scan.sessions, 0); + assert.equal(scan.sessionCountComplete, false); + assert.equal(scan.complete, false); +}); diff --git a/tests/kit/providers-status-section.test.mjs b/tests/kit/providers-status-section.test.mjs index 8f98fc3b..1498b0af 100644 --- a/tests/kit/providers-status-section.test.mjs +++ b/tests/kit/providers-status-section.test.mjs @@ -1,7 +1,7 @@ -// Branch 6a Task 7: the providers section's hostManagementRows() stops +// the providers section's hostManagementRows() stops // calling have(h.bin) for every non-enabled host — providers.mjs's -// detectHosts() (via collectIntegrationFacts, already evidence-cached by -// Task 5) already answers the exact same "is this host's bin on PATH" +// detectHosts() (via collectIntegrationFacts, already backed by the +// evidence cache) already answers the exact same "is this host's bin on PATH" // question once per collect(), for every host regardless of enablement, so // reusing integrationFacts.hosts[h.id].present removes a duplicate probe of // the identical fact rather than adding a second evidence kind for it. diff --git a/tests/kit/quota-codex-presence.test.mjs b/tests/kit/quota-codex-presence.test.mjs index 133aef1a..75a319ff 100644 --- a/tests/kit/quota-codex-presence.test.mjs +++ b/tests/kit/quota-codex-presence.test.mjs @@ -63,7 +63,8 @@ function spawnSpy() { } const callLimits = (extra = {}) => readLimits({ - now: NOW, claudeFile, claudeSettingsFile, codexCacheFile: cacheFile, ...extra, + now: NOW, claudeFile, claudeSettingsFile, claudeManagedSettingsFile: null, + codexCacheFile: cacheFile, ...extra, }); test('found Codex, stale cache: one app-server call', async () => { diff --git a/tests/kit/quota.test.mjs b/tests/kit/quota.test.mjs index 965c38b7..3a837519 100644 --- a/tests/kit/quota.test.mjs +++ b/tests/kit/quota.test.mjs @@ -8,13 +8,18 @@ import path from 'node:path'; import { EventEmitter } from 'node:events'; import { windowLabel, normalizeClaudeLimits, normalizeCodexLimits, readClaudeLimits, - collectCodexLimits, CODEX_TTL_MS, unsupportedQuotaHosts, readLimits, - classifyClaudeTeeChannel, CLAUDE_TEE_CHANNELS, + collectCodexLimits, CODEX_TTL_MS, unsupportedQuotaHosts, readLimits as rawReadLimits, + classifyClaudeTeeChannel as rawClassifyClaudeTeeChannel, CLAUDE_TEE_CHANNELS, collectCodexLimitsDetailed, CODEX_UNAVAILABLE_REASONS, } from '../../src/lib/quota.mjs'; import { tempDir } from './helpers/temp-dir.mjs'; +import { claudeManagedSettingsPath } from '../../src/lib/paths.mjs'; const tmp = () => tempDir('ak-quota'); +const classifyClaudeTeeChannel = (options) => rawClassifyClaudeTeeChannel({ + managedSettingsFile: null, ...options, +}); +const readLimits = (options) => rawReadLimits({ claudeManagedSettingsFile: null, ...options }); // ── windowLabel — duration-derived, never slot-derived ─────────────────────── @@ -104,6 +109,206 @@ function teeFixture({ statusLine, raw, scripts = {} } = {}) { } const cmd = (command) => ({ type: 'command', command }); +test('managed statusLine overrides a user footer with a custom command', () => { + const fx = teeFixture({ statusLine: cmd('node ~/.claude/helpers/statusline.cjs'), + scripts: { '.claude/helpers/statusline.cjs': FOOTER_SCRIPT } }); + const managedSettingsFile = path.join(fx.home, 'managed-settings.json'); + fs.writeFileSync(managedSettingsFile, JSON.stringify({ statusLine: cmd('echo managed') })); + assert.equal(classifyClaudeTeeChannel({ + settingsFile: fx.settingsFile, managedSettingsFile, home: fx.home, + }), 'custom'); +}); + +test('managed footer overrides a custom user statusLine', () => { + const fx = teeFixture({ statusLine: cmd('echo user'), + scripts: { '.claude/helpers/statusline.cjs': FOOTER_SCRIPT } }); + const managedSettingsFile = path.join(fx.home, 'managed-settings.json'); + fs.writeFileSync(managedSettingsFile, JSON.stringify({ + statusLine: cmd('node ~/.claude/helpers/statusline.cjs'), + })); + assert.equal(classifyClaudeTeeChannel({ + settingsFile: fx.settingsFile, managedSettingsFile, home: fx.home, + }), 'kit-footer'); +}); + +test('missing managed file or unrelated managed keys preserve the user statusLine', () => { + const fx = teeFixture({ statusLine: cmd('echo user') }); + const managedSettingsFile = path.join(fx.home, 'managed-settings.json'); + assert.equal(classifyClaudeTeeChannel({ + settingsFile: fx.settingsFile, managedSettingsFile, home: fx.home, + }), 'custom'); + fs.writeFileSync(managedSettingsFile, JSON.stringify({ model: 'synthetic' })); + assert.equal(classifyClaudeTeeChannel({ + settingsFile: fx.settingsFile, managedSettingsFile, home: fx.home, + }), 'custom'); +}); + +test('invalid or unreadable managed file stays unknown instead of using the user footer', () => { + const fx = teeFixture({ statusLine: cmd('node ~/.claude/helpers/statusline.cjs'), + scripts: { '.claude/helpers/statusline.cjs': FOOTER_SCRIPT } }); + const managedSettingsFile = path.join(fx.home, 'managed-settings.json'); + for (const body of ['{broken', '[]', JSON.stringify({ statusLine: {} }), + JSON.stringify({ statusLine: 'invalid' })]) { + fs.writeFileSync(managedSettingsFile, body); + assert.equal(classifyClaudeTeeChannel({ + settingsFile: fx.settingsFile, managedSettingsFile, home: fx.home, + }), 'unknown'); + } + fs.rmSync(managedSettingsFile); + fs.mkdirSync(managedSettingsFile); + assert.equal(classifyClaudeTeeChannel({ + settingsFile: fx.settingsFile, managedSettingsFile, home: fx.home, + }), 'unknown'); +}); + +test('managed null or false statusLine is unknown and cannot inherit the user footer', () => { + const fx = teeFixture({ statusLine: cmd('node ~/.claude/helpers/statusline.cjs'), + scripts: { '.claude/helpers/statusline.cjs': FOOTER_SCRIPT } }); + const managedSettingsFile = path.join(fx.home, 'managed-settings.json'); + for (const statusLine of [null, false]) { + fs.writeFileSync(managedSettingsFile, JSON.stringify({ statusLine })); + assert.equal(classifyClaudeTeeChannel({ + settingsFile: fx.settingsFile, managedSettingsFile, home: fx.home, + }), 'unknown'); + } +}); + +test('managed default path selection is platform-specific and unsupported platforms skip it', () => { + const fx = teeFixture({ statusLine: cmd('echo user') }); + for (const platform of ['darwin', 'linux', 'win32']) { + const expected = claudeManagedSettingsPath(platform); + const reads = []; + const fsImpl = { readFileSync(file, encoding) { + reads.push([file, encoding]); + if (file === expected) return JSON.stringify({ statusLine: cmd('echo managed') }); + if (file === fx.settingsFile) return JSON.stringify({ statusLine: cmd('echo user') }); + throw Object.assign(new Error('unexpected read'), { code: 'ENOENT' }); + } }; + assert.equal(rawClassifyClaudeTeeChannel({ + settingsFile: fx.settingsFile, platform, fsImpl, home: fx.home, + }), 'custom'); + assert.deepEqual(reads, [[expected, 'utf8']]); + } + const reads = []; + const fsImpl = { readFileSync(file) { reads.push(file); return JSON.stringify({ statusLine: cmd('echo user') }); } }; + assert.equal(rawClassifyClaudeTeeChannel({ + settingsFile: fx.settingsFile, platform: 'unsupported', fsImpl, home: fx.home, + }), 'custom'); + assert.deepEqual(reads, [fx.settingsFile]); +}); + +test('a shell chain is custom without reading or running its footer script', () => { + const settingsFile = '/synthetic/user.json'; + const managedSettingsFile = '/synthetic/managed.json'; + const script = '/synthetic/footer.cjs'; + const reads = []; + const fsImpl = { + readFileSync(file) { + reads.push(file); + if (file === managedSettingsFile) return JSON.stringify({ + statusLine: cmd(`node ${script} && echo should-not-run`), + }); + if (file === script) return FOOTER_SCRIPT; + throw new Error('unexpected read'); + }, + statSync(file) { + assert.equal(file, script); + return { isFile: () => true, size: FOOTER_SCRIPT.length }; + }, + }; + assert.equal(rawClassifyClaudeTeeChannel({ + settingsFile, managedSettingsFile, fsImpl, home: '/synthetic', + }), 'custom'); + assert.deepEqual(reads, [managedSettingsFile]); +}); + +test('shell wrappers around a footer helper are custom', () => { + const fx = teeFixture({ scripts: { '.claude/helpers/statusline.cjs': FOOTER_SCRIPT } }); + const script = path.join(fx.home, '.claude/helpers/statusline.cjs'); + const wrapped = [ + `sh -c 'node "${script}"'`, + `bash -c 'node "${script}"'`, + `zsh -c 'node "${script}"'`, + `cmd /c node "${script}"`, + `powershell -Command "node '${script}'"`, + 'sh -c \'node "$D/.claude/helpers/statusline.cjs"\'', + ]; + for (const command of wrapped) { + fs.writeFileSync(fx.settingsFile, JSON.stringify({ statusLine: cmd(command) })); + assert.equal(classifyClaudeTeeChannel({ settingsFile: fx.settingsFile, home: fx.home }), 'custom', command); + } +}); + +test('a JavaScript path in an argument or inline program does not prove the footer runs', () => { + const fx = teeFixture({ scripts: { '.claude/helpers/statusline.cjs': FOOTER_SCRIPT } }); + const script = path.join(fx.home, '.claude/helpers/statusline.cjs'); + for (const command of [ + `echo "${script}"`, + `node -e "console.log('${script}')"`, + `node -p "'${script}'"`, + `node -r "${script}" -e '0'`, + ]) { + fs.writeFileSync(fx.settingsFile, JSON.stringify({ statusLine: cmd(command) })); + assert.equal(classifyClaudeTeeChannel({ settingsFile: fx.settingsFile, home: fx.home }), 'custom', command); + } +}); + +test('option-shaped targets and unquoted comments cannot identify the project helper', () => { + const settingsFile = '/synthetic/settings.json'; + let command; + const reads = []; + const fsImpl = { + readFileSync(file) { + reads.push(file); + if (file === settingsFile) return JSON.stringify({ statusLine: cmd(command) }); + throw new Error('unexpected script read'); + }, + statSync() { throw new Error('unexpected script stat'); }, + }; + for (command of [ + 'node --eval=./.claude/helpers/statusline.cjs', + 'node --no-warnings=./.claude/helpers/statusline.cjs', + 'node #/.claude/helpers/statusline.cjs', + 'node "--eval=./.claude/helpers/statusline.cjs"', + ]) { + assert.equal(classifyClaudeTeeChannel({ settingsFile, fsImpl, home: '/synthetic' }), 'custom', command); + } + assert.deepEqual(reads, Array(4).fill(settingsFile)); +}); + +test('a quoted footer path keeps literal shell punctuation', () => { + const fx = teeFixture({ scripts: { 'My & Tools/#statusline.cjs': FOOTER_SCRIPT } }); + const script = path.join(fx.home, 'My & Tools/#statusline.cjs'); + fs.writeFileSync(fx.settingsFile, JSON.stringify({ statusLine: cmd(`node "${script}"`) })); + assert.equal(classifyClaudeTeeChannel({ settingsFile: fx.settingsFile, home: fx.home }), 'kit-footer'); +}); + +test('direct quoted helper invocation remains a footer with a Node option', () => { + const fx = teeFixture({ scripts: { 'My Tools/status line.cjs': FOOTER_SCRIPT } }); + const script = path.join(fx.home, 'My Tools/status line.cjs'); + for (const command of [`node --no-warnings "${script}"`, `node '${script}'`]) { + fs.writeFileSync(fx.settingsFile, JSON.stringify({ statusLine: cmd(command) })); + assert.equal(classifyClaudeTeeChannel({ settingsFile: fx.settingsFile, home: fx.home }), 'kit-footer', command); + } +}); + +test('direct Windows Node invocation can read a quoted footer path', () => { + const settingsFile = '/synthetic/settings.json'; + const script = 'C:\\Users\\Example User\\statusline.cjs'; + const fsImpl = { + readFileSync(file) { + if (file === settingsFile) return JSON.stringify({ statusLine: cmd(`node.exe "${script}"`) }); + if (file === script) return FOOTER_SCRIPT; + throw new Error('unexpected read'); + }, + statSync(file) { + assert.equal(file, script); + return { isFile: () => true, size: FOOTER_SCRIPT.length }; + }, + }; + assert.equal(classifyClaudeTeeChannel({ settingsFile, fsImpl, home: '/synthetic' }), 'kit-footer'); +}); + test('classifyClaudeTeeChannel: no settings file or no statusLine is "none"', () => { const { home, settingsFile } = teeFixture(); assert.equal(classifyClaudeTeeChannel({ settingsFile, home }), 'none', 'absent settings file'); @@ -509,6 +714,19 @@ test('readLimits carries the Claude tee channel class beside an unchanged claude assert.equal(out.claudeChannel, 'custom'); }); +test('readLimits forwards a managed settings path to the Claude classifier', async () => { + const fx = teeFixture({ statusLine: cmd('echo user') }); + const managedSettingsFile = path.join(fx.home, 'managed-settings.json'); + fs.writeFileSync(managedSettingsFile, JSON.stringify({ statusLine: null })); + const out = await rawReadLimits({ + now: 1000, claudeFile: path.join(fx.home, 'absent.json'), + codexCacheFile: path.join(fx.home, 'codex.json'), codexPresence: () => 'not-found', + claudeSettingsFile: fx.settingsFile, claudeManagedSettingsFile: managedSettingsFile, + home: fx.home, + }); + assert.equal(out.claudeChannel, 'unknown'); +}); + test('readLimits carries why Codex limits are unavailable beside an unchanged codex field', async () => { // codexPresence: () => 'found' pins this test to the app-server failure-class // propagation it exercises (unaffected by this task): whether the spawn is diff --git a/tests/kit/real-state-tripwire.test.mjs b/tests/kit/real-state-tripwire.test.mjs index ef911ec8..2a90383c 100644 --- a/tests/kit/real-state-tripwire.test.mjs +++ b/tests/kit/real-state-tripwire.test.mjs @@ -144,6 +144,43 @@ test('developer mode moves live-session writers to "concurrent"; strict mode fai assert.match(formatReport(dev), /concurrent writers \(not failing\)/); }); +test('Ruflo-created .claude-flow root and proven-config files are concurrent only in developer mode', (t) => { + const home = tmp(t, 'ak-trip-ruflo'); + const repo = path.join(home, 'repo'); + fs.mkdirSync(path.join(repo, '.claude'), { recursive: true }); + const roots = realStateRoots({ platform: process.platform, homedir: home, repoRoot: repo, env: {} }); + const before = snapshotRoots(roots); + fs.mkdirSync(path.join(repo, '.claude-flow')); + fs.writeFileSync(path.join(repo, '.claude-flow', 'session.json'), '{}'); + fs.writeFileSync(path.join(repo, '.claude', 'proven-config.json'), '{}'); + fs.writeFileSync(path.join(repo, '.claude', '.proven-config-version'), 'v1'); + fs.writeFileSync(path.join(repo, '.claude-flow', 'config.json'), '{}'); + const after = snapshotRoots(roots); + const dev = compareSnapshots(before, after, { strict: false }); + assert.deepEqual(dev.concurrent.map((c) => c.rel).sort(), [ + '.claude-flow/', '.claude-flow/session.json', '.claude/.proven-config-version', '.claude/proven-config.json', + ].sort()); + assert.deepEqual(dev.failing.map((c) => c.rel), ['.claude-flow/config.json']); + const strict = compareSnapshots(before, after, { strict: true }); + assert.deepEqual(strict.concurrent, []); + assert.deepEqual(strict.failing.map((c) => c.rel).sort(), [ + '.claude-flow/', '.claude-flow/config.json', '.claude-flow/session.json', + '.claude/.proven-config-version', '.claude/proven-config.json', + ].sort()); +}); + +test('removing the .claude-flow root is concurrent only in developer mode', (t) => { + const home = tmp(t, 'ak-trip-ruflo-remove'); + const repo = path.join(home, 'repo'); + fs.mkdirSync(path.join(repo, '.claude-flow'), { recursive: true }); + const roots = realStateRoots({ platform: process.platform, homedir: home, repoRoot: repo, env: {} }); + const before = snapshotRoots(roots); + fs.rmSync(path.join(repo, '.claude-flow'), { recursive: true }); + const after = snapshotRoots(roots); + assert.deepEqual(compareSnapshots(before, after, { strict: false }).concurrent.map((c) => c.rel), ['.claude-flow/']); + assert.deepEqual(compareSnapshots(before, after, { strict: true }).failing.map((c) => c.rel), ['.claude-flow/']); +}); + test('CI and AK_TRIPWIRE_STRICT make the comparison strict', () => { assert.equal(isStrict({ CI: 'true' }), true); assert.equal(isStrict({ CI: '1' }), true); diff --git a/tests/kit/refresh-vocabulary-guard.test.mjs b/tests/kit/refresh-vocabulary-guard.test.mjs index 7523a763..e36aeb38 100644 --- a/tests/kit/refresh-vocabulary-guard.test.mjs +++ b/tests/kit/refresh-vocabulary-guard.test.mjs @@ -1,21 +1,16 @@ // refresh-vocabulary-guard.test.mjs — ADR-0063 (the refresh vocabulary): no -// retired CLI spelling from before the one-refresh-flag vocabulary may +// retired CLI or dashboard spelling from before the one-refresh vocabulary may // reappear in help text, README, docs, or the installed `claude/` guidance. // -// Scope: src/** (comments included — they must describe current CLI -// behaviour), bin/**, claude/**, README.md, and living top-level docs/*.md. +// Scope: src/** (comments included — they must describe current behaviour), +// bin/**, claude/**, README.md, and current docs/**/*.md. // docs/adr/, docs/archive/, docs/plans/, and docs/proposals/ are records that // may preserve retired spellings. In // src/lib/hook-audit/agentic-dependency-constraints.json only the dated // watch[].history[].note strings are skipped — every other string, including // `adjustment`, is scanned like any other source text. // -// CLI patterns only: the dashboard's own retired spellings ("Full -// scan", "Refresh evidence", "Re-measure machine", "Check again", "refresh -// now") are out of this guard's scope until the dashboard half of this work -// lands in a later branch (see docs/superpowers/plans/2026-09-28-branch-6b- -// one-refresh-flag.md, "Closing this branch"). This guard never asserts an -// UPGRADING section or an old -> new table exists (no legacy, no hints). +// This guard never asserts an UPGRADING section or an old -> new table exists. import { test } from 'node:test'; import assert from 'node:assert/strict'; import fs from 'node:fs'; @@ -60,13 +55,25 @@ const RETIRED_CLI_PATTERNS = [ { label: 'ak usage prompts …--deep (flag anywhere on the same line)', pattern: /\bprompts\b[^\n]*\[?--deep\b/g }, ]; +const RETIRED_DASHBOARD_PATTERNS = [ + { label: 'retired dashboard scan control', pattern: /\bFull scan\b/g }, + { label: 'retired dashboard evidence control', pattern: /\bRefresh evidence\b/g }, + { label: 'retired dashboard measurement control', pattern: /\bRe-measure machine\b/g }, + { label: 'retired dashboard local-check control', pattern: /\bCheck again\b/g }, + { label: 'retired dashboard refresh prompt', pattern: /\brefresh now\b/g }, + { label: 'retired dashboard GET refresh query', pattern: /refresh=(?:deep|scan)/g }, + { label: 'retired dashboard host-health route', pattern: /\/api\/host-health\/local/g }, +]; + +const RETIRED_PATTERNS = [...RETIRED_CLI_PATTERNS, ...RETIRED_DASHBOARD_PATTERNS]; + function lineOf(text, offset) { return text.slice(0, offset).split('\n').length; } function violations(relPath, text) { const found = []; - for (const { label, pattern } of RETIRED_CLI_PATTERNS) { + for (const { label, pattern } of RETIRED_PATTERNS) { pattern.lastIndex = 0; for (const match of text.matchAll(pattern)) { found.push(`${relPath}:${lineOf(text, match.index)} ${label}: ${JSON.stringify(match[0])}`); @@ -157,14 +164,14 @@ function scopeFiles() { return files.filter((file) => file !== REGISTRY); } -test('no retired CLI spelling remains in help, README, docs, or installed guidance', () => { +test('no retired CLI or dashboard spelling remains in source, README, current docs, or installed guidance', () => { const found = []; for (const file of scopeFiles()) { const text = fs.readFileSync(file, 'utf8'); found.push(...violations(path.relative(ROOT, file), text)); } found.push(...registryViolations()); - assert.deepEqual(found, [], `retired CLI spellings remain:\n${found.join('\n')}`); + assert.deepEqual(found, [], `retired CLI or dashboard spellings remain:\n${found.join('\n')}`); }); test('the registry skip is narrow: dated watch[].history[].note strings still contain the old spellings they document', () => { diff --git a/tests/kit/refresh.test.mjs b/tests/kit/refresh.test.mjs index 8dcf0fae..0d5fed7e 100644 --- a/tests/kit/refresh.test.mjs +++ b/tests/kit/refresh.test.mjs @@ -163,7 +163,7 @@ test('runRefresh refuses an unknown strength or a missing stage', async () => { test('the refresh operation never references the paid connection check', () => { // Every module that runs refresh stages belongs in this list, including any // future server-side refresh module. - for (const file of [REFRESH_SOURCE]) { + for (const file of [REFRESH_SOURCE, path.join(PKG_ROOT, 'src/lib/dashboard/refresh-api.mjs')]) { const source = fs.readFileSync(file, 'utf8'); assert.doesNotMatch(source, /checkConnection|host-health-connected/, file); } diff --git a/tests/kit/ruflo-components-snapshot.test.mjs b/tests/kit/ruflo-components-snapshot.test.mjs index f1e99fdd..8fee7a6c 100644 --- a/tests/kit/ruflo-components-snapshot.test.mjs +++ b/tests/kit/ruflo-components-snapshot.test.mjs @@ -189,7 +189,7 @@ for (const decidedBy of ['package-default', 'project-config', 'user-config']) { } // Controller ruling 1: the shared, read-only projection every surface (status, -// Task 10's dashboard) builds a snapshot from. +// the dashboard) builds a snapshot from. test('rufloComponentsPayload has all 8 components, writes nothing, and reports unwritten typesafe as not applied', () => { const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-rc-payload-')); try { diff --git a/tests/kit/ruflo-components-status.test.mjs b/tests/kit/ruflo-components-status.test.mjs index 8e8be73b..15f468ae 100644 --- a/tests/kit/ruflo-components-status.test.mjs +++ b/tests/kit/ruflo-components-status.test.mjs @@ -1,6 +1,7 @@ import { test } from 'node:test'; import assert from 'node:assert/strict'; import { rufloComponentRows, formatComponentResults, componentResultReport, RESTART_REMINDER } from '../../src/commands/status/sections/ruflo-components.mjs'; +import { describeState } from '../../src/lib/ruflo-components/states.mjs'; import { rufloComponentsTrustGroup, trustManifestLines } from '../../src/lib/trust-manifest.mjs'; const view = (id, label, stateId, stateLabel, meaning, action = '') => ({ id, label, state: { id: stateId, label: stateLabel, meaning, action } }); @@ -23,6 +24,45 @@ test('rows lead with a summary and always carry the meaning', () => { assert.equal(rows[3].fix, null); }); +test('applied-unverified gives one restart instruction across message and manual fix', () => { + const [summary, unverified] = rufloComponentRows({ + rufloVersion: '3.44.0', summary: { active: 0, total: 1 }, components: [ + { id: 'minilmPicker', label: 'MiniLM agent picker', state: describeState('applied-unverified') }, + ], + }); + assert.equal(summary.level, 'info'); + assert.equal(unverified.state, 'applied-unverified'); + assert.equal(unverified.level, 'warn'); + assert.equal(unverified.repair, 'manual'); + assert.match(unverified.message, /MiniLM agent picker — applied, not verified: Set, but not yet confirmed/); + assert.match(unverified.fix, /restart Claude Code, Codex and OpenCode, then run ak status --refresh/i); + assert.equal(`${unverified.message} ${unverified.fix}`.match(/restart Claude Code, Codex and OpenCode/gi)?.length, 1); +}); + +test('neighboring component rows retain action, repair, and convergence contracts', () => { + const states = ['not-applied', 'drifted', 'blocked', 'active']; + const rows = rufloComponentRows({ + rufloVersion: '3.44.0', summary: { active: 1, total: 4 }, + components: states.map((id) => ({ id, label: `${id} component`, state: describeState(id) })), + }).slice(1); + for (const id of ['not-applied', 'drifted']) { + const rendered = rows.find((r) => r.state === id); + assert.equal(rendered.level, 'warn'); + assert.equal(rendered.repair, 'sync'); + assert.match(rendered.message, /Run ak sync/); + assert.match(rendered.fix, /sync applies/); + } + const blocked = rows.find((r) => r.state === 'blocked'); + assert.equal(blocked.level, 'fail'); + assert.equal(blocked.repair, 'sync'); + assert.match(blocked.message, /Follow the reason shown, then run ak sync/); + assert.match(blocked.fix, /sync applies/); + const active = rows.find((r) => r.state === 'active'); + assert.equal(active.level, 'ok'); + assert.equal(active.repair, null); + assert.equal(active.fix, null); +}); + test('setup results table lists state and meaning per component', () => { const lines = formatComponentResults(snapshot); assert.ok(lines.some((l) => /Learning profile.*user-managed.*leaves it alone/.test(l))); diff --git a/tests/kit/ruflo-daemon-config.test.mjs b/tests/kit/ruflo-daemon-config.test.mjs index 8c4ac413..30af764f 100644 --- a/tests/kit/ruflo-daemon-config.test.mjs +++ b/tests/kit/ruflo-daemon-config.test.mjs @@ -1,4 +1,4 @@ -// ak-managed Ruflo daemon settings (Branch 3, Task 2.2). The daemon reads FLAT +// ak-managed Ruflo daemon settings. The daemon reads FLAT // keys from /.claude-flow/config.json once, in its constructor // (worker-daemon.js:139-144, 354-416, Ruflo 3.46.1); `ruflo config set` // writes nested keys and relocates the memory root (ruvnet/ruflo#3449), so ak @@ -120,6 +120,174 @@ test('a malformed, non-object or symlinked config.json is user-managed and untou assert.equal(fs.readFileSync(target, 'utf8'), '{}'); }); +test('YAML markers hold JSON creation in apply and preview without hiding user daemon values', (t) => { + for (const extension of ['yaml', 'yml']) { + const root = tmpProject(t); + fs.mkdirSync(path.join(root, '.claude-flow')); + const yaml = path.join(root, '.claude-flow', `config.${extension}`); + fs.writeFileSync(yaml, 'daemon:\n maxConcurrent: 7\n'); + writeSettings(root, { claudeFlow: { daemon: { autoStart: false } } }); + const receipts = {}; + const preview = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts, dryRun: true }); + assert.equal(preview.config, 'user-managed'); + assert.equal(preview.held?.reason, 'yaml-shadow'); + assert.equal(preview.autostart, 'enabled'); + assert.deepEqual(receipts, {}); + const applied = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts }); + assert.deepEqual({ config: applied.config, held: applied.held, changed: applied.changed }, + { config: preview.config, held: preview.held, changed: preview.changed }); + assert.equal(fs.existsSync(configFile(root)), false); + assert.equal(fs.readFileSync(yaml, 'utf8'), 'daemon:\n maxConcurrent: 7\n'); + assert.equal(autoStart(root), true); + } +}); + +test('a root JSON config holds ineffective creation of lower-priority daemon JSON', (t) => { + const root = tmpProject(t); + fs.writeFileSync(path.join(root, 'claude-flow.config.json'), '{"daemon.maxConcurrent":7}'); + const result = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts: {} }); + assert.equal(result.held?.reason, 'higher-priority-json'); + assert.equal(fs.existsSync(configFile(root)), false); +}); + +test('a symlinked .claude-flow directory cannot redirect daemon JSON edits outside the project', (t) => { + const root = tmpProject(t); + const target = tmpProject(t); + fs.symlinkSync(target, path.join(root, '.claude-flow')); + const receipts = {}; + const result = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts }); + assert.equal(result.config, 'user-managed'); + assert.equal(result.held?.invalid, true); + assert.equal(fs.existsSync(path.join(target, 'config.json')), false); + assert.deepEqual(receipts, {}); + fs.writeFileSync(path.join(target, 'config.json'), '{}'); + const second = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts }); + assert.equal(second.config, 'user-managed'); + assert.equal(fs.readFileSync(path.join(target, 'config.json'), 'utf8'), '{}'); +}); + +test('root JSON becoming active holds new keys but permits receipted obsolete-key cleanup', (t) => { + const root = tmpProject(t); + const receipts = {}; + reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts }); + fs.writeFileSync(path.join(root, 'claude-flow.config.json'), '{"daemon.maxConcurrent":7}'); + const result = reconcileRufloDaemon(root, { rufloVersion: '3.46.1', platform: 'darwin', receipts }); + assert.equal(result.held?.reason, 'higher-priority-json'); + assert.deepEqual(readConfig(root), { 'daemon.resourceThresholds.minFreeMemoryPercent': 0 }); + assert.deepEqual(receipts[path.resolve(root)].configKeys, + { 'daemon.resourceThresholds.minFreeMemoryPercent': 0 }); +}); + +test('cleanup of inactive receipted JSON under root JSON does not restart a live daemon', async (t) => { + const root = rufloRepo(t); + const cfg = { rufloDaemon: { receipts: {} } }; + reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts: cfg.rufloDaemon.receipts }); + fs.writeFileSync(path.join(root, 'claude-flow.config.json'), '{"daemon.maxConcurrent":7}'); + const before = JSON.stringify(cfg.rufloDaemon.receipts); + const preview = reconcileRufloDaemon(root, { + rufloVersion: '3.46.1', platform: 'linux', receipts: cfg.rufloDaemon.receipts, dryRun: true, + }); + assert.equal(preview.config, 'removed'); + assert.equal(JSON.stringify(cfg.rufloDaemon.receipts), before); + const { calls, runner } = recorder(); + const applied = await applyRufloDaemon(root, { + cfg, rufloVersion: '3.46.1', platform: 'linux', runner, alive: () => true, + }); + assert.equal(applied.result.config, 'removed'); + assert.equal(applied.restarted, false); + assert.deepEqual(calls, []); + assert.equal(fs.existsSync(configFile(root)), false); + assert.deepEqual(cfg.rufloDaemon.receipts, {}); +}); + +test('an existing explicit config holds creation of daemon JSON in preview and apply', (t) => { + const root = tmpProject(t); + const external = tmpProject(t); + const custom = path.join(external, 'custom-config.json'); + fs.writeFileSync(custom, '{"daemon.maxConcurrent":7}'); + fs.mkdirSync(path.join(root, '.claude-flow')); + fs.writeFileSync(path.join(root, '.claude-flow', 'config.yaml'), 'daemon:\n maxConcurrent: 9\n'); + const receipts = {}; + const env = { CLAUDE_FLOW_CONFIG: custom }; + const before = fs.readFileSync(custom, 'utf8'); + const preview = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts, env, dryRun: true }); + assert.equal(preview.config, 'user-managed'); + assert.equal(preview.held?.reason, 'explicit-config'); + assert.deepEqual(receipts, {}); + const applied = reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts, env }); + assert.deepEqual(applied, preview); + assert.equal(fs.existsSync(configFile(root)), false); + assert.equal(fs.readFileSync(custom, 'utf8'), before); +}); + +test('relative explicit config resolves from process cwd, and absent path does not hold JSON creation', (t) => { + const root = tmpProject(t); + const external = tmpProject(t); + const originalCwd = process.cwd(); + fs.writeFileSync(path.join(external, 'custom-config.json'), '{"daemon.maxConcurrent":7}'); + try { + process.chdir(external); + const held = reconcileRufloDaemon(root, { + rufloVersion: '3.45.0', platform: 'darwin', receipts: {}, + env: { CLAUDE_FLOW_CONFIG: './custom-config.json' }, dryRun: true, + }); + assert.equal(held.held?.reason, 'explicit-config'); + const writable = reconcileRufloDaemon(root, { + rufloVersion: '3.45.0', platform: 'darwin', receipts: {}, + env: { CLAUDE_FLOW_CONFIG: './missing-config.json' }, dryRun: true, + }); + assert.equal(writable.config, 'written'); + } finally { process.chdir(originalCwd); } +}); + +test('an existing daemon JSON takes precedence over CLAUDE_FLOW_CONFIG', async (t) => { + const root = rufloRepo(t); + const external = tmpProject(t); + const custom = path.join(external, 'custom-config.json'); + fs.writeFileSync(custom, '{"daemon.maxConcurrent":7}'); + fs.writeFileSync(configFile(root), '{}'); + const { calls, runner } = recorder(); + const result = await applyRufloDaemon(root, { + cfg: { rufloDaemon: { receipts: {} } }, rufloVersion: '3.46.1', platform: 'darwin', + env: { CLAUDE_FLOW_CONFIG: custom }, runner, alive: () => true, + }); + assert.equal(result.result.config, 'written'); + assert.equal(result.restarted, true); + assert.deepEqual(readConfig(root), { 'daemon.resourceThresholds.minFreeMemoryPercent': 0 }); + assert.equal(fs.readFileSync(custom, 'utf8'), '{"daemon.maxConcurrent":7}'); + assert.deepEqual(calls, [['ruflo', 'daemon', 'stop', root], ['ruflo', 'daemon', 'start', root]]); +}); + +test('an existing JSON keeps its ownership rules beside YAML, and release exposes YAML again', (t) => { + const root = tmpProject(t); + const receipts = {}; + reconcileRufloDaemon(root, { rufloVersion: '3.45.0', platform: 'darwin', receipts }); + const yaml = path.join(root, '.claude-flow', 'config.yml'); + fs.writeFileSync(yaml, 'daemon:\n maxConcurrent: 7\n'); + const preview = reconcileRufloDaemon(root, { rufloVersion: '3.46.1', platform: 'linux', receipts, dryRun: true }); + assert.equal(preview.config, 'removed'); + assert.equal(fs.existsSync(configFile(root)), true); + assert.equal(reconcileRufloDaemon(root, { rufloVersion: '3.46.1', platform: 'linux', receipts }).config, 'removed'); + assert.equal(fs.existsSync(configFile(root)), false); + assert.equal(fs.readFileSync(yaml, 'utf8'), 'daemon:\n maxConcurrent: 7\n'); + assert.deepEqual(receipts, {}); +}); + +test('a YAML hold stops repeated low-memory restarts while leaving the YAML and receipts alone', async (t) => { + const root = rufloRepo(t); + fs.writeFileSync(path.join(root, '.claude-flow', 'config.yaml'), 'daemon:\n maxConcurrent: 7\n'); + fs.mkdirSync(path.join(root, '.claude-flow', 'logs')); + fs.writeFileSync(path.join(root, '.claude-flow', 'logs', 'daemon.log'), + `[${new Date().toISOString()}] [INFO] Worker consolidate deferred: Memory too low: 3.9% free\n`); + const { calls, runner } = recorder(); + const cfg = { rufloDaemon: { receipts: {} } }; + const result = await applyRufloDaemon(root, { cfg, rufloVersion: '3.46.1', platform: 'darwin', runner, alive: () => true }); + assert.equal(result.restarted, false); + assert.deepEqual(calls, []); + assert.equal(fs.existsSync(configFile(root)), false); + assert.deepEqual(cfg.rufloDaemon.receipts, {}); +}); + test('the memory pin still wins and .swarm stays the memory root', (t) => { const root = tmpProject(t); fs.writeFileSync(path.join(root, 'claude-flow.config.json'), JSON.stringify({ memory: { persistPath: '.swarm' } })); @@ -280,7 +448,7 @@ test('a daemon deferring under a user-managed config.json is not restarted: sync } }); -// F5 (Branch 3 fix round 2): Ruflo 3.46.1 treats a folder as a Ruflo project +// F5: Ruflo 3.46.1 treats a folder as a Ruflo project // only with a durable marker (services/daemon-autostart.js:90-123, isRufloProject): // .claude-flow/config.{yaml,yml,json}, claude-flow.config.json, // .swarm/memory.db, settings.json `claudeFlow`, or a ruflo/claude-flow server in diff --git a/tests/kit/ruflo-memory-location.test.mjs b/tests/kit/ruflo-memory-location.test.mjs index 212b8d43..32fa0c54 100644 --- a/tests/kit/ruflo-memory-location.test.mjs +++ b/tests/kit/ruflo-memory-location.test.mjs @@ -84,6 +84,30 @@ test('a disposable folder below a temporary root inside a tool folder keeps its assert.equal(launch.env.CLAUDE_FLOW_MEMORY_PATH, undefined); }); +test('a temporary root equal to a tool folder does not exempt its descendants', (t) => { + const home = sandbox(t); + const cache = mkdir(path.join(home, '.cache')); + const child = mkdir(path.join(cache, 'project')); + const at = (cwd) => rufloMemoryLocation(cwd, { home, env: { TMPDIR: cache } }); + assert.equal(at(cache).kind, 'user', 'the tool folder itself remains unsuitable'); + const location = at(child); + assert.equal(location.kind, 'user'); + assert.match(location.reason, /inside ~\/\.cache, a tool's own folder/); + assert.equal(location.db, path.join(userStore(home), 'memory.db')); +}); + +test('Windows same-path temp and tool roots are not deeper disposable boundaries', () => { + const home = 'C:\\Users\\Me'; + const local = `${home}\\AppData\\Local`; + const options = { home, platform: 'win32', p: path.win32, env: { LOCALAPPDATA: local, TMPDIR: local.toLowerCase() } }; + const tool = rufloMemoryLocation(local, options); + const child = rufloMemoryLocation(`${local}\\project`, options); + assert.equal(tool.kind, 'user'); + assert.equal(child.kind, 'user'); + assert.match(child.reason, /tool's own folder/); + assert.equal(child.db, `${home}\\.claude-flow\\memory\\memory.db`); +}); + test('the filesystem root, the home folder and a temporary root use the one user-level store', (t) => { const home = sandbox(t); for (const [cwd, reason] of [ @@ -167,6 +191,25 @@ test('a Git repository at the home folder does not pull a plain subfolder into ~ assert.equal(rufloMemoryLocation(home, { home }).kind, 'user'); }); +test('user-level location explains both an unsuitable folder and its unsuitable repository root', (t) => { + const home = sandbox(t); + fs.mkdirSync(path.join(home, '.git')); + const codex = mkdir(path.join(home, '.codex', 'sessions')); + const location = rufloMemoryLocation(codex, { home }); + assert.deepEqual([location.kind, location.root, location.dir, location.db], + ['user', userStore(home), userStore(home), path.join(userStore(home), 'memory.db')]); + assert.equal(location.reason, "inside ~/.codex, a tool's own folder, in a repository whose root is the home folder"); +}); + +test('location does not repeat a reason when root and folder share one tool area', (t) => { + const home = sandbox(t); + const root = mkdir(path.join(home, '.codex', 'workspace')); + fs.mkdirSync(path.join(root, '.git')); + const child = mkdir(path.join(root, 'src')); + assert.equal(rufloMemoryLocation(child, { home }).reason, "inside ~/.codex, a tool's own folder"); + assert.equal(rufloMemoryLocation(root, { home }).reason, "inside ~/.codex, a tool's own folder"); +}); + test('the launcher pins both memory variables to the user-level store and starts Ruflo inside it', (t) => { const home = sandbox(t); const launch = rufloMcpLaunch(home, { CLAUDE_FLOW_DB_PATH: '/.swarm/memory.db', KEEP: 'yes' }, { cfg, rufloVersion: '3.45.0', home }); @@ -215,6 +258,27 @@ test('status reports the user-level store and stray stores outside projects, for assert.match(strays[0].message, /leaves them in place/); }); +test('user status reports AQE home data separately without calling an empty folder healthy', async (t) => { + const home = sandbox(t); + const aqeDir = path.join(home, '.agentic-qe'); + fs.mkdirSync(aqeDir); + let rows = await userMemory.collect({ home, env: {} }); + let aqe = rows.find((r) => r.message.includes(aqeDir)); + assert.ok(aqe); + assert.equal(aqe.level, 'info'); + assert.equal(aqe.fix, null); + assert.match(aqe.message, /AQE.*memory\.db absent.*unverified/); + assert.doesNotMatch(aqe.message, /Ruflo|healthy|merge|move/i); + + fs.writeFileSync(path.join(aqeDir, 'memory.db'), 'placeholder'); + fs.writeFileSync(path.join(aqeDir, 'memory.db-wal'), 'wal'); + rows = await userMemory.collect({ home, env: {} }); + aqe = rows.find((r) => r.message.includes(aqeDir)); + assert.match(aqe.message, /AQE.*memory\.db present.*14 B.*unverified/); + assert.doesNotMatch(aqe.message, /Ruflo|healthy|merge|move/i); + assert.equal(rows.filter((r) => /stray Ruflo/.test(r.message)).length, 0); +}); + test('Claude mode from the home folder pins both memory variables to the user-level store', (t) => { const home = sandbox(t); const launch = rufloMcpLaunch(home, { RUFLO_INTELLIGENCE_MODE: 'fast' }, { cfg, rufloVersion: '3.46.1', home, host: 'claude' }); diff --git a/tests/kit/ruflo-windows-diagnostic-process.test.mjs b/tests/kit/ruflo-windows-diagnostic-process.test.mjs new file mode 100644 index 00000000..d902dcd2 --- /dev/null +++ b/tests/kit/ruflo-windows-diagnostic-process.test.mjs @@ -0,0 +1,191 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { EventEmitter } from 'node:events'; +import { PassThrough } from 'node:stream'; +import { spawn } from 'node:child_process'; +import { createDiagnosticScope } from '../live/ruflo-windows-diagnostic-process.mjs'; + +function fakeChild({ closes = true, code = 0 } = {}) { + const child = new EventEmitter(); + Object.assign(child, { pid: 123, exitCode: null, signalCode: null, + stdin: new PassThrough(), stdout: new PassThrough(), stderr: new PassThrough(), + unreferenced: false, killed: false }); + child.unref = () => { child.unreferenced = true; }; + child.kill = () => { + child.killed = true; + if (closes) queueMicrotask(() => { child.exitCode = code; child.emit('close', code, null); }); + }; + return child; +} + +function scopeFor(spawnFn) { + let released = false; + const scope = createDiagnosticScope({ spawnFn, platform: 'win32', + killTimeoutMs: 20, closeTimeoutMs: 20, + acquireHold: () => ({}), releaseHold: () => { released = true; } }); + return { scope, released: () => released }; +} + +test('every diagnostic role passes the exact environment, excluding parent-only credentials', async () => { + process.env.AK_DIAGNOSTIC_PARENT_ONLY = 'synthetic-do-not-inherit'; + const env = { PATH: 'fixture-only' }; + const seen = []; + let active; + const { scope } = scopeFor((command, _args, options) => { + seen.push({ command, env: options.env }); + const child = fakeChild(); + if (command === 'taskkill.exe') queueMicrotask(() => { active.kill(); child.kill(); }); + else active = child; + return child; + }); + try { + for (const command of ['version', 'legacy', 'initialize']) { + const r = await scope.launch({ command, args: [] }, { env, timeoutMs: 10 }); + assert.equal(r.cleanupComplete, true); + } + assert.deepEqual(seen.map((x) => x.command), ['version', 'taskkill.exe', 'legacy', 'taskkill.exe', 'initialize', 'taskkill.exe']); + for (const entry of seen) { + assert.strictEqual(entry.env, env); + assert.equal(entry.env.AK_DIAGNOSTIC_PARENT_ONLY, undefined); + } + assert.equal(scope.release(), true); + } finally { delete process.env.AK_DIAGNOSTIC_PARENT_ONLY; } +}); + +for (const failure of ['nonzero', 'spawn-error', 'stall']) { + test(`taskkill ${failure} always disposes handles and retains uncertainty`, async () => { + const children = []; + const { scope, released } = scopeFor((command) => { + const child = fakeChild({ closes: false }); children.push(child); + if (command === 'taskkill.exe' && failure !== 'stall') queueMicrotask(() => { + if (failure === 'spawn-error') child.emit('error', new Error('ENOENT')); + child.exitCode = 1; child.emit('close', 1, null); + }); + return child; + }); + const started = Date.now(); + const r = await scope.launch({ command: 'initialize', args: [] }, { env: {}, timeoutMs: 10 }); + assert.equal(r.cleanupComplete, false); + assert.ok(Date.now() - started < 1000); + assert.equal(children[0].killed, true, 'direct-child fallback attempted'); + assert.equal(children[0].unreferenced, true); + assert.equal(children[0].stdout.destroyed, true); + assert.equal(children[0].stderr.destroyed, true); + assert.equal(children[0].stdin.destroyed, true); + assert.equal(scope.release(), false); + assert.equal(released(), false); + if (failure === 'stall') assert.equal(children[1].unreferenced, true); + }); +} + +for (const command of ['version', 'legacy']) { + test(`${command} cleanup failure cannot be cleared by a later successful launch`, async () => { + const { scope, released } = scopeFor((cmd) => { + const child = fakeChild(); + if (cmd === 'later' || cmd === 'taskkill.exe') queueMicrotask(() => { + child.exitCode = cmd === 'taskkill.exe' ? 1 : 0; child.emit('close', child.exitCode, null); + }); + return child; + }); + assert.equal((await scope.launch({ command, args: [] }, { env: {}, timeoutMs: 10 })).cleanupComplete, false); + assert.equal((await scope.launch({ command: 'later', args: [] }, { env: {}, timeoutMs: 10 })).cleanupComplete, true); + assert.equal(scope.release(), false); + assert.equal(released(), false); + }); +} + +test('real version, legacy, initialize and taskkill children cannot see a parent-only sentinel', async () => { + const key = 'AK_DIAGNOSTIC_PARENT_ONLY'; + const previous = process.env[key]; + process.env[key] = 'synthetic-do-not-inherit'; + const env = { AK_DIAGNOSTIC_ALLOWED: 'yes' }; + const observed = []; + const scope = createDiagnosticScope({ + platform: 'win32', acquireHold: () => ({}), releaseHold: () => {}, + killTimeoutMs: 1000, closeTimeoutMs: 1000, + spawnFn: (command, args, options) => { + const payload = 'process.stdout.write(JSON.stringify({allowed:process.env.AK_DIAGNOSTIC_ALLOWED,sentinel:process.env.AK_DIAGNOSTIC_PARENT_ONLY}));'; + const code = command === 'taskkill.exe' + ? `${payload}process.kill(${Number(args[1])},'SIGKILL');` + : `${payload}${command === 'initialize' ? 'setInterval(()=>{},1000);' : ''}`; + const child = spawn(process.execPath, ['-e', code], { ...options, env: options.env }); + let output = ''; + child.stdout.on('data', (chunk) => { output += chunk; }); + child.on('close', () => observed.push({ command, ...JSON.parse(output) })); + return child; + }, + }); + try { + for (const command of ['version', 'legacy', 'initialize']) { + const result = await scope.launch({ command, args: [] }, { env, timeoutMs: 500 }); + assert.equal(result.cleanupComplete, true); + } + assert.deepEqual(observed.map((x) => x.command).sort(), ['initialize', 'legacy', 'taskkill.exe', 'version']); + for (const result of observed) { + assert.equal(result.allowed, 'yes'); + assert.equal(result.sentinel, undefined); + } + assert.equal(scope.release(), true); + } finally { + if (previous === undefined) delete process.env[key]; else process.env[key] = previous; + } +}); + +test('EOF response waits for natural pipe closure without racing taskkill', async () => { + let child; + const commands = []; + const { scope, released } = scopeFor((command) => { + commands.push(command); + child = fakeChild({ closes: false }); + queueMicrotask(() => child.stdout.write('initialized')); + setTimeout(() => { child.exitCode = 0; }, 5); + setTimeout(() => child.emit('close', 0, null), 60); + return child; + }); + const result = await scope.launch({ command: 'eof', args: [] }, { + env: {}, timeoutMs: 200, endInput: true, until: (text) => text === 'initialized', + }); + assert.equal(result.matched, true); + assert.equal(result.code, 0); + assert.equal(result.cleanupComplete, true, 'requires observed close, not just exit code'); + assert.equal(result.timedOut, false); + assert.equal(child.killed, false); + assert.deepEqual(commands, ['eof']); + assert.equal(scope.release(), true); + assert.equal(released(), true); +}); + +test('EOF match plus exit code zero without pipe closure retains uncertainty', async () => { + const commands = []; + let child; + const { scope, released } = scopeFor((command) => { + commands.push(command); + child = fakeChild({ closes: false }); + queueMicrotask(() => { child.stdout.write('initialized'); child.exitCode = 0; }); + return child; + }); + const result = await scope.launch({ command: 'eof', args: [] }, { + env: {}, timeoutMs: 30, endInput: true, until: (text) => text === 'initialized', + }); + assert.equal(result.matched, true); + assert.equal(result.timedOut, true); + assert.equal(result.cleanupComplete, false); + assert.deepEqual(commands, ['eof'], 'never target a reaped PID'); + assert.equal(child.unreferenced, true); + assert.equal(child.stdout.destroyed, true); + assert.equal(scope.release(), false); + assert.equal(released(), false); +}); + +test('a real EOF responder exits naturally after its reply before cleanup is accepted', async () => { + const scope = createDiagnosticScope({ acquireHold: () => ({}), releaseHold: () => {} }); + const code = "process.stdin.resume();process.stdin.on('end',()=>{process.stdout.write('initialized');setTimeout(()=>process.exit(0),80)});"; + const result = await scope.launch({ command: process.execPath, args: ['-e', code] }, { + env: {}, timeoutMs: 2000, endInput: true, until: (text) => text === 'initialized', + }); + assert.equal(result.matched, true); + assert.equal(result.code, 0); + assert.equal(result.signal, null, 'no forced termination after the reply'); + assert.equal(result.cleanupComplete, true); + assert.equal(scope.release(), true); +}); diff --git a/tests/kit/run-root-identities.test.mjs b/tests/kit/run-root-identities.test.mjs new file mode 100644 index 00000000..4b79d17b --- /dev/null +++ b/tests/kit/run-root-identities.test.mjs @@ -0,0 +1,139 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; +import { ownerRecord, writeOwner, readOwner, collectAbandonedRoots, OWNER_FILE } from '../../scripts/run-roots.mjs'; +import { runGuarded } from '../../scripts/run-tests.mjs'; + +const first = 2n ** 53n; +const second = first + 1n; +assert.equal(Number(first), Number(second), 'fixture must collide through Number'); + +// Model the OS exposing exact BigInt fields or lossy legacy Number fields. +function identity(stat, options, field, value) { + const exact = options?.bigint === true; + const key = !exact && field.endsWith('Ns') ? field.replace(/Ns$/, 'Ms') : field; + stat[key] = exact ? value : Number(value) / (field.endsWith('Ns') ? 1e6 : 1); + return stat; +} +function fixture(t) { + const parent = fs.realpathSync(tempDir('ak-exact-root', t)); + const root = fs.mkdtempSync(path.join(parent, 'ak-suite-')); + const home = path.join(parent, 'home'); + fs.mkdirSync(home); + writeOwner(root, ownerRecord()); + return { parent, root, home }; +} + +for (const field of ['dev', 'ino', 'ctimeNs']) { + for (const phase of ['open', 'read']) { + test(`owner ${field} collision at ${phase} refuses swapped file`, (t) => { + const { root } = fixture(t); + const file = path.join(root, OWNER_FILE); + const originalLstat = fs.lstatSync; + const originalFstat = fs.fstatSync; + let reads = 0; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = originalLstat(target, options); + return target === file ? identity(stat, options, field, ++reads === 1 ? first : second) : stat; + }); + t.mock.method(fs, 'fstatSync', (fd, options) => + identity(originalFstat(fd, options), options, field, phase === 'open' ? second : first)); + assert.equal(readOwner(root), null); + }); + } + + test(`sibling ${field} collision during injected proof retains root`, (t) => { + const { parent, root, home } = fixture(t); + const original = fs.lstatSync; + let swapped = false; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = original(target, options); + return target === root ? identity(stat, options, field, swapped ? second : first) : stat; + }); + let removals = 0; + const result = collectAbandonedRoots({ tmpdir: parent, homedir: home, log: () => {}, + probes: { alive: () => false, startedAfter: () => false, completeExit: () => { swapped = true; return true; } }, + remove: () => { removals++; }, + }); + assert.equal(removals, 0); + assert.deepEqual(result.removed, []); + assert.match(result.kept[0].reason, /changed/); + assert.ok(fs.existsSync(root)); + }); +} + +for (const field of ['dev', 'ino', 'birthtimeNs']) { + test(`call-owned ${field} collision retains replacement untouched`, (t) => { + const { parent, home } = fixture(t); + const tmpdir = path.join(parent, 'tmp'); + const repoRoot = path.join(home, 'repo'); + fs.mkdirSync(tmpdir); + fs.mkdirSync(repoRoot); + const env = spawnEnv(home); + t.mock.method(os, 'tmpdir', () => tmpdir); + const original = fs.lstatSync; + let swapped = false; + let ownRoot; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = original(target, options); + if (!stat || path.dirname(String(target)) !== tmpdir || !/^ak-suite-[A-Za-z0-9]{6}$/.test(path.basename(String(target)))) return stat; + ownRoot = target; + for (const key of ['dev', 'ino', 'birthtimeNs']) identity(stat, options, key, first); + return identity(stat, options, field, swapped ? second : first); + }); + const messages = []; + const code = runGuarded([], { env, homedir: home, repoRoot, log: (message) => { + messages.push(message); + if (message.startsWith('real-state tripwire:')) { + // Preserve metadata so the ownership/hold checks alone cannot catch this swap. + fs.cpSync(ownRoot, `${ownRoot}-saved`, { recursive: true }); + fs.rmSync(ownRoot, { recursive: true }); + fs.renameSync(`${ownRoot}-saved`, ownRoot); + swapped = true; + } + } }); + assert.equal(code, 4); + assert.match(messages.join('\n'), /directory identity changed/); + assert.ok(fs.existsSync(path.join(ownRoot, OWNER_FILE))); + assert.ok(fs.existsSync(path.join(ownRoot, '.ak-suite-holds'))); + }); +} + +test('unchanged exact owner identity accepts large IDs and preserves numeric attribution', (t) => { + const { root } = fixture(t); + const file = path.join(root, OWNER_FILE); + const lstat = fs.lstatSync; + const fstat = fs.fstatSync; + const patch = (stat, options) => { + for (const field of ['dev', 'ino', 'ctimeNs']) identity(stat, options, field, second); + return stat; + }; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = lstat(target, options); + return target === file ? patch(stat, options) : stat; + }); + t.mock.method(fs, 'fstatSync', (fd, options) => patch(fstat(fd, options), options)); + const owner = readOwner(root); + assert.equal(owner.pid, process.pid); + assert.equal(owner.uid, process.getuid?.() ?? null); + assert.equal(typeof owner.startedAt, 'number'); +}); + +test('exact owner stats reject foreign filesystem UID before opening', (t) => { + if (!process.getuid) return; // Windows has no UID boundary to validate. + const { root } = fixture(t); + const file = path.join(root, OWNER_FILE); + const lstat = fs.lstatSync; + t.mock.method(fs, 'lstatSync', (target, options) => { + const stat = lstat(target, options); + if (target === file) stat.uid = options?.bigint ? BigInt(process.getuid()) + 1n : process.getuid() + 1; + return stat; + }); + const open = t.mock.method(fs, 'openSync', () => { throw Error('must not open foreign owner'); }); + assert.equal(readOwner(root), null); + assert.equal(open.mock.callCount(), 0); +}); diff --git a/tests/kit/run-roots.test.mjs b/tests/kit/run-roots.test.mjs new file mode 100644 index 00000000..ae0b1154 --- /dev/null +++ b/tests/kit/run-roots.test.mjs @@ -0,0 +1,266 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { ownerRecord, writeOwner, readOwner, unsafeTempBase, removableRunRoot, + proveAbandoned, collectAbandonedRoots, defaultProbes, OWNER_FILE, + prepareRunRootHolds, acquireRunRootHold, releaseRunRootHold, inspectRunRootHolds } from '../../scripts/run-roots.mjs'; + +const uid = process.getuid?.() ?? null; +const complete = { alive: () => false, startedAfter: () => false, completeExit: () => true }; +function fixture(t) { + const tmpdir = tempDir('ak-root-fixture', t); + const root = fs.mkdtempSync(path.join(tmpdir, 'ak-suite-')); + const homedir = path.join(tmpdir, 'home'); + fs.mkdirSync(homedir); + writeOwner(root, ownerRecord()); + return { root, tmpdir, homedir, uid }; +} +function collect(f, probes = complete, extra = {}) { + return collectAbandonedRoots({ ...f, probes, log: () => {}, ...extra }); +} + +test('private atomic owner metadata round trips and does not leave staging data', (t) => { + const f = fixture(t); + const record = readOwner(f.root); + assert.equal(record.pid, process.pid); + assert.equal(record.root, f.root); + assert.equal(record.tmpdir, f.tmpdir); + assert.equal(record.proofMode, 'list-only'); + assert.deepEqual(fs.readdirSync(f.root), [OWNER_FILE]); + if (process.platform !== 'win32') assert.equal(fs.statSync(path.join(f.root, OWNER_FILE)).mode & 0o777, 0o600); +}); + +test('parallel prelaunch holds release only their own marker; uncertain inspection retains', (t) => { + const f = fixture(t); + const owner = readOwner(f.root); + prepareRunRootHolds(f.root, owner.runId); + const env = { AK_SUITE_ROOT: f.root, AK_SUITE_RUN_ID: owner.runId }; + const first = acquireRunRootHold({ env }); + const second = acquireRunRootHold({ env }); + assert.equal(inspectRunRootHolds(f.root, owner.runId).unresolved, true); + assert.throws(() => releaseRunRootHold({ ...first, pid: 0 })); + const dir = path.join(f.root, '.ak-suite-holds'); + const saved = path.join(f.root, 'saved-holds'); + fs.renameSync(dir, saved); + try { + fs.symlinkSync(saved, dir, 'junction'); + assert.throws(() => releaseRunRootHold(first)); + } finally { + if (fs.existsSync(dir)) fs.unlinkSync(dir); + fs.renameSync(saved, dir); + } + releaseRunRootHold(first); + assert.equal(inspectRunRootHolds(f.root, owner.runId).unresolved, true); + releaseRunRootHold(second); + assert.equal(inspectRunRootHolds(f.root, owner.runId).unresolved, false); + assert.throws(() => acquireRunRootHold({ env: { ...env, AK_SUITE_RUN_ID: 'foreign' } })); + assert.equal(acquireRunRootHold({ env: {} }), null); + fs.rmSync(path.join(f.root, '.ak-suite-holds'), { recursive: true }); + assert.equal(inspectRunRootHolds(f.root, owner.runId).unresolved, true); + assert.throws(() => acquireRunRootHold({ env })); +}); + +test('native defaults never prove abandonment, including dead owners and reused PIDs', (t) => { + const f = fixture(t); + for (const platform of ['darwin', 'linux', 'win32', 'other']) { + assert.equal(proveAbandoned(f.root, readOwner(f.root), defaultProbes(platform)).abandoned, false); + assert.deepEqual(collect(f, defaultProbes(platform)).removed, []); + } + for (const probes of [ { ...complete, alive: () => true }, { ...complete, alive: () => null }, + { ...complete, completeExit: () => null }, { ...complete, completeExit: () => false }, + { ...complete, alive: () => true, startedAfter: () => null }, + { ...complete, alive: () => { throw Error('uncertain'); } } ]) { + assert.equal(collect(f, probes).kept.length, 1); + assert.ok(fs.existsSync(f.root)); + } +}); + +test('only explicit complete fixture proof permits deletion; reused PID needs independent descendant proof', (t) => { + const f = fixture(t); + fs.writeFileSync(path.join(f.root, 'payload'), 'fixture'); + assert.deepEqual(collect(f, { ...complete, alive: () => true, startedAfter: () => true }).removed, [f.root]); + assert.equal(fs.existsSync(f.root), false); +}); + +test('missing, malformed, oversized, foreign and path-mismatched owners stay listed', (t) => { + const f = fixture(t); + const record = readOwner(f.root); + const file = path.join(f.root, OWNER_FILE); + const values = ['{', 'x'.repeat(8193), 'null', '[]', JSON.stringify({ ...record, schema: 2 }), + ...[{ pid: 0 }, { startedAt: -1 }, { hostname: 'foreign' }, { uid: 123456789 }, + { platform: 'foreign' }, { root: f.tmpdir }, { tmpdir: f.root }, { proofMode: 'delete' }, + { runId: '' }].map((patch) => JSON.stringify({ ...record, ...patch }))]; + fs.unlinkSync(file); + assert.equal(collect(f).kept.length, 1); + for (const value of values) { + fs.writeFileSync(file, value); + assert.equal(readOwner(f.root), null, value.slice(0, 100)); + assert.equal(collect(f).kept.length, 1); + assert.ok(fs.existsSync(f.root)); + } +}); + +test('symlink owner files and directory owners are never read', (t) => { + const f = fixture(t); + const file = path.join(f.root, OWNER_FILE); + fs.renameSync(file, path.join(f.tmpdir, 'record')); + fs.symlinkSync(path.join(f.tmpdir, 'record'), file); + assert.equal(readOwner(f.root), null); + assert.equal(collect(f).kept.length, 1); + fs.unlinkSync(file); fs.mkdirSync(file); + assert.equal(readOwner(f.root), null); +}); + +test('unsafe home and filesystem-root temp parents are refused for POSIX and Windows', () => { + for (const [tmp, home] of [['/', '/home/me'], ['/home/me', '/home/me'], ['C:\\', 'C:\\Users\\me'], + ['C:\\Users\\ME', 'c:\\users\\me'], ['\\\\server\\share\\', 'C:\\Users\\me']]) { + assert.ok(unsafeTempBase(tmp, home)); + } + assert.equal(unsafeTempBase('/tmp', '/home/me'), null); + assert.equal(unsafeTempBase('C:\\Temp', 'C:\\Users\\me'), null); +}); + +test('root guards reject path escapes, aliases, non-direct children, wrong owners and symlinks', (t) => { + const f = fixture(t); + assert.equal(removableRunRoot(f.root, { ...f, requireOwner: true }).ok, true); + for (const dir of ['relative', f.tmpdir, path.join(f.root, 'ak-suite-AAAAAA'), + `${f.tmpdir}/x/../${path.basename(f.root)}`]) { + assert.equal(removableRunRoot(dir, f).ok, false); + } + assert.equal(removableRunRoot(f.root, { ...f, tmpdir: f.homedir }).ok, false); + assert.equal(removableRunRoot(f.root, { ...f, homedir: f.tmpdir }).ok, false); + if (uid !== null) assert.equal(removableRunRoot(f.root, { ...f, uid: uid + 1 }).ok, false); + const alias = path.join(f.tmpdir, 'ak-suite-AAAAAA'); + fs.symlinkSync(f.root, alias, 'junction'); + fs.mkdirSync(path.join(f.tmpdir, 'ak-suite-not-exact')); + assert.equal(removableRunRoot(alias, f).ok, false); + const r = collect(f, complete, { selfRoot: f.root }); + assert.deepEqual(r.removed, []); + assert.equal(r.kept.length, 1); + assert.ok(fs.existsSync(f.root)); +}); + +test('revalidation refuses root replacement during the proof', (t) => { + const f = fixture(t); + const r = collect(f, { ...complete, completeExit: () => { + fs.renameSync(f.root, `${f.root}-saved`); + fs.mkdirSync(f.root); writeOwner(f.root, ownerRecord()); + return true; + } }); + assert.deepEqual(r.removed, []); + assert.match(r.kept[0].reason, /changed/); + assert.ok(fs.existsSync(f.root)); +}); + +test('removal errors disclose possible partial deletion and listing errors are reported', (t) => { + const f = fixture(t); + fs.writeFileSync(path.join(f.root, 'payload'), 'fixture'); + const r = collect(f, complete, { remove: (root) => { + fs.unlinkSync(path.join(root, 'payload')); throw Error('EBUSY'); + } }); + assert.deepEqual(r.removed, []); + assert.match(r.kept[0].reason, /partially removed.*EBUSY/); + assert.equal(fs.existsSync(path.join(f.root, 'payload')), false); + assert.equal(collect({ ...f, tmpdir: path.join(f.tmpdir, 'absent') }).kept.length, 1); +}); + +test('default probe functions explicitly report unknown without supplying authority', () => { + const probes = defaultProbes(); + assert.equal(probes.alive(process.pid), null); + assert.equal(probes.startedAfter(process.pid, Date.now()), null); + assert.equal(probes.completeExit('/unused', {}), null); +}); + +test('owner publication rejects invalid attribution and refuses existing staging links', (t) => { + const f = fixture(t); + assert.throws(() => writeOwner(`${f.root}/.`, ownerRecord()), /noncanonical/); + assert.throws(() => writeOwner(f.root, ownerRecord({ pid: 0 })), /invalid/); + const target = path.join(f.tmpdir, 'target'); + fs.writeFileSync(target, 'untouched'); + fs.symlinkSync(target, path.join(f.root, `${OWNER_FILE}.tmp`)); + assert.throws(() => writeOwner(f.root, ownerRecord()), /EEXIST/); + assert.equal(fs.readFileSync(target, 'utf8'), 'untouched'); +}); + +test('owner reader refuses oversized or replaced files at its descriptor boundary', (t) => { + const f = fixture(t); + const realFstat = fs.fstatSync; + const realRead = fs.readSync; + for (const patch of [{ size: 8193 }, { ino: -1 }, { isFile: () => false }]) { + const mock = t.mock.method(fs, 'fstatSync', (...args) => Object.assign(realFstat(...args), patch)); + assert.equal(readOwner(f.root), null); + mock.mock.restore(); + } + const mock = t.mock.method(fs, 'readSync', (...args) => { realRead(...args); return 8193; }); + assert.equal(readOwner(f.root), null); + mock.mock.restore(); + const file = path.join(f.root, OWNER_FILE); + const hardlink = path.join(f.tmpdir, 'hardlink'); + fs.linkSync(file, hardlink); + assert.equal(readOwner(f.root), null); +}); + +test('inspection races retain roots before any removal attempt', (t) => { + const f = fixture(t); + const original = fs.lstatSync; + let calls = 0; + const mock = t.mock.method(fs, 'lstatSync', (...args) => { + if (args[0] === f.root && ++calls === 2) throw Error('inspection denied'); + return original(...args); + }); + const r = collect(f); + assert.deepEqual(r.removed, []); + assert.match(r.kept[0].reason, /inspection.*failed/); + mock.mock.restore(); + const ownerFile = path.join(f.root, OWNER_FILE); + let reads = 0; + const mock2 = t.mock.method(fs, 'lstatSync', (...args) => { + if (args[0] === ownerFile && ++reads === 3) throw Error('owner vanished'); + return original(...args); + }); + assert.match(collect(f).kept[0].reason, /owner changed/); + mock2.mock.restore(); + assert.ok(fs.existsSync(f.root)); +}); + +test('default collection keeps valid roots without injected probes', (t) => { + const f = fixture(t); + const messages = []; + const r = collectAbandonedRoots({ ...f, log: (s) => messages.push(s) }); + assert.deepEqual(r.removed, []); + assert.match(messages[0], /list-only/); +}); + +test('missing roots and noncanonical parents fail closed', (t) => { + const f = fixture(t); + fs.rmSync(f.root, { recursive: true }); + assert.equal(removableRunRoot(f.root, f).ok, false); + assert.equal(removableRunRoot(f.root, { ...f, tmpdir: 'relative' }).ok, false); +}); + +test('platforms without numeric uid retain attributable roots by default', (t) => { + const original = Object.getOwnPropertyDescriptor(process, 'getuid'); + Object.defineProperty(process, 'getuid', { value: undefined, configurable: true }); + t.after(() => { if (original) Object.defineProperty(process, 'getuid', original); }); + const f = fixture(t); + assert.equal(readOwner(f.root).uid, null); + assert.deepEqual(collectAbandonedRoots({ ...f, uid: null, log: () => {} }).removed, []); +}); + +test('logging failure cannot turn a completed removal into an intact-preservation claim', (t) => { + const f = fixture(t); + assert.doesNotThrow(() => { + const result = collectAbandonedRoots({ ...f, probes: complete, log: () => { throw Error('closed pipe'); } }); + assert.deepEqual(result.removed, [f.root]); + assert.deepEqual(result.kept, []); + }); +}); + +test('direct abandonment proof refuses malformed metadata even with complete injected probes', (t) => { + const f = fixture(t); + for (const record of [null, {}, { ...readOwner(f.root), pid: -1 }]) { + assert.equal(proveAbandoned(f.root, record, complete).abandoned, false); + } +}); diff --git a/tests/kit/run-tests-cancellation.test.mjs b/tests/kit/run-tests-cancellation.test.mjs new file mode 100644 index 00000000..881edc46 --- /dev/null +++ b/tests/kit/run-tests-cancellation.test.mjs @@ -0,0 +1,120 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { spawnSync } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; + +const runner = fileURLToPath(new URL('../../scripts/run-tests.mjs', import.meta.url)); +const sandbox = new URL('./helpers/home-sandbox.mjs', import.meta.url).href; +const helper = new URL('./helpers/interruption-scope.mjs', import.meta.url).href; + +test('real test timeout closes owned runner and orphan before cwd removal and releases the outer hold', t => { + const home = tempDir('ak-cancel-check', t); + const evidence = path.join(home, 'evidence.json'); + const fixture = path.join(home, 'cancel.test.mjs'); + fs.writeFileSync(fixture, ` + import { test, after } from 'node:test'; + import assert from 'node:assert/strict'; + import fs from 'node:fs'; + import { spawn } from 'node:child_process'; + import { spawnEnv } from ${JSON.stringify(sandbox)}; + import { interruptionScope, childGone, until } from ${JSON.stringify(helper)}; + let scope, child, closed = false, removed = false; + test('intentional cancellation', { timeout: 500 }, async t => { + // Node 22 unrefs its test timeout; keep this fixture alive until it fires. + // This owned handle is bounded even if cancellation cleanup fails. + const keepAlive = setTimeout(() => {}, 10000); + t.after(() => clearTimeout(keepAlive)); + scope = interruptionScope(t, { beforeRemove(home) { + assert.equal(closed, true, 'close must precede cwd removal'); + assert.equal(childGone(child.pid), true, 'OS must report ESRCH before removal'); + assert.equal(fs.existsSync(home), true); + const orphan = JSON.parse(fs.readFileSync(home + '/ready', 'utf8')); + assert.equal(childGone(orphan.pid), true, 'orphan must also have native ESRCH before removal'); + fs.writeFileSync(${JSON.stringify(evidence)}, JSON.stringify({ pid: child.pid, orphanPid: orphan.pid, closed, gone: childGone(child.pid), home })); + removed = true; + }}); + fs.writeFileSync(scope.home + '/hold.mjs', \` + import fs from 'node:fs'; + fs.writeFileSync(process.argv[2] + '.tmp', JSON.stringify({ pid: process.pid })); + fs.renameSync(process.argv[2] + '.tmp', process.argv[2]); + setInterval(() => { if (fs.existsSync(process.argv[3])) process.exit(0); }, 25); + setTimeout(() => process.exit(8), 10000); + \`); + const env = spawnEnv(scope.home); + delete env.NODE_TEST_CONTEXT; + child = scope.launch(() => spawn(process.execPath, [${JSON.stringify(runner)}, 'exec', '--repo', scope.home, '--', scope.home + '/hold.mjs', scope.home + '/ready', scope.home + '/stop'], { env, stdio: 'ignore' }), + { handshake: scope.home + '/ready', stop: scope.home + '/stop' }); + child.once('close', () => { closed = true; }); + try { + await until(() => fs.existsSync(scope.home + '/ready'), 'orphan readiness', 5000, t.signal); + child.kill('SIGTERM'); + await until(() => closed, 'runner close', 5000, t.signal); + await new Promise(resolve => t.signal.addEventListener('abort', resolve, { once: true })); + } + finally { + await scope.cleanup(); + let attempted = false; + assert.throws(() => scope.launch(() => { attempted = true; return child; })); + assert.equal(attempted, false, 'cancellation must block the launch callback'); + } + }); + after(() => { + assert.equal(removed, true); + assert.equal(fs.existsSync(scope.home), false); + }); + `); + const env = spawnEnv(home, { CI: 'true' }); + delete env.NODE_TEST_CONTEXT; + const result = spawnSync(process.execPath, [runner, 'exec', '--repo', home, '--', '--test', fixture], { + env, encoding: 'utf8', + }); + assert.equal(result.error, undefined); + assert.equal(result.status, 1, result.stdout + result.stderr); + assert.match(result.stdout + result.stderr, /test timed out after 500ms/); + const proof = JSON.parse(fs.readFileSync(evidence, 'utf8')); + assert.equal(proof.closed, true); + assert.equal(proof.gone, true); + assert.equal(fs.existsSync(proof.home), false); + assert.throws(() => process.kill(proof.pid, 0), { code: 'ESRCH' }); + assert.throws(() => process.kill(proof.orphanPid, 0), { code: 'ESRCH' }); + assert.deepEqual(fs.readdirSync(env.TMPDIR), [], 'guarded timeout has no unresolved hold or fixture leftovers'); +}); + +test('uncertain fixture identity retains its local directory and enclosing guarded root', t => { + const home = tempDir('ak-cancel-retain-check', t); + const evidence = path.join(home, 'evidence.json'); + const fixture = path.join(home, 'uncertain.test.mjs'); + fs.writeFileSync(fixture, ` + import { test } from 'node:test'; + import fs from 'node:fs'; + import { spawn } from 'node:child_process'; + import { interruptionScope } from ${JSON.stringify(helper)}; + test('intentional missing child identity', async t => { + const scope = interruptionScope(t); + const child = scope.launch(() => spawn(process.execPath, ['-e', 'process.exit(0)'], { cwd: scope.home, stdio: 'ignore' }), + { handshake: scope.home + '/missing-handshake', stop: scope.home + '/stop' }); + await new Promise(resolve => child.once('close', resolve)); + fs.writeFileSync(${JSON.stringify(evidence)}, JSON.stringify({ pid: child.pid, home: scope.home, root: process.env.AK_SUITE_ROOT })); + await scope.cleanup(); + }); + `); + const env = spawnEnv(home, { CI: 'true' }); + delete env.NODE_TEST_CONTEXT; + const result = spawnSync(process.execPath, [runner, 'exec', '--repo', home, '--', '--test', fixture], { env, encoding: 'utf8' }); + assert.equal(result.error, undefined); + assert.equal(result.status, 1, result.stdout + result.stderr); + assert.match(result.stdout + result.stderr, /owned process exit uncertain/); + assert.match(result.stderr, /unresolved child hold/); + const proof = JSON.parse(fs.readFileSync(evidence, 'utf8')); + assert.throws(() => process.kill(proof.pid, 0), { code: 'ESRCH' }); + assert.equal(fs.existsSync(proof.home), true); + assert.equal(fs.existsSync(proof.root), true); + assert.equal(fs.readdirSync(path.join(proof.root, '.ak-suite-holds')).length, 1); + // This enclosing test knows its exact fixture ran only process.exit(0), and + // independently established ESRCH above; remove only that disposable root. + fs.rmSync(proof.root, { recursive: true }); +}); diff --git a/tests/kit/run-tests-interruption.test.mjs b/tests/kit/run-tests-interruption.test.mjs new file mode 100644 index 00000000..2f36f5f8 --- /dev/null +++ b/tests/kit/run-tests-interruption.test.mjs @@ -0,0 +1,112 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { spawn, spawnSync } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; +import { interruptionScope, childGone } from './helpers/interruption-scope.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..'); +const RUNNER = path.join(ROOT, 'scripts', 'run-tests.mjs'); +const ownerFile = '.ak-suite-owner.json'; + +function roots(parent) { + return fs.readdirSync(parent).filter((name) => /^ak-suite-[A-Za-z0-9]{6}$/.test(name)).sort(); +} + +function runnerFor(repo, script, handshake, stop, done, env, log) { + const fd = fs.openSync(log, 'w'); + try { + return spawn(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', script, handshake, stop, done], { + env, stdio: ['ignore', fd, fd], + }); + } finally { fs.closeSync(fd); } +} + +test('interrupted and live sibling roots stay listed, including after the orphan exits', { + skip: process.platform === 'win32' && 'POSIX runner termination probe; Windows retains siblings by list-only policy', + timeout: 20000, +}, async (t) => { + const scope = interruptionScope(t); + const { home, wait: until } = scope; + const repo = path.join(home, 'repo'); + fs.mkdirSync(path.join(repo, '.git'), { recursive: true }); + const env = spawnEnv(home, { APPDATA: path.join(home, 'AppData', 'Roaming'), CI: 'true' }); + delete env.NODE_TEST_CONTEXT; + const parent = env.TMPDIR; + const baseline = fs.readdirSync(parent); + const script = path.join(home, 'hold.mjs'); + fs.writeFileSync(script, `import fs from 'node:fs'; + const [handshake, stop, done] = process.argv.slice(2); + fs.writeFileSync(handshake + '.tmp', JSON.stringify({ pid: process.pid, cwd: process.cwd(), tmpdir: process.env.TMPDIR })); + fs.renameSync(handshake + '.tmp', handshake); + const timer = setInterval(() => { + if (!fs.existsSync(stop)) return; + const request = fs.readFileSync(stop, 'utf8'); + if (!fs.existsSync(done)) fs.writeFileSync(done, 'stopped'); + if (request === 'exit') { clearInterval(timer); process.exit(0); } + }, 25); + setTimeout(() => process.exit(8), 12000);`); + const clean = path.join(home, 'clean.test.mjs'); + fs.writeFileSync(clean, "import { test } from 'node:test'; test('clean', () => {});"); + const first = ['first-handshake', 'first-stop', 'first-done'].map((name) => path.join(home, name)); + const live = ['live-handshake', 'live-stop', 'live-done'].map((name) => path.join(home, name)); + let runner1; + let runner4; + let interruptedRoot; + try { + runner1 = scope.launch(() => runnerFor(repo, script, ...first, env, path.join(home, 'run1.log')), + { handshake: first[0], stop: first[1] }); + const firstData = await until(() => fs.existsSync(first[0]) && JSON.parse(fs.readFileSync(first[0], 'utf8')), 'first child handshake'); + const root1 = await until(() => roots(parent).map((name) => path.join(parent, name)).find((root) => { + try { return JSON.parse(fs.readFileSync(path.join(root, ownerFile), 'utf8')).pid === runner1.pid; } + catch { return false; } + }), 'first owner record'); + interruptedRoot = root1; + assert.equal(firstData.cwd, repo); + assert.equal(firstData.tmpdir, root1); + assert.ok(firstData.pid > 0); + assert.equal(runner1.kill('SIGTERM'), true); + await until(() => runner1.exitCode !== null || runner1.signalCode !== null, 'owned runner termination'); + assert.ok(fs.existsSync(root1)); + assert.equal(fs.existsSync(first[2]), false, 'orphan has not stopped'); + + const runFocus = () => { + scope.active(); + return spawnSync(process.execPath, [RUNNER, 'focus', clean], { env, encoding: 'utf8' }); + }; + const second = runFocus(); + assert.equal(second.status, 0, second.stdout + second.stderr); + assert.match(second.stderr, /kept run root.*cannot prove complete descendant exit \(list-only\)/); + assert.deepEqual(roots(parent), [path.basename(root1)]); + + runner4 = scope.launch(() => runnerFor(repo, script, ...live, env, path.join(home, 'run4.log')), + { handshake: live[0], stop: live[1] }); + const liveData = await until(() => fs.existsSync(live[0]) && JSON.parse(fs.readFileSync(live[0], 'utf8')), 'live child handshake'); + const root4 = await until(() => roots(parent).map((name) => path.join(parent, name)).find((root) => { + try { return JSON.parse(fs.readFileSync(path.join(root, ownerFile), 'utf8')).pid === runner4.pid; } + catch { return false; } + }), 'live owner record'); + assert.equal(liveData.tmpdir, root4); + fs.writeFileSync(first[1], 'stop'); + await until(() => fs.existsSync(first[2]), 'orphan private stop acknowledgement'); + assert.equal(childGone(firstData.pid), false, 'acknowledgement precedes child exit'); + fs.writeFileSync(first[1], 'exit'); + await until(() => childGone(firstData.pid), 'orphan PID absent after private exit request'); + assert.equal(childGone(liveData.pid), false, 'concurrent child remains alive'); + const third = runFocus(); + assert.equal(third.status, 0, third.stdout + third.stderr); + assert.match(third.stderr, /kept run root.*cannot prove complete descendant exit \(list-only\)/); + assert.deepEqual(roots(parent), [path.basename(root1), path.basename(root4)].sort()); + assert.ok(fs.existsSync(root1), 'known stopped child does not authorize sibling removal'); + assert.ok(fs.existsSync(root4), 'concurrently live sibling remains'); + } finally { + await scope.cleanup(); + } + if (t.signal.aborted) return; + // Only this test's disposable fixture root is removed after its child stops. + assert.deepEqual(roots(parent), [path.basename(interruptedRoot)]); + fs.rmSync(interruptedRoot, { recursive: true }); + assert.deepEqual(fs.readdirSync(parent), baseline); +}); diff --git a/tests/kit/run-tests-runner.test.mjs b/tests/kit/run-tests-runner.test.mjs index fb74f826..cfa70811 100644 --- a/tests/kit/run-tests-runner.test.mjs +++ b/tests/kit/run-tests-runner.test.mjs @@ -2,27 +2,203 @@ import { test } from 'node:test'; import assert from 'node:assert/strict'; import fs from 'node:fs'; -import os from 'node:os'; import path from 'node:path'; import { spawnSync } from 'node:child_process'; import { fileURLToPath } from 'node:url'; +import { tempDir } from './helpers/temp-dir.mjs'; import { spawnEnv } from './helpers/home-sandbox.mjs'; const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..'); const RUNNER = path.join(ROOT, 'scripts', 'run-tests.mjs'); function sandbox(t) { - const home = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'ak-runner-'))); - t.after(() => fs.rmSync(home, { recursive: true, force: true })); + const home = tempDir('ak-runner', t); const repo = path.join(home, 'repo'); fs.mkdirSync(path.join(repo, '.git'), { recursive: true }); const env = spawnEnv(home, { APPDATA: path.join(home, 'AppData', 'Roaming'), CI: 'true' }); delete env.AK_TRIPWIRE_STRICT; + // Nested CLI fixtures must run as independent test processes. + delete env.NODE_TEST_CONTEXT; return { home, repo, env }; } const stub = (dir, name, body) => { const f = path.join(dir, name); fs.writeFileSync(f, body); return f; }; +async function pidGone(pid, timeoutMs = 5_000) { + const end = Date.now() + timeoutMs; + while (Date.now() < end) { + try { process.kill(pid, 0); } catch (error) { if (error.code === 'ESRCH') return true; } + await new Promise((resolve) => setTimeout(resolve, 20)); + } + return false; +} + +test('focus runs a literal clean test file through the guarded root', (t) => { + const { home, env } = sandbox(t); + const file = stub(home, 'clean.test.mjs', "import { test } from 'node:test'; test('clean', () => {});"); + const before = fs.readdirSync(env.TMPDIR); + const r = spawnSync(process.execPath, [RUNNER, 'focus', file], { env, encoding: 'utf8' }); + assert.equal(r.status, 0, r.stdout + r.stderr); + assert.match(r.stderr, /real-state tripwire: watching/); + assert.match(r.stderr, /removed own run root /); + assert.deepEqual(fs.readdirSync(env.TMPDIR), before); +}); + +test('focus reports a leaked temp folder with hygiene exit code', (t) => { + const { home, env } = sandbox(t); + const file = stub(home, 'leaky.test.mjs', `import { test } from 'node:test'; + import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; + test('leak', () => { fs.mkdtempSync(path.join(os.tmpdir(), 'focus-leak-')); });`); + const r = spawnSync(process.execPath, [RUNNER, 'focus', file], { env, encoding: 'utf8' }); + assert.equal(r.status, 4, r.stdout + r.stderr); + assert.match(r.stderr, /temp folders left behind.*focus-leak-/s); +}); + +async function removeExitedFixture(root) { + for (let attempt = 0; ; attempt++) { + try { fs.rmSync(root, { recursive: true, force: true }); return; } + catch (error) { + // Match the fixture helpers' three-retry policy without depending on + // Node's JS/C++ rm implementation. Exit has already been established. + if (!['EBUSY', 'EPERM', 'ENOTEMPTY', 'EEXIST'].includes(error.code) || attempt === 3) throw error; + await new Promise(resolve => setTimeout(resolve, (attempt + 1) * 100)); + } + } +} + +async function heldRootFixture(t, { check = () => {}, beforeRemove = () => () => {} } = {}) { + const { home, repo, env } = sandbox(t); + const pidFile = path.join(home, 'held-child-pid'); + const child = stub(home, 'held-child.cjs', `const fs = require('node:fs'); + const release = process.argv[2]; + fs.writeFileSync(require('node:path').join(process.cwd(), 'held-child-ready'), 'ready'); + const timer = setInterval(() => { if (fs.existsSync(release)) { clearInterval(timer); process.exit(0); } }, 20); + setTimeout(() => process.exit(2), 10000);`); + const script = stub(home, 'unresolved-hold.mjs', `import fs from 'node:fs'; + import path from 'node:path'; import { spawn } from 'node:child_process'; + import { acquireRunRootHold } from ${JSON.stringify(new URL('../../scripts/run-roots.mjs', import.meta.url).href)}; + acquireRunRootHold(); + const root = process.env.AK_SUITE_ROOT; + fs.writeFileSync(path.join(root, 'held-sentinel'), 'keep'); + const child = spawn(process.execPath, [${JSON.stringify(child)}, path.join(root, 'release-child')], + { cwd: root, stdio: 'ignore', detached: true }); + child.once('error', (error) => { throw error; }); + fs.writeFileSync(${JSON.stringify(pidFile)}, String(child.pid)); + child.unref(); + const ready = path.join(root, 'held-child-ready'); + const deadline = Date.now() + 2000; + while (!fs.existsSync(ready) && Date.now() < deadline) await new Promise((resolve) => setTimeout(resolve, 10)); + if (!fs.existsSync(ready)) throw Error('fixture child did not become ready'); + process.exit(7);`); + let root; + let pid; + let failure; + try { + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', script], { env, encoding: 'utf8' }); + const roots = fs.readdirSync(env.TMPDIR).filter((name) => /^ak-suite-[A-Za-z0-9]{6}$/.test(name)); + if (roots.length === 1) root = path.join(env.TMPDIR, roots[0]); + if (fs.existsSync(pidFile)) pid = Number(fs.readFileSync(pidFile, 'utf8')); + assert.equal(r.status, 7, r.stdout + r.stderr); + assert.match(r.stderr, /kept own run root .*unresolved child hold/); + assert.equal(roots.length, 1, r.stderr); + assert.equal(fs.readFileSync(path.join(root, 'held-sentinel'), 'utf8'), 'keep'); + assert.equal(fs.readFileSync(path.join(root, 'held-child-ready'), 'utf8'), 'ready'); + assert.doesNotThrow(() => process.kill(pid, 0), 'the held child should still own the retained cwd'); + check(); + } catch (error) { failure = error; } + try { + if (root) fs.writeFileSync(path.join(root, 'release-child'), 'release'); + if (Number.isInteger(pid) && pid > 0) { + if (!root) { try { process.kill(pid); } catch { /* already exited */ } } + if (!await pidGone(pid)) { try { process.kill(pid); } catch { /* already exited */ } } + assert.ok(await pidGone(pid), `fixture child ${pid} did not exit`); + } + if (root) { + const restore = beforeRemove(root, pid); + // Exit proof above is required; retries only address post-exit filesystem refusal. + try { await removeExitedFixture(root); } + finally { restore(); } + } + } catch (error) { + failure = failure ? new AggregateError([failure, error], 'fixture assertion and cleanup failed') : error; + } + if (failure) throw failure; +} + +test('an unresolved prelaunch hold retains the guarded root and sentinel after test failure', heldRootFixture); + +function injectBusyRemoval(limit) { + let calls = 0; + return { + calls: () => calls, + beforeRemove(root, pid) { + const original = fs.rmSync; + fs.rmSync = (dir, ...args) => { + if (String(dir) === root) { + assert.throws(() => process.kill(pid, 0), { code: 'ESRCH' }); + if (++calls <= limit) throw Object.assign(new Error('injected post-exit EBUSY'), { code: 'EBUSY' }); + } + return original(dir, ...args); + }; + return () => { fs.rmSync = original; }; + }, + }; +} + +test('post-exit fixture removal retries transient EBUSY', async t => { + const busy = injectBusyRemoval(2); + await heldRootFixture(t, busy); + assert.equal(busy.calls(), 3); +}); + +test('exhausted post-exit retries preserve both assertion and cleanup failures', async t => { + const busy = injectBusyRemoval(Infinity); + const original = new assert.AssertionError({ message: 'original fixture assertion' }); + await assert.rejects(heldRootFixture(t, { ...busy, check: () => { throw original; } }), error => { + assert.ok(error instanceof AggregateError); + assert.equal(error.errors[0], original); + assert.equal(error.errors[1].code, 'EBUSY'); + return true; + }); + assert.equal(busy.calls(), 4, 'one attempt plus three bounded retries'); +}); + +test('unresolved holds preserve command, tripwire, then hygiene exit priority', (t) => { + for (const [mode, expected] of [['command', 7], ['tripwire', 3], ['hygiene', 4]]) { + const { home, repo, env } = sandbox(t); + const script = stub(home, `${mode}-hold.mjs`, `import fs from 'node:fs'; import path from 'node:path'; + import { acquireRunRootHold } from ${JSON.stringify(new URL('../../scripts/run-roots.mjs', import.meta.url).href)}; + acquireRunRootHold(); + if (process.env.HOLD_MODE === 'tripwire') { + const file = path.join(process.env.XDG_CONFIG_HOME, 'agentic-kit', 'kit.json'); + fs.mkdirSync(path.dirname(file), { recursive: true }); fs.writeFileSync(file, '{}'); + } + if (process.env.HOLD_MODE === 'command') process.exit(7);`); + let root; + try { + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', script], { + env: { ...env, HOLD_MODE: mode }, encoding: 'utf8', + }); + const roots = fs.readdirSync(env.TMPDIR).filter((name) => /^ak-suite-[A-Za-z0-9]{6}$/.test(name)); + if (roots.length === 1) root = path.join(env.TMPDIR, roots[0]); + assert.equal(r.status, expected, r.stdout + r.stderr); + assert.match(r.stderr, /kept own run root .*unresolved child hold/); + assert.equal(roots.length, 1); + } finally { if (root) fs.rmSync(root, { recursive: true, force: true }); } + } +}); + +test('focus rejects missing files and option-shaped filenames before creating roots', (t) => { + const { env } = sandbox(t); + const before = fs.readdirSync(env.TMPDIR); + for (const args of [[], ['--test-reporter=dot'], ['does-not-exist.test.mjs']]) { + const r = spawnSync(process.execPath, [RUNNER, 'focus', ...args], { env, encoding: 'utf8' }); + assert.equal(r.status, 2, r.stdout + r.stderr); + assert.match(r.stderr, /usage: run-tests\.mjs focus/); + assert.deepEqual(fs.readdirSync(env.TMPDIR), before); + } +}); + test('a command that writes real state fails the run and the path is named', (t) => { const { home, repo, env } = sandbox(t); const leak = stub(home, 'leak.mjs', `import fs from 'node:fs'; import path from 'node:path'; @@ -69,7 +245,7 @@ test('a clean command passes; concurrent-writer churn does not fail a developer assert.match(r.stderr, /concurrent writers \(not failing\)/); }); -test('SUITES keeps the exact commands package.json ran before', async () => { +test('SUITES preserves package scripts and includes the session-surfaces regression', async () => { const { SUITES } = await import('../../scripts/run-tests.mjs'); assert.deepEqual(SUITES.unit[0], ['--test', '--experimental-test-coverage', '--test-coverage-lines=70', '--test-coverage-branches=70', '--test-coverage-functions=70', 'tests/kit/*.test.mjs']); @@ -79,7 +255,7 @@ test('SUITES keeps the exact commands package.json ran before', async () => { assert.deepEqual(SUITES.ui[1], ['--test', 'tests/ui/dashboard-project-context.mjs', 'tests/ui/maintenance-projects.mjs', 'tests/ui/maintenance-host-alignment.mjs', 'tests/ui/intelligence-picker.mjs', 'tests/ui/usage-project-groups.mjs', 'tests/ui/context-coverage.mjs', 'tests/ui/host-readiness.mjs', 'tests/ui/maintenance-focus.mjs', - 'tests/ui/maintenance-guidance.mjs']); + 'tests/ui/maintenance-guidance.mjs', 'tests/ui/session-surfaces.mjs']); const pkg = JSON.parse(fs.readFileSync(path.join(ROOT, 'package.json'), 'utf8')); assert.equal(pkg.scripts.test, 'node scripts/run-tests.mjs unit'); assert.equal(pkg.scripts['test:ui'], 'node scripts/run-tests.mjs ui'); @@ -130,3 +306,182 @@ test('the suite runs without the shell FORCE_COLOR', (t) => { const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', probe], { env: { ...env, FORCE_COLOR: '3' }, encoding: 'utf8' }); assert.equal(r.status, 0, r.stdout + r.stderr); }); + +test('guarded runner preserves tool selectors while scrubbing FORCE_COLOR', (t) => { + const { home, repo, env } = sandbox(t); + const probe = stub(home, 'selectors.mjs', ` + import assert from 'node:assert/strict'; + assert.equal(process.env.AQE_EMBEDDER_PROVIDER, 'sentinel-provider'); + assert.equal(process.env.AQE_EMBEDDER_MODEL, 'sentinel-model'); + assert.equal(process.env.FORCE_COLOR, undefined); + `); + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', probe], { + env: { ...env, AQE_EMBEDDER_PROVIDER: 'sentinel-provider', AQE_EMBEDDER_MODEL: 'sentinel-model', FORCE_COLOR: '3' }, + encoding: 'utf8', + }); + assert.equal(r.status, 0, r.stdout + r.stderr); +}); + +test('home temp base is refused before creating a run root', (t) => { + const { home, repo, env } = sandbox(t); + const before = fs.readdirSync(home); + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', '-e', ''], { + env: { ...env, TMPDIR: home, TMP: home, TEMP: home }, encoding: 'utf8', + }); + assert.equal(r.status, 2, r.stderr); + assert.match(r.stderr, /unsafe temp base/); + assert.deepEqual(fs.readdirSync(home), before); +}); + +test('completed runners retain sibling roots and ignore their own owner metadata', async (t) => { + const { ownerRecord, writeOwner } = await import('../../scripts/run-roots.mjs'); + const { home, repo, env } = sandbox(t); + const parent = env.TMPDIR; + const sibling = fs.mkdtempSync(path.join(parent, 'ak-suite-')); + writeOwner(sibling, ownerRecord({ pid: process.pid })); + fs.writeFileSync(path.join(sibling, 'sentinel'), 'preserve'); + for (const exit of [0, 7]) { + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', '-e', `process.exit(${exit})`], { env, encoding: 'utf8' }); + assert.equal(r.status, exit, r.stderr); + assert.match(r.stderr, /kept run root.*cannot prove complete descendant exit/); + assert.doesNotMatch(r.stderr, /temp folders left behind/); + assert.equal(fs.readFileSync(path.join(sibling, 'sentinel'), 'utf8'), 'preserve'); + assert.deepEqual(fs.readdirSync(parent), [path.basename(sibling)]); + } + assert.ok(home); +}); + +test('a reaped owner does not authorize sibling deletion after a completed run', async (t) => { + const { ownerRecord, writeOwner } = await import('../../scripts/run-roots.mjs'); + const { repo, env } = sandbox(t); + const startedAt = Date.now(); + const child = spawnSync(process.execPath, ['-e', ''], { env }); + assert.equal(child.status, 0); + const sibling = fs.mkdtempSync(path.join(env.TMPDIR, 'ak-suite-')); + writeOwner(sibling, ownerRecord({ pid: child.pid, now: startedAt })); + const r = spawnSync(process.execPath, [RUNNER, 'exec', '--repo', repo, '--', '-e', ''], { env, encoding: 'utf8' }); + assert.equal(r.status, 0, r.stderr); + assert.match(r.stderr, /cannot prove complete descendant exit/); + assert.ok(fs.existsSync(sibling)); +}); + +test('native self-termination reports the platform-specific child result', (t) => { + const { env } = sandbox(t); + const result = spawnSync(process.execPath, ['-e', "process.kill(process.pid, 'SIGTERM')"], { env }); + assert.equal(result.error, undefined); + // Windows uv_kill uses TerminateProcess(1); only uv_process_kill records + // exit_signal on the parent's owned handle. A child's self-kill cannot do so. + const expected = process.platform === 'win32' + ? { status: 1, signal: null } : { status: null, signal: 'SIGTERM' }; + assert.deepEqual({ status: result.status, signal: result.signal }, expected); + assert.throws(() => process.kill(result.pid, 0), { code: 'ESRCH' }); + t.diagnostic(`native self-termination: ${JSON.stringify(expected)}`); +}); + +test('a native timeout signal retains its run root and performs no sibling collection', (t) => { + const { home, repo, env } = sandbox(t); + const sibling = fs.mkdtempSync(path.join(env.TMPDIR, 'ak-suite-')); + fs.writeFileSync(path.join(sibling, 'sentinel'), 'preserve'); + const evidence = path.join(home, 'signal-result.json'); + const command = stub(home, 'wait-for-signal.mjs', ` + import fs from 'node:fs'; import path from 'node:path'; + fs.writeFileSync(path.join(process.env.AK_SUITE_ROOT, 'sentinel'), 'preserve'); + setTimeout(() => {}, 10000); + `); + // Alter only the native spawn options in this isolated driver. The actual + // OS result is passed through unchanged; no signal/result is fabricated. + const driver = stub(home, 'owned-timeout.mjs', ` + import cp from 'node:child_process'; import fs from 'node:fs'; + import { syncBuiltinESMExports } from 'node:module'; + import { runGuarded } from ${JSON.stringify(new URL('../../scripts/run-tests.mjs', import.meta.url).href)}; + const nativeSpawn = cp.spawnSync; + cp.spawnSync = (file, args, options) => { + const result = nativeSpawn(file, args, { ...options, timeout: 1000, killSignal: 'SIGTERM' }); + fs.writeFileSync(${JSON.stringify(evidence)}, JSON.stringify({ + pid: result.pid, status: result.status, signal: result.signal, error: result.error?.code, + })); + return result; + }; + syncBuiltinESMExports(); + try { process.exitCode = runGuarded([[${JSON.stringify(command)}]], { repoRoot: ${JSON.stringify(repo)} }); } + finally { cp.spawnSync = nativeSpawn; syncBuiltinESMExports(); } + `); + const r = spawnSync(process.execPath, [driver], { env, encoding: 'utf8' }); + assert.equal(r.error, undefined); + assert.equal(r.status, 1, r.stderr); + const native = JSON.parse(fs.readFileSync(evidence, 'utf8')); + assert.deepEqual({ status: native.status, signal: native.signal, error: native.error }, + { status: null, signal: 'SIGTERM', error: 'ETIMEDOUT' }); + assert.throws(() => process.kill(native.pid, 0), { code: 'ESRCH' }); + assert.match(r.stderr, /interrupted run/); + assert.doesNotMatch(r.stderr, /^kept run root/m); + assert.equal(fs.readdirSync(env.TMPDIR).length, 2); + const own = fs.readdirSync(env.TMPDIR).map(name => path.join(env.TMPDIR, name)).find(root => root !== sibling); + assert.equal(fs.readFileSync(path.join(own, 'sentinel'), 'utf8'), 'preserve'); + assert.equal(fs.readFileSync(path.join(sibling, 'sentinel'), 'utf8'), 'preserve'); + t.diagnostic(`native timeout: ${JSON.stringify(native)}`); +}); + +// Inject failures only at this subprocess's disposable temp boundary. +function cleanupProbe(home) { + return stub(home, 'cleanup-errors.mjs', `import fs from 'node:fs'; + import os from 'node:os'; import path from 'node:path'; + import { runGuarded } from ${JSON.stringify(new URL('../../scripts/run-tests.mjs', import.meta.url).href)}; + const [repo, mode, code, tripwire] = process.argv.slice(2); + const parent = fs.realpathSync(os.tmpdir()); + const own = (p) => path.dirname(String(p)) === parent && /^ak-suite-[A-Za-z0-9]{6}$/.test(path.basename(String(p))); + const readdir = fs.readdirSync, remove = fs.rmSync, lstat = fs.lstatSync; + fs.readdirSync = (p, ...args) => { + if ((mode === 'inspection' && own(p)) || (mode === 'sibling' && p === parent)) throw Error('EACCES'); + return readdir(p, ...args); + }; + fs.rmSync = (p, ...args) => { + if (mode === 'removal' && own(p)) throw Error('EBUSY'); + return remove(p, ...args); + }; + let ownStats = 0; + fs.lstatSync = (p, ...args) => { + const stat = lstat(p, ...args); + if (own(p)) { + ownStats++; + if (mode === 'refusal' && ownStats >= 3) stat.isDirectory = () => false; + if (mode === 'identity' && ownStats >= 4) { + if (typeof stat.birthtimeNs === 'bigint') stat.birthtimeNs += 1n; + else stat.birthtimeMs += 1; + } + } + return stat; + }; + const child = "const fs=require('fs'),path=require('path'),os=require('os');" + + (mode === 'inspection' ? "fs.writeFileSync(path.join(os.tmpdir(),'leftover'),'retain me');" : '') + + (tripwire === 'yes' ? "fs.mkdirSync(path.join(process.env.XDG_CONFIG_HOME,'agentic-kit'),{recursive:true});fs.writeFileSync(path.join(process.env.XDG_CONFIG_HOME,'agentic-kit','kit.json'),'{}');" : '') + + 'process.exit(' + code + ')'; + process.exitCode = runGuarded([['-e', child]], { repoRoot: repo });`); +} + +for (const mode of ['inspection', 'removal', 'refusal', 'identity']) { + test(`own-root ${mode} failure fails hygiene and retains command/tripwire precedence`, (t) => { + for (const [commandCode, tripwire, expected] of [[0, 'no', 4], [7, 'no', 7], [0, 'yes', 3]]) { + const { home, repo, env } = sandbox(t); + const r = spawnSync(process.execPath, [cleanupProbe(home), repo, mode, String(commandCode), tripwire], { env, encoding: 'utf8' }); + assert.equal(r.status, expected, r.stderr); + const roots = fs.readdirSync(env.TMPDIR).filter((name) => /^ak-suite-[A-Za-z0-9]{6}$/.test(name)); + assert.equal(roots.length, 1, r.stderr); + if (mode === 'inspection') { + assert.match(r.stderr, /could not list own run root/); + assert.equal(fs.readFileSync(path.join(env.TMPDIR, roots[0], 'leftover'), 'utf8'), 'retain me'); + } else if (mode === 'removal') assert.match(r.stderr, /may be partially removed/); + else assert.match(r.stderr, /kept own run root/); + } + }); +} + +test('sibling listing failure stays nonfatal and does not mask command failure', (t) => { + for (const code of [0, 7]) { + const { home, repo, env } = sandbox(t); + const r = spawnSync(process.execPath, [cleanupProbe(home), repo, 'sibling', String(code), 'no'], { env, encoding: 'utf8' }); + assert.equal(r.status, code, r.stderr); + assert.match(r.stderr, /could not list run roots/); + assert.deepEqual(fs.readdirSync(env.TMPDIR), []); + } +}); diff --git a/tests/kit/session-presentation.test.mjs b/tests/kit/session-presentation.test.mjs new file mode 100644 index 00000000..23608787 --- /dev/null +++ b/tests/kit/session-presentation.test.mjs @@ -0,0 +1,72 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import * as presentation from '../../src/lib/session-surface.mjs'; +import { discoverProjectSources } from '../../src/lib/footprint/project-sources.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; +import fs from 'node:fs'; +import path from 'node:path'; + +test('shared presentation keeps legacy desktop mode unknown and rejects raw provider claims', () => { + assert.equal(typeof presentation.sessionPresentation, 'function'); + const legacy = presentation.sessionPresentation({ origin: 'codex-desktop' }); + assert.equal(legacy.surface, 'unknown'); + assert.equal(legacy.label, 'Unknown'); + assert.match(legacy.note, /ChatGPT desktop app.*mode.*not recorded/i); + assert.equal(presentation.sessionPresentation({ surface: 'claude-desktop', thirdPartyProvider: 'amazon-bedrock' }).provider, 'Unknown'); + assert.equal(presentation.sessionPresentation({ surface: 'claude-desktop', thirdPartyProvider: 'amazon-bedrock', thirdPartyProviderBasis: 'assistant-model-id' }).provider, 'Amazon Bedrock'); +}); + +test('project discovery preserves separate surfaces and bounded raw disagreements through JSON', (t) => { + const root = tempDir('ak-session-presentation-', t); + const claude = path.join(root, 'claude'); fs.mkdirSync(claude); + for (const [i, entrypoint] of ['cli', 'sdk-cli', 'future-a', 'future-b'].entries()) { + fs.writeFileSync(path.join(claude, `${i}.jsonl`), JSON.stringify({ type: 'user', sessionId: `id-${i}`, cwd: root, entrypoint })); + } + const census = discoverProjectSources({ claudeRoot: claude, codexRoot: path.join(root, 'absent'), opencodeDbFile: path.join(root, 'absent.db') }); + const row = JSON.parse(JSON.stringify(census.projects[0])); + assert.equal(row.sessions, 4); + assert.ok(Array.isArray(row.sessionSurfaces)); + assert.deepEqual(row.sessionSurfaces.map((entry) => entry.surface).sort(), ['claude-code-cli', 'claude-noninteractive', 'other-claude']); + const other = row.sessionSurfaces.find((entry) => entry.surface === 'other-claude'); + assert.deepEqual(other.rawEvidence.entrypoint, ['future-a', 'future-b']); + assert.equal(other.sessions, 2); + assert.equal(other.countBasis, 'declared-session-ids'); + assert.equal(other.initiator, 'unknown'); +}); +test('aggregation preserves declared attributes and separates supported provider observations', async () => { + const { mergeSessionSurfaces } = await import('../../src/lib/footprint/session-surfaces.mjs'); + const common = { host: 'claude', surface: 'claude-desktop', initiator: 'person', sessions: 1, + countBasis: 'declared-session-ids', attributes: ['on 3P'] }; + const rows = mergeSessionSurfaces([{ ...common, thirdPartyProvider: 'amazon-bedrock', thirdPartyProviderBasis: 'assistant-model-id' }, common]); + assert.equal(rows.length, 2); + assert.deepEqual(rows[0].attributes, ['on 3P']); + assert.equal(rows.filter((row) => row.thirdPartyProvider === 'amazon-bedrock').length, 1); +}); +test('bounded raw disagreements preserve an explicit incomplete flag, with order-independent values', async () => { + const { mergeSessionSurfaces } = await import('../../src/lib/footprint/session-surfaces.mjs'); + const entries = Array.from({ length: 20 }, (_, i) => ({ host: 'codex', surface: 'other-openai', initiator: 'unknown', + sessions: 1, countBasis: 'transcript-files', rawEvidence: { originator: `future-${i}` } })); + const forward = mergeSessionSurfaces(entries), reverse = mergeSessionSurfaces(entries.toReversed()); + assert.deepEqual(forward, reverse); + assert.equal(forward[0].rawEvidenceComplete, false); + assert.equal(forward[0].rawEvidence.originator.length, 16); + assert.equal(forward[0].sessions, 20); +}); +test('all import count fields survive collection and the summary; absent legacy fields stay unknown', async () => { + const { collectProjects } = await import('../../src/lib/footprint/projects.mjs'); + const { systemSummaryPayload } = await import('../../src/lib/dashboard/system-summary.mjs'); + const { censusDisclosure } = await import('../../src/lib/census-presentation.mjs'); + const projects = collectProjects({ sources: { projects: [], importedExcluded: 4, importedMixed: 2, + importedUnresolved: 3, everSeen: 0, onDisk: 0, gitRepos: 0, complete: false }, loc: false }); + const summary = systemSummaryPayload({ projects }).projects; + assert.equal(summary.importedExcluded, 4); assert.equal(summary.importedMixed, 2); assert.equal(summary.importedUnresolved, 3); + assert.match(censusDisclosure({}), /Unknown number of confirmed pure imported copies/); + assert.doesNotMatch(censusDisclosure({}), /0 confirmed/); +}); +test('surface aggregation tolerates malformed attributes and orders provider groups deterministically', async () => { + const { mergeSessionSurfaces } = await import('../../src/lib/footprint/session-surfaces.mjs'); + const base = { host: 'claude', surface: 'claude-desktop', sessions: 1, attributes: 'untrusted' }; + assert.doesNotThrow(() => mergeSessionSurfaces([base])); + const entries = [{ ...base, attributes: [], thirdPartyProvider: 'amazon-bedrock', thirdPartyProviderBasis: 'assistant-model-id' }, { ...base, attributes: [] }]; + assert.deepEqual(mergeSessionSurfaces(entries), mergeSessionSurfaces(entries.toReversed())); +}); diff --git a/tests/kit/session-surface-renderers.test.mjs b/tests/kit/session-surface-renderers.test.mjs new file mode 100644 index 00000000..33cf6399 --- /dev/null +++ b/tests/kit/session-surface-renderers.test.mjs @@ -0,0 +1,104 @@ +import { censusDisclosure } from '../../src/lib/census-presentation.mjs'; +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import vm from 'node:vm'; +import * as vocabulary from '../../src/lib/session-surface.mjs'; +const esc = (value) => String(value).replaceAll('&', '&').replaceAll('<', '<').replaceAll('>', '>').replaceAll('"', '"'); +function renderer(name, elements = {}) { + const context = vm.createContext({ ...vocabulary, censusDisclosure, MNT_CURATED_VIEW_LABELS: {}, esc, window: {}, document: { getElementById: (id) => elements[id] ?? null }, + fmtNum: String, kpi: () => '', ago: () => 'now', formatLocalDateTime: () => null }); + const read = (file) => fs.readFileSync(new URL(`../../src/lib/dashboard/client/${file}.mjs`, import.meta.url), 'utf8').replace(/^import .*;$/gm, '').replace(/\bexport /g, ''); + const helper = new URL('../../src/lib/dashboard/client/session-presentation.mjs', import.meta.url); + if (fs.existsSync(helper)) vm.runInContext(read('session-presentation'), context); + vm.runInContext(read(name), context); + return context; +} +const sessionOrigin = { surface: 'chatgpt-desktop-work', initiator: 'agent', rawEvidence: { originator: 'future-client' } }; +test('Usage detail renders independent surface, initiator and bounded raw evidence', () => { + const context = renderer('usage'); + const html = context.sdetail({ id: 'x', host: 'codex', sessionOrigin }); + assert.match(html, /ChatGPT desktop app · ChatGPT Work \(local\)/); + assert.match(html, /initiator.*Agent/); + assert.match(html, /future-client/); + const hostile = context.sdetail({ id: 'x', sessionOrigin: { surface: '', rawEvidence: { originator: '', source: 'vscode' } } }); + assert.doesNotMatch(hostile, / { + const elements = { 'mw-table': {}, 'mw-hero': {} }; + const context = renderer('intelligence', elements); + context.renderMachineWide({ totals: {}, perProject: [{ label: 'Repository', learningScope: 'repository', + sessionSurfaces: [{ ...sessionOrigin, host: 'codex', sessions: 1 }, { surface: 'cloud-session', initiator: 'automation', host: 'claude', sessions: 1 }] }] }); + assert.match(elements['mw-table'].innerHTML, /Git repository/); + assert.match(elements['mw-table'].innerHTML, /ChatGPT desktop app · ChatGPT Work \(local\)/); + assert.match(elements['mw-table'].innerHTML, /Cloud session/); + assert.match(elements['mw-table'].innerHTML, /Session surface/); +}); +test('Maintenance origin facet shares the same surface vocabulary and honest legacy fallback', () => { + const context = renderer('maintenance-filters'); + assert.equal(context.mntFacetValueLabel('sessionOrigin', 'chatgpt-desktop-work'), 'ChatGPT desktop app · ChatGPT Work (local)'); + assert.equal(context.mntFacetValueLabel('sessionOrigin', 'codex-desktop'), 'ChatGPT desktop app observed; mode not recorded in this legacy snapshot'); + assert.equal(context.mntFacetValueLabel('sessionOrigin', 'unknown'), 'Legacy origin: no declared desktop origin'); + assert.equal(context.mntFacetValueLabel('sessionOrigin', 'surface-unknown'), 'Unknown session surface'); +}); +test('Intelligence and System disclose pure, mixed and unresolved import counts, including an empty project census', () => { + const counts = { importedExcluded: 4, importedMixed: 2, importedUnresolved: 3 }; + const elements = { 'mw-census': {}, 'mw-census-body': {}, 'sys-projects': {} }; + const intel = renderer('intelligence', elements); + intel.renderCensus({ counts }); + const system = renderer('system-projects', elements); + system.sysEmpty = (text) => text; + system.renderSysProjects({ projects: { ...counts, projects: [], discoveryProjects: [] } }); + for (const id of ['mw-census-body', 'sys-projects']) { + assert.match(elements[id].innerHTML, /4 confirmed pure imported copies excluded/); + assert.match(elements[id].innerHTML, /2 mixed files retain proven native activity/); + assert.match(elements[id].innerHTML, /3 files have unresolved bounded ownership/); + assert.match(elements[id].innerHTML, /dedicated Cowork transcript source is not covered/); + } +}); +test('Maintenance project detail exposes independent evidence outside the navigation button', () => { + const context = renderer('maintenance-focus'); + context.MNT = { facets: {} }; context.mntIcon = () => ''; context.mntProjectKindBadge = () => 'Git repository'; + const html = context.mntFocusNode({ value: 'id', label: 'Example', projectKind: 'git', count: 1, + sessionSurfaces: [{ ...sessionOrigin, host: 'codex', sessions: 2 }] }, 0, 'project', false); + assert.match(html, /<\/button>
    { + const context = renderer('maintenance-filters'); + assert.equal(context.mntFacetValueLabel('sessionOrigin', '__proto__'), 'Unknown'); + assert.equal(vocabulary.sessionPresentation({ initiator: 'toString' }).initiator, 'Unknown'); +}); +test('Usage retains explicit observed provider IDs while refusing unproven or unsupported provider claims', () => { + const context = renderer('usage'); + const observed = context.sdetail({ id: 'x', provider: 'openrouter', providerProvenance: 'observed' }); + assert.match(observed, /OpenRouter/); + assert.match(observed, /recorded provider ID; not network attestation/); + const unknown = context.sdetail({ id: 'x', provider: 'openrouter', providerProvenance: 'unknown' }); + assert.doesNotMatch(unknown, /OpenRouter/); +}); +test('Intelligence refresh resets an unavailable surface selection before rendering rows', () => { + const elements = { 'mw-table': {}, 'mw-hero': {}, 'mw-surface-filter': {} }; + const context = renderer('intelligence', elements); + const cloud = { label: 'Cloud project', sessionSurfaces: [{ surface: 'cloud-session', sessions: 1 }] }; + const local = { label: 'Local project', sessionSurfaces: [{ surface: 'claude-code-cli', sessions: 1 }] }; + context.renderMachineWide({ totals: {}, perProject: [cloud, local] }); + elements['mw-surface-filter'].value = 'Cloud session'; + elements['mw-surface-filter'].onchange(); + assert.equal(context.machineWideSurfaceFilter, 'Cloud session'); + context.renderMachineWide({ totals: {}, perProject: [local] }); + assert.equal(context.machineWideSurfaceFilter, 'all'); + assert.match(elements['mw-table'].innerHTML, /Local project/); + assert.doesNotMatch(elements['mw-table'].innerHTML, /Cloud session/); +}); +test('shared legacy Claude Desktop fallback remains known in Intelligence and Maintenance', () => { + const origin = { origin: 'claude-desktop', sessions: 3 }; + assert.deepEqual(vocabulary.sessionPresentation(origin), { surface: 'claude-desktop', label: 'Claude Desktop', + initiator: 'Unknown', provider: 'Unknown', providerBasis: 'not established', note: 'Claude Desktop observed in this legacy snapshot' }); + const context = renderer('maintenance-filters'); + assert.deepEqual(Array.from(context.surfaceNames({ sessionOrigins: [origin] })), ['Claude Desktop']); + assert.equal(context.mntFacetValueLabel('sessionOrigin', 'claude-desktop'), 'Claude Desktop'); + assert.equal(vocabulary.sessionPresentation({ origin: 'codex-desktop' }).surface, 'unknown'); +}); diff --git a/tests/kit/session-surface.test.mjs b/tests/kit/session-surface.test.mjs new file mode 100644 index 00000000..a34b3992 --- /dev/null +++ b/tests/kit/session-surface.test.mjs @@ -0,0 +1,139 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { classifySessionSurface, sessionSurfaceLabel } from '../../src/lib/session-surface.mjs'; +import { transcriptSessionOrigin } from '../../src/lib/footprint/session-origin.mjs'; + +const claudeCases = [ + ['cli', 'claude-code-cli', 'person'], ['claude-vscode', 'claude-code-vscode', 'person'], + ['claude-desktop', 'claude-desktop', 'person'], ['claude-desktop-3p', 'claude-desktop', 'person'], + ['local-agent', 'cowork', 'person'], ['local_agent', 'cowork', 'person'], + ['remote_cowork', 'cowork', 'person'], ['remote', 'cloud-session', 'person'], + ['remote_desktop', 'cloud-session', 'person'], ['remote_mobile', 'cloud-session', 'person'], + ['remote_projects', 'cloud-session', 'person'], ['remote_trigger', 'cloud-session', 'automation'], + ['remote_cowork_trigger', 'cloud-session', 'automation'], + ['sdk-py', 'claude-agent-sdk', 'automation'], ['sdk-ts', 'claude-agent-sdk', 'automation'], + ['sdk-cli', 'claude-noninteractive', 'automation'], + ['claude-code-github-action', 'github-actions', 'automation'], + ['claude_in_slack', 'claude-tag', 'person'], ['claude-in-slack', 'claude-tag', 'person'], + ['claude-in-teams', 'claude-tag', 'person'], + ['mcp', 'other-claude', 'automation'], ['ssh-remote', 'other-claude', 'person'], + ['bench', 'other-claude', 'unknown'], +]; + +test('maps every ADR-0060 Claude raw value to one surface and initiator', () => { + for (const [entrypoint, surface, initiator] of claudeCases) { + const actual = classifySessionSurface({ host: 'claude', entrypoint }); + assert.equal(actual.surface, surface, entrypoint); + assert.equal(actual.initiator, initiator, entrypoint); + assert.equal(actual.label, sessionSurfaceLabel(surface), entrypoint); + assert.deepEqual(actual.rawEvidence, { entrypoint }, entrypoint); + } +}); + +test('maps every ADR-0060 OpenAI originator, including source-dependent MCP', () => { + const cases = [ + ['Codex Desktop', undefined, 'chatgpt-desktop-codex', 'person'], + ['codex_work_desktop', undefined, 'chatgpt-desktop-work', 'person'], + ['codex-tui', undefined, 'codex-cli', 'person'], + ['codex_exec', 'exec', 'codex-cli-exec', 'automation'], + ['codex_vscode', 'vscode', 'codex-ide', 'person'], + ['codex_sdk_ts', undefined, 'codex-sdk', 'automation'], + ['codex_python_sdk', undefined, 'codex-sdk', 'automation'], + ['codex_cli_rs', 'mcp', 'codex-mcp', 'agent'], + ['codex_work_web', undefined, 'chatgpt-work-cloud', 'person'], + ['codex_work_mobile', undefined, 'chatgpt-work-cloud', 'person'], + ['codex_work_cca', undefined, 'chatgpt-work-cloud', 'person'], + ['chatgpt_cca', undefined, 'chatgpt-work-cloud', 'person'], + ['future-client', 'vscode', 'other-openai', 'unknown'], + ]; + for (const [originator, source, surface, initiator] of cases) { + const actual = classifySessionSurface({ host: 'codex', originator, source }); + assert.equal(actual.surface, surface, originator); + assert.equal(actual.initiator, initiator, originator); + assert.equal(actual.label, sessionSurfaceLabel(surface), originator); + assert.deepEqual(actual.rawEvidence, source ? { originator, source } : { originator }, originator); + } +}); + +test('keeps initiator orthogonal to surface and import state', () => { + assert.equal(classifySessionSurface({ host: 'claude', entrypoint: 'cli', sessionKind: 'bg' }).initiator, 'automation'); + assert.equal(classifySessionSurface({ host: 'codex', originator: 'Codex Desktop', threadSource: 'guardian_review' }).initiator, 'agent'); + assert.equal(classifySessionSurface({ host: 'codex', originator: 'Codex Desktop', threadSource: 'subagent' }).initiator, 'agent'); + assert.equal(classifySessionSurface({ host: 'codex', originator: 'Codex Desktop', threadSource: 'chatgpt_handoff' }).initiator, 'person'); + assert.equal(classifySessionSurface({ host: 'codex', originator: 'Codex Desktop', threadSource: 'automation' }).initiator, 'automation'); + assert.equal(classifySessionSurface({ host: 'codex', originator: 'Codex Desktop', importedCopy: true }).initiator, 'imported-copy'); + assert.equal(classifySessionSurface({ host: 'codex', originator: 'codex_exec', threadSource: 'user' }).initiator, 'automation'); + for (const threadSource of ['user', 'chatgpt_handoff']) { + assert.equal(classifySessionSurface({ host: 'codex', originator: 'codex_cli_rs', source: 'mcp', threadSource }).initiator, 'agent'); + for (const originator of ['codex_sdk_ts', 'codex_python_sdk']) { + assert.equal(classifySessionSurface({ host: 'codex', originator, threadSource }).initiator, 'automation'); + } + } + assert.equal(classifySessionSurface({ host: 'codex', originator: 'codex_sdk_ts', threadSource: 'guardian_review' }).initiator, 'agent'); +}); + +test('unknown declarations remain bounded and do not become product claims', () => { + assert.deepEqual(classifySessionSurface({ host: 'claude' }), { + surface: 'unknown', initiator: 'unknown', label: 'Unknown', rawEvidence: {}, attributes: [], thirdPartyProvider: null, + }); + const ambiguous = classifySessionSurface({ host: 'codex', source: 'vscode' }); + assert.equal(ambiguous.surface, 'unknown'); + assert.equal(ambiguous.label, 'Unknown'); + assert.deepEqual(ambiguous.rawEvidence, { source: 'vscode' }); + const prompt = 'private prompt '.repeat(100); + const unknown = classifySessionSurface({ host: 'claude', entrypoint: prompt }); + assert.equal(unknown.surface, 'unknown'); + assert.deepEqual(unknown.rawEvidence, {}); + assert.ok(!JSON.stringify(unknown).includes('private prompt')); + assert.deepEqual(classifySessionSurface({ host: 'codex', originator: 'user prompt' }).rawEvidence, {}); + for (const field of ['entrypoint', 'originator', 'source', 'threadSource', 'sessionKind']) { + const value = 'future_enum_v2'; + const candidate = classifySessionSurface({ host: field === 'entrypoint' || field === 'sessionKind' ? 'claude' : 'codex', + [field]: value }); + assert.equal(candidate.rawEvidence[field], value, field); + } + assert.deepEqual(classifySessionSurface({ host: 'claude', entrypoint: 'x'.repeat(80) }).rawEvidence, + { entrypoint: 'x'.repeat(80) }); + assert.deepEqual(classifySessionSurface({ host: 'claude', entrypoint: 'x'.repeat(81) }).rawEvidence, {}); + for (const value of ['a\nsecret', 'a secret', '', { text: 'secret' }, ['secret']]) { + assert.deepEqual(classifySessionSurface({ host: 'codex', originator: value }).rawEvidence, {}); + } + assert.deepEqual(classifySessionSurface({ host: 'claude', entrypoint: 'future_enum_v2' }), { + surface: 'other-claude', initiator: 'unknown', label: 'Other Claude surface', + rawEvidence: { entrypoint: 'future_enum_v2' }, attributes: [], thirdPartyProvider: null, + }); + assert.equal(sessionSurfaceLabel('__proto__'), 'Unknown'); +}); + +test('records observed attributes and keeps third-party provider separate', () => { + const thirdParty = classifySessionSurface({ host: 'claude', entrypoint: 'claude-desktop-3p' }); + assert.deepEqual(thirdParty.attributes, ['on 3P']); + assert.equal(thirdParty.thirdPartyProvider, null); + const remote = classifySessionSurface({ host: 'claude', entrypoint: 'remote_desktop' }); + assert.equal(remote.surface, 'cloud-session'); + assert.deepEqual(remote.attributes, ['started from Claude Desktop']); +}); + +test('footprint adapter keeps legacy origin/evidence and latches first declaration', () => { + const lines = (...rows) => rows.map((row) => JSON.stringify(row)); + const codex = transcriptSessionOrigin(lines( + { type: 'session_meta', payload: { originator: 'codex-tui', source: 'vscode', thread_source: 'user' } }, + { type: 'session_meta', payload: { originator: 'Codex Desktop' } }, + ), 'codex'); + assert.equal(codex.origin, 'unknown'); + assert.equal(codex.evidence, 'desktop-origin-not-declared'); + assert.equal(codex.surface, 'codex-cli'); + assert.equal(codex.initiator, 'person'); + assert.deepEqual(codex.rawEvidence, { originator: 'codex-tui', source: 'vscode', threadSource: 'user' }); + const claude = transcriptSessionOrigin(lines({ entrypoint: 'claude-desktop' }, { entrypoint: 'cli' }), 'claude'); + assert.deepEqual([claude.origin, claude.evidence, claude.surface], + ['claude-desktop', 'entrypoint:claude-desktop', 'claude-desktop']); + assert.equal(transcriptSessionOrigin(lines({ entrypoint: 'remote_desktop' }), 'claude').origin, 'unknown'); + const future = transcriptSessionOrigin(lines( + { entrypoint: 'future_enum_v2', sessionKind: 'future_kind', message: { content: 'private prompt' } }, + { entrypoint: 'claude-desktop' }), 'claude'); + assert.deepEqual(future.rawEvidence, { entrypoint: 'future_enum_v2', sessionKind: 'future_kind' }); + assert.equal(future.surface, 'other-claude'); + assert.equal(future.origin, 'unknown'); + assert.equal(JSON.stringify(future).includes('private prompt'), false); +}); diff --git a/tests/kit/setup-command.test.mjs b/tests/kit/setup-command.test.mjs index 70fe19ae..2b918bfe 100644 --- a/tests/kit/setup-command.test.mjs +++ b/tests/kit/setup-command.test.mjs @@ -774,7 +774,7 @@ test('ak setup --opencode with an ABSENT CLI never fabricates the config home', test.after(() => rmrf(HOME)); -// Branch 3, Task 2.2: setup no longer turns Ruflo's start-on-use off, and it +// setup no longer turns Ruflo's start-on-use off, and it // writes the managed daemon settings BEFORE `ruflo daemon start`, because the // daemon reads .claude-flow/config.json once, in its constructor // (worker-daemon.js:139-144, 3.46.1). diff --git a/tests/kit/setup-host-rerecord.test.mjs b/tests/kit/setup-host-rerecord.test.mjs new file mode 100644 index 00000000..80cb1110 --- /dev/null +++ b/tests/kit/setup-host-rerecord.test.mjs @@ -0,0 +1,106 @@ +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import { registerHooks } from 'node:module'; +import { sandboxHome, rmrf } from './helpers/home-sandbox.mjs'; +import { isolateProject, REPO_ROOT } from './helpers/project-isolation.mjs'; + +const home = sandboxHome('ak-setup-host-rerecord'); +after(() => rmrf(home)); +isolateProject('ak-setup-host-rerecord'); +const setup = await import('../../src/commands/setup.mjs'); +const cfg = { integrations: { hosts: { claude: false, codex: true, opencode: false } } }; + +for (const [name, initial, ok, expected] of [ + ['successful install', 'absent', true, 2], + ['failed install', 'absent', false, 1], + ['present host', 'npm', true, 1], +]) { + test(`${name} records fresh host facts only when installation succeeds`, async () => { + const stateCalls = []; + const facts = []; + const installs = []; + await setup.installEnabledAbsentHosts(cfg, { yes: true }, { + installState: async (_host, opts) => { stateCalls.push(opts); return { method: stateCalls.length === 1 ? initial : 'npm', version: '1.0.0' }; }, + install: async id => { installs.push(id); return { ok, detail: ok ? 'installed' : 'failed' }; }, + collectFacts: async opts => { facts.push(opts); }, + }); + assert.equal(stateCalls.length, expected); + assert.deepEqual(installs, initial === 'absent' ? ['codex'] : []); + if (expected === 2) { + assert.deepEqual(stateCalls[1], { refresh: true, record: true, source: 'setup' }); + assert.deepEqual(facts, [{ cfg, refresh: true, record: true, source: 'setup' }]); + } else assert.deepEqual(facts, []); + }); +} + +test('disabled hosts do not probe, install, or record evidence', async () => { + await setup.installEnabledAbsentHosts({ integrations: { hosts: {} } }, { yes: true }, { + installState: async () => { throw new Error('disabled host probed'); }, + install: async () => { throw new Error('disabled host installed'); }, + collectFacts: async () => { throw new Error('disabled host recorded'); }, + }); +}); + +test('setup run passes its host lifecycle through the machine setup boundary', async () => { + const lifecycle = { installState: async () => { throw new Error('sentinel'); } }; + let received; + await setup.run({ + flags: { 'dry-run': true, minimal: true, 'no-aqe': true, 'no-ruvnet-brain': true, 'no-agent-browser': true, yes: true }, + pkgRoot: REPO_ROOT, + deps: { hostLifecycle: lifecycle }, + machineSetup: async args => { received = args.deps?.hostLifecycle; return false; }, + }); + assert.equal(received, lifecycle); +}); + +test('real run_machine passes injected lifecycle to the host installation loop', async () => { + const stubs = new Map([ + ['../lib/versions.mjs', "export const installedVersion = () => '3.48.0'; export const cmpVersions = () => 0;"], + ['../lib/heal.mjs', "export const healNatives = async () => ({ ok: true, detail: 'stub' }); export const healAidefence = async () => ({ ok: true, detail: 'stub' });"], + ['../lib/exec.mjs', "export const have = async () => false; export const run = async () => { throw Error('real command reached'); };"], + ['../lib/providers.mjs', ` + export const HOSTS = [{ id: 'codex', pkg: '@openai/codex' }]; + export const hostInstallState = async () => { throw Error('default host probe reached'); }; + export const installHost = async () => { throw Error('default installer reached'); }; + export const collectIntegrationFacts = async () => { throw Error('default facts collector reached'); }; + export const migrateRetiredRoutesInConfig = () => {}; + export const printActivityRoutingTable = () => {}; + export const convergeProviderStack = () => {}; + export const applySetupHostFlags = () => {}; + export const guidanceContext = () => {}; + export const reportRetiredRouteChanges = () => {}; + `], + ['../lib/adapters/lifecycle-registry.mjs', ` + export const hostsWithLifecycle = () => []; + export const lifecycleAdapterFor = () => { throw Error('host lifecycle reached'); }; + export const lifecycleExecutionEnabled = () => false; + export const detectionBinFor = () => { throw Error('host lifecycle reached'); }; + `], + ['../lib/ruflo-components/apply.mjs', "export const reconcileRufloComponents = async () => { throw Error('components reached'); };"], + ['./status/sections/ruflo-components.mjs', "export const componentResultReport = () => [];"], + ]); + registerHooks({ + resolve(specifier, context, nextResolve) { + if (context.parentURL?.includes('/src/commands/setup.mjs?b2-machine') && stubs.has(specifier)) { + return { url: `data:text/javascript,${encodeURIComponent(stubs.get(specifier))}`, shortCircuit: true }; + } + return nextResolve(specifier, context); + }, + }); + const isolatedSetup = await import('../../src/commands/setup.mjs?b2-machine'); + const sentinel = new Error('injected host lifecycle reached'); + const machineCfg = { + agentBrowser: false, aqe: false, ruvnetBrain: false, security: false, + integrations: { hosts: { codex: true } }, + }; + let calls = 0; + await assert.rejects(isolatedSetup.run_machine({ + flags: { yes: true }, cfg: machineCfg, pkgRoot: home, + deps: { hostLifecycle: { + installState: async () => { calls++; throw sentinel; }, + install: async () => { throw Error('injected installer reached'); }, + collectFacts: async () => { throw Error('injected facts reached'); }, + } }, + }), error => error === sentinel); + assert.equal(calls, 1); +}); diff --git a/tests/kit/setup-memory-probe.test.mjs b/tests/kit/setup-memory-probe.test.mjs index d477d976..e9842afd 100644 --- a/tests/kit/setup-memory-probe.test.mjs +++ b/tests/kit/setup-memory-probe.test.mjs @@ -10,6 +10,7 @@ import path from 'node:path'; import { DatabaseSync } from 'node:sqlite'; import { verifyProjectMemoryWrite } from '../../src/commands/setup.mjs'; import { captureLog } from './helpers/home-sandbox.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; function store(file) { fs.mkdirSync(path.dirname(file), { recursive: true }); @@ -109,3 +110,114 @@ test('the setup probe leaves nothing in a redirected memory root', async (t) => assert.deepEqual(keys(path.join(redirected, 'agentdb-memory.db')), ['user-row'], 'the redirected MCP store kept a setup probe'); for (const name of ['memory.db', 'agentdb-memory.db']) assert.deepEqual(keys(path.join(root, '.swarm', name)), ['user-row']); }); + +// Ruflo 3.48 writes its native mirror under CLAUDE_FLOW_MEMORY_PATH even when +// CLAUDE_FLOW_DB_PATH points elsewhere. The fake follows those two routes. +function routedProject(t) { + const root = tempDir('ak-setup-route', t); + const primary = path.join(root, '.swarm', 'memory.db'); + const canonical = path.join(root, '.swarm', 'agentdb-memory.db'); + const redirected = path.join(root, 'data', 'memory', 'agentdb-memory.db'); + for (const file of [primary, canonical, redirected]) { + const db = store(file); + db.prepare("INSERT INTO memory_entries VALUES ('real', 'user-row', 'active')").run(); + db.close(); + } + const mirrors = []; + const runner = async (_cmd, args, { env }) => { + const key = args[args.indexOf('-k') + 1]; + assert.equal(env.RUFLO_DAEMON_AUTOSTART, '0'); + assert.equal(env.CLAUDE_FLOW_DB_PATH, primary); + const mirror = path.join(env.CLAUDE_FLOW_MEMORY_PATH, 'agentdb-memory.db'); + mirrors.push(mirror); + for (const file of [env.CLAUDE_FLOW_DB_PATH, mirror]) { + fs.mkdirSync(path.dirname(file), { recursive: true }); + const db = new DatabaseSync(file); + db.exec('CREATE TABLE IF NOT EXISTS memory_entries (namespace TEXT, key TEXT, status TEXT)'); + db.prepare("INSERT INTO memory_entries VALUES ('_setup', ?, 'active')").run(key); + db.close(); + } + return { code: 0, stdout: '', stderr: '' }; + }; + return { root, primary, canonical, redirected, mirrors, runner }; +} + +test('setup verifies the primary while removing its private mirror and preserving native corpora', async (t) => { + const p = routedProject(t); + const { out } = await captureLog(() => verifyProjectMemoryWrite(p.root, { CLAUDE_FLOW_DB_PATH: p.primary, + CLAUDE_FLOW_MEMORY_PATH: path.dirname(p.redirected) }, { runner: p.runner })); + assert.match(out, /memory write VERIFIED/); + assert.deepEqual(keys(p.primary), ['user-row']); + assert.deepEqual(keys(p.canonical), ['user-row']); + assert.deepEqual(keys(p.redirected), ['user-row']); + assert.equal(p.mirrors.length, 1); + assert.equal(fs.existsSync(path.dirname(p.mirrors[0])), false); +}); + +test('setup pins the primary when caller env is absent and removes the mirror on a failed store', async (t) => { + const p = routedProject(t); + const { out } = await captureLog(() => verifyProjectMemoryWrite(p.root, undefined, { + runner: async (cmd, args, options) => { + await p.runner(cmd, args, options); + return { code: 1, stdout: '', stderr: 'failed after write' }; + }, + })); + assert.match(out, /memory write verification FAILED/); + assert.doesNotMatch(out, /memory write VERIFIED/); + assert.deepEqual(keys(p.canonical), ['user-row']); + assert.deepEqual(keys(p.redirected), ['user-row']); + assert.equal(fs.existsSync(path.dirname(p.mirrors[0])), false); +}); + +test('setup removes its private mirror after a runner throws', async (t) => { + const p = routedProject(t); + const { out } = await captureLog(() => verifyProjectMemoryWrite(p.root, {}, { + runner: async (cmd, args, options) => { + await p.runner(cmd, args, options); + throw new Error('runner failed'); + }, + })); + assert.match(out, /memory write verification FAILED/); + assert.equal(fs.existsSync(path.dirname(p.mirrors[0])), false); + assert.deepEqual(keys(p.canonical), ['user-row']); +}); + +test('setup reports a missing primary row and removes the private mirror', async (t) => { + const p = routedProject(t); + let mirror; + const { out } = await captureLog(() => verifyProjectMemoryWrite(p.root, {}, { + runner: async (_cmd, _args, { env }) => { + mirror = env.CLAUDE_FLOW_MEMORY_PATH; + const db = store(path.join(mirror, 'agentdb-memory.db')); + db.close(); + return { code: 0, stdout: '', stderr: '' }; + }, + })); + assert.match(out, /memory write verification FAILED/); + assert.doesNotMatch(out, /memory write VERIFIED/); + assert.equal(fs.existsSync(mirror), false); + assert.deepEqual(keys(p.primary), ['user-row']); +}); + +test('setup does not claim complete verification when private mirror removal fails', { skip: process.platform === 'win32' }, async (t) => { + const p = routedProject(t); + let mirror; + t.after(() => { + if (mirror && fs.existsSync(mirror)) { + fs.chmodSync(mirror, 0o700); + fs.rmSync(mirror, { recursive: true, force: true }); + } + }); + const { out } = await captureLog(() => verifyProjectMemoryWrite(p.root, {}, { + runner: async (cmd, args, options) => { + await p.runner(cmd, args, options); + mirror = options.env.CLAUDE_FLOW_MEMORY_PATH; + fs.chmodSync(mirror, 0o000); + return { code: 0, stdout: '', stderr: '' }; + }, + })); + assert.match(out, /temporary mirror cleanup failed/); + assert.match(out, /memory write verification FAILED/); + assert.doesNotMatch(out, /memory write VERIFIED/); + assert.deepEqual(keys(p.primary), ['user-row']); +}); diff --git a/tests/kit/status-command.test.mjs b/tests/kit/status-command.test.mjs index 27744242..5539389d 100644 --- a/tests/kit/status-command.test.mjs +++ b/tests/kit/status-command.test.mjs @@ -284,7 +284,7 @@ test('collect() writes nothing to HOME or the project, beyond its own probe-resu const beforeHome = snapshot(HOME); const beforeProject = snapshot(PROJECT); await collect(); - // Branch 6a Task 5: a plain status collect() call (refresh:false, the + // a plain status collect() call (refresh:false, the // default) still probes for real on a cache miss/stale/first run (Ruling // A) and records the result — so a LATER plain status call can reuse it. // That write lands only under the shared evidence store, never anywhere @@ -329,7 +329,7 @@ test('ruflo provider intent never claims registration alone is routed execution' assert.doesNotMatch(intent.message, /routable|executed successfully/); }); -// Branch 6a Task 9: dashboard-server.mjs now calls collect() in process, and +// dashboard-server.mjs now calls collect() in process, and // a live dashboard server can be asked (via its per-request `cwd`) for a // project other than its own launch cwd. Every cwd-sensitive section must // resolve against the PASSED cwd, never a silent process.cwd() fallback. @@ -658,6 +658,7 @@ test('Codex MCP topology fails recursive self-registration and reports missing A 'args = ["x", "ruflo-mcp"]', ].join('\n')); try { + const codexConfigBefore = fs.readFileSync(path.join(PROJECT, '.codex', 'config.toml'), 'utf8'); const rows = rowsFor(await collect(), 'codex-mcp'); assert.equal(rows.find((r) => /recursive codex/.test(r.message))?.level, 'fail'); assert.equal(rows.find((r) => /agentic-qe MCP is not concretely/.test(r.message))?.level, 'warn'); @@ -666,7 +667,12 @@ test('Codex MCP topology fails recursive self-registration and reports missing A // can prove (codexMcpRepairPlan); Agentic-QE owns its own Codex registration. assert.equal(rows.find((r) => /recursive codex/.test(r.message))?.repair, 'sync'); assert.equal(rows.find((r) => /duplicate Ruflo/.test(r.message))?.repair, 'sync'); - assert.equal(rows.find((r) => /agentic-qe MCP is not concretely/.test(r.message))?.repair, 'manual'); + const aqe = rows.find((r) => /agentic-qe MCP is not concretely/.test(r.message)); + assert.equal(aqe?.repair, 'manual'); + assert.match(aqe.fix, /aqe init --auto --with-codex --codex-guidance compact/); + assert.doesNotMatch(aqe.fix, /platform setup|codex mcp-server/); + assert.equal(fs.readFileSync(path.join(PROJECT, '.codex', 'config.toml'), 'utf8'), codexConfigBefore, + 'status only reports the AQE-owned initialization flow'); } finally { rmrf(path.join(PROJECT, '.codex')); } @@ -716,7 +722,7 @@ test('a user-owned deprecated codex mcp-server entry is a manual removal', async }); test('Codex MCP topology does not ask an aqe:false machine to register agentic-qe in Codex', async () => { - // #237 N1: with AQE opted out, `aqe platform setup codex` is advice for a + // #237 N1: with AQE opted out, Codex initialization advice is for a // tool the user declined; the topology rows must honor kit.json like the // aqe section does. seedHome(offlineKitConfig({ @@ -970,7 +976,7 @@ test('--refresh prints one line per finished stage, then the status table', asyn assert.match(r.out, /ak status\n.*versions +fake versions row/); }); -test('plain status runs no refresh stage', async () => { +test('plain status rejects a stray positional before any refresh stage', async () => { seedHome(); let calls = 0; const refreshStages = new Proxy({}, { get: () => async () => { calls += 1; return { ok: true }; } }); @@ -981,7 +987,8 @@ test('plain status runs no refresh stage', async () => { r = await captureLog(() => status.run({ flags: {}, positionals: ['extra'], pkgRoot: PKG_ROOT, deps: { refreshStages } })); } finally { process.chdir(cwd); } assert.equal(calls, 0); - assert.notEqual(r.result, 2, 'plain status still ignores a positional, as it always has'); + assert.equal(r.result, 2); + assert.match(r.out, /unexpected argument 'extra'/); }); test('a refresh with a stray argument is a usage error that names the one-token spelling', async () => { diff --git a/tests/kit/status-version-drift-refresh.test.mjs b/tests/kit/status-version-drift-refresh.test.mjs index 0f9bc8f7..1a0076e0 100644 --- a/tests/kit/status-version-drift-refresh.test.mjs +++ b/tests/kit/status-version-drift-refresh.test.mjs @@ -1,4 +1,4 @@ -// Task 6 (Branch 6a refactor/evidence-store): `ak status --refresh` had no +// `ak status --refresh` had no // effect on version-drift rows because four sections silently dropped the // `refresh` key their shared collect() ctx already carries. The four // libraries (versions.mjs/ruvector.mjs/ruvnet-brain.mjs's driftReport/ @@ -151,7 +151,7 @@ function seedSelfHome({ ageMs = 0 } = {}) { ttlHours: 24, last: Date.now(), seen: { ruflo: '9.9.9', 'agentic-qe': '9.9.9' }, - self: { last: Date.now() - ageMs, best: { version: '0.0.1', tag: 'latest' } }, + self: { last: Date.now() - ageMs, best: { version: '0.0.1', tag: 'latest' }, lastTags: ['latest', 'next'] }, }, }); writeKitConfig(HOME, cfg); diff --git a/tests/kit/status-zero-spawn.test.mjs b/tests/kit/status-zero-spawn.test.mjs index a14ea7e2..974ede8b 100644 --- a/tests/kit/status-zero-spawn.test.mjs +++ b/tests/kit/status-zero-spawn.test.mjs @@ -1,11 +1,11 @@ -// Branch 6a Task 8a built this harness: src/commands/status.mjs's collect() +// This harness covers src/commands/status.mjs's collect() // runs inside a real child Node process launched with `--import` of // tests/helpers/spawn-guard.mjs (Ruling C: a product-code ledger seam inside // exec.mjs would under-count — ~30 files spawn child_process directly, not // through it), so every spawn path is caught regardless of which module // makes it, with no product code change. // -// Task 7 closed every remaining spawn path on a plain `ak status` (native +// Evidence caching closed every remaining spawn path on a plain `ak status` (native // runtime, host setup/launch, deja-vu, version drift, npm's global root, the // daemon process sweep, the ak-launcher-availability check) and promotes the // zero-spawn assertion from `test.todo` (Ruling D — never commit a red test) @@ -17,12 +17,13 @@ // `refresh: true` (always probes again, even warm). import { test } from 'node:test'; import assert from 'node:assert/strict'; -import { execFileSync } from 'node:child_process'; +import { execFileSync, spawn } from 'node:child_process'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { fileURLToPath, pathToFileURL } from 'node:url'; import { spawnEnv, sandboxProject } from './helpers/home-sandbox.mjs'; +import { acquireRunRootHold, releaseRunRootHold } from '../../scripts/run-roots.mjs'; const HERE = path.dirname(fileURLToPath(import.meta.url)); const PKG_ROOT = path.resolve(HERE, '..', '..'); @@ -35,9 +36,30 @@ function readLedger(file) { return fs.readFileSync(file, 'utf8').split('\n').filter(Boolean).map((line) => JSON.parse(line)); } +async function waitUntil(check, timeoutMs = 5_000) { + const deadline = Date.now() + timeoutMs; + while (!check() && Date.now() < deadline) await new Promise((resolve) => setTimeout(resolve, 10)); + return check(); +} + +async function waitFor(promise, timeoutMs = 5_000) { + let timer; + try { + return await Promise.race([promise, new Promise((_, reject) => { + timer = setTimeout(() => reject(new Error('owned process did not close')), timeoutMs); + })]); + } finally { clearTimeout(timer); } +} + +function processGone(pid) { + if (!Number.isInteger(pid) || pid <= 0) return false; + try { process.kill(pid, 0); return false; } + catch (error) { return error.code === 'ESRCH'; } +} + /** Runs `fn` inside a disposable sandboxed HOME/project, with `extraEnv` - * merged into the child's environment. Cleans up unconditionally. */ -function inSandbox(prefix, extraEnv, fn) { + * merged into the child's environment. Retains both roots when a child may still use them. */ +async function inSandbox(prefix, extraEnv, fn) { const home = fs.mkdtempSync(path.join(os.tmpdir(), `${prefix}-home-`)); fs.mkdirSync(path.join(home, '.config'), { recursive: true }); const project = sandboxProject(prefix); @@ -48,16 +70,22 @@ function inSandbox(prefix, extraEnv, fn) { PATH: path.join(home, 'no-such-bin'), ...extraEnv, }); - try { - return fn({ home, project, env }); - } finally { - fs.rmSync(home, { recursive: true, force: true }); - fs.rmSync(project, { recursive: true, force: true }); + let result; + let failure; + let retained = false; + try { result = await fn({ home, project, env, retain: () => { retained = true; } }); } + catch (error) { failure = error; } + if (retained) throw failure ?? new Error(`retained uncertain sandbox: ${home}, ${project}`); + for (const root of [home, project]) { + try { fs.rmSync(root, { recursive: true, force: true }); } + catch (error) { failure = failure ? new AggregateError([failure, error], 'sandbox assertion and cleanup failed') : error; } } + if (failure) throw failure; + return result; } -test('spawn-guard is a no-op when AK_SPAWN_LEDGER_FILE is unset', () => { - inSandbox('ak-spawn-guard-noop', {}, ({ project, env }) => { +test('spawn-guard is a no-op when AK_SPAWN_LEDGER_FILE is unset', async () => { + await inSandbox('ak-spawn-guard-noop', {}, ({ project, env }) => { // A vacuous "no ledger file exists" check would pass even if the guard // patched child_process regardless of the env var — assert the function // ITSELF is untouched (native name `spawn`, not our wrapper's `patched`). @@ -68,25 +96,121 @@ test('spawn-guard is a no-op when AK_SPAWN_LEDGER_FILE is unset', () => { }); }); -test('spawn-guard records a spawn made inside the guarded child, by every wrapped form', () => { - inSandbox('ak-spawn-guard-smoke', {}, ({ project, env: baseEnv }) => { - const ledgerFile = path.join(os.tmpdir(), `ak-spawn-guard-smoke-${process.pid}.ndjson`); +async function runGuardedSmoke({ childExitCode = 0, childStallsOnRelease = false, + closeTimeoutMs = 5_000, launchFailure = false } = {}) { + // The guarded runner must know about uncertainty before any process uses + // its temp root. Standalone node --test has no runner hold to acquire. + const hold = acquireRunRootHold(); + await inSandbox('ak-spawn-guard-smoke', {}, async ({ project, env: baseEnv, retain }) => { + const ledgerFile = path.join(project, 'spawn-ledger.ndjson'); + const readyFile = path.join(project, 'fork-ready'); + const releaseFile = path.join(project, 'fork-release'); + const doneFile = path.join(project, 'fork-done'); + const pidFile = path.join(project, 'fork-pid'); + const attemptFile = path.join(project, 'fork-attempt'); + const waitingFile = path.join(project, 'parent-waiting-for-fork'); + const parentAckFile = path.join(project, 'fork-closed-by-parent'); const env = { ...baseEnv, AK_SPAWN_LEDGER_FILE: ledgerFile }; - const forkTarget = path.join(project, 'fork-target.mjs'); - fs.writeFileSync(forkTarget, 'process.exit(0);\n'); + const forkTarget = path.join(project, 'fork-target.cjs'); + fs.writeFileSync(forkTarget, [ + "const fs = require('node:fs');", + `fs.writeFileSync(${JSON.stringify(readyFile)}, 'ready');`, + `const release = ${JSON.stringify(releaseFile)};`, + 'const timer = setInterval(() => {', + ' if (!fs.existsSync(release)) return;', + ` fs.writeFileSync(${JSON.stringify(doneFile)}, 'done');`, + ...(childStallsOnRelease ? [' return;'] : []), + ' clearInterval(timer);', + ` process.exit(${childExitCode});`, + '}, 10);', + ].join('\n')); const script = [ + 'async function main() {', "const { spawnSync, execFileSync: ef, execSync: es, fork } = require('node:child_process');", "spawnSync(process.execPath, ['-e', '0']);", "ef(process.execPath, ['-e', '0']);", 'try { es(\'true\'); } catch {}', // shell builtin: exercised even with PATH broken - `fork(${JSON.stringify(forkTarget)}, [], { stdio: 'ignore' });`, - 'process.exit(0);', // do not wait on the forked grandchild's IPC channel + `require('node:fs').writeFileSync(${JSON.stringify(attemptFile)}, 'attempt');`, + `const child = fork(${JSON.stringify(forkTarget)}, [], { stdio: 'ignore' });`, + `require('node:fs').writeFileSync(${JSON.stringify(pidFile)}, String(child.pid));`, + `require('node:fs').writeFileSync(${JSON.stringify(waitingFile)}, 'waiting');`, + 'await new Promise((resolve, reject) => {', + ' child.once("error", reject);', + ' child.once("close", (code, signal) => code === 0 && !signal', + ' ? resolve() : reject(new Error(`fork failed: code=${code}, signal=${signal}`)));', + '});', + 'if (child.exitCode !== 0 || child.signalCode) throw new Error("fork has not exited cleanly");', + `require('node:fs').writeFileSync(${JSON.stringify(parentAckFile)}, 'fork closed');`, + '}', + 'main().catch((error) => { console.error(error); process.exitCode = 1; });', ].join(' '); - execFileSync(process.execPath, [ + const guarded = spawn(launchFailure ? path.join(project, 'missing-node') : process.execPath, [ `--import=${SPAWN_GUARD_URL}`, '-e', script, - ], { cwd: project, env, encoding: 'utf8' }); + ], { cwd: project, env, stdio: ['ignore', 'ignore', 'pipe'] }); + let stderr = ''; + let launchError; + guarded.once('error', (error) => { launchError = error; }); + guarded.stderr.on('data', (chunk) => { stderr += chunk; }); + let closed = false; + const parentClose = new Promise((resolve) => guarded.once('close', (code, signal) => { + closed = true; + resolve({ code, signal }); + })); + let failure; + try { + await waitUntil(() => (fs.existsSync(waitingFile) && fs.existsSync(readyFile)) || closed || launchError); + if (launchError) throw launchError; + assert.ok(fs.existsSync(waitingFile), `guarded parent did not reach fork wait: ${stderr}`); + assert.ok(fs.existsSync(readyFile), `fork did not become ready: ${stderr}`); + // A premature parent may write its acknowledgment or close on a later + // event-loop turn. Give those events time to surface before release. + await waitUntil(() => fs.existsSync(parentAckFile) || closed, 300); + assert.ok(!fs.existsSync(parentAckFile), 'parent acknowledged fork close before child release'); + assert.ok(!closed, 'guarded parent exited while its owned fork was still running'); + } catch (error) { failure = error; } + const cleanupErrors = []; + let parentExited = closed; + let pid = NaN; + try { fs.writeFileSync(releaseFile, 'release'); } catch (error) { cleanupErrors.push(error); } + try { pid = fs.existsSync(pidFile) ? Number(fs.readFileSync(pidFile, 'utf8')) : NaN; } + catch (error) { cleanupErrors.push(error); } + try { await waitFor(parentClose, closeTimeoutMs); parentExited = true; } + catch { + // Keep the guarded parent alive to receive its own fork's close event. + if (Number.isInteger(pid) && pid > 0) { + try { process.kill(pid); } catch (error) { if (error.code !== 'ESRCH') cleanupErrors.push(error); } + } + try { await waitFor(parentClose, closeTimeoutMs); parentExited = true; } + catch { + if (!closed) guarded.kill(); + try { await waitFor(parentClose, closeTimeoutMs); parentExited = true; } + catch (error) { cleanupErrors.push(error); } + } + } + parentExited ||= closed; + // The acknowledgment is written only after the guarded parent receives + // its fork's close event. Otherwise require independent OS exit evidence. + let forkExited = fs.existsSync(parentAckFile) || (!fs.existsSync(attemptFile) && parentExited) + || await waitUntil(() => processGone(pid)); + if (!forkExited && Number.isInteger(pid) && pid > 0) { + try { process.kill(pid); } catch (error) { if (error.code !== 'ESRCH') cleanupErrors.push(error); } + forkExited = await waitUntil(() => processGone(pid)); + } + if (!forkExited) cleanupErrors.push(new Error(`cannot establish owned fork exit: pid=${pid}`)); + if (!parentExited) cleanupErrors.push(new Error('cannot establish guarded parent exit')); + if (!parentExited || !forkExited) retain(); + else { + try { releaseRunRootHold(hold); } catch (error) { cleanupErrors.push(error); } + } + if (cleanupErrors.length) failure = failure + ? new AggregateError([failure, ...cleanupErrors], 'fork assertion and cleanup failed') + : new AggregateError(cleanupErrors, 'fork cleanup failed'); + if (failure) throw failure; + assert.ok(fs.existsSync(doneFile), 'owned fork completed before sandbox cleanup'); + const { code, signal } = await parentClose; + assert.strictEqual(code, 0, `guarded parent failed (${signal}): ${stderr}`); + assert.ok(fs.existsSync(parentAckFile), 'guarded parent did not acknowledge the fork close event'); const lines = readLedger(ledgerFile); - fs.rmSync(ledgerFile, { force: true }); // Each wrapped form by name, not an exact total: execSync goes through a // platform shell, and only its own record is what this test is about. const got = JSON.stringify(lines); @@ -96,6 +220,22 @@ test('spawn-guard records a spawn made inside the guarded child, by every wrappe assert.ok(lines.some((l) => l.cmd === forkTarget), `fork() records the module path as cmd; got ${got}`); assert.ok(lines.every((l) => typeof l.at === 'string' && !Number.isNaN(Date.parse(l.at))), 'every line has an ISO timestamp'); }); +} + +test('spawn-guard records every wrapped form and waits for its owned fork', async () => { + await runGuardedSmoke(); +}); + +test('nonzero owned fork exit fails after the guarded parent reaps it', async () => { + await assert.rejects(runGuardedSmoke({ childExitCode: 7 }), /guarded parent failed/); +}); + +test('stalled owned fork is signaled before its parent is reaped', async () => { + await assert.rejects(runGuardedSmoke({ childStallsOnRelease: true, closeTimeoutMs: 200 }), /guarded parent failed/); +}); + +test('guarded launch failure is reported and its sandbox is cleaned', async () => { + await assert.rejects(runGuardedSmoke({ launchFailure: true }), { code: 'ENOENT' }); }); // A ledger line from the fixture's own npm registry lookups @@ -130,15 +270,14 @@ function sliceByCallBoundary(lines, labels) { return slices; } -test('plain ak status spawns nothing on a warm cache; --refresh always re-probes', () => { - inSandbox('ak-status-zero-spawn', {}, ({ project, env: baseEnv }) => { - const ledgerFile = path.join(os.tmpdir(), `ak-status-zero-spawn-${process.pid}.ndjson`); +test('plain ak status spawns nothing on a warm cache; --refresh always re-probes', async () => { + await inSandbox('ak-status-zero-spawn', {}, ({ project, env: baseEnv }) => { + const ledgerFile = path.join(project, 'status-ledger.ndjson'); const env = { ...baseEnv, AK_SPAWN_LEDGER_FILE: ledgerFile }; execFileSync(process.execPath, [ `--import=${SPAWN_GUARD_URL}`, FIXTURE, PKG_ROOT, project, ], { cwd: project, env, encoding: 'utf8', timeout: 30_000 }); const lines = readLedger(ledgerFile); - fs.rmSync(ledgerFile, { force: true }); const { first, second, third } = sliceByCallBoundary(lines, ['first', 'second', 'third']); diff --git a/tests/kit/sync-command.test.mjs b/tests/kit/sync-command.test.mjs index 3883e066..64945d97 100644 --- a/tests/kit/sync-command.test.mjs +++ b/tests/kit/sync-command.test.mjs @@ -1248,7 +1248,7 @@ test('enabled + absent CLI: the install is attempted by hosts, the wiring is ski assert.ok(!fs.existsSync(ocHome()), 'the config home is never fabricated for an absent host'); }); -// final-review fix: a real (non-dry-run) sync forces host evidence fresh +// a real (non-dry-run) sync forces host evidence fresh // BEFORE the plan is read (refreshPlanHosts), so a cached row that has not // yet gone stale (host-setup's 6h TTL) but is simply WRONG — here, claude // went from "absent" to "on PATH" since it was last recorded — can never diff --git a/tests/kit/sync-daemon-repair.test.mjs b/tests/kit/sync-daemon-repair.test.mjs index 616841da..b13863f5 100644 --- a/tests/kit/sync-daemon-repair.test.mjs +++ b/tests/kit/sync-daemon-repair.test.mjs @@ -1,4 +1,4 @@ -// final-review fix: sync's 'daemons' step must re-record daemon-sweep +// sync's 'daemons' step must re-record daemon-sweep // evidence after a successful reap. The Important finding: listDaemons() // records the PRE-reap process list; without a re-list after reap() kills // something, that stale list is the last thing written — a later `ak @@ -16,6 +16,7 @@ import { sandboxHome, rmrf } from './helpers/home-sandbox.mjs'; const SANDBOX_HOME = sandboxHome('ak-sync-daemon-repair'); after(() => rmrf(SANDBOX_HOME)); const sync = await import('../../src/commands/sync.mjs'); +const { applyRufloDaemon } = await import('../../src/lib/ruflo-daemon-config.mjs'); const daemonsStep = sync.SYNC_STEPS.find((s) => s.id === 'daemons'); @@ -75,3 +76,49 @@ test('a reap attempt that fails to kill anything is not re-recorded either', asy assert.equal(listCalls.length, 1, 'a failed reap (pid already gone/reused) leaves the process list unchanged; no re-record is needed'); } finally { rmrf(cwd); } }); + +test('daemon step uses injected version, apply and save seams after its sweep', async () => { + const cwd = freshCwd(); + const calls = []; + try { + await daemonsStep.run({ + cwd, cfg: {}, + daemonLifecycle: { list: async () => [], reap: () => [] }, + daemonVersion: () => { calls.push('version'); return '3.48.0'; }, + daemonApply: async (_cwd, options) => { + calls.push(['apply', options.rufloVersion]); + return { result: { config: 'converged', autostart: 'converged', changed: false, held: null }, restarted: false }; + }, + saveConfig: () => calls.push('save'), + }); + assert.deepEqual(calls, ['version', ['apply', '3.48.0'], 'save']); + } finally { rmrf(cwd); } +}); + +test('versions-triggered daemon step applies only fixture settings with no external processes', async () => { + const cwd = freshCwd(); + const calls = []; + try { + fs.mkdirSync(path.join(cwd, '.git')); + fs.mkdirSync(path.join(cwd, '.claude-flow')); + fs.mkdirSync(path.join(cwd, '.swarm')); + fs.writeFileSync(path.join(cwd, '.swarm', 'memory.db'), ''); + const cfg = { rufloDaemon: { receipts: {} } }; + assert.ok(sync.activeSteps(new Set(['versions']), {}, cfg).includes('daemons')); + await daemonsStep.run({ + cwd, cfg, + daemonLifecycle: { list: async () => { calls.push('list'); return []; }, reap: () => [] }, + daemonVersion: () => '3.45.0', + daemonApply: (root, options) => applyRufloDaemon(root, { + ...options, platform: 'darwin', alive: () => { calls.push('alive'); return true; }, + runner: async (_tool, args) => { calls.push(args.join(' ')); return { code: 0 }; }, + }), + saveConfig: () => calls.push('save'), + }); + assert.deepEqual(calls, ['list', 'alive', 'daemon stop', 'daemon start', 'save']); + assert.deepEqual(JSON.parse(fs.readFileSync(path.join(cwd, '.claude-flow', 'config.json'), 'utf8')), + { 'daemon.idleSecs': 0, 'daemon.resourceThresholds.minFreeMemoryPercent': 0 }); + assert.deepEqual(cfg.rufloDaemon.receipts[path.resolve(cwd)].configKeys, + { 'daemon.idleSecs': 0, 'daemon.resourceThresholds.minFreeMemoryPercent': 0 }); + } finally { rmrf(cwd); } +}); diff --git a/tests/kit/sync-dry-run-preview.test.mjs b/tests/kit/sync-dry-run-preview.test.mjs index f3def05e..19c1b9ec 100644 --- a/tests/kit/sync-dry-run-preview.test.mjs +++ b/tests/kit/sync-dry-run-preview.test.mjs @@ -136,6 +136,44 @@ const DRY = (over = {}) => ({ 'dry-run': true, 'no-upgrade': false, yes: false, const failedDates = async () => ({ code: 1, stdout: '', stderr: 'offline' }); const NOT_CHECKED = /versions not checked online \(offline or timed out\); this plan uses the versions ak recorded/; +test('versions-only dry run previews conditional daemon convergence in a Ruflo project without writes', async () => { + seed(); + const marker = path.join(PROJECT, '.claude-flow', 'config.yaml'); + fs.mkdirSync(path.dirname(marker), { recursive: true }); + fs.writeFileSync(marker, 'daemon:\n maxConcurrent: 7\n'); + const beforeHome = snapshot(HOME); + const beforeProject = snapshot(PROJECT); + try { + const { result, out } = await syncLines({ + flags: DRY(), fetchLatest: async () => null, releaseDatesRunner: failedDates, + collectFn: async () => [{ subsystem: 'versions', level: 'warn', message: 'ruflo update available', fix: 'sync upgrades', repair: 'sync' }], + }); + assert.equal(result, 0, out); + assert.match(out, /\[daemons\].*installed Ruflo version/); + assert.doesNotMatch(out, /daemon\.idleSecs.*0/, 'the target version is not yet known'); + assertUnchanged(beforeHome, HOME); + assertUnchanged(beforeProject, PROJECT); + } finally { fs.rmSync(marker); } +}); + +test('versions-triggered daemon preview honors skip daemons and skip versions', async () => { + seed(); + const marker = path.join(PROJECT, '.claude-flow', 'config.yaml'); + fs.mkdirSync(path.dirname(marker), { recursive: true }); + fs.writeFileSync(marker, 'daemon:\n maxConcurrent: 7\n'); + const collectFn = async () => [{ subsystem: 'versions', level: 'warn', message: 'update available', fix: 'sync upgrades', repair: 'sync' }]; + try { + const daemonSkipped = await syncLines({ + flags: DRY({ skip: ['daemons'] }), fetchLatest: async () => null, releaseDatesRunner: failedDates, collectFn, + }); + assert.match(daemonSkipped.out, /skipped by request: \[daemons\]/); + const versionSkipped = await syncLines({ + flags: DRY({ skip: ['versions'] }), fetchLatest: async () => null, releaseDatesRunner: failedDates, collectFn, + }); + assert.doesNotMatch(versionSkipped.out, /\[daemons\]/); + } finally { fs.rmSync(marker); } +}); + test('a dry run looks the versions up online and plans the upgrade a fresh cache does not know about, recording nothing', () => { const root = seed({ age: HOUR }); const env = spawnEnv(HOME); diff --git a/tests/kit/sync-host-repair.test.mjs b/tests/kit/sync-host-repair.test.mjs index 99977555..b17e74a4 100644 --- a/tests/kit/sync-host-repair.test.mjs +++ b/tests/kit/sync-host-repair.test.mjs @@ -59,7 +59,7 @@ test('the versions step verifies host CLIs it upgrades', () => { assert.deepEqual(sync.hostUpgradeOptions('ruflo'), {}); }); -// ── final-review fix: re-record evidence after a repair succeeds ─────────── +// ── re-record evidence after a repair succeeds ─────────── // The Important finding: `installState()` at the top of the loop records the // PRE-repair evidence; without a re-probe after `install()` succeeds, that // stale row is the last thing written — a later `ak status`/this sync's own diff --git a/tests/kit/sync-self-freshness.test.mjs b/tests/kit/sync-self-freshness.test.mjs index 6d726340..2bfce5d2 100644 --- a/tests/kit/sync-self-freshness.test.mjs +++ b/tests/kit/sync-self-freshness.test.mjs @@ -113,6 +113,103 @@ test('a failed next lookup retains its cached candidate without claiming a fresh assert.equal(loadKitConfig().versionCheck.self.last, 1); }); +test('partial self answers retry once per TTL without renewing the cached observation', async t => { + seed(); + const cfg = loadKitConfig(); + cfg.versionCheck.self = { last: 100, observedAt: 80, best: { version: '4.0.0-alpha.50', tag: 'next' } }; + writeKitConfig(home, cfg); + let now = 200_000_000; + t.mock.method(Date, 'now', () => now); + const tags = []; + const fetchLatest = async (_pkg, tag) => { tags.push(tag); return tag === 'latest' ? '4.0.0-alpha.0' : null; }; + const first = await selfDrift({ pkgRoot, fetchLatest }); + assert.deepEqual([first.latest, first.tag], ['4.0.0-alpha.50', 'next']); + assert.deepEqual(loadKitConfig().versionCheck.self, { + last: 100, observedAt: 80, best: { version: '4.0.0-alpha.50', tag: 'next' }, + attempt: { at: now, tags: ['latest', 'next'] }, + }); + const afterFirst = fs.readFileSync(paths.kitConfigPath(), 'utf8'); + now += 1000; + assert.equal((await selfDrift({ pkgRoot, fetchLatest })).latest, first.latest); + assert.deepEqual(tags, ['latest', 'next']); + assert.equal(fs.readFileSync(paths.kitConfigPath(), 'utf8'), afterFirst); + await selfDrift({ pkgRoot, force: true, fetchLatest }); + assert.deepEqual(tags, ['latest', 'next', 'latest', 'next']); + now += 24 * 3600_000; + await selfDrift({ pkgRoot, fetchLatest }); + assert.deepEqual(tags, ['latest', 'next', 'latest', 'next', 'latest', 'next']); + assert.equal(loadKitConfig().versionCheck.self.observedAt, 80); + now += 1000; + const recovered = await selfDrift({ pkgRoot, force: true, fetchLatest: async (_pkg, tag) => + tag === 'next' ? '4.0.0-alpha.51' : null }); + assert.equal(recovered.latest, '4.0.0-alpha.51'); + assert.deepEqual(loadKitConfig().versionCheck.self, + { last: now, observedAt: now, best: { version: '4.0.0-alpha.51', tag: 'next' }, lastTags: ['latest', 'next'] }); +}); + +test('a stable lookup cannot make an untried prerelease channel fresh', async t => { + seed('4.0.0'); + const cfg = loadKitConfig(); + cfg.versionCheck.self = { last: 100, observedAt: 80, best: { version: '4.0.0', tag: 'latest' } }; + writeKitConfig(home, cfg); + let now = 200_000_000; + t.mock.method(Date, 'now', () => now); + const stableTags = []; + await selfDrift({ pkgRoot, fetchLatest: async (_pkg, tag) => { + stableTags.push(tag); return null; + } }); + assert.deepEqual(stableTags, ['latest']); + const stable = loadKitConfig().versionCheck.self; + assert.deepEqual(stable, { + last: now, observedAt: 80, best: { version: '4.0.0', tag: 'latest' }, lastTags: ['latest'], + }); + fs.writeFileSync(path.join(pkgRoot, 'package.json'), JSON.stringify({ name: KIT_PKG, version: '4.0.0-alpha.1' })); + now += 1000; + const prereleaseTags = []; + const offline = async (_pkg, tag) => { prereleaseTags.push(tag); return null; }; + const first = await selfDrift({ pkgRoot, fetchLatest: offline }); + assert.deepEqual(prereleaseTags, ['latest', 'next']); + assert.deepEqual([first.latest, first.tag], ['4.0.0', 'latest']); + const afterFirst = fs.readFileSync(paths.kitConfigPath(), 'utf8'); + await selfDrift({ pkgRoot, fetchLatest: offline }); + assert.deepEqual(prereleaseTags, ['latest', 'next']); + assert.equal(fs.readFileSync(paths.kitConfigPath(), 'utf8'), afterFirst); + assert.equal(loadKitConfig().versionCheck.self.observedAt, 80); +}); + +test('a legacy latest-only record does not suppress an untried next channel', async t => { + seed('4.0.0-alpha.1'); + const cfg = loadKitConfig(); + cfg.versionCheck.self = { last: 200_000_000, observedAt: 100, best: { version: '4.0.0', tag: 'latest' } }; + writeKitConfig(home, cfg); + t.mock.method(Date, 'now', () => 200_001_000); + const tags = []; + await selfDrift({ pkgRoot, fetchLatest: async (_pkg, tag) => { tags.push(tag); return null; } }); + assert.deepEqual(tags, ['latest', 'next']); + assert.equal(loadKitConfig().versionCheck.self.observedAt, 100); +}); + +test('a successful stable observation also leaves next untried after a channel switch', async t => { + seed('4.0.0'); + const cfg = loadKitConfig(); + cfg.versionCheck.self.last = 100; + writeKitConfig(home, cfg); + let now = 200_000_000; + t.mock.method(Date, 'now', () => now); + await selfDrift({ pkgRoot, fetchLatest: async () => '4.0.1' }); + assert.deepEqual(loadKitConfig().versionCheck.self.lastTags, ['latest']); + assert.equal(loadKitConfig().versionCheck.self.observedAt, now); + fs.writeFileSync(path.join(pkgRoot, 'package.json'), JSON.stringify({ name: KIT_PKG, version: '4.0.0-alpha.1' })); + now += 1000; + const tags = []; + const result = await selfDrift({ pkgRoot, fetchLatest: async (_pkg, tag) => { + tags.push(tag); return tag === 'next' ? '4.1.0-alpha.1' : null; + } }); + assert.deepEqual(tags, ['latest', 'next']); + assert.deepEqual([result.latest, result.tag], ['4.1.0-alpha.1', 'next']); + assert.deepEqual(loadKitConfig().versionCheck.self.lastTags, ['latest', 'next']); +}); + test('stable installs reject cached next-channel candidates when latest is unavailable', async () => { seed('4.0.0'); const cfg = loadKitConfig(); @@ -145,6 +242,68 @@ test('a stable install whose record holds only a next-channel candidate is not r assert.deepEqual(loadKitConfig().versionCheck.self.best, { version: '5.0.0-alpha.1', tag: 'next' }); }); +test('offline self attempts are scoped to tags and do not persist in read-only modes', async t => { + seed('4.0.0'); + const cfg = loadKitConfig(); + cfg.versionCheck.self = { last: 100, observedAt: 80, best: { version: '5.0.0-alpha.1', tag: 'next' } }; + writeKitConfig(home, cfg); + let now = 200_000_000; + t.mock.method(Date, 'now', () => now); + const tags = []; + const fetchLatest = async (_pkg, tag) => { tags.push(tag); return null; }; + assert.equal((await selfDrift({ pkgRoot, fetchLatest })).latest, null); + assert.deepEqual(loadKitConfig().versionCheck.self, { + last: 100, observedAt: 80, best: { version: '5.0.0-alpha.1', tag: 'next' }, + attempt: { at: now, tags: ['latest'] }, + }); + const afterFirst = fs.readFileSync(paths.kitConfigPath(), 'utf8'); + await selfDrift({ pkgRoot, fetchLatest }); + assert.deepEqual(tags, ['latest']); + assert.equal(fs.readFileSync(paths.kitConfigPath(), 'utf8'), afterFirst); + await selfDrift({ pkgRoot, force: true, fetchLatest }); + assert.deepEqual(tags, ['latest', 'latest']); + fs.writeFileSync(path.join(pkgRoot, 'package.json'), JSON.stringify({ name: KIT_PKG, version: '4.0.0-alpha.1' })); + await selfDrift({ pkgRoot, fetchLatest }); + assert.deepEqual(tags, ['latest', 'latest', 'latest', 'next'], 'the untried channel is probed'); + const beforeReadOnly = fs.readFileSync(paths.kitConfigPath(), 'utf8'); + await selfDrift({ pkgRoot, force: true, record: false, fetchLatest }); + assert.deepEqual(tags.slice(-2), ['latest', 'next']); + assert.equal(fs.readFileSync(paths.kitConfigPath(), 'utf8'), beforeReadOnly); + await selfDrift({ pkgRoot, force: true, cacheOnly: true, fetchLatest }); + assert.equal(fs.readFileSync(paths.kitConfigPath(), 'utf8'), beforeReadOnly); + assert.equal(tags.length, 6); + now += 24 * 3600_000; + await selfDrift({ pkgRoot, fetchLatest }); + assert.equal(tags.length, 8); +}); + +test('malformed self attempt stamps cannot suppress an offline retry', async t => { + seed('4.0.0'); + const cfg = loadKitConfig(); + cfg.versionCheck.self = { + last: 100, best: { version: '5.0.0-alpha.1', tag: 'next' }, + attempt: { at: Number.MAX_SAFE_INTEGER, tags: ['latest'] }, + }; + writeKitConfig(home, cfg); + t.mock.method(Date, 'now', () => 200_000_000); + let calls = 0; + await selfDrift({ pkgRoot, fetchLatest: async () => { calls += 1; return null; } }); + assert.equal(calls, 1); + assert.deepEqual(loadKitConfig().versionCheck.self.attempt, { at: 200_000_000, tags: ['latest'] }); + for (const attempt of [ + { at: '200000000', tags: ['latest'] }, + { at: 200_000_000.5, tags: ['latest'] }, + { at: 200_000_000, tags: ['next'] }, + { at: 200_000_000, tags: 'latest' }, + ]) { + const next = loadKitConfig(); + next.versionCheck.self.attempt = attempt; + writeKitConfig(home, next); + await selfDrift({ pkgRoot, fetchLatest: async () => { calls += 1; return null; } }); + } + assert.equal(calls, 5); +}); + test('successful registry observations supersede cached versions even after a channel rollback', async () => { seed(); const cfg = loadKitConfig(); diff --git a/tests/kit/system-command.test.mjs b/tests/kit/system-command.test.mjs index bff84153..d77828b0 100644 --- a/tests/kit/system-command.test.mjs +++ b/tests/kit/system-command.test.mjs @@ -243,3 +243,11 @@ test('ak system --help documents only the current spellings', () => { assert.match(r.stdout, /\[--refresh\[=live\|machine\]\] \[--project-trees\] \[--json\]/); assert.doesNotMatch(r.stdout, /--deep\b/, 'only the current spellings'); }); +test('system reports pure exclusions, mixed activity and unresolved ownership separately with Cowork coverage', async () => { + const collector = fakeCollector({ snapshot: { projects: { projects: [], importedExcluded: 4, importedMixed: 2, importedUnresolved: 3 } } }); + const result = await captureLog(() => system.run({ flags: {}, deps: { collector } })); + assert.match(result.out, /4 confirmed pure imported copies excluded/); + assert.match(result.out, /2 mixed files retain proven native activity/); + assert.match(result.out, /3 files have unresolved bounded ownership/); + assert.match(result.out, /dedicated Cowork transcript source is not covered/); +}); diff --git a/tests/kit/system-runtime-app-labels.test.mjs b/tests/kit/system-runtime-app-labels.test.mjs new file mode 100644 index 00000000..13ac7759 --- /dev/null +++ b/tests/kit/system-runtime-app-labels.test.mjs @@ -0,0 +1,36 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { run } from '../../src/commands/system.mjs'; +import { captureLog } from './helpers/home-sandbox.mjs'; + +const measured = (value) => ({ status: 'measured', value, partial: false }); +const unknown = (reason) => ({ status: 'unknown', reason }); + +test('ak system renders application, coding-agent host, service, and unknown runtime rows', async () => { + const row = (pid, host, application, source) => ({ pid, host, application, source, + cpuPercent: measured(2), rssBytes: measured(1000), uptimeMs: measured(2000) }); + const processes = [ + row(101, null, 'Claude Desktop', measured({ kind: 'desktop-app', label: 'Claude Desktop' })), + row(102, 'claude', null, measured({ kind: 'repository', label: 'work' })), + row(103, null, 'ChatGPT desktop app', measured({ kind: 'desktop-app', label: 'ChatGPT desktop app' })), + row(104, 'codex', null, measured({ kind: 'host-service', label: 'Codex app service' })), + row(105, null, null, unknown('not attributable — access denied')), + ]; + const snapshot = { platform: 'darwin', generatedAt: 'now', snapshot: { present: false, reason: 'not scanned' }, + runtime: { ephemeral: true, processes: measured(processes), totals: { + processCount: measured(5), rssBytes: measured(5000), cpuPercent: measured(10), + }, daemons: { count: measured(0), staleCount: measured(0) } } }; + const { result, out } = await captureLog(() => run({ flags: { json: false }, deps: { + collector: { read: async () => snapshot }, now: () => 0, + } })); + assert.equal(result, 0); + assert.match(out, /CODING-AGENT HOST \/ DESKTOP APPLICATION/); + assert.match(out, /Claude Desktop/); + assert.match(out, /ChatGPT desktop app/); + assert.match(out, /Codex app service/); + assert.match(out, /work/); + assert.match(out, /Unknown process/); + assert.match(out, /unattributed: not attributable — access denied/); + assert.match(out, /processes\s+5/); + assert.match(out, /memory \(RSS\)\s+5\.00 KB/); +}); diff --git a/tests/kit/system-summary.test.mjs b/tests/kit/system-summary.test.mjs index dbb9e4f2..441d92ec 100644 --- a/tests/kit/system-summary.test.mjs +++ b/tests/kit/system-summary.test.mjs @@ -1,3 +1,5 @@ +import * as sessionVocabulary from '../../src/lib/session-surface.mjs'; +import { censusDisclosure } from '../../src/lib/census-presentation.mjs'; // GET /api/system/summary (#237 M4, decision 8). The System page drew from // GET /api/system, which ships the whole persisted catalog: every presence // fact repeated in item.presence, item.consumerBindings, item.artifacts and @@ -366,7 +368,7 @@ test('the projects summary drops per-project stack detection and node_modules ro } const row = summary.projects.projects[0]; assert.deepEqual(Object.keys(row).sort(), - ['path', 'label', 'hosts', 'totalBytes', 'lastActivity', 'loc', 'remote', 'repository'].sort()); + ['path', 'label', 'hosts', 'totalBytes', 'lastActivity', 'loc', 'remote', 'repository', 'sessionOrigins'].sort()); assert.equal('stack' in row, false, 'framework/manifest detection is not rendered (system-projects.mjs langCell)'); assert.equal('nodeModulesRoots' in row, false); assert.deepEqual(Object.keys(row.loc).sort(), ['total', 'languages', 'byLanguage'].sort()); @@ -374,7 +376,7 @@ test('the projects summary drops per-project stack detection and node_modules ro assert.deepEqual(Object.keys(row.repository).sort(), ['repositoryId', 'kind', 'root'].sort()); assert.deepEqual(Object.keys(row.remote).sort(), ['status', 'webUrl', 'raw'].sort()); const discovery = summary.projects.discoveryProjects[0]; - assert.deepEqual(Object.keys(discovery).sort(), ['path', 'label', 'hosts', 'repository'].sort()); + assert.deepEqual(Object.keys(discovery).sort(), ['path', 'label', 'hosts', 'repository', 'sessionOrigins'].sort()); }); test('the projects summary keeps byLanguage for a pre-languages loc, so an old carried-forward snapshot still renders bars', () => { @@ -510,13 +512,15 @@ function systemClient({ fetchImpl } = {}) { 'storageHostTotals', 'renderSysSummary', 'renderSysConsumers', 'renderSysReclaim', 'CHART_EXCLUDED_CATEGORIES', 'transcriptIdOf', 'renderSysKpis']; const readout = load('system-readout', { esc, fmtNum, fmtTok, document, window }, readoutExports); + const surfaceHelpers = load('session-presentation', { ...sessionVocabulary, esc }, ['projectSurfacesHtml']); const projects = load('system-projects', { + ...surfaceHelpers, censusDisclosure, ...readout, esc, authHeaders: () => ({}), formatLocalDateTime: () => null, formatLocalDateTimeLong: () => null, shortSessionId: (s) => s, ago: () => 'just now', fmtNum, fmtTok, limAge, pct, repositoryTree, SYSTEM: null, systemBusy: false, systemPollTimer: null, consMode: 'ranked', document, window, fetch: fetchImpl ?? (() => Promise.reject(new Error('no fetch in this test'))), setTimeout: () => 0, clearTimeout: () => {}, - }, ['renderSysCatalog', 'loadSystem', 'renderSysStorage', 'renderSysProjects']); + }, ['renderSysCatalog', 'loadSystem', 'renderSysStorage', 'renderSysProjects', 'renderSysRuntime']); return { document, readout, projects }; } @@ -527,6 +531,29 @@ function catalogHtml(payload) { return Object.fromEntries([...client.document.elements].map(([id, el]) => [id, { html: el.innerHTML, text: el.textContent }])); } +test('Runtime process renderer names desktop applications separately from coding-agent hosts', () => { + const client = systemClient(); + const row = (pid, host, application, source) => ({ pid, host, application, source, + uptimeMs: meas(2000), cpuPercent: meas(2), rssBytes: meas(1000) }); + client.projects.renderSysRuntime({ runtime: { processes: meas([ + row(1, null, 'Claude Desktop', meas({ kind: 'desktop-app', label: 'Claude Desktop' })), + row(2, 'claude', null, meas({ kind: 'repository', label: 'work' })), + row(3, null, 'ChatGPT desktop app', meas({ kind: 'desktop-app', label: 'ChatGPT desktop app' })), + row(4, 'codex', null, meas({ kind: 'host-service', label: 'Codex app service' })), + row(5, null, '', { status: 'unknown', reason: 'not attributable — ' }), + ]) } }); + const html = client.document.getElementById('sys-procs').innerHTML; + assert.match(html, /Coding-agent host \/ desktop application/); + assert.match(html, /Claude Desktop/); + assert.match(html, /ChatGPT desktop app/); + assert.match(html, /Codex app service/); + assert.match(html, /work/); + assert.match(html, /not attributable/); + assert.match(html, /<unknown>/); + assert.match(html, /<denied>/); + assert.doesNotMatch(html, /|/); +}); + test('the KPI band and every catalog card render identically from the summary and the full payload', () => { const full = fullPayload(6); const fromFull = catalogHtml(full); @@ -592,13 +619,13 @@ test('the projects note says how many imported copies discovery set aside, and n // ── The page reads the slim endpoint ──────────────────────────────────────── -test('loadSystem fetches /api/system/summary, deep refresh parameters included', async () => { +test('loadSystem only re-reads /api/system/summary', async () => { const urls = []; const fetchImpl = (url) => { urls.push(url); return Promise.resolve({ json: () => Promise.resolve(systemSummaryPayload(fullPayload(1))) }); }; const { projects } = systemClient({ fetchImpl }); await projects.loadSystem(); await projects.loadSystem(true, false); - assert.deepEqual(urls, ['/api/system/summary', '/api/system/summary?refresh=deep&trees=0']); + assert.deepEqual(urls, ['/api/system/summary', '/api/system/summary']); }); // ── The routes ────────────────────────────────────────────────────────────── @@ -661,15 +688,14 @@ test('GET /api/system/summary serves the projection; GET /api/system stays compl assert.ok(complete.catalog.items[0].presence[0].itemPath); }); -test('GET /api/system/summary?refresh=deep starts the scan and answers with its running state', async (t) => { +test('GET /api/system/summary rejects measurement queries before reading the collector', async (t) => { const collector = fakeCollector(); const cwd = tempDir('ak-system-summary'); const server = await startDashboard({ port: 0, cwd, system: collector, usage: {}, ...hermeticMaintenance() }); t.after(() => server.close()); const r = await request(server, '/api/system/summary?refresh=deep&trees=0'); - assert.equal(r.status, 200); + assert.equal(r.status, 400); const body = JSON.parse(r.body); - assert.deepEqual(collector.calls.refreshDeep, [{ includeProjectTrees: false }]); - assert.deepEqual(body.scan, { running: true, phase: 'catalog' }); - assert.equal('artifacts' in body.catalog, false); + assert.deepEqual(body, { error: 'start a refresh with POST /api/refresh' }); + assert.deepEqual(collector.calls.refreshDeep, []); }); diff --git a/tests/kit/telemetry-cli.test.mjs b/tests/kit/telemetry-cli.test.mjs index f7f2ad7b..a91132f2 100644 --- a/tests/kit/telemetry-cli.test.mjs +++ b/tests/kit/telemetry-cli.test.mjs @@ -66,6 +66,15 @@ test('should_keepErrorsGeneric_when_malformedFilesContainSecrets', t => { const result = cli(['validate', file]); assert.equal(result.status, 2); assert.doesNotMatch(result.stdout + result.stderr, /SECRET/); + assert.deepEqual(Object.keys(JSON.parse(result.stdout)), ['error', 'exitCode']); + assert.equal(JSON.parse(result.stdout).exitCode, 2); +}); +test('telemetry parser rejection remains private and machine-readable', () => { + const result = cli(['schema', '--private-path=/secret/SECRET']); + assert.equal(result.status, 2); + assert.deepEqual(Object.keys(JSON.parse(result.stdout)), ['error', 'exitCode']); + assert.equal(JSON.parse(result.stdout).exitCode, 2); + assert.doesNotMatch(result.stdout + result.stderr, /SECRET|\/secret/); }); test('should_degradeSourcesIndependently_when_usageReaderFails', async () => { const { collectSnapshot } = await import('../../src/lib/telemetry/collect.mjs'); diff --git a/tests/kit/trace-ort.test.mjs b/tests/kit/trace-ort.test.mjs new file mode 100644 index 00000000..58e62773 --- /dev/null +++ b/tests/kit/trace-ort.test.mjs @@ -0,0 +1,105 @@ +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { spawnSync } from 'node:child_process'; +import { fileURLToPath, pathToFileURL } from 'node:url'; +import { test } from 'node:test'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; + +const hook = new URL('../../scripts/trace-ort.mjs', import.meta.url).href; + +function pkg(root, name, version, { commonjs = false, malformed = false } = {}) { + const dir = path.join(root, 'node_modules', ...name.split('/')); + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, 'package.json'), malformed ? '{bad' : JSON.stringify({ name, version, main: 'index.js', type: commonjs ? 'commonjs' : 'module' })); + fs.writeFileSync(path.join(dir, 'index.js'), commonjs ? 'module.exports = 1;\n' : 'export default 1;\n'); + return dir; +} + +function run(root, source, log = path.join(root, 'trace.jsonl'), hookUrl = hook) { + const home = path.join(root, 'home'); + const entry = path.join(root, 'entry.mjs'); + fs.writeFileSync(entry, source); + const env = spawnEnv(home, { NODE_OPTIONS: `--import=${hookUrl}`, TRACE_ORT_LOG: log }); + const child = spawnSync(process.execPath, [entry], { cwd: root, env, encoding: 'utf8' }); + const records = fs.existsSync(log) ? fs.readFileSync(log, 'utf8').trim().split('\n').filter(Boolean).map(JSON.parse) : []; + return { child, records }; +} + +test('records CJS and ESM resolution once per distinct package root', (t) => { + const root = tempDir('trace-ort', t); + const huggingface = pkg(root, '@huggingface/transformers', '4.3.0'); + const ort = pkg(root, 'onnxruntime-node', '1.30.0', { commonjs: true }); + const nested = path.join(root, 'nested'); + const old = pkg(nested, '@huggingface/transformers', '3.8.1'); + const { child, records } = run(root, `import '@huggingface/transformers'; +import '@huggingface/transformers'; +import { createRequire } from 'node:module'; +const require = createRequire(import.meta.url); +require('onnxruntime-node'); +require('onnxruntime-node'); +const nestedRequire = createRequire(${JSON.stringify(path.join(nested, 'entry.cjs'))}); +nestedRequire('@huggingface/transformers'); +`); + assert.equal(child.status, 0, child.stderr); + assert.equal(records.filter((r) => r.type === 'start').length, 1); + assert.deepEqual(records.filter((r) => r.type === 'package').map((r) => [r.name, r.version, r.root]).sort(), [ + ['@huggingface/transformers', '3.8.1', old], + ['@huggingface/transformers', '4.3.0', huggingface], + ['onnxruntime-node', '1.30.0', ort], + ].sort()); + assert.ok(records.every((r) => r.pid === records[0].pid)); +}); + +test('reports unknown version without executing package metadata', (t) => { + const root = tempDir('trace-ort-version', t); + pkg(root, '@xenova/transformers', undefined); + const { child, records } = run(root, "import '@xenova/transformers';\n"); + assert.equal(child.status, 0, child.stderr); + assert.equal(records.find((r) => r.type === 'package')?.version, 'unknown'); +}); + +test('an unusable or absent log target does not change the command result', (t) => { + const root = tempDir('trace-ort-failure', t); + pkg(root, 'onnxruntime-node', '1.30.0', { commonjs: true }); + const source = "import { createRequire } from 'node:module'; createRequire(import.meta.url)('onnxruntime-node');\n"; + const invalid = run(root, source, path.join(root, 'missing', 'trace.jsonl')); + assert.equal(invalid.child.status, 0, invalid.child.stderr); + assert.match(invalid.child.stderr, /trace-ort.*log/i); + const absent = run(root, source, ''); + assert.equal(absent.child.status, 0, absent.child.stderr); + assert.match(absent.child.stderr, /trace-ort.*log/i); +}); + +test('NODE_OPTIONS traces inherited child processes with separate start receipts', (t) => { + const root = tempDir('trace-ort-child', t); + pkg(root, 'onnxruntime-node', '1.21.0', { commonjs: true }); + const childFile = path.join(root, 'child.cjs'); + fs.writeFileSync(childFile, "require('onnxruntime-node');\n"); + const { child, records } = run(root, `import { spawnSync } from 'node:child_process'; +const result = spawnSync(process.execPath, [${JSON.stringify(childFile)}], { encoding: 'utf8' }); +if (result.status !== 0) process.exit(result.status || 1); +`); + assert.equal(child.status, 0, child.stderr); + const starts = records.filter((r) => r.type === 'start'); + assert.equal(starts.length, 2); + assert.equal(new Set(starts.map((r) => r.pid)).size, 2); + assert.equal(records.filter((r) => r.type === 'package').length, 1); + assert.equal(records.find((r) => r.type === 'package')?.version, '1.21.0'); +}); + +test('preloads a hook from a path containing spaces', (t) => { + const root = tempDir('trace ort space', t); + const hookDir = path.join(root, 'hook with spaces'); + fs.mkdirSync(hookDir); + const copiedHook = path.join(hookDir, 'trace ort.mjs'); + fs.copyFileSync(fileURLToPath(hook), copiedHook); + pkg(root, 'onnxruntime-node', '1.30.0', { commonjs: true }); + const { child, records } = run(root, + "import { createRequire } from 'node:module'; createRequire(import.meta.url)('onnxruntime-node');\n", + path.join(root, 'trace output.jsonl'), pathToFileURL(copiedHook).href); + assert.equal(child.status, 0, child.stderr); + assert.equal(records.filter((r) => r.type === 'start').length, 1); + assert.equal(records.find((r) => r.type === 'package')?.version, '1.30.0'); +}); diff --git a/tests/kit/trust-manifest.test.mjs b/tests/kit/trust-manifest.test.mjs index 30b07ea3..cea6267d 100644 --- a/tests/kit/trust-manifest.test.mjs +++ b/tests/kit/trust-manifest.test.mjs @@ -250,7 +250,7 @@ test('ruflo components governance disclosure states enforcement is scoped to the assert.match(text, /project's own existing \.harness\/mcp-policy\.json is left alone/); }); -// Branch 3, Task 2.2: project setup discloses both daemon writes and the opt-out. +// project setup discloses both daemon writes and the opt-out. test('project setup discloses the managed Ruflo daemon settings and how to opt out', () => { const group = (cfg, project) => setupTrustManifest(cfg, { hosts: [], project }) .find((g) => g.componentId === 'ruflo-daemon'); diff --git a/tests/kit/ui-chrome-launch.test.mjs b/tests/kit/ui-chrome-launch.test.mjs index e279710a..e0e43aa7 100644 --- a/tests/kit/ui-chrome-launch.test.mjs +++ b/tests/kit/ui-chrome-launch.test.mjs @@ -61,3 +61,44 @@ test('launchChrome removes its temp folder when Chrome fails to start', async (t await assert.rejects(launchChrome(), /no chrome/); assert.equal(fs.existsSync(dir), false); }); + +test('launchChrome excludes credentials and caller env while retaining launch options', async (t) => { + const { chromium } = await import('playwright'); + const { launchChrome } = await import('../ui/helpers/launch-chrome.mjs'); + const injected = ['AK_CHROME_SECRET_SENTINEL', 'AQE_EMBEDDER_PROVIDER', 'CODEX_HOME', 'CLAUDE_CONFIG_DIR']; + const original = Object.fromEntries(injected.map((key) => [key, process.env[key]])); + for (const key of injected) process.env[key] = `sentinel-${key}`; + t.after(() => { + for (const key of injected) { + if (original[key] === undefined) delete process.env[key]; + else process.env[key] = original[key]; + } + }); + let seen; + t.mock.method(chromium, 'launch', async (options) => { seen = options; return { close: async () => {} }; }); + const browser = await launchChrome({ headless: false, args: ['--disable-gpu'], env: { AK_CHROME_SECRET_SENTINEL: 'override' } }); + try { + assert.equal(seen.headless, false); + assert.deepEqual(seen.args, ['--disable-gpu']); + assert.equal(seen.env.AK_CHROME_SECRET_SENTINEL, undefined); + for (const key of injected) { + assert.equal(seen.env[key], undefined, `${key} should not reach Chrome`); + } + for (const key of ['HOME', 'USERPROFILE', 'XDG_CONFIG_HOME', 'XDG_CACHE_HOME', 'APPDATA', 'LOCALAPPDATA']) { + assert.ok(seen.env[key].startsWith(seen.env.TMPDIR), `${key} should be private`); + } + } finally { await browser.close(); } +}); + +test('Chrome environment selects Windows names without duplicate case variants', async () => { + const { chromeEnv } = await import('../ui/helpers/launch-chrome.mjs'); + const env = chromeEnv({ Path: 'first', PATH: 'second', display: ':8', SystemRoot: 'C:\\Windows', + temp: 'real-temp', TOKEN: 'secret' }, 'private-temp', 'win32'); + assert.equal(env.Path, 'second'); + assert.equal(env.SystemRoot, 'C:\\Windows'); + assert.equal(env.TEMP, 'private-temp'); + assert.equal(env.TMP, 'private-temp'); + assert.equal(Object.keys(env).filter((key) => key.toUpperCase() === 'PATH').length, 1); + assert.equal(Object.keys(env).filter((key) => key.toUpperCase() === 'TEMP').length, 1); + assert.equal(env.TOKEN, undefined); +}); diff --git a/tests/kit/uninstall-command.test.mjs b/tests/kit/uninstall-command.test.mjs index c443e2a1..73145ef0 100644 --- a/tests/kit/uninstall-command.test.mjs +++ b/tests/kit/uninstall-command.test.mjs @@ -492,7 +492,7 @@ test('a failed typesafe package removal keeps kit.json under --purge', async () test.after(() => rmrf(HOME)); -// Branch 3, Task 2.2: uninstall puts back what ak changed for Ruflo's daemon +// uninstall puts back what ak changed for Ruflo's daemon // in every receipted project: the autoStart value and the flat keys (and the // .claude-flow/config.json ak created). test('uninstall restores autoStart and removes the managed daemon keys in receipted projects', async () => { diff --git a/tests/kit/upstream-watch-dispatch.test.mjs b/tests/kit/upstream-watch-dispatch.test.mjs index 88966229..03976d76 100644 --- a/tests/kit/upstream-watch-dispatch.test.mjs +++ b/tests/kit/upstream-watch-dispatch.test.mjs @@ -5,7 +5,7 @@ import test from 'node:test'; import assert from 'node:assert/strict'; import { eventLine } from '../../scripts/upstream-watch/classify.mjs'; -import { FIRE_HEADERS, FIRE_URL, SAME_REPO_PR, createDispatcher, dispatch } from '../../scripts/upstream-watch/dispatch.mjs'; +import { FIRE_HEADERS, FIRE_URL, PR_OBSERVE_DAYS, SAME_REPO_PR, createDispatcher, dispatch, sessionList } from '../../scripts/upstream-watch/dispatch.mjs'; const NOW = new Date('2026-10-02T14:17:00Z'); const RECORDED_AT = '2026-10-02T14:17:00Z'; @@ -26,9 +26,15 @@ function fakeDispatcher({ exists = false, pr = null, session = 'https://claude.a fire: async (text) => { calls.fire.push(text); if (fireError) throw new Error(fireError); return session; }, }; } -const run = (dispatcher, records = [], { dryRun, list = [released] } = {}) => dispatch({ released: list, records, dispatcher, repo: 'pacphi/agentic-kit', sentinel: 'UPSTREAM-WATCH', now: NOW, recordedAt: RECORDED_AT, dryRun }); +const run = (dispatcher, records = [], { dryRun, list = [released], eligibleIds = new Set([ID]), now = NOW } = {}) => dispatch({ released: list, records, dispatcher, repo: 'pacphi/agentic-kit', sentinel: 'UPSTREAM-WATCH', now, recordedAt: RECORDED_AT, dryRun, eligibleIds }); const NOTHING = { records: [], errors: [], wouldFire: [], deferred: [] }; +test('exhausted firing links read naturally at any count', () => { + assert.equal(sessionList(['one']), 'one'); + assert.equal(sessionList(['one', 'two']), 'one and two'); + assert.equal(sessionList(['one', 'two', 'three']), 'one, two, and three'); +}); + test('a released fix without a branch fires once and is recorded with its session', async () => { const dispatcher = fakeDispatcher(); const result = await run(dispatcher); @@ -108,6 +114,22 @@ test('a fired branch with an open pull request records dispatch-pr once', async assert.deepEqual(again.calls.pr, []); }); +test('PR observation stops on ineligible status or seven days after latest firing', async () => { + assert.equal(PR_OBSERVE_DAYS, 7); + const first = fired('2026-09-20T14:17:00Z'); + const latest = fired('2026-09-26T14:17:00Z'); + const inactive = fakeDispatcher({ exists: true, pr: 261 }); + assert.deepEqual(await run(inactive, [latest], { list: [], eligibleIds: new Set() }), NOTHING); + assert.deepEqual(inactive.calls.pr, []); + const before = fakeDispatcher({ exists: true, pr: 261 }); + const inside = await run(before, [first, latest], { list: [], now: new Date('2026-10-03T14:16:59Z') }); + assert.equal(inside.records[0].event, 'dispatch-pr'); + assert.deepEqual(before.calls.pr, [['pacphi/agentic-kit', BRANCH]]); + const boundary = fakeDispatcher({ exists: true, pr: 261 }); + assert.deepEqual(await run(boundary, [first, latest], { list: [], now: new Date('2026-10-03T14:17:00Z'), dryRun: true }), NOTHING); + assert.deepEqual(boundary.calls.pr, []); +}); + const fix = (n) => eventLine('UPSTREAM-WATCH', `proffesor-for-testing/agentic-qe#${n}`, 'released', '2026-08-06', { version: '3.14.5', branch: `upstream/proffesor-for-testing-agentic-qe-${n}` }); const gentle = (dispatcher, list, pauses = []) => dispatch({ released: list, records: [], dispatcher, repo: 'pacphi/agentic-kit', sentinel: 'UPSTREAM-WATCH', now: NOW, recordedAt: RECORDED_AT, pause: async (ms) => { pauses.push(ms); } }); @@ -157,7 +179,7 @@ test('the trigger call sends the payload with the documented headers and never p await assert.rejects(createDispatcher({ fetchImpl: empty, env }).fire('x'), /without a session/); }); -test('the trigger call retries a 5xx answer, then reports the request id and body', async () => { +test('the trigger call retries HTTP 503, then reports the request id and body', async () => { const env = { UPSTREAM_DISPATCH_ROUTINE: 'trig_01LmNVKJ4K86joHPvvPtc7yx', UPSTREAM_DISPATCH_TOKEN: 'sk-secret-token' }; const ok = { status: 200, json: async () => ({ claude_code_session_url: 'https://claude.ai/code/session_r' }) }; const busy = { status: 503, headers: new Headers({ 'request-id': 'req_1' }), json: async () => ({ error: { message: 'overloaded' } }) }; @@ -211,3 +233,139 @@ test('a pull request from a fork is never taken for the dispatch pull request', assert.equal(await createDispatcher({ exec: forkOnly }).openPullRequest('pacphi/agentic-kit', BRANCH), null); assert.deepEqual(calls[0].slice(-4), ['--json', 'number,isCrossRepository', '--jq', SAME_REPO_PR]); }); + +// These cases catch retrying an ambiguous or undocumented outcome. +for (const status of [200, 400, 401, 403, 404, 429, 502, 504, 529]) { + test(`HTTP ${status} without a session is not retried`, async () => { + let calls = 0; + const waits = []; + const dispatcher = createDispatcher({ + env: { UPSTREAM_DISPATCH_ROUTINE: 'trig_test', UPSTREAM_DISPATCH_TOKEN: 'test-secret' }, + fetchImpl: async () => { calls++; return { status, json: async () => ({}) }; }, + sleep: async (ms) => { waits.push(ms); }, + }); + await assert.rejects(dispatcher.fire('x'), new RegExp(`HTTP ${status}`)); + assert.equal(calls, 1); + assert.deepEqual(waits, []); + }); +} + +for (const status of [500, 503]) { + test(`HTTP ${status} retries exactly twice with bounded waits`, async () => { + let calls = 0; + const waits = []; + const dispatcher = createDispatcher({ + env: { UPSTREAM_DISPATCH_ROUTINE: 'trig_test', UPSTREAM_DISPATCH_TOKEN: 'test-secret' }, + fetchImpl: async () => { calls++; return { status, json: async () => ({}) }; }, + sleep: async (ms) => { waits.push(ms); }, + }); + await assert.rejects(dispatcher.fire('x'), new RegExp(`HTTP ${status}`)); + assert.equal(calls, 3); + assert.deepEqual(waits, [2000, 4000]); + }); + test(`HTTP ${status} with a session URL never repeats the request`, async () => { + let calls = 0; + const waits = []; + const dispatcher = createDispatcher({ + env: { UPSTREAM_DISPATCH_ROUTINE: 'trig_test', UPSTREAM_DISPATCH_TOKEN: 'test-secret' }, + fetchImpl: async () => { calls++; return { status, json: async () => ({ claude_code_session_url: 'https://claude.ai/code/session_seen' }) }; }, + sleep: async (ms) => { waits.push(ms); }, + }); + await assert.rejects(dispatcher.fire('x'), new RegExp(`HTTP ${status}`)); + assert.equal(calls, 1); + assert.deepEqual(waits, []); + }); +} + +for (const failure of [new Error('connection reset'), new DOMException('timed out', 'TimeoutError')]) { + test(`a thrown ${failure.name} is not retried`, async () => { + let calls = 0; + const waits = []; + const dispatcher = createDispatcher({ + env: { UPSTREAM_DISPATCH_ROUTINE: 'trig_test', UPSTREAM_DISPATCH_TOKEN: 'test-secret' }, + fetchImpl: async () => { calls++; throw failure; }, + sleep: async (ms) => { waits.push(ms); }, + }); + await assert.rejects(dispatcher.fire('x'), { message: failure.message }); + assert.equal(calls, 1); + assert.deepEqual(waits, []); + }); +} + +test('error metadata redacts the configured token before truncation and removes control characters', async () => { + const token = 'synthetic-secret-longer-than-the-remaining-space'; + const dispatcher = createDispatcher({ + env: { UPSTREAM_DISPATCH_ROUTINE: 'trig_test', UPSTREAM_DISPATCH_TOKEN: token }, + fetchImpl: async () => ({ + status: 401, + headers: { get: () => `req\n\x1b${token} ${'r'.repeat(190)}${token}` }, + json: async () => ({ error: { message: `message\r\n${token} ${'m'.repeat(170)}${token}` } }), + }), + }); + const error = await dispatcher.fire('x').catch((failure) => failure); + assert.doesNotMatch(error.message, /synthetic|secret|\p{Cc}/u); + assert.match(error.message, /request-id req.*\[REDACTED\]/); + assert.match(error.message, /message.*\[REDACTED\]/); + assert.ok(error.message.length < 500, 'both metadata fields are bounded'); +}); + +test('non-string error metadata is ignored without coercing objects', async () => { + const dispatcher = createDispatcher({ + env: { UPSTREAM_DISPATCH_ROUTINE: 'trig_test', UPSTREAM_DISPATCH_TOKEN: 'test-secret' }, + fetchImpl: async () => ({ status: 401, headers: { get: () => ({ invalid: true }) }, json: async () => ({ error: { message: { invalid: true } } }) }), + }); + await assert.rejects(dispatcher.fire('x'), { message: 'the routine trigger answered HTTP 401 without a session' }); +}); + +test('a dry run previews three eligible fixes, defers the rest and never sleeps', async () => { + const dispatcher = fakeDispatcher(); + const result = await dispatch({ released: [1, 2, 3, 4, 5].map(fix), records: [], dispatcher, repo: 'pacphi/agentic-kit', sentinel: 'UPSTREAM-WATCH', now: NOW, recordedAt: RECORDED_AT, dryRun: true, pause: async () => { assert.fail('dry run slept'); } }); + assert.deepEqual(result.wouldFire.map((item) => item.id), [1, 2, 3].map((n) => `proffesor-for-testing/agentic-qe#${n}`)); + assert.deepEqual(result.deferred.map((item) => item.id), [4, 5].map((n) => `proffesor-for-testing/agentic-qe#${n}`)); + assert.deepEqual(dispatcher.calls.fire, []); + assert.deepEqual(result.records, []); +}); + +for (const fail of [false, true]) { + test(`PR observation remains bounded after ${fail ? 'a fire failure' : 'the fire cap'}`, async () => { + const dispatcher = fakeDispatcher({ pr: 261, fireError: fail ? 'unavailable' : null }); + const waits = []; + const observed = (n, at) => ({ ...fired(at), id: `proffesor-for-testing/agentic-qe#${n}`, fields: { branch: fix(n).fields.branch, session: `https://claude.ai/code/session_${n}` } }); + const records = [observed(10, '2026-10-01T14:17:00Z'), observed(11, '2026-09-25T14:17:00Z'), observed(12, '2026-10-01T14:17:00Z'), observed(13, '2026-10-01T14:17:00Z'), { ...observed(13, '2026-10-01T14:17:00Z'), event: 'dispatch-pr' }]; + const result = await dispatch({ released: [10, 1, 2, 3, 4, 5].map(fix), records, dispatcher, repo: 'pacphi/agentic-kit', sentinel: 'UPSTREAM-WATCH', now: NOW, recordedAt: RECORDED_AT, eligibleIds: new Set([10, 11, 13].map((n) => `proffesor-for-testing/agentic-qe#${n}`)), pause: async (ms) => { waits.push(ms); } }); + assert.equal(dispatcher.calls.fire.length, fail ? 1 : 3); + assert.deepEqual(waits, fail ? [] : [15000, 15000]); + assert.equal(result.deferred.length, fail ? 4 : 2); + assert.deepEqual(dispatcher.calls.pr, [['pacphi/agentic-kit', 'upstream/proffesor-for-testing-agentic-qe-10']]); + assert.deepEqual(result.records.filter((item) => item.event === 'dispatch-pr').map((item) => item.id), ['proffesor-for-testing/agentic-qe#10']); + }); +} + +for (const status of [500, 503]) { + for (const failure of [new DOMException('body timed out', 'TimeoutError'), new DOMException('body aborted', 'AbortError'), new TypeError('body stream network failure')]) { + test(`HTTP ${status} body ${failure.name} propagates without another POST`, async () => { + let calls = 0; + const waits = []; + const dispatcher = createDispatcher({ + env: { UPSTREAM_DISPATCH_ROUTINE: 'trig_test', UPSTREAM_DISPATCH_TOKEN: 'test-secret' }, + fetchImpl: async () => { calls++; return { status, json: async () => { throw failure; } }; }, + sleep: async (ms) => { waits.push(ms); }, + }); + await assert.rejects(dispatcher.fire('x'), (error) => error === failure); + assert.equal(calls, 1); + assert.deepEqual(waits, []); + }); + } + test(`HTTP ${status} complete malformed JSON retains the bounded status retry`, async () => { + let calls = 0; + const waits = []; + const dispatcher = createDispatcher({ + env: { UPSTREAM_DISPATCH_ROUTINE: 'trig_test', UPSTREAM_DISPATCH_TOKEN: 'test-secret' }, + fetchImpl: async () => { calls++; return new Response('not JSON', { status }); }, + sleep: async (ms) => { waits.push(ms); }, + }); + await assert.rejects(dispatcher.fire('x'), new RegExp(`HTTP ${status} without a session`)); + assert.equal(calls, 3); + assert.deepEqual(waits, [2000, 4000]); + }); +} diff --git a/tests/kit/upstream-watch-fixtures.mjs b/tests/kit/upstream-watch-fixtures.mjs index bb2dee0c..7d440260 100644 --- a/tests/kit/upstream-watch-fixtures.mjs +++ b/tests/kit/upstream-watch-fixtures.mjs @@ -7,6 +7,7 @@ import path from 'node:path'; import { UPSTREAM_REGISTRY_FILE } from '../../src/lib/hook-audit/upstream.mjs'; import { releaseFacts } from '../../scripts/upstream-watch/classify.mjs'; +import { PermanentFetchError } from '../../scripts/upstream-watch/fetch.mjs'; const FIXTURES = path.resolve('tests/fixtures/upstream-watch'); const threads = JSON.parse(fs.readFileSync(path.join(FIXTURES, 'threads.json'), 'utf8')).threads; @@ -55,7 +56,7 @@ export function fixtureFetcher({ authenticated = true, failing = new Set(), flak thread: async (id) => { if (failing.has(id)) throw new Error('HTTP 502'); if (flaky.get(id) > 0) { flaky.set(id, flaky.get(id) - 1); throw new Error('HTTP 502'); } - if (!threads[id]) throw new Error(`no fixture for ${id}`); + if (!threads[id]) throw new PermanentFetchError(`no fixture for ${id}`); return clone(threads[id]); }, release: async ({ name }) => releaseFacts('npm', npm[name]), diff --git a/tests/kit/upstream-watch-ledger-branch.test.mjs b/tests/kit/upstream-watch-ledger-branch.test.mjs index e7447970..231f2eaa 100644 --- a/tests/kit/upstream-watch-ledger-branch.test.mjs +++ b/tests/kit/upstream-watch-ledger-branch.test.mjs @@ -61,6 +61,33 @@ test('an absent ledger branch reads as an empty ledger without a fetch', async ( assert.deepEqual(calls.map((call) => call.args), [['ls-remote', '--exit-code', '--heads', 'origin', 'upstream-watch-ledger']]); }); +test('runWithInput reports a spawn failure without rejecting', async () => { + const result = await runWithInput('ak-missing-command-for-test', []); + assert.equal(result.status, null); + assert.equal(result.error?.code, 'ENOENT'); +}); + +test('read reports failures from rev-parse, show and log', async () => { + for (const failedCommand of ['rev-parse', 'show', 'log']) { + const sha = 'a'.repeat(40); + const responses = [ + { stdout: `${sha}\trefs/heads/upstream-watch-ledger\n` }, {}, + { stdout: `${sha}\n` }, { stdout: '' }, { stdout: '' }, + ]; + const index = { 'rev-parse': 2, show: 3, log: 4 }[failedCommand]; + responses[index] = { status: 128, stderr: `${failedCommand} failed` }; + const { exec, calls } = fakeExec(responses); + await assert.rejects(createLedgerStore({ exec }).read('upstream-watch-ledger', { now: NOW }), new RegExp(`git ${failedCommand} failed: ${failedCommand} failed`)); + assert.equal(calls.at(-1).args[0], failedCommand); + } +}); + +test('read rejects an invalid branch before any git call', async () => { + const { exec, calls } = fakeExec([]); + await assert.rejects(createLedgerStore({ exec }).read('../ledger'), /not a branch name/); + assert.equal(calls.length, 0); +}); + test('a failed ls-remote or fetch throws', async () => { const lookup = fakeExec([{ status: 128, stderr: 'fatal: unable to access: HTTP 403\n' }]); await assert.rejects(createLedgerStore({ exec: lookup.exec }).read('upstream-watch-ledger', { now: NOW }), /git ls-remote origin upstream-watch-ledger failed: fatal: unable to access/); diff --git a/tests/kit/upstream-watch-notice.test.mjs b/tests/kit/upstream-watch-notice.test.mjs index 7735d3ef..78c17bcf 100644 --- a/tests/kit/upstream-watch-notice.test.mjs +++ b/tests/kit/upstream-watch-notice.test.mjs @@ -33,6 +33,18 @@ test('a quiet run has no notice; an action run mentions the maintainer first', ( assert.match(body, /\nThe full record: `node scripts\/upstream-watch\.mjs ledger --recorded-since 2026-10-02T14:17:00Z`\n$/); }); +test('one action uses singular wording and the latest fired session for an id', () => { + const records = [ + rec('released', { version: '1.0.0', branch: 'upstream/ruvnet-ruflo-1' }), + rec('fired', { branch: 'upstream/ruvnet-ruflo-1', session: 'https://claude.ai/code/session_old' }), + rec('fired', { branch: 'upstream/ruvnet-ruflo-1', session: 'https://claude.ai/code/session_new' }), + ]; + const body = renderNotice({ records, mention: 'pacphi', date: '2026-10-02', recordedAt: '2026-10-02T14:17:00Z' }); + assert.match(body, /^@pacphi upstream watch: 1 item needs you/); + assert.match(body, /Routine session: https:\/\/claude\.ai\/code\/session_new/); + assert.doesNotMatch(body, /session_old/); +}); + test('thread ids never autolink; only a dispatch pull request number does', () => { const body = renderNotice({ records: [rec('reply', { by: 'x', at: '10:00:00Z' }, 'a/b#5'), rec('dispatch-pr', { branch: 'upstream/a-b-5', pr: 261 }, 'a/b#5')], mention: 'pacphi', date: '2026-10-02', recordedAt: '2026-10-02T14:17:00Z' }); const prose = body.replace(/`[^`]*`/g, ''); @@ -46,6 +58,17 @@ test('a notice too long for GitHub lists what fits and says how many more', () = assert.ok(body.length <= NOTICE_MAX, String(body.length)); assert.match(body, /\n\d+ more; see the ledger\.\n/); assert.ok(body.includes('`owner/repo#1`') && !body.includes('`owner/repo#3000`')); + const items = records.filter(isActionRecord); + const bullets = items.map((item) => `- ${sentence(item)}`); + const head = `@pacphi upstream watch: ${items.length} items need you (2026-10-02).`; + const foot = 'The full record: `node scripts/upstream-watch.mjs ledger --recorded-since 2026-10-02T14:17:00Z`'; + let expected; + for (let count = bullets.length; count > 0; count--) { + const more = bullets.length - count; + const candidate = `${[head, '', ...bullets.slice(0, count), ...(more ? ['', `${more} more; see the ledger.`] : []), '', foot].join('\n')}\n`; + if (candidate.length <= NOTICE_MAX) { expected = candidate; break; } + } + assert.equal(body, expected, 'the optimized truncation keeps the exact previous body'); }); // A commit message is plain text: GitHub turns `owner/repo#n` or `#n` there into diff --git a/tests/kit/upstream-watch-query.test.mjs b/tests/kit/upstream-watch-query.test.mjs index c13b3506..b064d4b6 100644 --- a/tests/kit/upstream-watch-query.test.mjs +++ b/tests/kit/upstream-watch-query.test.mjs @@ -60,6 +60,17 @@ test('--recorded-since belongs to ledger only', async () => { }); }); +test('ledger fails visibly on an invalid registry without reading the ledger', async () => { + await withRegistryFile([entry('ruvnet/ruflo#3153', { status: 'done' })], async (file) => { + let reads = 0; + const result = await query(file, [], { read: async () => { reads++; throw new Error('unexpected read'); } }); + assert.equal(result.code, 3); + assert.equal(reads, 0); + assert.match(result.err, /upstream registry is invalid/); + assert.doesNotMatch(result.out, /No report/); + }); +}); + test('the fetcher reads the last successful scheduled watch run from the Actions API', async () => { const calls = []; const exec = async (command, args) => { diff --git a/tests/kit/upstream-watch-record.test.mjs b/tests/kit/upstream-watch-record.test.mjs index e854b218..30c1ec27 100644 --- a/tests/kit/upstream-watch-record.test.mjs +++ b/tests/kit/upstream-watch-record.test.mjs @@ -3,7 +3,7 @@ import test from 'node:test'; import assert from 'node:assert/strict'; -import { retrying } from '../../scripts/upstream-watch/fetch.mjs'; +import { PermanentFetchError, retrying } from '../../scripts/upstream-watch/fetch.mjs'; import { isActionRecord } from '../../scripts/upstream-watch/ledger.mjs'; import { main } from '../../scripts/upstream-watch.mjs'; import { @@ -17,15 +17,22 @@ async function record(file, argv, { fetcher = fixtureFetcher(), ledgerStore = me return { code, result: out.text() ? JSON.parse(out.text()) : null, err: err.text(), ledgerStore, dispatcher }; } -test('retrying retries every method but auth, then gives up', async () => { +test('retrying handles transient failures within the budget, but never retries auth or deterministic failures', async () => { let calls = 0; const waits = []; const fetcher = retrying({ auth: async () => { throw new Error('auth is not retried'); }, thread: async () => { calls += 1; if (calls < 3) throw new Error('HTTP 502'); return 'ok'; } }, { sleep: async (ms) => { waits.push(ms); } }); assert.equal(await fetcher.thread('a/b#1'), 'ok'); assert.deepEqual(waits, [2000, 10000]); await assert.rejects(fetcher.auth(), /auth is not retried/); - const always = retrying({ thread: async () => { throw new Error('HTTP 404'); } }, { sleep: noSleep }); - await assert.rejects(always.thread('a/b#1'), /HTTP 404/); + let exhausted = 0; + const always = retrying({ thread: async () => { exhausted++; throw new Error('HTTP 502'); } }, { delays: [1, 2], sleep: async (ms) => { waits.push(ms); } }); + await assert.rejects(always.thread('a/b#1'), /HTTP 502/); + assert.equal(exhausted, 3); + assert.deepEqual(waits, [2000, 10000, 1, 2]); + let deterministic = 0; + const invalid = retrying({ thread: async () => { deterministic++; throw new PermanentFetchError('no fixture'); } }, { sleep: async () => { assert.fail('deterministic failure slept'); } }); + await assert.rejects(invalid.thread('a/b#1'), /no fixture/); + assert.equal(deterministic, 1); }); test('record on an absent ledger starts from --since, commits every record and prints the notice', async () => { @@ -160,6 +167,7 @@ test('record is blind (exit 3) when gh, the ledger, the registry or every upstre assert.equal(invalid.code, 3); assert.deepEqual([invalid.result.blind, invalid.result.records, invalid.result.commit], [true, [], null]); assert.match(invalid.result.error, /registry/); + assert.equal(invalid.err.match(/upstream registry is/g)?.length, 1); }); }); @@ -182,3 +190,65 @@ test('future --since is a usage error', async () => { assert.match(err, /--since is in the future/); }); }); + +test('invalid registry takes precedence over a future --since', async () => { + await withRegistryFile([entry('ruvnet/ruflo#3153', { status: 'done' })], async (file) => { + const { code, result, err } = await record(file, ['--since', '2026-10-01T00:00:00Z']); + assert.equal(code, 3); + assert.equal(result.blind, true); + assert.match(err, /upstream registry is invalid/); + assert.doesNotMatch(err, /--since is in the future/); + }); +}); + +// Reuse the recorded release fact for distinct synthetic issue ids; all I/O is injected. +const backlog = () => [1, 2, 3, 4, 5].map((n) => entry(`proffesor-for-testing/agentic-qe#${n}`, { doneWhen: { state: 'closed-completed', release: { channel: 'npm', name: 'agentic-qe', minVersion: '3.13.10' } } })); +const backlogFetcher = () => { + const fetcher = fixtureFetcher(); + return { ...fetcher, thread: async () => fetcher.thread('proffesor-for-testing/agentic-qe#617') }; +}; + +test('advanced Checked-At preserves deferred releases for a later run', async () => { + await withRegistryFile(backlog(), async (file) => { + const first = await record(file, [], { fetcher: backlogFetcher() }); + assert.equal(first.code, 0); + assert.equal(first.dispatcher.fired.length, 3); + assert.deepEqual(first.result.deferred.map((item) => item.id), ['proffesor-for-testing/agentic-qe#4', 'proffesor-for-testing/agentic-qe#5']); + const second = await record(file, [], { fetcher: backlogFetcher(), ledgerStore: memoryLedger({ commit: first.result.commit, records: first.result.records, checkedAt: first.result.checkedAt }), now: new Date('2026-09-27T23:00:00Z') }); + assert.equal(second.result.since, '2026-09-26T23:00:00Z'); + assert.equal(second.dispatcher.fired.length, 2); + assert.deepEqual(second.result.records.map((item) => [item.id, item.event]), [['proffesor-for-testing/agentic-qe#4', 'fired'], ['proffesor-for-testing/agentic-qe#5', 'fired']]); + assert.deepEqual(second.result.deferred, []); + }); +}); + +test('ledger build failure retains successful firings, dispatch errors and the deferred backlog', async () => { + await withRegistryFile(backlog(), async (file) => { + const dispatcher = fakeDispatcher(); + const fire = dispatcher.fire; + dispatcher.fire = async (text) => { if (dispatcher.fired.length === 1) throw new Error('HTTP 503'); return fire(text); }; + const ledgerStore = { read: async () => ({ commit: null, records: [], checkedAt: null }), build: async () => { throw new Error('disk full'); } }; + const { code, result } = await record(file, [], { fetcher: backlogFetcher(), dispatcher, ledgerStore }); + assert.equal(code, 3); + assert.equal(result.blind, true); + assert.equal(result.fired.length, 1); + assert.equal(result.fired[0].fields.session, 'https://claude.ai/code/session_new'); + assert.deepEqual(result.dispatchErrors, [{ id: 'proffesor-for-testing/agentic-qe#2', error: 'HTTP 503' }]); + assert.deepEqual(result.deferred.map((item) => item.id), ['proffesor-for-testing/agentic-qe#3', 'proffesor-for-testing/agentic-qe#4', 'proffesor-for-testing/agentic-qe#5']); + }); +}); + +test('record dry run reports its bounded preview and deferred backlog in JSON and console', async () => { + await withRegistryFile(backlog(), async (file) => { + const preview = await record(file, ['--dry-run'], { fetcher: backlogFetcher() }); + assert.equal(preview.result.wouldFire.length, 3); + assert.equal(preview.result.deferred.length, 2); + assert.deepEqual(preview.dispatcher.fired, []); + assert.deepEqual(preview.ledgerStore.built, []); + const out = capture(); + const code = await main(['record', '--dry-run', '--registry', file], { fetcher: backlogFetcher(), ledgerStore: memoryLedger(), dispatcher: fakeDispatcher(), sleep: async () => { assert.fail('preview slept'); }, stdout: out.stream, stderr: capture().stream, now: NOW }); + assert.equal(code, 0); + assert.equal(out.text().match(/^Would fire /gm)?.length, 3); + assert.equal(out.text().match(/^Deferred /gm)?.length, 2); + }); +}); diff --git a/tests/kit/upstream-watch-registry.test.mjs b/tests/kit/upstream-watch-registry.test.mjs index 7b1d43b9..18eca394 100644 --- a/tests/kit/upstream-watch-registry.test.mjs +++ b/tests/kit/upstream-watch-registry.test.mjs @@ -193,11 +193,10 @@ test('the tracking issues carry their whole upstream remainder', () => { const t240 = entry(doc, 'pacphi/agentic-kit#240'); assert.deepEqual([...t240.tracks].sort(), ['proffesor-for-testing/agentic-qe#574', 'proffesor-for-testing/agentic-qe#719']); assert.match(t240.adjustment, /agentic-qe#574/); - assert.ok(t240.kitImpact.files.includes('src/lib/aqe-readiness.mjs')); + assert.ok(t240.kitImpact.files.includes('docs/troubleshooting.md')); }); -// agentic-qe#719 is a partial fix for #574: releasing it alone must not dispatch removing the busy rule. -test('the partial fix agentic-qe#719 is context only; agentic-qe#574 drives the dispatch', () => { +test('the #574 exception retirement records released conformance without closing #240', () => { const doc = document(); const partial = entry(doc, 'proffesor-for-testing/agentic-qe#719'); assert.equal(partial.mapping, 'unmapped'); @@ -206,7 +205,13 @@ test('the partial fix agentic-qe#719 is context only; agentic-qe#574 drives the assert.match(partial.note, /agentic-qe#574/); const driver = entry(doc, 'proffesor-for-testing/agentic-qe#574'); assert.equal(driver.mapping, 'mapped'); - assert.match(driver.adjustment, /busy rule/); + assert.equal(driver.status, 'retired'); + assert.match(driver.adjustment, /macOS and Linux/); + assert.match(driver.adjustment, /not a universal AQE minimum/); + assert.ok(driver.history.some((item) => item.event === 'retired' && item.date === '2026-09-29')); + const tracker = entry(doc, 'pacphi/agentic-kit#240'); + assert.equal(tracker.status, 'watching'); + assert.match(tracker.adjustment, /final main PR/); }); test('stale threads are mapped to what ak carries, or retired with a reason', () => { diff --git a/tests/kit/usage-claude-cost-state.test.mjs b/tests/kit/usage-claude-cost-state.test.mjs new file mode 100644 index 00000000..66ca4160 --- /dev/null +++ b/tests/kit/usage-claude-cost-state.test.mjs @@ -0,0 +1,174 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { parseClaude } from '../../src/lib/usage-parsers.mjs'; +import { sessionCostEvidence, reconcileClaudeCostState } from '../../src/lib/usage-cost.mjs'; +import { costOf } from '../../src/lib/pricing.mjs'; +import { aggregate, sessionPayload } from '../../src/lib/usage-aggregate.mjs'; +import { buildIndex, SCHEMA_VERSION, _resetForTest } from '../../src/lib/usage-index.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; +import fs from 'node:fs'; +import path from 'node:path'; + +const line = (value) => JSON.stringify(value); +const usage = { input_tokens: 10, output_tokens: 20, cache_read_input_tokens: 30, cache_creation_input_tokens: 40 }; +const assistant = line({ type: 'assistant', timestamp: '2026-09-28T12:00:00Z', message: { + id: 'm1', role: 'assistant', model: 'claude-opus-5', usage, content: [{ type: 'text', text: 'ok' }], +} }); +const snapshot = (totalCostUSD, overrides = {}) => line({ + type: 'cost-state', sessionId: 's1', totalCostUSD, startTime: Date.parse('2026-09-28T11:00:00Z'), + modelUsage: { 'claude-opus-5': { + inputTokens: 10, outputTokens: 20, cacheReadInputTokens: 30, cacheCreationInputTokens: 40, + thinkingTokens: 0, webSearchRequests: 0, costUSD: totalCostUSD, + } }, hasUnknownModelCost: false, ...overrides, +}); +const parse = (...records) => parseClaude([assistant, ...records].join('\n'), { id: 's1' }).session; +const deps = { costOf, classify: () => ({ category: 'Unclassified', confidence: 0, basis: 'test' }), detectInsights: () => [] }; + +test('latest valid cumulative snapshot remains separate from message cost', () => { + const rec = parse(snapshot(1), snapshot(2)); + assert.equal(rec.claudeCostState.reportedUsd, 2); + assert.equal(rec.claudeCostState.currency, 'USD'); + assert.equal(rec.claudeCostState.provenance, 'claude-code-cost-state'); + assert.equal(rec.claudeCostState.validSnapshots, 2); + assert.equal(rec.claudeCostState.coverage, 'session-cumulative'); + const evidence = sessionCostEvidence(rec, deps); + assert.equal(evidence.estimatedUsd, costOf({ model: 'claude-opus-5', day: '2026-09-28', input: 10, output: 20, cacheRead: 30, cacheWrite: 40 })); + const reconciled = reconcileClaudeCostState(rec); + assert.equal(reconciled.status, 'scope-unknown'); + assert.equal(reconciled.reportedUsd, 2); + assert.equal(reconciled.estimatedUsd, null); + assert.equal(reconciled.checkpointStartMs, Date.parse('2026-09-28T11:00:00Z')); + assert.equal(reconciled.checkpointEndMs, null); + assert.equal(reconciled.messageFirstAtMs, Date.parse('2026-09-28T12:00:00Z')); + assert.equal(reconciled.messageLastAtMs, Date.parse('2026-09-28T12:00:00Z')); +}); + +test('malformed and unsupported later checkpoints cannot replace latest valid', () => { + const rec = parse(snapshot(1), snapshot(-1), snapshot('2'), + line({ type: 'cost-state', sessionId: 's1', futureCost: 3 }), snapshot(4, { sessionId: 'other' })); + assert.equal(rec.claudeCostState.reportedUsd, 1); + assert.equal(rec.claudeCostState.validSnapshots, 1); + assert.equal(rec.claudeCostState.malformedSnapshots, 2); + assert.equal(rec.claudeCostState.unsupportedSnapshots, 2); +}); + +test('scope mismatch or unknown model prevents a cost equality claim', () => { + const mismatch = parse(snapshot(1, { modelUsage: { 'claude-opus-5': { + inputTokens: 11, outputTokens: 20, cacheReadInputTokens: 30, cacheCreationInputTokens: 40, + webSearchRequests: 0, costUSD: 1, + } } })); + assert.deepEqual(reconcileClaudeCostState(mismatch).scopeReasons, ['model-token-totals-differ']); + const unknown = parse(snapshot(1, { hasUnknownModelCost: true })); + assert.equal(reconcileClaudeCostState(unknown).status, 'scope-unknown'); + const routed = parse(snapshot(1)); + routed.sessionOrigin.thirdPartyProvider = 'amazon-bedrock'; + assert.equal(reconcileClaudeCostState(routed).status, 'scope-unknown'); +}); + +test('matching amount and exact model/token totals do not attest time or provider scope', () => { + const amount = costOf({ model: 'claude-opus-5', day: '2026-09-28', input: 10, output: 20, cacheRead: 30, cacheWrite: 40 }); + const rec = parse(snapshot(amount)); + assert.equal(rec.providerProvenance, 'unknown'); + assert.ok(reconcileClaudeCostState(rec).scopeReasons.includes('serving-provider-unverified')); + rec.inferenceProvider = 'amazon-bedrock'; + rec.providerProvenance = 'observed'; + assert.ok(reconcileClaudeCostState(rec).scopeReasons.includes('serving-provider-different')); +}); + +test('a Bedrock assistant model is provider evidence, not an Anthropic serving attestation', () => { + const model = 'us.anthropic.claude-sonnet-4-6'; + const message = JSON.parse(assistant); + message.message.model = model; + const state = JSON.parse(snapshot(1)); + state.modelUsage = { [model]: state.modelUsage['claude-opus-5'] }; + const rec = parseClaude([line(message), line(state)].join('\n'), { id: 's1' }).session; + assert.equal(rec.sessionOrigin.thirdPartyProvider, 'amazon-bedrock'); + assert.ok(reconcileClaudeCostState(rec).scopeReasons.includes('serving-provider-different')); +}); + +test('missing and malformed start times stay diagnostic without replacing a valid checkpoint', () => { + const rec = parse(snapshot(1), snapshot(2, { startTime: undefined }), + snapshot(3, { startTime: 'bad' }), snapshot(4, { startTime: 1_780_000_000 })); + assert.equal(rec.claudeCostState.reportedUsd, 1); + assert.equal(rec.claudeCostState.malformedSnapshots, 3); + assert.equal(reconcileClaudeCostState(rec).status, 'scope-unknown'); +}); + +test('checkpoint starting after a charged message has mismatched time scope', () => { + const rec = parse(snapshot(1, { startTime: Date.parse('2026-09-28T12:01:00Z') })); + assert.equal(rec.claudeMessageCoverage.firstAtMs, Date.parse('2026-09-28T12:00:00Z')); + assert.deepEqual(reconcileClaudeCostState(rec).scopeReasons, ['checkpoint-start-after-message']); +}); + +test('an earlier zero-token assistant does not widen charged-message time coverage', () => { + const zero = JSON.parse(assistant); + zero.timestamp = '2026-09-28T11:00:00Z'; + zero.message.id = 'zero'; + zero.message.usage = { input_tokens: 0, output_tokens: 0 }; + const rec = parseClaude([line(zero), assistant, + snapshot(1, { startTime: Date.parse('2026-09-28T11:30:00Z') })].join('\n'), { id: 's1' }).session; + assert.equal(rec.claudeMessageCoverage.firstAtMs, Date.parse('2026-09-28T12:00:00Z')); + assert.equal(reconcileClaudeCostState(rec).status, 'scope-unknown'); +}); + +test('overlong model keys remain diagnostic and never enter the cache', () => { + const allowedModel = 'a'.repeat(100); + const longModel = 'a'.repeat(101); + const counts = { + inputTokens: 10, outputTokens: 20, cacheReadInputTokens: 30, cacheCreationInputTokens: 40, + webSearchRequests: 0, costUSD: 2, + }; + const rec = parse(snapshot(1, { modelUsage: { [allowedModel]: counts } }), + snapshot(2, { modelUsage: { [longModel]: counts } })); + assert.equal(rec.claudeCostState.reportedUsd, 1); + assert.equal(rec.claudeCostState.malformedSnapshots, 1); + assert.equal(Object.keys(rec.claudeCostState.modelUsage)[0], allowedModel); + assert.equal(JSON.stringify(rec).includes(longModel), false); +}); + +test('aggregate and session detail expose reconciliation without adding the snapshot to totals', () => { + const rec = parse(snapshot(2)); + const now = Date.parse('2026-09-29T00:00:00Z'); + const agg = aggregate([rec], { days: 7, now, cutoff: now - 7 * 86400000, deps }); + const messageCost = sessionCostEvidence(rec, deps).estimatedUsd; + assert.equal(agg.sessions[0].cost, messageCost); + assert.equal(agg.totals.cost, messageCost); + assert.equal(agg.sessions[0].claudeCostState.status, 'scope-unknown'); + assert.equal(agg.sessions[0].claudeCostState.reportedUsd, 2); + assert.equal(sessionPayload(rec, [], deps).meta.claudeCostState.status, 'scope-unknown'); +}); + +test('schema 26 cold and warm cache reads retain the checkpoint diagnostic', async () => { + assert.equal(SCHEMA_VERSION, 26); + const root = tempDir('ak-claude-cost-state'); + const project = path.join(root, 'claude', 'project'); + fs.mkdirSync(project, { recursive: true }); + const overlongModel = 'a'.repeat(101); + const invalid = JSON.parse(snapshot(3)); + invalid.modelUsage = { [overlongModel]: invalid.modelUsage['claude-opus-5'] }; + fs.writeFileSync(path.join(project, 's1.jsonl'), [assistant, snapshot(2), line(invalid)].join('\n')); + const options = { + days: 7, now: Date.parse('2026-09-29T00:00:00Z'), + roots: { claude: path.join(root, 'claude'), codex: path.join(root, 'codex') }, + cachePath: path.join(root, 'cache', 'usage-index.json'), deps, + }; + _resetForTest(); + const cold = await buildIndex(options); + assert.equal(cold.sessions[0].claudeCostState.malformedSnapshots, 1); + assert.equal(fs.readFileSync(options.cachePath, 'utf8').includes(overlongModel), false); + const cache = JSON.parse(fs.readFileSync(options.cachePath, 'utf8')); + for (const entry of Object.values(cache.entries)) { + delete entry.session.claudeCostState.startMs; + delete entry.session.claudeCostState.endMs; + delete entry.session.claudeMessageCoverage; + } + fs.writeFileSync(options.cachePath, JSON.stringify(cache)); + _resetForTest(); + const reparsed = await buildIndex(options); + assert.equal(reparsed.sessions[0].claudeCostState.status, 'scope-unknown'); + _resetForTest(); + const warm = await buildIndex(options); + assert.deepEqual(warm.sessions[0].claudeCostState, cold.sessions[0].claudeCostState); + assert.equal(warm.sessions[0].claudeCostState.status, 'scope-unknown'); + assert.equal(warm.totals.cost, cold.totals.cost); +}); diff --git a/tests/kit/usage-claude-dedup.test.mjs b/tests/kit/usage-claude-dedup.test.mjs index 36a861d9..240fec17 100644 --- a/tests/kit/usage-claude-dedup.test.mjs +++ b/tests/kit/usage-claude-dedup.test.mjs @@ -9,6 +9,7 @@ import assert from 'node:assert/strict'; import fs from 'node:fs'; import path from 'node:path'; import { parseClaude } from '../../src/lib/usage-parsers.mjs'; +import { reconcileClaudeMessages } from '../../src/lib/usage-claude-dedup.mjs'; import { decodeClaudeRecord } from '../../src/lib/telemetry-records.mjs'; import { buildIndex, SCHEMA_VERSION, _resetForTest } from '../../src/lib/usage-index.mjs'; import { costOf, priceFor } from '../../src/lib/pricing.mjs'; @@ -16,6 +17,10 @@ import { tempDir } from './helpers/temp-dir.mjs'; const T0 = Date.parse('2026-08-20T10:00:00.000Z'); const at = (s) => new Date(T0 + s * 1000).toISOString(); +const dayAt = (s) => { + const d = new Date(at(s)); + return `${d.getFullYear()}-${String(d.getMonth() + 1).padStart(2, '0')}-${String(d.getDate()).padStart(2, '0')}`; +}; const line = (o) => JSON.stringify(o); const USAGE = { input_tokens: 10, output_tokens: 200, cache_read_input_tokens: 5000, cache_creation_input_tokens: 700 }; @@ -185,6 +190,355 @@ test('turn rows (reader path) are still emitted per transcript line', () => { assert.equal(turns.filter((t) => t.role === 'assistant').length, 2); }); +test('cross-file copies charge one richest message while source sessions keep their response counts', () => { + const a = parse([prompt(), asst({ id: 'msg_shared', s: 5, block: text(), usage: { ...USAGE, output_tokens: 12 } })]); + const b = parse([prompt(), asst({ id: 'msg_shared', s: 6, block: text(), usage: { ...USAGE, output_tokens: 200 } })]); + a.id = 'a'; b.id = 'b'; + const [aa, bb] = reconcileClaudeMessages([a, b]); + assert.equal(a.responses + b.responses, 2, 'cached transcript observations are untouched'); + assert.equal(aa.responses + bb.responses, 2, 'session counts still describe each file'); + assert.equal(aa.accountedResponses + bb.accountedResponses, 1); + assert.equal(aa.usage[0].output + bb.usage[0].output, 200); + assert.equal(aa.usage[0].cacheRead + bb.usage[0].cacheRead, 5000); + assert.equal(aa.usage[0].responses + bb.usage[0].responses, 1); + assert.deepEqual(reconcileClaudeMessages([b, a]).map((r) => [r.id, r.usage[0].output]), + [[bb.id, bb.usage[0].output], [aa.id, aa.usage[0].output]], 'file traversal cannot elect a different charge'); +}); + +test('missing IDs and distinct same-count IDs never collapse across files', () => { + const a = parse([prompt(), asst({ s: 5, block: text() }), asst({ id: 'msg_one', s: 6, block: text() })]); + const b = parse([prompt(), asst({ s: 5, block: text() }), asst({ id: 'msg_two', s: 6, block: text() })]); + const rows = reconcileClaudeMessages([a, b]); + assert.equal(rows.reduce((n, r) => n + r.accountedResponses, 0), 4); + assert.equal(rows.reduce((n, r) => n + r.usage[0].output, 0), 800); +}); + +test('malformed repeated identifiers are not cross-file identity proof', () => { + const a = parse([asst({ id: 'not-an-api-message-id', s: 5, block: text() })]); + const b = parse([asst({ id: 'not-an-api-message-id', s: 5, block: text() })]); + const rows = reconcileClaudeMessages([a, b]); + assert.equal(rows.reduce((n, r) => n + r.accountedResponses, 0), 2); + assert.equal(rows.reduce((n, r) => n + r.usage[0].output, 0), 400); +}); + +test('message ID, request ID and serving-provider namespaces stay separate', () => { + const plain = parse([asst({ id: 'same', s: 5, block: text() })]); + const request = parse([asst({ requestId: 'same', s: 5, block: text() })]); + const bedrock = parse([asst({ id: 'same', s: 5, block: text(), model: 'us.anthropic.claude-sonnet-4-20250514-v1:0' })]); + const rows = reconcileClaudeMessages([plain, request, bedrock]); + assert.equal(rows.reduce((n, r) => n + r.accountedResponses, 0), 3); + assert.equal(rows.reduce((n, r) => n + r.usage[0].output, 0), 600); +}); + +test('partial copied snapshots reconcile component maxima once', () => { + const a = parse([asst({ id: 'msg_partial', s: 5, block: text(), usage: { ...USAGE, cache_read_input_tokens: 0 } })]); + const b = parse([asst({ id: 'msg_partial', s: 6, block: text(), usage: { ...USAGE, output_tokens: 20 } })]); + const rows = reconcileClaudeMessages([a, b]); + assert.equal(rows.reduce((n, r) => n + r.usage[0].output, 0), 200); + assert.equal(rows.reduce((n, r) => n + r.usage[0].cacheRead, 0), 5000); + assert.equal(rows.reduce((n, r) => n + r.accountedResponses, 0), 1); +}); + +test('copies entirely outside the fixed pool still have one historical charge', () => { + const a = parse([asst({ id: 'msg_old', s: 5, block: text(), usage: { ...USAGE, output_tokens: 12 } })]); + const b = parse([asst({ id: 'msg_old', s: 6, block: text(), usage: USAGE })]); + a.claudeIdentityEligible = false; b.claudeIdentityEligible = false; + const rows = reconcileClaudeMessages([a, b]); + assert.equal(rows.reduce((n, r) => n + r.accountedResponses, 0), 1); + assert.equal(rows.reduce((n, r) => n + r.usage[0].output, 0), 200); +}); + +test('equal claims have a stable accounting owner when sidechain and source differ', () => { + const main = parse([asst({ id: 'msg_tie', s: 5, block: text() })]); + const side = parse([asst({ id: 'msg_tie', s: 5, block: text() })]); + main.sidechain = false; side.sidechain = true; + main.claudeSourceKey = 'a'; side.claudeSourceKey = 'b'; + const forward = reconcileClaudeMessages([main, side]); + const reverse = reconcileClaudeMessages([side, main]); + assert.equal(forward.find((r) => r.claudeSourceKey === 'a').accountedResponses, 1); + assert.equal(reverse.find((r) => r.claudeSourceKey === 'a').accountedResponses, 1); + assert.equal(forward.find((r) => r.claudeSourceKey === 'b').accountedResponses, 0); + assert.equal(reverse.find((r) => r.claudeSourceKey === 'b').accountedResponses, 0); + const x = parse([asst({ id: 'msg_same_metadata', s: 5, block: text() })]); + const y = parse([asst({ id: 'msg_same_metadata', s: 5, block: text() })]); + x.claudeSourceKey = 'x'; y.claudeSourceKey = 'y'; + assert.equal(reconcileClaudeMessages([x, y]).find((r) => r.claudeSourceKey === 'x').accountedResponses, 1); + assert.equal(reconcileClaudeMessages([y, x]).find((r) => r.claudeSourceKey === 'x').accountedResponses, 1, + 'source identity settles a tie even when every visible session field matches'); +}); + +test('one identity is charged once across current, previous and combined windows regardless of lookback', async () => { + _resetForTest(); + const dir = tempDir('ak-cross-window'); + const root = path.join(dir, 'claude'); + const proj = path.join(root, '-Users-me-proj'); + fs.mkdirSync(proj, { recursive: true }); + const oldFile = path.join(proj, 'old.jsonl'); + const newFile = path.join(proj, 'new.jsonl'); + const write = (file, s, output) => { + fs.writeFileSync(file, [prompt(s), asst({ id: 'msg_shared', s: s + 5, block: text(), + usage: { ...USAGE, output_tokens: output } })].join('\n') + '\n'); + fs.utimesSync(file, new Date(at(s)), new Date(at(s + 5))); + }; + write(oldFile, 0, 200); + write(newFile, 86_400, 100); + const o = { days: 1, now: T0 + 2 * 86_400_000, + roots: { claude: root, codex: path.join(dir, 'codex') }, + cachePath: path.join(dir, 'usage-index.json'), codexState: null, + deps: { costOf: ({ output }) => output / 100, pricesAsOf: 'fixture', + classify: () => ({ category: 'Build', confidence: 1, basis: 'fixture' }), detectInsights: () => [] } }; + const plain = await buildIndex(o); + assert.equal(plain.totals.output, 0, 'the richer older copy is the single accounting owner'); + assert.equal(plain.totals.responses, 0); + assert.equal(plain.totals.cost, 0); + assert.deepEqual(plain.sourceHealth.claude.identityCoverage, + { horizonDays: 2, horizonCoversComparison: true, horizonCoversRequestedHistory: true, + outOfPoolRecords: 0, unknownEligibilityRecords: 0, outsideCurrentExcluded: 0, + basis: 'file-mtime-and-session-end' }); + _resetForTest(); + const widenedCurrent = await buildIndex({ ...o, lookbackDays: 2 }); + assert.equal(widenedCurrent.totals.output, plain.totals.output); + assert.equal(widenedCurrent.totals.responses, plain.totals.responses); + _resetForTest(); + const widened = await buildIndex({ ...o, lookbackDays: 2, previous: true }); + assert.equal(widened.totals.output, plain.totals.output, 'comparison cannot change the displayed charge'); + assert.equal(widened.totals.responses, plain.totals.responses); + assert.equal(widened.previous.totals.output, 200); + assert.equal(widened.previous.totals.responses, 1); + assert.equal(widened.previous.totals.cost, 2); + _resetForTest(); + const combined = await buildIndex({ ...o, days: 2 }); + assert.equal(combined.totals.output, 200); + assert.equal(combined.totals.responses, 1); + assert.equal(combined.totals.cost, 2); + assert.equal(widened.totals.output + widened.previous.totals.output, combined.totals.output); + assert.equal(widened.totals.responses + widened.previous.totals.responses, combined.totals.responses); + _resetForTest(); + const warmPlain = await buildIndex(o); + assert.equal(warmPlain.totals.output, plain.totals.output); + const outsideFile = path.join(proj, 'outside.jsonl'); + write(outsideFile, -86_400, 300); + _resetForTest(); + const longLookback = await buildIndex({ ...o, lookbackDays: 5, previous: true }); + assert.equal(longLookback.totals.output, plain.totals.output, + 'a third copy outside the fixed accounting horizon cannot change current ownership'); + assert.equal(longLookback.previous.totals.output, 200); + _resetForTest(); + assert.equal((await buildIndex(o)).totals.output, plain.totals.output, + 'cached older history cannot change the plain query'); + _resetForTest(); + const capped = await buildIndex({ ...o, days: 400 }); + assert.deepEqual(capped.sourceHealth.claude.identityCoverage, + { horizonDays: 730, horizonCoversComparison: false, horizonCoversRequestedHistory: true, + outOfPoolRecords: 0, unknownEligibilityRecords: 0, outsideCurrentExcluded: 0, + basis: 'file-mtime-and-session-end' }, + 'a caller wider than the supported dashboard window sees the identity cap'); +}); + +test('equal copied usage still yields one charge across adjacent windows', async () => { + _resetForTest(); + const dir = tempDir('ak-equal-cross-window'); + const root = path.join(dir, 'claude'); + const proj = path.join(root, '-Users-me-proj'); + fs.mkdirSync(proj, { recursive: true }); + for (const [name, seconds] of [['old', 0], ['new', 86_400]]) { + const file = path.join(proj, `${name}.jsonl`); + fs.writeFileSync(file, asst({ id: 'msg_equal', s: seconds + 5, block: text() }) + '\n'); + fs.utimesSync(file, new Date(at(seconds)), new Date(at(seconds + 5))); + } + const o = { days: 1, now: T0 + 2 * 86_400_000, + roots: { claude: root, codex: path.join(dir, 'codex') }, + cachePath: path.join(dir, 'usage-index.json'), codexState: null, + deps: { costOf: ({ output }) => output / 100, pricesAsOf: 'fixture', + classify: () => ({ category: 'Build', confidence: 1, basis: 'fixture' }), detectInsights: () => [] } }; + const split = await buildIndex({ ...o, previous: true }); + assert.equal(split.previous.totals.sessions, 1, 'comparison is acquired within the common identity horizon'); + _resetForTest(); + const combined = await buildIndex({ ...o, days: 2 }); + assert.equal(split.totals.output + split.previous.totals.output, 200); + assert.equal(split.totals.responses + split.previous.totals.responses, 1); + assert.equal(combined.totals.output, 200); + assert.equal(combined.totals.responses, 1); + assert.equal(split.totals.cost + split.previous.totals.cost, combined.totals.cost); +}); + +test('deeper lookback cannot promote a copy whose file mtime is outside the fixed identity pool', async () => { + _resetForTest(); + const dir = tempDir('ak-identity-mtime'); + const root = path.join(dir, 'claude'); + const proj = path.join(root, '-Users-me-proj'); + fs.mkdirSync(proj, { recursive: true }); + const olderMtime = path.join(proj, 'older-mtime.jsonl'); + const newerMtime = path.join(proj, 'newer-mtime.jsonl'); + const write = (file, seconds, output, mtimeSeconds, id = 'msg_same') => { + fs.writeFileSync(file, asst({ id, s: seconds, block: text(), + usage: { ...USAGE, output_tokens: output } }) + '\n'); + fs.utimesSync(file, new Date(at(mtimeSeconds)), new Date(at(mtimeSeconds))); + }; + write(olderMtime, 5, 200, -86_400); // transcript Aug 20, file mtime Aug 19 + write(newerMtime, 86_405, 100, 86_405); // transcript and mtime Aug 21 + const o = { days: 1, now: T0 + 2 * 86_400_000, + roots: { claude: root, codex: path.join(dir, 'codex') }, + cachePath: path.join(dir, 'usage-index.json'), codexState: null, + deps: { costOf: ({ output }) => output / 100, pricesAsOf: 'fixture', + classify: () => ({ category: 'Build', confidence: 1, basis: 'fixture' }), detectInsights: () => [] } }; + const plain = await buildIndex(o); + assert.equal(plain.totals.output, 100); + _resetForTest(); + const wider = await buildIndex({ ...o, lookbackDays: 5, previous: true }); + assert.equal(wider.totals.output, 100); + assert.equal(wider.totals.cost, 1); + assert.equal(wider.totals.responses, 1); + assert.equal(wider.previous.totals.output, 0, 'the observed duplicate still charges only once'); + assert.equal(wider.sourceHealth.claude.identityCoverage.horizonCoversRequestedHistory, false); + assert.equal(wider.sourceHealth.claude.identityCoverage.outOfPoolRecords, 1); + _resetForTest(); + assert.equal((await buildIndex(o)).totals.output, 100, 'cached older history cannot promote it'); + + const distinctHistory = path.join(proj, 'distinct-history.jsonl'); + write(distinctHistory, 6, 50, -86_400, 'msg_distinct_history'); + _resetForTest(); + const historical = await buildIndex({ ...o, lookbackDays: 5, previous: true }); + assert.equal(historical.totals.output, 100); + assert.equal(historical.previous.totals.output, 50, + 'a distinct outside-pool historical message remains visible when requested'); + assert.equal(historical.sourceHealth.claude.identityCoverage.outOfPoolRecords, 2); + + // Inclusive boundary: a file at the fixed mtime cutoff belongs to the + // identity pool even when its transcript timestamp is earlier. + fs.utimesSync(olderMtime, new Date(at(0)), new Date(at(0))); + _resetForTest(); + const boundary = await buildIndex(o); + assert.equal(boundary.totals.output, 0, 'the richer boundary copy is eligible'); + _resetForTest(); + const boundaryWide = await buildIndex({ ...o, lookbackDays: 5, previous: true }); + assert.equal(boundaryWide.totals.output, 0); + assert.equal(boundaryWide.previous.totals.output, 250); +}); + +test('malformed cached Claude claim is reparsed instead of crashing or trusted', async () => { + _resetForTest(); + const dir = tempDir('ak-malformed-claims'); + const root = path.join(dir, 'claude'); + const proj = path.join(root, '-Users-me-proj'); + fs.mkdirSync(proj, { recursive: true }); + const file = path.join(proj, 'one.jsonl'); + fs.writeFileSync(file, asst({ id: 'msg_one', s: 5, block: text() }) + '\n'); + const o = { days: 14, now: T0 + 86_400_000, + roots: { claude: root, codex: path.join(dir, 'codex') }, + cachePath: path.join(dir, 'usage-index.json'), codexState: null, + deps: { costOf: () => 0, pricesAsOf: 'fixture', + classify: () => ({ category: 'Build', confidence: 1, basis: 'fixture' }), detectInsights: () => [] } }; + assert.equal((await buildIndex(o)).totals.output, 200); + const original = JSON.parse(fs.readFileSync(o.cachePath, 'utf8')); + const claim = original.entries[file].session.claudeMessages[0]; + for (const malformed of [[null], [{ ...claim, usage: { ...claim.usage, output: 201 } }], + [claim, claim]]) { + const cache = structuredClone(original); + cache.entries[file].session.claudeMessages = malformed; + fs.writeFileSync(o.cachePath, JSON.stringify(cache)); + _resetForTest(); + assert.equal((await buildIndex(o)).totals.output, 200); + const repaired = JSON.parse(fs.readFileSync(o.cachePath, 'utf8')); + assert.equal(repaired.entries[file].session.claudeMessages.length, 1); + assert.equal(typeof repaired.entries[file].session.claudeMessages[0].identity, 'string'); + } +}); + +test('invalid cached claims do not mask a source that becomes unreadable before reparse', async () => { + _resetForTest(); + const dir = tempDir('ak-claim-fallback'); + const root = path.join(dir, 'claude'); + const proj = path.join(root, '-Users-me-proj'); + fs.mkdirSync(proj, { recursive: true }); + const file = path.join(proj, 'one.jsonl'); + fs.writeFileSync(file, asst({ id: 'msg_one', s: 5, block: text() }) + '\n'); + const o = { days: 14, now: T0 + 86_400_000, + roots: { claude: root, codex: path.join(dir, 'codex') }, + cachePath: path.join(dir, 'usage-index.json'), codexState: null, + deps: { costOf: () => 0, pricesAsOf: 'fixture', + classify: () => ({ category: 'Build', confidence: 1, basis: 'fixture' }), detectInsights: () => [] } }; + await buildIndex(o); + const cache = JSON.parse(fs.readFileSync(o.cachePath, 'utf8')); + cache.entries[file].session.claudeMessages = [null]; + fs.writeFileSync(o.cachePath, JSON.stringify(cache)); + _resetForTest(); + const result = await buildIndex({ ...o, onProgress: ({ phase, scanned }) => { + if (phase === 'scan' && scanned === 0) { + fs.unlinkSync(file); + fs.mkdirSync(file); // still stats, but is no longer a readable transcript + } + } }); + assert.equal(result.totals.sessions, 0, 'an invalid cache is not a fallback observation'); + assert.equal(result.sourceHealth.claude.status, 'degraded'); + assert.equal(result.sourceHealth.claude.diagnostics.common.unitsSeen, 1); + assert.equal(result.sourceHealth.claude.diagnostics.common.unitsParsed, 0); + assert.equal(Object.keys(JSON.parse(fs.readFileSync(o.cachePath, 'utf8')).entries).length, 0); +}); + +test('index keeps cross-file accounting on cold, warm, add, change and removal scans', async () => { + _resetForTest(); + const dir = tempDir('ak-cross-file'); + const root = path.join(dir, 'claude'); + const proj = path.join(root, '-Users-me-proj'); + const other = path.join(root, '-Users-me-other'); + fs.mkdirSync(proj, { recursive: true }); + fs.mkdirSync(other, { recursive: true }); + const a = path.join(proj, 'a.jsonl'); + const b = path.join(other, 'b.jsonl'); + const c = path.join(proj, 'c.jsonl'); + const write = (file, messages) => fs.writeFileSync(file, + [prompt(), ...messages].join('\n') + '\n'); + write(a, [asst({ id: 'msg_shared', s: 5, block: text(), usage: { ...USAGE, output_tokens: 12 } })]); + write(b, [asst({ id: 'msg_shared', s: 86_406, block: text(), usage: USAGE })]); + const o = { days: 14, now: T0 + 2 * 86_400_000, + roots: { claude: root, codex: path.join(dir, 'codex') }, + cachePath: path.join(dir, 'usage-index.json'), codexState: null, + deps: { costOf: ({ output }) => output / 100, pricesAsOf: 'fixture', + classify: () => ({ category: 'Build', confidence: 1, basis: 'fixture' }), detectInsights: () => [] } }; + const check = (result, count, output, responses, sharedProject) => { + assert.equal(result.totals.sessions, count); + assert.equal(result.totals.responses, responses); + assert.equal(result.totals.output, output); + assert.equal(result.totals.cost, output / 100); + assert.equal(result.totals.tokens, result.totals.input + result.totals.output + + result.totals.cacheRead + result.totals.cacheWrite); + assert.equal(Object.values(result.byDay).reduce((n, row) => n + row.tokens, 0), result.totals.tokens); + assert.ok(result.byDay[dayAt(sharedProject === 'other' ? 86_406 : 5)].tokens > 0, + 'the elected copy owns its local billing day'); + assert.equal(Object.values(result.byModel).reduce((n, row) => n + row.output, 0), output); + assert.equal(Object.values(result.byProvider).reduce((n, row) => n + row.output, 0), output); + assert.equal(Object.values(result.byProject).reduce((n, row) => n + row.output, 0), output); + assert.equal(result.byProject[sharedProject].output, + sharedProject === 'other' ? output - (count === 3 ? 200 : 0) : output); + assert.equal(Object.values(result.bySource).reduce((n, row) => n + row.responses, 0), responses); + assert.equal(Object.values(result.punchcard).reduce((n, value) => n + value, 0), responses); + assert.equal(result.projectTree.reduce((n, row) => n + row.tokens, 0), result.totals.tokens); + assert.equal(result.sessions.reduce((n, row) => n + row.output, 0), output); + assert.equal(result.sessions.reduce((n, row) => n + row.responses, 0), count, + 'each session retains its own observed response count'); + assert.equal(result.sessions.reduce((n, row) => n + row.accountedResponses, 0), responses); + }; + check(await buildIndex(o), 2, 200, 1, 'other'); + const legacy = JSON.parse(fs.readFileSync(o.cachePath, 'utf8')); + for (const entry of Object.values(legacy.entries)) delete entry.session.claudeMessages; + fs.writeFileSync(o.cachePath, JSON.stringify(legacy)); + _resetForTest(); + check(await buildIndex(o), 2, 200, 1, 'other'); // compatible schema-26 cache reparses old Claude entries + _resetForTest(); + check(await buildIndex(o), 2, 200, 1, 'other'); // genuinely warm + write(c, [asst({ id: 'msg_distinct', s: 7, block: text(), usage: USAGE })]); + _resetForTest(); + check(await buildIndex(o), 3, 400, 2, 'other'); + write(b, [asst({ id: 'msg_shared', s: 86_408, block: text(), usage: { ...USAGE, output_tokens: 300 } })]); + fs.utimesSync(b, new Date(T0), new Date(T0 + 20_000)); + _resetForTest(); + check(await buildIndex(o), 3, 500, 2, 'other'); + fs.unlinkSync(b); + _resetForTest(); + check(await buildIndex(o), 2, 212, 2, 'proj'); +}); + // ── schema version ────────────────────────────────────────────────────────── test('SCHEMA_VERSION is at least 22 and a v21 cache is discarded and re-parsed de-duplicated', async () => { diff --git a/tests/kit/usage-claude-provider.test.mjs b/tests/kit/usage-claude-provider.test.mjs new file mode 100644 index 00000000..f0870160 --- /dev/null +++ b/tests/kit/usage-claude-provider.test.mjs @@ -0,0 +1,96 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { parseClaude, parseCodex } from '../../src/lib/usage-parsers.mjs'; +import { buildIndex, _resetForTest } from '../../src/lib/usage-index.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { Rollout } from './helpers/codex-rollout.mjs'; + +const line = (value) => JSON.stringify(value); +const assistant = (model, extra = {}) => line({ type: 'assistant', timestamp: '2026-09-28T10:00:00Z', + entrypoint: 'claude-desktop-3p', ...extra, + message: { id: `m-${model}`, role: 'assistant', model, + usage: { input_tokens: 3, output_tokens: 2 }, content: [{ type: 'text', text: 'ok' }] } }); +const parse = (...lines) => parseClaude(lines.join('\n'), { id: 'provider-fixture' }).session; + +test('session-bound Bedrock and Vertex model IDs establish a separate provider detail', () => { + for (const [model, provider] of [ + ['us.anthropic.claude-sonnet-4-5-20250929-v1:0', 'amazon-bedrock'], + ['anthropic.claude-haiku-4-5', 'amazon-bedrock'], + ['anthropic.claude-opus-4-6-v1', 'amazon-bedrock'], + ['anthropic.claude-fable-5-1', 'amazon-bedrock'], + ['claude-sonnet-4-5@20250929', 'google-vertex-ai'], + ['claude-sonnet-4-5@20240229', 'google-vertex-ai'], + ]) { + const rec = parse(assistant(model)); + assert.equal(rec.sessionOrigin.surface, 'claude-desktop'); + assert.deepEqual(rec.sessionOrigin.attributes, ['on 3P']); + assert.equal(rec.sessionOrigin.thirdPartyProvider, provider); + assert.equal(rec.sessionOrigin.thirdPartyProviderBasis, 'assistant-model-id'); + assert.equal(rec.inferenceProvider, null, 'existing pricing/provider axis is unchanged'); + } +}); + +test('ordinary models, unrelated current configuration, unknown gateways and malformed metadata stay unknown', () => { + for (const model of ['claude-sonnet-4-5', 'anthropic/claude-sonnet-4-5', + 'https://secret:token@private.example/model', '//private.example/model', 'claude-sonnet-4-5@bad', + 'anthropic.claude-sonnet-4-this-is-not-a-model', 'claude-sonnet-4-5@20999999', + 'claude-sonnet-4-5@20260229', 'anthropic.claude-sonnet-4-5-20260229-v1:0', + { value: 'us.anthropic.claude-sonnet-4-5' }]) { + const { session: rec, turns } = parseClaude(assistant(model, { + env: { CLAUDE_CODE_USE_BEDROCK: '1', ANTHROPIC_BASE_URL: 'https://secret:token@private.example' }, + settings: { env: { CLAUDE_CODE_USE_VERTEX: '1' } }, + }), { id: 'provider-fixture', withTurns: true }); + assert.equal(rec.sessionOrigin.thirdPartyProvider, null); + assert.equal(JSON.stringify(rec.sessionOrigin).includes('private.example'), false); + assert.equal(JSON.stringify(rec.sessionOrigin).includes('token'), false); + assert.equal(JSON.stringify(rec).includes('private.example'), false); + assert.equal(JSON.stringify(rec).includes('secret:token'), false); + assert.equal(JSON.stringify(turns).includes('private.example'), false); + } +}); + +test('a local API-error placeholder cannot erase a preceding completed provider observation', () => { + const actual = assistant('us.anthropic.claude-sonnet-4-6'); + const error = assistant('', { isApiErrorMessage: true }); + const rec = parse(actual, error); + assert.equal(rec.sessionOrigin.thirdPartyProvider, 'amazon-bedrock'); + assert.equal(rec.responses, 1); + assert.equal(rec.exceptions, 1); + assert.deepEqual(rec.models, ['us.anthropic.claude-sonnet-4-6']); +}); + +test('conflicting provider-specific assistant IDs leave session provider unknown', () => { + const rec = parse(assistant('us.anthropic.claude-sonnet-4-6'), assistant('claude-haiku-4-5@20251001')); + assert.equal(rec.sessionOrigin.thirdPartyProvider, null); + assert.equal(rec.sessionOrigin.thirdPartyProviderBasis, undefined); + assert.equal(parse(assistant('us.anthropic.claude-sonnet-4-6'), assistant('claude-opus-5')) + .sessionOrigin.thirdPartyProvider, null, 'an unmatched model may have used a different route'); +}); + +test('imported Codex copy never inherits a Claude provider claim', () => { + const copy = new Rollout({ id: 'copy' }).meta({ originator: 'Codex Desktop' }) + .taskStarted('external-import-turn-1').user('copy').agent('response'); + const rec = parseCodex(copy.toString(), { id: 'copy' }).session; + assert.equal(rec.imported, true); + assert.equal(rec.sessionOrigin.thirdPartyProvider, null); +}); + +test('provider detail survives the existing schema-26 cold and warm cache', async (t) => { + _resetForTest(); + const dir = tempDir('ak-provider-', t), project = path.join(dir, 'claude', '-synthetic-project'); + fs.mkdirSync(project, { recursive: true }); + fs.writeFileSync(path.join(project, 'provider-fixture.jsonl'), `${assistant('us.anthropic.claude-sonnet-4-6')}\n`); + const options = { days: 2, now: Date.parse('2026-09-29T12:00:00Z'), + roots: { claude: path.join(dir, 'claude'), codex: path.join(dir, 'codex') }, + cachePath: path.join(dir, 'cache', 'usage-index.json'), + deps: { costOf: () => 0, pricesAsOf: 'fixture', classify: () => ({ category: 'Build', confidence: 1, basis: 'fixture' }), detectInsights: () => [] } }; + const cold = await buildIndex(options); + assert.equal(cold.sessions[0].sessionOrigin.thirdPartyProvider, 'amazon-bedrock'); + const cache = JSON.parse(fs.readFileSync(options.cachePath, 'utf8')); + assert.equal(Object.values(cache.entries)[0].session.sessionOrigin.thirdPartyProviderBasis, 'assistant-model-id'); + _resetForTest(); + const warm = await buildIndex(options); + assert.deepEqual(warm.sessions[0].sessionOrigin, cold.sessions[0].sessionOrigin); +}); diff --git a/tests/kit/usage-claude-record-coverage.test.mjs b/tests/kit/usage-claude-record-coverage.test.mjs new file mode 100644 index 00000000..bb1f3605 --- /dev/null +++ b/tests/kit/usage-claude-record-coverage.test.mjs @@ -0,0 +1,133 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { parseClaude } from '../../src/lib/usage-parsers.mjs'; +import { buildIndex, _resetForTest } from '../../src/lib/usage-index.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; + +const at = '2026-09-28T12:00:00Z'; +const line = (value) => JSON.stringify(value); +const assistant = { type: 'assistant', timestamp: at, message: { + id: 'm1', role: 'assistant', model: 'claude-opus-5', + usage: { input_tokens: 3, output_tokens: 5 }, content: [{ type: 'text', text: 'ok' }], +} }; +const deps = { costOf: () => 2, pricesAsOf: '2026-09-01', + classify: () => ({ category: 'Build', confidence: 1, basis: 'test' }), detectInsights: () => [] }; + +const raw = [ + line({ type: 'user', timestamp: at, message: { role: 'user', content: 'hello' } }), + line(assistant), line({ type: 'ai-title', aiTitle: 'safe title' }), + line({ type: 'cost-state', sessionId: 's1', totalCostUSD: 1 }), + line({ type: 'system', timestamp: at }), line({ type: 'attachment', timestamp: at }), + line({ type: 'future-private-type', timestamp: at, message: { usage: { output_tokens: 900 } } }), + line({ type: 42, timestamp: at }), '{bad json', +].join('\n'); + +test('Claude record coverage separates handled, ignored, unknown, invalid type and malformed JSON', () => { + const { session, parseStats } = parseClaude(raw, { id: 's1' }); + assert.deepEqual(parseStats, { + knownHandledRecords: 4, knownIgnoredRecords: 2, unknownRecords: 1, + invalidTypeRecords: 1, malformedRecords: 1, + }); + assert.equal(session.responses, 1); + assert.equal(session.usage[0].input, 3); + assert.equal(session.usage[0].output, 5); + assert.equal(JSON.stringify({ session, parseStats }).includes('future-private-type'), false); +}); + +test('valid whitespace-prefixed objects and non-object JSON have separate coverage from invalid JSON', () => { + const source = [' ' + line({ type: 'user', timestamp: at, message: { role: 'user', content: 'hello' } }), + '\t' + line(assistant), ' [1,2]', '"private scalar"', 'null', '42', ' ', '{bad json'].join('\n'); + const { session, parseStats } = parseClaude(source, { id: 's1' }); + assert.deepEqual(parseStats, { + knownHandledRecords: 2, knownIgnoredRecords: 0, unknownRecords: 0, + invalidTypeRecords: 4, malformedRecords: 1, + }); + assert.equal(session.responses, 1); + assert.equal(session.usage[0].input, 3); + assert.equal(session.usage[0].output, 5); + assert.equal(JSON.stringify({ session, parseStats }).includes('private scalar'), false); +}); + +test('whitespace-prefixed known records stay healthy across cold and warm cache reads', async () => { + const root = tempDir('ak-claude-whitespace-health'); + const project = path.join(root, 'claude', 'project'); + fs.mkdirSync(project, { recursive: true }); + fs.writeFileSync(path.join(project, 's1.jsonl'), [' ' + line(assistant), '\t' + line({ type: 'system' })].join('\n')); + const options = { days: 7, now: Date.parse('2026-09-29T00:00:00Z'), + roots: { claude: path.join(root, 'claude'), codex: path.join(root, 'codex') }, + cachePath: path.join(root, 'cache.json'), deps }; + for (let i = 0; i < 2; i++) { + _resetForTest(); + const result = await buildIndex(options); + assert.equal(result.sourceHealth.claude.status, 'ok'); + assert.deepEqual(result.sourceHealth.claude.diagnostics.records, { + knownHandledRecords: 1, knownIgnoredRecords: 1, unknownRecords: 0, + invalidTypeRecords: 0, malformedRecords: 0, coverage: 'complete', + }); + assert.equal(result.totals.responses, 1); + assert.equal(result.totals.cost, 2); + } +}); + +test('cold, warm and legacy v26 cache expose count-only incomplete coverage without changing totals', async () => { + const root = tempDir('ak-claude-record-coverage'); + const project = path.join(root, 'claude', 'project'); + fs.mkdirSync(project, { recursive: true }); + const file = path.join(project, 's1.jsonl'); + fs.writeFileSync(file, raw); + const options = { days: 7, now: Date.parse('2026-09-29T00:00:00Z'), + roots: { claude: path.join(root, 'claude'), codex: path.join(root, 'codex') }, + cachePath: path.join(root, 'cache.json'), deps }; + _resetForTest(); + const cold = await buildIndex(options); + const expected = { knownHandledRecords: 4, knownIgnoredRecords: 2, unknownRecords: 1, + invalidTypeRecords: 1, malformedRecords: 1, coverage: 'incomplete' }; + assert.deepEqual(cold.sourceHealth.claude.diagnostics.records, expected); + assert.equal(cold.sourceHealth.claude.status, 'degraded'); + assert.equal(cold.totals.responses, 1); + assert.equal(cold.totals.cost, 2); + const cached = JSON.parse(fs.readFileSync(options.cachePath, 'utf8')); + assert.deepEqual(cached.entries[file].parseStats, { + knownHandledRecords: 4, knownIgnoredRecords: 2, unknownRecords: 1, + invalidTypeRecords: 1, malformedRecords: 1, + }); + assert.equal(JSON.stringify(cached).includes('future-private-type'), false); + _resetForTest(); + const warm = await buildIndex(options); + assert.deepEqual(warm.sourceHealth.claude.diagnostics.records, expected); + assert.equal(warm.totals.cost, cold.totals.cost); + + delete cached.entries[file].parseStats; + fs.writeFileSync(options.cachePath, JSON.stringify(cached)); + _resetForTest(); + const repaired = await buildIndex(options); + assert.deepEqual(repaired.sourceHealth.claude.diagnostics.records, expected); + assert.deepEqual(JSON.parse(fs.readFileSync(options.cachePath, 'utf8')).entries[file].parseStats, + { knownHandledRecords: 4, knownIgnoredRecords: 2, unknownRecords: 1, + invalidTypeRecords: 1, malformedRecords: 1 }); +}); + +test('known-only files report complete coverage and an unreadable root reports unknown coverage', async () => { + const root = tempDir('ak-claude-record-health'); + const project = path.join(root, 'claude', 'project'); + fs.mkdirSync(project, { recursive: true }); + fs.writeFileSync(path.join(project, 's1.jsonl'), [line(assistant), line({ type: 'system' })].join('\n')); + const options = { days: 7, now: Date.parse('2026-09-29T00:00:00Z'), + roots: { claude: path.join(root, 'claude'), codex: path.join(root, 'codex') }, + cachePath: path.join(root, 'cache.json'), deps }; + _resetForTest(); + const good = await buildIndex(options); + assert.deepEqual(good.sourceHealth.claude.diagnostics.records, { + knownHandledRecords: 1, knownIgnoredRecords: 1, unknownRecords: 0, + invalidTypeRecords: 0, malformedRecords: 0, coverage: 'complete', + }); + assert.equal(good.sourceHealth.claude.status, 'ok'); + const badRoot = path.join(root, 'not-a-directory'); + fs.writeFileSync(badRoot, ''); + _resetForTest(); + const bad = await buildIndex({ ...options, roots: { ...options.roots, claude: badRoot } }); + assert.equal(bad.sourceHealth.claude.status, 'degraded'); + assert.equal(bad.sourceHealth.claude.diagnostics.records.coverage, 'unknown'); +}); diff --git a/tests/kit/usage-codex-effort-timing-compaction.test.mjs b/tests/kit/usage-codex-effort-timing-compaction.test.mjs new file mode 100644 index 00000000..ef4adbad --- /dev/null +++ b/tests/kit/usage-codex-effort-timing-compaction.test.mjs @@ -0,0 +1,162 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { parseCodex } from '../../src/lib/usage-parsers.mjs'; +import { aggregate } from '../../src/lib/usage-aggregate.mjs'; +import { buildIndex, _resetForTest } from '../../src/lib/usage-index.mjs'; +import { Rollout, usage, codexSandbox, stubDeps } from './helpers/codex-rollout.mjs'; +import { forkedSubagent } from './helpers/codex-rollout.mjs'; + +const now = Date.parse('2026-07-25T12:00:00Z'); +const compacted = { window_number: 2, replacement_history: [], latest_token_usage_record: {} }; +const sample = () => new Rollout({ id: 'unit10' }).meta() + .taskStarted('t1').turn('gpt-5.6', { effort: 'high' }).user('work') + .tokenCount(usage({ input: 100, output: 5 })) + .raw('event_msg', { type: 'task_complete', duration_ms: 9000, time_to_first_token_ms: 1200 }) + .raw('compacted', compacted) + .item('ContextCompaction', { id: 'compaction-1' }) + .taskStarted('t2').turn('gpt-5.6', { effort: 'low' }).user('continue') + .tokenCount(usage({ input: 40, output: 3 })) + .raw('event_msg', { type: 'task_complete', duration_ms: 7000, time_to_first_token_ms: 800 }); + +test('native effort, measured first-token times and paired compaction reconcile without token replay', () => { + const rec = parseCodex(String(sample()), { id: 'fallback' }).session; + assert.deepEqual(rec.codexEffort, { last: 'low', counts: { high: 1, low: 1 } }); + assert.deepEqual(rec.firstTokenMs, { count: 2, total: 2000, min: 800, max: 1200, provenance: 'host-observed' }); + assert.equal(rec.compactions, 1); + const a = aggregate([rec], { days: 14, now, cutoff: now - 14 * 86400000, deps: stubDeps() }); + assert.equal(a.totals.tokens, 148); + assert.equal(a.sessions[0].compactions, 1); + assert.deepEqual(a.sessions[0].codexEffort, rec.codexEffort); + assert.deepEqual(a.sessions[0].firstTokenMs, rec.firstTokenMs); + assert.equal(a.totals.compactions, 1); + assert.deepEqual(a.totals.firstTokenMs, { count: 2, total: 2000, min: 800, max: 1200, provenance: 'host-observed' }); +}); + +test('missing or malformed timing and unbounded effort never become measurements', () => { + const r = new Rollout({ id: 'invalid' }).meta() + .raw('event_msg', { type: 'task_started', turn_id: 't1', started_at: '2026-07-24T09:00:00.000Z' }) + .turn('gpt-5.6', { effort: 'arbitrary provider string' }).user('work') + .tokenCount(usage({ input: 20, output: 2 })) + .raw('event_msg', { type: 'task_complete', duration_ms: 4500 }) + .turn('gpt-5.6', { effort: 'high'.repeat(100) }) + .raw('event_msg', { type: 'task_complete', duration_ms: 1, time_to_first_token_ms: -1 }); + const rec = parseCodex(String(r), { id: 'fallback' }).session; + assert.equal(rec.codexEffort, null); + assert.equal(rec.firstTokenMs, null); + assert.equal(rec.compactions, 0); + assert.equal(rec.latCount > 0, true, 'existing total-duration latency remains separate'); +}); + +test('cold and warm cache retain the same observed detail', async () => { + const sb = codexSandbox({ 'rollout-unit10.jsonl': sample() }); + const options = { days: 14, now, roots: sb.roots, cachePath: sb.cachePath, deps: stubDeps() }; + for (let i = 0; i < 2; i++) { + _resetForTest(); + const a = await buildIndex(options); + assert.equal(a.sessions[0].codexEffort.last, 'low'); + assert.equal(a.sessions[0].firstTokenMs.total, 2000); + assert.equal(a.sessions[0].compactions, 1); + assert.equal(a.totals.tokens, 148); + if (i) assert.equal(a.sourceHealth.codex.diagnostics.cachedFiles, 1); + } +}); + +test('imported turns cannot contribute effort, first-token time or compaction', () => { + const r = new Rollout({ id: 'mixed' }).meta({ originator: 'Codex Desktop' }) + .taskStarted('external-import-turn-1').turn('gpt-5.6', { turn_id: 'external-import-turn-1', effort: 'high' }) + .raw('event_msg', { type: 'task_complete', time_to_first_token_ms: 999 }) + .raw('compacted', compacted) + .taskStarted('native-1').turn('gpt-5.6', { turn_id: 'native-1', effort: 'low' }) + .tokenCount(usage({ input: 20, output: 2 })) + .raw('event_msg', { type: 'task_complete', time_to_first_token_ms: 20 }); + const rec = parseCodex(String(r), { id: 'fallback' }).session; + assert.deepEqual(rec.codexEffort, { last: 'low', counts: { low: 1 } }); + assert.deepEqual(rec.firstTokenMs, { count: 1, total: 20, min: 20, max: 20, provenance: 'host-observed' }); + assert.equal(rec.compactions, 0); +}); + +test('subagent replay contributes no parent effort, timing or compaction', () => { + const parent = new Rollout({ id: 'parent' }).meta().taskStarted('parent-turn') + .turn('gpt-5.6', { effort: 'high' }) + .raw('event_msg', { type: 'task_complete', time_to_first_token_ms: 900 }) + .raw('compacted', compacted); + const child = forkedSubagent({ id: 'child', parent, own: (r) => r + .turn('gpt-5.6', { effort: 'medium' }) + .tokenCount(usage({ input: 10, output: 2 })) + .raw('event_msg', { type: 'task_complete', time_to_first_token_ms: 30 }) }); + const rec = parseCodex(String(child), { id: 'fallback' }).session; + assert.deepEqual(rec.codexEffort, { last: 'medium', counts: { medium: 1 } }); + assert.equal(rec.firstTokenMs.total, 30); + assert.equal(rec.compactions, 0); +}); + +test('older v26 records with absent or malformed optional detail remain unmeasured', () => { + const old = parseCodex(String(new Rollout({ id: 'old' }).meta().taskStarted('t1') + .turn().tokenCount(usage({ input: 10, output: 2 }))), { id: 'fallback' }).session; + delete old.codexEffort; + delete old.firstTokenMs; + delete old.compactions; + const malformed = { ...old, id: 'malformed', codexEffort: { last: 'arbitrary', counts: { arbitrary: 1 } }, + firstTokenMs: { count: 1, total: -3, min: -3, max: -3, provenance: 'derived' }, compactions: -2 }; + const a = aggregate([old, malformed], { days: 14, now, cutoff: now - 14 * 86400000, deps: stubDeps() }); + for (const row of a.sessions) { + assert.equal(row.codexEffort, null); + assert.equal(row.firstTokenMs, null); + assert.equal(row.compactions, 0); + } + assert.equal(a.totals.compactions, 0); + assert.equal(a.totals.firstTokenMs, null); +}); + +test('compaction observations on separate turns count separately', () => { + const r = new Rollout({ id: 'separate' }).meta().taskStarted('a').turn() + .tokenCount(usage({ input: 10, output: 2 })).raw('compacted', compacted) + .taskStarted('b').turn().raw('event_msg', { + type: 'item_completed', turn_id: 'b', item: { type: 'ContextCompaction' }, + }); + const rec = parseCodex(String(r), { id: 'fallback' }).session; + assert.equal(rec.compactions, 2); + assert.deepEqual(rec.compactionEvidence, { lowerBound: 2, upperBound: 2 }); +}); + +test('partially paired compaction shapes retain an explicit uncertainty bound', () => { + const r = new Rollout({ id: 'partial' }).meta().taskStarted('a').turn() + .tokenCount(usage({ input: 10, output: 2 })).raw('compacted', compacted) + .raw('event_msg', { type: 'item_completed', turn_id: 'a', item: { type: 'ContextCompaction' } }) + .taskStarted('b').turn().raw('compacted', { ...compacted, window_number: 3 }); + const rec = parseCodex(String(r), { id: 'fallback' }).session; + assert.equal(rec.compactions, 2); + assert.deepEqual(rec.compactionEvidence, { lowerBound: 2, upperBound: 3 }); + const a = aggregate([rec], { days: 14, now, cutoff: now - 14 * 86400000, deps: stubDeps() }); + assert.equal(a.totals.compactions, 2); + assert.deepEqual(a.totals.compactionEvidence, { lowerBound: 2, upperBound: 3 }); +}); + +test('an unmatched item turn ID cannot prove it differs from an ID-less compacted record', () => { + const r = new Rollout({ id: 'unmatched' }).meta().turn() + .tokenCount(usage({ input: 10, output: 2 })).raw('compacted', compacted) + .raw('event_msg', { type: 'item_completed', turn_id: 'unseen', item: { type: 'ContextCompaction' } }); + const rec = parseCodex(String(r), { id: 'fallback' }).session; + assert.deepEqual(rec.compactionEvidence, { lowerBound: 1, upperBound: 2 }); +}); + +test('pre-ordinal subagent history cannot supply effort, first-token time or compactions', () => { + const r = new Rollout({ id: 'old-child' }).meta({ thread_source: 'subagent' }) + .taskStarted('parent-turn').turn('gpt-5.6', { effort: 'high' }) + .tokenCount(usage({ input: 100, output: 5 })) + .raw('event_msg', { type: 'task_complete', time_to_first_token_ms: 950 }) + .raw('compacted', compacted); + const preOrdinal = r.lines.map((line) => { + const entry = JSON.parse(line); + delete entry.ordinal; + return JSON.stringify(entry); + }).join('\n'); + const rec = parseCodex(preOrdinal, { id: 'fallback' }).session; + assert.deepEqual(rec.usage, []); + assert.equal(rec.codexEffort, null); + assert.equal(rec.firstTokenMs, null); + assert.equal(rec.compactions, 0); + const a = aggregate([rec], { days: 14, now, cutoff: now - 14 * 86400000, deps: stubDeps() }); + assert.equal(a.totals.compactions, 0); + assert.equal(a.totals.firstTokenMs, null); +}); diff --git a/tests/kit/usage-codex-import-gaps.test.mjs b/tests/kit/usage-codex-import-gaps.test.mjs new file mode 100644 index 00000000..a42467e8 --- /dev/null +++ b/tests/kit/usage-codex-import-gaps.test.mjs @@ -0,0 +1,115 @@ +// Regression for lost or explicitly invalid ownership boundaries in mixed files. +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import path from 'node:path'; +import { parseCodex } from '../../src/lib/usage-parsers.mjs'; +import { openCodexRollout } from '../../src/lib/codex-rollout-reader.mjs'; +import { scanTranscriptCwds } from '../../src/lib/footprint/project-sources.mjs'; +import { buildIndex, _resetForTest } from '../../src/lib/usage-index.mjs'; +import { Rollout, usage, codexSandbox, stubDeps } from './helpers/codex-rollout.mjs'; + +function beforeGap({ nativeActivity = true, imports = true } = {}) { + const r = new Rollout({ id: 'mixed' }).meta({ cwd: '/copied', originator: 'Codex Desktop' }); + if (imports) r.taskStarted('external-import-turn-1').agent('copy').tokenCount(usage({ input: 1000 })); + r.taskStarted('native-1').turn('gpt-5.6', { turn_id: 'native-1', cwd: '/native' }); + if (nativeActivity) r.agent('own').tokenCount(usage({ input: 100 })); + return r; +} +function afterGap(r) { return r.agent('owner-unproved').tokenCount(usage({ input: 500 })); } +function parseBoth(r) { + const sb = codexSandbox({ 'rollout-gap.jsonl': r }); + const file = path.join(sb.roots.codex, '2026', '07', '24', 'rollout-gap.jsonl'); + return [parseCodex(String(r), { id: 'fallback' }), + parseCodex(openCodexRollout(file, { chunkBytes: 37 }), { id: 'fallback' }), + parseCodex(openCodexRollout(file), { id: 'fallback' })]; +} +function assertExcluded(result) { + assert.equal(result.session.responses, 0, 'incomplete mixed files are conservatively excluded'); + assert.deepEqual(result.session.usage, [], 'copied cumulative usage cannot enter native totals'); + assert.equal(result.session.importEvidence.ownershipComplete, false); + assert.equal(result.session.imported, true); +} + +for (const broken of ['{"type":"event_msg","payload":{"type":"task_started","turn_id":"external-import-turn-2"', + 'not-json', '[]', 'null']) { + test(`string and streaming parsing disclose a lost ownership boundary: ${broken.slice(0, 12)}`, () => { + const r = beforeGap(); + r.lines.push(broken); + afterGap(r); + for (const result of parseBoth(r)) { + assertExcluded(result); + assert.equal(result.session.importEvidence.malformedRecords, 1); + } + const sb = codexSandbox({ 'rollout-gap.jsonl': r }); + const scan = scanTranscriptCwds(sb.roots.codex, 'codex'); + assert.deepEqual(scan.sightings, []); + assert.equal(scan.importedUnresolved, 1); + assert.equal(scan.complete, false); + }); +} + +test('string and streaming Codex imports keep whitespace-prefixed records fail closed', () => { + const r = beforeGap(); + r.lines.push(' {"type":"event_msg","payload":{"type":"task_started","turn_id":"native-2"}}'); + afterGap(r); + for (const result of parseBoth(r)) { + assertExcluded(result); + assert.equal(result.session.importEvidence.malformedRecords, 1); + } +}); + +for (const turnId of [null, '', 'bad id', 1, {}, [], 'x'.repeat(257)]) { + test(`explicitly invalid context ID breaks adjacency: ${JSON.stringify(turnId).slice(0, 24)}`, () => { + const r = afterGap(beforeGap().turn('copied-model', { turn_id: turnId, cwd: '/copied' })); + for (const result of parseBoth(r)) { + assertExcluded(result); + assert.ok(result.session.importEvidence.ambiguousRecords >= 3); + } + const noOwn = afterGap(beforeGap({ nativeActivity: false }) + .turn('copied-model', { turn_id: turnId, cwd: '/copied' })); + const sb = codexSandbox({ 'rollout-gap.jsonl': noOwn }); + const scan = scanTranscriptCwds(sb.roots.codex, 'codex'); + assert.deepEqual(scan.sightings, [], 'unproved native adjacency must not establish an app or project'); + assert.equal(scan.importedMixed, 0); + assert.equal(scan.importedUnresolved, 1); + assert.equal(scan.complete, false); + assert.equal(scan.sessionCountComplete, false); + }); +} + +test('absent context ID can enrich an identified native turn without breaking ownership', () => { + const r = afterGap(beforeGap().turn('gpt-5.6', { cwd: '/native' })); + for (const result of parseBoth(r)) { + assert.equal(result.session.responses, 2); + assert.equal(result.session.usage[0].input, 600); + assert.equal(result.session.importEvidence.ownershipComplete, true); + } +}); + +test('all-native legacy files retain their permissive skip behavior', () => { + const r = beforeGap({ imports: false }); + r.lines.push('{bad-json'); + afterGap(r.turn('gpt-5.6', { turn_id: null })); + for (const result of parseBoth(r)) { + assert.equal(result.session.responses, 2); + assert.equal(result.session.usage[0].input, 600); + assert.equal(result.session.imported, undefined); + } +}); + +test('malformed mixed-file ownership remains disclosed in cold and warm index diagnostics', async () => { + const r = beforeGap(); + r.lines.push('{broken-boundary'); + afterGap(r); + for (const streamAboveBytes of [0, 1_000_000]) { + const sb = codexSandbox({ 'rollout-gap.jsonl': r }); + for (let n = 0; n < 2; n++) { + _resetForTest(); + const a = await buildIndex({ days: 14, now: Date.parse('2026-07-25T12:00:00Z'), roots: sb.roots, + cachePath: sb.cachePath, deps: stubDeps(), readLimits: { streamAboveBytes } }); + assert.equal(a.totals.responses, 0); + assert.equal(a.totals.input, 0); + assert.equal(a.sourceHealth.codex.diagnostics.importOwnershipIncompleteFiles, 1); + } + } +}); diff --git a/tests/kit/usage-codex-import-turns.test.mjs b/tests/kit/usage-codex-import-turns.test.mjs new file mode 100644 index 00000000..591b2873 --- /dev/null +++ b/tests/kit/usage-codex-import-turns.test.mjs @@ -0,0 +1,261 @@ +// Ownership is exercised through parser, aggregate/cache and bounded discovery. +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { parseCodex } from '../../src/lib/usage-parsers.mjs'; +import { buildIndex, _resetForTest } from '../../src/lib/usage-index.mjs'; +import { scanTranscriptCwds, discoverProjectSources } from '../../src/lib/footprint/project-sources.mjs'; +import { Rollout, usage, codexSandbox, stubDeps, forkedSubagent } from './helpers/codex-rollout.mjs'; + +const imported = (originator = 'Codex Desktop') => new Rollout({ id: 'mixed' }) + .meta({ originator, cwd: '/copied', thread_source: undefined }) + .taskStarted('external-import-turn-1').turn('copied-model', { turn_id: 'external-import-turn-1', cwd: '/copied' }) + .user('copied secret prompt').agent('copied secret answer').item('FileChange') + .tokenCount(usage({ input: 1000, cached: 200, output: 100 })); +const own = (r) => r.taskStarted('native-1').turn('gpt-5.6', { turn_id: 'native-1', cwd: '/genuine' }) + .user('genuine prompt').agent('genuine answer').item('CommandExecution') + .tokenCount(usage({ input: 100, cached: 20, output: 10 })); +const parse = (r) => parseCodex(String(r), { id: 'fallback', withTurns: true }); + +test('mixed ownership excludes copied content and cumulative baseline but retains genuine project and surface', () => { + const { session: s, turns, parseStats } = parse(own(imported())); + assert.equal(s.imported, undefined); + assert.equal(s.id, 'mixed'); + assert.equal(s.prompts, 1); + assert.equal(s.responses, 1); + assert.deepEqual(s.models, ['gpt-5.6']); + assert.deepEqual(s.tools, { CommandExecution: 1 }); + assert.equal(s.contextEvidence.input.samples, 1); + assert.equal(s.contextEvidence.input.peak, 100); + assert.equal(s.usage[0].input, 80); + assert.equal(s.usage[0].cacheRead, 20); + assert.equal(s.usage[0].output, 10); + assert.equal(s.project, 'genuine'); + assert.equal(s.sessionOrigin.surface, 'chatgpt-desktop-codex'); + assert.equal(s.importEvidence.importedTurns, 1); + assert.equal(parseStats.prompts, 1); + assert.equal(parseStats.responses, 1); + assert.equal(JSON.stringify({ s, turns }).includes('copied secret'), false); +}); + +test('native CLI continuation keeps its declaration and handles a cumulative reset', () => { + const { session: s } = parse(own(imported('codex-tui').resetTotal())); + assert.equal(s.sessionOrigin.surface, 'codex-cli'); + assert.equal(s.usage[0].input, 80); + assert.equal(s.usage[0].output, 10); +}); + +test('native before late import stays counted and repeated metadata cannot replace identity or native origin', () => { + const r = new Rollout({ id: 'first' }).meta({ originator: 'codex-tui' }) + .taskStarted('native-1').turn().user('own').agent().tokenCount(usage({ input: 100, output: 10 })) + .meta({ id: 'parent', originator: 'Codex Desktop' }).taskStarted('external-import-turn-1') + .user('copied').agent().tokenCount(usage({ input: 2000, output: 200 })); + const { session: s } = parse(r); + assert.equal(s.id, 'first'); + assert.equal(s.responses, 1); + assert.equal(s.usage[0].input, 100); + assert.equal(s.sessionOrigin.surface, 'codex-cli'); +}); + +test('foreign imported completion does not close a native turn; a new imported start does', () => { + const r = imported().taskStarted('native-1') + .raw('event_msg', { type: 'item_completed', turn_id: 'native-1', item: { type: 'AgentMessage', text: 'own' } }) + .raw('event_msg', { type: 'task_complete', turn_id: 'external-import-turn-1' }) + .tokenCount(usage({ input: 100, output: 10 })) + .raw('event_msg', { type: 'task_complete', turn_id: 'native-1' }) + .taskStarted('external-import-turn-2').agent('copy').tokenCount(usage({ input: 500, output: 50 })); + const { session: s } = parse(r); + assert.equal(s.responses, 1); + assert.equal(s.usage[0].input, 100); + assert.equal(s.importEvidence.importedTurns, 2); +}); + +test('missing or mismatched turn IDs cannot establish native ownership after import', () => { + const r = imported().taskStarted(undefined); + // Builder has a default ID; replace this boundary with a truly missing ID. + r.lines.pop(); + r.raw('event_msg', { type: 'task_started' }).user('ambiguous').agent('ambiguous') + .tokenCount(usage({ input: 100, output: 10 })); + const { session: s } = parse(r); + assert.equal(s.imported, true); + assert.equal(s.responses, 0); + assert.ok(s.importEvidence.ambiguousRecords > 0); + const mismatch = own(imported()).raw('event_msg', { type: 'agent_message', turn_id: 'unopened', message: 'ambiguous' }); + assert.equal(parse(mismatch).session.responses, 1); +}); + +test('a marker embedded in prompt text changes no ownership', () => { + const r = new Rollout({ id: 'ordinary' }).meta().turn().user('external-import-turn-1').agent() + .tokenCount(usage({ input: 100, output: 10 })); + assert.equal(parse(r).session.responses, 1); + assert.equal(parse(r).session.imported, undefined); +}); + +test('replayed imports do not replace child identity or count parent native activity', () => { + const parent = own(imported()); + const child = forkedSubagent({ id: 'child', parent, + own: (r) => r.user('child').agent().tokenCount(usage({ input: 50, output: 5 })) }); + const { session: s } = parse(child); + assert.equal(s.id, 'child'); + assert.equal(s.threadSource, 'subagent'); + assert.equal(s.responses, 1); + assert.equal(s.usage[0].input, 50); +}); + +test('cold and warm aggregate retain mixed usage and exclusion diagnostics', async () => { + const sb = codexSandbox({ 'rollout-mixed.jsonl': own(imported()), 'rollout-copy.jsonl': imported() }); + const options = { days: 14, now: Date.parse('2026-07-25T12:00:00Z'), roots: sb.roots, + cachePath: sb.cachePath, deps: stubDeps() }; + for (let n = 0; n < 2; n++) { + _resetForTest(); + const a = await buildIndex(options); + assert.equal(a.totals.sessions, 1); + assert.equal(a.totals.responses, 1); + assert.equal(a.totals.input, 80); + assert.equal(a.sessions[0].importEvidence.importedTurns, 1); + assert.equal(a.sessions[0].cost, 1); + assert.equal(a.sourceHealth.codex.diagnostics.importedExcluded, 1); + assert.equal(a.sourceHealth.codex.diagnostics.importedMixed, 1); + assert.equal(a.sourceHealth.codex.diagnostics.importedTurnsExcluded, 2); + if (n) assert.equal(a.sourceHealth.codex.diagnostics.cachedFiles, 2); + } +}); + +test('bounded tail discovery finds interleaved native Desktop activity beyond the head', () => { + const r = imported().raw('response_item', { type: 'message', content: 'x'.repeat(300_000) }); + own(r).taskStarted('external-import-turn-2').agent('later copied answer'); + const sb = codexSandbox({ 'rollout-mixed.jsonl': r }); + const scan = scanTranscriptCwds(sb.roots.codex, 'codex'); + assert.equal(scan.sightings.length, 1); + assert.equal(scan.sightings[0].cwd, '/genuine'); + assert.equal(scan.sightings[0].sessionOrigin.surface, 'chatgpt-desktop-codex'); + assert.equal(scan.importedExcluded, 0); + assert.equal(scan.importedMixed, 1); +}); + +test('bounded observations cannot label a large unseen middle imported-only or recover its encoded directory', () => { + const r = imported().raw('response_item', { type: 'message', content: 'x'.repeat(3_000_000) }); + const sb = codexSandbox({ 'rollout-copy.jsonl': r }); + const scan = scanTranscriptCwds(sb.roots.codex, 'codex', { decodeDir: () => '/not-evidence' }); + assert.equal(scan.sightings.length, 0); + assert.equal(scan.importedExcluded, 0); + assert.equal(scan.importedUnresolved, 1); + assert.equal(scan.complete, false); + assert.equal(scan.sessionCountComplete, false); +}); + +test('a pure import whose whole bounded file is read remains excluded', () => { + const sb = codexSandbox({ 'rollout-copy.jsonl': imported() }); + const scan = scanTranscriptCwds(sb.roots.codex, 'codex'); + assert.equal(scan.importedExcluded, 1); + assert.equal(scan.sightings.length, 0); + assert.equal(scan.complete, true); + assert.equal(fs.existsSync(path.join(sb.dir, 'cache')), false); +}); + +// A reset can restart above the old imported total; last===total is explicit +// first-call evidence even when no monotonic field decreased. +test('a native counter reset above the imported baseline books the whole first call', () => { + const r = imported().resetTotal().taskStarted('native-1').turn() + .agent().tokenCount(usage({ input: 2000, cached: 400, output: 200 })); + const s = parse(r).session; + assert.equal(s.usage[0].input, 1600); + assert.equal(s.usage[0].cacheRead, 400); + assert.equal(s.usage[0].output, 200); +}); + +test('imported discovery excludes replayed native parent activity in a child with no own work', () => { + const r = forkedSubagent({ parent: own(imported()), own: () => {} }); + const sb = codexSandbox({ 'rollout-child.jsonl': r }); + const scan = scanTranscriptCwds(sb.roots.codex, 'codex'); + assert.equal(scan.sightings.length, 0); +}); + +test('mixed streaming and string parsing agree without leaking copied context', async () => { + const { openCodexRollout } = await import('../../src/lib/codex-rollout-reader.mjs'); + const r = own(imported()); + const sb = codexSandbox({ 'rollout-mixed.jsonl': r }); + const file = path.join(sb.roots.codex, '2026', '07', '24', 'rollout-mixed.jsonl'); + const source = openCodexRollout(file); + { + const streamed = parseCodex(source, { id: 'fallback', withTurns: true }); + assert.deepEqual(streamed, parse(r)); + assert.equal(JSON.stringify(streamed).includes('copied-model'), false); + } +}); + +test('discovery reports ambiguous missing-ID activity as unresolved even at EOF', () => { + const r = imported().raw('event_msg', { type: 'task_started' }).agent('unknown owner'); + const sb = codexSandbox({ 'rollout-ambiguous.jsonl': r }); + const scan = scanTranscriptCwds(sb.roots.codex, 'codex'); + assert.equal(scan.importedExcluded, 0); + assert.equal(scan.importedUnresolved, 1); + assert.equal(scan.complete, false); +}); + +test('discovery summary carries mixed and unresolved counts for consumers', () => { + const sb = codexSandbox({ 'rollout-mixed.jsonl': own(imported()) }); + const result = discoverProjectSources({ claudeRoot: sb.roots.claude, codexRoot: sb.roots.codex, + opencodeDbFile: path.join(sb.dir, 'absent.db') }); + assert.equal(result.importedMixed, 1); + assert.equal(result.importedUnresolved, 0); +}); + +test('bounded import inspection honors a shared byte budget without reading payload bytes', async () => { + const { inspectCodexImport } = await import('../../src/lib/footprint/codex-import-discovery.mjs'); + const sb = codexSandbox({ 'rollout-copy.jsonl': imported() }); + const file = path.join(sb.roots.codex, '2026', '07', '24', 'rollout-copy.jsonl'); + let reads = 0; + const fsImpl = { ...fs, readSync: (...args) => { reads++; return fs.readSync(...args); } }; + const result = inspectCodexImport(file, { fsImpl, headBytes: 262144, budget: { remaining: 0 } }); + assert.equal(result.kind, 'unresolved'); + assert.equal(reads, 0); +}); + +test('a native interval wholly inside an unread gap cannot establish a project', () => { + const r = imported().raw('response_item', { content: 'x'.repeat(300_000) }); + own(r).taskStarted('external-import-turn-2').raw('response_item', { content: 'y'.repeat(3_000_000) }); + const sb = codexSandbox({ 'rollout-gap.jsonl': r }); + const scan = scanTranscriptCwds(sb.roots.codex, 'codex'); + assert.equal(scan.importedMixed, 0); + assert.equal(scan.importedUnresolved, 1); + assert.deepEqual(scan.sightings, []); +}); + +test('streaming mixed files with clipped ownership evidence are excluded and disclosed', async () => { + const { openCodexRollout } = await import('../../src/lib/codex-rollout-reader.mjs'); + const r = own(imported()).raw('event_msg', { type: 'item_completed', turn_id: 'external-import-turn-2', + item: { type: 'FileChange', output: 'x'.repeat(9000) } }).agent('uncertain ownership'); + const sb = codexSandbox({ 'rollout-clipped.jsonl': r }); + const file = path.join(sb.roots.codex, '2026', '07', '24', 'rollout-clipped.jsonl'); + const result = parseCodex(openCodexRollout(file, { maxLineBytes: 4096 }), { id: 'fallback' }); + assert.equal(result.session.responses, 0); + assert.equal(result.session.importEvidence.ownershipComplete, false); + assert.equal(result.parseStats.clippedLines, 1); +}); + +test('import turn diagnostics remain bounded and disclose a truncated unique-turn count', () => { + const r = imported(); + for (let i = 2; i <= 4097; i++) r.taskStarted(`external-import-turn-${i}`); + const { session: s } = parse(r); + assert.equal(s.importEvidence.importedTurns, 4096); + assert.equal(s.importEvidence.importedTurnCountComplete, false); + assert.equal(s.imported, true); +}); + +test('clipped mixed-file exclusion remains visible in cold and warm source-health diagnostics', async () => { + const r = own(imported()).raw('response_item', { content: 'x'.repeat(9000) }); + const sb = codexSandbox({ 'rollout-clipped.jsonl': r }); + for (let i = 0; i < 2; i++) { + _resetForTest(); + const a = await buildIndex({ days: 14, now: Date.parse('2026-07-25T12:00:00Z'), + roots: sb.roots, cachePath: sb.cachePath, deps: stubDeps(), + readLimits: { streamAboveBytes: 0, maxLineBytes: 4096 } }); + const d = a.sourceHealth.codex.diagnostics; + assert.equal(a.totals.responses, 0); + assert.equal(d.importOwnershipIncompleteFiles, 1); + assert.equal(d.clippedLines, 1); + assert.ok(d.warnings.includes('oversized-lines-clipped')); + } +}); diff --git a/tests/kit/usage-codex-thread-source.test.mjs b/tests/kit/usage-codex-thread-source.test.mjs new file mode 100644 index 00000000..03c650a4 --- /dev/null +++ b/tests/kit/usage-codex-thread-source.test.mjs @@ -0,0 +1,150 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { classifySessionSurface } from '../../src/lib/session-surface.mjs'; +import { parseCodex } from '../../src/lib/usage-parsers.mjs'; +import { applyCodexLedger } from '../../src/lib/usage-aggregate.mjs'; +import { rowCostEvidence } from '../../src/lib/usage-cost.mjs'; +import { costOf } from '../../src/lib/pricing.mjs'; +import { buildIndex, _resetForTest } from '../../src/lib/usage-index.mjs'; +import { Rollout, usage, codexSandbox, stubDeps } from './helpers/codex-rollout.mjs'; + +test('enumerated Codex thread sources classify without guessing an unknown source', () => { + for (const [threadSource, initiator] of [ + ['user', 'person'], ['chatgpt_handoff', 'person'], ['guardian_review', 'agent'], + ['subagent', 'agent'], ['agent_created_thread', 'agent'], ['automation', 'automation'], + ['future_source', 'unknown'], + ]) { + assert.equal(classifySessionSurface({ host: 'codex', originator: 'Codex Desktop', threadSource }).initiator, + initiator, threadSource); + } + assert.equal(classifySessionSurface({ host: 'codex', originator: 'codex_exec', threadSource: 'user' }).initiator, 'automation'); + assert.equal(classifySessionSurface({ host: 'codex', originator: 'codex_cli_rs', source: 'mcp', threadSource: 'user' }).initiator, 'agent'); + for (const threadSource of [{}, 'future source']) { + const result = classifySessionSurface({ host: 'codex', originator: 'Codex Desktop', threadSource }); + assert.equal(result.initiator, 'unknown'); + assert.deepEqual(result.rawEvidence, { originator: 'Codex Desktop' }); + } + assert.equal(classifySessionSurface({ host: 'codex', originator: 'Codex Desktop' }).initiator, 'person'); + for (const originator of ['codex_exec', 'codex_sdk_ts']) { + assert.equal(classifySessionSurface({ host: 'codex', originator, threadSource: {} }).initiator, 'automation'); + } + assert.equal(classifySessionSurface({ host: 'codex', originator: 'codex_cli_rs', source: 'mcp', + threadSource: {} }).initiator, 'agent'); +}); + +test('malformed first thread source remains unknown through parser and ledger overlay', () => { + for (const threadSource of [{}, 'future source']) { + const rollout = new Rollout({ id: 'malformed' }).meta({ originator: 'Codex Desktop', thread_source: threadSource }) + .turn().agent().tokenCount(usage({ input: 10, output: 2 })); + const rec = parseCodex(rollout.toString(), { id: 'malformed' }).session; + assert.equal(rec.sessionOrigin.initiator, 'unknown'); + assert.deepEqual(rec.sessionOrigin.rawEvidence, { originator: 'Codex Desktop' }); + const ledger = { threads: new Map([['malformed', { threadSource: 'future source' }]]), parents: new Map() }; + const overlaid = applyCodexLedger([rec], ledger)[0]; + assert.equal(overlaid.sessionOrigin.initiator, 'unknown'); + assert.deepEqual(overlaid.sessionOrigin.rawEvidence, { originator: 'Codex Desktop' }); + } +}); + +test('structured source proves parent only with the observed thread_spawn shape', () => { + const parentId = '123e4567-e89b-42d3-a456-426614174000'; + const child = new Rollout({ id: 'child' }).meta({ originator: 'Codex Desktop', thread_source: 'subagent', + source: { subagent: { thread_spawn: { parent_thread_id: parentId, depth: 1 } } } }) + .turn('gpt-5.6-sol').agent().tokenCount(usage({ input: 100, output: 20 })); + const parsed = parseCodex(child.toString(), { id: 'child' }).session; + assert.equal(parsed.parentSessionId, parentId); + assert.equal(parsed.threadSource, 'subagent'); + assert.equal(parsed.sessionOrigin.initiator, 'agent'); + assert.ok(parsed.usage.length > 0, 'own usage is retained'); + const malformed = new Rollout({ id: 'other' }).meta({ source: { subagent: { thread_spawn: { parent_thread_id: '../secret' } } } }); + assert.equal(parseCodex(malformed.toString(), { id: 'other' }).session.parentSessionId, null); +}); + +test('ledger backfill updates session origin and safely rolls child under its observed parent surface', () => { + const origin = (threadSource, surface) => ({ ...classifySessionSurface({ host: 'codex', + originator: surface === 'codex-ide' ? 'codex_vscode' : 'Codex Desktop', threadSource }), + origin: 'unknown', evidence: 'desktop-origin-not-declared' }); + const parent = { id: 'parent', provider: 'codex', threadSource: 'user', sessionOrigin: origin('user', 'codex-ide') }; + const child = { id: 'child', provider: 'codex', threadSource: null, sessionOrigin: origin(null, 'desktop'), + usage: [{ input: 20 }], reasoningOutput: 2 }; + const ledger = { threads: new Map([['child', { threadSource: 'guardian_review' }]]), + parents: new Map([['child', 'parent']]) }; + const [p, c] = applyCodexLedger([parent, child], ledger); + assert.equal(p, parent); + assert.equal(c.threadSource, 'guardian_review'); + assert.equal(c.sessionOrigin.initiator, 'agent'); + assert.equal(c.sessionOrigin.surface, 'codex-ide'); + assert.equal(c.parentSessionId, 'parent'); + assert.deepEqual(c.usage, child.usage, 'reviewer own tokens are not stripped'); + const noParent = applyCodexLedger([child], ledger)[0]; + assert.equal(noParent.parentSessionId, null); + assert.equal(noParent.sessionOrigin.surface, 'chatgpt-desktop-codex'); + const declared = { ...child, threadSource: 'subagent', parentSessionId: 'parent', + sessionOrigin: origin('subagent', 'desktop') }; + assert.equal(applyCodexLedger([parent, declared], null)[1].sessionOrigin.surface, 'codex-ide', + 'first declaration supplies a parent even when the optional ledger is absent'); + const a = { ...declared, id: 'a', parentSessionId: 'b' }; + const b = { ...parent, id: 'b', threadSource: null, parentSessionId: 'a' }; + assert.equal(applyCodexLedger([a, b], null)[0].parentSessionId, null, + 'cyclic parent declarations supply no rollup evidence'); +}); + +test('Auto-review tokens remain counted but the unpublished model is unpriced', () => { + const row = { model: 'codex-auto-review', provider: 'openai', day: '2026-09-29', + input: 100, output: 20, cacheRead: 0, cacheWrite: 0, responses: 1 }; + const evidence = rowCostEvidence(row, { provider: 'codex' }, { costOf }); + assert.equal(evidence.estimatedUsd, 0); + assert.equal(evidence.unpricedMessages, 1); + assert.equal(rowCostEvidence({ ...row, model: 'gpt-5.6-sol' }, { provider: 'codex' }, { costOf }).unpricedMessages, 0); +}); + +test('ledger enrichment cannot relabel an imported copy as a reviewer', () => { + const copy = new Rollout({ id: 'imported' }).meta({ originator: 'Codex Desktop', thread_source: null }) + .taskStarted('external-import-turn-1').user('synthetic copied text'); + const rec = parseCodex(copy.toString(), { id: 'imported' }).session; + const ledger = { threads: new Map([['imported', { threadSource: 'guardian_review' }]]), parents: new Map() }; + const enriched = applyCodexLedger([rec], ledger)[0]; + assert.equal(enriched.sessionOrigin.initiator, 'imported-copy'); + assert.equal(enriched.sessionOrigin.evidence, 'imported-copy'); + assert.deepEqual(enriched.sessionOrigin.rawEvidence, {}); + assert.deepEqual(enriched.usage, []); +}); + +test('cold and warm aggregates retain child own usage and unpriced reviewer coverage under parent surface', async () => { + const parentId = '123e4567-e89b-42d3-a456-426614174000'; + const childId = '123e4567-e89b-42d3-a456-426614174001'; + const reviewId = '123e4567-e89b-42d3-a456-426614174002'; + const parent = new Rollout({ id: parentId }).meta({ originator: 'codex_vscode' }) + .turn('gpt-5.6-sol').user().agent().tokenCount(usage({ input: 100, output: 20 })); + const child = new Rollout({ id: childId }).meta({ originator: 'Codex Desktop', thread_source: 'subagent', + source: { subagent: { thread_spawn: { parent_thread_id: parentId, depth: 1 } } } }) + .turn('gpt-5.6-sol').agent().tokenCount(usage({ input: 50, output: 20 })); + const reviewer = new Rollout({ id: reviewId }).meta({ originator: 'Codex Desktop', thread_source: 'guardian_review', + source: { subagent: { other: 'guardian' } } }) + .turn('codex-auto-review').agent().tokenCount(usage({ input: 25, output: 5 })); + const sb = codexSandbox({ + 'rollout-2026-07-24T09-00-00-parent.jsonl': parent, + 'rollout-2026-07-24T09-01-00-child.jsonl': child, + 'rollout-2026-07-24T09-02-00-review.jsonl': reviewer, + }); + const options = { days: 14, now: Date.parse('2026-07-25T12:00:00Z'), roots: sb.roots, + cachePath: sb.cachePath, codexState: { threads: new Map(), parents: new Map([[reviewId, parentId]]) }, + deps: { ...stubDeps(), costOf } }; + _resetForTest(); + const cold = await buildIndex(options); + _resetForTest(); + const warm = await buildIndex(options); + for (const agg of [cold, warm]) { + assert.equal(agg.totals.tokens, 220); + assert.equal(agg.totals.humanPrompts, 1); + assert.equal(agg.bySource.subagent.tokens, 100); + assert.equal(agg.sessions.find((s) => s.id === childId).tokens, 70); + const review = agg.sessions.find((s) => s.id === reviewId); + assert.equal(review.tokens, 30); + assert.equal(review.cost, 0); + assert.equal(review.costEvidence.unpricedMessages, 1); + assert.equal(review.sessionOrigin.surface, 'codex-ide'); + } + assert.deepEqual(warm.totals, cold.totals); + assert.deepEqual(warm.sessions.map((s) => s.sessionOrigin), cold.sessions.map((s) => s.sessionOrigin)); +}); diff --git a/tests/kit/usage-codex-zero-response.test.mjs b/tests/kit/usage-codex-zero-response.test.mjs new file mode 100644 index 00000000..d5465f7b --- /dev/null +++ b/tests/kit/usage-codex-zero-response.test.mjs @@ -0,0 +1,98 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { parseCodex } from '../../src/lib/usage-parsers.mjs'; +import { aggregate } from '../../src/lib/usage-aggregate.mjs'; +import { buildIndex, _resetForTest } from '../../src/lib/usage-index.mjs'; +import { Rollout, usage, codexSandbox, stubDeps } from './helpers/codex-rollout.mjs'; + +const now = Date.parse('2026-07-25T12:00:00Z'); +const native = () => new Rollout({ id: 'tool-only' }).meta().taskStarted('native-1').turn() + .item('CommandExecution').tokenCount(usage({ input: 120, cached: 20, output: 7 })) + .raw('event_msg', { type: 'turn_aborted' }); +const totalOnly = () => new Rollout({ id: 'total-only' }).meta().taskStarted('native-1').turn() + .raw('event_msg', { type: 'token_count', info: { + total_token_usage: { total_tokens: 127 }, last_token_usage: { total_tokens: 127 }, + } }); + +test('native tool-only component usage counts with zero normalized responses', () => { + const parsed = parseCodex(String(native()), { id: 'fallback' }); + assert.equal(parsed.session.responses, 0); + assert.deepEqual(parsed.session.usage.map(({ input, cacheRead, output }) => ({ input, cacheRead, output })), + [{ input: 100, cacheRead: 20, output: 7 }]); + const result = aggregate([parsed.session], { days: 14, now, cutoff: now - 14 * 86400000, deps: stubDeps() }); + assert.equal(result.totals.sessions, 1); + assert.equal(result.totals.responses, 0); + assert.equal(result.totals.input, 100); + assert.equal(result.totals.cacheRead, 20); + assert.equal(result.totals.output, 7); + assert.equal(result.totals.tokens, 127); +}); + +test('cold and warm cache count component usage without responses and retain missing-breakdown diagnostics', async () => { + const sb = codexSandbox({ 'rollout-native.jsonl': native(), 'rollout-total.jsonl': totalOnly() }); + const options = { days: 14, now, roots: sb.roots, cachePath: sb.cachePath, deps: stubDeps() }; + for (let i = 0; i < 2; i++) { + _resetForTest(); + const result = await buildIndex(options); + assert.equal(result.totals.sessions, 1); + assert.equal(result.totals.responses, 0); + assert.equal(result.totals.tokens, 127); + assert.equal(result.sourceHealth.codex.diagnostics.zeroResponseUsageFiles, 1); + assert.equal(result.sourceHealth.codex.diagnostics.totalOnlyTokenCountEvents, 1); + assert.equal(result.sourceHealth.codex.diagnostics.zeroResponseUnsupportedFiles, 1); + assert.ok(result.sourceHealth.codex.diagnostics.warnings.includes('total-only-token-count')); + assert.equal(result.sourceHealth.codex.status, 'degraded'); + if (i) assert.equal(result.sourceHealth.codex.diagnostics.cachedFiles, 2); + } +}); + +test('total-only counter cannot create a priced row or fabricated components', () => { + const parsed = parseCodex(String(totalOnly()), { id: 'fallback' }); + assert.equal(parsed.session.responses, 0); + assert.deepEqual(parsed.session.usage, []); + assert.equal(parsed.parseStats.totalOnlyTokenCountEvents, 1); + const result = aggregate([parsed.session], { days: 14, now, cutoff: now - 14 * 86400000, deps: stubDeps() }); + assert.equal(result.totals.sessions, 0); + assert.equal(result.totals.tokens, 0); + assert.equal(result.totals.cost, 0); +}); + +test('a total-only gap cannot make a later component snapshot double count earlier usage', () => { + const r = native(); + r.raw('event_msg', { type: 'token_count', info: { total_token_usage: { total_tokens: 200 } } }); + r.tokenCount(usage({ input: 50, output: 5 })); + const parsed = parseCodex(String(r), { id: 'fallback' }); + assert.equal(parsed.parseStats.totalOnlyTokenCountEvents, 1); + assert.deepEqual(parsed.session.usage.map(({ input, cacheRead, output }) => ({ input, cacheRead, output })), + [{ input: 100, cacheRead: 20, output: 7 }]); +}); + +test('import copies and incomplete mixed ownership cannot become zero-response billable sessions', async () => { + const copy = new Rollout({ id: 'copy' }).meta({ originator: 'Codex Desktop' }) + .taskStarted('external-import-turn-1').tokenCount(usage({ input: 500 })); + const mixed = new Rollout({ id: 'mixed' }).meta({ originator: 'Codex Desktop' }) + .taskStarted('external-import-turn-1').tokenCount(usage({ input: 500 })) + .taskStarted('native-1').tokenCount(usage({ input: 50 })); + mixed.lines.push('{broken-boundary'); + const sb = codexSandbox({ 'rollout-copy.jsonl': copy, 'rollout-mixed.jsonl': mixed }); + const result = await buildIndex({ days: 14, now, roots: sb.roots, cachePath: sb.cachePath, deps: stubDeps() }); + assert.equal(result.totals.sessions, 0); + assert.equal(result.totals.tokens, 0); + assert.equal(result.sourceHealth.codex.diagnostics.importedExcluded, 2); + assert.equal(result.sourceHealth.codex.diagnostics.importOwnershipIncompleteFiles, 1); +}); + +test('an imported total-only baseline cannot bill the next native cumulative snapshot', () => { + const r = new Rollout({ id: 'mixed-total' }).meta({ originator: 'Codex Desktop' }) + .taskStarted('external-import-turn-1') + .raw('event_msg', { type: 'token_count', info: { total_token_usage: { total_tokens: 500 } } }) + .taskStarted('native-1').turn('gpt-5.6', { turn_id: 'native-1' }) + .raw('event_msg', { type: 'token_count', info: { + total_token_usage: usage({ input: 550, output: 5 }), + last_token_usage: usage({ input: 50, output: 5 }), + } }); + const parsed = parseCodex(String(r), { id: 'fallback' }); + assert.equal(parsed.session.responses, 0); + assert.deepEqual(parsed.session.usage, []); + assert.equal(parsed.session.importEvidence.ownershipComplete, true); +}); diff --git a/tests/kit/usage-index-opencode-source.test.mjs b/tests/kit/usage-index-opencode-source.test.mjs new file mode 100644 index 00000000..344f9e59 --- /dev/null +++ b/tests/kit/usage-index-opencode-source.test.mjs @@ -0,0 +1,265 @@ +// usage-index × opencode — the third transcript source through scan(), +// aggregate(), and readSession(). Hermetic: fixture claude/codex corpora and a +// fixture opencode.db, all in tmp; injected pricing/classification stubs so +// the arithmetic is exact. The real stores are never touched (the roots seam +// is also what is under test: overridden roots must NOT read the real db). +import { test, after } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { DatabaseSync } from 'node:sqlite'; + +const NOW = Date.parse('2026-07-29T12:00:00Z'); +const DAY = 86_400_000; +const tmp = (p) => fs.mkdtempSync(path.join(os.tmpdir(), p)); +const rm = (d) => fs.rmSync(d, { recursive: true, force: true }); + +const { sandboxHome } = await import('./helpers/home-sandbox.mjs'); +const testHome = sandboxHome('ak-oc-source'); +after(() => rm(testHome)); + +const { buildIndex, readIndex, readSession, _resetForTest } = await import('../../src/lib/usage-index.mjs'); + +/** Pricing stub: prices EVERY token at 1/1000 — deliberately different from + * the fixture's observed costs so the preference is provable. */ +const deps = () => ({ + costOf: ({ input, output, cacheRead, cacheWrite }) => (input + output + cacheRead + cacheWrite) / 1000, + pricesAsOf: '2026-07-01', + classify: ({ title }) => (title + ? { category: 'Build', confidence: 0.9, basis: 'title+tools' } + : { category: 'Unclassified', confidence: 0, basis: 'no signal' }), + detectInsights: () => [], +}); + +const assistantMsg = (id, sessionId, at, { model = 'kimi-k3', provider = 'opencode', cost = null, tokens = {} } = {}) => ({ + id, sessionId, at, + data: { + role: 'assistant', agent: 'build', modelID: model, providerID: provider, + tokens: { input: 1000, output: 100, reasoning: 10, cache: { read: 200, write: 10 }, ...tokens }, + ...(cost != null ? { cost } : {}), + time: { created: at, completed: at + 1000 }, finish: 'stop', + }, +}); + +function buildDb(file, { sessions = [], messages = [] } = {}) { + const db = new DatabaseSync(file); + db.exec(` + CREATE TABLE session (id text PRIMARY KEY, project_id text NOT NULL, workspace_id text, + parent_id text, slug text NOT NULL, directory text NOT NULL, path text, title text NOT NULL, + version text NOT NULL, share_url text, summary_additions integer, summary_deletions integer, + summary_files integer, summary_diffs text, metadata text, cost real DEFAULT 0 NOT NULL, + tokens_input integer DEFAULT 0 NOT NULL, tokens_output integer DEFAULT 0 NOT NULL, + tokens_reasoning integer DEFAULT 0 NOT NULL, tokens_cache_read integer DEFAULT 0 NOT NULL, + tokens_cache_write integer DEFAULT 0 NOT NULL, revert text, permission text, agent text, + model text, time_created integer NOT NULL, time_updated integer NOT NULL, + time_compacting integer, time_archived integer); + CREATE TABLE message (id text PRIMARY KEY, session_id text NOT NULL, + time_created integer NOT NULL, time_updated integer NOT NULL, data text NOT NULL); + CREATE INDEX message_session_time_created_id_idx ON message (session_id, time_created, id); + CREATE TABLE part (id text PRIMARY KEY, message_id text NOT NULL, session_id text NOT NULL, + time_created integer NOT NULL, time_updated integer NOT NULL, data text NOT NULL); + `); + const insS = db.prepare('INSERT INTO session (id, project_id, parent_id, slug, directory, title, version, time_created, time_updated) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)'); + const insM = db.prepare('INSERT INTO message (id, session_id, time_created, time_updated, data) VALUES (?, ?, ?, ?, ?)'); + for (const s of sessions) insS.run(s.id, 'proj-1', s.parentId ?? null, 'slug-x', s.directory, s.title, '1.18.8', s.timeCreated ?? NOW - DAY, s.timeUpdated ?? NOW - DAY); + for (const m of messages) insM.run(m.id, m.sessionId, m.at, m.at, JSON.stringify(m.data)); + db.close(); + return file; +} + +/** A sandbox: empty claude/codex corpora + a fixture opencode.db + cache path. */ +function sandbox({ sessions = [], messages = [] } = {}) { + const dir = tmp('ak-uio-'); + fs.mkdirSync(path.join(dir, 'corpus', 'claude'), { recursive: true }); + fs.mkdirSync(path.join(dir, 'corpus', 'codex'), { recursive: true }); + const dbFile = buildDb(path.join(dir, 'corpus', 'opencode.db'), { sessions, messages }); + return { + dir, dbFile, + roots: { + claude: path.join(dir, 'corpus', 'claude'), + codex: path.join(dir, 'corpus', 'codex'), + opencode: dbFile, + }, + cachePath: path.join(dir, 'cache', 'usage-index.json'), + }; +} + +const opts = (sb, extra = {}) => ({ days: 14, now: NOW, roots: sb.roots, cachePath: sb.cachePath, deps: deps(), ...extra }); + +// O9: equal session IDs and stamps in separate stores must never share parses. +test('OpenCode database switches isolate warm and degraded cache entries', async () => { + const at = NOW - DAY; + const data = (title, cost) => ({ sessions: [{ id: 'ses_shared', directory: '/x', title, timeCreated: at }], + messages: [assistantMsg('a1', 'ses_shared', at + 1000, { cost })] }); + const sb = sandbox(data('first', 0.2)); + try { + await buildIndex(opts(sb)); + const second = buildDb(path.join(sb.dir, 'second.db'), data('second', 0.8)); + const options = opts(sb, { roots: { ...sb.roots, opencode: second } }); + const switched = await buildIndex(options); + assert.equal(switched.sessions[0].title, 'second'); + assert.equal(switched.totals.cost, 0.8); + assert.equal((await readSession('ses_shared', options)).meta.title, 'second'); + fs.writeFileSync(sb.dbFile, 'corrupt'); + const degraded = await buildIndex(opts(sb)); + assert.equal(degraded.sessions.length, 0, 'second database cannot supply first database fallback'); + } finally { _resetForTest(); rm(sb.dir); } +}); + +test('OpenCode legacy source identity reparses and cannot carry through a degraded store', async () => { + const at = NOW - DAY; + const sb = sandbox({ sessions: [{ id: 'ses_identity', directory: '/x', title: 'real' }], + messages: [assistantMsg('a1', 'ses_identity', at, { cost: 0.2 })] }); + try { + await buildIndex(opts(sb)); + const cache = JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')); + delete cache.entries['opencode://ses_identity'].sourceIdentity; + cache.entries['opencode://ses_identity'].session.title = 'old'; + fs.writeFileSync(sb.cachePath, JSON.stringify(cache)); _resetForTest(); + assert.equal((await buildIndex(opts(sb))).sessions[0].title, 'real'); + fs.writeFileSync(sb.cachePath, JSON.stringify(cache)); + fs.writeFileSync(sb.dbFile, 'corrupt'); _resetForTest(); + assert.equal((await buildIndex(opts(sb))).sessions.length, 0); + } finally { _resetForTest(); rm(sb.dir); } +}); + +test('ambiguous default stores expose health, drop cache fallback, and agree with selected-session reads', async () => { + const sb = sandbox({ sessions: [{ id: 'ses_choice', directory: '/x', title: 'chosen' }], + messages: [assistantMsg('a1', 'ses_choice', NOW - DAY, { cost: 0.4 })] }); + const keys = ['XDG_DATA_HOME', 'OPENCODE_DB', 'OPENCODE_DISABLE_CHANNEL_DB']; + const before = Object.fromEntries(keys.map((key) => [key, process.env[key]])); + try { + process.env.XDG_DATA_HOME = sb.dir; + delete process.env.OPENCODE_DB; delete process.env.OPENCODE_DISABLE_CHANNEL_DB; + const root = path.join(sb.dir, 'opencode'); fs.mkdirSync(root); + fs.copyFileSync(sb.dbFile, path.join(root, 'opencode-preview.db')); + const options = opts(sb, { roots: undefined }); + assert.equal((await readIndex(options)).sessions[0].title, 'chosen'); + assert.equal((await readSession('ses_choice', options)).meta.title, 'chosen'); + fs.copyFileSync(sb.dbFile, path.join(root, 'opencode.db')); + const ambiguous = await readIndex(options); + assert.equal(ambiguous.sourceHealth.opencode.reason, 'database-selection-ambiguous'); + assert.equal(ambiguous.sessions.length, 0); + assert.equal(await readSession('ses_choice', options), null); + const { discoverProjectSources } = await import('../../src/lib/footprint/project-sources.mjs'); + const projects = discoverProjectSources({ claudeRoot: sb.roots.claude, codexRoot: sb.roots.codex }); + assert.equal(projects.sources.opencode.reason, 'database-selection-ambiguous'); + assert.equal(projects.complete, false); + process.env.OPENCODE_DB = 'opencode-preview.db'; + assert.equal((await readIndex(options)).sessions[0].title, 'chosen'); + assert.equal((await readSession('ses_choice', options)).meta.title, 'chosen'); + process.env.OPENCODE_DB = ':memory:'; + assert.equal((await buildIndex(options)).sourceHealth.opencode.reason, 'database-in-memory'); + assert.equal(await readSession('ses_choice', options), null); + assert.equal(fs.existsSync(path.join(root, ':memory:')), false); + assert.equal((await buildIndex(opts(sb))).sessions[0].title, 'chosen', 'explicit roots ignore environment'); + assert.equal((await buildIndex(opts(sb, { roots: {} }))).sessions.length, 0); + } finally { + for (const key of keys) { if (before[key] === undefined) delete process.env[key]; else process.env[key] = before[key]; } + _resetForTest(); rm(sb.dir); + } +}); + +// O10: coverage belongs to the source, independent of V1 candidate/cache yield. +for (const state of ['missing', 'empty', 'present', 'unknown']) { + test(`OpenCode unsupported V2 ${state} is observed with zero V1 candidates`, async () => { + const sb = sandbox(); + try { + const db = new DatabaseSync(sb.dbFile); + if (state === 'unknown') db.exec('CREATE VIEW session_message AS SELECT 1'); + else if (state !== 'missing') { + db.exec('CREATE TABLE session_message (payload text)'); + if (state === 'present') db.exec("INSERT INTO session_message VALUES ('private content')"); + } + db.close(); + const result = await buildIndex(opts(sb)); + const health = result.sourceHealth.opencode; + assert.equal(health.status, 'ok'); + assert.equal(health.storageCoverage.v2.status, state); + assert.deepEqual(health.diagnostics.common.warnings, state === 'present' + ? ['opencode-v2-session-message-present'] : state === 'unknown' + ? ['opencode-v2-observation-incomplete'] : []); + assert.equal(result.sessions.length, 0); + assert.equal(JSON.stringify(health).includes('private content'), false); + assert.equal(JSON.stringify(health).includes(sb.dir), false); + } finally { _resetForTest(); rm(sb.dir); } + }); +} + +test('OpenCode storage warnings refresh on all cache hits and survive an empty window without changing usage', async () => { + const sb = sandbox({ sessions: [{ id: 'ses_coverage', directory: '/x', title: 'known' }], + messages: [assistantMsg('a1', 'ses_coverage', NOW - DAY, { cost: 0.2 })] }); + try { + const first = await buildIndex(opts(sb)); + const cache = JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')); + cache.entries['opencode://ses_coverage'].session.title = 'cache witness'; + fs.writeFileSync(sb.cachePath, JSON.stringify(cache)); _resetForTest(); + const db = new DatabaseSync(sb.dbFile); + // Source-level V2 activity elsewhere must not invalidate this unchanged V1 session. + db.exec("CREATE TABLE session_message (session_id text, payload text); INSERT INTO session_message VALUES ('unrelated', NULL)"); db.close(); + fs.mkdirSync(path.join(path.dirname(sb.dbFile), 'storage')); + fs.writeFileSync(path.join(path.dirname(sb.dbFile), 'storage', 'legacy.json'), 'not read'); + const warm = await buildIndex(opts(sb)); + assert.equal(warm.sessions[0].title, 'cache witness', 'all V1 parses came from cache'); + assert.deepEqual(warm.totals, first.totals); + const expected = ['opencode-v2-session-message-present', 'opencode-legacy-json-present']; + assert.deepEqual(warm.sourceHealth.opencode.diagnostics.common.warnings, expected); + const empty = await buildIndex(opts(sb, { now: NOW + 100 * DAY })); + assert.equal(empty.sessions.length, 0); + assert.deepEqual(empty.sourceHealth.opencode.diagnostics.common.warnings, expected); + } finally { _resetForTest(); rm(sb.dir); } +}); + +for (const condition of ['missing', 'corrupt', 'v1-missing', 'legacy-inaccessible']) { + test(`OpenCode ${condition} preserves independent legacy coverage and availability`, async () => { + const sb = sandbox(); + try { + const legacy = path.join(path.dirname(sb.dbFile), 'storage'); + if (condition === 'legacy-inaccessible') fs.writeFileSync(legacy, 'not a directory'); + else { fs.mkdirSync(legacy); fs.writeFileSync(path.join(legacy, 'record.json'), 'private'); } + if (condition === 'missing') fs.unlinkSync(sb.dbFile); + if (condition === 'corrupt') fs.writeFileSync(sb.dbFile, 'invalid'); + if (condition === 'v1-missing') { + const db = new DatabaseSync(sb.dbFile); + db.exec('DROP TABLE message; CREATE TABLE session_message (payload text); INSERT INTO session_message VALUES (NULL)'); db.close(); + } + const health = (await buildIndex(opts(sb))).sourceHealth.opencode; + assert.equal(health.status, condition === 'missing' ? 'absent' : condition === 'legacy-inaccessible' ? 'ok' : 'degraded'); + assert.equal(health.storageCoverage.legacy.status, condition === 'legacy-inaccessible' ? 'unknown' : 'present'); + assert.ok(health.diagnostics.common.warnings.includes(condition === 'legacy-inaccessible' + ? 'opencode-legacy-observation-incomplete' : 'opencode-legacy-json-present')); + if (condition === 'corrupt') assert.equal(health.storageCoverage.v2.status, 'unknown'); + if (condition === 'v1-missing') assert.ok(health.diagnostics.common.warnings.includes('opencode-v2-session-message-present')); + assert.equal(fs.existsSync(sb.dbFile), condition !== 'missing', 'missing DB stays missing'); + const isolated = (await buildIndex(opts(sb, { roots: {} }))).sourceHealth.opencode; + assert.equal(isolated.storageCoverage.legacy.status, 'not-observed'); + assert.deepEqual(isolated.diagnostics.common.warnings, []); + } finally { _resetForTest(); rm(sb.dir); } + }); +} + +test('OpenCode default legacy root stays observable during ambiguous database selection', async () => { + const sb = sandbox(); + const keys = ['XDG_DATA_HOME', 'OPENCODE_DB', 'OPENCODE_DISABLE_CHANNEL_DB']; + const before = Object.fromEntries(keys.map((key) => [key, process.env[key]])); + try { + process.env.XDG_DATA_HOME = sb.dir; + delete process.env.OPENCODE_DB; delete process.env.OPENCODE_DISABLE_CHANNEL_DB; + const root = path.join(sb.dir, 'opencode'); fs.mkdirSync(path.join(root, 'storage'), { recursive: true }); + fs.writeFileSync(path.join(root, 'storage', 'legacy.json'), 'never parsed'); + fs.copyFileSync(sb.dbFile, path.join(root, 'opencode.db')); + fs.copyFileSync(sb.dbFile, path.join(root, 'opencode-preview.db')); + const health = (await buildIndex(opts(sb, { roots: undefined }))).sourceHealth.opencode; + assert.equal(health.reason, 'database-selection-ambiguous'); + assert.equal(health.storageCoverage.v2.status, 'not-observed'); + assert.ok(health.diagnostics.common.warnings.includes('opencode-legacy-json-present')); + assert.deepEqual((await buildIndex(opts(sb))).sourceHealth.opencode.diagnostics.common.warnings, [], 'explicit source never observes the global legacy root'); + process.env.OPENCODE_DB = sb.dbFile; + const overridden = (await buildIndex(opts(sb, { roots: undefined }))).sourceHealth.opencode; + assert.ok(overridden.diagnostics.common.warnings.includes('opencode-legacy-json-present'), 'DB override does not relocate upstream legacy storage'); + } finally { + for (const key of keys) { if (before[key] === undefined) delete process.env[key]; else process.env[key] = before[key]; } + _resetForTest(); rm(sb.dir); + } +}); diff --git a/tests/kit/usage-index-opencode.test.mjs b/tests/kit/usage-index-opencode.test.mjs index 9bb4f4ad..1e8ea9ad 100644 --- a/tests/kit/usage-index-opencode.test.mjs +++ b/tests/kit/usage-index-opencode.test.mjs @@ -86,6 +86,66 @@ function sandbox({ sessions = [], messages = [] } = {}) { const opts = (sb, extra = {}) => ({ days: 14, now: NOW, roots: sb.roots, cachePath: sb.cachePath, deps: deps(), ...extra }); +test('cached OpenCode child fingerprints cannot enter current or previous prompt projections', async () => { + const current = NOW - DAY; + const prior = NOW - 15 * DAY; + const sb = sandbox({ + sessions: [ + { id: 'parent', directory: '/x', title: 'parent', timeCreated: current }, + { id: 'child', directory: '/x', title: 'child', parentId: 'parent', timeCreated: current }, + { id: 'prior-child', directory: '/x', title: 'prior child', parentId: 'parent', timeCreated: prior }, + ], + messages: [ + userMsg('pu', 'parent', current), assistantMsg('pa', 'parent', current + 1000, { cost: 0.2 }), + userMsg('cu', 'child', current), assistantMsg('ca', 'child', current + 1000, { provider: 'alpha', cost: 0.3 }), + userMsg('ou', 'prior-child', prior), assistantMsg('oa', 'prior-child', prior + 1000, { cost: 0.4 }), + ], + }); + try { + const db = new DatabaseSync(sb.dbFile); + const insert = db.prepare('INSERT INTO part (id, message_id, session_id, time_created, time_updated, data) VALUES (?, ?, ?, ?, ?, ?)'); + for (const [id, messageId, sessionId, at] of [ + ['p1', 'pu', 'parent', current], ['p2', 'cu', 'child', current], ['p3', 'ou', 'prior-child', prior], + ]) insert.run(id, messageId, sessionId, at, at, JSON.stringify({ type: 'text', text: 'Run the tests' })); + db.close(); + const options = opts(sb, { lookbackDays: 28, previous: true, prompts: true }); + const cold = await buildIndex(options); + assert.equal(cold.totals.typedPrompts, 1); + assert.equal(cold.totals.humanPrompts, 1); + assert.equal(cold.totals.cost, 0.5); + assert.equal(cold.sessions.find((s) => s.id === 'child').typedPrompts, 0); + assert.equal(cold.previous.totals.typedPrompts, 0); + assert.deepEqual(cold.promptPatterns.corpus, { fingerprints: 1, typed: 1 }); + assert.deepEqual(Object.keys(cold.promptBaselines), [], 'a prior child alone does not define an operator baseline'); + const selected = await readSession('child', { roots: sb.roots, deps: deps() }); + assert.equal(selected.meta.sidechain, true); + assert.equal(selected.meta.cost, 0.3); + assert.equal(selected.turns[0].text, 'Run the tests'); + const cache = JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')); + const parentFP = cache.entries['opencode://parent'].session.promptFPs[0]; + for (const id of ['child', 'prior-child']) { + cache.entries[`opencode://${id}`].session.promptFPs = [parentFP]; + } + fs.writeFileSync(sb.cachePath, JSON.stringify(cache)); + _resetForTest(); + const warm = await buildIndex(options); + assert.equal(warm.totals.typedPrompts, 1); + assert.equal(warm.sessions.find((s) => s.id === 'child').typedPrompts, 0); + assert.equal(warm.previous.totals.typedPrompts, 0); + assert.deepEqual(warm.promptPatterns.corpus, { fingerprints: 1, typed: 1 }); + assert.deepEqual(warm.promptPatterns.exactRepeats, []); + assert.deepEqual(Object.keys(warm.promptBaselines), []); + assert.equal(warm.totals.cost, 0.5); + assert.equal(warm.byProvider.alpha.cost, 0.3); + assert.equal(warm.byProvider.alpha.tokens, 1320); + assert.equal(warm.byProvider.opencode.cost, 0.2); + assert.equal(warm.totals.tokens, 2640); + assert.equal(warm.sessions.find((s) => s.id === 'child').threadSource, 'subagent'); + assert.equal(JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')).entries['opencode://child'].session.promptFPs.length, 1, + 'the warm scan reused the old cached record, so aggregation must defend itself'); + } finally { _resetForTest(); rm(sb.dir); } +}); + test('scan aggregates opencode sessions: host bucket, provider bucket, tokens, and OBSERVED cost preferred over the pricing stub', async () => { const at = NOW - DAY; const sb = sandbox({ @@ -113,11 +173,108 @@ test('scan aggregates opencode sessions: host bucket, provider bucket, tokens, a assert.ok(agg.byHost.opencode, 'byHost gains the opencode bucket'); assert.equal(agg.byHost.opencode.cost, 0.5); assert.ok(agg.byProvider.opencode, 'byProvider gains the observed provider bucket'); + assert.ok(s.minutes > 0, 'the fixture has measured duration'); + assert.equal(agg.byProvider.opencode.minutes, s.minutes, 'single-provider duration follows the session'); + assert.equal(agg.byProvider.opencode.confidence, 0.9, 'classifier confidence survives provider folding'); assert.equal(agg.totals.cost, 0.5); assert.equal(agg.byModel['kimi-k3'].cost, 0.5); rm(sb.dir); }); +test('one OpenCode session partitions provider usage on cold and warm scans without changing global totals', async () => { + const at = NOW - DAY; + const sb = sandbox({ + sessions: [{ id: 'ses_switch', directory: '/x', title: 'switch', timeCreated: at }], + messages: [ + assistantMsg('a1', 'ses_switch', at + 1000, { model: 'shared', provider: 'alpha', cost: 0.25, + tokens: { input: 10, output: 2, reasoning: 0, cache: { read: 3, write: 1 } } }), + assistantMsg('a2', 'ses_switch', at + 2000, { model: 'shared', provider: 'beta', + tokens: { input: 20, output: 4, reasoning: 0, cache: { read: 5, write: 2 } } }), + assistantMsg('a3', 'ses_switch', at + 3000, { model: 'shared', provider: 'alpha', cost: 0, + tokens: { input: 0, output: 0, reasoning: 0, cache: { read: 0, write: 0 } } }), + ], + }); + try { + const cold = await buildIndex(opts(sb)); + const cache = JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')); + for (const entry of Object.values(cache.entries)) { + if (entry.session.host === 'opencode') { + entry.session.inferenceProvider = 'alpha'; // pre-repair v26 last-wins cache + entry.session.providerProvenance = 'observed'; + } + } + fs.writeFileSync(sb.cachePath, JSON.stringify(cache)); + _resetForTest(); + const warm = await buildIndex(opts(sb)); + const reused = JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')); + assert.equal(Object.values(reused.entries).find((entry) => entry.session.host === 'opencode') + .session.inferenceProvider, 'alpha', 'warm scan reused the old record'); + for (const agg of [cold, warm]) { + assert.equal(agg.sessions.length, 1); + assert.equal(agg.sessions[0].provider, null, 'mixed providers have no unique session provider'); + assert.equal(agg.sessions[0].providerProvenance, 'unknown'); + assert.equal(agg.totals.sessions, 1); + assert.equal(agg.totals.responses, 3); + assert.equal(agg.totals.tokens, 47); + assert.equal(agg.totals.cost, 0.281); + assert.deepEqual(agg.sessions[0].costEvidence, { + observedUsd: 0.25, estimatedUsd: 0.031, + observedMessages: 2, estimatedMessages: 1, unpricedMessages: 0, + }); + assert.equal(agg.byHost.opencode.sessions, 1); + assert.equal(agg.byHost.opencode.responses, 3); + assert.ok(agg.sessions[0].minutes > 0, 'the fixture has measured duration'); + assert.deepEqual(Object.keys(agg.byProvider).sort(), ['alpha', 'beta']); + for (const provider of ['alpha', 'beta']) { + assert.equal(agg.byProvider[provider].minutes, agg.sessions[0].minutes, + 'each provider session count carries the session duration'); + assert.equal(agg.byProvider[provider].confidence, 0.9, + 'each provider session count carries classifier confidence'); + } + assert.deepEqual( + ['sessions', 'responses', 'input', 'output', 'cacheRead', 'cacheWrite', 'tokens', 'cost'] + .map((key) => agg.byProvider.alpha[key]), + [1, 2, 10, 2, 3, 1, 16, 0.25], + ); + assert.deepEqual( + ['sessions', 'responses', 'input', 'output', 'cacheRead', 'cacheWrite', 'tokens', 'cost'] + .map((key) => agg.byProvider.beta[key]), + [1, 1, 20, 4, 5, 2, 31, 0.031], + ); + assert.equal(agg.byModel.shared.responses, 3); + assert.equal(agg.byModel.shared.tokens, 47); + } + } finally { rm(sb.dir); } +}); + +test('an unreported zero-token completion belongs to its observed provider and an absent provider stays unknown', async () => { + const at = NOW - DAY; + const sb = sandbox({ + sessions: [{ id: 'ses_unknown', directory: '/x', title: 'unknown', timeCreated: at }], + messages: [ + assistantMsg('a1', 'ses_unknown', at + 1000, { provider: 'alpha', cost: 0, + tokens: { input: 0, output: 0, reasoning: 0, cache: { read: 0, write: 0 } } }), + assistantMsg('a2', 'ses_unknown', at + 2000, { provider: null, cost: 0.1, + tokens: { input: 0, output: 0, reasoning: 0, cache: { read: 0, write: 0 } } }), + ], + }); + try { + const agg = await buildIndex(opts(sb)); + assert.equal(agg.sessions[0].provider, null); + assert.equal(agg.totals.responses, 2); + assert.equal(agg.totals.tokens, 0); + assert.equal(agg.totals.cost, 0.1); + assert.equal(agg.byHost.opencode.responses, 2); + assert.deepEqual(Object.keys(agg.byProvider).sort(), ['alpha', 'unknown']); + assert.equal(agg.byProvider.alpha.sessions, 1); + assert.equal(agg.byProvider.alpha.responses, 1); + assert.equal(agg.byProvider.alpha.tokens, 0); + assert.equal(agg.byProvider.unknown.sessions, 1); + assert.equal(agg.byProvider.unknown.responses, 1); + assert.equal(agg.byProvider.unknown.cost, 0.1); + } finally { rm(sb.dir); } +}); + test('sessions with NO observed cost fall back to the pricing table (never a fabricated $0)', async () => { const at = NOW - DAY; const sb = sandbox({ @@ -152,6 +309,96 @@ test('the incremental cache: a warm scan reuses unchanged sessions and picks up rm(sb.dir); }); +test('unchanged schema 26 OpenCode rows without parse semantics are reparsed, including mixed untrusted zero cost', async () => { + const at = NOW - DAY; + const sb = sandbox({ + sessions: [{ id: 'ses_legacy26', directory: '/x', title: 'real title', timeCreated: at }], + messages: [ + assistantMsg('a1', 'ses_legacy26', at + 1000, { cost: 0.4 }), + assistantMsg('a2', 'ses_legacy26', at + 2000, { cost: 0 }), + ], + }); + try { + const cold = await buildIndex(opts(sb)); + const file = `opencode://ses_legacy26`; + const cache = JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')); + assert.equal(cache.schemaVersion, 26); + assert.ok(cache.entries[file].parseSemantics, 'every new OpenCode entry identifies its parser semantics'); + delete cache.entries[file].parseSemantics; + cache.entries[file].session.title = 'FORGED-LEGACY'; + cache.entries[file].session.usage[0].costObserved = 0.4; + delete cache.entries[file].session.usage[0].costUntrustedMessages; + fs.writeFileSync(sb.cachePath, JSON.stringify(cache)); + _resetForTest(); + const repaired = await buildIndex(opts(sb)); + const session = repaired.sessions.find((s) => s.id === 'ses_legacy26'); + assert.equal(session.title, 'real title'); + assert.equal(session.responses, 2); + assert.equal(session.costEvidence.observedMessages, 1); + assert.equal(session.costEvidence.unpricedMessages, 1); + assert.equal(session.cost, cold.sessions[0].cost); + assert.equal(repaired.totals.cost, repaired.byHost.opencode.cost); + assert.equal(repaired.totals.cost, repaired.byProvider.opencode.cost); + assert.equal(repaired.totals.responses, repaired.byProvider.opencode.responses); + assert.equal(repaired.totals.tokens, repaired.byProvider.opencode.tokens); + assert.ok(JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')).entries[file].parseSemantics); + } finally { _resetForTest(); rm(sb.dir); } +}); + +test('schema 26 OpenCode rows lacking a marker reparse even when every cost was trusted; marked rows reuse warm', async () => { + const at = NOW - DAY; + const sb = sandbox({ sessions: [{ id: 'ses_trusted26', directory: '/x', title: 'real', timeCreated: at }], + messages: [assistantMsg('a1', 'ses_trusted26', at + 1000, { cost: 0.2 })] }); + try { + await buildIndex(opts(sb)); + const file = 'opencode://ses_trusted26'; + const cache = JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')); + delete cache.entries[file].parseSemantics; + cache.entries[file].session.title = 'OLD'; + fs.writeFileSync(sb.cachePath, JSON.stringify(cache)); + _resetForTest(); + await buildIndex(opts(sb)); + const reparsed = JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')); + assert.equal(reparsed.entries[file].session.title, 'real'); + reparsed.entries[file].session.title = 'WARM-MARKED'; + fs.writeFileSync(sb.cachePath, JSON.stringify(reparsed)); + _resetForTest(); + const warm = await buildIndex(opts(sb)); + assert.equal(warm.sessions[0].title, 'WARM-MARKED'); + } finally { _resetForTest(); rm(sb.dir); } +}); + +test('degraded OpenCode store excludes legacy cache accounting but retains compatible cache accounting', async () => { + const at = NOW - DAY; + const sb = sandbox({ sessions: [ + { id: 'ses_old', directory: '/x', title: 'old', timeCreated: at }, + { id: 'ses_new', directory: '/x', title: 'new', timeCreated: at }, + ], messages: [ + assistantMsg('a1', 'ses_old', at + 1000, { cost: 0.3 }), + assistantMsg('a2', 'ses_new', at + 1000, { cost: 0.4 }), + ] }); + try { + await buildIndex(opts(sb)); + const cache = JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')); + delete cache.entries['opencode://ses_old'].parseSemantics; + fs.writeFileSync(sb.cachePath, JSON.stringify(cache)); + fs.rmSync(sb.dbFile); + fs.writeFileSync(sb.dbFile, 'not a sqlite database'); + _resetForTest(); + const degraded = await buildIndex(opts(sb)); + assert.equal(degraded.sourceHealth.opencode.status, 'degraded'); + assert.equal(degraded.sourceHealth.opencode.reason, 'corrupt'); + assert.equal(degraded.sourceHealth.opencode.legacyCacheEntriesExcluded, 1); + assert.deepEqual(degraded.sessions.map((s) => s.id), ['ses_new']); + assert.equal(degraded.totals.cost, 0.4); + assert.equal(degraded.byProvider.opencode.cost, 0.4); + assert.equal(degraded.totals.responses, 1); + const retained = JSON.parse(fs.readFileSync(sb.cachePath, 'utf8')); + assert.ok(retained.entries['opencode://ses_old']); + assert.equal(retained.entries['opencode://ses_old'].parseSemantics, undefined); + } finally { _resetForTest(); rm(sb.dir); } +}); + test('a corrupt OpenCode store preserves last-good usage and surfaces degraded source health', async () => { const at = NOW - DAY; const sb = sandbox({ diff --git a/tests/kit/usage-index-v6.test.mjs b/tests/kit/usage-index-v6.test.mjs index ab71807d..6066bf83 100644 --- a/tests/kit/usage-index-v6.test.mjs +++ b/tests/kit/usage-index-v6.test.mjs @@ -138,9 +138,10 @@ test('parseCodex normalizes current item_completed messages and exposes bounded // genuinely new shapes. assert.equal(s.tools.CommandExecution, 1, 'tallied as the tool it is'); assert.deepEqual(agg.sourceHealth.codex.diagnostics, { - files: 1, cachedFiles: 0, parsedFiles: 1, unparsedFiles: 0, unparsedReasons: {}, importedExcluded: 0, - filesWithTokens: 1, filesWithResponses: 1, - legacyEvents: 0, itemCompletedEvents: 3, tokenCountEvents: 1, + files: 1, cachedFiles: 0, parsedFiles: 1, unparsedFiles: 0, unparsedReasons: {}, importedExcluded: 0, importedMixed: 0, importedTurnsExcluded: 0, importAmbiguousRecords: 0, + importOwnershipIncompleteFiles: 0, importedTurnCountIncompleteFiles: 0, + filesWithTokens: 1, filesWithResponses: 1, zeroResponseUsageFiles: 0, zeroResponseUnsupportedFiles: 0, + legacyEvents: 0, itemCompletedEvents: 3, tokenCountEvents: 1, totalOnlyTokenCountEvents: 0, prompts: 1, responses: 1, unknownItemTypes: {}, unknownItemTypeOverflow: 0, clippedLines: 0, warnings: [], common: { @@ -169,7 +170,7 @@ test('parseCodex accepts a mixed legacy/current rollout without dropping either assert.equal(agg.sourceHealth.codex.diagnostics.itemCompletedEvents, 3); }); -test('Codex source health degrades when token-bearing files yield zero normalized responses', async () => { +test('Codex source health counts component usage with zero normalized responses', async () => { _resetForTest(); const id = 'item-completed-zero'; const line = (o) => `${JSON.stringify(o)}\n`; @@ -180,13 +181,14 @@ test('Codex source health degrades when token-bearing files yield zero normalize } }); const sb = sandbox({ [`rollout-2026-07-24T09-00-00-${id}.jsonl`]: raw }); const agg = await buildIndex(opts(sb)); - assert.equal(agg.totals.sessions, 0); - assert.equal(agg.sourceHealth.codex.status, 'degraded'); - assert.equal(agg.sourceHealth.codex.reason, 'parse-yield-zero'); - assert.deepEqual(agg.sourceHealth.codex.diagnostics.warnings, ['zero-response-yield']); + assert.equal(agg.totals.sessions, 1); + assert.equal(agg.totals.tokens, 120); + assert.equal(agg.sourceHealth.codex.status, 'ok'); + assert.equal(agg.sourceHealth.codex.diagnostics.zeroResponseUsageFiles, 1); + assert.deepEqual(agg.sourceHealth.codex.diagnostics.warnings, []); }); -test('Codex source health exposes partial response yield across token-bearing files', async () => { +test('Codex source health counts a zero-response component file alongside responses', async () => { _resetForTest(); const id = 'item-completed-partial'; const line = (o) => `${JSON.stringify(o)}\n`; @@ -199,10 +201,9 @@ test('Codex source health exposes partial response yield across token-bearing fi 'rollout-2026-07-24T09-00-01-zero-yield.jsonl': zero, }); const agg = await buildIndex(opts(sb)); - assert.equal(agg.totals.sessions, 1); - assert.equal(agg.sourceHealth.codex.status, 'degraded'); - assert.equal(agg.sourceHealth.codex.reason, 'parse-yield-partial'); - assert.deepEqual(agg.sourceHealth.codex.diagnostics.warnings, ['partial-response-yield']); + assert.equal(agg.totals.sessions, 2); + assert.equal(agg.sourceHealth.codex.status, 'ok'); + assert.deepEqual(agg.sourceHealth.codex.diagnostics.warnings, []); assert.equal(agg.sourceHealth.codex.diagnostics.filesWithTokens, 2); assert.equal(agg.sourceHealth.codex.diagnostics.filesWithResponses, 1); }); diff --git a/tests/kit/usage-index.test.mjs b/tests/kit/usage-index.test.mjs index 07e32530..f67adbe9 100644 --- a/tests/kit/usage-index.test.mjs +++ b/tests/kit/usage-index.test.mjs @@ -947,12 +947,13 @@ test('an empty corpus yields a zeroed Aggregate rather than throwing', async () assert.equal(agg.sourceHealth.codex.diagnostics.files, 0); }); -test('buildIndex reports ok claude/codex root health when the transcript roots exist', async () => { +test('buildIndex reports malformed Claude coverage while Codex root health remains ok', async () => { _resetForTest(); const sb = sandbox(); const agg = await buildIndex(opts(sb)); - assert.equal(agg.sourceHealth.claude.status, 'ok'); - assert.equal(agg.sourceHealth.claude.reason, null); + assert.equal(agg.sourceHealth.claude.status, 'degraded'); + assert.equal(agg.sourceHealth.claude.reason, 'transcript-record-coverage-incomplete'); + assert.equal(agg.sourceHealth.claude.diagnostics.records.malformedRecords, 1); // Source health reports what was READ, never what the parser could report: // the old per-host capability matrix is not part of this payload. assert.ok(!Object.hasOwn(agg.sourceHealth.claude, 'capabilities'), @@ -1857,7 +1858,7 @@ test('blankSession v14 fingerprint fields default honest-empty', () => { assert.equal(rec.promptFPOverflow, 0); }); -// ── v11 index carry-through + lookback (Task 5) ───────────────────────────── +// ── v11 index carry-through + lookback ───────────────────────────── /** Carries BOTH v16 shape flags: it opens with a persona assignment and it ends * with a question mark. Shared by the fixture and its assertions so the two @@ -1958,7 +1959,7 @@ test('cached session entries round-trip the v11 and v14 fields across a cache hi assertBoth(); // cache-hit read: identical values prove the round trip }); -// ── lookback (Task 5) ──────────────────────────────────────────────────────── +// ── lookback ──────────────────────────────────────────────────────── test("buildIndex({ days, lookbackDays }) widens discovery/parse; unset stays exactly today's behavior", async () => { _resetForTest(); @@ -1985,7 +1986,7 @@ test("buildIndex({ days, lookbackDays }) widens discovery/parse; unset stays exa assert.equal(byId(undefinedLookback, 'old-session'), undefined, 'lookbackDays: undefined must not widen the window either'); assert.deepEqual(undefinedLookback.totals, plain.totals, 'unset lookbackDays is identical to omitting it entirely'); - // Task 7 (2026-08-28-scorecard-matrix-a SDD) ruling B: lookbackDays alone + // lookbackDays alone // no longer widens what the CURRENT window's sessions/totals contain — only // discovery/parse/cache. `aggregate` is always called with the DISPLAY // cutoff (now - days*DAY_MS), so a `previous: true`-less caller must see @@ -2029,7 +2030,7 @@ test('scanKey distinguishes calls that differ only by lookbackDays', async () => 'a call with lookbackDays set must not collide with one that omits it — they must not share the in-flight promise / result object'); }); -// Task 7 fix round 1: the same F-08-style regression guard as the lookbackDays +// The same F-08-style regression guard as the lookbackDays // test above, now for `previous` — added after review flagged that scanKey // folded lookbackDays into its identity but not previous, so a {previous:true} // caller (e.g. /api/usage) racing a {previous:false} caller (e.g. the Models @@ -2044,7 +2045,7 @@ test('scanKey distinguishes calls that differ only by previous', async () => { 'a call with previous:true must not collide with one that omits it — they must not share the in-flight promise / result object'); }); -// ── aggregate: buckets, rhythm, per-day engaged, previous window (Task 6) ─── +// ── aggregate: buckets, rhythm, per-day engaged, previous window ─── // // These suites drive `aggregate()` directly instead of through buildIndex: the // previous-window projection reads records OLDER than the cutoff, which only diff --git a/tests/kit/usage-limits-empty-state.test.mjs b/tests/kit/usage-limits-empty-state.test.mjs index 6d882681..ff880eaa 100644 --- a/tests/kit/usage-limits-empty-state.test.mjs +++ b/tests/kit/usage-limits-empty-state.test.mjs @@ -33,17 +33,17 @@ const text = (html) => html.replace(/<[^>]+>/g, '').replace(/—/g, '—').r test('a custom user-level statusLine is named, with the project precedence that still fills the panel', () => { const html = text(renderLimitsWith(empty({ claudeChannel: 'custom' }))['u-lim-claude'].innerHTML); - assert.match(html, /user-level statusLine runs a custom script/); + assert.match(html, /effective statusLine runs a custom script/); assert.match(html, /does not report limits to ak/); - assert.match(html, /project’s own statusLine takes precedence over your user-level one/); + assert.match(html, /local or managed settings may override the project and user settings/); assert.match(html, /ak setup --project/); assert.doesNotMatch(html, /Run one session, then revisit/, 'running more sessions is not the fix when the effective statusline cannot tee'); }); -test('no user-level statusLine says only footer-carrying projects report limits', () => { +test('no effective statusLine says only footer-carrying projects report limits', () => { const html = text(renderLimitsWith(empty({ claudeChannel: 'none' }))['u-lim-claude'].innerHTML); - assert.match(html, /no user-level statusLine/); + assert.match(html, /no effective statusLine/); assert.match(html, /ak setup --project/); }); @@ -117,7 +117,7 @@ test('a stale Codex answer served after a failed refresh says the refresh failed assert.doesNotMatch(fresh['u-lim-codex-note'].textContent, /refresh failed/); }); -// Fix round 1: a cached figure served under a presence-gated reason (no spawn +// a cached figure served under a presence-gated reason (no spawn // was ever attempted) must say "not refreshed", never "last refresh failed" — // that phrase implies an attempt that did not happen. test('a cached Codex figure served under a presence-gated reason says "not refreshed", never "failed"', () => { diff --git a/tests/kit/usage-opencode-compaction.test.mjs b/tests/kit/usage-opencode-compaction.test.mjs new file mode 100644 index 00000000..0d6e8c51 --- /dev/null +++ b/tests/kit/usage-opencode-compaction.test.mjs @@ -0,0 +1,236 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { DatabaseSync } from 'node:sqlite'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { parseSession } from '../../src/lib/usage-opencode.mjs'; +import { aggregate } from '../../src/lib/usage-aggregate.mjs'; +import { buildIndex, _resetForTest } from '../../src/lib/usage-index.mjs'; + +const NOW = Date.parse('2026-09-29T12:00:00Z'); +const TOKENS = { input: 100, output: 20, reasoning: 5, cache: { read: 40, write: 3 } }; +const assistant = (extra = {}) => ({ role: 'assistant', summary: true, parentID: 'u', finish: 'stop', + providerID: 'openrouter', modelID: 'test', tokens: TOKENS, cost: 0.25, + time: { created: NOW - 2000, completed: NOW - 1000 }, ...extra }); +function fixture(t, { messages = [['u', { role: 'user' }], ['a', assistant()]], + parts = [['u', { type: 'compaction' }], ['a', { type: 'step-finish', tokens: TOKENS, cost: 0.25 }]], + metadata = {} } = {}) { + const dir = tempDir('ak-oc-compaction-'); + t.after(() => { _resetForTest(); fs.rmSync(dir, { recursive: true, force: true }); }); + const dbFile = path.join(dir, 'opencode.db'); + const db = new DatabaseSync(dbFile); + db.exec(`CREATE TABLE session (id TEXT PRIMARY KEY, parent_id TEXT, directory TEXT, title TEXT, + version TEXT, time_created INTEGER, time_updated INTEGER, time_compacting INTEGER, + cost REAL, tokens_input INTEGER, tokens_output INTEGER, tokens_reasoning INTEGER, + tokens_cache_read INTEGER, tokens_cache_write INTEGER); + CREATE TABLE message (id TEXT PRIMARY KEY, session_id TEXT, time_created INTEGER, time_updated INTEGER, data TEXT); + CREATE TABLE part (id TEXT PRIMARY KEY, message_id TEXT, session_id TEXT, data TEXT);`); + db.prepare('INSERT INTO session VALUES (?,NULL,?,?,?,?,?,NULL,?,?,?,?,?,?)') + .run('s', dir, 'fixture', '1.18.33', NOW - 3000, NOW, 0.25, 100, 20, 5, 40, 3); + for (const [key, value] of Object.entries(metadata)) db.prepare(`UPDATE session SET ${key} = ?`).run(value); + for (const [id, data] of messages) db.prepare('INSERT INTO message VALUES (?,?,?,?,?)') + .run(id, 's', NOW - 2000, NOW, JSON.stringify(data)); + parts.forEach(([id, data], i) => db.prepare('INSERT INTO part VALUES (?,?,?,?)').run(`p${i}`, id, 's', JSON.stringify(data))); + db.close(); + const roots = { claude: path.join(dir, 'claude'), codex: path.join(dir, 'codex'), opencode: dbFile }; + fs.mkdirSync(roots.claude); fs.mkdirSync(roots.codex); + return { dbFile, id: 's', roots, cachePath: path.join(dir, 'cache.json'), days: 14, now: NOW, + deps: { costOf: () => 99, classify: () => ({ category: 'Build', confidence: 1 }), detectInsights: () => [] } }; +} + +test('linked completed compaction counts once on scan/detail without charging its parts', t => { + const opts = fixture(t, { parts: [['u', { type: 'compaction' }], ['u', { type: 'compaction' }], + ['a', { type: 'step-finish', tokens: TOKENS, cost: 0.25 }]] }); + for (const withTurns of [false, true]) { + const { session } = parseSession({ ...opts, withTurns }); + assert.equal(session.compactions, 1); + assert.deepEqual(session.compactionEvidence, { lowerBound: 1, upperBound: 1 }); + assert.equal(session.usage[0].input, 100); + assert.equal(session.usage[0].output, 25); + assert.equal(session.usage[0].costObserved, 0.25); + assert.equal(session.opencodeReconciliation.state, 'matched'); + } +}); + +for (const [name, data, parts, meta, bounds] of [ + ['failed', assistant({ error: { name: 'APIError' } }), [['u', { type: 'compaction' }]], {}, [0, 0]], + ['aborted', assistant({ error: { name: 'MessageAbortedError' } }), [['u', { type: 'compaction' }]], {}, [0, 0]], + ['inflight', assistant({ finish: undefined, time: { created: NOW - 2000 } }), [['u', { type: 'compaction' }]], { time_compacting: NOW }, [0, 1]], + ['orphan summary', assistant({ parentID: 'absent' }), [], {}, [0, 1]], + ['request only', { role: 'user' }, [['u', { type: 'compaction' }]], {}, [0, 1]], +]) test(`${name} cannot claim a completed compaction`, t => { + const { session } = parseSession(fixture(t, { messages: [['u', { role: 'user' }], ['a', data]], parts, metadata: meta })); + assert.equal(session.compactions, bounds[0]); + assert.deepEqual(session.compactionEvidence, { lowerBound: bounds[0], upperBound: bounds[1] }); +}); + +test('time_compacting alone is incomplete evidence, never completion', t => { + const { session } = parseSession(fixture(t, { parts: [], messages: [], metadata: { time_compacting: NOW } })); + assert.equal(session.compactions, 0); + assert.deepEqual(session.compactionEvidence, { lowerBound: 0, upperBound: null }); +}); + +for (const [name, changes, state, reason] of [ + ['mismatch', { metadata: { tokens_input: 101 } }, 'mismatch', null], + ['default zero', { metadata: { cost: 0, tokens_input: 0, tokens_output: 0, tokens_reasoning: 0, tokens_cache_read: 0, tokens_cache_write: 0 } }, 'unknown', 'unpopulated-session-counters'], + ['malformed session', { metadata: { tokens_input: -1 } }, 'unknown', 'invalid-session-counters'], + ['partial session counters', { metadata: { tokens_reasoning: null } }, 'unknown', 'invalid-session-counters'], + ['unsupported inclusive token basis', { messages: [['a', assistant({ tokens: { ...TOKENS, total: 163 } })]] }, 'unknown', 'invalid-message-counters'], + ['unsupported version', { metadata: { version: '0.9.10' } }, 'unknown', 'unsupported-version'], + ['compacting', { metadata: { time_compacting: NOW } }, 'unknown', 'incomplete-messages'], + ['missing steps', { parts: [] }, 'unknown', 'unproved-step-scope'], + ['multiple steps', { parts: [['a', { type: 'step-finish', tokens: TOKENS, cost: 0.25 }], ['a', { type: 'step-finish', tokens: TOKENS, cost: 0.25 }]] }, 'unknown', 'unproved-step-scope'], + ['inflight message', { messages: [['a', assistant({ time: { created: NOW - 2000 } })]] }, 'unknown', 'incomplete-messages'], + ['invalid message counters', { messages: [['a', assistant({ tokens: { ...TOKENS, input: '100' } })]] }, 'unknown', 'invalid-message-counters'], +]) test(`reconciliation ${name} preserves message usage`, t => { + const { session } = parseSession(fixture(t, changes)); + assert.equal(session.opencodeReconciliation.state, state); + if (reason) assert.equal(session.opencodeReconciliation.reason, reason); + assert.equal(session.usage[0].input, 100); + assert.equal(session.usage[0].costObserved, 0.25); +}); + +test('aggregate current/previous, warm cache and old parse marker retain observations', async t => { + const opts = fixture(t); + const cold = await buildIndex(opts); + assert.equal(cold.sessions[0].compactions, 1); + assert.equal(cold.totals.compactions, 1); + assert.equal(cold.sessions[0].opencodeReconciliation.state, 'matched'); + _resetForTest(); + const warm = await buildIndex(opts); + assert.equal(warm.totals.compactions, 1); + const cache = JSON.parse(fs.readFileSync(opts.cachePath, 'utf8')); + const entry = Object.values(cache.entries).find(e => e.session?.host === 'opencode'); + entry.parseSemantics = 'cost-trust-v2'; entry.session.compactions = 0; + fs.writeFileSync(opts.cachePath, JSON.stringify(cache)); _resetForTest(); + assert.equal((await buildIndex(opts)).totals.compactions, 1); + _resetForTest(); + const later = await buildIndex({ ...opts, now: NOW + 15 * 86400000, previous: true, lookbackDays: 28 }); + assert.equal(later.totals.compactions, 0); + assert.equal(later.previous.totals.compactions, 1); +}); + + +test('untrusted zero cost and unsupported V2 scope cannot claim matched reconciliation', t => { + const opts = fixture(t, { messages: [['a', assistant({ cost: 0 })]], + parts: [['a', { type: 'step-finish', tokens: TOKENS, cost: 0 }]], metadata: { cost: 0 } }); + assert.equal(parseSession(opts).session.opencodeReconciliation.reason, 'untrusted-message-cost'); + const db = new DatabaseSync(opts.dbFile); + db.exec("CREATE TABLE session_message (session_id TEXT); INSERT INTO session_message VALUES ('s')"); db.close(); + assert.equal(parseSession(opts).session.opencodeReconciliation.reason, 'unsupported-v2-scope'); +}); + +test('oversized selected metadata is bounded before acquisition', t => { + const opts = fixture(t, { metadata: { version: 'x'.repeat(5000) } }); + const { session } = parseSession({ ...opts, maxSessionBytes: 4000 }); + assert.equal(session.acquisitionCoverage.reason, 'session-byte-limit'); +}); + + +test('duplicate summary evidence is counted once per actual request parent', t => { + const opts = fixture(t, { messages: [['u', { role: 'user' }], ['a', assistant()], ['b', assistant()]], + parts: [['u', { type: 'compaction' }], ['u', { type: 'compaction' }]] }); + const { session } = parseSession(opts); + assert.equal(session.compactions, 1); + assert.deepEqual(session.compactionEvidence, { lowerBound: 1, upperBound: 1 }); + assert.equal(session.responses, 2, 'distinct assistant rows keep their recorded usage'); + assert.equal(session.usage[0].costObserved, 0.5); +}); + + +function mutateDb(opts, action) { + const db = new DatabaseSync(opts.dbFile); + try { action(db); } finally { db.close(); } + _resetForTest(); +} + +for (const [name, mutation, reason] of [ + ['upstream part removal and counter subtraction', db => db.exec(`DELETE FROM part WHERE id = 'p1'; + UPDATE session SET cost = 0, tokens_input = 0, tokens_output = 0, tokens_reasoning = 0, + tokens_cache_read = 0, tokens_cache_write = 0`), 'unpopulated-session-counters'], + ['same-count step rewrite', db => db.prepare('UPDATE part SET data = ? WHERE id = ?') + .run(JSON.stringify({ type: 'step-finish', tokens: { ...TOKENS, input: 101 }, cost: 0.25 }), 'p1'), 'unproved-step-scope'], + ['part addition', db => db.exec("INSERT INTO part SELECT 'extra', message_id, session_id, data FROM part WHERE id = 'p1'"), 'unproved-step-scope'], + ['metadata-only rewrite', db => db.exec("UPDATE session SET time_compacting = 123"), 'incomplete-messages'], + ['new V2 scope', db => db.exec("CREATE TABLE session_message (session_id TEXT); INSERT INTO session_message VALUES ('s')"), 'unsupported-v2-scope'], +]) test(`warm observation cache invalidates after ${name} without timestamp changes`, async t => { + const opts = fixture(t); + assert.equal((await buildIndex(opts)).sessions[0].opencodeReconciliation.state, 'matched'); + mutateDb(opts, mutation); + const warm = await buildIndex(opts); + assert.equal(warm.sessions[0].opencodeReconciliation.reason, reason); + assert.deepEqual(warm.sessions[0].opencodeReconciliation, parseSession(opts).session.opencodeReconciliation); + assert.equal(warm.totals.responses, 1); + assert.equal(warm.totals.cost, 0.25); +}); + +test('same-size compaction part rewrite invalidates a current marker while unchanged evidence reuses it', async t => { + const opts = fixture(t); + await buildIndex(opts); + const cache = JSON.parse(fs.readFileSync(opts.cachePath, 'utf8')); + cache.entries['opencode://s'].session.title = 'cache reuse sentinel'; + fs.writeFileSync(opts.cachePath, JSON.stringify(cache)); _resetForTest(); + assert.equal((await buildIndex(opts)).sessions[0].title, 'cache reuse sentinel', 'unchanged source reuses the parse'); + mutateDb(opts, db => db.prepare('UPDATE part SET data = ? WHERE id = ?') + .run(JSON.stringify({ type: 'xxxxxxxxxx' }), 'p0')); + const warm = await buildIndex(opts); + assert.equal(warm.sessions[0].title, 'fixture'); + assert.equal(warm.totals.compactions, 0); + assert.deepEqual(warm.totals.compactionEvidence, { lowerBound: 0, upperBound: 1 }); +}); + +for (const [name, maxSessionBytes, upperBound] of [ + ['request-only', undefined, 1], ['acquisition-incomplete', 64, null], +]) test(`${name} observation uncertainty survives current and previous aggregate windows`, t => { + const opts = fixture(t, { messages: [['u', { role: 'user' }]], parts: [['u', { type: 'compaction' }]] }); + const { session } = parseSession({ ...opts, maxSessionBytes }); + for (const offset of [0, 15]) { + const now = NOW + offset * 86400000; + const result = aggregate([session], { days: 14, now, cutoff: now - 14 * 86400000, previous: true, deps: opts.deps }); + const totals = offset ? result.previous.totals : result.totals; + assert.deepEqual(totals.compactionEvidence, { lowerBound: 0, upperBound }); + assert.equal(totals.sessions, maxSessionBytes ? 0 : 1, 'refused acquisition is not a normal session'); + assert.equal(totals.responses, 0); + assert.equal(totals.tokens, 0); + assert.equal(totals.cost, 0); + if (offset) assert.equal(result.totals.sessions, 0); + } +}); + +test('a tighter acquisition budget cannot reuse a cached full observation', async t => { + const opts = fixture(t); + await buildIndex(opts); _resetForTest(); + const restricted = await buildIndex({ ...opts, readLimits: { maxSessionBytes: 64 } }); + assert.equal(restricted.totals.responses, 0); + assert.deepEqual(restricted.totals.compactionEvidence, { lowerBound: 0, upperBound: null }); + assert.equal(restricted.totals.cost, 0); +}); + + +test('unknown V2 schema preserves V1 usage and refuses a matched reconciliation', async t => { + const opts = fixture(t); + await buildIndex(opts); + mutateDb(opts, db => db.exec('CREATE TABLE session_message (payload TEXT)')); + const warm = await buildIndex(opts); + assert.equal(warm.totals.responses, 1); + assert.equal(warm.totals.cost, 0.25); + assert.equal(warm.sessions[0].opencodeReconciliation.reason, 'unsupported-v2-scope'); +}); + +test('an observation entry without its input digest reparses rather than laundering stale evidence', async t => { + const opts = fixture(t); + await buildIndex(opts); + const cache = JSON.parse(fs.readFileSync(opts.cachePath, 'utf8')); + delete cache.entries['opencode://s'].observationFingerprint; + cache.entries['opencode://s'].session.title = 'unbound cache'; + fs.writeFileSync(opts.cachePath, JSON.stringify(cache)); _resetForTest(); + assert.equal((await buildIndex(opts)).sessions[0].title, 'fixture'); +}); + +test('read-limit controls cannot override the explicitly selected database', async t => { + const opts = fixture(t); + const unrelated = fixture(t, { metadata: { title: 'other source' } }); + const result = await buildIndex({ ...opts, readLimits: { dbFile: unrelated.dbFile, id: unrelated.id } }); + assert.equal(result.sessions[0].title, 'fixture'); +}); diff --git a/tests/kit/usage-opencode-selection.test.mjs b/tests/kit/usage-opencode-selection.test.mjs new file mode 100644 index 00000000..b77bfac7 --- /dev/null +++ b/tests/kit/usage-opencode-selection.test.mjs @@ -0,0 +1,125 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { tempDir } from './helpers/temp-dir.mjs'; +import * as source from '../../src/lib/usage-opencode.mjs'; +import { scanOpencodeDirectories } from '../../src/lib/footprint/project-sources.mjs'; + +test('OpenCode selection honors explicit authority, detects ambiguity, and rejects unsafe paths', () => { + const dir = tempDir('ak-oc-selection-'); + try { + const root = path.join(dir, 'opencode'); fs.mkdirSync(root); + const channel = path.join(root, 'opencode-review_42.db'); fs.writeFileSync(channel, ''); + const env = { HOME: dir, XDG_DATA_HOME: dir }; + const select = (extra = {}) => source.selectOpencodeSource({ env, ...extra }); + assert.equal(typeof source.selectOpencodeSource, 'function'); + assert.equal(select().dbFile, fs.realpathSync(channel)); + fs.writeFileSync(path.join(root, 'opencode.db'), ''); + assert.equal(select().health.reason, 'database-selection-ambiguous'); + assert.equal(select().dbFile, null); + assert.equal(select({ env: { ...env, OPENCODE_DISABLE_CHANNEL_DB: 'true' } }).dbFile, path.join(root, 'opencode.db')); + assert.equal(select({ env: { ...env, OPENCODE_DB: 'opencode-review_42.db' } }).dbFile, fs.realpathSync(channel)); + assert.equal(select({ env: { ...env, OPENCODE_DB: channel } }).dbFile, fs.realpathSync(channel)); + assert.equal(select({ roots: {} }).dbFile, null); + assert.equal(select({ roots: { opencode: channel }, env: { OPENCODE_DB: ':memory:' } }).dbFile, fs.realpathSync(channel)); + assert.equal(select({ env: { ...env, OPENCODE_DB: ':memory:' } }).health.reason, 'database-in-memory'); + for (const value of ['../outside.db', 'bad\0.db', 'bad\n.db']) { + assert.equal(select({ env: { ...env, OPENCODE_DB: value } }).health.status, 'degraded'); + } + assert.equal(select({ roots: { opencode: 'relative.db' } }).health.status, 'degraded'); + const project = scanOpencodeDirectories({ selection: select(), withDb: () => { throw Error('must not open ambiguous source'); } }); + assert.equal(project.reason, 'database-selection-ambiguous'); + assert.equal(project.complete, false); + } finally { fs.rmSync(dir, { recursive: true, force: true }); } +}); + +test('source discovery is bounded and unreadable candidates never trigger a fallback', () => { + const dir = tempDir('ak-oc-bound-'); + try { + const root = path.join(dir, 'opencode'); fs.mkdirSync(root); + const env = { HOME: dir, XDG_DATA_HOME: dir }; + for (let i = 0; i < 257; i++) fs.writeFileSync(path.join(root, `other-${i}`), ''); + assert.equal(source.selectOpencodeSource({ env }).health.reason, 'database-discovery-limit'); + const denied = { ...fs, opendirSync: () => { throw Object.assign(Error('private path'), { code: 'EACCES' }); } }; + assert.equal(source.selectOpencodeSource({ env, fsImpl: denied }).health.reason, 'database-discovery-unreadable'); + const loop = { ...fs, realpathSync: () => { throw Object.assign(Error('private path'), { code: 'ELOOP' }); } }; + const result = source.selectOpencodeSource({ roots: { opencode: path.join(root, 'chosen.db') }, fsImpl: loop }); + assert.equal(result.dbFile, null); + assert.equal(result.health.reason, 'database-path-unreadable'); + assert.ok(!JSON.stringify(result.health).includes(dir)); + assert.equal(source.selectOpencodeSource({ env: { HOME: dir, XDG_DATA_HOME: 'relative', OPENCODE_DB: 'selected.db' } }).dbFile, + path.join(dir, '.local', 'share', 'opencode', 'selected.db')); + const missing = source.selectOpencodeSource({ roots: { opencode: path.join(root, 'missing.db') } }); + assert.equal(scanOpencodeDirectories({ selection: missing }).status, 'absent'); + assert.equal(fs.existsSync(missing.dbFile), false, 'read-only discovery does not create missing stores'); + } finally { fs.rmSync(dir, { recursive: true, force: true }); } +}); + +test('project source health reports unreadable database categories without private error text', () => { + const result = scanOpencodeDirectories({ dbFile: path.resolve('unreadable.db'), + withDb: () => ({ ok: false, error: { kind: 'permission', message: 'private database location' } }) }); + assert.equal(result.status, 'degraded'); + assert.equal(result.reason, 'permission'); +}); + +// The deterministic iterator orders an actual dangling link between two actual +// stores. Production must not treat a candidate's ENOENT as a missing root. +test('a dangling eligible candidate cannot hide a second real database', { skip: process.platform === 'win32' }, () => { + const dir = tempDir('ak-oc-dangling-'); + try { + const root = path.join(dir, 'opencode'); fs.mkdirSync(root); + fs.writeFileSync(path.join(root, 'opencode-a.db'), ''); + fs.symlinkSync(path.join(root, 'missing.db'), path.join(root, 'opencode-b.db')); + fs.writeFileSync(path.join(root, 'opencode-c.db'), ''); + const entries = fs.readdirSync(root, { withFileTypes: true }).sort((a, b) => a.name.localeCompare(b.name)); + let closed = false; + const fsImpl = { ...fs, opendirSync: () => ({ readSync: () => entries.shift() ?? null, closeSync: () => { closed = true; } }) }; + const selection = source.selectOpencodeSource({ env: { HOME: dir, XDG_DATA_HOME: dir }, fsImpl }); + assert.equal(selection.dbFile, null); + assert.equal(selection.health.status, 'degraded'); + assert.equal(closed, true); + const projects = scanOpencodeDirectories({ selection, withDb: () => { throw Error('uncertain source must not be opened'); } }); + assert.equal(projects.complete, false); + } finally { fs.rmSync(dir, { recursive: true, force: true }); } +}); + +for (const failure of ['stat-ENOENT', 'stat-EACCES', 'read-ENOENT', 'read-EIO', 'close-EIO', 'limit']) { + test(`partial discovery never establishes uniqueness after ${failure}`, () => { + const dir = tempDir('ak-oc-partial-'); + try { + const root = path.join(dir, 'opencode'); fs.mkdirSync(root); + const first = path.join(root, 'opencode-a.db'); fs.writeFileSync(first, ''); + const second = path.join(root, 'opencode-b.db'); fs.writeFileSync(second, ''); + let count = 0; + const [operation, code] = failure.split('-'); + const fail = () => { throw Object.assign(Error('private path'), { code }); }; + const fsImpl = { ...fs, + opendirSync: () => ({ + readSync: () => { + count++; + if (count === 1) return { name: 'opencode-a.db' }; + if (operation === 'read') return fail(); + if (operation === 'limit') return { name: `unrelated-${count}` }; + return count === 2 ? { name: 'opencode-b.db' } : null; + }, + closeSync: () => { if (operation === 'close') fail(); }, + }), + statSync: (file) => file === second && operation === 'stat' ? fail() : fs.statSync(file), + }; + const result = source.selectOpencodeSource({ env: { HOME: dir, XDG_DATA_HOME: dir }, fsImpl }); + assert.equal(result.dbFile, null); + assert.equal(result.health.status, 'degraded'); + assert.ok(!JSON.stringify(result.health).includes(dir)); + } finally { fs.rmSync(dir, { recursive: true, force: true }); } + }); +} + +test('absolute OpenCode authority survives invalid ambient home when legacy root cannot be resolved', () => { + const dir = tempDir('ak-oc-absolute'); + const dbFile = path.join(dir, 'explicit.db'); + const selected = source.selectOpencodeSource({ env: { HOME: 'relative', OPENCODE_DB: dbFile } }); + assert.equal(selected.dbFile, dbFile); + assert.equal(selected.legacyRoot, null); + assert.equal(source.selectOpencodeSource({ env: { HOME: 'relative', OPENCODE_DB: ':memory:' } }).health.reason, 'database-in-memory'); +}); diff --git a/tests/kit/usage-opencode-storage-coverage.test.mjs b/tests/kit/usage-opencode-storage-coverage.test.mjs new file mode 100644 index 00000000..78af4053 --- /dev/null +++ b/tests/kit/usage-opencode-storage-coverage.test.mjs @@ -0,0 +1,103 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { createHash } from 'node:crypto'; +import { DatabaseSync } from 'node:sqlite'; +import { observeOpencodeStorageCoverage } from '../../src/lib/usage-opencode-storage-coverage.mjs'; + +function fixture(fn) { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-oc-coverage-')); + const file = path.join(root, 'opencode.db'); + const db = new DatabaseSync(file); + try { return fn({ root, file, db }); } + finally { if (db.isOpen) db.close(); fs.rmSync(root, { recursive: true, force: true }); } +} + +const digest = (file) => createHash('sha256').update(fs.readFileSync(file)).digest('hex'); + +test('null inputs remain not observed, with no completeness claim', () => { + assert.deepEqual(observeOpencodeStorageCoverage(), { + v2: { status: 'not-observed' }, legacy: { status: 'not-observed' }, warnings: [], + }); +}); + +test('older schema, empty V2 table, and populated V2 table have distinct states', () => fixture(({ db }) => { + assert.equal(observeOpencodeStorageCoverage({ db }).v2.status, 'missing'); + db.exec('CREATE TABLE session_message (id TEXT, data TEXT)'); + assert.equal(observeOpencodeStorageCoverage({ db }).v2.status, 'empty'); + db.prepare('INSERT INTO session_message VALUES (?, ?)').run('one', 'private body'); + const result = observeOpencodeStorageCoverage({ db }); + assert.deepEqual(result.v2, { status: 'present' }); + assert.deepEqual(result.warnings, ['opencode-v2-session-message-present']); + assert.equal(JSON.stringify(result).includes('private body'), false); +})); + +test('non-table schema and failed query stay unknown without leaking error detail', () => fixture(({ db }) => { + db.exec('CREATE VIEW session_message AS SELECT 1 AS id'); + assert.deepEqual(observeOpencodeStorageCoverage({ db }).v2, { status: 'unknown' }); + db.exec('DROP VIEW session_message; CREATE TABLE session_message (id TEXT)'); + db.close(); + const result = observeOpencodeStorageCoverage({ db }); + assert.deepEqual(result.v2, { status: 'unknown' }); + assert.deepEqual(result.warnings, ['opencode-v2-observation-incomplete']); +})); + +test('legacy absent and regular JSON presence are distinguished without reading content', () => fixture(({ root, db, file }) => { + const missing = path.join(root, 'missing'); + assert.equal(observeOpencodeStorageCoverage({ legacyRoot: missing }).legacy.status, 'absent'); + const storage = path.join(root, 'storage'); + fs.mkdirSync(path.join(storage, 'session'), { recursive: true }); + const json = path.join(storage, 'session', 'one.json'); + fs.writeFileSync(json, '{"private":"body"}'); + const before = [digest(file), digest(json)]; + const result = observeOpencodeStorageCoverage({ db, legacyRoot: storage }); + assert.deepEqual(result.legacy, { status: 'present' }); + assert.deepEqual(result.warnings, ['opencode-legacy-json-present']); + assert.equal(JSON.stringify(result).includes('private'), false); + assert.deepEqual([digest(file), digest(json)], before); +})); + +test('unreadable shape, entry cap, depth cap, and symlink-only tree are unknown', () => fixture(({ root }) => { + const fileRoot = path.join(root, 'file'); + fs.writeFileSync(fileRoot, 'not a directory'); + assert.equal(observeOpencodeStorageCoverage({ legacyRoot: fileRoot }).legacy.status, 'unknown'); + const storage = path.join(root, 'storage'); + fs.mkdirSync(storage); + fs.writeFileSync(path.join(storage, 'a.txt'), 'a'); + fs.writeFileSync(path.join(storage, 'b.txt'), 'b'); + assert.equal(observeOpencodeStorageCoverage({ legacyRoot: storage, maxEntries: 1 }).legacy.status, 'unknown'); + fs.mkdirSync(path.join(storage, 'nested')); + fs.writeFileSync(path.join(storage, 'nested', 'one.json'), '{}'); + assert.equal(observeOpencodeStorageCoverage({ legacyRoot: storage, maxDepth: 0 }).legacy.status, 'unknown'); + const linkOnly = path.join(root, 'links'); + fs.mkdirSync(linkOnly); + fs.symlinkSync(path.join(storage, 'nested'), path.join(linkOnly, 'nested')); + assert.equal(observeOpencodeStorageCoverage({ legacyRoot: linkOnly }).legacy.status, 'unknown'); +})); + +test('established presence survives incomplete traversal, and combined warnings are fixed shape', () => fixture(({ root, db }) => { + db.exec('CREATE TABLE session_message (id TEXT)'); + db.prepare('INSERT INTO session_message VALUES (?)').run('one'); + const storage = path.join(root, 'storage'); + fs.mkdirSync(storage); + fs.writeFileSync(path.join(storage, 'a.json'), '{}'); + fs.symlinkSync(path.join(root, 'outside'), path.join(storage, 'z-link')); + assert.deepEqual(observeOpencodeStorageCoverage({ db, legacyRoot: storage }), { + v2: { status: 'present' }, legacy: { status: 'present' }, + warnings: ['opencode-v2-session-message-present', 'opencode-legacy-json-present'], + }); +})); + +test('caller-owned read-only handle remains open and database bytes remain unchanged', () => fixture(({ db, file }) => { + db.exec('CREATE TABLE session_message (id TEXT); INSERT INTO session_message VALUES (1)'); + db.close(); + const before = digest(file); + const reader = new DatabaseSync(file, { readOnly: true }); + try { + assert.equal(observeOpencodeStorageCoverage({ db: reader }).v2.status, 'present'); + assert.equal(reader.isOpen, true); + assert.equal(digest(file), before); + } finally { reader.close(); } +})); diff --git a/tests/kit/usage-opencode.test.mjs b/tests/kit/usage-opencode.test.mjs index 8327587d..9826a10b 100644 --- a/tests/kit/usage-opencode.test.mjs +++ b/tests/kit/usage-opencode.test.mjs @@ -10,6 +10,8 @@ import path from 'node:path'; import { DatabaseSync } from 'node:sqlite'; import { listSessions, parseSession, sessionExists } from '../../src/lib/usage-opencode.mjs'; import { promptFingerprint } from '../../src/lib/usage-parsers.mjs'; +import { sessionCostEvidence } from '../../src/lib/usage-cost.mjs'; +import { buildIndex, _resetForTest } from '../../src/lib/usage-index.mjs'; const tmp = () => fs.mkdtempSync(path.join(os.tmpdir(), 'ak-uo-')); const rm = (d) => fs.rmSync(d, { recursive: true, force: true }); @@ -73,6 +75,88 @@ const assistantMsg = (id, sessionId, at, { model = 'kimi-k3', provider = 'openco }, }); +test('positive-token reported zero is unpriced for hosted and unknown providers, but observed for local providers', () => { + const d = tmp(); + try { + const dbFile = buildDb(path.join(d, 'opencode.db'), { + sessions: [{ id: 'cost-zero', directory: '/x', title: 'cost zero' }], + messages: [ + assistantMsg('hosted', 'cost-zero', T, { provider: 'openrouter', model: 'unknown-model', cost: 0 }), + assistantMsg('unknown', 'cost-zero', T + 1000, { provider: '', model: 'unknown-model', cost: 0 }), + assistantMsg('local', 'cost-zero', T + 2000, { provider: 'lmstudio', model: 'unknown-model', cost: 0 }), + ], + }); + const { session } = parseSession({ dbFile, id: 'cost-zero' }); + const evidence = sessionCostEvidence(session, { costOf: () => { throw Error('reported zero must not be estimated'); } }); + assert.deepEqual(evidence, { observedUsd: 0, estimatedUsd: 0, observedMessages: 1, estimatedMessages: 0, unpricedMessages: 2 }); + assert.equal(session.usage.length, 3, 'provider attribution remains separate'); + assert.equal(session.usage.find(r => r.provider === 'openrouter').costObserved, null); + assert.equal(session.usage.find(r => r.provider === 'lmstudio').costObserved, 0); + } finally { rm(d); } +}); + +test('reported zero without measured tokens and positive observed cost preserve their own evidence', () => { + const d = tmp(); + try { + const dbFile = buildDb(path.join(d, 'opencode.db'), { + sessions: [{ id: 'cost-mixed', directory: '/x', title: 'mixed' }], + messages: [ + assistantMsg('positive', 'cost-mixed', T, { provider: 'openrouter', cost: 0.25 }), + assistantMsg('untrusted', 'cost-mixed', T + 1000, { provider: 'openrouter', cost: 0 }), + assistantMsg('missing', 'cost-mixed', T + 2000, { provider: 'openrouter' }), + assistantMsg('zero-tokens', 'cost-mixed', T + 3000, { provider: 'openrouter', cost: 0, + tokens: { input: 0, output: 0, reasoning: 0, cache: { read: 0, write: 0 } } }), + ], + }); + const { session } = parseSession({ dbFile, id: 'cost-mixed' }); + const evidence = sessionCostEvidence(session, { costOf: () => 0.5 }); + assert.deepEqual(evidence, { observedUsd: 0.25, estimatedUsd: 0.5, observedMessages: 2, estimatedMessages: 1, unpricedMessages: 1 }); + } finally { rm(d); } +}); + +test('malformed recorded costs are missing coverage, not trusted charges', () => { + const d = tmp(); + try { + const dbFile = buildDb(path.join(d, 'opencode.db'), { + sessions: [{ id: 'cost-bad', directory: '/x', title: 'bad' }], + messages: [ + assistantMsg('negative', 'cost-bad', T, { cost: -1 }), + assistantMsg('string', 'cost-bad', T + 1000, { cost: '0' }), + assistantMsg('nan', 'cost-bad', T + 2000, { cost: Number.NaN }), + ], + }); + const { session } = parseSession({ dbFile, id: 'cost-bad' }); + const evidence = sessionCostEvidence(session, { costOf: () => 0.1 }); + assert.deepEqual(evidence, { observedUsd: 0, estimatedUsd: 0.1, observedMessages: 0, estimatedMessages: 3, unpricedMessages: 0 }); + } finally { rm(d); } +}); + +test('a cold and warm OpenCode scan conserve unpriced zero-cost coverage', async () => { + const d = tmp(); + _resetForTest(); + try { + const dbFile = buildDb(path.join(d, 'opencode.db'), { + sessions: [{ id: 'cost-cache', directory: '/x', title: 'cache' }], + messages: [assistantMsg('remote-zero', 'cost-cache', T, { provider: 'openrouter', model: 'unknown-model', cost: 0 })], + }); + const opts = { now: T + DAY, days: 14, cachePath: path.join(d, 'cache', 'index.json'), + roots: { claude: path.join(d, 'claude'), codex: path.join(d, 'codex'), opencode: dbFile } }; + const cold = await buildIndex(opts); + _resetForTest(); + const warm = await buildIndex(opts); + for (const agg of [cold, warm]) { + const session = agg.sessions.find((row) => row.id === 'cost-cache'); + assert.equal(session.costEvidence.unpricedMessages, 1); + assert.equal(session.costEvidence.observedMessages, 0); + assert.equal(session.costEvidence.estimatedMessages, 0); + assert.equal(session.cost, 0); + assert.equal(agg.totals.cost, 0); + } + assert.deepEqual(warm.sessions.find((row) => row.id === 'cost-cache').costEvidence, + cold.sessions.find((row) => row.id === 'cost-cache').costEvidence); + } finally { _resetForTest(); rm(d); } +}); + test('listSessions filters by the latest message time and keys on mtime+count', () => { const d = tmp(); const dbFile = buildDb(path.join(d, 'opencode.db'), { @@ -117,17 +201,18 @@ test('parseSession maps a session to the index record: identity, usage rows with assert.equal(rec.exceptions, 0); assert.equal(rec.sidechain, false); assert.equal(rec.threadSource, null); - // provider is the LAST observed assistant providerID — never the host - assert.equal(rec.inferenceProvider, 'openrouter'); - assert.equal(rec.providerProvenance, 'observed'); + assert.equal(rec.inferenceProvider, null, 'a session spanning providers has no single inference provider'); + assert.equal(rec.providerProvenance, 'unknown'); // usage rows per (day, model) with summed observed cost const day1 = rec.usage.find((r) => r.model === 'kimi-k3'); + assert.equal(day1.provider, 'opencode'); assert.deepEqual( { input: day1.input, output: day1.output, cacheRead: day1.cacheRead, cacheWrite: day1.cacheWrite, responses: day1.responses, costObserved: day1.costObserved }, // output = 2 x (20 text + 5 reasoning): OpenCode stores output NET of reasoning { input: 200, output: 50, cacheRead: 80, cacheWrite: 6, responses: 2, costObserved: 0.03 }, ); const day2 = rec.usage.find((r) => r.model === 'moonshotai/kimi-k3'); + assert.equal(day2.provider, 'openrouter'); assert.equal(day2.costObserved, 0.03); assert.equal(day2.day !== day1.day, true, 'rows keyed by day'); assert.deepEqual(rec.models, ['kimi-k3', 'moonshotai/kimi-k3']); @@ -312,6 +397,59 @@ test('parseSession fingerprints user messages on the scan path, not only withTur rm(d); }); +test('an OpenCode child keeps its prompt and usage but never fingerprints its user turns', () => { + const d = tmp(); + try { + const dbFile = buildDb(path.join(d, 'opencode.db'), { + sessions: [ + { id: 'parent', directory: '/x', title: 'parent' }, + { id: 'child', directory: '/x', title: 'child', parentId: 'parent' }, + { id: 'different-child', directory: '/x', title: 'other', parentId: 'parent' }, + ], + messages: [ + userMsg('pu', 'parent', T), assistantMsg('pa', 'parent', T + 1000, { cost: 0.2 }), + userMsg('cu', 'child', T), assistantMsg('ca', 'child', T + 1000, { cost: 0.3 }), + userMsg('du', 'different-child', T), + ], + parts: [ + { id: 'pp', messageId: 'pu', sessionId: 'parent', at: T, data: { type: 'text', text: 'Run the tests' } }, + { id: 'cp', messageId: 'cu', sessionId: 'child', at: T, data: { type: 'text', text: 'Run the tests' } }, + { id: 'dp', messageId: 'du', sessionId: 'different-child', at: T, data: { type: 'text', text: 'Review the database migration' } }, + ], + }); + for (const withTurns of [false, true]) { + const parent = parseSession({ dbFile, id: 'parent', withTurns }).session; + const child = parseSession({ dbFile, id: 'child', withTurns }).session; + assert.equal(parent.promptFPs.length, 1); + assert.deepEqual(child.promptFPs, []); + assert.deepEqual(parseSession({ dbFile, id: 'different-child', withTurns }).session.promptFPs, []); + assert.equal(child.prompts, 1); + assert.equal(child.sidechain, true); + assert.equal(child.threadSource, 'subagent'); + assert.equal(child.usage[0].costObserved, 0.3); + } + } finally { rm(d); } +}); + +test('only a nonempty parent_id is child evidence, even when the parent row is absent', () => { + const d = tmp(); + try { + const dbFile = buildDb(path.join(d, 'opencode.db'), { + sessions: [ + { id: 'orphan', directory: '/x', title: 'orphan', parentId: 'missing' }, + { id: 'blank', directory: '/x', title: 'blank', parentId: '' }, + ], + messages: [userMsg('ou', 'orphan', T), userMsg('bu', 'blank', T)], + }); + const orphan = parseSession({ dbFile, id: 'orphan' }).session; + const blank = parseSession({ dbFile, id: 'blank' }).session; + assert.equal(orphan.sidechain, true); + assert.deepEqual(orphan.promptFPs, []); + assert.equal(blank.sidechain, false); + assert.equal(blank.promptFPs.length, 1); + } finally { rm(d); } +}); + test('a user message with no text part fingerprints as an attachment-only control turn', () => { const d = tmp(); const dbFile = buildDb(path.join(d, 'opencode.db'), { @@ -562,7 +700,7 @@ test('the same modelID under two providers stays two usage rows, each carrying i assert.equal(local.costObserved, null, 'the local turn recorded no cost and is not charged with the cloud turn\'s'); assert.equal(cloud.input, 75); assert.ok(Math.abs(cloud.costObserved - 0.3) < 1e-9); - assert.equal(session.inferenceProvider, 'openrouter', 'the session-level provider stays the last observed one'); + assert.equal(session.inferenceProvider, null, 'two observed providers cannot be a single session provider'); } finally { rm(d); } }); diff --git a/tests/kit/usage-session-surface.test.mjs b/tests/kit/usage-session-surface.test.mjs new file mode 100644 index 00000000..b19510f6 --- /dev/null +++ b/tests/kit/usage-session-surface.test.mjs @@ -0,0 +1,142 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import { parseClaude, parseCodex } from '../../src/lib/usage-parsers.mjs'; +import { buildIndex, SCHEMA_VERSION, _resetForTest } from '../../src/lib/usage-index.mjs'; +import { Rollout, usage, codexSandbox, stubDeps } from './helpers/codex-rollout.mjs'; + +const NOW = Date.parse('2026-07-25T12:00:00.000Z'); +const options = (sandbox) => ({ days: 14, now: NOW, roots: sandbox.roots, + cachePath: sandbox.cachePath, deps: stubDeps() }); + +test('Claude declarations persist the first known SDK classification without changing legacy origin', () => { + const line = (value) => `${JSON.stringify(value)}\n`; + const raw = line({ type: 'user', timestamp: '2026-07-24T09:00:00Z', sessionId: 'sdk', + entrypoint: 'sdk-py', message: { role: 'user', content: 'synthetic request' } }) + + line({ type: 'assistant', timestamp: '2026-07-24T09:01:00Z', sessionId: 'sdk', + entrypoint: 'claude-desktop', message: { id: 'a', role: 'assistant', model: 'claude-opus-5', + usage: { input_tokens: 5, output_tokens: 2 }, content: [{ type: 'text', text: 'synthetic response' }] } }); + const { session } = parseClaude(raw, { id: 'sdk' }); + assert.deepEqual(session.sessionOrigin, { + origin: 'unknown', evidence: 'desktop-origin-not-declared', surface: 'claude-agent-sdk', + initiator: 'automation', label: 'Claude Agent SDK', rawEvidence: { entrypoint: 'sdk-py' }, + attributes: [], thirdPartyProvider: null, + }); +}); + +test('Codex MCP declaration and imported copy use factory fields with existing accounting', () => { + const native = new Rollout({ id: 'mcp' }).meta({ originator: 'codex_cli_rs', source: 'mcp' }) + .meta({ originator: 'Codex Desktop', source: 'vscode' }).turn().user().agent() + .tokenCount(usage({ input: 100, output: 20 })); + const parsed = parseCodex(native.toString(), { id: 'mcp' }).session; + assert.deepEqual(parsed.sessionOrigin, { + origin: 'unknown', evidence: 'desktop-origin-not-declared', surface: 'codex-mcp', + initiator: 'agent', label: 'Codex MCP server', + rawEvidence: { originator: 'codex_cli_rs', source: 'mcp', threadSource: 'user' }, + attributes: [], thirdPartyProvider: null, + }); + assert.equal(parsed.prompts, 1); + assert.equal(parsed.responses, 1); + + const imported = new Rollout({ id: 'copy' }).meta({ originator: 'Codex Desktop' }) + .taskStarted('external-import-turn-1').user('copied request').agent('copied response'); + const copy = parseCodex(imported.toString(), { id: 'copy' }).session; + assert.equal(copy.imported, true); + assert.equal(copy.sessionOrigin.origin, 'unknown'); + assert.equal(copy.sessionOrigin.evidence, 'imported-copy'); + assert.equal(copy.sessionOrigin.surface, 'unknown'); + assert.equal(copy.sessionOrigin.initiator, 'imported-copy'); + assert.deepEqual(copy.sessionOrigin.rawEvidence, {}); + assert.equal(copy.prompts, 0); + assert.equal(copy.responses, 0); + assert.deepEqual(copy.usage, []); +}); + +test('a late import marker still removes copied Desktop classification', () => { + const copied = new Rollout({ id: 'late-copy' }).meta({ originator: 'Codex Desktop' }); + for (let i = 0; i < 45; i++) copied.raw('event_msg', { type: 'task_complete' }); + copied.taskStarted('external-import-turn-1'); + const session = parseCodex(copied.toString(), { id: 'late-copy' }).session; + assert.equal(session.imported, true); + assert.deepEqual(session.sessionOrigin, { + origin: 'unknown', evidence: 'imported-copy', surface: 'unknown', initiator: 'imported-copy', + label: 'Unknown', rawEvidence: {}, attributes: [], thirdPartyProvider: null, + }); +}); + +test('malformed declaration metadata cannot copy prompt text into classification', () => { + const privateText = 'synthetic private prompt content'; + const raw = `${JSON.stringify({ type: 'session_meta', payload: { + id: 'bad', originator: { text: privateText }, source: ['mcp'], thread_source: privateText, + } })}\n`; + const session = parseCodex(raw, { id: 'bad' }).session; + assert.equal(JSON.stringify(session.sessionOrigin).includes(privateText), false); + assert.deepEqual(session.sessionOrigin.rawEvidence, {}); +}); + +test('unfamiliar bounded origin survives parser, aggregate and warm cache without a product guess', async () => { + _resetForTest(); + const native = new Rollout({ id: 'future' }).meta({ originator: 'future_client_v2', + source: 'future_transport', thread_source: 'future_trigger' }) + .turn().user().agent().tokenCount(usage({ input: 100, output: 20 })); + const sandbox = codexSandbox({ 'rollout-2026-07-24T09-00-00-future.jsonl': native.toString() }); + const parsed = parseCodex(native.toString(), { id: 'future' }).session; + assert.equal(parsed.sessionOrigin.surface, 'other-openai'); + assert.equal(parsed.sessionOrigin.initiator, 'unknown'); + assert.deepEqual(parsed.sessionOrigin.rawEvidence, { originator: 'future_client_v2', + source: 'future_transport', threadSource: 'future_trigger' }); + const cold = await buildIndex(options(sandbox)); + assert.equal(cold.sessions[0].sessionOrigin.surface, 'other-openai'); + assert.deepEqual(cold.sessions[0].sessionOrigin.rawEvidence, parsed.sessionOrigin.rawEvidence); + const cache = JSON.parse(fs.readFileSync(sandbox.cachePath, 'utf8')); + assert.deepEqual(Object.values(cache.entries)[0].session.sessionOrigin.rawEvidence, + parsed.sessionOrigin.rawEvidence); + _resetForTest(); + const warm = await buildIndex(options(sandbox)); + assert.equal(warm.sourceHealth.codex.diagnostics.cachedFiles, 1); + assert.deepEqual(warm.sessions[0].sessionOrigin.rawEvidence, parsed.sessionOrigin.rawEvidence); + assert.deepEqual(warm.totals, cold.totals); +}); + +test('schema 25 reparses unchanged files into schema 26 and warm cache preserves classification and accounting', async () => { + _resetForTest(); + const native = new Rollout({ id: 'sdk' }).meta({ originator: 'codex_sdk_ts' }) + .turn().user().agent().tokenCount(usage({ input: 100, output: 20 })); + const imported = new Rollout({ id: 'copy' }).meta({ originator: 'Codex Desktop' }) + .taskStarted('external-import-turn-1').user('copied request'); + const sandbox = codexSandbox({ + 'rollout-2026-07-24T09-00-00-sdk.jsonl': native.toString(), + 'rollout-2026-07-24T09-00-00-copy.jsonl': imported.toString(), + }); + const before = await buildIndex(options(sandbox)); + assert.deepEqual({ sessions: before.totals.sessions, prompts: before.totals.prompts, + responses: before.totals.responses, tokens: before.totals.tokens, cost: before.totals.cost, + importedExcluded: before.sourceHealth.codex.diagnostics.importedExcluded }, + { sessions: 1, prompts: 1, responses: 1, tokens: 120, cost: 1, importedExcluded: 1 }); + const old = JSON.parse(fs.readFileSync(sandbox.cachePath, 'utf8')); + old.schemaVersion = 25; + for (const entry of Object.values(old.entries)) { + entry.session.sessionOrigin = { origin: entry.session.sessionOrigin.origin, + evidence: entry.session.sessionOrigin.evidence }; + } + fs.writeFileSync(sandbox.cachePath, JSON.stringify(old)); + + _resetForTest(); + const cold = await buildIndex(options(sandbox)); + assert.equal(SCHEMA_VERSION, 26); + assert.equal(cold.sourceHealth.codex.diagnostics.cachedFiles, 0); + assert.equal(cold.sourceHealth.codex.diagnostics.importedExcluded, 1); + assert.deepEqual(cold.totals, before.totals); + assert.equal(cold.sessions[0].sessionOrigin.surface, 'codex-sdk'); + assert.equal(cold.sessions[0].sessionOrigin.initiator, 'automation'); + const cache = JSON.parse(fs.readFileSync(sandbox.cachePath, 'utf8')); + assert.equal(cache.schemaVersion, 26); + assert.ok(Object.values(cache.entries).some((entry) => entry.session.sessionOrigin.surface === 'codex-sdk')); + + _resetForTest(); + const warm = await buildIndex(options(sandbox)); + assert.equal(warm.sourceHealth.codex.diagnostics.cachedFiles, 2); + assert.equal(warm.sourceHealth.codex.diagnostics.importedExcluded, 1); + assert.deepEqual(warm.totals, cold.totals); + assert.deepEqual(warm.sessions[0].sessionOrigin, cold.sessions[0].sessionOrigin); +}); diff --git a/tests/kit/usage-timezone-cache.test.mjs b/tests/kit/usage-timezone-cache.test.mjs new file mode 100644 index 00000000..b1fbf6c6 --- /dev/null +++ b/tests/kit/usage-timezone-cache.test.mjs @@ -0,0 +1,191 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { DatabaseSync } from 'node:sqlite'; +import { spawnSync } from 'node:child_process'; +import { tempDir } from './helpers/temp-dir.mjs'; +import { selectOpencodeSource } from '../../src/lib/usage-opencode-source.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; + +const indexUrl = new URL('../../src/lib/usage-index.mjs', import.meta.url).href; +const zones = ['America/Los_Angeles', 'Asia/Tokyo']; +// Local midnight, spring DST gap, and autumn repeated hour. Also spans a +// synthetic dated price boundary, preserving the existing local-row-day basis. +const stamps = ['2026-03-08T07:59:00Z', '2026-03-08T08:01:00Z', + '2026-03-08T09:59:00Z', '2026-03-08T10:01:00Z', + '2026-11-01T08:30:00Z', '2026-11-01T09:30:00Z']; +function fixture(t) { + const dir = tempDir('ak-timezone'); + t.after(() => fs.rmSync(dir, { recursive: true, force: true })); + const claude = path.join(dir, 'claude', 'project'); + const codex = path.join(dir, 'codex'); + fs.mkdirSync(claude, { recursive: true }); + fs.mkdirSync(path.join(codex, '2026', '03', '08'), { recursive: true }); + const rows = stamps.map((timestamp, i) => ({ type: 'assistant', timestamp, + message: { id: `message-${i}`, role: 'assistant', model: 'claude-opus-5', + usage: { input_tokens: 100, output_tokens: 20, cache_read_input_tokens: 50, + cache_creation_input_tokens: 10 }, content: [] } })); + fs.writeFileSync(path.join(claude, 'fixture.jsonl'), rows.map(JSON.stringify).join('\n')); + const cx = [{ type: 'session_meta', timestamp: stamps[0], payload: { id: 'codex-fixture', model_provider: 'openai' } }, + { type: 'turn_context', timestamp: stamps[0], payload: { model: 'gpt-5.6', effort: 'high' } }, + ...stamps.map((timestamp, i) => ({ type: 'event_msg', timestamp, payload: { type: 'token_count', info: { + total_token_usage: { input_tokens: (i + 1) * 100, cached_input_tokens: (i + 1) * 50, output_tokens: (i + 1) * 20 }, + } } }))]; + fs.writeFileSync(path.join(codex, '2026', '03', '08', 'rollout-fixture.jsonl'), cx.map(JSON.stringify).join('\n')); + return dir; +} +function openCodeFixture(dir) { + const db = new DatabaseSync(path.join(dir, 'broken.db')); + db.exec(`CREATE TABLE session (id TEXT, parent_id TEXT, directory TEXT, title TEXT, time_created INTEGER, time_updated INTEGER); + CREATE TABLE message (id TEXT, session_id TEXT, time_created INTEGER, time_updated INTEGER, data TEXT); + CREATE TABLE part (message_id TEXT, data TEXT);`); + db.prepare('INSERT INTO session VALUES (?, NULL, ?, ?, ?, ?)').run('oc-native', '/fixture', 'fixture', Date.parse(stamps[0]), Date.parse(stamps.at(-1))); + const insert = db.prepare('INSERT INTO message VALUES (?, ?, ?, ?, ?)'); + stamps.forEach((timestamp, i) => { + const at = Date.parse(timestamp); + insert.run(`a${i}`, 'oc-native', at, at, JSON.stringify({ role: 'assistant', + modelID: 'gpt-5.6', providerID: 'openai', cost: 0.25, + tokens: { input: 100, output: 20, reasoning: 5, cache: { read: 50, write: 10 } }, + time: { created: at, completed: at + 1000 }, finish: 'stop' })); + }); + db.close(); +} +function run(dir, zone, extra = '', days = 240, runtimeSetup = '') { + const script = ` + import fs from 'node:fs'; + import path from 'node:path'; + import { buildIndex, readIndex } from ${JSON.stringify(indexUrl)}; + const dir = process.argv[1]; + const cachePath = path.join(dir, 'cache.json'); + const options = { roots: { claude: path.join(dir, 'claude'), codex: path.join(dir, 'codex'), + opencode: path.join(dir, 'broken.db') }, cachePath, + now: Date.parse('2026-11-02T12:00:00Z'), days: ${days}, lookbackDays: 240, previous: true, + deps: { costOf: (u) => (u.input + u.output + u.cacheRead + u.cacheWrite) / 1000 + * (u.day < '2026-03-08' ? 2 : 1), + classify: () => ({category:'Build', confidence:1, basis:'fixture'}), detectInsights: () => [] } }; + // Filesystem provenance uses Date.now independently of the query clock. + Date.now = () => options.now; + ${runtimeSetup} + const first = await readIndex(options); + ${extra} + const agg = ${extra ? 'await readIndex(options)' : 'first'}; + const cache = JSON.parse(fs.readFileSync(cachePath, 'utf8')); + const {sourceHealth, ...stable} = agg; + console.log(JSON.stringify({ stable, sourceHealth, entries: Object.values(cache.entries) })); + `; + const out = spawnSync(process.execPath, ['--input-type=module', '-e', script, dir], { + env: spawnEnv(path.join(dir, 'home'), { TZ: zone }), encoding: 'utf8', timeout: 30000, + }); + assert.equal(out.status, 0, out.stderr); + return JSON.parse(out.stdout); +} +function poisonContext(dir, context) { + const file = path.join(dir, 'cache.json'); + const cache = JSON.parse(fs.readFileSync(file, 'utf8')); + for (const entry of Object.values(cache.entries)) { + if (context === undefined) delete entry.localTimeContext; + else entry.localTimeContext = context; + entry.session.usage.forEach((row) => { row.day = '1900-01-01'; }); + } + fs.writeFileSync(file, JSON.stringify(cache)); +} + +test('cold and warm indexes agree after timezone changes, including DST and dated pricing', (t) => { + const dir = fixture(t); + openCodeFixture(dir); + const la = run(dir, zones[0]); + assert.equal(la.stable.sessions.length, 3); + const warm = run(dir, zones[0]); + assert.deepEqual(warm.stable, la.stable); + assert.ok(warm.sourceHealth.codex.diagnostics.cachedFiles > 0); + const changed = run(dir, zones[1]); + fs.unlinkSync(path.join(dir, 'cache.json')); + const cold = run(dir, zones[1]); + assert.deepEqual(changed.stable, cold.stable); + assert.deepEqual(changed.entries, cold.entries); + assert.equal(changed.sourceHealth.codex.diagnostics.cachedFiles, 0); + assert.notDeepEqual(la.stable.byDay, cold.stable.byDay); + assert.equal(la.stable.totals.tokens, cold.stable.totals.tokens); + assert.notEqual(la.stable.totals.cost, cold.stable.totals.cost, + 'existing local day pricing intentionally changes at a dated rate boundary'); + const back = run(dir, zones[0]); + assert.deepEqual(back.stable, la.stable); +}); + +for (const context of [undefined, null, 'Invalid/Zone', {}, { zone: 'UTC' }]) { + test(`missing or invalid cached timezone reparses: ${JSON.stringify(context)}`, (t) => { + const dir = fixture(t); + const expected = run(dir, zones[0]); + poisonContext(dir, context); + const actual = run(dir, zones[0]); + assert.deepEqual(actual.stable, expected.stable); + assert.equal(actual.sourceHealth.codex.diagnostics.cachedFiles, 0); + }); +} + +test('in-process memo observes a child-only TZ change', (t) => { + const dir = fixture(t); + const changed = run(dir, zones[0], `process.env.TZ = ${JSON.stringify(zones[1])};`); + fs.unlinkSync(path.join(dir, 'cache.json')); + assert.deepEqual(changed.stable, run(dir, zones[1]).stable); +}); + +for (const unavailable of ['return { timeZone: undefined };', 'throw new Error("zone unavailable");']) { + test(`unknown runtime timezone declines persistent cache reuse: ${unavailable}`, (t) => { + const dir = fixture(t); + const expected = run(dir, zones[0]); + // Invalid TZ strings can resolve to a fallback zone on Windows. Model the + // unavailable Intl boundary directly, inside this disposable child only. + const setup = `Intl.DateTimeFormat.prototype.resolvedOptions = function () { ${unavailable} };`; + for (let refresh = 0; refresh < 2; refresh++) { + const actual = run(dir, zones[0], '', 240, setup); + assert.deepEqual(actual.stable, expected.stable); + assert.equal(actual.sourceHealth.codex.diagnostics.cachedFiles, 0); + assert.ok(actual.entries.length > 0); + assert.ok(actual.entries.every((entry) => entry.localTimeContext === null)); + } + }); +} + +test('degraded OpenCode retains old timezone evidence without contributing stale buckets', (t) => { + const dir = fixture(t); + const initial = run(dir, zones[0]); + const file = path.join(dir, 'cache.json'); + const cache = JSON.parse(fs.readFileSync(file, 'utf8')); + const entry = structuredClone(initial.entries[0]); + entry.session.id = 'oc-fixture'; + entry.session.host = entry.session.provider = 'opencode'; + entry.parseSemantics = 'cost-trust-v2-observations-v1'; + entry.dbFile = path.join(dir, 'broken.db'); + cache.entries['opencode://oc-fixture'] = entry; + fs.writeFileSync(entry.dbFile, 'not a database'); + entry.sourceIdentity = selectOpencodeSource({ roots: { opencode: entry.dbFile } }).sourceIdentity; + fs.writeFileSync(file, JSON.stringify(cache)); + const same = run(dir, zones[0]); + assert.ok(same.stable.sessions.some((s) => s.id === 'oc-fixture')); + const changed = run(dir, zones[1]); + assert.equal(changed.sourceHealth.opencode.status, 'degraded'); + assert.equal(changed.sourceHealth.opencode.timezoneCacheEntriesExcluded, 1); + assert.ok(!changed.stable.sessions.some((s) => s.id === 'oc-fixture')); + assert.deepEqual(changed.entries.find((e) => e.session.id === 'oc-fixture').localTimeContext, + entry.localTimeContext, 'carry forward must not relabel the old evidence'); + assert.ok(run(dir, zones[0]).stable.sessions.some((s) => s.id === 'oc-fixture')); +}); + +for (const days of [1, 120]) { + test(`timezone refresh preserves current and previous ${days}-day windows`, (t) => { + const dir = fixture(t); + openCodeFixture(dir); + run(dir, zones[0], '', days); + const changed = run(dir, zones[1], '', days); + fs.unlinkSync(path.join(dir, 'cache.json')); + assert.deepEqual(changed.stable, run(dir, zones[1], '', days).stable); + }); +} + +test('unset TZ uses the resolved machine zone and remains incremental', (t) => { + const dir = fixture(t); + run(dir, undefined); + assert.equal(run(dir, undefined).sourceHealth.codex.diagnostics.cachedFiles, 1); +}); diff --git a/tests/kit/verify-memory-routes.test.mjs b/tests/kit/verify-memory-routes.test.mjs index b1e87d23..cf530a51 100644 --- a/tests/kit/verify-memory-routes.test.mjs +++ b/tests/kit/verify-memory-routes.test.mjs @@ -241,8 +241,11 @@ test('the quick live memory check proves the CLI round trip without starting an assert.ok(!calls.includes('mcp start'), 'the live check stays quick: routing is observed only by the memory-routes proof'); }); -test('--only memory-routes runs the memory proof with the route observation and remembers it as memory', posix, async () => { +test('--only memory-routes runs the memory proof with the route observation and remembers it separately', posix, async () => { const cfg = offlineKitConfig(); + const evidence = await import('../../src/lib/live-check-evidence.mjs'); + evidence.recordLiveCheck({ id: 'memory', status: 'failed', source: 'status-refresh-live', + inputsKey: evidence.liveCheckInputsKey('memory') }); const { results, calls } = await withFakeRuflo('aligned', async () => ({ results: await verify.runLiveChecks({ cfg, cwd: PROJECT, only: ['memory-routes'] }), })); @@ -250,4 +253,21 @@ test('--only memory-routes runs the memory proof with the route observation and assert.equal(results[0].status, 'passed', JSON.stringify(results[0])); assert.ok(calls.includes('mcp start'), 'the named proof observes CLI↔MCP routing'); assert.ok(results[0].entries.some((e) => /see each other's writes/.test(e.text)), JSON.stringify(results[0].entries)); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'passed'); + assert.equal(evidence.readLiveCheck('memory').status, 'passed', + 'the successful CLI round trip refreshes the generic memory row'); +}); + +test('an unavailable route observation is inconclusive even after a successful CLI round trip', posix, async () => { + const cfg = offlineKitConfig(); + const evidence = await import('../../src/lib/live-check-evidence.mjs'); + evidence.recordLiveCheck({ id: 'memory', status: 'failed', source: 'status-refresh-live', + inputsKey: evidence.liveCheckInputsKey('memory') }); + const { results } = await withFakeRuflo('mcp-down', async () => ({ + results: await verify.runLiveChecks({ cfg, cwd: PROJECT, only: ['memory-routes'] }), + })); + assert.equal(results[0].status, 'inconclusive'); + assert.equal(evidence.readLiveCheck('memory-routes').status, 'inconclusive'); + assert.equal(evidence.readLiveCheck('memory').status, 'passed', + 'MCP unavailability does not undo the successful CLI proof'); }); diff --git a/tests/kit/windows-npm-shim.test.mjs b/tests/kit/windows-npm-shim.test.mjs new file mode 100644 index 00000000..fc965df1 --- /dev/null +++ b/tests/kit/windows-npm-shim.test.mjs @@ -0,0 +1,151 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { resolveShim, run } from '../../src/lib/exec.mjs'; +import { callMcpTools } from '../../src/lib/mcp-tool-call.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; + +const templates = Object.fromEntries(['cmd', 'ps1'].map((ext) => [ext, + fs.readFileSync(new URL(`../fixtures/npm-windows-shim/ruflo.${ext}`, import.meta.url), 'utf8')])); +function fixture(t) { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-npm-shim-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + const entry = path.join(root, 'node_modules', 'ruflo', 'bin', 'ruflo.js'); + const manifest = path.join(root, 'node_modules', 'ruflo', 'package.json'); + fs.mkdirSync(path.dirname(entry), { recursive: true }); + fs.writeFileSync(entry, '#!/usr/bin/env node\n'); + fs.writeFileSync(manifest, JSON.stringify({ name: 'ruflo', bin: { ruflo: 'bin/ruflo.js' } })); + for (const ext of ['cmd', 'ps1']) fs.writeFileSync(path.join(root, `ruflo.${ext}`), templates[ext]); + const powershell = path.join(root, 'System32', 'WindowsPowerShell', 'v1.0', 'powershell.exe'); + fs.mkdirSync(path.dirname(powershell), { recursive: true }); fs.writeFileSync(powershell, 'fixture'); + const node = path.join(root, 'node.exe'); fs.writeFileSync(node, 'fixture'); + const env = { PATH: root, PATHEXT: '.EXE;.CMD', SystemRoot: root }; + return { root, entry, manifest, node, powershell, env }; +} + +test('recognized npm shim maps to its declared public bin with literal argv and adjacent Node', (t) => { + const f = fixture(t); + const args = ['mcp', 'start', 'a & b', 'quoted " argument', '$(ignored)', '', '\n']; + assert.deepEqual(resolveShim('ruflo', args, { windows: true, env: f.env }), { + command: f.node, args: [f.entry, ...args], resolved: true, + }); + assert.deepEqual(resolveShim(path.join(f.root, 'ruflo.cmd'), args, { windows: true, env: f.env }), { + command: f.node, args: [f.entry, ...args], resolved: true, + }); + assert.equal(resolveShim('ruflo', args, { windows: true, env: f.env, npmBin: false }).command, f.powershell); + for (const ext of ['cmd', 'ps1']) { + fs.writeFileSync(path.join(f.root, `ruflo.${ext}`), templates[ext].replaceAll('\r\n', '\n').replaceAll('\n', '\r\n')); + } + assert.equal(resolveShim('ruflo', args, { windows: true, env: f.env }).command, f.node, 'CRLF templates remain recognized'); +}); + +test('PATH-selected installation and case-insensitive Node lookup beat current runtime/global guesses', (t) => { + const f = fixture(t); const other = fixture(t); + fs.rmSync(f.node); + const env = { pAtH: `${f.root}${path.delimiter}${other.root}`, pAtHeXt: '.CMD', sYsTeMrOoT: f.root }; + assert.deepEqual(resolveShim('ruflo', ['mcp', 'start'], { windows: true, env }), { + command: other.node, args: [f.entry, 'mcp', 'start'], resolved: true, + }); +}); + +for (const mutation of ['cmd', 'ps1', 'manifest', 'foreign-package', 'undeclared-bin', 'traversal', 'shebang', 'missing-bin', 'missing-node']) { + test(`${mutation} cannot silently bypass a custom or unverified wrapper`, (t) => { + const f = fixture(t); + if (mutation === 'cmd' || mutation === 'ps1') fs.appendFileSync(path.join(f.root, `ruflo.${mutation}`), '\ncustom-behavior\n'); + if (mutation === 'manifest') fs.writeFileSync(f.manifest, 'broken json'); + if (mutation === 'foreign-package') fs.writeFileSync(f.manifest, JSON.stringify({ name: 'other', bin: { ruflo: 'bin/ruflo.js' } })); + if (mutation === 'undeclared-bin') fs.writeFileSync(f.manifest, JSON.stringify({ name: 'ruflo', bin: { other: 'bin/ruflo.js' } })); + if (mutation === 'traversal') fs.writeFileSync(f.manifest, JSON.stringify({ name: 'ruflo', bin: { ruflo: '../other.js' } })); + if (mutation === 'shebang') fs.writeFileSync(f.entry, '#!/usr/bin/env node --require injected\n'); + if (mutation === 'missing-bin') fs.rmSync(f.entry); + if (mutation === 'missing-node') fs.rmSync(f.node); + assert.equal(resolveShim('ruflo', [], { windows: true, env: f.env }).command, f.powershell); + }); +} + +test('custom wrapper first on PATH and native executable precedence remain authoritative', (t) => { + const first = fixture(t); const second = fixture(t); + fs.appendFileSync(path.join(first.root, 'ruflo.ps1'), '\n# user customization\n'); + const env = { ...first.env, PATH: `${first.root}${path.delimiter}${second.root}` }; + assert.equal(resolveShim('ruflo', [], { windows: true, env }).command, first.powershell); + fs.writeFileSync(path.join(first.root, 'ruflo.exe'), 'native'); + assert.deepEqual(resolveShim('ruflo', ['literal'], { windows: true, env }), { + command: path.join(first.root, 'ruflo.exe'), args: ['literal'], resolved: true, + }); +}); + +test('bin symlink escaping its package cannot authorize bypass', { skip: process.platform === 'win32' }, (t) => { + const f = fixture(t); + const outside = path.join(f.root, 'outside.js'); fs.writeFileSync(outside, '#!/usr/bin/env node\n'); + fs.rmSync(f.entry); fs.symlinkSync(outside, f.entry); + assert.equal(resolveShim('ruflo', [], { windows: true, env: f.env }).command, f.powershell); +}); + +test('recognized public entry point answers MCP initialize without stdin EOF', { skip: process.platform === 'win32' }, async (t) => { + const f = fixture(t); + fs.rmSync(f.node); fs.symlinkSync(process.execPath, f.node); + fs.writeFileSync(f.entry, `#!/usr/bin/env node +require('node:readline').createInterface({input:process.stdin}).on('line', line => { + const r=JSON.parse(line); + if(r.method==='initialize') process.stdout.write(JSON.stringify({jsonrpc:'2.0',id:r.id,result:{protocolVersion:'2024-11-05'}})+'\\n'); +}); +`); + const spec = resolveShim('ruflo', ['mcp', 'start'], { windows: true, env: f.env }); + assert.equal(spec.command, f.node); + const result = await callMcpTools({ ...spec, cwd: f.root, env: spawnEnv(f.root), calls: [], timeoutMs: 1000 }); + assert.equal(result.status, 'ok'); +}); + +test('scoped package aliases and string bin declarations require manifest agreement', (t) => { + const f = fixture(t); + fs.writeFileSync(f.manifest, JSON.stringify({ name: 'ruflo', bin: './bin/ruflo.js' })); + assert.equal(resolveShim('ruflo', [], { windows: true, env: f.env }).command, f.node); + const target = 'node_modules/@scope/tool/bin/cli.js'; + const entry = path.join(f.root, ...target.split('/')); + fs.mkdirSync(path.dirname(entry), { recursive: true }); + fs.writeFileSync(entry, '#!/usr/bin/env node\n'); + const manifest = path.join(f.root, 'node_modules', '@scope', 'tool', 'package.json'); + fs.writeFileSync(manifest, JSON.stringify({ name: '@scope/tool', bin: { ruflo: 'bin/cli.js' } })); + for (const ext of ['cmd', 'ps1']) { + const oldTarget = ext === 'cmd' ? 'node_modules\\ruflo\\bin\\ruflo.js' : 'node_modules/ruflo/bin/ruflo.js'; + fs.writeFileSync(path.join(f.root, `ruflo.${ext}`), templates[ext].replaceAll(oldTarget, ext === 'cmd' ? target.replaceAll('/', '\\') : target)); + } + assert.deepEqual(resolveShim('ruflo', [], { windows: true, env: f.env }), { + command: f.node, args: [entry], resolved: true, + }); + fs.writeFileSync(manifest, JSON.stringify({ name: '@scope/tool', bin: 'bin/cli.js' })); + assert.equal(resolveShim('ruflo', [], { windows: true, env: f.env }).command, f.powershell); +}); + +test('run honors a differently cased PATH override without duplicate environment keys', { skip: process.platform === 'win32' }, async (t) => { + const f = fixture(t); + fs.rmSync(f.node); fs.symlinkSync(process.execPath, f.node); + fs.writeFileSync(f.entry, `#!/usr/bin/env node +process.stdout.write(JSON.stringify({argv:process.argv.slice(2),paths:Object.keys(process.env).filter(k=>k.toUpperCase()==='PATH')})); +`); + const args = ['mcp', 'quoted " value', 'a & b', '']; + const result = await run('ruflo', args, { windows: true, env: { + pAtH: f.root, pAtHeXt: '.CMD', sYsTeMrOoT: f.root, + } }); + assert.equal(result.code, 0, result.stderr); + assert.deepEqual(JSON.parse(result.stdout), { argv: args, paths: ['pAtH'] }); +}); + +test('an earlier cmd with no safe sibling does not fall through to another installation', (t) => { + const first = fixture(t); const second = fixture(t); + fs.rmSync(path.join(first.root, 'ruflo.ps1')); + const env = { ...first.env, PATH: `${first.root}${path.delimiter}${second.root}` }; + assert.deepEqual(resolveShim('ruflo', ['mcp'], { windows: true, env }), { + command: 'ruflo', args: ['mcp'], resolved: false, + }); +}); + +test('package symlink escaping the selected installation cannot authorize bypass', { skip: process.platform === 'win32' }, (t) => { + const first = fixture(t); const second = fixture(t); + const pkg = path.join(first.root, 'node_modules', 'ruflo'); + fs.rmSync(pkg, { recursive: true }); + fs.symlinkSync(path.join(second.root, 'node_modules', 'ruflo'), pkg); + assert.equal(resolveShim('ruflo', [], { windows: true, env: first.env }).command, first.powershell); +}); diff --git a/tests/kit/xdg-relative.test.mjs b/tests/kit/xdg-relative.test.mjs new file mode 100644 index 00000000..b0ad4e42 --- /dev/null +++ b/tests/kit/xdg-relative.test.mjs @@ -0,0 +1,111 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { spawnSync } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; +import * as pathModule from '../../src/lib/paths.mjs'; +import { defaultStorageRoots } from '../../src/lib/footprint/storage.mjs'; +import { spawnEnv } from './helpers/home-sandbox.mjs'; +import { tempDir } from './helpers/temp-dir.mjs'; + +const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..'); + +test('xdgBase keeps only absolute XDG values for both path flavors', () => { + assert.equal(typeof pathModule.xdgBase, 'function'); + for (const [p, absolute] of [[path.posix, '/opt/cfg'], [path.win32, 'C:\\cfg']]) { + const fallback = p.join(absolute, 'fallback'); + for (const value of [undefined, '', 'rel/cfg', './cfg']) { + assert.equal(pathModule.xdgBase('XDG_CONFIG_HOME', fallback, { env: { XDG_CONFIG_HOME: value }, p }), fallback); + } + assert.equal(pathModule.xdgBase('XDG_CONFIG_HOME', fallback, { env: { XDG_CONFIG_HOME: absolute }, p }), absolute); + } +}); + +test('Windows deep runtime-log root follows LOCALAPPDATA with a distinct XDG state base', () => { + const home = 'C:\\Users\\Ada'; + const env = { + LOCALAPPDATA: 'C:\\Users\\Ada\\AppData\\Local', + XDG_STATE_HOME: 'D:\\xdg-state', + }; + const expected = 'C:\\Users\\Ada\\AppData\\Local\\agentic-kit\\runtime-debug.log'; + const options = { env, home, platform: 'win32', p: path.win32 }; + assert.equal(pathModule.stateBase(options), env.LOCALAPPDATA); + assert.equal(defaultStorageRoots(options).find((row) => row.id === 'ak-runtime-debug')?.path, expected); +}); + +test('runtime-log writer, known-file reader, and deep root agree with distinct native state bases', (t) => { + const home = tempDir('ak-xdg-state-agreement', t); + const local = path.join(home, 'native-local'); + const xdg = path.join(home, 'xdg-state'); + const pathsUrl = new URL('../../src/lib/paths.mjs', import.meta.url).href; + const footprintUrl = new URL('../../src/lib/footprint/index.mjs', import.meta.url).href; + const storageUrl = new URL('../../src/lib/footprint/storage.mjs', import.meta.url).href; + const script = `import path from 'node:path'; +import { stateBase } from ${JSON.stringify(pathsUrl)}; +import { knownFileSpecs } from ${JSON.stringify(footprintUrl)}; +import { defaultStorageRoots } from ${JSON.stringify(storageUrl)}; +console.log(JSON.stringify({ writer: path.join(stateBase(), 'agentic-kit', 'runtime-debug.log'), + known: knownFileSpecs().find((row) => row.id === 'ak-runtime-debug').path, + deep: defaultStorageRoots().find((row) => row.id === 'ak-runtime-debug').path }));`; + const child = spawnSync(process.execPath, ['--input-type=module', '-e', script], { + cwd: home, + env: spawnEnv(home, { LOCALAPPDATA: local, XDG_STATE_HOME: xdg }), + encoding: 'utf8', + }); + assert.equal(child.status, 0, child.stderr); + const paths = JSON.parse(child.stdout); + const base = process.platform === 'win32' ? local : xdg; + const expected = path.join(base, 'agentic-kit', 'runtime-debug.log'); + assert.deepEqual(paths, { writer: expected, known: expected, deep: expected }); +}); + +test('relative XDG values cannot redirect live paths or tool root discovery into cwd', { skip: process.platform === 'win32' }, (t) => { + const home = tempDir('ak-xdg-relative-home', t); + const cwd = path.join(home, 'work'); + fs.mkdirSync(cwd); + const pathsUrl = new URL('../../src/lib/paths.mjs', import.meta.url).href; + const footprintUrl = new URL('../../src/lib/footprint/index.mjs', import.meta.url).href; + const script = `import * as paths from ${JSON.stringify(pathsUrl)}; +import { knownFileSpecs } from ${JSON.stringify(footprintUrl)}; +console.log(JSON.stringify([paths.configDir(), paths.evidenceDir(), + ...knownFileSpecs().map((row) => row.path), ...paths.toolInternalDirs()]));`; + const child = spawnSync(process.execPath, ['--input-type=module', '-e', script], { + cwd, + env: spawnEnv(home, { + XDG_CONFIG_HOME: 'rel/cfg', XDG_STATE_HOME: 'rel/state', + XDG_DATA_HOME: 'rel/data', XDG_CACHE_HOME: 'rel/cache', + }), + encoding: 'utf8', + }); + assert.equal(child.status, 0, child.stderr); + const paths = JSON.parse(child.stdout); + assert.ok(paths.length > 10); + for (const candidate of paths) { + assert.ok(path.isAbsolute(candidate), candidate); + assert.ok(candidate.startsWith(`${home}${path.sep}`), candidate); + assert.doesNotMatch(candidate, /(?:^|[/\\])rel(?:[/\\]|$)/); + } + assert.ok(paths.includes(path.join(home, '.local', 'state', 'agentic-kit', 'runtime-debug.log'))); +}); + +test('new source readers use the shared XDG base validator', () => { + const allowed = new Set(['src/lib/paths.mjs', 'src/templates/statusline-footer.cjs', + 'src/lib/adapters/manifest.mjs']); + const matches = []; + const visit = (dir) => { + for (const entry of fs.readdirSync(dir, { withFileTypes: true })) { + const at = path.join(dir, entry.name); + if (entry.isDirectory()) { visit(at); continue; } + if (!entry.isFile()) continue; + const relative = path.relative(root, at).split(path.sep).join('/'); + if (allowed.has(relative)) continue; + const source = fs.readFileSync(at, 'utf8') + .replace(/\/\*[\s\S]*?\*\//g, '') + .replace(/\/\/[^\n]*/g, ''); + if (/\b(?:process\.)?env\.XDG_[A-Z_]+|env\[['"]XDG_/.test(source)) matches.push(relative); + } + }; + visit(path.join(root, 'src')); + assert.deepEqual(matches, []); +}); diff --git a/tests/live/aqe-live-lock-conformance.test.mjs b/tests/live/aqe-live-lock-conformance.test.mjs new file mode 100644 index 00000000..bf5e6f2b --- /dev/null +++ b/tests/live/aqe-live-lock-conformance.test.mjs @@ -0,0 +1,150 @@ +// Opt-in native contract for agentic-qe#574. No installed package is patched. +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { createHash } from 'node:crypto'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { pathToFileURL } from 'node:url'; +import { createProcessScope } from './aqe-live-lock-process.mjs'; + +const required = process.env.AK_AQE_LOCK_LIVE === '1'; +const digest = (file) => createHash('sha256').update(fs.readFileSync(file)).digest('hex'); +const pause = (ms) => new Promise((resolve) => setTimeout(resolve, ms)); + +function installedRoot() { + if (process.env.AK_AQE_PACKAGE_ROOT) { + assert.ok(path.isAbsolute(process.env.AK_AQE_PACKAGE_ROOT), 'package root must be absolute'); + return fs.realpathSync(process.env.AK_AQE_PACKAGE_ROOT); + } + for (const dir of (process.env.PATH ?? '').split(path.delimiter)) { + const bin = path.join(dir, 'aqe'); + if (dir && fs.existsSync(bin)) { + const root = path.resolve(path.dirname(fs.realpathSync(bin)), '../..'); + if (fs.existsSync(path.join(root, 'package.json'))) return root; + } + } + throw new Error('AQE missing: set AK_AQE_PACKAGE_ROOT to absolute installed package root'); +} + +function childEnv(root, project) { + const home = path.join(root, 'home'); + const tmp = path.join(root, 'tmp'); + fs.mkdirSync(home); fs.mkdirSync(tmp); + return { + PATH: process.env.PATH ?? '', + ...(process.platform === 'win32' ? { + SystemRoot: process.env.SystemRoot ?? 'C:\\Windows', + ComSpec: process.env.ComSpec ?? 'C:\\Windows\\System32\\cmd.exe', + PATHEXT: process.env.PATHEXT ?? '.COM;.EXE;.BAT;.CMD', + } : {}), + HOME: home, USERPROFILE: home, TMPDIR: tmp, TMP: tmp, TEMP: tmp, + LANG: 'en_US.UTF-8', CI: '1', NO_COLOR: '1', + XDG_CONFIG_HOME: path.join(home, 'config'), XDG_STATE_HOME: path.join(home, 'state'), + XDG_DATA_HOME: path.join(home, 'data'), XDG_CACHE_HOME: path.join(home, 'cache'), + APPDATA: path.join(home, 'appdata'), LOCALAPPDATA: path.join(home, 'localappdata'), + CODEX_HOME: path.join(home, 'codex'), CLAUDE_CONFIG_DIR: path.join(home, 'claude'), + HERMES_HOME: path.join(home, 'hermes'), npm_config_prefix: path.join(home, 'npm-prefix'), + npm_config_cache: path.join(home, 'npm-cache'), + MISE_DATA_DIR: path.join(home, 'mise-data'), MISE_CONFIG_DIR: path.join(home, 'mise-config'), + MISE_CACHE_DIR: path.join(home, 'mise-cache'), AQE_PROJECT_ROOT: project, + AQE_MEMORY_PATH: path.join(project, '.agentic-qe', 'memory.db'), + AQE_STORAGE_PATH: path.join(project, '.agentic-qe'), + CLAUDE_FLOW_MEMORY_PATH: path.join(project, '.swarm'), + CLAUDE_FLOW_DB_PATH: path.join(project, '.swarm', 'memory.db'), + RUFLO_DAEMON_AUTOSTART: '0', + }; +} + +async function bounded(scope, file, args, options, limit) { + const run = scope.launch(process.execPath, [file, ...args], options); + return scope.wait(run, limit); +} + +function snapshot(store) { + return Object.fromEntries(fs.readdirSync(store).filter((n) => n.startsWith('patterns.rvf')) + .sort().map((name) => { + const file = path.join(store, name); + return [name, { sha256: digest(file), bytes: fs.statSync(file).size }]; + })); +} + +function check(label, result, owner, before, store) { + const output = result.stdout + result.stderr; + assert.equal(result.timedOut, false, `${label} timed out`); + assert.equal(result.code, 0, `${label} exited ${result.code}: ${output.slice(-1200)}`); + assert.equal(owner.closed, false, `native holder closed during ${label}`); + assert.equal(owner.child.exitCode, null, `native holder exited during ${label}`); + assert.equal(owner.child.signalCode, null, `native holder signaled during ${label}`); + assert.ifError(owner.error); + assert.match(output, /is locked by a live process/, `${label} missed live-lock warning`); + assert.match(output, /LockHeld|0x0300/, `${label} missed LockHeld fallback`); + assert.doesNotMatch(output, /FsyncFailed|0x0303/, `${label} emitted FsyncFailed`); + assert.deepEqual(snapshot(store), before, `${label} changed RVF bytes`); + assert.ok(!fs.readdirSync(store).some((n) => n.includes('.corrupt-')), `${label} quarantined RVF`); +} + +test('installed AQE degrades under a live native RVF lock without changing the store', + { skip: !required && 'set AK_AQE_LOCK_LIVE=1 for native proof', timeout: 420_000 }, async (t) => { + const root = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'ak-aqe-live-lock-'))); + const scope = createProcessScope(t.signal); + t.after(async () => { + try { + await scope.closeAll(); + } catch (error) { + throw new Error(`owned child closure unverified; retained ${root}`, { cause: error }); + } + fs.rmSync(root, { recursive: true, force: true, maxRetries: 3 }); + }); + const project = path.join(root, 'project'); fs.mkdirSync(project); + fs.writeFileSync(path.join(project, 'package.json'), '{"name":"aqe-live-lock-probe","version":"1.0.0","type":"module"}\n'); + const env = childEnv(root, project); + const packageRoot = installedRoot(); + const pkg = JSON.parse(fs.readFileSync(path.join(packageRoot, 'package.json'), 'utf8')); + assert.equal(pkg.name, 'agentic-qe'); + if (process.env.AK_AQE_EXPECTED_VERSION) assert.equal(pkg.version, process.env.AK_AQE_EXPECTED_VERSION); + const entry = path.join(packageRoot, 'dist', 'cli', 'bundle.js'); + const adapter = path.join(packageRoot, 'dist', 'integrations', 'ruvector', 'shared-rvf-adapter.js'); + assert.ok(fs.existsSync(entry) && fs.existsSync(adapter), 'installed AQE CLI and adapter required'); + const sourceHashes = { entry: digest(entry), adapter: digest(adapter) }; + const options = { cwd: project, env }; + const init = await bounded(scope, entry, ['init', '--minimal', '--auto'], options, 150_000); + assert.equal(init.timedOut, false); + assert.equal(init.code, 0, `aqe init failed: ${(init.stdout + init.stderr).slice(-1200)}`); + const store = path.join(project, '.agentic-qe'); + const ready = path.join(root, 'ready'); + const holderFile = path.join(root, 'holder.mjs'); + const moduleUrl = pathToFileURL(adapter).href; + fs.writeFileSync(holderFile, `import { createRequire } from 'node:module'; import fs from 'node:fs';\nglobalThis.require=createRequire(${JSON.stringify(adapter)});\nconst {getSharedRvfAdapter}=await import(${JSON.stringify(moduleUrl)});\nglobalThis.hold=getSharedRvfAdapter(${JSON.stringify(store)},384);\nif(!globalThis.hold)process.exit(2);\nfs.writeFileSync(${JSON.stringify(ready)},String(process.pid));\nconst timer=setInterval(()=>{if(!globalThis.hold)process.exit(3)},1000);\nprocess.on('SIGTERM',()=>{clearInterval(timer);globalThis.hold.close();process.exit(0)});\n`); + const owner = scope.launch(process.execPath, [holderFile], options); + try { + const deadline = Date.now() + 30_000; + while (!fs.existsSync(ready) && !owner.closed && !owner.error && owner.child.exitCode === null + && owner.child.signalCode === null && Date.now() < deadline) await pause(50); + assert.ifError(owner.error); + assert.ok(fs.existsSync(ready), 'holder did not signal ready'); + assert.equal(Number(fs.readFileSync(ready, 'utf8')), owner.child.pid, 'holder PID mismatch'); + assert.equal(owner.closed, false, 'holder closed after ready'); + assert.equal(owner.child.exitCode, null, 'holder exited after ready'); + assert.equal(owner.child.signalCode, null, 'holder signaled after ready'); + const before = snapshot(store); + assert.ok(before['patterns.rvf'] && before['patterns.rvf.lock'], 'native RVF and lock required'); + const status = await bounded(scope, entry, ['status'], options, 150_000); + check('aqe status', status, owner, before, store); + const challengerFile = path.join(root, 'challenger.mjs'); + fs.writeFileSync(challengerFile, `import {createRequire} from 'node:module';\nglobalThis.require=createRequire(${JSON.stringify(adapter)});\nconst {getSharedRvfAdapter}=await import(${JSON.stringify(moduleUrl)});\nconst adapter=getSharedRvfAdapter(${JSON.stringify(store)},384);\nconsole.log(JSON.stringify({fallback:adapter===null}));\nif(adapter){adapter.close();process.exitCode=2}\n`); + const challenger = await bounded(scope, challengerFile, [], options, 30_000); + check('shipped adapter', challenger, owner, before, store); + assert.match(challenger.stdout, /"fallback":true/, 'adapter did not fall back'); + console.log(JSON.stringify({ aqeVersion: pkg.version, platform: process.platform, + node: process.version, sourceHashes, ownerPid: owner.child.pid, + status: { exit: status.code, liveLock: true, lockHeld: true, fsyncFailed: false }, + adapter: { exit: challenger.code, fallback: true, liveLock: true, lockHeld: true, fsyncFailed: false }, + before, after: snapshot(store) })); + } finally { + const closed = await scope.stop(owner, 10_000); + if (!t.signal.aborted) { + assert.equal(closed.code, 0, `holder failed to close: ${closed.stderr.slice(-1000)}`); + } + } + }); diff --git a/tests/live/aqe-live-lock-process.mjs b/tests/live/aqe-live-lock-process.mjs new file mode 100644 index 00000000..433ab8a3 --- /dev/null +++ b/tests/live/aqe-live-lock-process.mjs @@ -0,0 +1,107 @@ +import { spawn } from 'node:child_process'; +import { acquireRunRootHold, releaseRunRootHold } from '../../scripts/run-roots.mjs'; +import { envValue } from '../kit/helpers/home-sandbox.mjs'; + +const CLOSE_LIMIT_MS = 10_000; + +function closedWithin(run, ms) { + if (run.closed) return Promise.resolve(run.result); + let timer; + return Promise.race([ + run.done, + new Promise((_, reject) => { + timer = setTimeout(() => reject(new Error(`call-owned child ${run.child.pid ?? 'unspawned'} did not close`)), ms); + }), + ]).finally(() => clearTimeout(timer)); +} + +export function createProcessScope(signal, { closeLimitMs = CLOSE_LIMIT_MS, platform = process.platform } = {}) { + // Register uncertainty with the enclosing guarded runner before any child starts. + const hold = acquireRunRootHold(); + const runs = new Set(); + let closed = false; + let closing = false; + const killLive = () => { + for (const run of runs) if (!run.closed) run.child.kill('SIGKILL'); + }; + signal.addEventListener('abort', killLive, { once: true }); + + function launch(command, args, options) { + if (closing || signal.aborted) throw Error('cannot launch after process scope closing or aborted'); + if (!options?.env || typeof options.env !== 'object' || Array.isArray(options.env)) { + throw Error('call-owned child requires an explicit sandbox env'); + } + const env = options.env; + const missing = (key) => { + const value = envValue(env, key, platform); + return typeof value !== 'string' || !value; + }; + const required = ['HOME', 'USERPROFILE', 'TMPDIR', 'TEMP', 'TMP', 'XDG_CONFIG_HOME', + 'XDG_STATE_HOME', 'APPDATA', 'LOCALAPPDATA']; + if (required.some(missing)) { + throw Error('call-owned child requires sandbox home, temp, and state env'); + } + if (platform === 'win32' && ['SystemRoot', 'ComSpec', 'PATHEXT'].some(missing)) { + throw Error('call-owned child requires Windows process env'); + } + const child = spawn(command, args, { ...options, env, stdio: ['ignore', 'pipe', 'pipe'] }); + const run = { child, closed: false, error: null, stdout: '', stderr: '', done: null, result: null }; + runs.add(run); + child.stdout.setEncoding('utf8'); child.stderr.setEncoding('utf8'); + child.stdout.on('data', (s) => { run.stdout += s; }); + child.stderr.on('data', (s) => { run.stderr += s; }); + // Resolve on close even after spawn error. The caller can inspect error immediately + // while polling readiness, and no delayed rejection can go unhandled. + run.done = new Promise((resolve) => { + child.once('error', (error) => { run.error = error; }); + child.once('close', (code, childSignal) => { + run.closed = true; + run.result = { code, signal: childSignal, stdout: run.stdout, stderr: run.stderr }; + resolve(run.result); + }); + }); + if (signal.aborted) child.kill('SIGKILL'); + return run; + } + + async function wait(run, limit) { + let timedOut = false; + const timer = setTimeout(() => { + timedOut = true; + if (!run.closed) run.child.kill('SIGKILL'); + }, limit); + try { + const result = await closedWithin(run, limit + CLOSE_LIMIT_MS); + if (run.error) throw run.error; + return { ...result, timedOut }; + } finally { clearTimeout(timer); } + } + + async function stop(run, grace = CLOSE_LIMIT_MS) { + if (!run.closed) run.child.kill('SIGTERM'); + let timer; + try { + await Promise.race([ + run.done, + new Promise((resolve) => { timer = setTimeout(resolve, grace); }), + ]); + } finally { + clearTimeout(timer); + if (!run.closed) run.child.kill('SIGKILL'); + } + return closedWithin(run, CLOSE_LIMIT_MS); + } + + async function closeAll() { + if (closed) return; + closing = true; + killLive(); + // A failed close keeps the caller's temporary root intact for diagnosis. + await Promise.all([...runs].map((run) => closedWithin(run, closeLimitMs))); + releaseRunRootHold(hold); + closed = true; + signal.removeEventListener('abort', killLive); + } + + return { launch, wait, stop, closeAll }; +} diff --git a/tests/live/ruflo-windows-diagnostic-process.mjs b/tests/live/ruflo-windows-diagnostic-process.mjs new file mode 100644 index 00000000..5ec1b20a --- /dev/null +++ b/tests/live/ruflo-windows-diagnostic-process.mjs @@ -0,0 +1,103 @@ +// Test-only exact-environment launcher. Every attempt contributes independently +// to the scope's cleanup receipt; uncertainty is sticky until handoff. +import { spawn } from 'node:child_process'; +import { acquireRunRootHold, releaseRunRootHold } from '../../scripts/run-roots.mjs'; + +const delay = (ms) => new Promise((resolve) => setTimeout(resolve, ms)); + +export function createDiagnosticScope({ spawnFn = spawn, platform = process.platform, + killTimeoutMs = 3000, closeTimeoutMs = 5000, + acquireHold = acquireRunRootHold, releaseHold = releaseRunRootHold } = {}) { + const hold = acquireHold(); + const receipts = []; + let released = false; + + async function launch(spec, { cwd, env, timeoutMs, input = '', endInput = true, + until = () => false, tree = true }) { + if (released) throw Error('diagnostic scope already released'); + if (!env || typeof env !== 'object') throw Error('explicit diagnostic environment required'); + const result = { code: null, signal: null, stdout: '', stderr: '', error: null, + timedOut: false, overflow: false, matched: false, cleanupComplete: false, elapsedMs: 0 }; + receipts.push(result); + let child; + let closed = false; + let bytes = 0; + const started = Date.now(); + const live = () => child?.pid && child.exitCode === null && child.signalCode === null; + const capture = (key, chunk) => { + bytes += chunk.length; + result.overflow ||= bytes > 65536; + result[key] = (result[key] + chunk.toString('utf8')).slice(0, 16384); + }; + const waitClose = async () => { + const deadline = Date.now() + closeTimeoutMs; + while (!closed && Date.now() < deadline) await delay(10); + }; + try { + child = spawnFn(spec.command, spec.args, { + cwd, env, shell: false, stdio: ['pipe', 'pipe', 'pipe'], + detached: tree && platform !== 'win32', + }); + child.on('error', (e) => { result.error = e.message; }); + child.once('close', (code, signal) => { + closed = true; result.code = code; result.signal = signal; + }); + child.stdout.on('data', (chunk) => capture('stdout', chunk)); + child.stderr.on('data', (chunk) => capture('stderr', chunk)); + child.stdin.on('error', (e) => { result.error = e.message; }); + child.stdin.write(input); + if (endInput) child.stdin.end(); + while (!closed && !result.error && !result.overflow && Date.now() - started < timeoutMs) { + result.matched = until(result.stdout); + // EOF comparisons must observe natural closure. Stopping on the + // first response can race a server already exiting after stdin end. + if (result.matched && !endInput) break; + await delay(10); + } + result.matched ||= until(result.stdout); + result.timedOut = !closed && !(result.matched && !endInput) && !result.error && !result.overflow; + } catch (error) { + result.error = error.message; + } finally { + // Never throw before cleanup. Failed tree termination is uncertainty even + // if the direct-child fallback subsequently closes its own pipes. + let treeStopped = closed || !child; + if (child && !closed) { + if (live()) { + if (tree && platform === 'win32') { + const killed = await launch({ command: 'taskkill.exe', args: ['/PID', String(child.pid), '/T', '/F'] }, + { cwd, env, timeoutMs: killTimeoutMs, tree: false }); + treeStopped = killed.cleanupComplete && killed.code === 0 && !killed.timedOut && !killed.error; + } else { + try { + if (tree) process.kill(-child.pid, 'SIGKILL'); + else child.kill('SIGKILL'); + treeStopped = true; + } catch { treeStopped = false; } + } + if (!treeStopped && live()) { + try { child.kill('SIGKILL'); } catch { /* preserve uncertainty */ } + } + } + await waitClose(); + } + result.cleanupComplete = treeStopped && (closed || !child); + if (child && !closed) { + // Close local handles and release event-loop references without claiming + // the process tree exited. The scope hold remains active. + child.stdin.destroy(); child.stdout.destroy(); child.stderr.destroy(); + child.unref(); + } + result.elapsedMs = Date.now() - started; + } + return result; + } + + function release() { + if (receipts.some((receipt) => !receipt.cleanupComplete)) return false; + if (!released) releaseHold(hold); + released = true; + return true; + } + return { launch, release, receipts }; +} diff --git a/tests/live/ruflo-windows-transport.test.mjs b/tests/live/ruflo-windows-transport.test.mjs new file mode 100644 index 00000000..d4c91e08 --- /dev/null +++ b/tests/live/ruflo-windows-transport.test.mjs @@ -0,0 +1,95 @@ +// Opt-in diagnosis only. A backend response never substitutes for public routing +// proof. Run alongside (not instead of) ruflo-memory-routing.test.mjs. +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import crypto from 'node:crypto'; +import { spawnEnv, envValue } from '../kit/helpers/home-sandbox.mjs'; +import { resolveShim } from '../../src/lib/exec.mjs'; +import { createDiagnosticScope } from './ruflo-windows-diagnostic-process.mjs'; + +const sha = (file) => crypto.createHash('sha256').update(fs.readFileSync(file)).digest('hex'); +const initialized = (stdout) => stdout.split('\n').some((line) => { + try { const r = JSON.parse(line); return r.id === 1 && Boolean(r.result?.protocolVersion); } + catch { return false; } +}); +const input = `${JSON.stringify({ jsonrpc: '2.0', id: 1, method: 'initialize', params: { + protocolVersion: '2024-11-05', capabilities: {}, clientInfo: { name: 'ak-native-diagnostic', version: '1' }, +} })}\n`; + +test('diagnose installed Ruflo Windows public shim versus installed bin transport', { + skip: process.env.AK_RUFLO_WINDOWS_DIAGNOSTIC !== '1', timeout: 180000, +}, async () => { + assert.equal(process.platform, 'win32', 'diagnostic requires native Windows'); + const packageRoot = process.env.AK_RUFLO_PACKAGE_ROOT; + assert.ok(packageRoot && path.isAbsolute(packageRoot), 'explicit installed Ruflo package root required'); + const packageFile = path.join(packageRoot, 'package.json'); + const pkg = JSON.parse(fs.readFileSync(packageFile, 'utf8')); + assert.equal(pkg.name, 'ruflo'); + const bin = path.resolve(packageRoot, typeof pkg.bin === 'string' ? pkg.bin : pkg.bin.ruflo); + assert.ok(bin.startsWith(`${fs.realpathSync(packageRoot)}${path.sep}`), 'bin belongs to installed package'); + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ak-ruflo-win-diagnostic-')); + const scope = createDiagnosticScope(); + let clean = false; + try { + const bootstrap = {}; + for (const key of ['PATH', 'SystemRoot', 'WINDIR', 'ComSpec', 'PATHEXT']) { + const value = envValue(process.env, key); + if (value) bootstrap[key] = value; + } + const env = spawnEnv(root, { + CLAUDE_FLOW_DB_PATH: path.join(root, '.swarm', 'memory.db'), + CLAUDE_FLOW_MEMORY_PATH: path.join(root, '.swarm'), RUFLO_DAEMON_AUTOSTART: '0', + }, { env: bootstrap }); + fs.writeFileSync(path.join(root, 'claude-flow.config.json'), JSON.stringify({ daemon: { autostart: false } })); + const invocation = resolveShim('ruflo', ['mcp', 'start'], { env }); + assert.equal(invocation.resolved, true, 'installed public shim must resolve'); + const powershell = resolveShim('ruflo', ['mcp', 'start'], { env, npmBin: false }); + const scriptIndex = powershell.args.indexOf('-File'); + assert.ok(scriptIndex >= 0, 'receipt requires the native PowerShell transport'); + const shim = powershell.args[scriptIndex + 1]; + const cmdShim = shim.slice(0, -4) + '.cmd'; + const version = await scope.launch(resolveShim('ruflo', ['--version'], { env }), + { cwd: root, env, timeoutMs: 10000 }); + console.log(JSON.stringify({ diagnostic: 'inputs', node: process.version, uv: process.versions.uv, + platform: process.platform, packageVersion: pkg.version, version, + packageSha: sha(packageFile), bin, binSha: sha(bin), invocation, powershell, + cmdSha: sha(cmdShim), cmdSource: fs.readFileSync(cmdShim, 'utf8').slice(0, 8192), + shimSha: sha(shim), shimSource: fs.readFileSync(shim, 'utf8').slice(0, 8192), + sourceSha: sha(new URL(import.meta.url)), + launcherSha: sha(new URL('./ruflo-windows-diagnostic-process.mjs', import.meta.url)) })); + assert.equal(version.cleanupComplete, true, 'version launch cleanup incomplete'); + // Capture the original fixture's native argument forwarding without + // creating any descendants: this isolates quoting from tree ownership. + const legacyShim = path.join(root, 'legacy-fixture.ps1'); + const marker = path.join(root, 'legacy-marker'); + fs.writeFileSync(legacyShim, `& '${process.execPath.replaceAll("'", "''")}' -e $args[0]\nexit $LASTEXITCODE\n`); + const legacyCode = `const {spawn}=require('node:child_process'); + require('node:fs').writeFileSync(${JSON.stringify(marker)},JSON.stringify([process.pid]));`; + const legacy = await scope.launch({ command: powershell.command, args: [ + ...powershell.args.slice(0, scriptIndex + 1), legacyShim, legacyCode, + ] }, { cwd: root, env, timeoutMs: 5000 }); + console.log(JSON.stringify({ diagnostic: 'legacy-multiline-node-e', + marker: fs.existsSync(marker), ...legacy })); + assert.equal(legacy.cleanupComplete, true, 'legacy launch cleanup incomplete'); + for (const [transport, spec, eof] of [ + ['public-open-stdin', invocation, false], + ['public-eof', invocation, true], + ['powershell-open-stdin', powershell, false], + ['powershell-eof', powershell, true], + ['installed-bin-open-stdin', { command: process.execPath, args: [bin, 'mcp', 'start'] }, false], + ]) { + const observation = await scope.launch(spec, { cwd: root, env, input, endInput: eof, + timeoutMs: 15000, until: initialized }); + console.log(JSON.stringify({ diagnostic: transport, initialized: observation.matched, endInput: eof, ...observation })); + assert.equal(observation.cleanupComplete, true, `${transport} cleanup incomplete`); + } + clean = scope.release(); + assert.equal(clean, true, 'every diagnostic launch must prove cleanup before root removal'); + } finally { + if (clean) fs.rmSync(root, { recursive: true, force: true }); + else console.error(`diagnostic root retained: ${root}`); + } +}); diff --git a/tests/quality/comment-label-guard.test.mjs b/tests/quality/comment-label-guard.test.mjs new file mode 100644 index 00000000..f26e1132 --- /dev/null +++ b/tests/quality/comment-label-guard.test.mjs @@ -0,0 +1,100 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; +import { execFileSync } from 'node:child_process'; +import { readFileSync } from 'node:fs'; +import { fileURLToPath } from 'node:url'; +import { inspectCommentLabels } from '../helpers/comment-label-guard.mjs'; +import { spawnEnv } from '../kit/helpers/home-sandbox.mjs'; +import { tempDir } from '../kit/helpers/temp-dir.mjs'; + +const root = fileURLToPath(new URL('../../', import.meta.url)); + +test('finds transient labels in all comment positions and reports their lines', () => { + const source = '// Task 1\nfunction f() { /* Fix round 2 */ }\nconst x = 1; // final-review fix\n/* Branch 6a shared Task */\n'; + assert.deepEqual(inspectCommentLabels(source).map(({ line, kind }) => [line, kind]), + [[1, 'comment'], [2, 'comment'], [3, 'comment'], [4, 'comment']]); +}); + +test('finds ordinary and member test titles including template segments', () => { + for (const callee of ['test', 'it', 'describe', 'suite', 'test.skip', 'describe.only', 't.test', "test['todo']", 'test.only.each([])']) { + assert.equal(inspectCommentLabels(`${callee}('Task 2.3a: behavior', () => {});`).length, 1, callee); + } + assert.equal(inspectCommentLabels('test(`Fix round 3`, () => {});').length, 1); + assert.equal(inspectCommentLabels('test(`Task 3: ${value}`, () => {});').length, 1); +}); + +test('ignores ordinary member calls and parameterized test data', () => { + for (const source of [ + '/Task/.test("Task 1");', + 'const pattern = /Task/; pattern.test("Task 1");', + 'const t = /Task/; t.test("Task 1");', + 'test.log("Task 1");', + 'describe.toString("Task 1");', + 'other.test("Task 1");', + 'test.each("Task 1")("ordinary title", () => {});', + ]) assert.deepEqual(inspectCommentLabels(source), [], source); +}); + +test('parameterized callback arguments are table data, not Node contexts', () => { + for (const api of ['test.each', 'test.only.each', "test['each']"]) { + assert.deepEqual(inspectCommentLabels(`${api}([/Task/])('matches text', (pattern) => pattern.test('Task 1'));`), [], api); + const findings = inspectCommentLabels(`${api}([/Task/])('Task 2: matches text', (t) => t.test('Task 1'));`); + assert.equal(findings.length, 1, api); + assert.equal(findings[0].text, 'Task 2: matches text'); + } + assert.equal(inspectCommentLabels('test.each([1])("row", () => { test("outer", (context) => context.test("Task 1", () => {})); });').length, 1); +}); + +test('recognizes lexically declared nested Node test contexts', () => { + assert.equal(inspectCommentLabels('test("outer", async (context) => { await context.test("Task 1", async (child) => { await child.test("Task 2", () => {}); }); });').length, 2); + assert.deepEqual(inspectCommentLabels('other("outer", (context) => { context.test("Task 1"); });'), []); +}); + +test('ignores strings, URLs, regular expressions, template text and durable audit IDs', () => { + const source = String.raw` +const url = 'https://example.test/Task 1'; +const text = "// Task 2"; +const pattern = /\/\/ Task 3/; +const template = \`/* Task 4 */ \${'// Task 5'}\`; +// B3-D2 / B5-D1 / ADR-0063 / agentic-qe#759 +other('Task 6'); +`.replaceAll('\\`', '`').replaceAll('\\${', '${'); + assert.deepEqual(inspectCommentLabels(source), []); +}); + +test('finds comments inside template expressions without flagging raw segments', () => { + assert.equal(inspectCommentLabels('const x = `Task 1 ${ /* Task 2 */ 1 } // Task 3`;').length, 1); +}); + +test('handles JSDoc, empty files with comments, CRLF, division and regex boundaries', () => { + assert.equal(inspectCommentLabels('/** Task 1 */\r\nconst x = /Task 2/.test("x") / 2; // Task 3').length, 2); + assert.equal(inspectCommentLabels('/* Task 1 */').length, 1); + assert.equal(inspectCommentLabels('function f() {} /* Task 1 */').length, 1); +}); + +test('rejects malformed JavaScript rather than silently overlooking labels', () => { + assert.throws(() => inspectCommentLabels('const x = "unterminated'), /Cannot parse/); +}); + +test('inline lint directives cannot suppress inspection', () => { + assert.equal(inspectCommentLabels('/* eslint-disable */\n// Task 1').length, 1); + assert.equal(inspectCommentLabels('/* eslint-disable labels/references */\ntest("Task 2", () => {});').length, 1); +}); + +test('parses CommonJS return and module imports using the matching source mode', () => { + assert.equal(inspectCommentLabels('return; // Task 1', 'fixture.cjs').length, 1); + assert.equal(inspectCommentLabels('import x from "x"; // Task 1', 'fixture.mjs').length, 1); + assert.throws(() => inspectCommentLabels('return;', 'fixture.mjs'), /Cannot parse/); +}); + +test('tracked JavaScript comments and test titles use durable references', (t) => { + const home = tempDir('ak-comment-label-git', t); + const files = execFileSync('git', ['ls-files', '-z', '--', 'src', 'scripts', 'bin', 'tests'], { + cwd: root, encoding: 'utf8', env: spawnEnv(home), + }) + .split('\0').filter(file => /\.(?:mjs|cjs|js)$/.test(file)); + assert.ok(files.length > 0); + const findings = files.flatMap(file => inspectCommentLabels(readFileSync(new URL(`../../${file}`, import.meta.url), 'utf8'), file) + .map(hit => `${file}:${hit.line}: ${hit.kind}: ${hit.text}`)); + assert.deepEqual(findings, []); +}); diff --git a/tests/ui/dashboard-ui.mjs b/tests/ui/dashboard-ui.mjs index 5fff42c7..8ed62e36 100644 --- a/tests/ui/dashboard-ui.mjs +++ b/tests/ui/dashboard-ui.mjs @@ -789,10 +789,9 @@ const MAINTENANCE_PAYLOAD = { receipts: [], }; -let chainedMaintenanceScans = 0; const MAINTENANCE_STUB = { async report() { return MAINTENANCE_PAYLOAD; }, - async scan() { chainedMaintenanceScans += 1; return MAINTENANCE_PAYLOAD; }, + async scan() { return MAINTENANCE_PAYLOAD; }, async plan() { return {}; }, }; @@ -1133,6 +1132,12 @@ async function main() { console.log(`\ncorpus: ${REAL ? 'REAL (~/.claude, ~/.codex)' : 'fixtures (deterministic)'}`); console.log(`cache : ${cachePath} (temp — your real index is untouched)\n`); + const refreshStageGates = new Map(); + const refreshStages = Object.fromEntries(['machine', 'maintenance', 'inventory', 'live', 'local'].map((id) => [id, async () => { + const gate = refreshStageGates.get(id); + if (gate) await gate; + return { ok: true }; + }])); const srv = await startDashboard({ port: 0, fetchStatus: STATUS_STUB, @@ -1145,6 +1150,7 @@ async function main() { modelScopeKey: 'ab'.repeat(32), system: SYSTEM_STUB, maintenance: MAINTENANCE_STUB, + refreshStages, }); const ORIGIN = new URL(srv.url).origin; const modelHeaders = { 'x-dash-token': srv.token }; @@ -1184,11 +1190,9 @@ async function main() { const group = document.getElementById('secondary-system')?.getBoundingClientRect(); const tabs = document.getElementById('system-seg')?.getBoundingClientRect(); const status = document.getElementById('system-freshness')?.getBoundingClientRect(); - const button = document.getElementById('sys-rescan'); return { statusText: document.getElementById('sys-asof')?.innerText, running: document.getElementById('system-freshness')?.getAttribute('data-running'), - buttonHidden: button?.hidden, statusBesideTabs: !!status && !!tabs && status.top < tabs.bottom && status.bottom > tabs.top && status.left >= tabs.right + 12, trailingSegmentSpace: Math.abs((tabs?.right ?? 0) @@ -1199,10 +1203,9 @@ async function main() { documentFits: document.documentElement.scrollWidth <= globalThis.innerWidth, }; }); - check('running full-scan progress sits beside a content-width System menu on wide screens', + check('running machine measurement progress sits beside a content-width System menu on wide screens', runningScanLayout.running === '1' - && /Full scan running.*Ranking disk use.*15 of 15/.test(runningScanLayout.statusText ?? '') - && runningScanLayout.buttonHidden === true + && /Machine measurement running.*Ranking disk use.*15 of 15/.test(runningScanLayout.statusText ?? '') && runningScanLayout.statusBesideTabs && runningScanLayout.trailingSegmentSpace < 6 && runningScanLayout.statusInsideGroup @@ -1226,7 +1229,7 @@ async function main() { ? getComputedStyle(document.getElementById('sys-asof')).whiteSpace : null, }; }); - check('narrow System navigation scrolls internally while scan status stays in its own rail', + check('narrow System navigation scrolls internally while measurement status stays in its own rail', narrowScanLayout.documentFits && narrowScanLayout.tabsScrollInternally && narrowScanLayout.statusBelowTabs @@ -1264,9 +1267,37 @@ async function main() { `blocked-storage startup was ${JSON.stringify(blockedStorageStartup)} with status ${blockedStatus.join(',')}`); await storageBlockedPage.close(); + const idlePage = await browser.newPage(); + const idleRequests = []; + idlePage.on('request', request => { + const route = new URL(request.url()).pathname; + if (route === '/api/status' || route === '/api/refresh') idleRequests.push({ route, method: request.method() }); + }); + await idlePage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }); + await idlePage.click('#poll-ivl'); + await idlePage.click('#poll-menu [data-ms="15000"]'); + const idleDeadline = Date.now() + 35_000; + while (idleRequests.filter(request => request.route === '/api/status').length < 3 && Date.now() < idleDeadline) { + await idlePage.waitForTimeout(250); + } + check('two idle poll ticks issue no /api/refresh request or POST', + idleRequests.filter(request => request.route === '/api/status').length >= 3 + && idleRequests.every(request => request.route !== '/api/refresh'), + JSON.stringify(idleRequests)); + await idlePage.close(); + const page = await browser.newPage({ viewport: { width: 1440, height: 900 }, locale: 'en-US', timezoneId: 'America/Los_Angeles', }); + const refreshRequests = []; + const statusRequests = []; + const systemSummaryRequests = []; + page.on('request', (request) => { + const pathname = new URL(request.url()).pathname; + if (pathname === '/api/refresh') refreshRequests.push({ method: request.method(), body: request.method() === 'POST' ? request.postDataJSON() : null }); + if (pathname === '/api/status') statusRequests.push(request.url()); + if (pathname === '/api/system/summary') systemSummaryRequests.push(request.url()); + }); // Anything the page logs as an error, or any request it fails, is a defect — // collected globally so a failure in one view is not silently swallowed. @@ -1297,7 +1328,8 @@ async function main() { page.on('console', (m) => { if (m.type() !== 'error') return; const loc = m.location(); - if (/status of 409 \(Conflict\)/.test(m.text()) && loc?.url && expectedHttpConsoleErrors.delete(loc.url)) return; + if (/status of (?:409 \(Conflict\)|503 \(Service Unavailable\))/.test(m.text()) + && loc?.url && expectedHttpConsoleErrors.delete(loc.url)) return; const where = loc?.url ? ` @ ${loc.url}` : ''; consoleErrors.push(`${m.text()}${where}`); }); @@ -1424,7 +1456,6 @@ async function main() { // check: the inventory stub answers `running` that many times, then flips // scanRequired off so the workspace's bounded polling sees the built page. let maintenanceBuildPollsRemaining = 0; - let maintenanceRunningPollsServed = 0; // Per-label override for a coverage entry's `filesystem` flag, applied // on top of whatever the sentinel fixture's coverage() helper produced // (which never sets `filesystem` at all, exercising the "flag absent -> @@ -1472,7 +1503,8 @@ async function main() { for (const entry of entries) { counts[entry.lane] += 1; lanes[entry.lane].push(entry); } return { lanes, counts, entries }; } - await page.route(/\/api\/maintenance\/v2\//, async (route) => { + const maintenanceV2Route = /\/api\/maintenance\/v2\//; + const maintenanceV2Stub = async (route) => { const request = route.request(); const url = new URL(request.url()); const pathname = url.pathname; @@ -1496,7 +1528,6 @@ async function main() { maintenanceInventoryRequests.push(params); if (maintenanceBuildPollsRemaining > 0) { maintenanceBuildPollsRemaining -= 1; - maintenanceRunningPollsServed += 1; if (maintenanceBuildPollsRemaining === 0) maintenanceScanRequired = false; return reply(200, { scanRequired: true, total: 0, groups: [], facetCounts: {}, sortGroups: [], partialSources: [], @@ -1630,7 +1661,12 @@ async function main() { } if (request.method() === 'GET' && pathname === '/api/maintenance/v2/activity') { return reply(200, buildActivity({ - receipts: [INTERRUPTED_RECEIPT], dispositions: [], recipeEvents: [], scanHistory: [], inProgress: [], + receipts: [INTERRUPTED_RECEIPT], dispositions: [], recipeEvents: [], inProgress: [], + scanHistory: [ + { sourceId: 'src-claude', environmentId: 'env-local', label: 'Claude user configuration', state: 'complete', completedAt: '2026-09-07T12:00:00.000Z', visited: 12 }, + { sourceId: 'src-claude', environmentId: 'env-local', label: 'Claude user configuration', state: 'paused', recordedAt: '2026-09-09T12:00:00.000Z', completedAt: null, visited: 18 }, + { sourceId: 'src-other', environmentId: 'env-local', label: 'Codex configuration', state: 'complete', completedAt: '2026-09-08T12:00:00.000Z', visited: 8 }, + ], })); } const receiptMatch = pathname.match(/^\/api\/maintenance\/v2\/receipts\/([^/]+)$/); @@ -1762,7 +1798,8 @@ async function main() { }); } return reply(404, { code: 'NOT_FOUND' }); - }); + }; + await page.route(maintenanceV2Route, maintenanceV2Stub); // ── Refresh evidence (MNT-DSC-010): the workspace's explicit control reuses // the v1 endpoint verbatim, `?refresh=scan` then polling until settled. ── let maintenanceProviderPollCount = 0; @@ -1816,6 +1853,11 @@ async function main() { // connections. DOM readiness plus the application shell is the stable // navigation contract; network-idle can never be guaranteed by a Live UI. await page.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }); + check('Refresh and Reload controls replace the retired scan buttons', + await page.locator('#refresh-run').count() === 1 + && await page.locator('#refresh-strength').count() === 1 + && await page.locator('#poll-now').getAttribute('aria-label') === 'Reload — re-read this view; runs no checks' + && await page.locator('#sys-rescan, #mnt-check-providers, #mnt-remeasure, #host-health-refresh').count() === 0); await page.waitForSelector('#panel-overview', { state: 'attached' }); // ── ADR-0026 · About leads the bar but must NOT hijack the landing view ── @@ -2588,7 +2630,7 @@ async function main() { check('a fresh install names coverage gaps and keeps measurement in the toolbar only', /^4 sources have not been scanned yet\./.test((freshInstallBanner || '').trim()) && await page.locator('#mnt-partial button').count() === 0 - && await page.isVisible('#mnt-remeasure') + && await page.locator('#mnt-remeasure').count() === 0 && !/Fresh source/.test(freshInstallBanner || ''), `fresh-install banner read ${JSON.stringify(freshInstallBanner)}`); @@ -2602,8 +2644,7 @@ async function main() { await page.fill('#mnt-search', ''); await page.waitForFunction(() => document.querySelectorAll('#mnt-results .mnt-row').length > 3); - // ── Refresh evidence (MNT-DSC-010): explicit, labeled; disables Apply/ - // Undo while it runs; never fires on its own ── + // The shared Refresh control owns provider checks; opening Maintenance does not start one. await page.click('[data-mnt-dest="guidance"]'); await page.waitForSelector('#mnt-tab-guidance[aria-selected="true"]'); await page.waitForSelector('[data-mnt-plan-plc]'); @@ -2611,29 +2652,8 @@ async function main() { check('MNT-GUD-001: Guidance has exactly the visible lanes Can apply here, Steps available, Decisions to make, Updates available, and Recovery to finish', JSON.stringify(guidanceLaneLabels) === JSON.stringify([ 'Can apply here', 'Steps available', 'Decisions to make', 'Updates available', 'Recovery to finish', - ]), - `guidance lane labels read ${JSON.stringify(guidanceLaneLabels)}`); - check('a provider check never starts on its own', - maintenanceCheckProvidersReads === 0, 'the workspace probed providers without an explicit click'); - const providerRefreshResponse = page.waitForResponse((response) => new URL(response.url()).searchParams.get('refresh') === 'scan'); - await page.click('#mnt-check-providers'); - await providerRefreshResponse; - // The button disables synchronously; Apply's disabled attribute follows - // once the guidance re-render that mntCheckProviders() triggers resolves. - await page.waitForFunction(() => document.querySelector('[data-mnt-plan-plc]')?.disabled === true, null, { timeout: 5000 }); - const providersRunning = await page.evaluate(() => ({ - buttonDisabled: document.getElementById('mnt-check-providers')?.disabled, - applyDisabled: document.querySelector('[data-mnt-plan-plc]')?.disabled, - })); - check('Refresh evidence is explicit, labeled, and disables Apply while it runs', - maintenanceCheckProvidersReads === 1 && providersRunning.buttonDisabled === true - && providersRunning.applyDisabled === true, - `providers-running state was ${JSON.stringify(providersRunning)}`); - await page.waitForFunction(() => document.getElementById('mnt-check-providers')?.disabled === false, null, { timeout: 8000 }); - check('the provider check settles and re-enables Apply', - await page.$eval('[data-mnt-plan-plc]', (button) => button.disabled === false), - 'Apply stayed disabled after the provider check settled'); - + ])); + check('a provider check never starts on its own', maintenanceCheckProvidersReads === 0); // ── Dispositions (MNT-GUD-009/011): explained before confirmation, one // exact guidanceId per write ── await page.waitForSelector('.mnt-dispositions [data-mnt-disposition-open="acknowledged"]'); @@ -2716,6 +2736,18 @@ async function main() { // typed-confirmation dialog ── await page.click('[data-mnt-dest="activity"]'); await page.waitForSelector('#mnt-tab-activity[aria-selected="true"]'); + await page.waitForSelector('#mnt-scan-history .mnt-history-day'); + const scanRows = await page.$$eval('#mnt-scan-history tbody', (groups) => groups.map((group) => ({ + heading: group.querySelector('.mnt-history-day')?.textContent?.trim(), + rows: [...group.querySelectorAll('tr:not(.mnt-history-day)')].map((row) => row.textContent?.trim()), + }))); + check('paused scan renders at its recorded time ahead of older completed scans', + scanRows.length === 3 && /Paused/.test(scanRows[0].rows[0]) + && /Claude user configuration/.test(scanRows[0].rows[0]) + && !/Time not recorded/.test(scanRows[0].rows[0]) + && /Codex configuration/.test(scanRows[1].rows[0]) + && /Complete/.test(scanRows[2].rows[0]), + `scan rows read ${JSON.stringify(scanRows)}`); await page.waitForSelector('[data-mnt-audit-receipt]'); const auditTriggerLabel = await page.textContent('[data-mnt-audit-receipt]'); check('MNT-RCV-001: an interrupted receipt offers Audit interruption, not generic Verify again', @@ -2776,12 +2808,6 @@ async function main() { `export requests were ${JSON.stringify(maintenanceExportRequests)}`); await page.click('#mnt-receipt-close'); - // ── Discovery: a light smoke check of the fourth destination. Waits for - // the CONTENT of #mnt-scan-progress specifically (not just an
  • in - // #mnt-automatic-sources, whose two rows are static and already present - // from the earlier group-collapse fixture's own stale Discovery visits) - // — a fresh fetch against base()'s real coverage can otherwise still be - // in flight when a weaker wait resolves on leftover data. ── await page.click('[data-mnt-dest="discovery"]'); await page.waitForSelector('#mnt-tab-discovery[aria-selected="true"]'); await page.waitForFunction(() => /Claude user configuration/.test( @@ -2893,37 +2919,22 @@ async function main() { )); const failedRefreshEmpty = await visibleText(page, '#mnt-results'); const failedRefreshStatus = await page.textContent('#mnt-status'); - check('a failed lastRefresh names "did not complete" plus the sanitized message, never the raw code, and keeps Refresh evidence available', + check('a failed lastRefresh names "did not complete" plus the sanitized message, never the raw code, and keeps Refresh available', /did not complete/.test(failedRefreshEmpty) && /did not respond before the timeout/.test(failedRefreshEmpty) && !/PROVIDER_TIMEOUT/.test(failedRefreshEmpty) && !/PROVIDER_TIMEOUT/.test(failedRefreshStatus || '') && /did not respond before the timeout/.test(failedRefreshStatus || '') - && await page.isEnabled('#mnt-check-providers'), + && await page.isEnabled('#refresh-run'), `empty state read ${JSON.stringify(failedRefreshEmpty)}, status read ${JSON.stringify(failedRefreshStatus)}`); maintenanceInventoryLastRefresh = null; - // ── D5: scanRequired points at Refresh evidence, and settling it - // refreshes Inventory in place, with no page reload. Hops off Inventory - // and back so the destination switch re-fetches against the reset - // (no-lastRefresh) fixture state above, without duplicating route - // handlers on a fresh page. ── + // A missing inventory points to the shared Refresh control. await page.click('[data-mnt-dest="discovery"]'); await page.click('[data-mnt-dest="inventory"]'); await page.waitForFunction(() => /No inventory has been built yet/.test( document.getElementById('mnt-results')?.innerText || '', )); const scanRequiredEmpty = await visibleText(page, '#mnt-results'); - check('the scanRequired empty state points at Refresh evidence on this workspace, not a System full scan', - /Refresh evidence/.test(scanRequiredEmpty) && !/full scan from System/i.test(scanRequiredEmpty), - `scanRequired empty state read ${JSON.stringify(scanRequiredEmpty)}`); - await page.click('#mnt-check-providers'); - await page.waitForFunction(() => document.getElementById('mnt-check-providers')?.disabled === false, null, { timeout: 8000 }); - await page.waitForSelector('#mnt-results .mnt-row', { timeout: 8000 }); - check('settling Refresh evidence clears scanRequired and shows Inventory results in place, with no reload', - await page.$$eval('#mnt-results .mnt-row', (els) => els.length) > 0, - 'Inventory did not refresh in place once the provider check settled'); - check('D8: the workspace polled the inventory through the running build (bounded, 1.5 s apart) instead of re-fetching once', - maintenanceRunningPollsServed >= 2 && maintenanceBuildPollsRemaining === 0, - `running polls served: ${maintenanceRunningPollsServed}, remaining ${maintenanceBuildPollsRemaining}`); + check('the scanRequired empty state points at Refresh', /Refresh/.test(scanRequiredEmpty)); maintenanceScanRequired = false; // ── D8: while the server reports a running build, the empty state says @@ -2937,68 +2948,13 @@ async function main() { document.getElementById('mnt-results')?.innerText || '', )); const runningEmpty = await visibleText(page, '#mnt-results'); - check('a running lastRefresh renders "Building the inventory…" and keeps Refresh evidence available', + check('a running lastRefresh renders "Building the inventory…" and keeps Refresh available', /Building the inventory/.test(runningEmpty) && !/not been built yet/.test(runningEmpty) - && await page.isEnabled('#mnt-check-providers'), + && await page.isEnabled('#refresh-run'), `running empty state read ${JSON.stringify(runningEmpty)}`); maintenanceInventoryLastRefresh = null; maintenanceScanRequired = false; - // ── Re-measure machine (System Full scan, exposed beside Refresh - // evidence): delegates to System's own #sys-rescan, then blocks writes - // (mntWritesBlocked) for the whole measurement + provider-check + - // inventory-rebuild chain. Uses its own temporary /api/system route - // rather than the real SYSTEM_STUB — that stub's systemDeepScans counter - // is asserted against a clean slate by the dedicated System-tab Rescan - // tests later in this run, and this block must neither depend on nor - // perturb that count. ── - await page.click('[data-mnt-dest="guidance"]'); - await page.waitForSelector('[data-mnt-plan-plc]'); - await page.waitForFunction(() => document.querySelector('[data-mnt-plan-plc]')?.disabled === false); - let remeasureSystemReadCount = 0; - let remeasureDeepScanRequests = 0; - // Deterministic on REQUEST COUNT, not wall-clock: the deep-scan kickoff - // itself is always the first read (reports running), every read after is - // settled. This cannot race system-projects.mjs's own poll cadence and - // mntPollSystemMeasurement's independent one against a Node-side timer. - await page.route(/\/api\/system(\/summary)?(\?|$)/, (route) => { - const reqUrl = new URL(route.request().url()); - if (reqUrl.searchParams.get('refresh') === 'deep') remeasureDeepScanRequests += 1; - remeasureSystemReadCount += 1; - if (remeasureSystemReadCount === 2) { - // Deep measurement also completes a fresh provider check and inventory build. - maintenanceCheckProvidersReads += 1; - maintenanceProviderPollCount = 2; - } - return route.fulfill({ - status: 200, contentType: 'application/json', - body: JSON.stringify({ ...SYSTEM_PAYLOAD, scan: { ...SYSTEM_PAYLOAD.scan, running: remeasureSystemReadCount <= 1 } }), - }); - }); - await page.click('#mnt-remeasure'); - // mntRemeasureMachine() re-renders the active destination the instant it - // sets MNT.remeasureBusy — before it even clicks #sys-rescan — so both - // of these are observable synchronously, exactly like Refresh evidence. - const remeasureStarted = await page.evaluate(() => ({ - remeasureDisabled: document.getElementById('mnt-remeasure')?.disabled, - applyDisabled: document.querySelector('[data-mnt-plan-plc]')?.disabled, - })); - check('Re-measure machine disables itself and blocks writes (mntWritesBlocked) the instant it starts', - remeasureStarted.remeasureDisabled === true && remeasureStarted.applyDisabled === true, - `remeasure-start state was ${JSON.stringify(remeasureStarted)}`); - await page.waitForFunction(() => document.getElementById('mnt-remeasure')?.disabled === false, null, { timeout: 15_000 }); - check('Re-measure machine delegates to #sys-rescan (a real deep scan) and settles, re-enabling itself and Apply', - remeasureDeepScanRequests === 1 && await page.$eval('[data-mnt-plan-plc]', (b) => b.disabled === false), - `deep scan requests: ${remeasureDeepScanRequests}, Apply stayed disabled after Re-measure machine settled: ${await page.$eval('[data-mnt-plan-plc]', (b) => b.disabled)}`); - // A stale scheduled System poll used to restart a completed owned scan as - // an external operation, waiting forever for a second evidence generation. - await page.waitForTimeout(3200); - check('a completed remeasurement stays settled after the System poll interval', - await page.isEnabled('#mnt-remeasure') && await page.isEnabled('[data-mnt-plan-plc]') - && remeasureSystemReadCount === 2, - `system reads: ${remeasureSystemReadCount}; operation: ${await page.textContent('#mnt-check-providers-status')}`); - await page.unroute(/\/api\/system(\/summary)?(\?|$)/); - // ── #system/catalog redirects to Maintenance Inventory (ADR-0048) ── await page.evaluate(() => { location.hash = '#system/catalog'; }); await page.reload({ waitUntil: 'domcontentloaded' }); @@ -3518,8 +3474,6 @@ async function main() { text: el?.textContent.trim(), stale: el?.getAttribute('data-stale'), title: el?.getAttribute('title'), - rescanDisabled: document.getElementById('sys-rescan')?.disabled, - fullScanLabel: document.getElementById('sys-rescan')?.innerText, live: el?.getAttribute('aria-live'), }; }); @@ -3528,28 +3482,12 @@ async function main() { `the freshness label read ${JSON.stringify(freshness)} — the snapshot is nine days old`); check('past the staleness horizon the label nudges without scanning', freshness.stale === '1' && /stale/i.test(String(freshness.text)) - && freshness.rescanDisabled === false && /Full scan/.test(String(freshness.fullScanLabel)) && freshness.live === 'polite', `staleness presentation was ${JSON.stringify(freshness)}`); check('opening System never starts a deep scan', systemDeepScans === 0, `${systemDeepScans} deep scan(s) had already run after opening the area and all five views`); - const deepResponse = page.waitForResponse( - (r) => r.url().includes('/api/system') && r.url().includes('refresh=deep'), - { timeout: 8000 }, - ).catch(() => null); - await page.click('#sys-rescan'); - await deepResponse; - await page.waitForTimeout(200); - check('Rescan is the only thing that starts a deep scan, and it starts exactly one', - systemDeepScans === 1, - `the collector saw ${systemDeepScans} deep scan(s) after one Rescan click`); - await page.waitForTimeout(50); - check('a successful deep System rescan refreshes Maintenance provider evidence once', - chainedMaintenanceScans === 1, - `the Maintenance service saw ${chainedMaintenanceScans} scan(s)`); - // ── Observability: execution workspace + synchronized evidence ── await page.click('[data-tab="observability"]'); await page.waitForSelector('#live-nodes .live-node', { timeout: 8000 }); @@ -5771,6 +5709,330 @@ async function main() { check('and survives a reload rather than snapping back to the default', c2.expanded === 'true' && c2.hidden === false, JSON.stringify(c2)); + // Refresh is the only route that starts checks. Its stage state is server authored. + check('idle dashboard requested no refresh state or operation', refreshRequests.length === 0, + JSON.stringify(refreshRequests)); + await page.click('#tab-system'); + await page.click('[data-system-view="maintenance"]'); + await page.click('[data-mnt-dest="guidance"]'); + await page.waitForSelector('[data-mnt-plan-plc]'); + // The audit and undo actions are conditional on retained receipts. Keep + // their actual selectors in the fixture while this run has no such receipt. + await page.evaluate(() => { + const fixture = globalThis.document.createElement('div'); + fixture.id = 'refresh-write-fixture'; + fixture.hidden = true; + fixture.innerHTML = ''; + globalThis.document.body.appendChild(fixture); + }); + const stageRelease = new Map(); + for (const id of ['maintenance', 'inventory', 'local']) { + refreshStageGates.set(id, new Promise((resolve) => stageRelease.set(id, resolve))); + } + const statusReadsBefore = statusRequests.length; + await page.click('#refresh-run'); + await page.waitForFunction(() => document.getElementById('refresh-status')?.textContent.includes('Refreshing Maintenance evidence')); + check('Refresh status names the server stage and elapsed time', + /Refreshing Maintenance evidence · \d+s/.test(await page.locator('#refresh-status').innerText())); + check('Refresh starts exactly one local POST', refreshRequests.filter((r) => r.method === 'POST').length === 1 + && JSON.stringify(refreshRequests.find((r) => r.method === 'POST')?.body) === JSON.stringify({ strength: 'local' })); + await page.waitForFunction(() => document.querySelector('[data-mnt-plan-plc]')?.disabled === true); + check('Maintenance Apply, Undo and Record are disabled during Refresh', + await page.locator('[data-mnt-plan-plc]').first().isDisabled() + && await page.locator('#refresh-write-fixture [data-mnt-undo-receipt]').isDisabled() + && await page.locator('#refresh-write-fixture [data-mnt-reconcile-receipt]').isDisabled()); + await page.click('[data-mnt-dest="inventory"]'); + await page.waitForSelector('#mnt-tab-inventory[aria-selected="true"]'); + const activeMaintenanceReadsBefore = maintenanceInventoryRequests.length; + for (const [id, label] of [['maintenance', 'Rebuilding the inventory'], ['inventory', 'Re-checking local evidence and versions']]) { + stageRelease.get(id)(); + await page.waitForFunction((text) => document.getElementById('refresh-status')?.textContent.includes(text), label); + } + stageRelease.get('local')(); + await page.waitForFunction(() => document.getElementById('refresh-status')?.textContent === 'Refresh complete.'); + for (let attempt = 0; attempt < 30 && maintenanceInventoryRequests.length <= activeMaintenanceReadsBefore; attempt++) { + await page.waitForTimeout(100); + } + await page.click('[data-mnt-dest="guidance"]'); + await page.waitForFunction(() => document.querySelector('[data-mnt-plan-plc]')?.disabled === false); + check('Maintenance write controls are restored after Refresh', + await page.locator('[data-mnt-plan-plc]').first().isEnabled() + && await page.locator('#refresh-write-fixture [data-mnt-undo-receipt]').isEnabled() + && await page.locator('#refresh-write-fixture [data-mnt-reconcile-receipt]').isEnabled()); + await page.locator('#refresh-write-fixture').evaluate(element => element.remove()); + await page.waitForTimeout(200); + check('Refresh re-reads the active Maintenance view and host readiness after completion', + statusRequests.length > statusReadsBefore && maintenanceInventoryRequests.length > activeMaintenanceReadsBefore); + const localRefreshReads = refreshRequests.filter(request => request.method === 'GET').length; + await page.waitForTimeout(1700); + check('completed Refresh stops status polling', refreshRequests.filter(request => request.method === 'GET').length === localRefreshReads); + await page.selectOption('#refresh-strength', 'machine'); + await page.check('#refresh-project-trees'); + await page.click('#refresh-run'); + await page.waitForFunction(() => document.getElementById('refresh-status')?.textContent === 'Refresh complete.'); + check('machine Refresh carries the project tree scope', refreshRequests.filter((r) => r.method === 'POST').length === 2 + && JSON.stringify(refreshRequests.filter((r) => r.method === 'POST')[1].body) === JSON.stringify({ strength: 'machine', projectTrees: true })); + const postCount = refreshRequests.filter((r) => r.method === 'POST').length; + const reloadRequests = []; + const captureReload = request => reloadRequests.push(request.method()); + await page.click('#poll-play'); + page.on('request', captureReload); + await page.waitForTimeout(3100); + const reloadResponse = page.waitForResponse(response => new URL(response.url()).pathname === '/api/status'); + await page.click('#poll-now'); + await reloadResponse; + page.off('request', captureReload); + check('Reload issues only GETs and starts no checks', reloadRequests.length > 0 + && reloadRequests.every(method => method === 'GET') + && refreshRequests.filter((r) => r.method === 'POST').length === postCount); + for (const view of ['summary', 'storage', 'projects']) { + await page.click(`[data-system-view="${view}"]`); + await page.waitForTimeout(3100); + const before = systemSummaryRequests.length; + await page.click('#poll-now'); + await page.waitForTimeout(300); + check(`Reload re-reads active System ${view} without measuring`, + systemSummaryRequests.length > before + && systemSummaryRequests.slice(before).every(url => !new URL(url).searchParams.has('refresh')), + `System reads before/after: ${before}/${systemSummaryRequests.length}`); + } + await page.click('[data-system-view="storage"]'); + await page.selectOption('#refresh-strength', 'local'); + const beforeCompletion = systemSummaryRequests.length; + await page.click('#refresh-run'); + await page.waitForFunction(() => document.getElementById('refresh-status')?.textContent === 'Refresh complete.'); + await page.waitForTimeout(300); + check('completed Refresh re-reads active System Storage without measuring', + systemSummaryRequests.length > beforeCompletion + && systemSummaryRequests.slice(beforeCompletion).every(url => !new URL(url).searchParams.has('refresh')), + `System reads before/after: ${beforeCompletion}/${systemSummaryRequests.length}`); + await page.click('#poll-ivl'); + await page.click('#poll-menu [data-ms="15000"]'); + const backgroundBefore = systemSummaryRequests.length; + await page.click('#poll-play'); + await page.waitForTimeout(16_000); + await page.click('#poll-play'); + check('background poll keeps System Storage on the existing cheap-read policy', + systemSummaryRequests.length === backgroundBefore, + `System reads before/after background tick: ${backgroundBefore}/${systemSummaryRequests.length}`); + console.log(`refresh requests: idle 0; local POST 1, GET ${localRefreshReads}; machine POST 1, total GET ${refreshRequests.filter(request => request.method === 'GET').length}; Reload ${reloadRequests.length} GET`); + await page.setViewportSize({ width: 390, height: 844 }); + await page.evaluate(() => { localStorage.setItem('ak-dash-theme', 'light'); location.reload(); }); + await page.waitForFunction(() => document.documentElement.getAttribute('data-theme') === 'light'); + await page.screenshot({ path: path.join(SHOTS, 'refresh-header-light-390.png'), animations: 'disabled' }); + await page.evaluate(() => { localStorage.setItem('ak-dash-theme', 'dark'); location.reload(); }); + await page.waitForFunction(() => document.documentElement.getAttribute('data-theme') === 'dark'); + await page.screenshot({ path: path.join(SHOTS, 'refresh-header-dark-390.png'), animations: 'disabled' }); + check('Refresh header fits at 390px in light and dark themes', + await page.evaluate(() => { + const header = document.querySelector('.band'); + const refresh = document.querySelector('.refresh-control'); + return header.getBoundingClientRect().right <= globalThis.innerWidth + && refresh.getBoundingClientRect().right <= globalThis.innerWidth; + })); + await page.setViewportSize({ width: 1440, height: 900 }); + check('wide dark screenshot has the dark theme at capture time', + await page.evaluate(() => document.documentElement.getAttribute('data-theme') === 'dark' + && getComputedStyle(document.documentElement).getPropertyValue('--bg').trim() === '#000000')); + await page.screenshot({ path: path.join(SHOTS, 'refresh-header-dark-1440.png'), animations: 'disabled' }); + + // A rejected status read cannot release the write guard for an operation + // this page already started. The next valid state reconciles it. + await page.click('[data-system-view="maintenance"]'); + await page.click('[data-mnt-dest="guidance"]'); + await page.waitForSelector('[data-mnt-plan-plc]'); + let releaseStatusStage; + refreshStageGates.set('maintenance', new Promise(resolve => { releaseStatusStage = resolve; })); + let statusFailures = 2; + const rejectStatus = route => { + if (route.request().method() === 'GET' && statusFailures > 0) { + statusFailures--; + if (statusFailures === 1) expectedHttpConsoleErrors.add(route.request().url()); + return route.fulfill(statusFailures === 1 + ? { status: 503, contentType: 'application/json', body: JSON.stringify({ error: 'temporary status failure' }) } + : { status: 200, contentType: 'application/json', body: JSON.stringify({ running: 'unknown', stages: [] }) }); + } + return route.continue(); + }; + await page.route('**/api/refresh', rejectStatus); + const rejectedStatus = page.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' && response.status() === 503); + await page.click('#refresh-run'); + await rejectedStatus; + await page.waitForFunction(() => /retry/i.test(document.getElementById('refresh-status')?.textContent || '')); + check('rejected refresh-status GET keeps the owned operation and Maintenance writes blocked', + await page.locator('#refresh-run').isDisabled() + && await page.locator('[data-mnt-plan-plc]').first().isDisabled() + && /retry/i.test(await page.locator('#refresh-status').innerText())); + const readsAfterError = refreshRequests.filter(request => request.method === 'GET').length; + await page.waitForTimeout(1700); + check('refresh-status read retries after non-2xx JSON and keeps invalid success state blocked', + refreshRequests.filter(request => request.method === 'GET').length > readsAfterError + && await page.locator('#refresh-run').isDisabled() + && await page.locator('[data-mnt-plan-plc]').first().isDisabled()); + releaseStatusStage(); + await page.waitForFunction(() => document.getElementById('refresh-status')?.textContent === 'Refresh complete.', null, { timeout: 8000 }); + await page.waitForFunction(() => document.querySelector('[data-mnt-plan-plc]')?.disabled === false); + check('owned refresh completes after status recovery and then unblocks writes', + await page.locator('#refresh-run').isEnabled() + && await page.locator('[data-mnt-plan-plc]').first().isEnabled()); + await page.unroute('**/api/refresh', rejectStatus); + + // Two real dashboard pages can supersede a completed operation before its + // first polling GET. Test both a newer terminal state and a newer running + // state against the actual server operation, not a route-shaped fake. + activeMaintenanceInventory = structuredClone(SENTINEL_FIXTURES.base()); + const supersessionUpdate = activeMaintenanceInventory.guidanceEntries.find(entry => entry.lane === 'apply'); + Object.assign(supersessionUpdate, { verb: 'update', outcome: 'Update Claude plugin', + providerCapabilityId: 'claude-plugin:v1:update:user', + verifiedPremises: ['placement', 'installedVersion', 'published-update', 'consumers', 'impact'], + impact: { summary: 'Installs the verified newer frontend-design version.' } }); + async function latestRefresh() { + const response = await fetch(`${ORIGIN}/api/refresh`, { headers: modelHeaders }); + return response.json(); + } + async function waitServerRefresh(running, differentFrom) { + const deadline = Date.now() + 5000; + while (Date.now() < deadline) { + const state = await latestRefresh(); + const identity = state.operationId; + if (state.running === running && identity !== differentFrom && identity) return state; + await new Promise(resolve => setTimeout(resolve, 20)); + } + throw new Error('the fixture refresh did not reach the expected server state'); + } + async function exerciseSupersession(holdPeer) { + const firstPage = await browser.newPage(); + const peerPage = await browser.newPage(); + let releaseFirstPoll; + const firstPollGate = new Promise(resolve => { releaseFirstPoll = resolve; }); + let interceptFirstPoll = true; + let releasePeerStage; + const holdPeerStage = new Promise(resolve => { releasePeerStage = resolve; }); + try { + await firstPage.route(maintenanceV2Route, maintenanceV2Stub); + await Promise.all([ + firstPage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }), + peerPage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }), + ]); + await firstPage.route('**/api/refresh', async route => { + if (route.request().method() === 'GET' && interceptFirstPoll) { + interceptFirstPoll = false; + await firstPollGate; + } + await route.continue(); + }); + await firstPage.click('#tab-system'); + await firstPage.click('[data-system-view="maintenance"]'); + await firstPage.click('[data-mnt-dest="guidance"]'); + await firstPage.waitForSelector('[data-mnt-plan-plc]'); + refreshStageGates.set('maintenance', Promise.resolve()); + const firstPost = firstPage.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' + && response.request().method() === 'POST'); + await firstPage.click('#refresh-run'); + await firstPost; + const firstDone = await waitServerRefresh(false, null); + const firstIdentity = firstDone.operationId; + if (holdPeer) refreshStageGates.set('maintenance', holdPeerStage); + const peerPost = peerPage.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' + && response.request().method() === 'POST'); + await peerPage.click('#refresh-run'); + await peerPost; + const newer = await waitServerRefresh(holdPeer, firstIdentity); + const firstStatus = firstPage.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' + && response.request().method() === 'GET'); + releaseFirstPoll(); + await firstStatus; + await firstPage.waitForTimeout(100); + if (holdPeer) { + check('superseding running refresh keeps the first page and real Apply blocked', + !!newer.operationId && newer.operationId !== firstDone.operationId + && await firstPage.locator('#refresh-run').isDisabled() + && await firstPage.locator('[data-mnt-plan-plc]').first().isDisabled() + && /another refresh|supersed/i.test(await firstPage.locator('#refresh-status').innerText())); + releasePeerStage(); + await waitServerRefresh(false, firstIdentity); + await firstPage.waitForFunction(() => !document.getElementById('refresh-run')?.disabled, + null, { timeout: 8000 }).catch(() => {}); + } + check(`first page recovers from newer ${holdPeer ? 'running' : 'completed'} refresh without claiming its outcome`, + await firstPage.locator('#refresh-run').isEnabled() + && await firstPage.locator('[data-mnt-plan-plc]').first().isEnabled() + && /outcome unavailable|supersed/i.test(await firstPage.locator('#refresh-status').innerText()) + && !/Refresh complete\.|Refresh did not complete\./.test(await firstPage.locator('#refresh-status').innerText())); + } finally { + releaseFirstPoll(); + releasePeerStage(); + await Promise.all([firstPage.close(), peerPage.close()]); + } + } + await exerciseSupersession(false); + await exerciseSupersession(true); + + // A page that loses the single-flight POST race must respect the running + // operation reported in the 409 response before accepting Maintenance writes. + { + const ownerPage = await browser.newPage(); + const losingPage = await browser.newPage(); + let releaseOwner; + const ownerStage = new Promise(resolve => { releaseOwner = resolve; }); + try { + await losingPage.route(maintenanceV2Route, maintenanceV2Stub); + await Promise.all([ + ownerPage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }), + losingPage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }), + ]); + await losingPage.click('#tab-system'); + await losingPage.click('[data-system-view="maintenance"]'); + await losingPage.click('[data-mnt-dest="guidance"]'); + await losingPage.waitForSelector('[data-mnt-plan-plc]'); + refreshStageGates.set('maintenance', ownerStage); + await ownerPage.click('#refresh-run'); + await waitServerRefresh(true, null); + const conflict = losingPage.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' + && response.request().method() === 'POST' && response.status() === 409); + await losingPage.click('#refresh-run'); + await conflict; + check('409 refresh conflict blocks the losing page and real Apply while the winner runs', + await losingPage.locator('#refresh-run').isDisabled() + && await losingPage.locator('[data-mnt-plan-plc]').first().isDisabled()); + releaseOwner(); + await losingPage.waitForFunction(() => !document.getElementById('refresh-run')?.disabled, + null, { timeout: 8000 }).catch(() => {}); + check('409 losing page restores Apply with an outcome-unavailable message after the winner ends', + await losingPage.locator('[data-mnt-plan-plc]').first().isEnabled() + && /outcome unavailable/i.test(await losingPage.locator('#refresh-status').innerText())); + } finally { + releaseOwner(); + await Promise.all([ownerPage.close(), losingPage.close()]); + } + } + + // An unconfirmed POST cannot adopt an older terminal operation as proof + // that its own request ended. The page must retain the write guard. + { + const uncertainPage = await browser.newPage(); + try { + await uncertainPage.route(maintenanceV2Route, maintenanceV2Stub); + await uncertainPage.goto(srv.urlWithToken, { waitUntil: 'domcontentloaded' }); + await uncertainPage.click('#tab-system'); + await uncertainPage.click('[data-system-view="maintenance"]'); + await uncertainPage.click('[data-mnt-dest="guidance"]'); + await uncertainPage.waitForSelector('[data-mnt-plan-plc]'); + await uncertainPage.route('**/api/refresh', route => route.request().method() === 'POST' + ? route.abort() : route.continue()); + const staleRead = uncertainPage.waitForResponse(response => new URL(response.url()).pathname === '/api/refresh' + && response.request().method() === 'GET'); + await uncertainPage.click('#refresh-run'); + await staleRead; + check('uncertain POST does not adopt an older completed operation or release real Apply', + await uncertainPage.locator('#refresh-run').isDisabled() + && await uncertainPage.locator('[data-mnt-plan-plc]').first().isDisabled() + && /retry|unavailable/i.test(await uncertainPage.locator('#refresh-status').innerText())); + } finally { + await uncertainPage.close(); + } + } + // ── nothing errored anywhere along the way ── // A 404 from /api/session/ is CORRECT behaviour for a session that does // not exist — the route was changed to stop answering 200-with-a-null-body. diff --git a/tests/ui/helpers/launch-chrome.mjs b/tests/ui/helpers/launch-chrome.mjs index bb38d9bb..25f485f8 100644 --- a/tests/ui/helpers/launch-chrome.mjs +++ b/tests/ui/helpers/launch-chrome.mjs @@ -12,6 +12,35 @@ import os from 'node:os'; import path from 'node:path'; import { chromium } from 'playwright'; +// Chrome needs the executable search path, display connection and a few Windows +// process basics. Its home and temp state belong to this launch, not the caller. +const CHROME_KEYS = ['PATH', 'DISPLAY', 'WAYLAND_DISPLAY', 'XAUTHORITY', 'XDG_RUNTIME_DIR', + 'DBUS_SESSION_BUS_ADDRESS', 'SYSTEMROOT', 'WINDIR', 'COMSPEC', 'PATHEXT']; +const WINDOWS_NAMES = { PATH: 'Path', SYSTEMROOT: 'SystemRoot', WINDIR: 'windir', COMSPEC: 'ComSpec', PATHEXT: 'PATHEXT' }; + +/** @param {NodeJS.ProcessEnv} source @param {string} dir @param {string} [platform] */ +export function chromeEnv(source, dir, platform = process.platform) { + const windows = platform === 'win32'; + const env = {}; + for (const key of CHROME_KEYS) { + let value = source[key]; + if (windows) { + const matches = Object.keys(source).filter((name) => name.toUpperCase() === key); + const chosen = matches.includes(key) ? key : matches.sort()[0]; + value = chosen === undefined ? undefined : source[chosen]; + } + if (value !== undefined) env[windows ? (WINDOWS_NAMES[key] ?? key) : key] = value; + } + return { + ...env, + HOME: dir, USERPROFILE: dir, + XDG_CONFIG_HOME: path.join(dir, 'config'), XDG_CACHE_HOME: path.join(dir, 'cache'), + XDG_DATA_HOME: path.join(dir, 'data'), APPDATA: path.join(dir, 'appdata'), + LOCALAPPDATA: path.join(dir, 'localappdata'), + TMPDIR: dir, TEMP: dir, TMP: dir, MAC_CHROMIUM_TMPDIR: dir, + }; +} + /** * @param {import('playwright').LaunchOptions} [options] merged over { channel: 'chrome', headless: true } * @returns {Promise} a browser whose close() also removes Chrome's temp folder @@ -19,12 +48,11 @@ import { chromium } from 'playwright'; export async function launchChrome(options = {}) { const dir = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'ak-ui-chrome-'))); const remove = () => fs.rmSync(dir, { recursive: true, force: true, maxRetries: 3 }); - const temp = { TMPDIR: dir, TEMP: dir, TMP: dir, MAC_CHROMIUM_TMPDIR: dir }; let browser; try { browser = await chromium.launch({ channel: 'chrome', headless: true, ...options, - env: { ...process.env, ...temp }, // spawn-env: inherits (the browser needs the display and PATH; only its temp dir moves) + env: chromeEnv(process.env, dir), }); } catch (error) { remove(); throw error; } const close = browser.close.bind(browser); diff --git a/tests/ui/host-readiness.mjs b/tests/ui/host-readiness.mjs index d8d8f491..247a47c9 100644 --- a/tests/ui/host-readiness.mjs +++ b/tests/ui/host-readiness.mjs @@ -19,7 +19,12 @@ const report = () => ({ checkedAt: '2026-09-20T10:00:00Z', scope: 'Dashboard lau host, status: 'ok', level: 'local', checks, evidenceKey: 'a'.repeat(64), canCheckConnection: true, checkedAt: '2026-09-20T10:00:00Z', connection: { state: 'not-run' }, target: { nativeDefault: true }, }])) }); - +const luminance = color => { + const rgb = color.match(/[\d.]+/g).slice(0,3).map(Number).map(value => value/255) + .map(value => value <= .04045 ? value/12.92 : ((value+.055)/1.055)**2.4); + return rgb[0]*.2126 + rgb[1]*.7152 + rgb[2]*.0722; +}; +const contrast = (a,b) => { const values=[luminance(a),luminance(b)].sort((x,y)=>y-x);return (values[0]+.05)/(values[1]+.05); }; test('all hosts have qualified OK, accessible details and explicitly confirmed connection checks', async t => { const browser = await launchChrome(); t.after(() => browser.close()); @@ -42,7 +47,7 @@ test('all hosts have qualified OK, accessible details and explicitly confirmed c return route.fulfill({contentType:'text/html',body:renderPage({name:'Health fixture',version:'test'}).replace(/]*>[\s\S]*?<\/script>/gi,'')}); }); await page.goto('http://health.test/'); - await page.addScriptTag({content:`${esc.toString()}\nfunction authHeaders(){return {'x-dash-token':'fixture'};}\n${source('usage')}\n${source('host-readiness')}\nwireHostHealth();`}); + await page.addScriptTag({content:`${esc.toString()}\nfunction authHeaders(){return {'x-dash-token':'fixture'};}\n${source('usage')}\nfunction refreshRunning(){return false;}\nfunction startRefresh(strength){(window.__refreshCalls ||= []).push(strength);return Promise.resolve(true); }\n${source('host-readiness')}\nwireHostHealth();`}); assert.equal(await page.locator('[data-health-host="codex"] .sp-status').innerText(),'Checking'); await page.evaluate(data=>globalThis.renderHostReadiness(data),report()); for(const [host,name] of [['claude','Claude Code'],['codex','Codex'],['opencode','OpenCode']]){ @@ -83,6 +88,20 @@ test('all hosts have qualified OK, accessible details and explicitly confirmed c await page.locator('.usage-source-details summary').click(); assert.match(await page.locator('.source-diagnostics').innerText(),/parse-yield-partial/); assert.equal(await page.locator('[data-health-host="codex"] .sp-status').innerText(),'OK'); + const unassessed=report(); + unassessed.hosts.claude={...unassessed.hosts.claude,checks:{...checks,configuration:{state:'unknown',reason:'Configuration was not assessed.'}}}; + await page.evaluate(data=>globalThis.renderHostReadiness(data),unassessed); + assert.equal(await page.locator('[data-health-host="claude"] .sp-status').innerText(),'Unknown'); + const iconColors=await page.evaluate(() => ['light','dark'].map(theme=>{ + globalThis.document.documentElement.setAttribute('data-theme',theme); + const chip=globalThis.document.querySelector('[data-health-host="codex"] .live-host'); + return {theme,fill:globalThis.getComputedStyle(chip.querySelector('path')).fill,background:globalThis.getComputedStyle(chip).backgroundColor}; + })); + for(const row of iconColors) { + console.log(`Codex icon ${row.theme}: ${row.fill} on ${row.background}, ${contrast(row.fill,row.background).toFixed(2)}:1`); + assert.ok(contrast(row.fill,row.background)>=3, + `Codex icon ${row.theme}: ${row.fill} on ${row.background}, ratio ${contrast(row.fill,row.background).toFixed(2)}:1`); + } await page.evaluate(()=>globalThis.renderHostReadiness(null)); assert.equal(await page.locator('[data-health-host="codex"] .sp-status').innerText(),'Unknown'); assert.equal(errors.length,0,errors.join('\n')); @@ -121,7 +140,7 @@ test('unmanaged hosts read their management state everywhere, with information-o return route.fulfill({contentType:'text/html',body:renderPage({name:'Health fixture',version:'test'}).replace(/]*>[\s\S]*?<\/script>/gi,'')}); }); await page.goto('http://health.test/'); - await page.addScriptTag({content:`${esc.toString()}\nfunction authHeaders(){return {'x-dash-token':'fixture'};}\n${source('usage')}\n${source('host-readiness')}\nwireHostHealth();`}); + await page.addScriptTag({content:`${esc.toString()}\nfunction authHeaders(){return {'x-dash-token':'fixture'};}\n${source('usage')}\nfunction refreshRunning(){return false;}\nfunction startRefresh(strength){(window.__refreshCalls ||= []).push(strength);return Promise.resolve(true); }\n${source('host-readiness')}\nwireHostHealth();`}); await page.evaluate(data=>globalThis.renderHostReadiness(data),unmanagedReport()); // Header pills: health for the managed host, the management words otherwise, never amber. @@ -144,10 +163,10 @@ test('unmanaged hosts read their management state everywhere, with information-o assert.match(await page.locator('#host-health-participation').innerText(),/not participating/i); assert.equal(await page.locator('#host-health-participation code').innerText(),'ak host pick --host claude,codex'); assert.equal(await page.locator('#host-health-participation [data-copy]').getAttribute('data-copy'),'ak host pick --host claude,codex'); - assert.equal(await page.locator('#host-health-refresh').innerText(),'Check again'); - await page.locator('#host-health-refresh').click(); - await page.waitForFunction(()=>globalThis.document.getElementById('host-health-message').textContent==='Check completed.'); - assert.deepEqual(requests,['/api/host-health/local']); + assert.equal(await page.locator('#host-health-run-refresh').innerText(),'Refresh'); + await page.locator('#host-health-run-refresh').click(); + assert.deepEqual(await page.evaluate(() => globalThis.__refreshCalls), ['local']); + assert.deepEqual(requests, []); await page.keyboard.press('Escape'); // Participation view (Overview → Hosts & Routing): one row per host, the hint copyable text. @@ -200,7 +219,7 @@ test('a dialog close that lands after the user moved on does not steal focus bac await page.route('http://health.test/**', route => route.fulfill({ contentType: 'text/html', body: renderPage({ name: 'Health fixture', version: 'test' }).replace(/]*>[\s\S]*?<\/script>/gi, '') })); await page.goto('http://health.test/'); - await page.addScriptTag({ content: `${esc.toString()}\nfunction authHeaders(){return {};}\n${source('usage')}\n${source('host-readiness')}\nwireHostHealth();` }); + await page.addScriptTag({ content: `${esc.toString()}\nfunction authHeaders(){return {};}\n${source('usage')}\nfunction refreshRunning(){return false;}\nfunction startRefresh(strength){(window.__refreshCalls ||= []).push(strength);return Promise.resolve(true); }\nfunction refreshRunning(){return false;}\nfunction startRefresh(strength){(window.__refreshCalls ||= []).push(strength);return Promise.resolve(true); }\n${source('host-readiness')}\nwireHostHealth();` }); await page.evaluate(data => globalThis.renderHostReadiness(data), report()); await page.locator('[data-health-host="claude"]').click(); // The race, made deterministic: close the dialog and move focus in the SAME diff --git a/tests/ui/intelligence-picker.mjs b/tests/ui/intelligence-picker.mjs index 719670d1..ece6e6ec 100644 --- a/tests/ui/intelligence-picker.mjs +++ b/tests/ui/intelligence-picker.mjs @@ -51,7 +51,7 @@ test('Intelligence picker labels every learning location with the inventory desi values: Array.from(node.children, option => option.value), labels: Array.from(node.children, option => option.textContent) }))); assert.deepEqual(groups.map(group => [group.label, group.values]), [ ['Git repositories', ['repo-a', 'repo-z']], ['Git worktrees', ['tree-a', 'tree-z']], - ['User-level learning', ['user-a', 'user-z']], ['Other / unclassified', ['unknown-a', 'unknown-z']], + ['User-level learning', ['user-a', 'user-z']], ['Unknown', ['unknown-a', 'unknown-z']], ]); assert.equal(groups[0].labels[0], 'alpha — Git repository'); assert.equal(groups[1].labels[0], 'Alpha tree — Git worktree'); @@ -117,7 +117,7 @@ test('Intelligence table keeps every learning location in one filterable invento assert.equal(await table.getByRole('columnheader').count(), 5); assert.deepEqual(await table.getByRole('columnheader').allTextContents(), ['Name', 'Designation', 'Patterns learned', 'Pattern store', 'Last active']); - assert.deepEqual(await table.locator('.mw-filter-pill').allTextContents(), ['All', 'Directory', 'Git repository', 'Git worktree']); + assert.deepEqual(await table.locator('.mw-filter-pill').allTextContents(), ['All', 'Git repository', 'Git worktree', 'Unknown', 'User-level learning']); const shots = process.env.AK_UI_ARTIFACTS; if (shots) fs.mkdirSync(shots, { recursive: true }); for (const width of [1360, 1100, 390]) { @@ -127,7 +127,7 @@ test('Intelligence table keeps every learning location in one filterable invento header: region.querySelector('.mw-head').getBoundingClientRect().height, row: region.querySelector('.mw-data-row').getBoundingClientRect().height, }))); - for (const size of sizes) { assert.equal(size.row, 34); assert.ok(size.scroll > size.height); } + for (const size of sizes) { assert.ok(size.row >= 34); assert.ok(size.scroll > size.height); } assert.ok(await table.evaluate(el => el.getBoundingClientRect().height) <= 520); assert.equal(await page.evaluate(() => globalThis.document.documentElement.scrollWidth <= globalThis.innerWidth), true, `no horizontal overflow at ${width}px`); assert.equal(await page.locator('#mw-hero').innerText(), hero); @@ -143,6 +143,6 @@ test('Intelligence table keeps every learning location in one filterable invento assert.equal(await table.locator('.mw-data-row').count(), 32); await table.getByRole('button', { name: 'Git repository', exact: true }).click(); assert.equal(await table.locator('.mw-data-row').count(), 8); - assert.deepEqual(await table.locator('.mw-designation').allTextContents(), Array(8).fill('Git repository')); + assert.deepEqual(await table.locator('.mw-scope-value').allTextContents(), Array(8).fill('Git repository')); assert.deepEqual(errors, []); }); diff --git a/tests/ui/maintenance-focus.mjs b/tests/ui/maintenance-focus.mjs index 22b82fb9..f8e11e22 100644 --- a/tests/ui/maintenance-focus.mjs +++ b/tests/ui/maintenance-focus.mjs @@ -1,3 +1,4 @@ +import { SESSION_SURFACE_LABELS, SESSION_HOST_LABELS, SESSION_INITIATOR_LABELS, SESSION_PROVIDER_LABELS, sessionPresentation } from '../../src/lib/session-surface.mjs'; // Real focus queries, public DTOs, page markup and browser modules. import { test } from 'node:test'; import assert from 'node:assert/strict'; @@ -36,7 +37,9 @@ test('focus browser progressively narrows to exact installations and preserves f function esc(v){return String(v).replace(/&/g,'&').replace(/

    Projects

    '); await page.addScriptTag({content:'var MNT='+JSON.stringify(state)+';var MNT_SCOPE_LABELS={};function esc(s){return String(s).replace(/&/g,"&").replace(/"']/g,function(c){return {'&':'&','<':'<','>':'>','"':'"',"'":'''}[c];});} function ago(){return '';} + function refreshRunning(){return false;} function beginMaintPreview(button, request){window.selectedPreview=request;} - ${['maintenance-workspace','maintenance-operation','maintenance-cards','maintenance-filters','maintenance-guidance','maintenance-relationships','maintenance-inspector','maintenance-language-logos','maintenance-focus','maintenance-inventory'].map(clientSource).join('\n')} + const SESSION_SURFACE_LABELS=${JSON.stringify(SESSION_SURFACE_LABELS)},SESSION_HOST_LABELS=${JSON.stringify(SESSION_HOST_LABELS)},SESSION_INITIATOR_LABELS=${JSON.stringify(SESSION_INITIATOR_LABELS)},SESSION_PROVIDER_LABELS=${JSON.stringify(SESSION_PROVIDER_LABELS)}; + ${sessionPresentation.toString()} + ${['session-presentation','maintenance-workspace','maintenance-operation','maintenance-cards','maintenance-filters','maintenance-guidance','maintenance-relationships','maintenance-inspector','maintenance-language-logos','maintenance-focus','maintenance-inventory'].map(clientSource).join('\n')} MNT.scope='user';MNT.view='host-alignment';wireMntInventory();wireMntInspector();wireMntGuidance();loadMntInventory(); ` }); await page.locator('[data-mnt-plc]').first().waitFor(); diff --git a/tests/ui/maintenance-projects.mjs b/tests/ui/maintenance-projects.mjs index c7a1ffa5..08cf3ed9 100644 --- a/tests/ui/maintenance-projects.mjs +++ b/tests/ui/maintenance-projects.mjs @@ -1,3 +1,4 @@ +import { SESSION_SURFACE_LABELS, SESSION_HOST_LABELS, SESSION_INITIATOR_LABELS, SESSION_PROVIDER_LABELS, sessionPresentation } from '../../src/lib/session-surface.mjs'; // Focused browser regression: real workspace markup, styles, client modules, // and public inventory projection; HTTP evidence is supplied by a fixed fixture. import { test } from 'node:test'; @@ -16,6 +17,7 @@ function fixture() { const projects = ['ampel', 'boon-worthy', 'emailibrium', 'finima', 'keel', 'prompt-genie', 'ampel-feature'].map((name) => ({ loc: { languages: (name === 'ampel' ? ['javascript', 'python', 'rust', 'java', 'ada'] : ['typescript']).map(id => ({ id })) }, repository: ['ampel','ampel-feature'].includes(name)?{repositoryId:'repository:0123456789abcdef0123',kind:name==='ampel-feature'?'worktree':'git',root:'/fixture/projects/ampel',evidence:name==='ampel-feature'?'git-common-directory-and-backlink':'git-directory',observedAt:Date.parse('2026-09-09T12:00:00Z')}:null, + sessionSurfaces: [{host:'codex',surface:name==='ampel-feature'?'chatgpt-desktop-work':'unknown',initiator:'person',sessions:1}], sessionOrigins: [{origin:name==='ampel-feature'?'codex-desktop':name==='ampel'?'claude-desktop':'unknown',sessions:1}], path: '/fixture/projects/'+name, label: name, hosts: ['claude', 'codex'], projectKind: name==='ampel-feature'?'worktree':'git', })); @@ -68,7 +70,9 @@ test('project worktree visibility and all-installations navigation work on deskt function authHeaders(){return {};} function esc(value){return String(value).replace(/[&<>"']/g,function(c){return {'&':'&','<':'<','>':'>','"':'"',"'":'''}[c];});} function ago(){return '';} - ${['maintenance-workspace', 'maintenance-cards', 'maintenance-filters', 'maintenance-guidance', 'maintenance-relationships', 'maintenance-inspector', 'maintenance-language-logos','maintenance-focus', 'maintenance-inventory'].map(clientSource).join('\n')} + const SESSION_SURFACE_LABELS=${JSON.stringify(SESSION_SURFACE_LABELS)},SESSION_HOST_LABELS=${JSON.stringify(SESSION_HOST_LABELS)},SESSION_INITIATOR_LABELS=${JSON.stringify(SESSION_INITIATOR_LABELS)},SESSION_PROVIDER_LABELS=${JSON.stringify(SESSION_PROVIDER_LABELS)}; + ${sessionPresentation.toString()} + ${['session-presentation','maintenance-workspace', 'maintenance-cards', 'maintenance-filters', 'maintenance-guidance', 'maintenance-relationships', 'maintenance-inspector', 'maintenance-language-logos','maintenance-focus', 'maintenance-inventory'].map(clientSource).join('\n')} MNT.scope='project';wireMntInventory();wireMntInspector();loadMntInventory(); ` }); await page.locator('#mnt-results [data-mnt-focus]').first().waitFor(); @@ -82,11 +86,11 @@ test('project worktree visibility and all-installations navigation work on deskt assert.equal(await page.locator('[data-mnt-focus="'+worktreeId+'"]').count(), 1); const sharedGroup=page.locator('.mnt-repository-group').filter({has:page.locator('[data-mnt-focus="'+worktreeId+'"]')}); assert.equal(await sharedGroup.locator('[data-mnt-level="project"]').count(),2); - const originFilter=page.locator('#mnt-facets input[data-mnt-facet="sessionOrigin"][value="codex-desktop"]'); + const originFilter=page.locator('#mnt-facets input[data-mnt-facet="sessionOrigin"][value="chatgpt-desktop-work"]'); await originFilter.check();await page.waitForFunction(()=>!globalThis.mntInventoryBusy); assert.equal(await page.locator('#mnt-results [data-mnt-level="project"]').count(),1); - assert.match(await page.locator('#mnt-facets').innerText(),/ChatGPT Desktop/); - assert.doesNotMatch(await page.locator('#mnt-results').innerText(),/ChatGPT Desktop/); + assert.match(await page.locator('#mnt-facets').innerText(),/ChatGPT desktop app · ChatGPT Work/); + assert.match(await page.locator('#mnt-results').innerText(),/ChatGPT desktop app · ChatGPT Work/); await originFilter.uncheck();await page.waitForFunction(()=>!globalThis.mntInventoryBusy); await page.locator('#mnt-facets [data-mnt-include-worktrees]').uncheck(); diff --git a/tests/ui/session-surfaces.mjs b/tests/ui/session-surfaces.mjs new file mode 100644 index 00000000..7bd22283 --- /dev/null +++ b/tests/ui/session-surfaces.mjs @@ -0,0 +1,64 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { launchChrome } from './helpers/launch-chrome.mjs'; +import { renderPage } from '../../src/lib/dashboard/page.mjs'; +const sessionSurfaces = [{ host: 'codex', surface: 'chatgpt-desktop-work', initiator: 'agent', sessions: 2, + countBasis: 'transcript-files', rawEvidence: { originator: ['codex_work_desktop'], source: ['vscode'] } }, +{ host: 'claude', surface: 'cloud-session', initiator: 'automation', sessions: 1 }]; +const project = { key: 'example', path: '/fixture/example', label: '', learningScope: 'repository', sessionSurfaces, + repository: { kind: 'git', repositoryId: 'fixture-repo', root: '/fixture/example' } }; +const counts = { everSeen: 1, onDisk: 1, gitRepos: 1, learning: 1, importedExcluded: 4, importedMixed: 2, importedUnresolved: 3 }; +test('served dashboard renders independent session evidence and preserves observed surface filters', async t => { + const browser = await launchChrome(); t.after(() => browser.close()); + const page = await browser.newPage({ viewport: { width: 1440, height: 1000 } }); + let currentProject = project; + const errors = []; page.on('pageerror', error => errors.push(error.message)); + await page.route('http://surfaces.test/**', async route => { + const url = new URL(route.request().url()); + if (url.pathname === '/') return route.fulfill({ contentType: 'text/html', body: renderPage({ name: 'Surfaces', version: 'test' }) }); + const body = url.pathname === '/api/status' ? { overall: 'ok', rows: [], intel: { projects: [currentProject], + census: { counts }, machineWide: { totals: { projectCount: 1 }, perProject: [currentProject] } } } + : url.pathname === '/api/usage' ? { totals: { sessions: 1 }, sessions: [], projectTree: [{ project: 'Example', sessions: 1, rows: [{ id: 'fixture-session', host: 'claude', sessionOrigin: { surface: 'claude-desktop', initiator: 'person', thirdPartyProvider: 'amazon-bedrock', thirdPartyProviderBasis: 'assistant-model-id', rawEvidence: { entrypoint: 'claude-desktop-3p' } } }] }] } + : url.pathname === '/api/system/summary' ? { projects: { ...counts, projects: [project], discoveryProjects: [project] } } : {}; + return route.fulfill({ contentType: 'application/json', body: JSON.stringify(body) }); + }); + await page.goto('http://surfaces.test/#token=fixture'); + await page.click('[data-overview-view="intel"]'); + const filter = page.locator('#mw-surface-filter'); await filter.waitFor(); + assert.deepEqual(await filter.locator('option').allTextContents(), ['All', 'ChatGPT desktop app · ChatGPT Work (local)', 'Cloud session']); + await filter.selectOption({ label: 'Cloud session' }); + await page.locator('#mw-table .session-surface-detail summary').click(); + assert.match(await page.locator('#mw-table').innerText(), /Git repository/); + assert.match(await page.locator('#mw-table').innerText(), /initiator: Agent/); + assert.match(await page.locator('#mw-table').innerText(), /source: vscode/); + assert.doesNotMatch(await page.locator('#mw-table').innerText(), /Codex IDE extension/); + await page.locator('#mw-census summary').click(); + assert.match(await page.locator('#mw-census-body').innerText(), /3 files have unresolved bounded ownership/); + for (const width of [1440, 390]) { + await page.setViewportSize({ width, height: 1000 }); + assert.equal(await page.evaluate(() => globalThis.document.documentElement.scrollWidth <= globalThis.innerWidth), true); + } + await page.click('#tab-system'); await page.click('[data-system-view="projects"]'); + await page.locator('#sys-projects .session-surface-detail summary').click(); + assert.match(await page.locator('#sys-projects').innerText(), /ChatGPT desktop app · ChatGPT Work \(local\)/); + assert.match(await page.locator('#sys-projects').innerText(), /4 confirmed pure imported copies excluded/); + assert.match(await page.locator('#sys-projects').innerText(), /dedicated Cowork transcript source is not covered/); + assert.equal(await page.locator('#sys-projects img').count(), 0); + await page.click('#tab-overview'); await page.click('[data-overview-view="intel"]'); + assert.equal(await filter.inputValue(), 'Cloud session'); + currentProject = { ...project, sessionSurfaces: undefined, sessionOrigins: [{ origin: 'claude-desktop', sessions: 3 }] }; + await page.click('#poll-now'); + await page.waitForFunction(() => globalThis.document.querySelector('#mw-surface-filter').value === 'all'); + assert.deepEqual(await filter.locator('option').allTextContents(), ['All', 'Claude Desktop']); + assert.equal(await page.locator('#mw-table .mw-data-row').count(), 1); + assert.match(await page.locator('#mw-table').innerText(), /Claude Desktop/); + await page.setViewportSize({ width: 1440, height: 1000 }); + await page.click('#tab-usage'); await page.click('#usage-tab-sessions'); + await page.locator('#u-tree .phead').click(); + await page.locator('#u-tree .s-exp').click(); + const detail = await page.locator('#sd-fixture-session').innerText(); + assert.match(detail, /Claude Desktop/); assert.match(detail, /Amazon Bedrock/); + assert.match(detail, /assistant-model-id; not network attestation/); + assert.match(detail, /claude-desktop-3p/); + assert.deepEqual(errors, []); +});