From b78d4fdd80a2457ec41598e75c64290cfec2438d Mon Sep 17 00:00:00 2001 From: Chris Phillipson Date: Mon, 27 Jul 2026 08:21:16 -0700 Subject: [PATCH 1/3] Usage panel: vendor-reported plan limits, Codex ledger attribution MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The scorecard could say what tokens cost but never how much of the plan they consumed — ADR-0009 §3 excluded that deliberately, because a locally computed percentage needs a denominator no vendor publishes. Both vendors now hand over their own percentages through supported channels, so the denominator is no longer invented and the exclusion no longer applies. ADR-0010 defines the two admissible channels, both credential-free for ak: claude — Claude Code PUSHES rate_limits (session/weekly/per-model used percentage + reset epochs) into every statusLine invocation on Pro/Max. The managed footer tees that payload to claude-rate-limits.json. codex — one initialize -> account/rateLimits/read exchange with a spawned `codex app-server`, which authenticates itself. Same shell-out trust model as `ak status --json`; TTL-cached. Explicit non-paths: no /api/oauth/usage (undocumented, and consumer-OAuth use outside Claude Code is ToS-prohibited and server-enforced since Jan 2026), no Keychain reads, no chatgpt.com backend endpoints, and no auto-consumption of Codex reset credits. Windows are keyed by DURATION, never by the vendor's primary/secondary slot: a live prolite account reported `primary` as the 10080-minute weekly window. Field-name trust would have mislabelled every bar. Codex attribution stops being heuristic. Codex keeps its own thread ledger (state_N.sqlite — the suffix is a migration generation, so glob it) with thread_source and spawn edges; a ledger-identified subagent has its usage stripped, since its rollout replays the parent's whole token history (ccusage#950 measured up to 91x inflation). The rollout sniff from #60 remains the fallback. Rollouts also yield reasoning tokens (a subset of output — annotation only, never summed) and embedded rate-limit snapshots, making a utilization history reconstructable with zero network. Six new detectors under the existing evidence rules — vendor percentages are the user's own data, and no dollar impact is ever claimed from a percentage. "Now" is the payload's generatedAt, never a clock, so every firing is reproducible from its input. Also: the classify-coverage finding advertised `ak x usage classify --enrich`, which does not exist — a dead command in a diagnostic is worse than none. And the Sonnet 5 introductory price was a comment promising a 2026-09-01 revert with nothing enforcing it; it is now a test that fails the suite from that date until the table is corrected. The Providers strip is retitled "routed models" and explains itself: it is the per-activity policy projected into agentic-qe agent overrides and `ak dual run`, not a record of what ran. agentic-qe carries its own model router, so a route there is an assignment, not a guarantee. Verified: pnpm run check green (569 unit + all cjs suites, incl. 47 new tests and the self-verifying doc citations, re-anchored after drift); Playwright artifact net 102/102, now also failing on any visible ADR id. --- docs/TRANSCRIPTS.md | 106 +++---- docs/USAGE-SCORECARD-METRICS.md | 191 ++++++++----- ...ge-scorecard-local-transcript-analytics.md | 7 + .../adr/0010-provider-mediated-quota-reads.md | 98 +++++++ src/lib/codex-state.mjs | 79 ++++++ src/lib/dashboard-server.mjs | 216 +++++++++++++-- src/lib/quota.mjs | 258 ++++++++++++++++++ src/lib/usage-index.mjs | 103 ++++++- src/lib/usage-insights.mjs | 217 ++++++++++++++- src/templates/statusline-footer.cjs | 33 +++ tests/dashboard.test.cjs | 48 +++- tests/kit/codex-state.test.mjs | 121 ++++++++ tests/kit/pricing-revert.test.mjs | 28 ++ tests/kit/quota.test.mjs | 194 +++++++++++++ tests/kit/usage-index-v6.test.mjs | 132 +++++++++ tests/kit/usage-insights.test.mjs | 4 +- tests/kit/usage-limit-insights.test.mjs | 203 ++++++++++++++ tests/ui/dashboard-ui.mjs | 29 ++ 18 files changed, 1910 insertions(+), 157 deletions(-) create mode 100644 docs/adr/0010-provider-mediated-quota-reads.md create mode 100644 src/lib/codex-state.mjs create mode 100644 src/lib/quota.mjs create mode 100644 tests/kit/codex-state.test.mjs create mode 100644 tests/kit/pricing-revert.test.mjs create mode 100644 tests/kit/quota.test.mjs create mode 100644 tests/kit/usage-index-v6.test.mjs create mode 100644 tests/kit/usage-limit-insights.test.mjs diff --git a/docs/TRANSCRIPTS.md b/docs/TRANSCRIPTS.md index ccc660d2..58c59f20 100644 --- a/docs/TRANSCRIPTS.md +++ b/docs/TRANSCRIPTS.md @@ -35,44 +35,44 @@ rewritten; rule 3 of the module header, `usage-index.mjs:22-29`): | Provider | Store | Discovered by | |---|---|---| -| Claude Code | `~/.claude/projects//.jsonl` | `listClaude` (`usage-index.mjs:607`) — exactly one level of project directories | -| Codex CLI | `~/.codex/sessions///
/rollout--.jsonl` | `listCodex` (`usage-index.mjs:622`) — the `yyyy/mm/dd` tree walk | +| Claude Code | `~/.claude/projects//.jsonl` | `listClaude` (`usage-index.mjs:648`) — exactly one level of project directories | +| Codex CLI | `~/.codex/sessions///
/rollout--.jsonl` | `listCodex` (`usage-index.mjs:663`) — the `yyyy/mm/dd` tree walk | -Roots come from `defaultRoots()` (`usage-index.mjs:599`) and are injectable +Roots come from `defaultRoots()` (`usage-index.mjs:640`) and are injectable for tests. A malformed line is skipped, never fatal (`jsonLines`, -`usage-index.mjs:274` — one corrupt line must not cost a whole file). +`usage-index.mjs:280` — one corrupt line must not cost a whole file). ### 1.1 Claude entry vocabulary Each line has a top-level `type`. The parser (`parseClaude`, -`usage-index.mjs:417-500`) reads: +`usage-index.mjs:427-510`) reads: | `type` | What the parser takes from it | |---|---| -| `ai-title` | The model-written session title (`usage-index.mjs:425`) — preferred over the first-prompt fallback | +| `ai-title` | The model-written session title (`usage-index.mjs:435`) — preferred over the first-prompt fallback | | `user` | A user-**role** turn — which is *not* the same as "the human"; see §3 | -| `assistant` | A model turn: `model` id, per-turn `usage` token counts, `tool_use` blocks (`usage-index.mjs:445-488`) | -| any | Side-band fields read regardless of type: `attributionSkill`/`attributionPlugin` (`usage-index.mjs:426-427`), `isSidechain` (`usage-index.mjs:428`), `cwd` for project derivation | +| `assistant` | A model turn: `model` id, per-turn `usage` token counts, `tool_use` blocks (`usage-index.mjs:455-498`) | +| any | Side-band fields read regardless of type: `attributionSkill`/`attributionPlugin` (`usage-index.mjs:436-437`), `isSidechain` (`usage-index.mjs:438`), `cwd` for project derivation | An assistant entry with `isApiErrorMessage: true` is a **local placeholder** Claude Code writes when a request dies before a real completion (connection drop, rate limit, auth failure — `model: ""`, all-zero usage). It is real engaged time but not a model attempt: counted as an *exception*, never -pushed into `models` or priced (`usage-index.mjs:459-468`; the full story is +pushed into `models` or priced (`usage-index.mjs:469-478`; the full story is [`USAGE-SCORECARD-METRICS.md`](USAGE-SCORECARD-METRICS.md) §10). ### 1.2 Codex entry vocabulary Codex rollout lines carry `type` + `payload`. The parser (`parseCodex`, -`usage-index.mjs:517-587`) reads: +`usage-index.mjs:527-628`) reads: | `type` / `payload.type` | What the parser takes from it | |---|---| -| `session_meta` | Authoritative session id, `cwd`, and `thread_source` (`usage-index.mjs:529-534`) — `"subagent"` marks a thread_spawn replay whose tokens are excluded from aggregation (`usage-index.mjs:572`; `USAGE-SCORECARD-METRICS.md` Appendix A, Bug B) | -| `turn_context` | The model id in effect from this point on (`usage-index.mjs:535`) | -| `event_msg` → `token_count` | A **cumulative** usage snapshot; only the last one is kept (`usage-index.mjs:542`) | -| `event_msg` → `user_message` | A real human prompt — Codex does not route tool output through this event (`usage-index.mjs:547-555`) | -| `event_msg` → `agent_message` | A model response (`usage-index.mjs:557-568`) | +| `session_meta` | Authoritative session id, `cwd`, and `thread_source` (`usage-index.mjs:539-544`) — `"subagent"` marks a thread_spawn replay whose tokens are excluded from aggregation (`usage-index.mjs:609`; `USAGE-SCORECARD-METRICS.md` Appendix A, Bug B) | +| `turn_context` | The model id in effect from this point on (`usage-index.mjs:545`) | +| `event_msg` → `token_count` | A **cumulative** usage snapshot; only the last one is kept (`usage-index.mjs:552`) | +| `event_msg` → `user_message` | A real human prompt — Codex does not route tool output through this event (`usage-index.mjs:584-592`) | +| `event_msg` → `agent_message` | A model response (`usage-index.mjs:594-605`) | Codex tool calls and tool outputs travel in event types the parser does not surface as turns at all — so a Codex transcript renders as a prompt/response @@ -88,8 +88,8 @@ The same parsers serve two very different callers, switched by `withTurns`: | Path | Entry point | `withTurns` | Message bodies | Cached? | |---|---|---|---|---| -| **Scan** — the aggregate index behind the Scorecard/Findings/Sessions views | `buildIndex` → `parseFile` (`usage-index.mjs:652`) | `false` | never held — holding them would balloon memory across 3,000+ files (`usage-index.mjs:413-416`) | yes: per-file derived records in `~/.config/agentic-kit/usage-index.json`, keyed `(path, mtime, size)`, invalidated wholesale by `SCHEMA_VERSION` (`usage-index.mjs:50`) | -| **Reader** — one transcript for the Transcript view | `readSession` (`usage-index.mjs:1078`) | `true` | full turn list built | **never** — every call re-reads and re-parses the one file | +| **Scan** — the aggregate index behind the Scorecard/Findings/Sessions views | `buildIndex` → `parseFile` (`usage-index.mjs:693`) | `false` | never held — holding them would balloon memory across 3,000+ files (`usage-index.mjs:423-426`) | yes: per-file derived records in `~/.config/agentic-kit/usage-index.json`, keyed `(path, mtime, size)`, invalidated wholesale by `SCHEMA_VERSION` (`usage-index.mjs:51`) | +| **Reader** — one transcript for the Transcript view | `readSession` (`usage-index.mjs:1173`) | `true` | full turn list built | **never** — every call re-reads and re-parses the one file | ![Figure: one parser, two read paths — the scan path (withTurns false) caches per-file records keyed by path, mtime and size; the reader path (withTurns true) builds full turns and is never cached](assets/transcript-read-paths.svg) @@ -97,7 +97,7 @@ The reader path being cache-free is load-bearing for maintainers: **turn-shape changes (like the `kind` field, §3) need no `SCHEMA_VERSION` bump**, because no turn is ever served from cache — whereas *session-record* fields (like `exceptions`) do, since stale cached records would otherwise sum `undefined` -into totals (`usage-index.mjs:39-49`; the incidents behind that rule are +into totals (`usage-index.mjs:40-50`; the incidents behind that rule are recorded in `USAGE-SCORECARD-METRICS.md` Appendix A). --- @@ -110,11 +110,11 @@ recorded in `USAGE-SCORECARD-METRICS.md` Appendix A). |---|---|---| | `role` | all | `"user"` or `"assistant"` — the **Messages-API role**, not the author (see below) | | `at` | all | ISO timestamp | -| `text` | all | Flattened display text (`claudeText`, `usage-index.mjs:338` — binary payloads dropped: a pasted screenshot renders as `[image]`, a tool result is prefixed `[tool result]`) | -| `model` | assistant | The model id; the literal string `exception` for an API-error placeholder turn (`usage-index.mjs:463`) | +| `text` | all | Flattened display text (`claudeText`, `usage-index.mjs:348` — binary payloads dropped: a pasted screenshot renders as `[image]`, a tool result is prefixed `[tool result]`) | +| `model` | assistant | The model id; the literal string `exception` for an API-error placeholder turn (`usage-index.mjs:473`) | | `tools` | assistant | Tool names invoked in the turn | -| `prompt` | user | `isHumanPrompt`'s verdict (`usage-index.mjs:378-387`) — drives the **prompt counts** | -| `kind` | user | `'prompt'` \| `'tool-result'` \| `'context'` — drives the **attribution label** (`userTurnKind`, `usage-index.mjs:405-416`) | +| `prompt` | user | `isHumanPrompt`'s verdict (`usage-index.mjs:388-397`) — drives the **prompt counts** | +| `kind` | user | `'prompt'` \| `'tool-result'` \| `'context'` — drives the **attribution label** (`userTurnKind`, `usage-index.mjs:415-426`) | | `exception` | assistant | `true` on API-error placeholder turns | | `truncated`, `originalChars` | any | Present **only** when the turn was abridged (§4.3) | @@ -134,7 +134,7 @@ story is [Appendix A](#appendix-a--fix-history).) ### 3.2 `kind` — the attribution field -`userTurnKind` (`usage-index.mjs:405-416`) classifies every user-role turn: +`userTurnKind` (`usage-index.mjs:415-426`) classifies every user-role turn: | `kind` | Test | Meaning | |---|---|---| @@ -152,12 +152,12 @@ Two deliberate subtleties: - **Harness-output envelopes are excluded from the prompt *count* too.** `isHumanPrompt` shares `HARNESS_OUTPUT_RE`, so a session's `prompts` figure never counts stdout dumps or task notifications as things the person said - (`SCHEMA_VERSION` 5, `usage-index.mjs:46-50`; the correction this shipped + (`SCHEMA_VERSION` 5, `usage-index.mjs:47-51`; the correction this shipped with is in [Appendix A](#appendix-a--fix-history)). - **`tool-result` outranks `context`**: a `tool_result` block on an `isMeta` entry is still tool feedback. -Codex user turns are `kind: 'prompt'` by construction (`usage-index.mjs:547-555`) +Codex user turns are `kind: 'prompt'` by construction (`usage-index.mjs:584-592`) — rollouts only record real prompts as `user_message` events (§1.2). Coverage: `tests/kit/usage-index.test.mjs` — "user-role turns carry a kind" @@ -168,31 +168,31 @@ and image-only pastes get the right kind" (the two edges). ## 4. The `readSession` pipeline — how one session becomes a payload -`readSession(id, opts)` (`usage-index.mjs:1078-1161`) is the only way +`readSession(id, opts)` (`usage-index.mjs:1173-1256`) is the only way transcript content leaves the module, and every step is a gate: ### 4.1 Locate, contain, bound 1. **Id grammar before any filesystem access** — `VALID_ID` - (`/^[A-Za-z0-9._-]{1,128}$/`, `usage-index.mjs:65`) rejects traversal - shapes with `ERR_INVALID_SESSION_ID` (`usage-index.mjs:1079`). -2. **Locate by id** across both roots (`locate`, `usage-index.mjs:1037`), + (`/^[A-Za-z0-9._-]{1,128}$/`, `usage-index.mjs:71`) rejects traversal + shapes with `ERR_INVALID_SESSION_ID` (`usage-index.mjs:1174`). +2. **Locate by id** across both roots (`locate`, `usage-index.mjs:1132`), consulting the scan cache when present but never requiring it — `readSession` works with no prior `buildIndex`. -3. **Realpath containment** (`usage-index.mjs:1085-1099`) — the resolved file +3. **Realpath containment** (`usage-index.mjs:1180-1194`) — the resolved file must live under a transcript root *after* `realpathSync` collapses symlinks; a symlink planted inside a root pointing at `/etc/anything` passes a lexical `startsWith` but fails this. Roots are realpath'd too so a symlinked dotfiles setup still works. -4. **Size cap** — `MAX_SESSION_BYTES` (64 MB, `usage-index.mjs:64`): a +4. **Size cap** — `MAX_SESSION_BYTES` (64 MB, `usage-index.mjs:70`): a transcript is read whole and JSON-expands ~5×, so an unbounded read is a memory-amplification primitive. Oversized reads as unavailable, not risky. ### 4.2 Parse and price The file is parsed with `withTurns: true` by the provider's parser -(`usage-index.mjs:1110-1116`), and `meta` is assembled -(`usage-index.mjs:1124-1143`) with the same fields the Sessions view rows +(`usage-index.mjs:1205-1211`), and `meta` is assembled +(`usage-index.mjs:1219-1238`) with the same fields the Sessions view rows carry — `prompts`, `responses`, `exceptions`, `sidechain`, `threadSource`, `models`, `tools`, `skill`/`plugin`, worktree — plus a `cost` priced from the same per-model usage rows `aggregate()` uses (the header used to render a @@ -200,11 +200,11 @@ hardcoded `$0.00`; the comment at the site records why). ### 4.3 Mask, then truncate — both marked, differently -Every turn body is passed through `maskSecrets` (`usage-index.mjs:166` — the +Every turn body is passed through `maskSecrets` (`usage-index.mjs:172` — the 21 secret shapes) **server-side, before serialization**, then length-capped at `MAX_TURN_CHARS` (40,000, -`usage-index.mjs:59`) with the marker appended -(`usage-index.mjs:1144-1160`). Two invariants: +`usage-index.mjs:65`) with the marker appended +(`usage-index.mjs:1239-1255`). Two invariants: - **Presence is the signal.** `truncated`/`originalChars` are emitted only when the slice fired, so a complete turn cannot be misread as abridged. @@ -213,8 +213,8 @@ serialization**, then length-capped at `MAX_TURN_CHARS` (40,000, The two kinds of withholding keep distinct vocabulary end-to-end: masking renders as `…redacted` marks (`markRedactions`, -`dashboard-server.mjs:2184`), truncation as a `truncated · N of M` badge -(`truncBadge`, `dashboard-server.mjs:2228`, deriving N from the received +`dashboard-server.mjs:2333`), truncation as a `truncated · N of M` badge +(`truncBadge`, `dashboard-server.mjs:2377`, deriving N from the received text so a changed constant can't desync the display). ![Figure: a turn body passes through maskSecrets (leaving redaction marks) and then the 40,000-character cap (leaving a truncated · N of M badge); originalChars is measured after masking](assets/transcript-mask-truncate.svg) @@ -229,9 +229,9 @@ preamble). Transcript-relevant routes: | Route | Serves | Notes | |---|---|---| -| `GET /api/usage?days=N` | the aggregate minus `sessions[]` (`dashboard-server.mjs:337`) | Scorecard + Findings + the project tree | -| `GET /api/sessions` | session rows, filtered/paginated (`dashboard-server.mjs:355`) | the Sessions view's "load all" | -| `GET /api/session/:id` | one transcript (`dashboard-server.mjs:372-403`) | the Transcript view | +| `GET /api/usage?days=N` | the aggregate minus `sessions[]` (`dashboard-server.mjs:341`) | Scorecard + Findings + the project tree | +| `GET /api/sessions` | session rows, filtered/paginated (`dashboard-server.mjs:375`) | the Sessions view's "load all" | +| `GET /api/session/:id` | one transcript (`dashboard-server.mjs:392-423`) | the Transcript view | `/api/session/:id` order of operations, each step deliberate: @@ -260,13 +260,13 @@ nothing on the page to reveal. ### 6.1 Sessions view — the row and its expander -`renderSessions` (`dashboard-server.mjs:2133`) renders the project tree +`renderSessions` (`dashboard-server.mjs:2282`) renders the project tree (collapsed by default; every project starts closed so the cross-project comparison stays above the fold). Each session is a `sessionRow` -(`dashboard-server.mjs:2107-2131`): host chip (claude/codex), title, +(`dashboard-server.mjs:2256-2280`): host chip (claude/codex), title, worktree glyph, category chip (dimmed when confidence < 0.6 or Unclassified), start, duration, `prompts/responses`, tokens, cost — and an -expander (`sdetail`, `dashboard-server.mjs:2074-2104`) carrying the +expander (`sdetail`, `dashboard-server.mjs:2220-2253`) carrying the per-session detail fields: classification `basis` + confidence, per-session `models`, the token split, top tools, and the `skill`/`plugin`/`sidechain`/`worktree` flags. A measured-but-absent value renders as `—`, never disappears — a @@ -274,15 +274,15 @@ field that vanishes when null teaches the reader it doesn't exist. ### 6.2 Transcript view — attribution, redaction, truncation -`renderTranscript` (`dashboard-server.mjs:2244`) renders the crumb (title, +`renderTranscript` (`dashboard-server.mjs:2393`) renders the crumb (title, project, duration, `prompts/responses`, tokens, cost — all from masked `meta`) and the turn list. **The label comes from `kind`, never from role** -(`dashboard-server.mjs:2260-2277`): +(`dashboard-server.mjs:2409-2426`): | Turn | Label | Styling | |---|---|---| -| user, `kind: 'prompt'` | `you` | accent — reserved for the person (`.t-user .t-who`, `dashboard-server.mjs:1270`) | -| user, `kind: 'tool-result'` | `tool result` | purple, rhyming with the tool chips (`.t-tool .t-who`, `dashboard-server.mjs:1274`); hover title states the harness — not the person — sent it | +| user, `kind: 'prompt'` | `you` | accent — reserved for the person (`.t-user .t-who`, `dashboard-server.mjs:1311`) | +| user, `kind: 'tool-result'` | `tool result` | purple, rhyming with the tool chips (`.t-tool .t-who`, `dashboard-server.mjs:1315`); hover title states the harness — not the person — sent it | | user, `kind: 'context'` | `context` | same purple + hover title | | assistant | the model id | dim mono (`exception` placeholder turns label as `exception`) | @@ -299,8 +299,8 @@ blocks (envelope census on the real corpus: task-notification 550, local-command-caveat 183, command-name 180, bash-input 85, bash-stdout 85, local-command-stdout 60; the stderr variants are the symmetric error-path siblings). Rendered literally they read as angle-bracket soup, so -`fmtHarness` (`dashboard-server.mjs:2197-2216`; CSS -`dashboard-server.mjs:1275-1287`) reformats them client-side: the command +`fmtHarness` (`dashboard-server.mjs:2346-2365`; CSS +`dashboard-server.mjs:1316-1328`) reformats them client-side: the command triple and `bash-input` become chips (`/clear`-style; the bash chip prefixed `!` so it reads as the shell invocation it was), and the block wrappers become quiet labelled @@ -312,8 +312,8 @@ verbatim; only the wrapper tags become styling.** It runs on escaped text turn truncation — is left raw rather than half-formatted. Deep links: `#usage/` opens the Transcript view directly -(`syncHash`, `dashboard-server.mjs:1403-1404`); the view lazy-fetches via -`loadTranscript` (`dashboard-server.mjs:1877`). +(`syncHash`, `dashboard-server.mjs:1444-1445`); the view lazy-fetches via +`loadTranscript` (`dashboard-server.mjs:1918`). --- @@ -351,7 +351,7 @@ was wrong before, for the curious. `isHumanPrompt` once counted harness-output envelopes as human prompts — 32 claimed vs 20 real on the reference session. Cached session records carried the inflated counts, hence the wholesale `SCHEMA_VERSION` 5 cache - invalidation (`usage-index.mjs:46-50`). + invalidation (`usage-index.mjs:47-51`). - **Session expander fields shipped but unrendered.** The per-session fields §6.1's expander now renders (classification `basis` + confidence, the token split, flags) once travelled on the wire and rendered nowhere. diff --git a/docs/USAGE-SCORECARD-METRICS.md b/docs/USAGE-SCORECARD-METRICS.md index a3f045db..72b61b99 100644 --- a/docs/USAGE-SCORECARD-METRICS.md +++ b/docs/USAGE-SCORECARD-METRICS.md @@ -57,15 +57,15 @@ Every metric section below follows the same shape: Two transcript stores, read-only, parsed at most once per file (cache keyed by `(path, mtime, size)`; `SCHEMA_VERSION` invalidates the whole cache on a -schema change — `src/lib/usage-index.mjs:50`): +schema change — `src/lib/usage-index.mjs:51`): | Provider | Store | Format | |---|---|---| | Claude Code | `~/.claude/projects//.jsonl` | one JSON object per line: `user`/`assistant` turns, each assistant turn carrying its own `usage` object | | Codex CLI | `~/.codex/sessions///
/rollout--.jsonl` | one JSON object per line: `session_meta`, `turn_context`, and `event_msg` records, the latter carrying **cumulative** `token_count` snapshots, not per-turn deltas | -The parsers are `parseClaude` (`usage-index.mjs:417-500`) and `parseCodex` -(`usage-index.mjs:517-587`). Both are pure functions over the raw file bytes — +The parsers are `parseClaude` (`usage-index.mjs:427-510`) and `parseCodex` +(`usage-index.mjs:527-628`). Both are pure functions over the raw file bytes — no network, no clock dependency beyond the transcript's own timestamps — so every downstream number traces back to bytes already on the user's disk. Nothing in this pipeline calls a provider API or a billing endpoint; **no @@ -89,16 +89,16 @@ responses = Σ over included sessions of session.responses **Source:** - Filter: a parsed record with zero assistant turns is dropped entirely — "no - assistant turn → not a session" (`usage-index.mjs:768`) — and a record whose + assistant turn → not a session" (`usage-index.mjs:809`) — and a record whose last activity falls outside the requested window is dropped too - (`usage-index.mjs:769`). + (`usage-index.mjs:810`). - `responses` accumulation: Claude increments per assistant message - (`usage-index.mjs:447`); Codex increments per `agent_message` event - (`usage-index.mjs:558`). + (`usage-index.mjs:457`); Codex increments per `agent_message` event + (`usage-index.mjs:595`). - Totals: `totals.responses += s.responses` per included session - (`usage-index.mjs:839`). + (`usage-index.mjs:884`). - Render: `kpi("sessions", fmtNum(t.sessions), fmtNum(t.responses)+" assistant - turns", "")` (`dashboard-server.mjs:1933`). + turns", "")` (`dashboard-server.mjs:1977`). **Worked example.** A session with 3 user turns and 2 assistant turns contributes `sessions += 1, responses += 2` — prompts (user turns) are tracked @@ -210,23 +210,23 @@ tokens = input + output + cacheRead + cacheWrite (summed across all rows in wi ``` **Source:** `t.tokens` from `totals`, accumulated per row at -`usage-index.mjs:783` (`rowTokens = row.input + row.output + row.cacheRead + +`usage-index.mjs:824` (`rowTokens = row.input + row.output + row.cacheRead + row.cacheWrite`) and rolled into `totals.tokens` via `addTo` -(`usage-index.mjs:738`). Rendered with `fmtTok()` -(`dashboard-server.mjs:1843-1849`): `≥1e9` → `"X.XB"`, `≥1e6` → `"X.XM"`, +(`usage-index.mjs:779`). Rendered with `fmtTok()` +(`dashboard-server.mjs:1884-1890`): `≥1e9` → `"X.XB"`, `≥1e6` → `"X.XM"`, `≥1e3` → `"X.XK"`, else the rounded integer. The **token composition bar** immediately below the hero row (cache read / cache write / output / input, as four coloured segments) is the same four -numbers as percentages of `t.tokens` (`dashboard-server.mjs:1969-1976`, -`pct(a,b) = b ? a/b*100 : 0`, `dashboard-server.mjs:1857`). +numbers as percentages of `t.tokens` (`dashboard-server.mjs:2013-2020`, +`pct(a,b) = b ? a/b*100 : 0`, `dashboard-server.mjs:1898`). **What "input" excludes.** For both providers, the `input` counter recorded per row is **gross input minus cached input** — Claude's parser reads `cache_read_input_tokens` and `cache_creation_input_tokens` as separate fields -the provider already reports separately (`usage-index.mjs:475-478`); Codex's +the provider already reports separately (`usage-index.mjs:485-488`); Codex's parser subtracts `cached_input_tokens` from `input_tokens` explicitly -(`usage-index.mjs:573-577`, `input: Math.max(0, gross - cacheRead)`) because +(`usage-index.mjs:610-614`, `input: Math.max(0, gross - cacheRead)`) because Codex's own `input_tokens` field **includes** cached tokens and would double-count them against the separately-reported `cacheRead` figure if left as-is. This is asserted by test: @@ -262,8 +262,8 @@ cacheRead + cacheWrite), **not** as a share of input alone. On the reference figures (`cacheRead = 1464.3B`, `tokens = 1495.2B`): `1464.3 / 1495.2 × 100 = 97.93%`, which rounds to the displayed `97.9%` — confirming the denominator. -**Source:** `dashboard-server.mjs:1931` (`cacheShare = pct(t.cacheRead, -t.tokens)`), rendered `dashboard-server.mjs:1939`. +**Source:** `dashboard-server.mjs:1975` (`cacheShare = pct(t.cacheRead, +t.tokens)`), rendered `dashboard-server.mjs:1983`. **Why this number matters more than it looks like it should.** On the reference corpus, 96.3% of tokens were cache reads — pricing them as fresh @@ -280,7 +280,7 @@ Codex CLI do by default. **Displayed as:** `ENGAGED TIME` hero tile — `403h`, subtitle `3286h summed` plus a `sessions overlap` note. Hovering the tile reveals a tooltip with all -three tiers (the "ladder," `dashboard-server.mjs:1921-1927`). +three tiers (the "ladder," `dashboard-server.mjs:1965-1971`). This is the single most heavily-caveated metric on the tab, because a naive version of it is **wrong by roughly 3×** on the reference corpus — worth @@ -311,27 +311,27 @@ session data, and each needs its own fix: human, or genuinely idle) donates its *entire* idle stretch to the span, even though no work happened during it. Fix: split each session into active sub-intervals wherever the gap between two consecutive timestamps - exceeds `IDLE_GAP_MS` (15 minutes, `usage-index.mjs:56`), then union + exceeds `IDLE_GAP_MS` (15 minutes, `usage-index.mjs:62`), then union *those* sub-intervals — this is `engagedSeconds`. **Source:** -- `mergeIntervals()` (`usage-index.mjs:83-108`) — the pure union primitive, +- `mergeIntervals()` (`usage-index.mjs:89-114`) — the pure union primitive, sorts intervals and merges any two that overlap **or exactly touch** - (`s <= curEnd`, `usage-index.mjs:99`), returning total covered seconds + (`s <= curEnd`, `usage-index.mjs:105`), returning total covered seconds rounded to the nearest second. -- `activeIntervals()` (`usage-index.mjs:306-318`) — splits one session's +- `activeIntervals()` (`usage-index.mjs:316-328`) — splits one session's sorted timestamp list into sub-intervals wherever a gap exceeds `IDLE_GAP_MS`; "a run of one timestamp yields a zero-length interval and so - contributes nothing" (comment, `usage-index.mjs:304`). + contributes nothing" (comment, `usage-index.mjs:314`). - Aggregation: `totals.engagedSeconds = mergeIntervals(sessions.flatMap(s => - s._active))` (`usage-index.mjs:882`); `totals.spanUnionSeconds = - mergeIntervals(sessions.map(s => s._span))` (`usage-index.mjs:881`); + s._active))` (`usage-index.mjs:927`); `totals.spanUnionSeconds = + mergeIntervals(sessions.map(s => s._span))` (`usage-index.mjs:926`); `totals.spanMinutes` is a running sum of `s._span[1] - s._span[0]` across - the loop (`usage-index.mjs:843`, finalized `usage-index.mjs:880`). -- Render: `fmtHours()` (`dashboard-server.mjs:1850`, `≥10h` rounds to the + the loop (`usage-index.mjs:888`, finalized `usage-index.mjs:925`). +- Render: `fmtHours()` (`dashboard-server.mjs:1891`, `≥10h` rounds to the nearest hour, else one decimal place) and `fmtMins()` - (`dashboard-server.mjs:1851`, `≥60min` rounds to hours, else whole + (`dashboard-server.mjs:1892`, `≥60min` rounds to hours, else whole minutes). **Worked example**, the reference-corpus measurement (14-day window, 582–584 @@ -377,13 +377,13 @@ byDay[day].cost = Σ costOf(row) for every usage row whose day == that key **Source:** the day key is the row's own `row.day`, computed once at parse time as **local calendar day**, not UTC -(`usage-index.mjs:474`/`usage-index.mjs:576` call `localDay(at)`) — so a +(`usage-index.mjs:484`/`usage-index.mjs:613` call `localDay(at)`) — so a session that runs from 23:58 local to 00:05 local is billed to the day its *first* row landed on (test: `tests/kit/usage-index.test.mjs:608`, "a session that opens before midnight is counted on its first billed day"). Accumulation: -`byDay[row.day].cost += rowCost` (`usage-index.mjs:786`). Bar height: -`h = maxDay ? max(2, cost/maxDay*100) : 2` (`dashboard-server.mjs:1950`) — +`byDay[row.day].cost += rowCost` (`usage-index.mjs:827`). Bar height: +`h = maxDay ? max(2, cost/maxDay*100) : 2` (`dashboard-server.mjs:1994`) — every non-empty day gets a visually nonzero bar (floor of 2%), so a very cheap day is never rendered as invisible. @@ -400,14 +400,14 @@ rather than a continuous 30-day series. **Displayed as:** two cards, `claude` and `codex`, each showing cost, session count, and total tokens; an idle host (`sessions == 0 && cost == 0`) renders "no sessions in window" instead of zeroed figures -(`dashboard-server.mjs:1962-1966`). +(`dashboard-server.mjs:2006-2010`). **Formula:** identical aggregation to every other bucket -(`byProvider[s.provider]`, populated via `addTo()`, `usage-index.mjs:734-743`, -called once per session at `usage-index.mjs:845`), keyed by the literal string +(`byProvider[s.provider]`, populated via `addTo()`, `usage-index.mjs:775-784`, +called once per session at `usage-index.mjs:890`), keyed by the literal string `"claude"` or `"codex"` assigned at parse time (`blankSession(id, 'claude')` / `blankSession(id, 'codex')`, -`usage-index.mjs:285-291`, `parseClaude`/`parseCodex` entry points). +`usage-index.mjs:291-301`, `parseClaude`/`parseCodex` entry points). **Why this pairing is the one under the most scrutiny.** Both providers' tokens are summed into the *same* `tokens`/`cost` fields using the *same* @@ -442,11 +442,11 @@ punchcard[dow + "-" + hour] += 1 per assistant/agent_message response, at its ``` **Source:** incremented once per Claude assistant turn -(`usage-index.mjs:449-450`, keyed by `punchKey(at)`) and once per Codex -`agent_message` (`usage-index.mjs:560-561`), merged into the window-level -`punchcard` object per session (`usage-index.mjs:860`). Cell intensity is +(`usage-index.mjs:459-460`, keyed by `punchKey(at)`) and once per Codex +`agent_message` (`usage-index.mjs:597-598`), merged into the window-level +`punchcard` object per session (`usage-index.mjs:905`). Cell intensity is linear against the single busiest cell in the window: -`v = pcMax ? n/pcMax : 0` (`dashboard-server.mjs:1986`) — this is a +`v = pcMax ? n/pcMax : 0` (`dashboard-server.mjs:2030`) — this is a **relative**, not absolute, scale, so the heatmap's brightest cell is always "the busiest hour-of-week in *this* window," not a fixed response-count threshold, and comparing brightness across two different date-range views is @@ -477,9 +477,9 @@ byModel[model].sessions = count of DISTINCT sessions whose s.models includes th ``` **Source:** cost/tokens/responses accumulate inside the usage-row loop -(`usage-index.mjs:773-790`); the `sessions` count is deliberately computed +(`usage-index.mjs:814-831`); the `sessions` count is deliberately computed **separately**, once per session over its `s.models` array -(`usage-index.mjs:853-858`) rather than inside the cost loop, precisely +(`usage-index.mjs:898-903`) rather than inside the cost loop, precisely **so that a model can appear in `byModel` — with a nonzero session count — even in a session that contributed zero cost/tokens/responses for that model.** This is not an edge case invented for this document: it is the @@ -489,16 +489,16 @@ excluded subagent-replay session still shows up as "used," at zero cost, rather than vanishing. `byModel[...].responses` is populated from `row.responses` -(`usage-index.mjs:788`), which in turn comes from the `responses` field +(`usage-index.mjs:829`), which in turn comes from the `responses` field passed into `addUsage()` at the call site — `1` per Claude assistant turn -(`usage-index.mjs:474-480`), or `rec.responses` (the session's whole response +(`usage-index.mjs:484-490`), or `rec.responses` (the session's whole response count) once per Codex session, passed at the single point Codex calls -`addUsage` (`usage-index.mjs:576-583`). +`addUsage` (`usage-index.mjs:613-624`). **Render:** `bar(name, fmtUsd(cost), fmtTok(tokens)+" · "+fmtNum(responses)+" -resp", pct(cost, topModelCost), false)` (`dashboard-server.mjs:1999-2002`), +resp", pct(cost, topModelCost), false)` (`dashboard-server.mjs:2043-2046`), list itself sorted cost-descending by the shared `entries()` helper -(`dashboard-server.mjs:1858-1863`). +(`dashboard-server.mjs:1899-1904`). **Exceptions — a turn that never resolved to a model is excluded here, not shown as a $0 row.** A dropped connection, rate limit, or authentication @@ -509,17 +509,17 @@ split `server_error` 27, `authentication_failed` 3, `rate_limit` 3 — three distinct underlying causes, one placeholder shape). The parser branches on `isApiErrorMessage === true` -(`usage-index.mjs:459-468`): the turn still increments `rec.responses` -and the punchcard (`usage-index.mjs:447-450`) — it *is* real engaged +(`usage-index.mjs:469-478`): the turn still increments `rec.responses` +and the punchcard (`usage-index.mjs:457-460`) — it *is* real engaged time, someone was genuinely waiting on it — but it is never pushed into `rec.models` and `addUsage()` is never called for it, so it can no longer create a `byModel` row of any kind. It increments a separate -`rec.exceptions` counter instead (`usage-index.mjs:460`), rolled up into -`totals.exceptions` (`usage-index.mjs:828-839`) and surfaced per-session -(`usage-index.mjs:803`, alongside the existing `sidechain`/`threadSource` +`rec.exceptions` counter instead (`usage-index.mjs:470`), rolled up into +`totals.exceptions` (`usage-index.mjs:873-884`) and surfaced per-session +(`usage-index.mjs:844`, alongside the existing `sidechain`/`threadSource` flags — inspectable in the Sessions tab, never hidden). When `totals.exceptions > 0`, the panel header shows a small `"· N -dropped/errored turns excluded"` note (`dashboard-server.mjs:2003-2008`); +dropped/errored turns excluded"` note (`dashboard-server.mjs:2047-2052`); when it's zero, the note renders empty rather than always claiming a count of zero. @@ -545,7 +545,7 @@ N"` when more exist; each row shows `cost`, `N sess · minutes`. it masquerade as a sibling project. (The mislabelling this rule corrected is recorded in [Appendix A](#appendix-a--fix-history).) -**Source:** ranking and truncation, `dashboard-server.mjs:2010-2017` +**Source:** ranking and truncation, `dashboard-server.mjs:2054-2061` (`shown = projects.slice(0,8)`); accumulation via the same `addTo()`/ `entries()` machinery as §10, keyed by `s.project` instead of `s.models`. @@ -561,7 +561,7 @@ rather than silently absent from the total. **Displayed as:** a ranked bar list of categories, bar width relative to the top category's cost; each row shows `cost`, `N sess · $/sess`; a small confidence dot on classified rows (opacity `0.5 + confidence×0.5`, -`dashboard-server.mjs:2024-2025`); `Unclassified` is always shown, never +`dashboard-server.mjs:2068-2069`); `Unclassified` is always shown, never hidden, and carries no confidence dot. This is the only Scorecard metric that is **not** pure arithmetic over @@ -632,9 +632,9 @@ coverage was `ai-title` 93%, tool mix 100%, cannot classify most sessions and layer 2 (title + tool-mix rules) carries the bulk of the load. -**Render:** `dashboard-server.mjs:2020-2031`, sorted cost-descending via the +**Render:** `dashboard-server.mjs:2064-2075`, sorted cost-descending via the shared `entries()` helper, with `$/sess = cost / max(sessions, 1)` guarding -the zero-session edge case (`dashboard-server.mjs:2022`). +the zero-session edge case (`dashboard-server.mjs:2066`). **What this does not model:** the classifier reads only the session's *title* (Claude's own `ai-title`, written by the model itself at session @@ -699,7 +699,7 @@ Anthropic which publishes a fetchable pricing document. OpenAI's rates in this table are therefore maintained by hand against OpenAI's own developer documentation and are the most drift-prone entries in the file — this is explicitly why `PRICES_AS_OF` is surfaced in the UI (`u-asof`, -`dashboard-server.mjs:1940`) rather than assumed current. +`dashboard-server.mjs:1984`) rather than assumed current. | Model (kit key) | Input | Output | Cache read (0.1×, derived) | |---|---|---|---| @@ -770,6 +770,67 @@ Recorded verbatim from `pricing.mjs:82-99` (`UNMODELLED_PRICING_FACTORS`, --- +## 13b. Limits view — vendor-reported plan utilization (ADR-0010) + +Every other figure in this document is computed locally from transcripts. The +Limits sub-view is different by design: its percentages are **vendor-reported** +— the plan's own denominator, which ADR-0009 §3 correctly said local parsing +could never honestly invent. ADR-0010 defines the only two admissible channels, +both credential-free for ak: + +- **Claude** — Claude Code pushes `rate_limits` (session/weekly/per-model + `used_percentage` + `resets_at`) into every statusLine invocation on Pro/Max. + The kit's managed statusline footer tees that JSON to + `~/.config/agentic-kit/claude-rate-limits.json` (throttled, atomic, 0600) — + the `quota tee` block in `statusline-footer.cjs:9`. The dashboard reads the + tee via `normalizeClaudeLimits` (`quota.mjs:62`), which maps `five_hour` / + `seven_day` / `seven_day_` keys to duration-labelled windows. +- **Codex** — one `initialize` → `account/rateLimits/read` JSON-RPC exchange + with a spawned `codex app-server`, implemented by `codexAppServerRateLimits` + (`quota.mjs:165`) and TTL-cached (`CODEX_TTL_MS`, `:36`) by + `collectCodexLimits` (`:220`). Lanes come from `rateLimitsByLimitId` in + `normalizeCodexLimits` (`:106`), including per-model pools and + rate-limit reset credits. + +**The primary/secondary trap.** Codex's `primary` window is *not* reliably the +5-hour window — a live `prolite` account reported `primary` with +`windowDurationMins: 10080` (the weekly). Windows are therefore keyed and +labelled by duration (`windowLabel`, `quota.mjs:44`), never by slot name. The +same rule applies to the historical snapshots parsed out of rollouts: the +normalizer at `usage-index.mjs:560` keeps a flat `windows` list keyed by +`window_minutes`. + +**Freshness is part of the number.** Both sides carry `fetchedAt`; the view +renders "as of Nm ago" and a `stale` badge (Claude's tee is push-only, so it +ages the moment sessions stop). Served by the `/api/limits` route +(`dashboard-server.mjs:363`) and rendered by `renderLimits` (`:2120`), with +`limRow` (`:2111`) coloring each bar by proximity to its cap. + +**Limit-aware findings.** `detectLimitInsights` (`usage-insights.mjs:625`) +applies the same evidence rules as every other detector — vendor percentages +are the user's own data; no dollar impact is ever claimed from a percentage; +"now" is the payload's `generatedAt`, never a clock. Detectors: pacing +against the window's own elapsed share, cross-host arbitrage between the two +plan pools, and expiring Codex reset credits (reported, never auto-consumed). + +**What still has no supported channel:** Claude extra-usage credit balance and +subscription tier. They stay absent rather than approximated. + +## 13c. Codex thread ledger — authoritative subagent attribution + +Codex ≥0.140 maintains its own SQLite thread ledger (`~/.codex/state_N.sqlite` +— the `N` is a migration generation, so `codexStateDb` (`codex-state.mjs:30`) +globs and takes the newest). `readCodexState` (`:49`) reads per-thread +`thread_source` (`user` vs `subagent`) plus `thread_spawn_edges`, and +`applyCodexLedger` (`usage-index.mjs:1090`) overlays that onto parsed +sessions: a ledger-identified subagent has its token usage stripped — its +rollout replays the parent's entire token history (ccusage/ccusage#950 +measured up to 91× inflation) — while the session record stays visible. The +rollout's own `session_meta.thread_source` sniff remains as the fallback when +the ledger is absent or migrated beyond recognition. Codex sessions also carry +`reasoningOutput` (`usage-index.mjs:623`) — reasoning tokens are a **subset** +of output tokens and are annotation only, never added to any sum. + ## 14. Known limitations, restated as a single checklist For a reviewer who wants the "does this number lie to me" answer without @@ -812,14 +873,14 @@ commit `540be18` on this branch. `parseCodex`'s single `addUsage()` call never included a `responses` field — Claude's parser passes `responses: 1` per assistant turn -(`usage-index.mjs:479`, as it existed before this fix), but Codex's call +(`usage-index.mjs:489`, as it existed before this fix), but Codex's call passed no such field at all. Because `byModel[model].responses` is summed -directly from each usage row's `responses` field (`usage-index.mjs:788`, +directly from each usage row's `responses` field (`usage-index.mjs:829`, `m.responses += row.responses`), **every** Codex model in §10's "Models in Play" list displayed `0 resp` regardless of real token/cost volume or actual `agent_message` count. **Fix:** `parseCodex` now passes `responses: rec.responses` (the session's own tallied response count, -`usage-index.mjs:558`) on its `addUsage()` call. +`usage-index.mjs:595`) on its `addUsage()` call. #### Bug B — subagent thread-replay could double-bill tokens @@ -839,13 +900,13 @@ at face value (correctly avoiding the separate naive-summing bug **[C5]** documents, since it already used last-event-only logic — see §4's worked example) but performed **no de-duplication** against a parent session a subagent file might be replaying. **Fix:** the parser now reads -`session_meta.thread_source` (`usage-index.mjs:532`, confirmed as a real +`session_meta.thread_source` (`usage-index.mjs:542`, confirmed as a real Codex rollout field by **[C7]**) and skips the `addUsage()` call entirely -when its value is `'subagent'` (`usage-index.mjs:572`, guard condition +when its value is `'subagent'` (`usage-index.mjs:609`, guard condition `rec.threadSource !== 'subagent'`). The session record itself is **not** dropped — it remains visible in the Sessions tab with `threadSource` surfaced (mirroring the existing `sidechain` flag Claude sessions already -carry, `usage-index.mjs:288`), so a maintainer auditing the raw data can +carry, `usage-index.mjs:294`), so a maintainer auditing the raw data can still see it; it simply contributes zero tokens/cost, exactly as intended by the "models still shows up in §10's list, with zero cost" mechanism §10 describes. @@ -907,7 +968,7 @@ promise. `totals.exceptions: null` and still showed the `` row in `byModel` on the first run after the change, purely because the cache predated it; every unit test still passed, since tests only exercise a - fresh parse. `SCHEMA_VERSION` went to `4` (`usage-index.mjs:35-50`; since + fresh parse. `SCHEMA_VERSION` went to `4` (`usage-index.mjs:36-51`; since superseded by `5`, above) specifically to force the one-time re-parse. Re-querying the same live server after the bump returned `totals.exceptions: 20` with `` absent from `byModel` — diff --git a/docs/adr/0009-usage-scorecard-local-transcript-analytics.md b/docs/adr/0009-usage-scorecard-local-transcript-analytics.md index dbafbb05..34282803 100644 --- a/docs/adr/0009-usage-scorecard-local-transcript-analytics.md +++ b/docs/adr/0009-usage-scorecard-local-transcript-analytics.md @@ -70,6 +70,13 @@ panel therefore labels every figure **"API-equivalent — not your plan billing" element, not a footnote. We do not model plan utilisation: Anthropic does not publish the limits that would make such a percentage honest, and an invented denominator is worse than no number. +> **Amended by [ADR-0010](0010-provider-mediated-quota-reads.md) (2026-07-27):** the exclusion above +> was about denominators the kit would have to *invent*. Both vendors now hand over their own +> percentages through supported channels (Claude Code's statusLine `rate_limits` push; Codex's +> `app-server` RPC), and the Limits sub-view renders those vendor-reported figures — with +> provenance and freshness — under ADR-0010's provider-mediated rules. Locally *computed* +> plan percentages remain excluded, exactly as stated here. + OpenAI publishes no pricing in `~/.codex/models_cache.json` (verified), so Codex rates are a maintained table and are **date-stamped** in the UI so staleness is visible rather than silent. diff --git a/docs/adr/0010-provider-mediated-quota-reads.md b/docs/adr/0010-provider-mediated-quota-reads.md new file mode 100644 index 00000000..445d8ec0 --- /dev/null +++ b/docs/adr/0010-provider-mediated-quota-reads.md @@ -0,0 +1,98 @@ +# ADR-0010 — Provider-mediated quota reads (the Limits view) + +Date: 2026-07-27 · Status: **Accepted** · Amends: ADR-0009 §3 + +## Context + +ADR-0009 §3 deliberately excluded plan/limit modeling from the usage scorecard: +Anthropic and OpenAI publish no limit values, so any locally computed +"percentage of plan used" would rest on an invented denominator — and "an +invented denominator is worse than no number". That reasoning was about +denominators the kit would have to *fabricate*. Research (2026-07-27, cited in +§Research below) established that both vendors now *hand over* their own +percentages through supported channels: + +- **Claude Code** pushes a `rate_limits` object — `five_hour` / `seven_day` + (and per-model weekly buckets) with `used_percentage` and `resets_at` — into + every statusLine invocation for Pro/Max subscribers. This is documented + behavior, not an internal. +- **Codex** answers `account/rateLimits/read` over its `codex app-server` + JSON-RPC surface with per-lane (`rateLimitsByLimitId`) used-percent, window + duration, reset time, plan type, and rate-limit reset credits. + +A vendor-reported percentage IS an honest denominator. What remains dishonest — +and stays excluded — is fetching those numbers through channels the vendors +prohibit or do not support. + +## Decision + +Quota data enters the kit **only through the vendors' own software**, with ak +never reading, storing, or refreshing a vendor credential: + +1. **Claude — statusline tee.** The kit's managed statusline footer tees the + pushed `rate_limits` (plus `context_window` and `cost`) to + `~/.config/agentic-kit/claude-rate-limits.json` (0600, atomic, throttled to + one write per minute). Push, not pull: with no recent Claude session the + file goes stale, and the UI labels it stale rather than hiding it. +2. **Codex — app-server subprocess.** `src/lib/quota.mjs` spawns + `codex -s read-only -a untrusted app-server` (codex authenticates itself), + performs one `initialize` → `account/rateLimits/read` exchange with a hard + timeout and kill, and caches the normalized answer for `CODEX_TTL_MS`. + This is the same shell-out trust model the dashboard already uses for + `ak status --json`. +3. **The dashboard server itself still opens no sockets to the internet.** + ADR-0005/0007's egress split is unchanged: the Codex subprocess is vendor + code using vendor auth, and the Claude path is a local file read. + +Normalization rules (all in `quota.mjs`, all pinned by tests): + +- **Windows are keyed by duration, never by slot.** Codex's `primary` / + `secondary` fields do not reliably mean "5-hour" / "weekly" — a live + `prolite` account answered with `primary.windowDurationMins = 10080`. +- `null` timestamps stay `null`; they are never coerced to epoch 0. +- Every payload carries `fetchedAt`, and the UI renders freshness ("as of Nm + ago") next to every number. + +### Explicit non-paths + +- **No `api.anthropic.com/api/oauth/usage`.** Undocumented; hostile + rate-limiting to unrecognized clients; and Anthropic's consumer ToS bars + subscription OAuth tokens "in any other product, tool, or service", enforced + server-side since January 2026. +- **No Keychain or credential-file reads** (the macOS item stopped carrying + the OAuth token in Claude Code 2.1.x; refresh tokens rotate destructively). +- **No `chatgpt.com/backend-api/wham/*`** — private endpoints that would + require ak to hold and refresh the user's bearer token. +- **No auto-consumption of Codex reset credits.** The panel reports them; + redeeming a finite grant is the user's action in codex's own `/usage`. + +## Consequences + +- The Limits sub-view can show authoritative, cross-device session/weekly + utilization — data local transcript parsing can never produce (ADR-0009 §3's + table of unknowables shrinks to: extra-usage credit balance and subscription + tier, which have no supported channel). +- `usage-insights.mjs` gains limit-aware detectors (`detectLimitInsights`) + under the same evidence rules; vendor percentages count as the user's own + data, and no dollar impact is ever claimed from a percentage. +- Codex attribution stops being heuristic: `codex-state.mjs` reads Codex's own + SQLite thread ledger (`state_*.sqlite`, glob — the numeric suffix is a + migration generation) for `thread_source` and spawn edges, demoting the + rollout-sniffing subagent guard to a fallback. +- Two new 0600 cache files exist under `~/.config/agentic-kit/`; both are + derived, deletable, and self-healing. +- The `codex app-server` surface is flagged experimental upstream; the client + pins nothing beyond two method names and degrades to "no data" with the + stale cache visible if the protocol shifts. + +## Research + +Grounded findings behind this ADR (full citations in the planning record): +statusLine `rate_limits` — code.claude.com/docs/en/statusline.md; Admin +Usage/Cost + Claude Code Analytics APIs are org-only — +platform.claude.com/docs/en/manage-claude/usage-cost-api; consumer-OAuth +prohibition — code.claude.com/docs/en/legal-and-compliance; Codex app-server +schema — `codex app-server generate-json-schema` (verified live on codex +0.145.0, 2026-07-27); Codex reset-credit detail — openai/codex#29618 (shipped +via PR #30488); subagent rollout replay inflation — ccusage/ccusage#950; +wham endpoint traffic complaints — openai/codex#10869, #27952. diff --git a/src/lib/codex-state.mjs b/src/lib/codex-state.mjs new file mode 100644 index 00000000..0d451051 --- /dev/null +++ b/src/lib/codex-state.mjs @@ -0,0 +1,79 @@ +// Codex's own thread ledger (~/.codex/state_N.sqlite) — the authoritative +// source for per-thread attribution that the rollout JSONL can only guess at. +// +// Codex ≥0.140 maintains a `threads` table with a pre-aggregated `tokens_used` +// column plus `thread_source` ('user' | 'subagent') and `source` +// ('cli' | 'exec' | 'subagent'), and a `thread_spawn_edges` table recording +// parent→child delegation. That replaces the kit's heuristic subagent-replay +// guard (a `session_meta.thread_source` sniff, PR #60) with Codex's own +// bookkeeping: a subagent rollout replays its parent's ENTIRE token history +// (ccusage/ccusage#950 measured up to 91x inflation), so knowing which threads +// are subagents is what keeps the totals honest. +// +// Deliberate limits: +// - READ-ONLY, always. `withDb` opens readonly and swallows every error — +// a locked or half-migrated db yields null, never a crash and never a lock +// held against the codex CLI itself. +// - The filename suffix (`state_5`) is a MIGRATION GENERATION, not a stable +// name. We glob `state_*.sqlite` and take the highest generation rather +// than hardcoding today's. +// - Column names are probed before use: a future migration that drops or +// renames a column degrades this module to null (callers fall back to the +// JSONL heuristic), not to a throw. +import fs from 'node:fs'; +import path from 'node:path'; +import { codexDir } from './paths.mjs'; +import { withDb } from './sqlite.mjs'; + +/** Newest-generation state db file under ~/.codex, or null. Exported for test + * via the `dir` override. */ +export function codexStateDb(dir = codexDir()) { + let names; + try { names = fs.readdirSync(dir); } catch { return null; } + const gens = names + .map((n) => /^state_(\d+)\.sqlite$/.exec(n)) + .filter(Boolean) + .map((m) => ({ name: m[0], gen: Number(m[1]) })) + .sort((a, b) => b.gen - a.gen); + return gens.length ? path.join(dir, gens[0].name) : null; +} + +/** + * Read Codex's thread ledger. Returns + * { threads: Map, + * parents: Map } + * or null when the db is absent, unreadable, or shaped unexpectedly. + * + * @param {{ dir?: string, file?: string }} [opts] test seams + */ +export function readCodexState(opts = {}) { + const file = opts.file ?? codexStateDb(opts.dir); + if (!file) return null; + return withDb(file, (db) => { + const cols = new Set(db.prepare('PRAGMA table_info(threads)').all().map((c) => c.name)); + // id + thread_source are what the attribution fix rests on; without them + // this ledger cannot answer the question and the caller must fall back. + if (!cols.has('id') || !cols.has('thread_source')) return null; + const pick = ['id', 'thread_source'] + .concat(['tokens_used', 'source', 'model', 'git_branch'].filter((c) => cols.has(c))); + const threads = new Map(); + for (const row of db.prepare(`SELECT ${pick.join(', ')} FROM threads`).all()) { + threads.set(String(row.id), { + tokensUsed: Number(row.tokens_used) || 0, + threadSource: typeof row.thread_source === 'string' ? row.thread_source : null, + source: typeof row.source === 'string' ? row.source : null, + model: typeof row.model === 'string' ? row.model : null, + gitBranch: typeof row.git_branch === 'string' ? row.git_branch : null, + }); + } + const parents = new Map(); + try { + for (const e of db.prepare('SELECT parent_thread_id, child_thread_id FROM thread_spawn_edges').all()) { + if (e.child_thread_id != null && e.parent_thread_id != null) { + parents.set(String(e.child_thread_id), String(e.parent_thread_id)); + } + } + } catch { /* the edges table is additive detail — attribution works without it */ } + return { threads, parents }; + }, null); +} diff --git a/src/lib/dashboard-server.mjs b/src/lib/dashboard-server.mjs index bc7fbe90..5dcbc3f8 100644 --- a/src/lib/dashboard-server.mjs +++ b/src/lib/dashboard-server.mjs @@ -262,12 +262,16 @@ function sendJson(res, status, payload) { /** * Start the dashboard HTTP server, bound to loopback only. - * @param {{ port?: number, cwd?: string, fetchStatus?: () => Promise, usage?: any }} [opts] + * @param {{ port?: number, cwd?: string, fetchStatus?: () => Promise, usage?: any, + * limits?: () => Promise }} [opts] * @returns {Promise<{ url: string, port: number, close: () => Promise }>} */ -export function startDashboard({ port = 7431, cwd = process.cwd(), fetchStatus, usage } = {}) { +export function startDashboard({ port = 7431, cwd = process.cwd(), fetchStatus, usage, limits } = {}) { const provide = fetchStatus || shellOutStatus(cwd); const usageApi = usage || lazyUsage(); + // Injectable like `usage`: tests must never spawn a real codex or read the + // real ~/.config through this route. Lazy for the same reason lazyUsage is. + const provideLimits = limits || (async () => (await import('./quota.mjs')).readLimits()); const html = renderPage({ name: '@pacphi/agentic-kit', version: kitVersion() }); const server = http.createServer(async (req, res) => { @@ -352,6 +356,22 @@ export function startDashboard({ port = 7431, cwd = process.cwd(), fetchStatus, return; } + // Plan-limit utilization (ADR-0010). Claude side is a pure file read (the + // statusline tee); Codex side may spawn ONE vendor subprocess (`codex + // app-server`), TTL-cached — the same shell-out trust model as + // `ak status --json` above. No vendor credential is ever read here. + if (url === '/api/limits') { + try { + const agg = await usageApi.readIndex({ days: clampDays(query.get('days')) }).catch(() => null); + const payload = await provideLimits(); + const { detectLimitInsights } = await import('./usage-insights.mjs'); + sendJson(res, 200, { ...payload, insights: detectLimitInsights(payload, agg) ?? [] }); + } catch (e) { + sendJson(res, 500, { error: String(e && e.message || e) }); + } + return; + } + if (url === '/api/sessions') { try { const agg = await usageApi.readIndex({ days: clampDays(query.get('days')) }); @@ -517,9 +537,16 @@ function renderPage({ name, version }) {
@@ -552,6 +579,7 @@ function renderPage({ name, version }) {
+ @@ -587,7 +615,7 @@ function renderPage({ name, version }) {
-

models in play

by api-equivalent cost
+

models in play

observed in transcripts · by api-equivalent cost
@@ -606,6 +634,26 @@ function renderPage({ name, version }) {
+ +