diff --git a/apps/api/src/db/migrations/0176_workspace_resource_working_set.sql b/apps/api/src/db/migrations/0176_workspace_resource_working_set.sql new file mode 100644 index 000000000..f1fa05247 --- /dev/null +++ b/apps/api/src/db/migrations/0176_workspace_resource_working_set.sql @@ -0,0 +1,8 @@ +ALTER TABLE workspace_resource_summaries + ADD COLUMN memory_working_set_mean_bytes INTEGER; + +ALTER TABLE workspace_resource_summaries + ADD COLUMN memory_working_set_peak_bytes INTEGER; + +ALTER TABLE workspace_resource_summaries + ADD COLUMN memory_working_set_sample_count INTEGER NOT NULL DEFAULT 0; diff --git a/apps/api/src/db/schema.ts b/apps/api/src/db/schema.ts index f568819aa..fdae14e07 100644 --- a/apps/api/src/db/schema.ts +++ b/apps/api/src/db/schema.ts @@ -1582,6 +1582,9 @@ export const workspaceResourceSummaries = sqliteTable( memoryMeanBytes: integer('memory_mean_bytes'), memoryPeakBytes: integer('memory_peak_bytes'), memoryKernelPeakBytes: integer('memory_kernel_peak_bytes'), + memoryWorkingSetMeanBytes: integer('memory_working_set_mean_bytes'), + memoryWorkingSetPeakBytes: integer('memory_working_set_peak_bytes'), + memoryWorkingSetSampleCount: integer('memory_working_set_sample_count').notNull().default(0), ioReadBytes: integer('io_read_bytes'), ioWriteBytes: integer('io_write_bytes'), oomCount: integer('oom_count').notNull().default(0), diff --git a/apps/api/src/durable-objects/sam-session/tools/get-resource-history.ts b/apps/api/src/durable-objects/sam-session/tools/get-resource-history.ts index 2fa56ef87..88e4181d9 100644 --- a/apps/api/src/durable-objects/sam-session/tools/get-resource-history.ts +++ b/apps/api/src/durable-objects/sam-session/tools/get-resource-history.ts @@ -11,6 +11,7 @@ export const getResourceHistoryDef: AnthropicToolDef = { description: 'Inspect bounded workspace resource history for a project session, task, or workspace. ' + 'Returns a cheap summary with server-resolved agentProfileId, skillId, and agentType plus a chunk index by default. Pass chunkId to load bounded downsampled samples and tool-span correlation for one chunk. ' + + 'Working-set memory is the sizing figure; total memory includes reclaimable file cache. ' + 'Tool spans include the ACP kind and metadata-provided tool name when available. They are correlation windows, not causal per-process attribution, and stored payloads omit titles, prompts, commands, tool args/output, file paths, env, and secrets.', input_schema: { type: 'object', @@ -84,6 +85,8 @@ export async function getResourceHistory( ...history, notes: [ 'Samples are workspace-level cgroup observations, not per-process attribution.', + 'memoryWorkingSetMeanBytes and memoryWorkingSetPeakBytes estimate memory needed by excluding reclaimable inactive file cache; null means the VM agent did not report them.', + 'memoryMeanBytes, memoryPeakBytes, and memoryKernelPeakBytes include cache and remain available for historical comparison.', 'Tool spans are timestamp correlation windows and may include an ACP kind and metadata-provided tool name; titles and inputs are never returned.', 'Chunk detail is returned only when chunkId is supplied; summary reads stay bounded.', ], diff --git a/apps/api/src/routes/mcp/session-tools.ts b/apps/api/src/routes/mcp/session-tools.ts index 122745a82..e8c014fe9 100644 --- a/apps/api/src/routes/mcp/session-tools.ts +++ b/apps/api/src/routes/mcp/session-tools.ts @@ -215,6 +215,8 @@ export async function handleGetResourceHistory( ...history, notes: [ 'Samples are workspace-level cgroup observations, not per-process attribution.', + 'memoryWorkingSetMeanBytes and memoryWorkingSetPeakBytes estimate memory needed by excluding reclaimable inactive file cache; null means the VM agent did not report them.', + 'memoryMeanBytes, memoryPeakBytes, and memoryKernelPeakBytes include cache and remain available for historical comparison.', 'Tool spans are timestamp correlation windows and may include an ACP kind and metadata-provided tool name; titles and inputs are never returned.', 'Chunk detail is returned only when chunkId is supplied; summary reads stay bounded.', ], diff --git a/apps/api/src/routes/mcp/tool-definitions-project-awareness.ts b/apps/api/src/routes/mcp/tool-definitions-project-awareness.ts index 2e3ad0a32..afe953d80 100644 --- a/apps/api/src/routes/mcp/tool-definitions-project-awareness.ts +++ b/apps/api/src/routes/mcp/tool-definitions-project-awareness.ts @@ -171,7 +171,7 @@ export const PROJECT_AWARENESS_TOOLS = [ { name: 'get_resource_history', description: - 'Inspect bounded workspace resource history for the current project. By default, MCP callers read their current session/task/workspace summary, including server-resolved agentProfileId, skillId, and agentType, plus the chunk index. Pass sessionId, taskId, or workspaceId to inspect a related scope. Pass chunkId to lazily load downsampled raw samples and tool-span correlation for that chunk, including ACP kind and metadata-provided tool name when available. This reports correlation, not causal per-process attribution, and never includes titles, prompts, commands, tool args/output, file paths, env, or secrets.', + 'Inspect bounded workspace resource history for the current project. By default, MCP callers read their current session/task/workspace summary, including server-resolved agentProfileId, skillId, and agentType, plus the chunk index. Pass sessionId, taskId, or workspaceId to inspect a related scope. Pass chunkId to lazily load downsampled raw samples and tool-span correlation for that chunk, including ACP kind and metadata-provided tool name when available. Working-set memory is the sizing figure; total memory includes reclaimable file cache. This reports correlation, not causal per-process attribution, and never includes titles, prompts, commands, tool args/output, file paths, env, or secrets.', inputSchema: { type: 'object' as const, properties: { diff --git a/apps/api/src/services/workspace-resource-history.ts b/apps/api/src/services/workspace-resource-history.ts index c212a28a6..e48a6cb8d 100644 --- a/apps/api/src/services/workspace-resource-history.ts +++ b/apps/api/src/services/workspace-resource-history.ts @@ -69,6 +69,9 @@ export interface WorkspaceResourceSummaryPayload { memoryMeanBytes?: number | null; memoryPeakBytes?: number | null; memoryKernelPeakBytes?: number | null; + memoryWorkingSetMeanBytes?: number | null; + memoryWorkingSetPeakBytes?: number | null; + memoryWorkingSetSampleCount?: number | null; ioReadBytes?: number | null; ioWriteBytes?: number | null; oomCount?: number | null; @@ -88,6 +91,7 @@ export interface ResourceSamplePoint { cpuMillis?: number; memoryBytes?: number; memoryPeakBytes?: number; + memoryWorkingSetBytes?: number; ioReadBytes?: number; ioWriteBytes?: number; oom?: number; @@ -144,6 +148,8 @@ export interface PublicWorkspaceResourceSummary { memoryMeanBytes: number | null; memoryPeakBytes: number | null; memoryKernelPeakBytes: number | null; + memoryWorkingSetMeanBytes: number | null; + memoryWorkingSetPeakBytes: number | null; ioReadBytes: number | null; ioWriteBytes: number | null; oomCount: number; @@ -752,6 +758,25 @@ export async function storeWorkspaceResourceChunk( const skillId = workspace.skill_id; const agentType = workspace.agent_type; const runtime = normalizeNullable(body.runtime) ?? 'vm'; + const memoryWorkingSetMeanBytes = assertFiniteMetric( + body.summary.memoryWorkingSetMeanBytes, + 'summary.memoryWorkingSetMeanBytes' + ); + const memoryWorkingSetPeakBytes = assertFiniteMetric( + body.summary.memoryWorkingSetPeakBytes, + 'summary.memoryWorkingSetPeakBytes' + ); + const memoryWorkingSetSampleCount = + memoryWorkingSetMeanBytes == null + ? 0 + : (assertFiniteMetric( + body.summary.memoryWorkingSetSampleCount, + 'summary.memoryWorkingSetSampleCount' + ) ?? body.sampleCount); + assertFiniteInteger(memoryWorkingSetSampleCount, 'summary.memoryWorkingSetSampleCount'); + if (memoryWorkingSetSampleCount > body.sampleCount) { + throw errors.badRequest('summary.memoryWorkingSetSampleCount cannot exceed sampleCount'); + } const values = { id: summaryId, projectId, @@ -776,6 +801,9 @@ export async function storeWorkspaceResourceChunk( body.summary.memoryKernelPeakBytes, 'summary.memoryKernelPeakBytes' ), + memoryWorkingSetMeanBytes, + memoryWorkingSetPeakBytes, + memoryWorkingSetSampleCount, ioReadBytes: assertFiniteMetric(body.summary.ioReadBytes, 'summary.ioReadBytes'), ioWriteBytes: assertFiniteMetric(body.summary.ioWriteBytes, 'summary.ioWriteBytes'), oomCount: Math.trunc(assertFiniteMetric(body.summary.oomCount, 'summary.oomCount') ?? 0), @@ -829,6 +857,27 @@ export async function storeWorkspaceResourceChunk( END`, memoryPeakBytes: sql`MAX(COALESCE(${schema.workspaceResourceSummaries.memoryPeakBytes}, 0), ${values.memoryPeakBytes ?? 0})`, memoryKernelPeakBytes: sql`MAX(COALESCE(${schema.workspaceResourceSummaries.memoryKernelPeakBytes}, 0), ${values.memoryKernelPeakBytes ?? 0})`, + memoryWorkingSetMeanBytes: + values.memoryWorkingSetMeanBytes == null || values.memoryWorkingSetSampleCount === 0 + ? schema.workspaceResourceSummaries.memoryWorkingSetMeanBytes + : sql`CASE + WHEN ${schema.workspaceResourceSummaries.memoryWorkingSetMeanBytes} IS NULL + OR ${schema.workspaceResourceSummaries.memoryWorkingSetSampleCount} = 0 + THEN ${values.memoryWorkingSetMeanBytes} + ELSE CAST(( + (${schema.workspaceResourceSummaries.memoryWorkingSetMeanBytes} * ${schema.workspaceResourceSummaries.memoryWorkingSetSampleCount}) + + (${values.memoryWorkingSetMeanBytes} * ${values.memoryWorkingSetSampleCount}) + ) / (${schema.workspaceResourceSummaries.memoryWorkingSetSampleCount} + ${values.memoryWorkingSetSampleCount}) AS INTEGER) + END`, + memoryWorkingSetPeakBytes: + values.memoryWorkingSetPeakBytes == null + ? schema.workspaceResourceSummaries.memoryWorkingSetPeakBytes + : sql`CASE + WHEN ${schema.workspaceResourceSummaries.memoryWorkingSetPeakBytes} IS NULL + THEN ${values.memoryWorkingSetPeakBytes} + ELSE MAX(${schema.workspaceResourceSummaries.memoryWorkingSetPeakBytes}, ${values.memoryWorkingSetPeakBytes}) + END`, + memoryWorkingSetSampleCount: sql`${schema.workspaceResourceSummaries.memoryWorkingSetSampleCount} + ${values.memoryWorkingSetSampleCount}`, ioReadBytes: sql`COALESCE(${schema.workspaceResourceSummaries.ioReadBytes}, 0) + ${values.ioReadBytes ?? 0}`, ioWriteBytes: sql`COALESCE(${schema.workspaceResourceSummaries.ioWriteBytes}, 0) + ${values.ioWriteBytes ?? 0}`, oomCount: sql`${schema.workspaceResourceSummaries.oomCount} + ${values.oomCount}`, @@ -908,6 +957,8 @@ function publicSummary(row: schema.WorkspaceResourceSummaryRow): PublicWorkspace memoryMeanBytes: row.memoryMeanBytes, memoryPeakBytes: row.memoryPeakBytes, memoryKernelPeakBytes: row.memoryKernelPeakBytes, + memoryWorkingSetMeanBytes: row.memoryWorkingSetMeanBytes, + memoryWorkingSetPeakBytes: row.memoryWorkingSetPeakBytes, ioReadBytes: row.ioReadBytes, ioWriteBytes: row.ioWriteBytes, oomCount: row.oomCount, @@ -947,6 +998,7 @@ function sampleScore(sample: ResourceSamplePoint): number { Number(sample.cpuMillis ?? 0), Number(sample.memoryBytes ?? 0) / (1024 * 1024), Number(sample.memoryPeakBytes ?? 0) / (1024 * 1024), + Number(sample.memoryWorkingSetBytes ?? 0) / (1024 * 1024), sample.gap ? Number.MAX_SAFE_INTEGER : 0 ); } diff --git a/apps/api/tests/helpers/resource-history.ts b/apps/api/tests/helpers/resource-history.ts new file mode 100644 index 000000000..e9d3ced62 --- /dev/null +++ b/apps/api/tests/helpers/resource-history.ts @@ -0,0 +1,28 @@ +export function base64(bytes: Uint8Array): string { + let binary = ''; + for (const byte of bytes) binary += String.fromCharCode(byte); + return btoa(binary); +} + +export async function sha256Hex(bytes: Uint8Array): Promise { + const digest = await crypto.subtle.digest('SHA-256', bytes); + return [...new Uint8Array(digest)].map((byte) => byte.toString(16).padStart(2, '0')).join(''); +} + +export async function gzipText(value: string): Promise { + return gzipBytes(new TextEncoder().encode(value)); +} + +export async function gzipBytes(value: Uint8Array): Promise { + const stream = new Blob([value]).stream().pipeThrough(new CompressionStream('gzip')); + return new Uint8Array(await new Response(stream).arrayBuffer()); +} + +export function gzipJson(value: unknown): Promise { + return gzipText(JSON.stringify(value)); +} + +export async function gunzipJson(bytes: Uint8Array): Promise { + const stream = new Blob([bytes]).stream().pipeThrough(new DecompressionStream('gzip')); + return JSON.parse(await new Response(stream).text()) as unknown; +} diff --git a/apps/api/tests/integration/workspace-resource-history-http.test.ts b/apps/api/tests/integration/workspace-resource-history-http.test.ts new file mode 100644 index 000000000..a247c3b8a --- /dev/null +++ b/apps/api/tests/integration/workspace-resource-history-http.test.ts @@ -0,0 +1,192 @@ +import Database from 'better-sqlite3'; +import { Hono } from 'hono'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; + +import * as schema from '../../src/db/schema'; +import type { Env } from '../../src/env'; +import { handleAppError } from '../../src/middleware/app-error-handler'; +import type { AuthContext } from '../../src/middleware/auth'; +import { projectResourceHistoryRoutes } from '../../src/routes/projects/workspace-resource-history'; +import { workspaceResourceHistoryCallbackRoute } from '../../src/routes/projects/workspace-resource-history-callback'; +import { verifyCallbackToken } from '../../src/services/jwt'; +import { base64, gzipJson, sha256Hex } from '../helpers/resource-history'; +import { createSchemaTables, createSqliteD1 } from '../helpers/sqlite-d1'; + +vi.mock('../../src/services/jwt', async (importOriginal) => ({ + ...(await importOriginal()), + verifyCallbackToken: vi.fn(), +})); + +function authContext(): AuthContext { + return { + user: { + id: 'member-1', + email: 'member-1@example.test', + name: 'Resource history owner', + avatarUrl: null, + role: 'user', + status: 'active', + }, + session: { + id: 'browser-session-1', + token: 'browser-token', + expiresAt: new Date('2027-01-01T00:00:00.000Z'), + }, + }; +} + +function makeR2() { + const objects = new Map(); + return { + objects, + binding: { + put: async (key: string, value: Uint8Array) => { + objects.set(key, value); + return null; + }, + delete: async (key: string) => objects.delete(key), + get: async (key: string) => { + const value = objects.get(key); + if (!value) return null; + return { + arrayBuffer: async () => + value.buffer.slice(value.byteOffset, value.byteOffset + value.byteLength), + body: new Blob([value]).stream(), + }; + }, + } as unknown as R2Bucket, + }; +} + +describe('workspace resource history HTTP vertical slice', () => { + let sqlite: Database.Database; + let env: Env; + let app: Hono<{ Bindings: Env }>; + let r2: ReturnType; + + beforeEach(() => { + vi.clearAllMocks(); + vi.mocked(verifyCallbackToken).mockResolvedValue({ + workspace: 'node-1', + type: 'callback', + scope: 'node', + }); + + sqlite = new Database(':memory:'); + createSchemaTables(sqlite, [ + schema.projects, + schema.projectMembers, + schema.workspaces, + schema.tasks, + schema.agentSessions, + schema.agentProfiles, + schema.skills, + schema.workspaceResourceSummaries, + schema.workspaceResourceChunks, + ]); + sqlite + .prepare(`INSERT INTO projects (id, user_id, name) VALUES ('proj-1', 'member-1', 'Project')`) + .run(); + sqlite + .prepare( + `INSERT INTO project_members (project_id, user_id, role, status) + VALUES ('proj-1', 'member-1', 'owner', 'active')` + ) + .run(); + sqlite + .prepare( + `INSERT INTO workspaces (id, project_id, node_id, chat_session_id) + VALUES ('ws-1', 'proj-1', 'node-1', 'session-1')` + ) + .run(); + + r2 = makeR2(); + env = { + DATABASE: createSqliteD1(sqlite), + PROJECT_DATA_ARCHIVE_R2: r2.binding, + } as unknown as Env; + app = new Hono<{ Bindings: Env }>(); + app.use('*', async (c, next) => { + c.set('auth', authContext()); + await next(); + }); + app.onError(handleAppError); + app.route('/api/projects', workspaceResourceHistoryCallbackRoute); + app.route('/api/projects', projectResourceHistoryRoutes); + }); + + afterEach(() => sqlite.close()); + + it('uploads and reads working-set summaries and raw samples through real routes', async () => { + const payload = { + samples: [ + { t: 1_000, memoryBytes: 1024, memoryWorkingSetBytes: 512 }, + { t: 1_500, memoryBytes: 4096, memoryWorkingSetBytes: 1024 }, + { t: 2_000, memoryBytes: 2048, memoryWorkingSetBytes: 768 }, + ], + toolSpans: [], + gaps: [], + }; + const compressed = await gzipJson(payload); + const body = { + workspaceId: 'ws-1', + nodeId: 'node-1', + sessionId: 'session-1', + taskId: null, + sourceVersion: 1, + chunkSequence: 0, + startedAt: 1_000, + endedAt: 2_000, + sampleCount: 3, + gapCount: 0, + toolSpanCount: 0, + compressedBase64: base64(compressed), + compressedBytes: compressed.byteLength, + uncompressedBytes: JSON.stringify(payload).length, + sha256: await sha256Hex(compressed), + completeness: { status: 'complete' }, + summary: { + memoryMeanBytes: 2389, + memoryPeakBytes: 4096, + memoryKernelPeakBytes: 8192, + memoryWorkingSetMeanBytes: 768, + memoryWorkingSetPeakBytes: 1024, + memoryWorkingSetSampleCount: 3, + }, + }; + + const upload = await app.fetch( + new Request('https://api.test/api/projects/proj-1/workspace-resource-history', { + method: 'POST', + headers: { Authorization: 'Bearer callback-token', 'Content-Type': 'application/json' }, + body: JSON.stringify(body), + }), + env + ); + expect(upload.status).toBe(200); + const uploaded = (await upload.json()) as { chunkId: string }; + expect(r2.objects.size).toBe(1); + + const read = await app.fetch( + new Request( + `https://api.test/api/projects/proj-1/sessions/session-1/resource-history?chunkId=${encodeURIComponent(uploaded.chunkId)}` + ), + env + ); + expect(read.status).toBe(200); + const history = (await read.json()) as { + summary: Record; + detail: { samples: Array> }; + }; + expect(history.summary).toMatchObject({ + memoryPeakBytes: 4096, + memoryKernelPeakBytes: 8192, + memoryWorkingSetMeanBytes: 768, + memoryWorkingSetPeakBytes: 1024, + }); + expect(history.detail.samples.map((sample) => sample.memoryBytes)).toEqual([1024, 4096, 2048]); + expect(history.detail.samples.map((sample) => sample.memoryWorkingSetBytes)).toEqual([ + 512, 1024, 768, + ]); + }); +}); diff --git a/apps/api/tests/unit/resource-history-tools.test.ts b/apps/api/tests/unit/resource-history-tools.test.ts index f59e643be..79e6c951c 100644 --- a/apps/api/tests/unit/resource-history-tools.test.ts +++ b/apps/api/tests/unit/resource-history-tools.test.ts @@ -44,6 +44,8 @@ describe('resource history MCP tool', () => { agentProfileId: 'profile-1', skillId: 'skill-1', agentType: 'openai-codex', + memoryWorkingSetMeanBytes: 512, + memoryWorkingSetPeakBytes: 1024, }, chunks: [{ id: 'wrchunk:1', sampleCount: 12 }], detail: { @@ -85,6 +87,8 @@ describe('resource history MCP tool', () => { agentProfileId: 'profile-1', skillId: 'skill-1', agentType: 'openai-codex', + memoryWorkingSetMeanBytes: 512, + memoryWorkingSetPeakBytes: 1024, }, detail: { toolSpans: [{ id: 'hashed', kind: 'execute', toolName: 'Bash', startedAt: 1 }], @@ -149,6 +153,8 @@ describe('SAM native get_resource_history tool', () => { agentProfileId: 'profile-1', skillId: 'skill-1', agentType: 'openai-codex', + memoryWorkingSetMeanBytes: 512, + memoryWorkingSetPeakBytes: 1024, }, chunks: [{ id: 'wrchunk:1', sampleCount: 7 }], detail: { @@ -197,6 +203,8 @@ describe('SAM native get_resource_history tool', () => { agentProfileId: 'profile-1', skillId: 'skill-1', agentType: 'openai-codex', + memoryWorkingSetMeanBytes: 512, + memoryWorkingSetPeakBytes: 1024, }, detail: { toolSpans: [ diff --git a/apps/api/tests/unit/workspace-resource-history-vertical.test.ts b/apps/api/tests/unit/workspace-resource-history-vertical.test.ts index cbf0c3af5..c77b2a758 100644 --- a/apps/api/tests/unit/workspace-resource-history-vertical.test.ts +++ b/apps/api/tests/unit/workspace-resource-history-vertical.test.ts @@ -8,23 +8,13 @@ import { handleAppError } from '../../src/middleware/app-error-handler'; import { workspaceResourceHistoryCallbackRoute } from '../../src/routes/projects/workspace-resource-history-callback'; import { verifyCallbackToken } from '../../src/services/jwt'; import { getWorkspaceResourceHistory } from '../../src/services/workspace-resource-history'; +import { base64, gzipText, sha256Hex } from '../helpers/resource-history'; import { createSchemaTables, createSqliteD1 } from '../helpers/sqlite-d1'; vi.mock('../../src/services/jwt', () => ({ verifyCallbackToken: vi.fn(), })); -function base64(bytes: Uint8Array): string { - let binary = ''; - for (const byte of bytes) binary += String.fromCharCode(byte); - return btoa(binary); -} - -async function sha256Hex(bytes: Uint8Array): Promise { - const digest = await crypto.subtle.digest('SHA-256', bytes); - return [...new Uint8Array(digest)].map((byte) => byte.toString(16).padStart(2, '0')).join(''); -} - describe('workspace resource history vertical slice', () => { it('resolves server attribution through the callback and exposes the persisted summary', async () => { const sqlite = new Database(':memory:'); @@ -54,11 +44,7 @@ describe('workspace resource history vertical slice', () => { `); const samples = JSON.stringify({ samples: [{ t: 1, cpuMillis: 1, memoryBytes: 2 }] }); - const compressed = new Uint8Array( - await new Response( - new Blob([samples]).stream().pipeThrough(new CompressionStream('gzip')) - ).arrayBuffer() - ); + const compressed = await gzipText(samples); const storedObjects = new Map(); const env = { DATABASE: createSqliteD1(sqlite), diff --git a/apps/api/tests/unit/workspace-resource-history.test.ts b/apps/api/tests/unit/workspace-resource-history.test.ts index 003f38ab7..60c4264b0 100644 --- a/apps/api/tests/unit/workspace-resource-history.test.ts +++ b/apps/api/tests/unit/workspace-resource-history.test.ts @@ -12,46 +12,38 @@ import { storeWorkspaceResourceChunk, type WorkspaceResourceUploadBody, } from '../../src/services/workspace-resource-history'; +import { base64, gunzipJson, gzipBytes, gzipJson, sha256Hex } from '../helpers/resource-history'; import { createSchemaTables, createSqliteD1 } from '../helpers/sqlite-d1'; -function base64(bytes: Uint8Array): string { - let binary = ''; - for (const byte of bytes) binary += String.fromCharCode(byte); - return btoa(binary); -} - -async function sha256Hex(bytes: Uint8Array): Promise { - const digest = await crypto.subtle.digest('SHA-256', bytes); - return [...new Uint8Array(digest)].map((byte) => byte.toString(16).padStart(2, '0')).join(''); -} - -async function gzipJson(value: unknown): Promise { - return gzipBytes(new TextEncoder().encode(JSON.stringify(value))); -} - -async function gzipBytes(value: Uint8Array): Promise { - const stream = new Blob([value]).stream().pipeThrough(new CompressionStream('gzip')); - return new Uint8Array(await new Response(stream).arrayBuffer()); -} - -async function gunzipJson(bytes: Uint8Array): Promise { - const stream = new Blob([bytes]).stream().pipeThrough(new DecompressionStream('gzip')); - return JSON.parse(await new Response(stream).text()) as unknown; -} - -function samplePayload() { +function samplePayload({ includeWorkingSet = true }: { includeWorkingSet?: boolean } = {}) { return { samples: [ - { t: 1_000, cpuMillis: 0, memoryBytes: 1024, ioReadBytes: 0, ioWriteBytes: 0 }, + { + t: 1_000, + cpuMillis: 0, + memoryBytes: 1024, + ...(includeWorkingSet ? { memoryWorkingSetBytes: 512 } : {}), + ioReadBytes: 0, + ioWriteBytes: 0, + }, { t: 1_500, cpuMillis: 40, memoryBytes: 2048, memoryPeakBytes: 4096, + ...(includeWorkingSet ? { memoryWorkingSetBytes: 1024 } : {}), ioReadBytes: 10, ioWriteBytes: 20, }, - { t: 2_000, cpuMillis: 900, memoryBytes: 1536, ioReadBytes: 5, ioWriteBytes: 7, oom: 1 }, + { + t: 2_000, + cpuMillis: 900, + memoryBytes: 1536, + ...(includeWorkingSet ? { memoryWorkingSetBytes: 768 } : {}), + ioReadBytes: 5, + ioWriteBytes: 7, + oom: 1, + }, ], toolSpans: [ { @@ -69,9 +61,10 @@ function samplePayload() { } async function uploadBody( - overrides: Partial = {} + overrides: Partial = {}, + payload = samplePayload() ): Promise { - const compressed = await gzipJson(samplePayload()); + const compressed = await gzipJson(payload); return { workspaceId: 'ws-1', nodeId: 'node-1', @@ -86,7 +79,7 @@ async function uploadBody( toolSpanCount: 1, compressedBase64: base64(compressed), compressedBytes: compressed.byteLength, - uncompressedBytes: JSON.stringify(samplePayload()).length, + uncompressedBytes: JSON.stringify(payload).length, sha256: await sha256Hex(compressed), completeness: { status: 'complete' }, summary: { @@ -95,6 +88,9 @@ async function uploadBody( memoryMeanBytes: 1536, memoryPeakBytes: 2048, memoryKernelPeakBytes: 4096, + memoryWorkingSetMeanBytes: 768, + memoryWorkingSetPeakBytes: 1024, + memoryWorkingSetSampleCount: 3, ioReadBytes: 15, ioWriteBytes: 27, oomCount: 1, @@ -154,7 +150,7 @@ function makeEnv(sqlite: Database.Database, r2: R2Bucket): Env { } as unknown as Env; } -function makePersistedResourceTestEnv() { +function makePersistedResourceTestEnv({ includeSummary = true } = {}) { const sqlite = new Database(':memory:'); createSchemaTables(sqlite, [ schema.workspaces, @@ -162,7 +158,7 @@ function makePersistedResourceTestEnv() { schema.agentSessions, schema.agentProfiles, schema.skills, - schema.workspaceResourceSummaries, + ...(includeSummary ? [schema.workspaceResourceSummaries] : []), schema.workspaceResourceChunks, ]); sqlite @@ -218,6 +214,118 @@ describe('workspace resource history', () => { ]); }); + it('aggregates working-set samples without treating old-agent omissions as zero', async () => { + const { sqlite, env } = makePersistedResourceTestEnv(); + + await storeWorkspaceResourceChunk(env, 'proj-1', await uploadBody(), 'node-1'); + await storeWorkspaceResourceChunk( + env, + 'proj-1', + await uploadBody({ + chunkSequence: 1, + startedAt: 2_001, + endedAt: 3_000, + sampleCount: 2, + summary: { + memoryWorkingSetMeanBytes: 1536, + memoryWorkingSetPeakBytes: 2048, + memoryWorkingSetSampleCount: 2, + }, + }), + 'node-1' + ); + await storeWorkspaceResourceChunk( + env, + 'proj-1', + await uploadBody({ + chunkSequence: 2, + startedAt: 3_001, + endedAt: 4_000, + summary: {}, + }), + 'node-1' + ); + + const history = await getWorkspaceResourceHistory(env, { + projectId: 'proj-1', + sessionId: 'session-1', + }); + expect(history.summary).toMatchObject({ + sampleCount: 8, + memoryWorkingSetMeanBytes: 1075, + memoryWorkingSetPeakBytes: 2048, + }); + expect( + sqlite + .prepare( + `SELECT memory_working_set_sample_count AS count + FROM workspace_resource_summaries` + ) + .get() + ).toEqual({ count: 5 }); + }); + + it('keeps working-set summary fields null for old-agent uploads', async () => { + const { env } = makePersistedResourceTestEnv(); + + const oldAgentPayload = samplePayload({ includeWorkingSet: false }); + const stored = await storeWorkspaceResourceChunk( + env, + 'proj-1', + await uploadBody({ summary: {} }, oldAgentPayload), + 'node-1' + ); + + const history = await getWorkspaceResourceHistory(env, { + projectId: 'proj-1', + sessionId: 'session-1', + detailChunkId: stored.chunkId, + }); + expect(history.summary).toMatchObject({ + memoryWorkingSetMeanBytes: null, + memoryWorkingSetPeakBytes: null, + }); + expect(history.detail?.samples).toHaveLength(3); + expect(history.detail?.samples.every((sample) => sample.memoryWorkingSetBytes == null)).toBe( + true + ); + }); + + it('initializes working-set aggregates when a new-agent chunk follows old-agent history', async () => { + const { sqlite, env } = makePersistedResourceTestEnv(); + + await storeWorkspaceResourceChunk( + env, + 'proj-1', + await uploadBody({ summary: {} }, samplePayload({ includeWorkingSet: false })), + 'node-1' + ); + await storeWorkspaceResourceChunk( + env, + 'proj-1', + await uploadBody({ chunkSequence: 1, startedAt: 2_001, endedAt: 3_000 }), + 'node-1' + ); + + const history = await getWorkspaceResourceHistory(env, { + projectId: 'proj-1', + sessionId: 'session-1', + }); + expect(history.summary).toMatchObject({ + sampleCount: 6, + memoryWorkingSetMeanBytes: 768, + memoryWorkingSetPeakBytes: 1024, + }); + expect( + sqlite + .prepare( + `SELECT memory_working_set_sample_count AS count + FROM workspace_resource_summaries` + ) + .get() + ).toEqual({ count: 3 }); + }); + it('keeps separate bounded summaries for reused workspace sessions while indexing raw chunks in R2', async () => { const sqlite = new Database(':memory:'); createSchemaTables(sqlite, [ @@ -301,6 +409,9 @@ describe('workspace resource history', () => { gaps: [expect.objectContaining({ reason: 'sample_error' })], }); expect(history.detail?.samples.some((sample) => sample.cpuMillis === 900)).toBe(true); + expect(history.detail?.samples.some((sample) => sample.memoryWorkingSetBytes === 1024)).toBe( + true + ); const summaries = sqlite .prepare( @@ -497,6 +608,7 @@ describe('workspace resource history', () => { sampleCount: 1, gapCount: 0, toolSpanCount: 3, + summary: {}, compressedBase64: base64(compressed), compressedBytes: compressed.byteLength, uncompressedBytes: new TextEncoder().encode(JSON.stringify(payload)).byteLength, @@ -634,23 +746,7 @@ describe('workspace resource history', () => { }); it('deletes the uploaded R2 object when D1 indexing fails', async () => { - const sqlite = new Database(':memory:'); - createSchemaTables(sqlite, [ - schema.workspaces, - schema.tasks, - schema.agentSessions, - schema.agentProfiles, - schema.skills, - schema.workspaceResourceChunks, - ]); - sqlite - .prepare( - `INSERT INTO workspaces (id, project_id, node_id, chat_session_id) - VALUES ('ws-1', 'proj-1', 'node-1', 'session-1')` - ) - .run(); - const r2 = makeR2(); - const env = makeEnv(sqlite, r2.binding); + const { r2, env } = makePersistedResourceTestEnv({ includeSummary: false }); await expect( storeWorkspaceResourceChunk(env, 'proj-1', await uploadBody(), 'node-1') diff --git a/apps/web/src/components/chat/SessionResourceHistoryDrawer.tsx b/apps/web/src/components/chat/SessionResourceHistoryDrawer.tsx index 953ac165a..7fc29fe13 100644 --- a/apps/web/src/components/chat/SessionResourceHistoryDrawer.tsx +++ b/apps/web/src/components/chat/SessionResourceHistoryDrawer.tsx @@ -30,7 +30,8 @@ interface SessionResourceHistoryDrawerProps { } function formatBytes(value: number | null | undefined): string { - if (!value || value <= 0) return '—'; + if (value == null || !Number.isFinite(value)) return '—'; + if (value <= 0) return '0 B'; const units = ['B', 'KB', 'MB', 'GB', 'TB']; let next = value; let unit = 0; @@ -65,6 +66,11 @@ function sampleMemoryMiB(sample: WorkspaceResourceSample): number { return Number(sample.memoryBytes ?? 0) / (1024 * 1024); } +function sampleWorkingSetMiB(sample: WorkspaceResourceSample): number | null { + if (sample.memoryWorkingSetBytes == null) return null; + return Number(sample.memoryWorkingSetBytes) / (1024 * 1024); +} + function sampleCpuMillis(sample: WorkspaceResourceSample): number { return Number(sample.cpuMillis ?? 0); } @@ -84,24 +90,61 @@ function sampleX(samples: WorkspaceResourceSample[], sample: WorkspaceResourceSa function seriesPoints( samples: WorkspaceResourceSample[], - valueForSample: (sample: WorkspaceResourceSample) => number + valueForSample: (sample: WorkspaceResourceSample) => number | null, + scaleMax?: number ): string { - const max = Math.max(1, ...samples.map(valueForSample)); + const values = samples.map(valueForSample).filter((value): value is number => value != null); + const max = Math.max(1, scaleMax ?? 0, ...values); return samples + .filter((sample) => valueForSample(sample) != null) .map((sample) => { const x = sampleX(samples, sample); - const y = 100 - (valueForSample(sample) / max) * 84 - 8; + const value = valueForSample(sample) ?? 0; + const y = 100 - (value / max) * 84 - 8; return `${x.toFixed(2)},${y.toFixed(2)}`; }) .join(' '); } +function seriesSegments( + samples: WorkspaceResourceSample[], + valueForSample: (sample: WorkspaceResourceSample) => number | null, + scaleMax?: number +): string[] { + const values = samples.map(valueForSample).filter((value): value is number => value != null); + const max = Math.max(1, scaleMax ?? 0, ...values); + const segments: string[][] = []; + let current: string[] = []; + + for (const sample of samples) { + const value = valueForSample(sample); + if (value == null) { + if (current.length > 0) segments.push(current); + current = []; + continue; + } + const x = sampleX(samples, sample); + const y = 100 - (value / max) * 84 - 8; + current.push(`${x.toFixed(2)},${y.toFixed(2)}`); + } + if (current.length > 0) segments.push(current); + return segments.map((segment) => segment.join(' ')); +} + export function ResourceSparkline({ samples, toolSpans, }: Readonly<{ samples: WorkspaceResourceSample[]; toolSpans: WorkspaceResourceToolSpan[] }>) { const cpuPoints = useMemo(() => seriesPoints(samples, sampleCpuMillis), [samples]); - const memoryPoints = useMemo(() => seriesPoints(samples, sampleMemoryMiB), [samples]); + const memoryScaleMax = useMemo(() => Math.max(1, ...samples.map(sampleMemoryMiB)), [samples]); + const totalMemoryPoints = useMemo( + () => seriesPoints(samples, sampleMemoryMiB, memoryScaleMax), + [memoryScaleMax, samples] + ); + const workingSetSegments = useMemo( + () => seriesSegments(samples, sampleWorkingSetMiB, memoryScaleMax), + [memoryScaleMax, samples] + ); const eventMarkers = useMemo( () => samples @@ -172,13 +215,25 @@ export function ResourceSparkline({ vectorEffect="non-scaling-stroke" /> + {workingSetSegments.map((points, index) => ( + + ))} {eventMarkers.map(({ sample, x }) => (
CPU: green solid line, normalized to the CPU peak for this chunk. - RAM: purple dashed line, normalized to the RAM peak for this chunk. + Memory needed: purple solid line (working set, when reported). + Total RAM: faint purple dashed line, including reclaimable file cache. Blue bands: concurrent tool windows. Dashed markers: gaps, counter resets, or OOM samples. @@ -328,7 +384,17 @@ export function ResourceHistoryContent({ + + = 24 ? 1_342_177_280 : 610_000_000 + i * 8_000_000, + memoryWorkingSetBytes: + i === 12 ? undefined : i === 24 ? 671_088_640 : 275_000_000 + i * 5_000_000, ioReadBytes: i % 8 === 0 ? 1_048_576 : 16_384, ioWriteBytes: i % 9 === 0 ? 4_194_304 : 65_536, oom: i === 24 ? 1 : 0, @@ -307,6 +311,8 @@ interface MockOptions extends SessionOptions { empty?: boolean; /** Seeds a long conversation so the scroll-to-bottom button can actually appear. */ manyMessages?: boolean; + legacyResourceHistory?: boolean; + resourceHistoryScenario?: 'normal' | 'empty' | 'error' | 'many'; } async function setupMocks(page: Page, options: MockOptions = {}) { @@ -316,6 +322,8 @@ async function setupMocks(page: Page, options: MockOptions = {}) { messagesLong = false, empty = false, manyMessages = false, + legacyResourceHistory = false, + resourceHistoryScenario = 'normal', } = options; await page.addInitScript( @@ -368,12 +376,47 @@ async function setupMocks(page: Page, options: MockOptions = {}) { } if (pathname === `/api/projects/${PROJECT_ID}/sessions/${SESSION_ID}/resource-history`) { + if (resourceHistoryScenario === 'error') { + await route.fulfill({ status: 500, json: { error: 'resource_history_unavailable' } }); + return; + } + if (resourceHistoryScenario === 'empty') { + await route.fulfill({ json: { summary: null, chunks: [] } }); + return; + } const includeDetail = new URL(url).searchParams.get('chunkId') === 'wrchunk-rail-2'; + const chunks = + resourceHistoryScenario === 'many' + ? Array.from({ length: 30 }, (_, index) => ({ + ...RESOURCE_HISTORY_CHUNKS[0], + id: `wrchunk-many-${index}`, + chunkSequence: 30 - index, + startedAt: NOW - (index + 2) * 900_000, + endedAt: NOW - (index + 1) * 900_000, + })) + : RESOURCE_HISTORY_CHUNKS; await route.fulfill({ json: { - summary: RESOURCE_HISTORY_SUMMARY, - chunks: RESOURCE_HISTORY_CHUNKS, - ...(includeDetail ? { detail: RESOURCE_HISTORY_DETAIL } : {}), + summary: legacyResourceHistory + ? { + ...RESOURCE_HISTORY_SUMMARY, + memoryWorkingSetMeanBytes: null, + memoryWorkingSetPeakBytes: null, + } + : RESOURCE_HISTORY_SUMMARY, + chunks, + ...(includeDetail + ? { + detail: legacyResourceHistory + ? { + ...RESOURCE_HISTORY_DETAIL, + samples: RESOURCE_HISTORY_DETAIL.samples.map( + ({ memoryWorkingSetBytes: _memoryWorkingSetBytes, ...sample }) => sample + ), + } + : RESOURCE_HISTORY_DETAIL, + } + : {}), }, }); return; @@ -1063,7 +1106,9 @@ test.describe('Session resource history drawer', () => { // Stat cards — use exact match to avoid ambiguity with chart legend text await expect(page.getByText('CPU peak', { exact: true })).toBeVisible(); - await expect(page.getByText('RAM peak', { exact: true })).toBeVisible(); + await expect(page.getByText('Memory needed (peak)', { exact: true })).toBeVisible(); + await expect(page.getByText('640 MB', { exact: true })).toBeVisible(); + await expect(page.getByText('Total RAM (incl. cache)', { exact: true })).toBeVisible(); await expect(page.getByText('1 OOM event observed in retained samples.')).toBeVisible(); // Chart auto-loads via useEffect selecting newest chunk — wait for it @@ -1074,8 +1119,16 @@ test.describe('Session resource history drawer', () => { page.getByText('CPU: green solid line, normalized to the CPU peak for this chunk.') ).toBeVisible(); await expect( - page.getByText('RAM: purple dashed line, normalized to the RAM peak for this chunk.') + page.getByText('Memory needed: purple solid line (working set, when reported).') + ).toBeVisible(); + await expect( + page.getByText('Total RAM: faint purple dashed line, including reclaimable file cache.') ).toBeVisible(); + await expect( + page + .getByRole('img', { name: 'CPU and memory resource timeline' }) + .locator('polyline[data-series="working-set"]') + ).toHaveCount(2); await expect(page.getByText(/Blue bands: concurrent tool windows/)).toBeVisible(); await expect(page.getByText('Tool windows', { exact: true })).toBeVisible(); await expect(page.getByText(/Bash ·/)).toBeVisible(); @@ -1116,4 +1169,49 @@ test.describe('Session resource history drawer', () => { }); await capture(page, `resource-history-chunks-${page.viewportSize()?.width ?? 'viewport'}`); }); + + test('shows unknown working set for history from an older VM agent', async ({ page }) => { + await openChat(page, { state: 'active', legacyResourceHistory: true }); + await page.getByTestId('session-tool-resources').click(); + + for (const label of ['Memory needed (peak)', 'Memory needed (mean)']) { + const neededCard = page.getByText(label, { exact: true }).locator('..'); + await expect(neededCard).toContainText('—'); + await expect(neededCard).not.toContainText('0 B'); + } + await expect( + page + .getByRole('img', { name: 'CPU and memory resource timeline' }) + .locator('polyline[data-series="working-set"]') + ).toHaveCount(0); + await capture(page, `resource-history-legacy-${page.viewportSize()?.width ?? 'viewport'}`); + }); + + test('shows the empty resource-history state', async ({ page }) => { + await openChat(page, { state: 'active', resourceHistoryScenario: 'empty' }); + await page.getByTestId('session-tool-resources').click(); + + await expect( + page.getByText('No retained resource history is available for this session yet.') + ).toBeVisible(); + await capture(page, `resource-history-empty-${page.viewportSize()?.width ?? 'viewport'}`); + }); + + test('shows the resource-history error state', async ({ page }) => { + await openChat(page, { state: 'active', resourceHistoryScenario: 'error' }); + await page.getByTestId('session-tool-resources').click(); + + await expect(page.getByText('Resource history could not be loaded.')).toBeVisible(); + await capture(page, `resource-history-error-${page.viewportSize()?.width ?? 'viewport'}`); + }); + + test('keeps a long chunk list usable', async ({ page }) => { + await openChat(page, { state: 'active', resourceHistoryScenario: 'many' }); + await page.getByTestId('session-tool-resources').click(); + + const chunksToggle = page.getByRole('button', { name: '30 chunks' }); + await chunksToggle.click(); + await expect(page.getByRole('button', { name: /samples/ })).toHaveCount(30); + await capture(page, `resource-history-many-${page.viewportSize()?.width ?? 'viewport'}`); + }); }); diff --git a/apps/www/public/images/docs/session-resources-drawer-mobile.png b/apps/www/public/images/docs/session-resources-drawer-mobile.png index f40950b9d..5ac9822b2 100644 Binary files a/apps/www/public/images/docs/session-resources-drawer-mobile.png and b/apps/www/public/images/docs/session-resources-drawer-mobile.png differ diff --git a/apps/www/public/images/docs/session-resources-drawer.png b/apps/www/public/images/docs/session-resources-drawer.png index 6af8b0163..112b9d7da 100644 Binary files a/apps/www/public/images/docs/session-resources-drawer.png and b/apps/www/public/images/docs/session-resources-drawer.png differ diff --git a/apps/www/src/content/docs/docs/guides/session-resources.md b/apps/www/src/content/docs/docs/guides/session-resources.md index 888924099..6dc514abf 100644 --- a/apps/www/src/content/docs/docs/guides/session-resources.md +++ b/apps/www/src/content/docs/docs/guides/session-resources.md @@ -24,12 +24,12 @@ The button is always there, including on sessions that already ended — which i want it, because the workspace is gone and this is the only record left. It is also there on sessions that never collected anything, where it shows an empty state rather than hiding itself. -![The Resources drawer for a chat session: stat cards reading CPU peak 4120 ms/sample, RAM peak 3.4 GB, I/O total 384 MB read and 1.1 GB write, and 360 samples with 1 gap; an amber banner reading "1 OOM event observed in retained samples"; a detail timeline chart with a green CPU line, a dashed purple RAM line, blue tool-window bands and an amber OOM marker; Tool windows labeled Bash, search, and tool; and a collapsed "2 chunks" disclosure.](/images/docs/session-resources-drawer.png) +![The Resources drawer for a chat session: stat cards distinguish peak and mean memory needed from total RAM including cache, alongside CPU peak, I/O totals, and sample counts; an amber OOM banner; a detail timeline chart with CPU, working-set memory, cache-inclusive memory, named tool-window bands and an OOM marker; a Tool windows list labeled Bash, search, and tool; and a collapsed chunk disclosure.](/images/docs/session-resources-drawer.png) On mobile the same panel fills the screen and scrolls, with the stat cards and the OOM banner first so the answer is above the fold. -![The same Resources panel on a phone, filling the whole screen: the four stat cards stacked two by two, the amber OOM banner, the full timeline chart with its four-line legend and chunk I/O totals, and Tool windows labeled Bash, search, and tool, with the rest reachable by scrolling.](/images/docs/session-resources-drawer-mobile.png) +![The same Resources panel on a phone, filling the whole screen: six stat cards stacked two by two, the amber OOM banner, the full timeline with separate working-set and cache-inclusive memory lines, and Tool windows labeled Bash, search, and tool, with the rest reachable by scrolling.](/images/docs/session-resources-drawer-mobile.png) ## Where resource history exists — and where it doesn't @@ -51,14 +51,18 @@ The drawer stacks its content top to bottom in the order you normally need it. ### Stat cards -Four numbers for the whole session: +Six cards for the whole session: -| Card | What it means | -| ------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **CPU peak** | The busiest single sample, in milliseconds of CPU time. See the conversion below. | -| **RAM peak** | The highest total memory the container held at any sampled moment — **including page cache**, so read it with [this caveat](#what-this-does-not-tell-you). | -| **I/O total** | Bytes read and written over the session. | -| **Samples** | How many observations were retained, and how many **gaps** there are (see [Gaps and resets](#gaps-and-resets)). | +| Card | What it means | +| ------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | +| **CPU peak** | The busiest single sample, in milliseconds of CPU time. See the conversion below. | +| **Memory needed (peak / mean)** | The non-reclaimable working set: `memory.current - inactive_file`, clamped at zero. Use the peak when sizing a workspace. | +| **Total RAM (incl. cache)** | The highest `memory.current` sample, including reclaimable page cache. Keep it for comparison, but do not size from it. | +| **I/O total** | Bytes read and written over the session. | +| **Samples** | How many observations were retained, and how many **gaps** there are (see [Gaps and resets](#gaps-and-resets)). | + +History uploaded by an older VM agent has no working-set fields. SAM shows an em dash for those +cards rather than treating an absent measurement as zero. **Converting CPU peak to cores.** CPU is reported as CPU-milliseconds consumed per sample, and SAM samples every 5 seconds by default (`RESOURCE_HISTORY_SAMPLE_INTERVAL`). One core running flat @@ -93,16 +97,17 @@ hitting the limit, raise the memory the work asks for. See The chart loads automatically for the most recent slice of the session: - **Green solid line** — CPU, normalized to this slice's own CPU peak. -- **Purple dashed line** — RAM, normalized to this slice's own RAM peak. +- **Purple solid line** — memory needed (working set), when the VM agent reported it. +- **Faint purple dashed line** — total RAM including reclaimable file cache. - **Blue bands** — tool windows: stretches where the agent had one or more tool calls in flight. A fainter band means SAM inferred the end of the window rather than observing it. - **An amber marker at the top**, with a dashed line down the chart — an out-of-memory sample. - **A small grey dot at the bottom** — a gap or a counter reset. Easy to miss, and worth not missing: see [Gaps and resets](#gaps-and-resets). -Each line is scaled to its **own** peak within the slice, so the two lines are shaped for reading -against the tool bands — not against each other. A tall green line does not mean CPU is higher -than RAM. +CPU is scaled to its own peak within the slice. Both memory lines share the total-RAM scale, so +their vertical separation shows how much of the cache-inclusive total is the working set. CPU and +memory still use different units, so their heights cannot be compared with each other. Under the chart sits its legend and the I/O read/write totals for this slice, then **Tool windows** — each window with its start time, duration, and how many tool calls overlapped. SAM uses @@ -151,10 +156,10 @@ stretch. The panel is easy to over-read. Four things it cannot tell you: -- **RAM peak is not "how much memory the program needed."** It is the container's total cgroup +- **Total RAM is not "how much memory the program needed."** It is the container's total cgroup memory, which includes reclaimable page cache. A workspace that reads or writes large files — - a clone, a build, a test run — climbs toward the machine's limit as a matter of course, with no - memory pressure at all. **Treat the OOM banner, not RAM peak, as evidence that memory ran out.** + a clone, a build, a test run — can make total RAM climb while the working set stays much lower. + Use **Memory needed (peak)** for sizing and the OOM banner as evidence that memory ran out. - **It is not per-process attribution.** Samples come from the workspace's cgroup — the whole container, including the agent harness, your dev server, test runners, and background jobs. A tool window that overlaps a CPU spike is a _correlation_, not proof that the tool caused the spike. @@ -192,9 +197,10 @@ the project default for "everything here needs more", the agent profile for "thi the task itself for a one-off. Remember the host reserve: a 4 GiB machine can only back a ~3.5 GiB reservation, so asking for exactly 4 GB pushes you onto an 8 GiB machine. -Do **not** use "RAM peak looks close to the machine size" as your trigger. That number includes -page cache (see [What this does not tell you](#what-this-does-not-tell-you)), so a workspace that -reads large files reaches it while having plenty of memory to spare. The OOM banner is the signal. +Use **Memory needed (peak)** as the historical sizing input. Do **not** use "Total RAM looks close +to the machine size" as your trigger: that number includes page cache, so a workspace that reads +large files can reach it while having plenty of reclaimable memory. The OOM banner remains the +strongest signal that the configured limit was insufficient. **If CPU peak never approached one core and there was no OOM**, you are likely paying for headroom you never used. Lower the requirement, or set the pool's **Workspace strategy** to **Smallest fit** @@ -231,12 +237,12 @@ unit on the node; the defaults are what every managed node runs. ## What is actually stored -The retained payload is deliberately narrow: timestamps, CPU-milliseconds, memory bytes, I/O bytes, -process counts, OOM flags, and hashed tool-call IDs with their start and end times. Tool spans may -also carry an ACP kind and a bounded metadata tool name; older history may contain neither. Its -summary also retains nullable `agentProfileId`, `skillId`, and `agentType` attribution. SAM resolves -those fields from server-owned records for the same project and workspace instead of trusting upload -values. +The retained payload is deliberately narrow: timestamps, CPU-milliseconds, total and working-set +memory bytes, I/O bytes, process counts, OOM flags, and hashed tool-call IDs with their start and end +times. Tool spans may also carry an ACP kind and a bounded metadata tool name; older history may +contain neither. Its summary also retains nullable `agentProfileId`, `skillId`, and `agentType` +attribution. SAM resolves those fields from server-owned records for the same project and workspace +instead of trusting upload values. It contains **no** prompts, messages, tool-call titles, commands, tool inputs or arguments, tool output, file paths, environment variables, or secrets. That is what makes it safe to keep for months @@ -255,7 +261,7 @@ stays cheap. There is no `projectId` parameter — the project comes from the ag This turns a vague complaint into a checkable one. For example: > "Task `01M2…` failed near the end. Use `get_resource_history` for that task, tell me whether it -> hit an OOM, and if so what its RAM peak was." +> hit an OOM, and if so what its peak working set was." The same information is available over HTTP at `GET /api/projects/:projectId/sessions/:sessionId/resource-history` (also `…/tasks/:taskId/…` and diff --git a/apps/www/src/content/docs/docs/reference/api.md b/apps/www/src/content/docs/docs/reference/api.md index 8d8be444f..4805836de 100644 --- a/apps/www/src/content/docs/docs/reference/api.md +++ b/apps/www/src/content/docs/docs/reference/api.md @@ -175,7 +175,7 @@ Retained CPU, memory, I/O, and out-of-memory observations for VM-backed workspac | GET | `/api/projects/:id/tasks/:taskId/resource-history` | History scoped to one task | | GET | `/api/projects/:id/workspaces/:workspaceId/resource-history` | History scoped to one workspace | -All three return `{ summary, chunks }`, where `summary` carries the session's peaks, sample and gap counts, I/O totals, OOM count, and nullable `agentProfileId`, `skillId`, and `agentType` attribution. SAM resolves those attribution fields from server-owned records scoped to the same project and workspace; upload values cannot override them. `chunks` is the index of retained time slices, newest first and capped at `WORKSPACE_RESOURCE_LIST_LIMIT` (24) with no cursor or offset — `?chunkId=` still resolves a chunk outside that window, but nothing enumerates the older IDs. Add `?chunkId=` to include a `detail` object with that chunk's samples and tool-correlation spans, downsampled to `WORKSPACE_RESOURCE_DETAIL_MAX_POINTS` (720) with spikes preserved. Each `detail.toolSpans` entry may include an ACP `kind` and metadata-provided `toolName`; either can be absent in history from older VM agents. Tool names are capped by `WORKSPACE_RESOURCE_TOOL_NAME_MAX_BYTES` (256 by default). +All three return `{ summary, chunks }`, where `summary` carries the session's peaks, sample and gap counts, I/O totals, OOM count, and nullable `agentProfileId`, `skillId`, and `agentType` attribution. SAM resolves those attribution fields from server-owned records scoped to the same project and workspace; upload values cannot override them. `memoryWorkingSetMeanBytes` and `memoryWorkingSetPeakBytes` exclude reclaimable inactive file cache and are the sizing figures; they are `null` for history uploaded by older VM agents. `memoryMeanBytes`, `memoryPeakBytes`, and `memoryKernelPeakBytes` remain cache-inclusive for comparison. `chunks` is the index of retained time slices, newest first and capped at `WORKSPACE_RESOURCE_LIST_LIMIT` (24) with no cursor or offset — `?chunkId=` still resolves a chunk outside that window, but nothing enumerates the older IDs. Add `?chunkId=` to include a `detail` object with that chunk's samples and tool-correlation spans, downsampled to `WORKSPACE_RESOURCE_DETAIL_MAX_POINTS` (720) with spikes preserved. Each sample can include `memoryWorkingSetBytes`; when it is absent, that sample's working set is unknown rather than zero. Each `detail.toolSpans` entry may include an ACP `kind` and metadata-provided `toolName`; either can be absent in history from older VM agents. Tool names are capped by `WORKSPACE_RESOURCE_TOOL_NAME_MAX_BYTES` (256 by default). Samples are workspace-cgroup observations, not per-process attribution. Stored payloads deliberately exclude prompts, tool-call titles, commands, tool inputs and arguments, tool output, file paths, environment values, and secrets; tool-call IDs are hashed. A tool name identifies the tool implementation, such as `Bash`, without exposing what it ran. Instant (Cloudflare Container) sessions have no resource history: the response is `200` with `summary: null` and an empty `chunks` array. diff --git a/packages/vm-agent/internal/resourcehistory/collector.go b/packages/vm-agent/internal/resourcehistory/collector.go index e75b1f7f5..76f9a844e 100644 --- a/packages/vm-agent/internal/resourcehistory/collector.go +++ b/packages/vm-agent/internal/resourcehistory/collector.go @@ -94,19 +94,20 @@ type Collector struct { } type Sample struct { - T int64 `json:"t"` - IntervalMillis int64 `json:"intervalMillis,omitempty"` - CPUMillis int64 `json:"cpuMillis,omitempty"` - MemoryBytes uint64 `json:"memoryBytes,omitempty"` - MemoryPeakBytes uint64 `json:"memoryPeakBytes,omitempty"` - IOReadBytes uint64 `json:"ioReadBytes,omitempty"` - IOWriteBytes uint64 `json:"ioWriteBytes,omitempty"` - OOM uint64 `json:"oom,omitempty"` - OOMKill uint64 `json:"oomKill,omitempty"` - PidsCurrent uint64 `json:"pidsCurrent,omitempty"` - CounterReset bool `json:"counterReset,omitempty"` - Unsupported string `json:"unsupported,omitempty"` - Gap bool `json:"gap,omitempty"` + T int64 `json:"t"` + IntervalMillis int64 `json:"intervalMillis,omitempty"` + CPUMillis int64 `json:"cpuMillis,omitempty"` + MemoryBytes uint64 `json:"memoryBytes,omitempty"` + MemoryPeakBytes uint64 `json:"memoryPeakBytes,omitempty"` + MemoryWorkingSetBytes *uint64 `json:"memoryWorkingSetBytes,omitempty"` + IOReadBytes uint64 `json:"ioReadBytes,omitempty"` + IOWriteBytes uint64 `json:"ioWriteBytes,omitempty"` + OOM uint64 `json:"oom,omitempty"` + OOMKill uint64 `json:"oomKill,omitempty"` + PidsCurrent uint64 `json:"pidsCurrent,omitempty"` + CounterReset bool `json:"counterReset,omitempty"` + Unsupported string `json:"unsupported,omitempty"` + Gap bool `json:"gap,omitempty"` } type ToolSpan struct { @@ -128,16 +129,19 @@ type chunkPayload struct { } type summaryPayload struct { - CPUMeanMillis *float64 `json:"cpuMeanMillis,omitempty"` - CPUPeakMillis *int64 `json:"cpuPeakMillis,omitempty"` - MemoryMeanBytes *uint64 `json:"memoryMeanBytes,omitempty"` - MemoryPeakBytes *uint64 `json:"memoryPeakBytes,omitempty"` - MemoryKernelPeakBytes *uint64 `json:"memoryKernelPeakBytes,omitempty"` - IOReadBytes *uint64 `json:"ioReadBytes,omitempty"` - IOWriteBytes *uint64 `json:"ioWriteBytes,omitempty"` - OOMCount *uint64 `json:"oomCount,omitempty"` - SampleIntervalMillis int64 `json:"sampleIntervalMillis"` - WeightedMeanWallMillis int64 `json:"weightedMeanWallMillis"` + CPUMeanMillis *float64 `json:"cpuMeanMillis,omitempty"` + CPUPeakMillis *int64 `json:"cpuPeakMillis,omitempty"` + MemoryMeanBytes *uint64 `json:"memoryMeanBytes,omitempty"` + MemoryPeakBytes *uint64 `json:"memoryPeakBytes,omitempty"` + MemoryKernelPeakBytes *uint64 `json:"memoryKernelPeakBytes,omitempty"` + MemoryWorkingSetMeanBytes *uint64 `json:"memoryWorkingSetMeanBytes,omitempty"` + MemoryWorkingSetPeakBytes *uint64 `json:"memoryWorkingSetPeakBytes,omitempty"` + MemoryWorkingSetSampleCount int `json:"memoryWorkingSetSampleCount,omitempty"` + IOReadBytes *uint64 `json:"ioReadBytes,omitempty"` + IOWriteBytes *uint64 `json:"ioWriteBytes,omitempty"` + OOMCount *uint64 `json:"oomCount,omitempty"` + SampleIntervalMillis int64 `json:"sampleIntervalMillis"` + WeightedMeanWallMillis int64 `json:"weightedMeanWallMillis"` } type uploadBody struct { @@ -434,7 +438,12 @@ func (c *Collector) sample(ctx context.Context) { func (c *Collector) sampleFromCounters(now time.Time, counters cgroupCounters) Sample { c.mu.Lock() defer c.mu.Unlock() - sample := Sample{T: now.UnixMilli(), MemoryBytes: counters.MemoryCurrent, MemoryPeakBytes: counters.MemoryPeak} + sample := Sample{ + T: now.UnixMilli(), + MemoryBytes: counters.MemoryCurrent, + MemoryPeakBytes: counters.MemoryPeak, + MemoryWorkingSetBytes: counters.MemoryWorkingSet, + } sample.PidsCurrent = counters.PidsCurrent if !c.lastSampleAt.IsZero() { sample.IntervalMillis = now.Sub(c.lastSampleAt).Milliseconds() @@ -541,6 +550,9 @@ func summarize(samples []Sample, interval time.Duration) summaryPayload { var totalMem uint64 var peakMem uint64 var kernelPeak uint64 + var totalWorkingSet uint64 + var peakWorkingSet uint64 + var workingSetSamples int var read uint64 var write uint64 var oom uint64 @@ -557,6 +569,13 @@ func summarize(samples []Sample, interval time.Duration) summaryPayload { if s.MemoryPeakBytes > kernelPeak { kernelPeak = s.MemoryPeakBytes } + if s.MemoryWorkingSetBytes != nil { + totalWorkingSet += *s.MemoryWorkingSetBytes + if *s.MemoryWorkingSetBytes > peakWorkingSet { + peakWorkingSet = *s.MemoryWorkingSetBytes + } + workingSetSamples++ + } read += s.IOReadBytes write += s.IOWriteBytes oom += s.OOM + s.OOMKill @@ -571,6 +590,12 @@ func summarize(samples []Sample, interval time.Duration) summaryPayload { out.MemoryMeanBytes = &meanMem out.MemoryPeakBytes = &peakMem out.MemoryKernelPeakBytes = &kernelPeak + if workingSetSamples > 0 { + meanWorkingSet := totalWorkingSet / uint64(workingSetSamples) + out.MemoryWorkingSetMeanBytes = &meanWorkingSet + out.MemoryWorkingSetPeakBytes = &peakWorkingSet + out.MemoryWorkingSetSampleCount = workingSetSamples + } out.IOReadBytes = &read out.IOWriteBytes = &write out.OOMCount = &oom @@ -776,14 +801,15 @@ func hashedToolID(raw string) string { } type cgroupCounters struct { - CPUUsageUsec uint64 - MemoryCurrent uint64 - MemoryPeak uint64 - IOReadBytes uint64 - IOWriteBytes uint64 - OOM uint64 - OOMKill uint64 - PidsCurrent uint64 + CPUUsageUsec uint64 + MemoryCurrent uint64 + MemoryPeak uint64 + MemoryWorkingSet *uint64 + IOReadBytes uint64 + IOWriteBytes uint64 + OOM uint64 + OOMKill uint64 + PidsCurrent uint64 } func (c *Collector) resolveCgroupPath(ctx context.Context) (string, error) { @@ -911,8 +937,18 @@ func readCgroupCounters(path string) (cgroupCounters, error) { return out, err } out.CPUUsageUsec = cpu["usage_usec"] - out.MemoryCurrent, _ = readUintFile(filepath.Join(path, "memory.current")) + var memoryCurrentErr error + out.MemoryCurrent, memoryCurrentErr = readUintFile(filepath.Join(path, "memory.current")) out.MemoryPeak, _ = readUintFile(filepath.Join(path, "memory.peak")) + if inactiveFile, statErr := readInactiveFile(filepath.Join(path, "memory.stat")); memoryCurrentErr == nil && statErr == nil { + workingSet := out.MemoryCurrent + if inactiveFile < workingSet { + workingSet -= inactiveFile + } else { + workingSet = 0 + } + out.MemoryWorkingSet = &workingSet + } ioStats, _ := readIOStat(filepath.Join(path, "io.stat")) out.IOReadBytes = ioStats[0] out.IOWriteBytes = ioStats[1] @@ -950,6 +986,28 @@ func readKeyedUintFile(path string) (map[string]uint64, error) { return result, nil } +func readInactiveFile(path string) (uint64, error) { + data, err := os.ReadFile(path) + if err != nil { + return 0, err + } + for _, line := range strings.Split(string(data), "\n") { + fields := strings.Fields(line) + if len(fields) == 0 || fields[0] != "inactive_file" { + continue + } + if len(fields) != 2 { + return 0, fmt.Errorf("invalid inactive_file entry in %s", path) + } + value, parseErr := strconv.ParseUint(fields[1], 10, 64) + if parseErr != nil { + return 0, fmt.Errorf("parse inactive_file in %s: %w", path, parseErr) + } + return value, nil + } + return 0, fmt.Errorf("inactive_file missing from %s", path) +} + func readIOStat(path string) ([2]uint64, error) { data, err := os.ReadFile(path) if err != nil { diff --git a/packages/vm-agent/internal/resourcehistory/collector_test.go b/packages/vm-agent/internal/resourcehistory/collector_test.go index 9ed8499e6..5a7e55b77 100644 --- a/packages/vm-agent/internal/resourcehistory/collector_test.go +++ b/packages/vm-agent/internal/resourcehistory/collector_test.go @@ -8,6 +8,7 @@ import ( "net/http/httptest" "os" "path/filepath" + "strconv" "strings" "testing" "time" @@ -30,6 +31,7 @@ func TestResolveCgroupPathFindsDockerScopeAndReadsCounters(t *testing.T) { writeFile(t, filepath.Join(path, "cpu.stat"), "usage_usec 123456\nuser_usec 1\n") writeFile(t, filepath.Join(path, "memory.current"), "4096\n") writeFile(t, filepath.Join(path, "memory.peak"), "8192\n") + writeFile(t, filepath.Join(path, "memory.stat"), "anon 1024\nfile 3072\ninactive_anon 256\nactive_anon 768\ninactive_file 3072\nactive_file 0\nslab 128\n") writeFile(t, filepath.Join(path, "io.stat"), "8:0 rbytes=100 wbytes=25 rios=1 wios=2\n8:16 rbytes=50 wbytes=75\n") writeFile(t, filepath.Join(path, "memory.events"), "oom 2\noom_kill 1\n") writeFile(t, filepath.Join(path, "pids.current"), "7\n") @@ -48,11 +50,115 @@ func TestResolveCgroupPathFindsDockerScopeAndReadsCounters(t *testing.T) { if counters.CPUUsageUsec != 123456 || counters.MemoryCurrent != 4096 || counters.MemoryPeak != 8192 { t.Fatalf("unexpected counters: %+v", counters) } + if counters.MemoryWorkingSet == nil || *counters.MemoryWorkingSet != 1024 { + t.Fatalf("working set = %v, want 1024", counters.MemoryWorkingSet) + } if counters.IOReadBytes != 150 || counters.IOWriteBytes != 100 || counters.OOM != 2 || counters.OOMKill != 1 || counters.PidsCurrent != 7 { t.Fatalf("unexpected io/events counters: %+v", counters) } } +func TestReadCgroupCountersWorkingSetMemory(t *testing.T) { + const gib = uint64(1024 * 1024 * 1024) + tests := []struct { + name string + current uint64 + memoryStat *string + want *uint64 + }{ + { + name: "large page cache", + current: 8 * gib, + memoryStat: stringPtr("anon 1610612736\nfile 6442450944\nkernel 536870912\ninactive_anon 268435456\nactive_anon 1342177280\ninactive_file 6442450944\nactive_file 0\nslab 268435456\n"), + want: uint64Ptr(2 * gib), + }, + { + name: "inactive file clamps at zero", + current: 2 * gib, + memoryStat: stringPtr("anon 536870912\nfile 3221225472\ninactive_file 3221225472\n"), + want: uint64Ptr(0), + }, + {name: "missing memory stat", current: 2 * gib, want: nil}, + { + name: "unparseable inactive file", + current: 2 * gib, + memoryStat: stringPtr("anon 536870912\ninactive_file not-a-number\nfile 1610612736\n"), + want: nil, + }, + { + name: "missing inactive file key", + current: 2 * gib, + memoryStat: stringPtr("anon 536870912\nfile 1610612736\n"), + want: nil, + }, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + path := t.TempDir() + writeFile(t, filepath.Join(path, "cpu.stat"), "usage_usec 1\n") + writeFile(t, filepath.Join(path, "memory.current"), strconv.FormatUint(test.current, 10)+"\n") + if test.memoryStat != nil { + writeFile(t, filepath.Join(path, "memory.stat"), *test.memoryStat) + } + + counters, err := readCgroupCounters(path) + if err != nil { + t.Fatalf("readCgroupCounters: %v", err) + } + if test.want == nil { + if counters.MemoryWorkingSet != nil { + t.Fatalf("working set = %d, want unknown", *counters.MemoryWorkingSet) + } + return + } + if counters.MemoryWorkingSet == nil || *counters.MemoryWorkingSet != *test.want { + t.Fatalf("working set = %v, want %d", counters.MemoryWorkingSet, *test.want) + } + }) + } +} + +func stringPtr(value string) *string { return &value } + +func uint64Ptr(value uint64) *uint64 { return &value } + +func TestSummarizeWorkingSetUsesOnlyKnownSamples(t *testing.T) { + known := uint64(768) + summary := summarize([]Sample{ + {T: 1, MemoryBytes: 4096}, + {T: 2, MemoryBytes: 4096, MemoryWorkingSetBytes: &known}, + }, 5*time.Second) + + if summary.MemoryWorkingSetMeanBytes == nil || *summary.MemoryWorkingSetMeanBytes != known { + t.Fatalf("working-set mean = %v, want %d", summary.MemoryWorkingSetMeanBytes, known) + } + if summary.MemoryWorkingSetPeakBytes == nil || *summary.MemoryWorkingSetPeakBytes != known { + t.Fatalf("working-set peak = %v, want %d", summary.MemoryWorkingSetPeakBytes, known) + } + if summary.MemoryWorkingSetSampleCount != 1 { + t.Fatalf("working-set sample count = %d, want 1", summary.MemoryWorkingSetSampleCount) + } + + encoded, err := json.Marshal(Sample{T: 1, MemoryBytes: 4096}) + if err != nil { + t.Fatalf("marshal unknown working set: %v", err) + } + if strings.Contains(string(encoded), "memoryWorkingSetBytes") { + t.Fatalf("unknown working set must be omitted, got %s", encoded) + } +} + +func TestSummarizeOmitsWorkingSetWhenAllSamplesAreUnknown(t *testing.T) { + summary := summarize([]Sample{{T: 1, MemoryBytes: 4096}}, 5*time.Second) + if summary.MemoryWorkingSetMeanBytes != nil || summary.MemoryWorkingSetPeakBytes != nil { + t.Fatalf("working-set summary = mean %v peak %v, want unknown", summary.MemoryWorkingSetMeanBytes, summary.MemoryWorkingSetPeakBytes) + } + if summary.MemoryWorkingSetSampleCount != 0 { + t.Fatalf("working-set sample count = %d, want 0", summary.MemoryWorkingSetSampleCount) + } +} + func TestResolveCgroupPathFindsNestedSystemdScopeWithShortContainerID(t *testing.T) { root := t.TempDir() fullID := "abcdef1234567890abcdef1234567890abcdef1234567890abcdef1234567890" @@ -144,6 +250,7 @@ func TestCollectorUploadsCompressedChunkAndHashesToolIDs(t *testing.T) { writeFile(t, filepath.Join(path, "cpu.stat"), "usage_usec 100000\n") writeFile(t, filepath.Join(path, "memory.current"), "1000\n") writeFile(t, filepath.Join(path, "memory.peak"), "1200\n") + writeFile(t, filepath.Join(path, "memory.stat"), "anon 600\nfile 400\ninactive_file 300\n") writeFile(t, filepath.Join(path, "io.stat"), "8:0 rbytes=10 wbytes=20\n") writeFile(t, filepath.Join(path, "memory.events"), "oom 0\noom_kill 0\n") writeFile(t, filepath.Join(path, "pids.current"), "5\n") @@ -170,6 +277,7 @@ func TestCollectorUploadsCompressedChunkAndHashesToolIDs(t *testing.T) { writeFile(t, filepath.Join(path, "cpu.stat"), "usage_usec 175000\n") writeFile(t, filepath.Join(path, "memory.current"), "2000\n") writeFile(t, filepath.Join(path, "memory.peak"), "2500\n") + writeFile(t, filepath.Join(path, "memory.stat"), "anon 900\nfile 1100\ninactive_file 800\n") writeFile(t, filepath.Join(path, "io.stat"), "8:0 rbytes=110 wbytes=220\n") collector.RecordACPToolCall("secret-tool-id", "in_progress", "execute", "Bash", now) collector.sample(context.Background()) @@ -186,6 +294,15 @@ func TestCollectorUploadsCompressedChunkAndHashesToolIDs(t *testing.T) { if received.CompressedBytes == 0 || received.SHA256 == "" || received.StorageFormat != StorageFormat { t.Fatalf("missing compressed payload metadata: %+v", received) } + if received.Summary.MemoryWorkingSetMeanBytes == nil || *received.Summary.MemoryWorkingSetMeanBytes != 950 { + t.Fatalf("working-set mean = %v, want 950", received.Summary.MemoryWorkingSetMeanBytes) + } + if received.Summary.MemoryWorkingSetPeakBytes == nil || *received.Summary.MemoryWorkingSetPeakBytes != 1200 { + t.Fatalf("working-set peak = %v, want 1200", received.Summary.MemoryWorkingSetPeakBytes) + } + if received.Summary.MemoryWorkingSetSampleCount != 2 { + t.Fatalf("working-set sample count = %d, want 2", received.Summary.MemoryWorkingSetSampleCount) + } } func TestCollectorRetrySpoolDropsPermanentClientErrorsAndContinues(t *testing.T) { @@ -274,6 +391,7 @@ func BenchmarkReadCgroupCounters(b *testing.B) { "cpu.stat": "usage_usec 123456\nuser_usec 1\nsystem_usec 2\n", "memory.current": "4096\n", "memory.peak": "8192\n", + "memory.stat": "anon 1024\nfile 3072\ninactive_file 3072\n", "io.stat": "8:0 rbytes=100 wbytes=25 rios=1 wios=2\n8:16 rbytes=50 wbytes=75\n", "memory.events": "oom 2\noom_kill 1\n", "pids.current": "7\n", diff --git a/tasks/active/2026-09-29-working-set-resource-history.md b/tasks/active/2026-09-29-working-set-resource-history.md new file mode 100644 index 000000000..97f1b35cc --- /dev/null +++ b/tasks/active/2026-09-29-working-set-resource-history.md @@ -0,0 +1,56 @@ +# Working-set memory in session resource history + +## Problem + +Session resource history records cgroup v2 `memory.current` and `memory.peak`. Both include reclaimable page cache, so file-heavy sessions can appear to require far more RAM than their non-reclaimable workload uses. Sizing consumers need a working-set metric while the existing totals remain available for historical comparison. + +## Research findings + +- `packages/vm-agent/internal/resourcehistory/collector.go` reads the cgroup counters, writes compressed sample chunks, and computes upload summaries. +- cAdvisor/kubelet-style working set is `memory.current - inactive_file`, clamped to zero. `inactive_file` is read from `memory.stat`; failure to read or parse that file must leave the metric absent. +- `apps/api/src/services/workspace-resource-history.ts` validates upload summaries, stores aggregate columns in `workspace_resource_summaries`, serves raw chunk detail from R2, and downsamples spike-preserving samples. +- Migration `0169_workspace_resource_history.sql` created the summary table. This change needs an additive migration with nullable columns. +- Summary upserts aggregate multiple chunks. A missing value from an old agent must retain an existing known value, and a known value arriving after unknown chunks must use only the known sample count when computing a mean. +- The session drawer currently calls cache-inclusive `memory.current` “RAM peak” and charts it. The UI needs to lead with working set as memory needed, retain a clearly labelled cache-inclusive figure, and render unknown as an em dash. +- Both the project MCP route and the native SAM session tool spread the service response, so their schemas remain compatible; their descriptions/notes need to define the new fields. +- VM-agent staging validation must delete existing staging nodes before deployment, provision a new node afterward, verify heartbeat and the uploaded working-set values, then remove the test resources. + +## Implementation checklist + +- [x] Add strict `memory.stat` parsing and optional working-set samples with realistic Go fixtures. +- [x] Summarize known working-set samples into mean, peak, and known-sample count fields. +- [x] Add nullable D1 columns and absence-safe aggregate upsert behavior. +- [x] Add working-set fields to public API and web client types, detail downsampling, and MCP documentation. +- [x] Update the drawer cards and timeline labels to distinguish memory needed from cache-inclusive memory. +- [x] Cover new/old-agent uploads and cross-chunk aggregation in API tests. +- [x] Update the public guide and API reference. +- [x] Run focused tests, full quality gates, visual audit, specialist review, and staging VM verification. +- [ ] Run CI and CodeRabbit review on the pull request. + +## Acceptance criteria + +- Raw samples include working-set bytes only when both `memory.current` and a valid `inactive_file` value are available; subtraction clamps at zero. +- Missing, unreadable, or malformed `memory.stat` yields unknown working set, never a numeric zero. +- Session summaries expose mean and peak working set while retaining `memory.current` mean/peak and kernel peak. +- Old-agent uploads cannot overwrite a known working-set aggregate or cause an unknown value to display as zero. +- HTTP and MCP reads expose the nullable fields, and the UI calls working set the needed figure while labelling totals as cache-inclusive. +- Unit tests cover realistic `memory.stat`, large page cache, missing file, malformed value, and mixed-version aggregation. +- A fresh staging VM reports a populated, plausible working set; heartbeat and workspace access are verified; staging resources are cleaned up. + +## Staging verification + +- Exact-head staging deploy run `36550714488` succeeded for commit `8f07e3712`. +- Fresh node `01M3P9M7KHACV1420HJEKEV82P` ran a 128 MiB file-cache workload in session `82e505de-1fd4-487d-830f-d981b9ac2d1c` and completed a final resource-history flush. +- The collector uploaded 53 samples, including 24 known working-set samples. Every known value satisfied `0 <= working set <= memory.current`. +- Working-set mean was 293,991,082 bytes (280 MB) and peak was 447,438,848 bytes (427 MB), versus a cache-inclusive sampled peak of 1,293,877,248 bytes (1.2 GB). +- The deployed drawer rendered the real summary correctly at desktop and mobile sizes. Dashboard, projects, and settings smoke routes remained healthy. +- The test session, workspace, and node were deleted after capture; the staging node list was empty at handoff. + +## References + +- `.claude/rules/73-optional-struct-fields-must-not-overwrite-on-absence.md` +- `packages/vm-agent/.claude/rules/27-vm-agent-staging-refresh.md` +- `packages/vm-agent/.claude/rules/54-vm-agent-rollout-compatibility.md` +- `apps/api/.claude/rules/31-migration-safety.md` +- `apps/web/.claude/rules/17-ui-visual-testing.md` +- `tasks/archive/2026-09-20-workspace-resource-history.md` diff --git a/tasks/evidence/2026-09-29-working-set-resource-history/README.md b/tasks/evidence/2026-09-29-working-set-resource-history/README.md new file mode 100644 index 000000000..7041fd958 --- /dev/null +++ b/tasks/evidence/2026-09-29-working-set-resource-history/README.md @@ -0,0 +1,9 @@ +# Working-set resource history staging evidence + +Staging deploy run `36550714488` installed commit `8f07e3712`. A fresh VM ran a 128 MiB file-cache workload, then completed a final resource-history flush. + +The captured summary in `staging-live-history.json` contains 53 samples, 24 with a known working set. Mean working set is 293,991,082 bytes (280 MB), peak working set is 447,438,848 bytes (427 MB), and the cache-inclusive sampled peak is 1,293,877,248 bytes (1.2 GB). Every known working-set sample was between zero and `memory.current`. + +`staging-live-drawer-desktop.png` and `staging-live-drawer-mobile.png` show the exact deployed Pages bundle rendering that captured real-VM summary. The authenticated session, workspace, and node were removed after the summary was captured, so the screenshot check replayed the captured response into the deployed UI while all other API traffic remained live. Dashboard, projects, and settings smoke routes also rendered without an error boundary. + +The temporary verification harnesses were removed after the run. Final repository tests cover the collector, storage/read paths, MCP output, and drawer behavior. diff --git a/tasks/evidence/2026-09-29-working-set-resource-history/staging-live-drawer-desktop.png b/tasks/evidence/2026-09-29-working-set-resource-history/staging-live-drawer-desktop.png new file mode 100644 index 000000000..b3d0801a9 Binary files /dev/null and b/tasks/evidence/2026-09-29-working-set-resource-history/staging-live-drawer-desktop.png differ diff --git a/tasks/evidence/2026-09-29-working-set-resource-history/staging-live-drawer-mobile.png b/tasks/evidence/2026-09-29-working-set-resource-history/staging-live-drawer-mobile.png new file mode 100644 index 000000000..03a5c0c86 Binary files /dev/null and b/tasks/evidence/2026-09-29-working-set-resource-history/staging-live-drawer-mobile.png differ diff --git a/tasks/evidence/2026-09-29-working-set-resource-history/staging-live-history.json b/tasks/evidence/2026-09-29-working-set-resource-history/staging-live-history.json new file mode 100644 index 000000000..2234cc2c1 --- /dev/null +++ b/tasks/evidence/2026-09-29-working-set-resource-history/staging-live-history.json @@ -0,0 +1,59 @@ +{ + "taskId": "01M3P9KYDBGAXTNKAZ6CECQ6XD", + "sessionId": "82e505de-1fd4-487d-830f-d981b9ac2d1c", + "workspaceId": "01M3P9VYRQSF7EJSNP00F9ZYFW", + "nodeId": "01M3P9M7KHACV1420HJEKEV82P", + "summary": { + "id": "workspace:01KJVGMWX26SGQ5DX94GMTJRQN:01M3P9VYRQSF7EJSNP00F9ZYFW:session:82e505de-1fd4-487d-830f-d981b9ac2d1c", + "projectId": "01KJVGMWX26SGQ5DX94GMTJRQN", + "workspaceId": "01M3P9VYRQSF7EJSNP00F9ZYFW", + "sessionId": "82e505de-1fd4-487d-830f-d981b9ac2d1c", + "taskId": "01M3P9KYDBGAXTNKAZ6CECQ6XD", + "nodeId": "01M3P9M7KHACV1420HJEKEV82P", + "agentProfileId": "01KQ4YQ3FQYQ19HWYGZPACGF73", + "skillId": null, + "agentType": null, + "runtime": "vm", + "sourceVersion": 1, + "startedAt": 1790676178712, + "endedAt": 1790676440691, + "sampleCount": 53, + "gapCount": 0, + "toolSpanCount": 5, + "cpuMeanMillis": 866.8301886792453, + "cpuPeakMillis": 7001, + "memoryMeanBytes": 403901691, + "memoryPeakBytes": 1293877248, + "memoryKernelPeakBytes": 1306095616, + "memoryWorkingSetMeanBytes": 293991082, + "memoryWorkingSetPeakBytes": 447438848, + "ioReadBytes": 241664, + "ioWriteBytes": 877498368, + "oomCount": 0, + "completeness": { + "finalFlush": true, + "nodeLossMayLoseUnflushedWindow": true, + "source": "cgroup-v2", + "status": "complete", + "unsupported": "" + }, + "summary": { + "cpuMeanMillis": 866.8301886792453, + "cpuPeakMillis": 7001, + "memoryMeanBytes": 403901691, + "memoryPeakBytes": 1293877248, + "memoryKernelPeakBytes": 1306095616, + "memoryWorkingSetMeanBytes": 293991082, + "memoryWorkingSetPeakBytes": 447438848, + "memoryWorkingSetSampleCount": 24, + "ioReadBytes": 241664, + "ioWriteBytes": 877498368, + "oomCount": 0, + "sampleIntervalMillis": 5000, + "weightedMeanWallMillis": 119991 + }, + "firstChunkId": "wrchunk:01KJVGMWX26SGQ5DX94GMTJRQN:01M3P9VYRQSF7EJSNP00F9ZYFW:session:82e505de-1fd4-487d-830f-d981b9ac2d1c:1:1790676178712", + "latestChunkId": "wrchunk:01KJVGMWX26SGQ5DX94GMTJRQN:01M3P9VYRQSF7EJSNP00F9ZYFW:session:82e505de-1fd4-487d-830f-d981b9ac2d1c:1:1790676178712" + }, + "knownSamples": 24 +} \ No newline at end of file