From 42079bd7a71197815fea7c20b9b54dbca4b5ec39 Mon Sep 17 00:00:00 2001 From: Jay Moran Date: Tue, 22 Sep 2026 14:50:31 -0700 Subject: [PATCH 1/3] Stop selecting Tool calls as a default metric An experiment is created before its canvas exists, so the default set can't know whether a tool will be wired. On a canvas without one, Tool calls can never measure anything. The metrics editor already offers it once a tool is attached. --- frontend/src/lib/metricCatalog.ts | 2 +- src/asaree/services/metrics.py | 6 ++++-- tests/test_experiments.py | 23 ----------------------- tests/test_metrics.py | 4 +--- 4 files changed, 6 insertions(+), 29 deletions(-) diff --git a/frontend/src/lib/metricCatalog.ts b/frontend/src/lib/metricCatalog.ts index 647f1ed..9d6a6de 100644 --- a/frontend/src/lib/metricCatalog.ts +++ b/frontend/src/lib/metricCatalog.ts @@ -36,7 +36,7 @@ export const METRIC_CATALOG: readonly MetricCatalogEntry[] = [ { key: 'input_tokens', name: 'Input tokens', shortDescription: 'Tokens sent to the model during the run.', kind: 'runtime', valueType: 'number', defaultDirection: 'minimize', aggregation: 'sum', unit: 'tokens', category: 'Tokens', source: 'Run summary' }, { key: 'output_tokens', name: 'Output tokens', shortDescription: 'Tokens generated by the model during the run.', kind: 'runtime', valueType: 'number', defaultDirection: 'minimize', aggregation: 'sum', unit: 'tokens', category: 'Tokens', source: 'Run summary' }, { key: 'total_tokens', name: 'Total tokens', shortDescription: 'Combined input and output tokens for the run.', kind: 'runtime', valueType: 'number', defaultDirection: 'minimize', aggregation: 'sum', unit: 'tokens', category: 'Tokens', source: 'Run summary', recommended: true }, - { key: 'tool_calls', name: 'Tool calls', shortDescription: 'Recorded tool-call attempts across attributed Agent runs.', kind: 'runtime', valueType: 'number', defaultDirection: 'minimize', aggregation: 'sum', unit: 'calls', category: 'Tools', source: 'Protocol activity', recommended: true }, + { key: 'tool_calls', name: 'Tool calls', shortDescription: 'Recorded tool-call attempts across attributed Agent runs.', kind: 'runtime', valueType: 'number', defaultDirection: 'minimize', aggregation: 'sum', unit: 'calls', category: 'Tools', source: 'Protocol activity' }, { key: 'tool_error_rate', name: 'Tool error rate', shortDescription: 'Failed tool calls divided by all tool-call attempts.', kind: 'runtime', valueType: 'number', defaultDirection: 'minimize', aggregation: 'mean', unit: 'rate', category: 'Tools', source: 'Protocol activity' }, { key: 'agent_loop_iterations', name: 'Agent-loop iterations', shortDescription: 'Distinct Agent reasoning-loop iterations in the run.', kind: 'runtime', valueType: 'number', defaultDirection: 'minimize', aggregation: 'sum', unit: 'iterations', category: 'Agents', source: 'Protocol activity' }, { key: 'critic_rejections', name: 'Critic rejections', shortDescription: 'Rejected critic-gate reviews during the run.', kind: 'runtime', valueType: 'number', defaultDirection: 'minimize', aggregation: 'sum', unit: 'reviews', category: 'Critics', source: 'Protocol activity' }, diff --git a/src/asaree/services/metrics.py b/src/asaree/services/metrics.py index 4e0b3e5..7505b54 100644 --- a/src/asaree/services/metrics.py +++ b/src/asaree/services/metrics.py @@ -116,12 +116,14 @@ }, ) _CATALOG_BY_KEY = {str(entry["key"]): entry for entry in METRIC_CATALOG} -RECOMMENDED_RUNTIME_METRIC_KEYS = ("cost_usd", "duration_seconds", "total_tokens", "tool_calls") +# Only metrics every run can produce. Tool calls is deliberately absent: an +# experiment is created before its canvas exists, and on a canvas with no tool +# wired it can never measure anything -- the metrics editor offers it once one is. +RECOMMENDED_RUNTIME_METRIC_KEYS = ("cost_usd", "duration_seconds", "total_tokens") _RECOMMENDED_RUNTIME_METRIC_IDS = { "cost_usd": "runtime-cost", "duration_seconds": "runtime-duration", "total_tokens": "runtime-total-tokens", - "tool_calls": "runtime-tool-calls", } _KINDS = {"runtime", "custom"} _VALUE_TYPES = {"number", "boolean", "opaque"} diff --git a/tests/test_experiments.py b/tests/test_experiments.py index aef9112..ec9bfef 100644 --- a/tests/test_experiments.py +++ b/tests/test_experiments.py @@ -132,18 +132,6 @@ async def test_create_experiment_returns_a_persisted_recommended_measurement_pla "primary": False, "unit": "tokens", }, - { - "id": "runtime-tool-calls", - "catalogKey": "tool_calls", - "name": "Tool calls", - "description": "Recorded tool-call attempts across attributed Agent runs.", - "kind": "runtime", - "valueType": "number", - "direction": "minimize", - "aggregation": "sum", - "primary": False, - "unit": "calls", - }, ] expected_plan = { "metrics": [ @@ -177,16 +165,6 @@ async def test_create_experiment_returns_a_persisted_recommended_measurement_pla "description": "Combined input and output tokens for the run.", "unit": "tokens", }, - { - "id": "runtime-tool-calls", - "name": "Tool calls", - "value_type": "number", - "direction": "minimize", - "aggregation": "sum", - "primary": False, - "description": "Recorded tool-call attempts across attributed Agent runs.", - "unit": "calls", - }, ], "producers": [ { @@ -197,7 +175,6 @@ async def test_create_experiment_returns_a_persisted_recommended_measurement_pla "cost_usd": "runtime-cost", "duration_seconds": "runtime-duration", "total_tokens": "runtime-total-tokens", - "tool_calls": "runtime-tool-calls", }, "artifacts": [], "config": {}, diff --git a/tests/test_metrics.py b/tests/test_metrics.py index 05beb27..55874a6 100644 --- a/tests/test_metrics.py +++ b/tests/test_metrics.py @@ -117,15 +117,13 @@ def test_recommended_runtime_measurements_form_one_normalized_non_primary_plan() "cost_usd", "duration_seconds", "total_tokens", - "tool_calls", ] - assert [metric["primary"] for metric in declarations] == [False, False, False, False] + assert [metric["primary"] for metric in declarations] == [False, False, False] assert normalize_measurement_plan(plan) == plan assert plan["producers"][0]["outputs"] == { "cost_usd": "runtime-cost", "duration_seconds": "runtime-duration", "total_tokens": "runtime-total-tokens", - "tool_calls": "runtime-tool-calls", } From 1530f83e7ded7cfe58525bc91a745ae2a3570077 Mon Sep 17 00:00:00 2001 From: Jay Moran Date: Tue, 22 Sep 2026 14:56:44 -0700 Subject: [PATCH 2/3] Let the metrics dialog save when design metrics and plan disagree Deselecting a built-in only removed it from the plan if design_spec.metrics also declared it. When that list was missing, the plan kept naming metrics nothing declared and every save and publish failed with 422. Removal now also goes by the plan's runtime outputs, and existing plan ids are reused. An unavailable metric that is already selected can now be unchecked, and the default 'Add metrics' selection skips tool and critic metrics the canvas can't produce. --- .../protocol/MetricsEditor.autosave.test.tsx | 7 +++- .../src/components/protocol/MetricsEditor.tsx | 41 +++++++++++++++---- 2 files changed, 38 insertions(+), 10 deletions(-) diff --git a/frontend/src/components/protocol/MetricsEditor.autosave.test.tsx b/frontend/src/components/protocol/MetricsEditor.autosave.test.tsx index 5b818e9..98f7e25 100644 --- a/frontend/src/components/protocol/MetricsEditor.autosave.test.tsx +++ b/frontend/src/components/protocol/MetricsEditor.autosave.test.tsx @@ -25,14 +25,17 @@ describe('MetricsEditor autosave', () => { }) }) - it('selects every built-in by default and disables contextual metrics without eligible nodes', async () => { + it('selects every producible built-in by default and disables contextual metrics without eligible nodes', async () => { const user = userEvent.setup() renderEditor() await user.click(screen.getByRole('button', { name: 'Add metrics' })) const dialog = await screen.findByRole('dialog', { name: 'Manage metrics' }) + const contextual = new Set(['tool_calls', 'tool_error_rate', 'critic_approvals', 'critic_rejections']) for (const entry of METRIC_CATALOG.filter((candidate) => candidate.kind === 'runtime')) { - expect(within(dialog).getByRole('checkbox', { name: new RegExp(`^${entry.name}`) })).toBeChecked() + const checkbox = within(dialog).getByRole('checkbox', { name: new RegExp(`^${entry.name}`) }) + if (contextual.has(entry.key)) expect(checkbox).not.toBeChecked() + else expect(checkbox).toBeChecked() } expect(within(dialog).getByRole('checkbox', { name: /^Tool calls/ })).toHaveAttribute('aria-disabled', 'true') expect(within(dialog).getByRole('checkbox', { name: /^Tool error rate/ })).toHaveAttribute('aria-disabled', 'true') diff --git a/frontend/src/components/protocol/MetricsEditor.tsx b/frontend/src/components/protocol/MetricsEditor.tsx index efe6ec4..667d81a 100644 --- a/frontend/src/components/protocol/MetricsEditor.tsx +++ b/frontend/src/components/protocol/MetricsEditor.tsx @@ -144,12 +144,13 @@ function MetricsDialog({ const canvasMetricKeys = new Set(contextualMetricSuggestions(graph).map((suggestion) => suggestion.key)) const hasValidTool = canvasMetricKeys.has('tool_error_rate') const hasCriticGate = graph?.nodes.some((node) => node.type === 'critic_gate') ?? false + const canvasCannotProduce = (key: string) => + ((key === 'tool_calls' || key === 'tool_error_rate') && !hasValidTool) + || ((key === 'critic_approvals' || key === 'critic_rejections') && !hasCriticGate) const unavailableBuiltInKeys = capabilitiesLoading || capabilitiesUnavailable ? [] : builtInEntries.flatMap((entry) => { - const unavailable = !supportedBuiltInKeys.has(entry.key) - || ((entry.key === 'tool_calls' || entry.key === 'tool_error_rate') && !hasValidTool) - || ((entry.key === 'critic_approvals' || entry.key === 'critic_rejections') && !hasCriticGate) + const unavailable = !supportedBuiltInKeys.has(entry.key) || canvasCannotProduce(entry.key) return unavailable ? [entry.key] : [] }) const unavailableBuiltInKeySignature = unavailableBuiltInKeys.join('\u0000') @@ -207,7 +208,12 @@ function MetricsDialog({ } if (wasOpenRef.current) return wasOpenRef.current = true - setDraftKeys(new Set(initialDraftKeySet)) + // A default selection (nothing saved yet) must not pre-check a metric this + // canvas can't produce -- it would save a metric that can never report. + // A saved selection is shown as-is, so it can still be unchecked. + setDraftKeys(new Set(initialDraftSignature === undefined + ? initialDraftKeySet + : [...initialDraftKeySet].filter((key) => !canvasCannotProduce(key)))) setSaveError(undefined) setCustomChanges([]) setCustomMetricIds(initialCustomMetricIds) @@ -215,6 +221,9 @@ function MetricsDialog({ setCustomMetricDirty(false) setCustomMetricPendingDelete(undefined) setCustomMetricDraft(undefined) + // Seeds once per open (wasOpenRef); canvas availability changing while the + // dialog is open is handled by the newly-unavailable effect below. + // eslint-disable-next-line react-hooks/exhaustive-deps }, [open, initialDraftKeySet, metrics, initialCustomMetricIds, initialCustomMetricSignature]) useEffect(() => { @@ -317,7 +326,7 @@ function MetricsDialog({ const unavailableReasonId = `metric-unavailable-${entry.key}` return