From aabbf1365a194f68b8ac9fab5919551be064a6bc Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 14:40:41 +0200 Subject: [PATCH 01/21] fix(orchestrator): carry worker failures into synthesis, abstain when all fail Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- .../ProductionControlTests.cs | 37 ++++++++++++++++++- OrchestratorWorkers.AgentFramework/Program.cs | 18 +++++++-- .../WorkerRegistry.cs | 16 +++++++- .../patterns/OrchestratorWorkers.md | 18 ++++++--- 4 files changed, 77 insertions(+), 12 deletions(-) diff --git a/AgenticPatterns.Tests/ProductionControlTests.cs b/AgenticPatterns.Tests/ProductionControlTests.cs index 736b0d4..076b038 100644 --- a/AgenticPatterns.Tests/ProductionControlTests.cs +++ b/AgenticPatterns.Tests/ProductionControlTests.cs @@ -319,8 +319,43 @@ public async Task ExecutionHonorsConcurrencyAndPreservesPartialSuccess() var synthesis = WorkerRegistry.BuildSynthesisInput(results); Assert.Contains("one", synthesis); Assert.Contains("three", synthesis); - Assert.DoesNotContain("simulated failure", synthesis); + Assert.Contains("simulated failure", synthesis); + Assert.Contains("bad", synthesis); } + + [Fact] + public void SynthesisInputNamesTheFailures() + { + var input = WorkerRegistry.BuildSynthesisInput([ + new WorkerResult("t1", "research", "found A", null), + new WorkerResult("t2", "research", null, "timed out")]); + Assert.Contains("t2", input); + Assert.Contains("FAILED", input); + Assert.Contains("timed out", input); + } + + [Fact] + public void EveryWorkerFailingAbstainsInsteadOfSynthesising() => + Assert.Equal(RunCompleteness.Abstained, + WorkerRegistry.Assess([new WorkerResult("t1", "r", null, "boom")], requiredQuorum: 1)); + + [Fact] + public void PartialSuccessIsLabelledPartial() => + Assert.Equal(RunCompleteness.Partial, WorkerRegistry.Assess( + [new WorkerResult("t1", "r", "ok", null), new WorkerResult("t2", "r", null, "boom")], + requiredQuorum: 1)); + + [Fact] + public void AllSucceedingAndMeetingQuorumIsComplete() => + Assert.Equal(RunCompleteness.Complete, WorkerRegistry.Assess( + [new WorkerResult("t1", "r", "ok", null), new WorkerResult("t2", "r", "ok", null)], + requiredQuorum: 2)); + + [Fact] + public void AllSucceedingButBelowQuorumIsPartial() => + Assert.Equal(RunCompleteness.Partial, WorkerRegistry.Assess( + [new WorkerResult("t1", "r", "ok", null), new WorkerResult("t2", "r", "ok", null)], + requiredQuorum: 3)); } public class GuardRailsTests diff --git a/OrchestratorWorkers.AgentFramework/Program.cs b/OrchestratorWorkers.AgentFramework/Program.cs index f4ebba8..324d999 100644 --- a/OrchestratorWorkers.AgentFramework/Program.cs +++ b/OrchestratorWorkers.AgentFramework/Program.cs @@ -29,11 +29,21 @@ foreach (var result in results) Console.WriteLine($"\n[{result.TaskId}/{result.Worker}] {(result.Succeeded ? result.Output : "FAILED: " + result.Error)}"); +var completeness = WorkerRegistry.Assess(results, requiredQuorum: plan.Tasks.Count); +if (completeness == RunCompleteness.Abstained) +{ + Console.WriteLine("\n=== Synthesis ===\nAbstained: every worker failed, so there is no evidence to synthesize."); + return; +} + +var instruction = completeness == RunCompleteness.Partial + ? "Some tasks failed. State clearly which conclusions are unsupported; do not fill the gaps. " + + "Synthesize the worker reports into a concise recommendation." + : "Synthesize the worker reports into a concise recommendation."; var evidence = WorkerRegistry.BuildSynthesisInput(results); -var synthesizer = new ChatClientAgent(client, - "Synthesize the successful worker reports into a concise recommendation.", - "Synthesizer"); -Console.WriteLine($"\n=== Synthesis ===\n{await synthesizer.RunAsync($"Request: {request}\n\nWorker reports:\n{evidence}")}"); +var synthesizer = new ChatClientAgent(client, instruction, "Synthesizer"); +var label = completeness == RunCompleteness.Partial ? "PARTIAL" : "COMPLETE"; +Console.WriteLine($"\n=== Synthesis ({label}) ===\n{await synthesizer.RunAsync($"Request: {request}\n\nWorker reports:\n{evidence}")}"); return; Func> Run(string roleInstruction) => async (task, cancellationToken) => diff --git a/OrchestratorWorkers.AgentFramework/WorkerRegistry.cs b/OrchestratorWorkers.AgentFramework/WorkerRegistry.cs index 1797ec5..7dac1bf 100644 --- a/OrchestratorWorkers.AgentFramework/WorkerRegistry.cs +++ b/OrchestratorWorkers.AgentFramework/WorkerRegistry.cs @@ -44,6 +44,18 @@ public async Task> ExecuteAsync(WorkPlan plan, int m } public static string BuildSynthesisInput(IEnumerable results) => - string.Join("\n\n", results.Where(r => r.Succeeded) - .Select(r => $"## {r.TaskId} ({r.Worker})\n{r.Output}")); + string.Join("\n\n", results.Select(r => r.Succeeded + ? $"## {r.TaskId} ({r.Worker})\nSTATUS: OK\n{r.Output}" + : $"## {r.TaskId} ({r.Worker})\nSTATUS: FAILED\nERROR: {r.Error}\nNO OUTPUT - do not infer one.")); + + public static RunCompleteness Assess(IReadOnlyList results, int requiredQuorum) + { + var succeeded = results.Count(r => r.Succeeded); + if (succeeded == 0) return RunCompleteness.Abstained; + return succeeded == results.Count && succeeded >= requiredQuorum + ? RunCompleteness.Complete + : RunCompleteness.Partial; + } } + +public enum RunCompleteness { Complete, Partial, Abstained } diff --git a/PatternExplorer/patterns/OrchestratorWorkers.md b/PatternExplorer/patterns/OrchestratorWorkers.md index 91b9fe4..86caa33 100644 --- a/PatternExplorer/patterns/OrchestratorWorkers.md +++ b/PatternExplorer/patterns/OrchestratorWorkers.md @@ -1,7 +1,7 @@ --- { "title": "Orchestrator-Workers", - "summary": "Let an orchestrator choose request-specific tasks, validate them, run fixed workers with bounded concurrency, and synthesize the results.", + "summary": "Let an orchestrator choose request-specific tasks, validate them, run fixed workers with bounded concurrency, and synthesize the results while naming any that failed.", "category": "Orchestration", "projects": [ { "flavor": "AgentFramework", "path": "OrchestratorWorkers.AgentFramework" } ] } @@ -11,7 +11,8 @@ Orchestrator-Workers uses a central model to decide which independent subtasks a particular request needs. The host validates that typed plan, dispatches each task to a fixed worker registry, -and asks a synthesizer to combine successful results. +and asks a synthesizer to combine the results — successes and failures alike, so the synthesis +never quietly infers over a gap. This is dynamic fan-out, but not an open-ended team protocol: @@ -50,7 +51,11 @@ flowchart LR ``` `WorkerRegistry` uses `SemaphoreSlim` to cap concurrency and captures per-task failures without -discarding successful reports. Only successful outputs enter the synthesis evidence. +discarding successful reports. Every result enters the synthesis evidence, failed tasks included, +each tagged `STATUS: FAILED` with its error so the synthesizer cannot silently paper over a gap. +`WorkerRegistry.Assess` labels the run `Complete`, `Partial`, or `Abstained`; an all-failed run +skips synthesis entirely, and a partial run tells the synthesizer to call out unsupported +conclusions instead of inferring them. ## Key APIs @@ -58,9 +63,12 @@ discarding successful reports. Only successful outputs enter the synthesis evide - `PlanValidator.Validate(...)` — trusted-host validation before dispatch. - `WorkerRegistry` — fixed role-to-executor mapping; the model cannot create arbitrary workers. - `Task.WhenAll(...)` + `SemaphoreSlim` — concurrent execution with a hard concurrency ceiling. +- `WorkerRegistry.Assess(...)` — labels a run `Complete`, `Partial`, or `Abstained` from its results. ## What to watch in the output First inspect the serialized validated plan: its tasks should reflect the request rather than a -hard-coded fan-out. Worker outputs follow, then one synthesis. A worker failure becomes a failed -`WorkerResult`; it does not erase independent successful evidence. +hard-coded fan-out. Worker outputs follow, then one synthesis, labelled `COMPLETE` or `PARTIAL` — +or, if every worker failed, an abstention with no synthesis at all. A worker failure becomes a +failed `WorkerResult`; it does not erase independent successful evidence, and it is not hidden +from the synthesizer either. From 7a779732d0bc5a1b3ec2c133e02a8ef166ce96aa Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 14:45:38 +0200 Subject: [PATCH 02/21] docs(orchestrator): mark the tautological quorum and stop overclaiming the gap Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- OrchestratorWorkers.AgentFramework/Program.cs | 4 +++- PatternExplorer/patterns/OrchestratorWorkers.md | 5 +++-- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/OrchestratorWorkers.AgentFramework/Program.cs b/OrchestratorWorkers.AgentFramework/Program.cs index 324d999..7a540d8 100644 --- a/OrchestratorWorkers.AgentFramework/Program.cs +++ b/OrchestratorWorkers.AgentFramework/Program.cs @@ -29,10 +29,12 @@ foreach (var result in results) Console.WriteLine($"\n[{result.TaskId}/{result.Worker}] {(result.Succeeded ? result.Output : "FAILED: " + result.Error)}"); +// ponytail: this demo demands every planned task, so the quorum term can never be the +// deciding one here; pass a real minimum-viable subset when a caller can act on less. var completeness = WorkerRegistry.Assess(results, requiredQuorum: plan.Tasks.Count); if (completeness == RunCompleteness.Abstained) { - Console.WriteLine("\n=== Synthesis ===\nAbstained: every worker failed, so there is no evidence to synthesize."); + Console.WriteLine("\n=== Synthesis (ABSTAINED) ===\nEvery worker failed, so there is no evidence to synthesize."); return; } diff --git a/PatternExplorer/patterns/OrchestratorWorkers.md b/PatternExplorer/patterns/OrchestratorWorkers.md index 86caa33..cf86fc6 100644 --- a/PatternExplorer/patterns/OrchestratorWorkers.md +++ b/PatternExplorer/patterns/OrchestratorWorkers.md @@ -11,8 +11,9 @@ Orchestrator-Workers uses a central model to decide which independent subtasks a particular request needs. The host validates that typed plan, dispatches each task to a fixed worker registry, -and asks a synthesizer to combine the results — successes and failures alike, so the synthesis -never quietly infers over a gap. +and asks a synthesizer to combine the results — successes and failures alike. A failed task is +carried in with its error and an explicit instruction not to invent an output, so the gap is at +least visible to the synthesizer; the host abstains outright when every worker fails. This is dynamic fan-out, but not an open-ended team protocol: From 41fa1d2070acc50cd083fb0c0375a2247155d381 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 14:50:48 +0200 Subject: [PATCH 03/21] fix(retry): rethrow cancellation and scope whole-turn retry to idempotent turns Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- AgenticPatterns.Tests/RetryTests.cs | 9 +++++++++ .../Retry.cs | 16 ++++++++++++++++ .../RetryAndFallbackFilter.cs | 11 +++++++++++ .../patterns/ExceptionHandlingAndRecovery.md | 9 ++++++++- 4 files changed, 44 insertions(+), 1 deletion(-) diff --git a/AgenticPatterns.Tests/RetryTests.cs b/AgenticPatterns.Tests/RetryTests.cs index 550edfd..2c4efa8 100644 --- a/AgenticPatterns.Tests/RetryTests.cs +++ b/AgenticPatterns.Tests/RetryTests.cs @@ -62,6 +62,15 @@ public async Task RecoveryOnSecondAttempt_ReturnsResponse() Assert.Equal(2, attempts); } + [Fact] + public async Task CallerCancellationIsNotTurnedIntoAFallback() + { + using var cts = new CancellationTokenSource(); + await cts.CancelAsync(); + await Assert.ThrowsAnyAsync(() => + Retry.RunAsync(_ => Task.FromCanceled(cts.Token), maxRetries: 3, NoBackoff)); + } + [Fact] public void HasToolError_DetectsExceptionAndErrorString() { diff --git a/ExceptionHandlingAndRecovery.AgentFramework/Retry.cs b/ExceptionHandlingAndRecovery.AgentFramework/Retry.cs index d222098..93f6d6d 100644 --- a/ExceptionHandlingAndRecovery.AgentFramework/Retry.cs +++ b/ExceptionHandlingAndRecovery.AgentFramework/Retry.cs @@ -17,6 +17,18 @@ public static bool HasToolError(AgentResponse response) => /// Runs attempt() up to maxRetries times. Returns (null, lastError) when every attempt threw /// OR the final attempt still reported a tool error — a persistent tool failure is never a success. /// + /// + /// Whole-turn retry replays every tool call the turn made. Only use it for turns whose tools are + /// read-only or idempotent; a turn that issues a refund must retry at the tool boundary with an + /// idempotency key instead (see IdempotentToolCalls). + /// + /// is always rethrown, never retried. RunAsync takes no + /// of its own, so it has nothing to test the exception against + /// (the caller's token lives inside ) — there is no way to tell "the + /// caller cancelled" apart from "the tool timed out internally" here. One consequence: a tool that + /// raises for its own internal timeout will no longer be + /// retried by this helper either. + /// public static async Task<(AgentResponse? Response, Exception? LastError)> RunAsync( Func> attempt, int maxRetries, Func backoff) { @@ -34,6 +46,10 @@ public static bool HasToolError(AgentResponse response) => if (attemptNumber < maxRetries) await backoff(attemptNumber); } + catch (OperationCanceledException) + { + throw; // the caller asked to stop; retrying is not recovery + } catch (Exception ex) { lastError = ex; diff --git a/ExceptionHandlingAndRecovery.SemanticKernel/RetryAndFallbackFilter.cs b/ExceptionHandlingAndRecovery.SemanticKernel/RetryAndFallbackFilter.cs index 0d7d9ab..ab145e1 100644 --- a/ExceptionHandlingAndRecovery.SemanticKernel/RetryAndFallbackFilter.cs +++ b/ExceptionHandlingAndRecovery.SemanticKernel/RetryAndFallbackFilter.cs @@ -13,6 +13,11 @@ public RetryAndFallbackFilter(ILogger logger) _logger = logger; } + /// + /// Whole-turn retry replays every tool call the turn made. Only use it for turns whose tools are + /// read-only or idempotent; a turn that issues a refund must retry at the tool boundary with an + /// idempotency key instead (see IdempotentToolCalls). + /// public async Task OnFunctionInvocationAsync( FunctionInvocationContext context, Func next) { @@ -38,6 +43,12 @@ public async Task OnFunctionInvocationAsync( _logger.LogInformation("[Recovery] {Function} succeeded on attempt {Attempt}", functionName, attempt); return; } + catch (OperationCanceledException) when (context.CancellationToken.IsCancellationRequested) + { + // the caller asked to stop; retrying (and then falling back to a different + // function) is not recovery + throw; + } catch (Exception ex) { _logger.LogWarning( diff --git a/PatternExplorer/patterns/ExceptionHandlingAndRecovery.md b/PatternExplorer/patterns/ExceptionHandlingAndRecovery.md index cab34a0..c1b41f1 100644 --- a/PatternExplorer/patterns/ExceptionHandlingAndRecovery.md +++ b/PatternExplorer/patterns/ExceptionHandlingAndRecovery.md @@ -1,7 +1,7 @@ --- { "title": "Exception Handling, Recovery, and Circuit Breaker", - "summary": "Retry transient faults, fail fast through an open dependency circuit, then degrade safely.", + "summary": "Retry transient faults on idempotent turns, fail fast through an open dependency circuit, then degrade safely — caller cancellation is never retried.", "category": "Production controls", "projects": [ { "flavor": "AgentFramework", "path": "ExceptionHandlingAndRecovery.AgentFramework" }, @@ -22,6 +22,13 @@ dependency, and degrade gracefully. A circuit breaker is distinct from **Bounded the breaker shares dependency health across calls through closed, open, and half-open states; a run budget limits one agent execution. +Both flavors retry the *whole turn*, which replays every tool call the turn made. That is only +safe when the turn's tools are read-only or idempotent, as `GetPreciseLocation` is here. A turn +that issues a refund must not be retried this way — retry at the tool boundary instead, with an +idempotency key, as **IdempotentToolCalls** does. And in both flavors, caller cancellation +(`OperationCanceledException`) is always rethrown, never retried or turned into a fallback — +the caller asked to stop, and retrying is not recovery. + ## When to use it - A tool depends on a network service, a rate-limited API, or anything with transient faults. From 2e0a099b1c4430427968738c620b107ff94f9c43 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 15:09:01 +0200 Subject: [PATCH 04/21] fix(tool-auth): capability commit point, unconditional amount validation, real approval Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- .../ProductionControlTests.cs | 122 ++++++++++++++++++ PatternExplorer/patterns/ToolAuthorization.md | 66 +++++++++- README.md | 2 +- .../AuthorizationDecision.cs | 39 ++++++ .../AuthorizedAIFunction.cs | 18 ++- ToolAuthorization.AgentFramework/Program.cs | 62 +++++++-- .../ToolAuthorizationPolicy.cs | 44 ++++++- 7 files changed, 327 insertions(+), 26 deletions(-) diff --git a/AgenticPatterns.Tests/ProductionControlTests.cs b/AgenticPatterns.Tests/ProductionControlTests.cs index 076b038..ee587b4 100644 --- a/AgenticPatterns.Tests/ProductionControlTests.cs +++ b/AgenticPatterns.Tests/ProductionControlTests.cs @@ -206,6 +206,128 @@ public void OneTimeGrantCannotBeReplayed() Assert.Equal(AuthorizationOutcome.Allowed, policy.Authorize(Principal, capability, "GetOrder", args).Outcome); Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "GetOrder", args).Outcome); } + + [Fact] + public void AVerifiedPreEffectFailureReleasesTheOneTimeCapability() + { + var policy = new ToolAuthorizationPolicy(Owners); + var capability = Grant("IssueRefund", maximum: 50m, oneTime: true); + var args = new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 25m }; + + Assert.Equal(AuthorizationOutcome.Allowed, policy.Authorize(Principal, capability, "IssueRefund", args).Outcome); + policy.Release(capability.Nonce); // the tool threw before doing anything, and the caller verified that + Assert.Equal(AuthorizationOutcome.Allowed, policy.Authorize(Principal, capability, "IssueRefund", args).Outcome); + } + + [Fact] + public void ACommittedOneTimeCapabilityCannotBeReused() + { + var policy = new ToolAuthorizationPolicy(Owners); + var capability = Grant("IssueRefund", maximum: 50m, oneTime: true); + var args = new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 25m }; + + Assert.Equal(AuthorizationOutcome.Allowed, policy.Authorize(Principal, capability, "IssueRefund", args).Outcome); + policy.Commit(capability.Nonce); + policy.Release(capability.Nonce); // a late release must not resurrect a committed capability + Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "IssueRefund", args).Outcome); + } + + [Fact] + public void ARefundWithoutAConfiguredMaximumStillRequiresAPositiveAmount() + { + var policy = new ToolAuthorizationPolicy(Owners); + var capability = Grant("IssueRefund", maximum: null); + + Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "IssueRefund", + new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = -5m }).Outcome); + Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "IssueRefund", + new AIFunctionArguments { ["orderId"] = "ORD-100" }).Outcome); + Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "IssueRefund", + new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = "not-money" }).Outcome); + } + + [Fact] + public void ApprovalIsAPendingRequestNotToolOutput() + { + var decision = new ToolAuthorizationPolicy(Owners).Authorize(Principal, Grant("IssueRefund", 50m), + "IssueRefund", new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 10_000m }); + + Assert.NotNull(decision.PendingApproval); + Assert.Equal("IssueRefund", decision.PendingApproval!.ToolName); + Assert.Equal(10_000m, decision.PendingApproval.Arguments["amount"]); + } + + [Fact] + public void PendingApprovalArgumentsAreSnapshottedNotAliased() + { + var arguments = new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 10_000m }; + var decision = new ToolAuthorizationPolicy(Owners).Authorize(Principal, Grant("IssueRefund", 50m), + "IssueRefund", arguments); + arguments["amount"] = 1m; + + Assert.Equal(10_000m, decision.PendingApproval!.Arguments["amount"]); + } + + [Fact] + public void ConcurrentAuthorizeReservesAOneTimeNonceExactlyOnce() + { + var policy = new ToolAuthorizationPolicy(Owners); + var capability = Grant("GetOrder", oneTime: true); + var allowed = 0; + + Parallel.For(0, 64, _ => + { + if (policy.Authorize(Principal, capability, "GetOrder", + new AIFunctionArguments { ["orderId"] = "ORD-100" }).Outcome == AuthorizationOutcome.Allowed) + Interlocked.Increment(ref allowed); + }); + + Assert.Equal(1, allowed); + } + + [Fact] + public async Task TheWrapperCommitsTheReservationOnlyAfterTheToolReturns() + { + var policy = new ToolAuthorizationPolicy(Owners); + var capability = Grant("GetOrder", oneTime: true); + var args = new AIFunctionArguments { ["orderId"] = "ORD-100" }; + var tool = new AuthorizedAIFunction( + AIFunctionFactory.Create((string orderId) => $"{orderId}: ok", "GetOrder"), Principal, capability, policy); + + Assert.Equal("ORD-100: ok", (await tool.InvokeAsync(args))?.ToString()); + policy.Release(capability.Nonce); // committed already, so this cannot hand the capability back + await Assert.ThrowsAsync(() => tool.InvokeAsync(args).AsTask()); + } + + [Fact] + public async Task AnUnverifiedToolFailureDoesNotReleaseTheReservation() + { + var policy = new ToolAuthorizationPolicy(Owners); + var capability = Grant("GetOrder", oneTime: true); + var args = new AIFunctionArguments { ["orderId"] = "ORD-100" }; + var tool = new AuthorizedAIFunction( + AIFunctionFactory.Create((Func)Boom, "GetOrder"), Principal, capability, policy); + + await Assert.ThrowsAsync(() => tool.InvokeAsync(args).AsTask()); + // Failing closed: from inside the wrapper the failure is unverified, so the reservation stands. + await Assert.ThrowsAsync(() => tool.InvokeAsync(args).AsTask()); + static string Boom(string orderId) => throw new InvalidOperationException("simulated tool failure"); + } + + [Fact] + public async Task TheWrapperNeverHandsAnApprovalRequestBackAsToolOutput() + { + var policy = new ToolAuthorizationPolicy(Owners); + var tool = new AuthorizedAIFunction( + AIFunctionFactory.Create((string orderId, decimal amount) => "refunded", "IssueRefund"), + Principal, Grant("IssueRefund", 50m), policy); + + var failure = await Assert.ThrowsAsync(() => + tool.InvokeAsync(new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 500m }).AsTask()); + + Assert.Equal(AuthorizationOutcome.ApprovalRequired, failure.Decision.Outcome); + Assert.NotNull(failure.Decision.PendingApproval); + } } public class IdempotentToolCallTests diff --git a/PatternExplorer/patterns/ToolAuthorization.md b/PatternExplorer/patterns/ToolAuthorization.md index 6acb2e9..f9f2cde 100644 --- a/PatternExplorer/patterns/ToolAuthorization.md +++ b/PatternExplorer/patterns/ToolAuthorization.md @@ -1,7 +1,7 @@ --- { "title": "Tool Authorization / Capability Scoping", - "summary": "Authorize each concrete tool invocation against caller, tenant, resource, amount, expiry, and replay constraints.", + "summary": "Authorize each concrete tool invocation against caller, tenant, resource, amount, expiry, and replay constraints, reserving a one-time capability until the effect is committed.", "category": "Production controls", "projects": [ { "flavor": "AgentFramework", "path": "ToolAuthorization.AgentFramework" } ] } @@ -34,7 +34,21 @@ A customer-support principal receives separate short-lived capabilities for `Get `IssueRefund`. `AuthorizedAIFunction` intercepts the concrete arguments before calling the inner function. The policy normalizes order IDs, checks tenant and ownership, enforces the exact tool name, denies missing or malformed authorization inputs, turns refunds over €50 into -`ApprovalRequired`, and can consume one-time nonces. +`ApprovalRequired`, and reserves one-time nonces. + +Two design choices carry as much weight as the checks themselves. + +**A refusal is not a tool result.** `AuthorizedAIFunction` throws `ToolAuthorizationException` +rather than returning `"ApprovalRequired: ..."` as the function's output. Handing a refusal back on +the tool channel gives the model a sentence to paraphrase — frequently into a claim that the work +was done — and puts an approval request somewhere the model can answer. The host catches the +exception, routes `decision.PendingApproval` to a human, and decides what the model is told. The +`PendingApproval` carries a *snapshot* of the arguments, not the caller's live dictionary, so the +approver judges the values that were actually authorized. + +**The amount check does not depend on the grant carrying a ceiling.** `IssueRefund` is a +money-moving tool: an absent, negative, or unparseable `amount` is refused whether or not +`MaximumAmount` is set. A configured maximum only adds the ceiling on top of that floor. ```mermaid flowchart LR @@ -43,7 +57,7 @@ flowchart LR C[Host-created capability] --> A W --> A A -->|allowed| T[Real tool] - A -->|high amount| H[Approval required] + A -->|high amount| H[PendingApproval → human channel] A -->|wrong tenant/resource/tool
expired or replayed| D[Denied] ``` @@ -51,15 +65,53 @@ flowchart LR capability cannot authorize another tool. Those are separate from the argument-level denial when the current customer asks for someone else's order. +## One-time capabilities: reserve, then commit + +This sample demonstrates **reserve/commit**, not the idempotency-key alternative — the two solve +the same replay problem and shipping both would just be two half-enforced ledgers. +(`IdempotentToolCalls` is where the idempotency-key design lives, and it keeps its dedup record +with the side effect, which is the right home for it.) + +`Authorize` moves a one-time nonce `Available -> Reserved`. `Commit` moves it `Reserved -> +Consumed` once the effect is durable; `Release` moves it back to `Available` after a *verified* +pre-effect failure. Burning the nonce inside `Authorize`, as an earlier version did, destroyed a +valid capability whenever the tool failed before doing anything. + +`AuthorizedAIFunction` commits after the inner call returns and deliberately does **not** release +when it throws: from inside the wrapper the failure is unverified — the effect may well have +happened — so the reservation stands and the capability fails closed. `Release` stays a caller- +driven act for a failure the caller has confirmed was pre-effect. + +The honest gap: **nothing here resolves a reservation whose commit never arrives.** If the host +crashes between reserve and commit, the capability is stranded in `Reserved` forever, and because +the ledger is an in-process `ConcurrentDictionary` a restart forgets it instead. A real system +gives a reservation a lease with an expiry and a sweeper that decides — by asking the downstream +system whether the effect landed, not by guessing — whether an expired reservation becomes +`Consumed` or `Available`. Better still, it stores the three states in the same transactional store +that owns the side effect, so the commit is atomic with the effect and the question never arises. + ## Key APIs - `DelegatingAIFunction.InvokeCoreAsync(...)` — the final host-side enforcement point. - `ToolCapability` — immutable subject, tenant, exact tool, resource, amount, expiry, and nonce. -- `ToolAuthorizationPolicy.Authorize(...)` — returns `Allowed`, `Denied`, or `ApprovalRequired`. +- `ToolAuthorizationPolicy.Authorize(...)` — returns `Allowed`, `Denied`, or `ApprovalRequired`, and + reserves a one-time capability rather than consuming it. +- `ToolAuthorizationPolicy.Commit(nonce)` / `Release(nonce)` — the two ends of the reservation. +- `CapabilityState` — `Available`, `Reserved`, `Consumed`. +- `AuthorizationDecision.PendingApproval` — tool name, argument snapshot, and reason for the human + channel. +- `ToolAuthorizationException` — how a refusal leaves the tool-result channel. - `RunPrincipal` — authenticated identity supplied by the application, never by the prompt. ## What to watch in the output -The five probes show: own order allowed, another customer's order denied, €25 refund allowed, -€500 refund escalated, and a tool absent from the grant denied. Each decision is printed before -the underlying function can run. +The probes show: own order allowed, another customer's order refused, €25 refund allowed, and a +€500 refund escalated — printed as an out-of-band approval request, then executed only after an +approver mints a *new* single-use capability sized to that exact request. The original capability +is never widened, and the model plays no part in producing the new grant. A one-time `GetOrder` +capability then succeeds once and is refused on replay, and a tool absent from the grant is +refused. Each decision is printed before the underlying function can run. + +The approver's answer in the sample is a constant, marked `// ponytail:` — the sample must run +unattended, so it cannot block on `Console.ReadLine`. `DurableHumanInTheLoop` is where the real +version of that wait lives. diff --git a/README.md b/README.md index 84e695d..93df299 100644 --- a/README.md +++ b/README.md @@ -152,7 +152,7 @@ the catalog together; each result states its scope limits and cites a primary so | IdempotentToolCalls | Retry-safe side effects: the dedup record lives with the side effect, not the caller | | Middleware | Agent-run and function-invocation middleware (logging, latency, tool guards) | | ResourceAwareOptimization | Model routing under a cost budget | -| ToolAuthorization | Capability-scoped, argument-level authorization before tool execution | +| ToolAuthorization | Capability-scoped, argument-level authorization before tool execution; one-time grants are reserved, then committed after the effect | ### Evaluation diff --git a/ToolAuthorization.AgentFramework/AuthorizationDecision.cs b/ToolAuthorization.AgentFramework/AuthorizationDecision.cs index 45bbeca..d0d4422 100644 --- a/ToolAuthorization.AgentFramework/AuthorizationDecision.cs +++ b/ToolAuthorization.AgentFramework/AuthorizationDecision.cs @@ -1,10 +1,49 @@ +using System.Collections.Frozen; + namespace ToolAuthorization.AgentFramework; public enum AuthorizationOutcome { Allowed, Denied, ApprovalRequired } +/// +/// A request for a human decision. It travels on the host's approval channel, never back to the +/// model as tool output. is a snapshot: the caller's live argument +/// dictionary stays mutable, so an approver must see the values that were actually judged. +/// +public sealed record PendingApproval +{ + public PendingApproval(string toolName, IEnumerable> arguments, string reason) + { + ToolName = toolName; + Arguments = arguments.ToFrozenDictionary(StringComparer.Ordinal); + Reason = reason; + } + + public string ToolName { get; } + public IReadOnlyDictionary Arguments { get; } + public string Reason { get; } +} + public sealed record AuthorizationDecision(AuthorizationOutcome Outcome, string Reason) { + /// Set only when is . + public PendingApproval? PendingApproval { get; init; } + public static AuthorizationDecision Allow() => new(AuthorizationOutcome.Allowed, "Capability permits this invocation."); public static AuthorizationDecision Deny(string reason) => new(AuthorizationOutcome.Denied, reason); public static AuthorizationDecision RequireApproval(string reason) => new(AuthorizationOutcome.ApprovalRequired, reason); + + public static AuthorizationDecision RequireApproval(string reason, string toolName, + IEnumerable> arguments) => + new(AuthorizationOutcome.ApprovalRequired, reason) { PendingApproval = new(toolName, arguments, reason) }; +} + +/// +/// Thrown when a tool invocation is not allowed. Refusals leave the tool-result channel entirely: +/// the host decides what, if anything, the model is told, and an approval request reaches a human +/// instead of becoming text the model can argue with or paraphrase into a false success. +/// +public sealed class ToolAuthorizationException(AuthorizationDecision decision) + : InvalidOperationException($"{decision.Outcome}: {decision.Reason}") +{ + public AuthorizationDecision Decision { get; } = decision; } diff --git a/ToolAuthorization.AgentFramework/AuthorizedAIFunction.cs b/ToolAuthorization.AgentFramework/AuthorizedAIFunction.cs index 246234b..1b0ce6b 100644 --- a/ToolAuthorization.AgentFramework/AuthorizedAIFunction.cs +++ b/ToolAuthorization.AgentFramework/AuthorizedAIFunction.cs @@ -13,8 +13,20 @@ public sealed class AuthorizedAIFunction( { var decision = policy.Authorize(principal, capability, Name, arguments); Console.WriteLine($" [authorization] {Name}: {decision.Outcome} — {decision.Reason}"); - return decision.Outcome == AuthorizationOutcome.Allowed - ? await base.InvokeCoreAsync(arguments, cancellationToken) - : $"{decision.Outcome}: {decision.Reason}"; + + // A refusal is not a tool result. Returning "ApprovalRequired: ..." as text hands the model + // a sentence to paraphrase — often into a claim that the work was done — and puts the + // approval request on the channel the model controls. The host gets an exception instead. + if (decision.Outcome != AuthorizationOutcome.Allowed) + throw new ToolAuthorizationException(decision); + + var result = await base.InvokeCoreAsync(arguments, cancellationToken); + + // Commit only after the inner call returns. There is deliberately no Release on the + // exception path: from inside this wrapper the failure is unverified — the effect may + // already have happened — so the reservation stands and the capability fails closed. + // Release is a caller-driven act for a failure the caller has verified was pre-effect. + policy.Commit(capability.Nonce); + return result; } } diff --git a/ToolAuthorization.AgentFramework/Program.cs b/ToolAuthorization.AgentFramework/Program.cs index 54df746..1afd6b4 100644 --- a/ToolAuthorization.AgentFramework/Program.cs +++ b/ToolAuthorization.AgentFramework/Program.cs @@ -9,21 +9,60 @@ }; var policy = new ToolAuthorizationPolicy(orderOwners); -ToolCapability Grant(string tool, decimal? maximumAmount = null) => new( +ToolCapability Grant(string tool, decimal? maximumAmount = null, bool oneTime = false) => new( principal.SubjectId, principal.TenantId, tool, new Dictionary(), maximumAmount, - DateTimeOffset.UtcNow.AddMinutes(5), Guid.NewGuid().ToString("N")); + DateTimeOffset.UtcNow.AddMinutes(5), Guid.NewGuid().ToString("N"), oneTime); -var getOrder = new AuthorizedAIFunction( - AIFunctionFactory.Create((string orderId) => $"{orderId}: paid, awaiting shipment", "GetOrder"), - principal, Grant("GetOrder"), policy); -var issueRefund = new AuthorizedAIFunction( - AIFunctionFactory.Create((string orderId, decimal amount) => $"Refunded €{amount:F2} for {orderId}", "IssueRefund"), - principal, Grant("IssueRefund", maximumAmount: 50m), policy); +var getOrderFunction = AIFunctionFactory.Create( + (string orderId) => $"{orderId}: paid, awaiting shipment", "GetOrder"); +var refundFunction = AIFunctionFactory.Create( + (string orderId, decimal amount) => $"Refunded €{amount:F2} for {orderId}", "IssueRefund"); + +var getOrder = new AuthorizedAIFunction(getOrderFunction, principal, Grant("GetOrder"), policy); +var issueRefund = new AuthorizedAIFunction(refundFunction, principal, Grant("IssueRefund", maximumAmount: 50m), policy); async Task Show(AIFunction tool, AIFunctionArguments arguments) { Console.WriteLine($"\n{tool.Name}({string.Join(", ", arguments.Select(a => $"{a.Key}={a.Value}"))})"); - Console.WriteLine($" result: {await tool.InvokeAsync(arguments)}"); + try + { + Console.WriteLine($" result: {await tool.InvokeAsync(arguments)}"); + } + catch (ToolAuthorizationException ex) + { + // The model never sees any of this: a refusal is a host event, not a tool result. + if (ex.Decision.PendingApproval is not { } pending) + { + Console.WriteLine($" refused: {ex.Decision.Reason}"); + return; + } + await Escalate(pending); + } +} + +async Task Escalate(PendingApproval pending) +{ + Console.WriteLine($" approval required: {pending.Reason}"); + Console.WriteLine(" → sent to the approver's channel, not returned to the model:"); + Console.WriteLine($" {pending.ToolName}({string.Join(", ", pending.Arguments.Select(a => $"{a.Key}={a.Value}"))})"); + + // ponytail: the approver's answer is a constant so the sample runs with no TTY and no + // credentials. Upgrade path: await a durable approval record (see DurableHumanInTheLoop) and + // resume from it — never Console.ReadLine, which would hang an unattended run. + var approverApproved = true; + if (!approverApproved) + { + Console.WriteLine(" approver declined — the tool is never invoked."); + return; + } + + // The approver issues a *new* capability, sized to this one request and single-use. The + // original capability is never widened, and the model had no part in producing this grant. + // The sample has exactly one escalating tool, so the inner function is named directly. + var approved = new AuthorizedAIFunction(refundFunction, principal, + Grant(pending.ToolName, maximumAmount: 500m, oneTime: true), policy); + Console.WriteLine(" approver granted a single-use capability for this exact request."); + Console.WriteLine($" result: {await approved.InvokeAsync(new AIFunctionArguments(pending.Arguments.ToDictionary()))}"); } Console.WriteLine("=== Capability-scoped tools ==="); @@ -32,6 +71,11 @@ async Task Show(AIFunction tool, AIFunctionArguments arguments) await Show(issueRefund, new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 25m }); await Show(issueRefund, new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 500m }); +Console.WriteLine("\n=== One-time capability: reserve, then commit ==="); +var oneTimeGetOrder = new AuthorizedAIFunction(getOrderFunction, principal, Grant("GetOrder", oneTime: true), policy); +await Show(oneTimeGetOrder, new AIFunctionArguments { ["orderId"] = "ORD-100" }); // reserved, then committed +await Show(oneTimeGetOrder, new AIFunctionArguments { ["orderId"] = "ORD-100" }); // refused: already consumed + var missingGrant = policy.Authorize(principal, Grant("GetOrder"), "GetInternalFraudScore", []); Console.WriteLine($"\nGetInternalFraudScore: {missingGrant.Outcome} — {missingGrant.Reason}"); Console.WriteLine("DeleteCustomer: not registered; the model never receives its definition."); diff --git a/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs b/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs index 2a88deb..0ce81ad 100644 --- a/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs +++ b/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs @@ -5,12 +5,25 @@ namespace ToolAuthorization.AgentFramework; +/// +/// The lifecycle of a one-time capability. reserves; the host commits once +/// the effect is durable, or releases after a verified pre-effect failure. +/// +public enum CapabilityState { Available, Reserved, Consumed } + public sealed class ToolAuthorizationPolicy( IReadOnlyDictionary orderOwners, TimeProvider? timeProvider = null) { + /// Tools that move money always need a present, positive, parseable amount. + private static readonly HashSet MoneyMovingTools = new(StringComparer.Ordinal) { "IssueRefund" }; + private readonly TimeProvider _timeProvider = timeProvider ?? TimeProvider.System; - private readonly ConcurrentDictionary _usedNonces = new(StringComparer.Ordinal); + + // ponytail: in-process nonce ledger, so a restart forgets every reservation and a crashed host + // leaks one. Upgrade path: the same three states in the transactional store that already owns + // the side effect, which is where the commit becomes atomic with the effect. + private readonly ConcurrentDictionary _nonceStates = new(StringComparer.Ordinal); public AuthorizationDecision Authorize(RunPrincipal principal, ToolCapability capability, string toolName, AIFunctionArguments arguments) @@ -33,7 +46,10 @@ public AuthorizationDecision Authorize(RunPrincipal principal, ToolCapability ca return AuthorizationDecision.Deny("Order is outside the capability resource scope."); } - if (capability.MaximumAmount is { } maximum) + // The amount is validated whenever the tool moves money, not only when the grant happens to + // carry a ceiling: an absent, negative, or unparseable amount is never authorizable. A + // configured maximum only adds the ceiling on top of that. + if (MoneyMovingTools.Contains(toolName) || capability.MaximumAmount is not null) { decimal? amount; try { amount = ReadDecimal(arguments, "amount"); } @@ -43,16 +59,32 @@ public AuthorizationDecision Authorize(RunPrincipal principal, ToolCapability ca } if (amount is null or <= 0) return AuthorizationDecision.Deny("A valid amount is required for authorization."); - if (amount > maximum) - return AuthorizationDecision.RequireApproval($"Amount €{amount:F2} exceeds the capability limit of €{maximum:F2}."); + if (capability.MaximumAmount is { } maximum && amount > maximum) + return AuthorizationDecision.RequireApproval( + $"Amount €{amount:F2} exceeds the capability limit of €{maximum:F2}.", toolName, arguments); } - if (capability.OneTimeUse && !_usedNonces.TryAdd(capability.Nonce, 0)) - return AuthorizationDecision.Deny("One-time capability has already been used."); + // Reserve last, so a refusal above never costs the caller a one-time capability. + if (capability.OneTimeUse && !TryReserve(capability.Nonce)) + return AuthorizationDecision.Deny("One-time capability is already reserved or consumed."); return AuthorizationDecision.Allow(); } + /// Call once the effect is durable. Moves Reserved -> Consumed. + public void Commit(string nonce) => _nonceStates.TryUpdate(nonce, CapabilityState.Consumed, CapabilityState.Reserved); + + /// + /// Call after a verified pre-effect failure — the caller must know the side effect did + /// not happen. Moves Reserved -> Available; a committed capability stays consumed. + /// + public void Release(string nonce) => _nonceStates.TryUpdate(nonce, CapabilityState.Available, CapabilityState.Reserved); + + /// Atomic Available -> Reserved; two racing callers cannot both win. + private bool TryReserve(string nonce) => + _nonceStates.TryAdd(nonce, CapabilityState.Reserved) || + _nonceStates.TryUpdate(nonce, CapabilityState.Reserved, CapabilityState.Available); + private static string? ReadString(AIFunctionArguments arguments, string name) => arguments.TryGetValue(name, out var value) ? value switch { From 438ecdd6ad6001e5b88a03b9b781c151b20e90d5 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 15:29:09 +0200 Subject: [PATCH 05/21] test(tool-auth): gate the race and the two lifecycle orderings, name the money-tool ceiling The concurrency test raced a single nonce once, which a read-then-write reserve survives 97-100% of the time; it now repeats 5000 trials on a fresh nonce and catches the reverted implementation 25/25. Commit-after-invoke and reserve-after-checks were asserted only in comments: Reserved and Consumed deny identically, so no test could tell them apart. Both now have an assertion that goes red when the ordering moves. MoneyMovingTools gets the ponytail comment naming its silent ceiling, the approver's grant is sized from the snapshot instead of a hard-coded figure and guarded so an unattended run cannot die on it, and both the sample and the doc now say that throwing keeps a refusal off the tool channel only because this host invokes the function directly. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- .../ProductionControlTests.cs | 47 +++++++++++++++---- PatternExplorer/patterns/ToolAuthorization.md | 8 +++- ToolAuthorization.AgentFramework/Program.cs | 29 +++++++++--- .../ToolAuthorizationPolicy.cs | 5 ++ 4 files changed, 73 insertions(+), 16 deletions(-) diff --git a/AgenticPatterns.Tests/ProductionControlTests.cs b/AgenticPatterns.Tests/ProductionControlTests.cs index ee587b4..94e2d6f 100644 --- a/AgenticPatterns.Tests/ProductionControlTests.cs +++ b/AgenticPatterns.Tests/ProductionControlTests.cs @@ -272,17 +272,41 @@ public void PendingApprovalArgumentsAreSnapshottedNotAliased() public void ConcurrentAuthorizeReservesAOneTimeNonceExactlyOnce() { var policy = new ToolAuthorizationPolicy(Owners); - var capability = Grant("GetOrder", oneTime: true); - var allowed = 0; - Parallel.For(0, 64, _ => + // One trial is not a gate: a read-then-write reserve wins the race the overwhelming + // majority of the time, and is *most* likely to win on a loaded machine, which is exactly + // how xUnit runs this. Repeat with a fresh nonce per trial so one lost race fails the test. + for (var trial = 0; trial < 5000; trial++) { - if (policy.Authorize(Principal, capability, "GetOrder", - new AIFunctionArguments { ["orderId"] = "ORD-100" }).Outcome == AuthorizationOutcome.Allowed) - Interlocked.Increment(ref allowed); - }); + var capability = Grant("GetOrder", oneTime: true); + var allowed = 0; + + Parallel.For(0, 4, _ => + { + if (policy.Authorize(Principal, capability, "GetOrder", + new AIFunctionArguments { ["orderId"] = "ORD-100" }).Outcome == AuthorizationOutcome.Allowed) + Interlocked.Increment(ref allowed); + }); - Assert.Equal(1, allowed); + Assert.Equal(1, allowed); + } + } + + [Fact] + public void ARefusedInvocationDoesNotBurnTheOneTimeCapability() + { + var policy = new ToolAuthorizationPolicy(Owners); + var capability = Grant("IssueRefund", maximum: 50m, oneTime: true); + + Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "IssueRefund", + new AIFunctionArguments { ["orderId"] = "ORD-OTHER-TENANT", ["amount"] = 25m }).Outcome); + Assert.Equal(AuthorizationOutcome.ApprovalRequired, policy.Authorize(Principal, capability, "IssueRefund", + new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 500m }).Outcome); + + // The reserve happens after every check, so neither refusal cost the caller the capability: + // the approved retry uses the very same grant. + Assert.Equal(AuthorizationOutcome.Allowed, policy.Authorize(Principal, capability, "IssueRefund", + new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 25m }).Outcome); } [Fact] @@ -311,6 +335,13 @@ public async Task AnUnverifiedToolFailureDoesNotReleaseTheReservation() await Assert.ThrowsAsync(() => tool.InvokeAsync(args).AsTask()); // Failing closed: from inside the wrapper the failure is unverified, so the reservation stands. await Assert.ThrowsAsync(() => tool.InvokeAsync(args).AsTask()); + + // ...and it is still merely Reserved, never Consumed. This is what pins Commit to its + // position *after* the inner call: committing first would deny identically here while + // leaving a capability that threw pre-effect permanently unreleasable. + policy.Release(capability.Nonce); + Assert.Equal(AuthorizationOutcome.Allowed, + policy.Authorize(Principal, capability, "GetOrder", args).Outcome); static string Boom(string orderId) => throw new InvalidOperationException("simulated tool failure"); } diff --git a/PatternExplorer/patterns/ToolAuthorization.md b/PatternExplorer/patterns/ToolAuthorization.md index f9f2cde..ff812b2 100644 --- a/PatternExplorer/patterns/ToolAuthorization.md +++ b/PatternExplorer/patterns/ToolAuthorization.md @@ -42,7 +42,13 @@ Two design choices carry as much weight as the checks themselves. rather than returning `"ApprovalRequired: ..."` as the function's output. Handing a refusal back on the tool channel gives the model a sentence to paraphrase — frequently into a claim that the work was done — and puts an approval request somewhere the model can answer. The host catches the -exception, routes `decision.PendingApproval` to a human, and decides what the model is told. The +exception, routes `decision.PendingApproval` to a human, and decides what the model is told. + +Throwing is necessary but not by itself sufficient: it keeps the refusal off the tool channel only +because this host invokes the function directly. Run the same wrapper under +`FunctionInvokingChatClient` — the loop the diagram above implies — and the framework catches the +function's exception and feeds the model a generic error, discarding the `PendingApproval` entirely. +A host on that path has to intercept the exception before the invocation loop does. The `PendingApproval` carries a *snapshot* of the arguments, not the caller's live dictionary, so the approver judges the values that were actually authorized. diff --git a/ToolAuthorization.AgentFramework/Program.cs b/ToolAuthorization.AgentFramework/Program.cs index 1afd6b4..27567f4 100644 --- a/ToolAuthorization.AgentFramework/Program.cs +++ b/ToolAuthorization.AgentFramework/Program.cs @@ -1,3 +1,4 @@ +using System.Globalization; using Microsoft.Extensions.AI; using ToolAuthorization.AgentFramework; @@ -30,7 +31,10 @@ async Task Show(AIFunction tool, AIFunctionArguments arguments) } catch (ToolAuthorizationException ex) { - // The model never sees any of this: a refusal is a host event, not a tool result. + // The model never sees any of this — because this host invokes the function directly. + // Throwing alone does not guarantee it: under FunctionInvokingChatClient the framework + // catches a function exception and feeds the model a generic error, discarding the + // PendingApproval. A host on that path must intercept before the loop does. if (ex.Decision.PendingApproval is not { } pending) { Console.WriteLine($" refused: {ex.Decision.Reason}"); @@ -56,13 +60,24 @@ async Task Escalate(PendingApproval pending) return; } - // The approver issues a *new* capability, sized to this one request and single-use. The - // original capability is never widened, and the model had no part in producing this grant. - // The sample has exactly one escalating tool, so the inner function is named directly. + // The approver issues a *new* capability, single-use and sized to this one request — the + // ceiling is read off the snapshot the approver saw, not hard-coded, so it stays correct when + // the probe changes. The original capability is never widened, and the model had no part in + // producing this grant. The sample has exactly one escalating tool, so the inner function is + // named directly. + var requested = Convert.ToDecimal(pending.Arguments["amount"], CultureInfo.InvariantCulture); var approved = new AuthorizedAIFunction(refundFunction, principal, - Grant(pending.ToolName, maximumAmount: 500m, oneTime: true), policy); - Console.WriteLine(" approver granted a single-use capability for this exact request."); - Console.WriteLine($" result: {await approved.InvokeAsync(new AIFunctionArguments(pending.Arguments.ToDictionary()))}"); + Grant(pending.ToolName, maximumAmount: requested, oneTime: true), policy); + Console.WriteLine($" approver granted a single-use capability capped at €{requested:F2}."); + try + { + Console.WriteLine($" result: {await approved.InvokeAsync(new AIFunctionArguments(pending.Arguments.ToDictionary()))}"); + } + catch (ToolAuthorizationException ex) + { + // The approver's own grant is enforced too, and the sample must run unattended to the end. + Console.WriteLine($" approved call still refused: {ex.Decision.Reason}"); + } } Console.WriteLine("=== Capability-scoped tools ==="); diff --git a/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs b/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs index 0ce81ad..b054e60 100644 --- a/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs +++ b/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs @@ -16,6 +16,11 @@ public sealed class ToolAuthorizationPolicy( TimeProvider? timeProvider = null) { /// Tools that move money always need a present, positive, parseable amount. + // ponytail: "which tools move money" is a hand-maintained set in the policy rather than a + // property of the tool registration. The ceiling is a silent one: register `IssueCredit`, grant + // it without a MaximumAmount, forget this line, and it gets no amount floor at all — absent and + // negative amounts authorize. Upgrade path: move the flag onto the capability/tool registration + // so adding a money tool cannot compile without declaring it. private static readonly HashSet MoneyMovingTools = new(StringComparer.Ordinal) { "IssueRefund" }; private readonly TimeProvider _timeProvider = timeProvider ?? TimeProvider.System; From 3ba7171ec32151e40b78c5d07f142035989be118 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 15:42:20 +0200 Subject: [PATCH 06/21] fix(react): enforce the tool-call bound in the host, not the prompt Add ToolCallBudgetFilter, a Semantic Kernel IFunctionInvocationFilter that throws once the 10-call budget is exhausted, so the call that would exceed it never runs. Registered on ReasoningAndActing's local kernel; the top-level catch prints a PARTIAL result in the same shape BoundedExecution uses. The "at most 10 tool calls" prompt sentence is now documented as a hint, not the control. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- .../AgenticPatterns.Tests.csproj | 1 + .../ToolCallBudgetFilterTests.cs | 65 +++++++++++++++++++ .../patterns/ReasoningAndActing.md | 19 ++++-- ReasoningAndActing/Program.cs | 27 ++++++-- ReasoningAndActing/ToolCallBudgetFilter.cs | 33 ++++++++++ 5 files changed, 135 insertions(+), 10 deletions(-) create mode 100644 AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs create mode 100644 ReasoningAndActing/ToolCallBudgetFilter.cs diff --git a/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj b/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj index 8851ea2..162987f 100644 --- a/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj +++ b/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj @@ -35,6 +35,7 @@ + diff --git a/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs b/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs new file mode 100644 index 0000000..8db01d3 --- /dev/null +++ b/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs @@ -0,0 +1,65 @@ +using ReasoningAndActing; +using Xunit; + +namespace AgenticPatterns.Tests; + +public class ToolCallBudgetFilterTests +{ + [Fact] + public async Task TenthCallIsAllowed_EleventhThrows() + { + var filter = new ToolCallBudgetFilter(); + var calls = 0; + + for (var i = 0; i < ToolCallBudgetFilter.MaxToolCalls; i++) + await filter.GuardAsync(() => + { + calls++; + return Task.CompletedTask; + }); + + Assert.Equal(ToolCallBudgetFilter.MaxToolCalls, calls); + + await Assert.ThrowsAsync(() => + filter.GuardAsync(() => + { + calls++; + return Task.CompletedTask; + })); + + // The 11th call must not have reached the inner delegate. + Assert.Equal(ToolCallBudgetFilter.MaxToolCalls, calls); + } + + [Fact] + public async Task OverBudgetCall_DoesNotInvokeInnerDelegate() + { + var filter = new ToolCallBudgetFilter(); + for (var i = 0; i < ToolCallBudgetFilter.MaxToolCalls; i++) + await filter.GuardAsync(() => Task.CompletedTask); + + var innerInvoked = false; + + await Assert.ThrowsAsync(() => + filter.GuardAsync(() => + { + innerInvoked = true; + return Task.CompletedTask; + })); + + Assert.False(innerInvoked); + } + + [Fact] + public async Task ExceptionMessage_NamesTheBudget() + { + var filter = new ToolCallBudgetFilter(); + for (var i = 0; i < ToolCallBudgetFilter.MaxToolCalls; i++) + await filter.GuardAsync(() => Task.CompletedTask); + + var ex = await Assert.ThrowsAsync(() => + filter.GuardAsync(() => Task.CompletedTask)); + + Assert.Contains(ToolCallBudgetFilter.MaxToolCalls.ToString(), ex.Message); + } +} diff --git a/PatternExplorer/patterns/ReasoningAndActing.md b/PatternExplorer/patterns/ReasoningAndActing.md index 21f752d..7583897 100644 --- a/PatternExplorer/patterns/ReasoningAndActing.md +++ b/PatternExplorer/patterns/ReasoningAndActing.md @@ -54,9 +54,13 @@ flowchart LR A --> R[Final answer with ratio] ``` -There is one honest wart worth knowing: SK 1.79 exposes no max-auto-invoke setting on -`FunctionChoiceBehavior`, so the loop is capped by a sentence in the system prompt — *"Use at -most 10 tool calls before giving your final answer."* A prompt is a request, not a guardrail. +SK 1.79 exposes no max-auto-invoke setting on `FunctionChoiceBehavior`, so the system prompt still +carries *"Use at most 10 tool calls before giving your final answer."* — but that sentence is only +a hint the model can ignore. The actual control is `ToolCallBudgetFilter`, an +`IFunctionInvocationFilter` registered on the local kernel: it counts every tool call and throws +once the 11th would run, so the call that would exceed the budget never executes. The top-level +`try`/`catch` turns that exception into a `PARTIAL` result — the same shape **Bounded Execution** +uses for its own hard stops — instead of letting the process crash mid-answer. ## Key APIs @@ -66,11 +70,16 @@ most 10 tool calls before giving your final answer."* A prompt is a request, not - `new OpenAIPromptExecutionSettings { FunctionChoiceBehavior = FunctionChoiceBehavior.Auto() }`. - `chatService.GetChatMessageContentAsync(history, settings, kernel)` — the kernel argument is what makes auto-invocation possible. +- `ToolCallBudgetFilter : IFunctionInvocationFilter` — the host-enforced call cap; see + **Bounded Execution** for the fuller pattern of hard, host-enforced run limits. ## What to watch in the output The demo prints a single block prefixed `ReAct Agent:`. The tool calls themselves are not logged, so the tell is in the content: the answer should quote the exact simulated figures — 40.1 million and 26.5 million — and a ratio near 1.5, none of which the model could produce without calling -the plugin. **Tool Use** is the single-call foundation this loops over; **Middleware** shows how -to log every invocation so the reasoning-acting alternation becomes visible. +the plugin. If the model instead wanders past the tool-call budget, the output switches to +`Result status: PARTIAL` with a `Stop reason:` line naming the exhausted budget — proof the bound +stopped the loop rather than the model choosing to stop. **Tool Use** is the single-call +foundation this loops over; **Middleware** shows how to log every invocation so the +reasoning-acting alternation becomes visible. diff --git a/ReasoningAndActing/Program.cs b/ReasoningAndActing/Program.cs index 8499081..a148b3e 100644 --- a/ReasoningAndActing/Program.cs +++ b/ReasoningAndActing/Program.cs @@ -1,3 +1,4 @@ +using Microsoft.Extensions.DependencyInjection; using Microsoft.SemanticKernel; using Microsoft.SemanticKernel.ChatCompletion; using Microsoft.SemanticKernel.Connectors.OpenAI; @@ -5,7 +6,9 @@ using Shared; // Local kernel so the shared Settings.Kernel singleton stays unmodified -var reactKernel = Settings.CreateKernelBuilder().Build(); +var reactBuilder = Settings.CreateKernelBuilder(); +reactBuilder.Services.AddSingleton(); +var reactKernel = reactBuilder.Build(); // Register tools the agent can use mid-reasoning reactKernel.Plugins.AddFromType(); @@ -33,12 +36,26 @@ Use at most 10 tool calls before giving your final answer. var reactSettings = new OpenAIPromptExecutionSettings { // SK 1.79 exposes no max-auto-invoke option on FunctionChoiceBehavior/FunctionChoiceBehaviorOptions, - // so the tool-call loop is capped via the "at most 10 tool calls" instruction above. + // so the "at most 10 tool calls" instruction above is only a hint to the model. ToolCallBudgetFilter, + // registered on reactKernel above, is the actual control: it throws once the budget is exhausted, + // stopping the loop regardless of what the model intends to do next. FunctionChoiceBehavior = FunctionChoiceBehavior.Auto() }; // The agent will autonomously call GetPopulation() for each country, // then reason about the results to compute the ratio. -var reactResponse = await reactService.GetChatMessageContentAsync( - reactHistory, reactSettings, reactKernel); -Console.WriteLine($"ReAct Agent:\n{reactResponse.Content}\n"); \ No newline at end of file +try +{ + var reactResponse = await reactService.GetChatMessageContentAsync( + reactHistory, reactSettings, reactKernel); + Console.WriteLine($"ReAct Agent:\n{reactResponse.Content}\n"); +} +catch (InvalidOperationException ex) when (ex.Message.Contains("Tool-call budget")) +{ + // Same shape as BoundedExecution: a PARTIAL result, the stop reason, and an explicit + // incomplete label rather than silently truncated output. + Console.WriteLine("Result status: PARTIAL"); + Console.WriteLine($"Stop reason: {ex.Message}"); + Console.WriteLine("ReAct Agent:\nStopped at the tool-call budget before a final answer was reached; " + + "any reasoning gathered so far is incomplete.\n"); +} \ No newline at end of file diff --git a/ReasoningAndActing/ToolCallBudgetFilter.cs b/ReasoningAndActing/ToolCallBudgetFilter.cs new file mode 100644 index 0000000..87a8507 --- /dev/null +++ b/ReasoningAndActing/ToolCallBudgetFilter.cs @@ -0,0 +1,33 @@ +using Microsoft.SemanticKernel; + +namespace ReasoningAndActing; + +/// +/// Hard bound on tool calls: the prompt asks the model to stop at 10, this enforces it. The +/// counting/throwing logic lives in , a context-free core that a test can +/// drive directly — 's constructor is internal to Semantic +/// Kernel, so a test cannot build one to call itself. +/// +public class ToolCallBudgetFilter : IFunctionInvocationFilter +{ + public const int MaxToolCalls = 10; + + private int _toolCalls; + + /// + /// Increments the call counter and invokes only while under budget. + /// Throws before calling once the budget is exhausted, so the call + /// that would exceed it never runs. + /// + public async Task GuardAsync(Func next) + { + if (Interlocked.Increment(ref _toolCalls) > MaxToolCalls) + throw new InvalidOperationException($"Tool-call budget of {MaxToolCalls} exhausted."); + + await next(); + } + + public Task OnFunctionInvocationAsync( + FunctionInvocationContext context, Func next) => + GuardAsync(() => next(context)); +} From c14f112d7f1a313507ea985b0a326c41214002fd Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 15:51:43 +0200 Subject: [PATCH 07/21] fix(react): catch the tool-call budget by type, not by message text Review round 1: replace the InvalidOperationException message-substring match in Program.cs's catch with a dedicated ToolCallBudgetExceededException, matching BoundedExecution.AgentFramework's BudgetExceededException shape - a reworded message could no longer silently kill the catch. Tests now pin the exception type itself. Also fixes csproj ProjectReference ordering (the exclusion comment must sit with the project it follows) and tags the GuardAsync test seam with a ponytail comment naming its ceiling and upgrade path. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- AgenticPatterns.Tests/AgenticPatterns.Tests.csproj | 2 +- AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs | 6 +++--- ReasoningAndActing/Program.cs | 2 +- .../ToolCallBudgetExceededException.cs | 10 ++++++++++ ReasoningAndActing/ToolCallBudgetFilter.cs | 12 +++++++----- 5 files changed, 22 insertions(+), 10 deletions(-) create mode 100644 ReasoningAndActing/ToolCallBudgetExceededException.cs diff --git a/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj b/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj index 162987f..846cb1d 100644 --- a/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj +++ b/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj @@ -35,8 +35,8 @@ - + diff --git a/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs b/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs index 8db01d3..849ae57 100644 --- a/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs +++ b/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs @@ -20,7 +20,7 @@ await filter.GuardAsync(() => Assert.Equal(ToolCallBudgetFilter.MaxToolCalls, calls); - await Assert.ThrowsAsync(() => + await Assert.ThrowsAsync(() => filter.GuardAsync(() => { calls++; @@ -40,7 +40,7 @@ public async Task OverBudgetCall_DoesNotInvokeInnerDelegate() var innerInvoked = false; - await Assert.ThrowsAsync(() => + await Assert.ThrowsAsync(() => filter.GuardAsync(() => { innerInvoked = true; @@ -57,7 +57,7 @@ public async Task ExceptionMessage_NamesTheBudget() for (var i = 0; i < ToolCallBudgetFilter.MaxToolCalls; i++) await filter.GuardAsync(() => Task.CompletedTask); - var ex = await Assert.ThrowsAsync(() => + var ex = await Assert.ThrowsAsync(() => filter.GuardAsync(() => Task.CompletedTask)); Assert.Contains(ToolCallBudgetFilter.MaxToolCalls.ToString(), ex.Message); diff --git a/ReasoningAndActing/Program.cs b/ReasoningAndActing/Program.cs index a148b3e..5a68b13 100644 --- a/ReasoningAndActing/Program.cs +++ b/ReasoningAndActing/Program.cs @@ -50,7 +50,7 @@ Use at most 10 tool calls before giving your final answer. reactHistory, reactSettings, reactKernel); Console.WriteLine($"ReAct Agent:\n{reactResponse.Content}\n"); } -catch (InvalidOperationException ex) when (ex.Message.Contains("Tool-call budget")) +catch (ToolCallBudgetExceededException ex) { // Same shape as BoundedExecution: a PARTIAL result, the stop reason, and an explicit // incomplete label rather than silently truncated output. diff --git a/ReasoningAndActing/ToolCallBudgetExceededException.cs b/ReasoningAndActing/ToolCallBudgetExceededException.cs new file mode 100644 index 0000000..f22ed3d --- /dev/null +++ b/ReasoningAndActing/ToolCallBudgetExceededException.cs @@ -0,0 +1,10 @@ +namespace ReasoningAndActing; + +/// +/// Thrown by once the tool-call budget is exhausted. A typed +/// exception, not a message-substring match, so the catch site in Program.cs can't go silently +/// dead if the message text is ever reworded — see BoundedExecution.AgentFramework's +/// BudgetExceededException for the same shape. +/// +public sealed class ToolCallBudgetExceededException(int maxToolCalls) + : InvalidOperationException($"Tool-call budget of {maxToolCalls} exhausted."); diff --git a/ReasoningAndActing/ToolCallBudgetFilter.cs b/ReasoningAndActing/ToolCallBudgetFilter.cs index 87a8507..509e60a 100644 --- a/ReasoningAndActing/ToolCallBudgetFilter.cs +++ b/ReasoningAndActing/ToolCallBudgetFilter.cs @@ -3,11 +3,13 @@ namespace ReasoningAndActing; /// -/// Hard bound on tool calls: the prompt asks the model to stop at 10, this enforces it. The -/// counting/throwing logic lives in , a context-free core that a test can -/// drive directly — 's constructor is internal to Semantic -/// Kernel, so a test cannot build one to call itself. +/// Hard bound on tool calls: the prompt asks the model to stop at 10, this enforces it. /// +// ponytail: GuardAsync(Func) is a context-free stand-in for OnFunctionInvocationAsync so +// the counting/throwing logic is unit-testable. Upgrade path: SemanticKernel's +// FunctionInvocationContext constructor is internal, so a test can't build a real one - if a +// future SK version makes it public, this seam can be dropped and the test can drive +// OnFunctionInvocationAsync directly. public class ToolCallBudgetFilter : IFunctionInvocationFilter { public const int MaxToolCalls = 10; @@ -22,7 +24,7 @@ public class ToolCallBudgetFilter : IFunctionInvocationFilter public async Task GuardAsync(Func next) { if (Interlocked.Increment(ref _toolCalls) > MaxToolCalls) - throw new InvalidOperationException($"Tool-call budget of {MaxToolCalls} exhausted."); + throw new ToolCallBudgetExceededException(MaxToolCalls); await next(); } From 4553fb9acc2972120bf1cb66fb18b24a97410fc0 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 16:07:09 +0200 Subject: [PATCH 08/21] fix(judge): strict verdicts, indeterminate on malformed output, multiple orderings Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- .../AgenticPatterns.Tests.csproj | 1 + AgenticPatterns.Tests/LlmAsJudgeTests.cs | 65 +++++++++++++++++++ LLMAsJudge.AgentFramework/JudgeParsing.cs | 57 ++++++++++++++++ LLMAsJudge.AgentFramework/Program.cs | 63 +++++++++++------- PatternExplorer/patterns/LLMAsJudge.md | 40 +++++++----- README.md | 2 +- 6 files changed, 188 insertions(+), 40 deletions(-) create mode 100644 AgenticPatterns.Tests/LlmAsJudgeTests.cs create mode 100644 LLMAsJudge.AgentFramework/JudgeParsing.cs diff --git a/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj b/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj index 846cb1d..ca389c1 100644 --- a/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj +++ b/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj @@ -26,6 +26,7 @@ (and Program.cs's top-level-statement type) are internal and invisible across the assembly boundary, so referencing it too would add nothing to test. --> + E[Quality evaluators
Relevance, Coherence, Groundedness] A --> R[RubricJudgeEvaluator
1-5 + justification] - P[Two candidate answers] --> J[Pairwise judge] - J -->|swap positions| J - J --> B[Position-bias verdict] + P[Two candidate answers] --> J[Pairwise judge x5
randomized position] + J --> Parse[JudgeParsing.Parse
A / B / Indeterminate] + Parse --> B[Preference distribution
+ position-bias rate] ``` -The pairwise section pits a precise answer against a vague one, judges them in both orders, and -reports whether the same candidate won regardless of slot. `PositionBiasDetected` is the whole -verdict: original pick ≠ swapped pick means bias. +The pairwise section pits a precise answer against a vague one across five randomized orderings, +parses each verdict with `JudgeParsing.Parse` (which returns `Indeterminate` — never a default +winner — for anything that doesn't match the `{"winner": "A"}` / `{"winner": "B"}` contract), and +reports the win distribution. `JudgeParsing.Summarize` excludes `Indeterminate` verdicts from the +position-bias rate and reports their count separately; a well-behaved judge should pick the +precise answer regardless of slot, so any determinate flip is the bias signal. ## Key APIs @@ -79,6 +85,8 @@ verdict: original pick ≠ swapped pick means bias. | `GroundednessEvaluatorContext(policy)` | Supplies grounding source to the evaluator | | `IEvaluator.EvaluateAsync(messages, response, chatConfig, contexts)` | The evaluator contract | | `result.Get(name)` | Reads a score back out of the `EvaluationResult` | +| `JudgeParsing.Parse(json)` | Strict verdict parse: `A` / `B` / `Indeterminate`, never throws | +| `JudgeParsing.Summarize(picks)` | Win/loss/indeterminate counts plus a bias rate that excludes indeterminates | ```bash dotnet run --project LLMAsJudge.AgentFramework @@ -88,9 +96,9 @@ dotnet run --project LLMAsJudge.AgentFramework -- --selfcheck # offline bias-l ## What to watch in the output The first block prints each answer with four scored lines (`Relevance`, `Coherence`, -`Groundedness`, `RubricScore`) and the judge's reason per metric. The second block prints the -original-order pick, the swapped-order pick, and a `► Position bias` verdict. A well-behaved -judge on a clear-cut pair should pick the precise answer both times — if it flips, you have just -measured your instrument, not your agent. **RegressionEvals** builds a gate on top of these -evaluators, and **EvaluationAndMonitoring** tracks the token cost of running a judge on every -answer. +`Groundedness`, `RubricScore`) and the judge's reason per metric. The second block runs five +randomized orderings and prints the win/loss/indeterminate counts, the position-bias rate (of +determinate verdicts only), and a `► Position bias` verdict. A well-behaved judge on a clear-cut +pair should pick the precise answer regardless of slot — if it flips, you have just measured your +instrument, not your agent. **RegressionEvals** builds a gate on top of these evaluators, and +**EvaluationAndMonitoring** tracks the token cost of running a judge on every answer. diff --git a/README.md b/README.md index 93df299..b5aaaab 100644 --- a/README.md +++ b/README.md @@ -158,7 +158,7 @@ the catalog together; each result states its scope limits and cites a primary so | Pattern | What it demonstrates | |---|---| -| LLMAsJudge | Judge-model rubric scoring plus a position-bias probe | +| LLMAsJudge | Judge-model rubric scoring plus a randomized-ordering position-bias probe | | RedTeaming | Deterministic leak checks first, judge second, against the real GuardRails filter | | RegressionEvals | Golden-dataset suite with tiered assertions, cached as a CI gate | | TrajectoryEvaluation | Scoring the agent's tool-use path with agent evaluators | From 35ec9a5f88b225d203716dbc471b9bc4b961c0e1 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 16:12:33 +0200 Subject: [PATCH 09/21] docs(judge): the probe is no longer a single swap Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- PatternExplorer/patterns/LLMAsJudge.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/PatternExplorer/patterns/LLMAsJudge.md b/PatternExplorer/patterns/LLMAsJudge.md index c18f673..10f6840 100644 --- a/PatternExplorer/patterns/LLMAsJudge.md +++ b/PatternExplorer/patterns/LLMAsJudge.md @@ -36,7 +36,7 @@ where this is what a second model says about the answer. A judge score is a proxy, and Goodhart's law applies the moment you optimize against it (see `docs/coordination-physics.md`): tune a prompt to please the judge and you may buy judge points -without buying quality. The position-swap probe is the honest correction — it measures the +without buying quality. The randomized-ordering probe is the honest correction — it measures the *instrument's* noise floor before you trust its readings. A judge that flips on position is not measuring answer quality; it is measuring slot order, and any score it emits is that much less information about the thing you actually care about. From 6396e223795d3941c991609a140aa8ec8cb39bda Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 16:25:36 +0200 Subject: [PATCH 10/21] fix(regression-evals): trace cases need human promotion; rename exact to contains Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- .../AgenticPatterns.Tests.csproj | 1 + AgenticPatterns.Tests/CasePartitionTests.cs | 67 +++++++++++++++++ .../patterns/EvaluationAndMonitoring.md | 6 +- PatternExplorer/patterns/RegressionEvals.md | 74 ++++++++++++------- README.md | 2 +- .../CasePartition.cs | 24 ++++++ RegressionEvals.AgentFramework/GoldenCase.cs | 7 ++ RegressionEvals.AgentFramework/Program.cs | 37 +++++++--- .../golden-cases.json | 10 +-- 9 files changed, 183 insertions(+), 45 deletions(-) create mode 100644 AgenticPatterns.Tests/CasePartitionTests.cs create mode 100644 RegressionEvals.AgentFramework/CasePartition.cs create mode 100644 RegressionEvals.AgentFramework/GoldenCase.cs diff --git a/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj b/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj index ca389c1..50acb13 100644 --- a/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj +++ b/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj @@ -39,6 +39,7 @@ + diff --git a/AgenticPatterns.Tests/CasePartitionTests.cs b/AgenticPatterns.Tests/CasePartitionTests.cs new file mode 100644 index 0000000..5a1a84d --- /dev/null +++ b/AgenticPatterns.Tests/CasePartitionTests.cs @@ -0,0 +1,67 @@ +using RegressionEvals.AgentFramework; +using Xunit; + +namespace AgenticPatterns.Tests; + +public class CasePartitionTests +{ + private static GoldenCase Case(string id, string reviewedBy) => + new(id, "Q?", "A.", "contains", reviewedBy); + + [Fact] + public void ReviewedCaseIsEvaluated() + { + var (evaluated, awaitingReview) = CasePartition.Partition([Case("reviewed", "alex")]); + + Assert.Equal(["reviewed"], evaluated.Select(c => c.Id)); + Assert.Empty(awaitingReview); + } + + // This is the guarantee this task exists to pin: a case with no reviewer - exactly the shape + // ExtractTraceCase used to hand straight to the evaluator - must be EXCLUDED from evaluation, + // not merely present somewhere. + [Fact] + public void UnreviewedCaseIsExcludedFromEvaluationAndReportedAsAwaitingReview() + { + var (evaluated, awaitingReview) = CasePartition.Partition([Case("from-trace", reviewedBy: null!)]); + + Assert.Empty(evaluated); + Assert.Equal(["from-trace"], awaitingReview.Select(c => c.Id)); + } + + [Fact] + public void EmptyReviewedByIsAlsoAwaitingReview() + { + var (evaluated, awaitingReview) = CasePartition.Partition([Case("blank", reviewedBy: "")]); + + Assert.Empty(evaluated); + Assert.Single(awaitingReview); + } + + [Fact] + public void MixedCorpusSplitsCorrectly() + { + var (evaluated, awaitingReview) = CasePartition.Partition( + [ + Case("a", "alex"), + Case("b", null!), + Case("c", "jamie"), + Case("d", "") + ]); + + Assert.Equal(["a", "c"], evaluated.Select(c => c.Id)); + Assert.Equal(["b", "d"], awaitingReview.Select(c => c.Id)); + } + + [Fact] + public void GateFailsWhenNothingWasEvaluatedEvenWithZeroFailures() => + Assert.Equal(1, CasePartition.GateExitCode(evaluatedCount: 0, failureCount: 0)); + + [Fact] + public void GatePassesWhenSomethingWasEvaluatedAndNothingFailed() => + Assert.Equal(0, CasePartition.GateExitCode(evaluatedCount: 3, failureCount: 0)); + + [Fact] + public void GateFailsOnAnyFailure() => + Assert.Equal(1, CasePartition.GateExitCode(evaluatedCount: 3, failureCount: 1)); +} diff --git a/PatternExplorer/patterns/EvaluationAndMonitoring.md b/PatternExplorer/patterns/EvaluationAndMonitoring.md index 4783b1a..521ce64 100644 --- a/PatternExplorer/patterns/EvaluationAndMonitoring.md +++ b/PatternExplorer/patterns/EvaluationAndMonitoring.md @@ -128,5 +128,7 @@ signals to an answer where this pattern measures cost and speed. This pattern measures cost and speed and records ground truth; the **Evaluation** category (**LLMAsJudge**, **RegressionEvals**, **TrajectoryEvaluation**, **RedTeaming**) judges quality — -**RegressionEvals** in particular turns the traces recorded here into a regression gate, and -**TrajectoryEvaluation** judges whether the tool calls this pattern counts were the right ones. +**RegressionEvals** in particular turns the traces recorded here into review candidates for its +regression gate (a trace records what happened, not what should have happened, so a human still +promotes each one), and **TrajectoryEvaluation** judges whether the tool calls this pattern counts +were the right ones. diff --git a/PatternExplorer/patterns/RegressionEvals.md b/PatternExplorer/patterns/RegressionEvals.md index 2f248f3..cec3508 100644 --- a/PatternExplorer/patterns/RegressionEvals.md +++ b/PatternExplorer/patterns/RegressionEvals.md @@ -1,7 +1,7 @@ --- { "title": "Regression Evals", - "summary": "A golden dataset with tiered assertions (string, NLP, judge) run as a gate, cached for CI.", + "summary": "A golden dataset of reviewed cases with tiered assertions (contains, NLP, judge) run as a gate, cached for CI.", "category": "Evaluation", "projects": [ { "flavor": "AgentFramework", "path": "RegressionEvals.AgentFramework" } @@ -21,24 +21,34 @@ tool renders as an HTML report. The assertions are **tiered cheapest-first**, because not every check needs a model: -- **exact** — a plain `Contains` string check. No model call. +- **contains** — a plain `Contains` string check. No model call. - **nlp** — `F1Evaluator` token overlap against the golden answer. No model call. - **judge** — `EquivalenceEvaluator`, an LLM-as-judge, for semantic equality. One model call. +Only **reviewed** cases run: a `GoldenCase` with a non-empty `ReviewedBy` is evaluated, everything +else is reported as awaiting review and skipped. The gate itself fails if nothing was evaluated — +a green run over zero reviewed cases is the same class of false confidence as a red regression +that ships anyway. + **Builds on:** **EvaluationAndMonitoring** records the trajectories this pattern turns into -golden cases — one case here is extracted straight from a recorded `run-trace.json`, the -canonical *production trace → eval case* pipeline. Where **SelfCorrectionLoop** runs an evaluator -inline to fix a single answer, this runs the same class of evaluators as a pre-merge gate over -many. The judge tier is **LLMAsJudge** embedded in a suite. +review candidates — one is extracted straight from a recorded `run-trace.json` into `candidates/` +as the pipeline `production trace → candidate case → reviewer supplies/verifies the expected +result → promoted to the golden set`. A trace is ground truth about what *happened*, never about +what *should have* happened, so extraction alone never gates anything. Where **SelfCorrectionLoop** +runs an evaluator inline to fix a single answer, this runs the same class of evaluators as a +pre-merge gate over many. The judge tier is **LLMAsJudge** embedded in a suite. ## Information-theoretic view An eval suite is a proxy for product quality, and a passing suite alongside a failing product means the proxy has drifted (see `docs/coordination-physics.md`). The working posture is that the suite is a hypothesis incidents revise: when a real regression ships green, the fix is a new -golden case, not a shrug. This is why the trace-sourced case matters — a captured trajectory is -ground truth about what actually happened, so growing the suite from real traces keeps the proxy -anchored to reality instead of to whatever the author imagined. +golden case, not a shrug. This is why the trace-sourced candidate matters — a captured trajectory +is ground truth about what actually happened, so growing the suite from real traces keeps the +proxy anchored to reality instead of to whatever the author imagined. But a trace only records +what happened, not what *should* have happened: a reviewer must supply or confirm the expected +result before a trace-derived case can gate anything, or the suite would simply freeze the +model's own historical mistakes into its own ground truth. ## When to use it @@ -51,28 +61,36 @@ the honest amount of machinery for a question you will not ask twice. ## How the demo works -Five golden cases live in `golden-cases.json`; a sixth is extracted at runtime from -`sample-run-trace.json` (a minimal `RunTrace` in EvaluationAndMonitoring's format) by reading its -first question and answer with `JsonDocument`. Each case runs through a `ScenarioRun` created from -a cache-enabled `DiskBasedReportingConfiguration`; the tier's assertion decides pass/fail; any -failure sets the process exit code to 1. +Five golden cases live in `golden-cases.json`, each already carrying a `reviewedBy`. At runtime a +sixth candidate is extracted from `sample-run-trace.json` (a minimal `RunTrace` in +EvaluationAndMonitoring's format) by reading its first question and observed answer with +`JsonDocument`, and written to `candidates/from-trace.json` with `reviewedBy: null` — it never +joins the evaluated set. `CasePartition.Partition` splits `golden-cases.json` into cases with a +reviewer (evaluated) and cases without one (reported and skipped); combined with the freshly +written candidate, the run prints `N candidate case(s) awaiting review — not evaluated`. Each +evaluated case runs through a `ScenarioRun` created from a cache-enabled +`DiskBasedReportingConfiguration`; the tier's assertion decides pass/fail; any failure, or an +empty evaluated set, sets the process exit code to 1. ```mermaid flowchart LR - G[golden-cases.json] --> S[Suite runner] - T[sample-run-trace.json] -->|extract Q and A| S + T[sample-run-trace.json] -->|extract Q and observed A| C[candidate case in candidates/] + C -->|reviewer supplies/verifies the expected result| G[golden-cases.json] + G --> P{CasePartition: reviewedBy?} + P -->|non-empty| S[Suite runner] + P -->|empty/missing| W[awaiting review — not evaluated] S --> A{tier} - A -->|exact| X[Contains check] + A -->|contains| X[Contains check] A -->|nlp| F[F1Evaluator] A -->|judge| E[EquivalenceEvaluator] X --> R[pass/fail + exit code] F --> R E --> R - S -->|via ReportingConfiguration| C[(response cache + result store)] + S -->|via ReportingConfiguration| Cache[(response cache + result store)] ``` -The `exact` and `nlp` tiers never call the model. The `judge` tier does, but its result is cached -by the reporting configuration, so a second run of an unchanged suite is free. +The `contains` and `nlp` tiers never call the model. The `judge` tier does, but its result is +cached by the reporting configuration, so a second run of an unchanged suite is free. ## Key APIs @@ -85,16 +103,18 @@ by the reporting configuration, so a second run of an unchanged suite is free. | `dotnet tool run aieval report --path --output report.html` | Renders the HTML report | ```bash -dotnet run --project RegressionEvals.AgentFramework # run the suite (exit 1 on failure) +dotnet run --project RegressionEvals.AgentFramework # run the suite (exit 1 on failure or 0 reviewed) dotnet run --project RegressionEvals.AgentFramework # re-run: judged case served from cache dotnet run --project RegressionEvals.AgentFramework -- --selfcheck # offline trace-extraction check ``` ## What to watch in the output -Each case prints a `[PASS]`/`[FAIL]` line with its tier and the assertion detail, then the -answer. The summary line reports the pass count and the `aieval` command to render the report; -the process exits non-zero if anything failed — that is the CI gate. Run it twice: the second run -returns the judged case from the response cache without a model call. **EvaluationAndMonitoring** -is where the trace-sourced case comes from, and **LLMAsJudge** documents the judge tier's own -failure modes. +The first line reports how many candidate cases are awaiting review and were skipped — the +trace-derived one every run, plus any hand-added `golden-cases.json` row missing a `reviewedBy`. +Each evaluated case then prints a `[PASS]`/`[FAIL]` line with its tier and the assertion detail, +then the answer. The summary line reports the pass count and the `aieval` command to render the +report; the process exits non-zero if anything failed *or* if nothing was evaluated at all — both +are the CI gate. Run it twice: the second run returns the judged case from the response cache +without a model call. **EvaluationAndMonitoring** is where the trace-sourced candidate comes +from, and **LLMAsJudge** documents the judge tier's own failure modes. diff --git a/README.md b/README.md index b5aaaab..97d1969 100644 --- a/README.md +++ b/README.md @@ -160,7 +160,7 @@ the catalog together; each result states its scope limits and cites a primary so |---|---| | LLMAsJudge | Judge-model rubric scoring plus a randomized-ordering position-bias probe | | RedTeaming | Deterministic leak checks first, judge second, against the real GuardRails filter | -| RegressionEvals | Golden-dataset suite with tiered assertions, cached as a CI gate | +| RegressionEvals | Golden-dataset suite of reviewed cases with tiered assertions, cached as a CI gate | | TrajectoryEvaluation | Scoring the agent's tool-use path with agent evaluators | ## Setup diff --git a/RegressionEvals.AgentFramework/CasePartition.cs b/RegressionEvals.AgentFramework/CasePartition.cs new file mode 100644 index 0000000..6846206 --- /dev/null +++ b/RegressionEvals.AgentFramework/CasePartition.cs @@ -0,0 +1,24 @@ +namespace RegressionEvals.AgentFramework; + +// The rule that decides which golden cases are trustworthy enough to gate a release: a case is +// evaluated only once a human has reviewed it (a non-empty ReviewedBy); everything else is +// awaiting review and must never be silently evaluated. This is what stops a trace-derived +// candidate - or any hand-added row missing a reviewer - from freezing an unreviewed answer into +// the suite. +public static class CasePartition +{ + public static (IReadOnlyList Evaluated, IReadOnlyList AwaitingReview) Partition( + IEnumerable cases) + { + var evaluated = new List(); + var awaitingReview = new List(); + foreach (var c in cases) + (string.IsNullOrEmpty(c.ReviewedBy) ? awaitingReview : evaluated).Add(c); + return (evaluated, awaitingReview); + } + + // A suite that evaluated zero cases is not a passing suite, even if nothing failed - the same + // class of defect as freezing an unreviewed trace answer into the gate. + public static int GateExitCode(int evaluatedCount, int failureCount) => + evaluatedCount == 0 || failureCount > 0 ? 1 : 0; +} diff --git a/RegressionEvals.AgentFramework/GoldenCase.cs b/RegressionEvals.AgentFramework/GoldenCase.cs new file mode 100644 index 0000000..ca9a56b --- /dev/null +++ b/RegressionEvals.AgentFramework/GoldenCase.cs @@ -0,0 +1,7 @@ +namespace RegressionEvals.AgentFramework; + +// A trace is ground truth about what HAPPENED, never about what SHOULD have happened. Extracting +// one gives a candidate; a reviewer supplies or confirms the expected result before it can gate a +// release. Promoting the model's own historical answer freezes its mistakes into the suite. +public record CandidateCase(string Id, string Question, string ObservedAnswer, string SourceTrace); +public record GoldenCase(string Id, string Question, string ExpectedAnswer, string Tier, string ReviewedBy); diff --git a/RegressionEvals.AgentFramework/Program.cs b/RegressionEvals.AgentFramework/Program.cs index 0943b82..f9f84f0 100644 --- a/RegressionEvals.AgentFramework/Program.cs +++ b/RegressionEvals.AgentFramework/Program.cs @@ -6,6 +6,7 @@ using Microsoft.Extensions.AI.Evaluation.Quality; using Microsoft.Extensions.AI.Evaluation.Reporting; using Microsoft.Extensions.AI.Evaluation.Reporting.Storage; +using RegressionEvals.AgentFramework; using Shared; if (args.FirstOrDefault() == "--selfcheck") { SelfCheck(); return; } @@ -14,11 +15,28 @@ var baseDir = AppContext.BaseDirectory; var web = new JsonSerializerOptions(JsonSerializerDefaults.Web); -var cases = JsonSerializer.Deserialize>( +var allCases = JsonSerializer.Deserialize>( File.ReadAllText(Path.Combine(baseDir, "golden-cases.json")), web)!; -// The canonical "production trace -> eval case" pipeline: one case comes straight from a -// recorded EvaluationAndMonitoring trajectory rather than being hand-written. -cases.Add(ExtractTraceCase(Path.Combine(baseDir, "sample-run-trace.json"))); +var (cases, awaitingReview) = CasePartition.Partition(allCases); + +// The canonical "production trace -> candidate case" pipeline: extraction pulls the question and +// the OBSERVED answer straight from a recorded EvaluationAndMonitoring trajectory. A trace is +// ground truth about what HAPPENED, never about what SHOULD have happened, so it lands in +// candidates/ for a reviewer to supply/confirm the expected answer - it never joins `cases` above. +var candidate = ExtractTraceCase(Path.Combine(baseDir, "sample-run-trace.json")); +var candidatesDir = Path.Combine(baseDir, "candidates"); +Directory.CreateDirectory(candidatesDir); +// ponytail: candidates/ lives under bin/, build output wiped by `dotnet clean`, so a candidate +// does not survive here to actually be reviewed. A real repo commits candidates beside +// golden-cases.json and reviews the promotion (filling in expectedAnswer/tier/reviewedBy) in a PR. +File.WriteAllText(Path.Combine(candidatesDir, $"{candidate.Id}.json"), JsonSerializer.Serialize(new +{ + candidate.Id, candidate.Question, candidate.ObservedAnswer, candidate.SourceTrace, + reviewedBy = (string?)null +}, web)); + +var awaitingCount = awaitingReview.Count + 1; // + the candidate just extracted from the trace +Console.WriteLine($"{awaitingCount} candidate case(s) awaiting review - not evaluated.\n"); const string policy = "TechCorp laptops include a two-year limited warranty. Defective products may be " + @@ -42,7 +60,7 @@ var (passed, detail) = c.Tier switch { - "exact" => (answer.Contains(c.ExpectedAnswer, StringComparison.OrdinalIgnoreCase), + "contains" => (answer.Contains(c.ExpectedAnswer, StringComparison.OrdinalIgnoreCase), $"contains \"{c.ExpectedAnswer}\""), "nlp" => await NlpTierAsync(run, c, answer), "judge" => await JudgeTierAsync(run, c, answer), @@ -57,7 +75,7 @@ Console.WriteLine($"\n{cases.Count - failures}/{cases.Count} passed. " + "Re-run to see cached (zero-call) evaluation; generate an HTML report with:"); Console.WriteLine($" dotnet tool run aieval report --path {Path.Combine(baseDir, "eval-results")} --output report.html"); -Environment.Exit(failures == 0 ? 0 : 1); +Environment.Exit(CasePartition.GateExitCode(cases.Count, failures)); async Task<(bool, string)> NlpTierAsync(ScenarioRun run, GoldenCase c, string answer) { @@ -81,13 +99,14 @@ [new ChatMessage(ChatRole.User, c.Question)], return (pass, $"equivalence={eq.Value} ({eq.Interpretation?.Rating})"); } -GoldenCase ExtractTraceCase(string tracePath) +CandidateCase ExtractTraceCase(string tracePath) { using var doc = JsonDocument.Parse(File.ReadAllText(tracePath)); var firstCall = doc.RootElement.GetProperty("modelCalls")[0]; string TextOf(string arrayName) => firstCall.GetProperty(arrayName)[0] .GetProperty("contents")[0].GetProperty("payload").GetProperty("value").GetString()!; - return new GoldenCase("from-trace", TextOf("messages"), TextOf("responseMessages"), "judge"); + return new CandidateCase( + "from-trace", TextOf("messages"), TextOf("responseMessages"), Path.GetFileName(tracePath)); } static void SelfCheck() @@ -103,5 +122,3 @@ static void SelfCheck() if (q != "Q?") throw new Exception("extraction broken"); Console.WriteLine("selfcheck ok"); } - -record GoldenCase(string Id, string Question, string ExpectedAnswer, string Tier); diff --git a/RegressionEvals.AgentFramework/golden-cases.json b/RegressionEvals.AgentFramework/golden-cases.json index 24aba0a..f645ae4 100644 --- a/RegressionEvals.AgentFramework/golden-cases.json +++ b/RegressionEvals.AgentFramework/golden-cases.json @@ -1,7 +1,7 @@ [ - { "id": "warranty-length", "question": "How long is the TechCorp laptop warranty?", "expectedAnswer": "two-year limited warranty", "tier": "exact" }, - { "id": "return-window", "question": "How many days do I have to return a defective product?", "expectedAnswer": "30 days", "tier": "exact" }, - { "id": "warranty-phrasing", "question": "What warranty comes with TechCorp laptops?", "expectedAnswer": "TechCorp laptops include a two-year limited warranty.", "tier": "nlp" }, - { "id": "returns-phrasing", "question": "How do I return a defective item?", "expectedAnswer": "Return defective products within 30 days with your order number.", "tier": "nlp" }, - { "id": "warranty-semantic", "question": "Tell me about laptop warranty coverage.", "expectedAnswer": "TechCorp laptops have a two-year limited warranty.", "tier": "judge" } + { "id": "warranty-length", "question": "How long is the TechCorp laptop warranty?", "expectedAnswer": "two-year limited warranty", "tier": "contains", "reviewedBy": "maintainer" }, + { "id": "return-window", "question": "How many days do I have to return a defective product?", "expectedAnswer": "30 days", "tier": "contains", "reviewedBy": "maintainer" }, + { "id": "warranty-phrasing", "question": "What warranty comes with TechCorp laptops?", "expectedAnswer": "TechCorp laptops include a two-year limited warranty.", "tier": "nlp", "reviewedBy": "maintainer" }, + { "id": "returns-phrasing", "question": "How do I return a defective item?", "expectedAnswer": "Return defective products within 30 days with your order number.", "tier": "nlp", "reviewedBy": "maintainer" }, + { "id": "warranty-semantic", "question": "Tell me about laptop warranty coverage.", "expectedAnswer": "TechCorp laptops have a two-year limited warranty.", "tier": "judge", "reviewedBy": "maintainer" } ] From fc86e993394fd1df98f49a835312a8f775df929e Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 16:32:01 +0200 Subject: [PATCH 11/21] fix(regression-evals): report awaiting-signoff and awaiting-review counts separately F1 from review round 1: folding unreviewed golden cases (already have an expected answer + tier, just need sign-off) and the trace-derived candidate (no expected answer yet, needs one written from scratch) into one "candidate case(s) awaiting review" count contradicted the very distinction this task introduced. Print and document them as two separate counts. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- PatternExplorer/patterns/RegressionEvals.md | 18 +++++++++++------- RegressionEvals.AgentFramework/Program.cs | 9 +++++++-- 2 files changed, 18 insertions(+), 9 deletions(-) diff --git a/PatternExplorer/patterns/RegressionEvals.md b/PatternExplorer/patterns/RegressionEvals.md index cec3508..6598eb8 100644 --- a/PatternExplorer/patterns/RegressionEvals.md +++ b/PatternExplorer/patterns/RegressionEvals.md @@ -66,9 +66,12 @@ sixth candidate is extracted from `sample-run-trace.json` (a minimal `RunTrace` EvaluationAndMonitoring's format) by reading its first question and observed answer with `JsonDocument`, and written to `candidates/from-trace.json` with `reviewedBy: null` — it never joins the evaluated set. `CasePartition.Partition` splits `golden-cases.json` into cases with a -reviewer (evaluated) and cases without one (reported and skipped); combined with the freshly -written candidate, the run prints `N candidate case(s) awaiting review — not evaluated`. Each -evaluated case runs through a `ScenarioRun` created from a cache-enabled +reviewer (evaluated) and cases without one (reported and skipped). The run prints two separate +counts, because the two states are not the same thing: `N golden case(s) awaiting sign-off` for +`golden-cases.json` rows that already have an expected answer and tier but no reviewer yet, and +`1 candidate case(s) awaiting review` for the trace-extracted `CandidateCase`, which has neither +and needs a reviewer to write the expected answer from scratch. Each evaluated case runs through +a `ScenarioRun` created from a cache-enabled `DiskBasedReportingConfiguration`; the tier's assertion decides pass/fail; any failure, or an empty evaluated set, sets the process exit code to 1. @@ -78,7 +81,7 @@ flowchart LR C -->|reviewer supplies/verifies the expected result| G[golden-cases.json] G --> P{CasePartition: reviewedBy?} P -->|non-empty| S[Suite runner] - P -->|empty/missing| W[awaiting review — not evaluated] + P -->|empty/missing| W[awaiting sign-off — not evaluated] S --> A{tier} A -->|contains| X[Contains check] A -->|nlp| F[F1Evaluator] @@ -110,9 +113,10 @@ dotnet run --project RegressionEvals.AgentFramework -- --selfcheck # offline t ## What to watch in the output -The first line reports how many candidate cases are awaiting review and were skipped — the -trace-derived one every run, plus any hand-added `golden-cases.json` row missing a `reviewedBy`. -Each evaluated case then prints a `[PASS]`/`[FAIL]` line with its tier and the assertion detail, +The first two lines report what was skipped, split by state: unsigned `golden-cases.json` rows +(fully specified, just need a reviewer's sign-off) and the trace-derived candidate (no expected +answer yet, needs one written from scratch). Each evaluated case then prints a `[PASS]`/`[FAIL]` +line with its tier and the assertion detail, then the answer. The summary line reports the pass count and the `aieval` command to render the report; the process exits non-zero if anything failed *or* if nothing was evaluated at all — both are the CI gate. Run it twice: the second run returns the judged case from the response cache diff --git a/RegressionEvals.AgentFramework/Program.cs b/RegressionEvals.AgentFramework/Program.cs index f9f84f0..c24e2a3 100644 --- a/RegressionEvals.AgentFramework/Program.cs +++ b/RegressionEvals.AgentFramework/Program.cs @@ -35,8 +35,13 @@ reviewedBy = (string?)null }, web)); -var awaitingCount = awaitingReview.Count + 1; // + the candidate just extracted from the trace -Console.WriteLine($"{awaitingCount} candidate case(s) awaiting review - not evaluated.\n"); +// Two different states, two different counts: an awaiting-review GoldenCase already has an +// ExpectedAnswer and Tier and just needs sign-off, while a CandidateCase has neither and needs a +// reviewer to write the expected answer from scratch. Folding them into one number would call a +// fully-specified unsigned golden case a "candidate", contradicting the distinction this task exists +// to enforce. +Console.WriteLine($"{awaitingReview.Count} golden case(s) awaiting sign-off - not evaluated."); +Console.WriteLine("1 candidate case(s) awaiting review - not evaluated.\n"); const string policy = "TechCorp laptops include a two-year limited warranty. Defective products may be " + From 388001619d3f46698ab5cfbd69972bf85acd3d23 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 16:33:50 +0200 Subject: [PATCH 12/21] fix(regression-evals): derive the candidate count instead of hard-coding it Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- RegressionEvals.AgentFramework/Program.cs | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/RegressionEvals.AgentFramework/Program.cs b/RegressionEvals.AgentFramework/Program.cs index c24e2a3..4dc9734 100644 --- a/RegressionEvals.AgentFramework/Program.cs +++ b/RegressionEvals.AgentFramework/Program.cs @@ -23,17 +23,18 @@ // the OBSERVED answer straight from a recorded EvaluationAndMonitoring trajectory. A trace is // ground truth about what HAPPENED, never about what SHOULD have happened, so it lands in // candidates/ for a reviewer to supply/confirm the expected answer - it never joins `cases` above. -var candidate = ExtractTraceCase(Path.Combine(baseDir, "sample-run-trace.json")); +List candidates = [ExtractTraceCase(Path.Combine(baseDir, "sample-run-trace.json"))]; var candidatesDir = Path.Combine(baseDir, "candidates"); Directory.CreateDirectory(candidatesDir); // ponytail: candidates/ lives under bin/, build output wiped by `dotnet clean`, so a candidate // does not survive here to actually be reviewed. A real repo commits candidates beside // golden-cases.json and reviews the promotion (filling in expectedAnswer/tier/reviewedBy) in a PR. -File.WriteAllText(Path.Combine(candidatesDir, $"{candidate.Id}.json"), JsonSerializer.Serialize(new -{ - candidate.Id, candidate.Question, candidate.ObservedAnswer, candidate.SourceTrace, - reviewedBy = (string?)null -}, web)); +foreach (var candidate in candidates) + File.WriteAllText(Path.Combine(candidatesDir, $"{candidate.Id}.json"), JsonSerializer.Serialize(new + { + candidate.Id, candidate.Question, candidate.ObservedAnswer, candidate.SourceTrace, + reviewedBy = (string?)null + }, web)); // Two different states, two different counts: an awaiting-review GoldenCase already has an // ExpectedAnswer and Tier and just needs sign-off, while a CandidateCase has neither and needs a @@ -41,7 +42,7 @@ // fully-specified unsigned golden case a "candidate", contradicting the distinction this task exists // to enforce. Console.WriteLine($"{awaitingReview.Count} golden case(s) awaiting sign-off - not evaluated."); -Console.WriteLine("1 candidate case(s) awaiting review - not evaluated.\n"); +Console.WriteLine($"{candidates.Count} candidate case(s) awaiting review - not evaluated.\n"); const string policy = "TechCorp laptops include a two-year limited warranty. Defective products may be " + From 2ccc76c29bd88ec5e7657d5cc250d3aa35e53972 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 16:39:45 +0200 Subject: [PATCH 13/21] fix(resource-aware): account for the fallback call and call the budget soft Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- .../patterns/ResourceAwareOptimization.md | 50 +++++++++++-------- README.md | 2 +- .../Program.cs | 16 +++--- 3 files changed, 41 insertions(+), 27 deletions(-) diff --git a/PatternExplorer/patterns/ResourceAwareOptimization.md b/PatternExplorer/patterns/ResourceAwareOptimization.md index c5a8d24..9671845 100644 --- a/PatternExplorer/patterns/ResourceAwareOptimization.md +++ b/PatternExplorer/patterns/ResourceAwareOptimization.md @@ -1,7 +1,7 @@ --- { "title": "Resource-Aware Optimization", - "summary": "Route each query to the cheapest model that can answer it, and stop spending when the budget runs out.", + "summary": "Route each query to the cheapest model that can answer it, and degrade — forcing the cheap tier or refusing reasoning-tier work — once a softly-enforced budget is crossed.", "category": "Production controls", "projects": [ { "flavor": "AgentFramework", "path": "ResourceAwareOptimization.AgentFramework" }, @@ -14,9 +14,12 @@ Not every question deserves the expensive model. Resource-aware optimization puts a cheap classifier in front of several model tiers, sends each request to the smallest tier that can -handle it, and tracks what the answers cost. When the running total crosses a budget, the system -degrades deliberately — forcing the cheap tier or refusing the work — instead of silently burning -money. +handle it, and tracks what the answers cost. The budget is **soft**: routing and refusal +decisions are made from the running total observed *after* each call returns, not reserved +before dispatch, so the call that pushes the total over the cap has already been paid for by the +time anything reacts to it. Once the prior total is over budget, the system degrades +deliberately — forcing the cheap tier for simple queries, refusing reasoning-tier work — instead +of silently burning money. For a hard, pre-call ceiling instead, see **Bounded Execution**. Two mechanisms combine: **tiered routing** (pick a model per request, with a fallback chain if a tier errors) and **budget enforcement** (measure token usage, convert to cents, gate on the total). @@ -24,7 +27,8 @@ tier errors) and **budget enforcement** (measure token usage, convert to cents, ## When to use it - Mixed traffic where most requests are trivial and a minority genuinely need a reasoning model. -- Hard cost ceilings per session, tenant, or user. +- Soft cost ceilings per session, tenant, or user, where degrading gracefully is acceptable and + a hard pre-call cap isn't required. - You want graceful degradation under provider outages: fall through to the next tier instead of failing the request. @@ -42,21 +46,24 @@ and each tier has a fallback chain if the call throws. ```mermaid flowchart LR - Q[User query] --> C[ClassifyQuery heuristic] + Q[User query] --> Chk[Check running total
from prior calls] + Chk -->|over budget, needs reasoning| G[Refuse, or force fast tier] + Chk -->|under budget, or simple query| C[ClassifyQuery heuristic] C -->|simple| F[Fast tier
gpt-4o-mini] C -->|reasoning| R[Reasoning tier
o4-mini] - F --> B[Budget tracker
tokens to cents] + F --> B[Record actual usage
after the call returns] R --> B - B -->|under budget| A[Answer] - B -->|over budget| G[Degrade to fast tier
or refuse] + B -.->|updates the total
for the next query| Chk ``` The flavors differ in where the logic lives. Agent Framework implements it as **two middleware layers**: `RoutingMiddleware` on the `IChatClient` picks the tier and calls the chosen client -directly instead of `next`, while `BudgetEnforcementMiddleware` on the agent short-circuits the -whole run once `BudgetState.Exceeded` is true, returning a canned "I've reached my processing -budget" message. Semantic Kernel does it with **keyed services**: three chat-completion services -registered under the ids `fast`, `reasoning`, and `default`, resolved per query via +directly instead of `next`, while `BudgetEnforcementMiddleware` on the agent short-circuits +reasoning-tier queries once `BudgetState.Exceeded` is true — via `QueryRouter.RefuseForBudget`, +so simple queries still get answered on the fast tier — returning a canned "I've reached my +processing budget" message. Semantic Kernel does it with **keyed services**: three +chat-completion services registered under the ids `fast`, `reasoning`, and `default`, resolved +per query via `GetRequiredKeyedService(sid)`, with the loop breaking out entirely once `BudgetTracker.BudgetExceeded` flips. Its budget is a deliberately tiny 2¢ so the reasoning query actually trips it; the Agent Framework budget is 50¢. @@ -77,10 +84,13 @@ Both cost models are approximations hard-coded per 1K tokens: `gpt-4o-mini` 0.01 ## What to watch in the output Each query prints `[Router] Classified as: simple|reasoning`, then `[Router] Trying: ` and -`[Router] Success with: `; a failed tier prints `[Fallback] failed: ...`. Every -completed call prints a `[Budget]` line with token count, incremental cost, and the running total -against the cap, and the run ends with `Total estimated cost: N.NN¢`. Watch for -`[Router] Budget exceeded — forcing fast tier.` and `[BudgetMiddleware] Budget exceeded. Returning -early.` in the Agent Framework flavor, and `Budget limit reached. Skipping remaining queries.` in -Semantic Kernel. **Routing** shows the same classify-then-dispatch idea without the cost angle, -and **Middleware** explains the interception layers this sample builds on. +`[Router] Success with: `; a failed tier prints `[Fallback] failed: ...`. If every +tier in the chain fails, Agent Framework prints `[Fallback] All tier models failed. Trying +original pipeline.` and makes one more call against the underlying client — that call's cost is +recorded too, so it is folded into the same running total. Every completed call, including that +last-resort one, prints a `[Budget]` line with token count, incremental cost, and the running +total against the cap, and the run ends with `Total estimated cost: N.NN¢`. Watch for +`[Router] Budget exceeded — forcing fast tier.` and `[BudgetMiddleware] Budget exceeded. Refusing +expensive-tier work.` in the Agent Framework flavor, and `Budget limit reached. Skipping remaining +queries.` in Semantic Kernel. **Routing** shows the same classify-then-dispatch idea without the +cost angle, and **Middleware** explains the interception layers this sample builds on. diff --git a/README.md b/README.md index 97d1969..4d7c47d 100644 --- a/README.md +++ b/README.md @@ -151,7 +151,7 @@ the catalog together; each result states its scope limits and cites a primary so | HumanInTheLoop | Tool-call approval gates | | IdempotentToolCalls | Retry-safe side effects: the dedup record lives with the side effect, not the caller | | Middleware | Agent-run and function-invocation middleware (logging, latency, tool guards) | -| ResourceAwareOptimization | Model routing under a cost budget | +| ResourceAwareOptimization | Model routing under a soft, post-call cost budget | | ToolAuthorization | Capability-scoped, argument-level authorization before tool execution; one-time grants are reserved, then committed after the effect | ### Evaluation diff --git a/ResourceAwareOptimization.AgentFramework/Program.cs b/ResourceAwareOptimization.AgentFramework/Program.cs index 1826e4f..c449b5e 100644 --- a/ResourceAwareOptimization.AgentFramework/Program.cs +++ b/ResourceAwareOptimization.AgentFramework/Program.cs @@ -17,7 +17,7 @@ var reasoningModel = (Client: azureClient.GetChatClient(ReasoningModelDeployment).AsIChatClient(), ModelId: ReasoningModelDeployment); -var budget = new BudgetState(50); +var softBudget = new BudgetState(50); async Task RoutingMiddleware( IEnumerable messages, @@ -34,7 +34,7 @@ async Task RoutingMiddleware( : new[] { fastModel, reasoningModel }; // If budget is exceeded, force the cheapest model - if (budget.Exceeded) + if (softBudget.Exceeded) { Console.WriteLine(" [Router] Budget exceeded — forcing fast tier."); chain = [fastModel]; @@ -50,7 +50,7 @@ async Task RoutingMiddleware( var response = await client.GetResponseAsync( messages.ToList(), options, cancellationToken); - budget.RecordUsage(modelId, response); + softBudget.RecordUsage(modelId, response); Console.WriteLine($" [Router] Success with: {modelId}"); return response; @@ -66,7 +66,11 @@ async Task RoutingMiddleware( // All models failed — call the original pipeline as last resort Console.WriteLine(" [Fallback] All tier models failed. Trying original pipeline."); - return await chatClient.GetResponseAsync(messages, options, cancellationToken); + var fallbackResponse = await chatClient.GetResponseAsync(messages, options, cancellationToken); + // chatClient wraps fastModel.Client (see the pipeline built below), so the cost belongs to + // fastModel.ModelId — not to whichever tier was last attempted in the chain above. + softBudget.RecordUsage(fastModel.ModelId, fallbackResponse); + return fallbackResponse; } async Task BudgetEnforcementMiddleware( @@ -78,7 +82,7 @@ async Task BudgetEnforcementMiddleware( { // Soft budget: only refuse work that would need the expensive tier — // simple queries still run and RoutingMiddleware forces the fast tier for them. - if (QueryRouter.RefuseForBudget(messages, budget.Exceeded)) + if (QueryRouter.RefuseForBudget(messages, softBudget.Exceeded)) { Console.WriteLine(" [BudgetMiddleware] Budget exceeded. Refusing expensive-tier work."); return new AgentResponse([ @@ -125,4 +129,4 @@ You are a helpful support agent. Answer user questions concisely and accurately. Console.WriteLine($"Agent: {result}"); } -Console.WriteLine($"\nTotal estimated cost: {budget.TotalCostCents:F2}¢"); \ No newline at end of file +Console.WriteLine($"\nTotal estimated cost: {softBudget.TotalCostCents:F2}¢"); \ No newline at end of file From 00e88002c1274082c56cb862ea246ab8251eadbb Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 16:42:02 +0200 Subject: [PATCH 14/21] ci: update and pin actions by commit sha Every action moves to its current major and is pinned by the commit SHA that tag points at, with the version in a trailing comment. A mutable tag is a supply-chain hole: whoever can move the tag can run code in a workflow that holds packages: write on this repo. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- .github/workflows/build.yml | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 61e6ca0..b3c1124 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -10,8 +10,8 @@ jobs: build: runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 - - uses: actions/setup-dotnet@v4 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - uses: actions/setup-dotnet@a98b56852c35b8e3190ac28c8c2271da59106c68 # v6.0.0 with: dotnet-version: 10.0.x - run: dotnet build "Agentic Patterns.slnx" --configuration Release @@ -24,17 +24,17 @@ jobs: contents: read packages: write steps: - - uses: actions/checkout@v4 - - uses: docker/setup-qemu-action@v3 - - uses: docker/setup-buildx-action@v3 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4.2.0 + - uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0 - if: github.event_name == 'push' - uses: docker/login-action@v3 + uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0 with: registry: ghcr.io username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} - id: meta - uses: docker/metadata-action@v5 + uses: docker/metadata-action@dc802804100637a589fabce1cb79ff13a1411302 # v6.2.0 with: images: ghcr.io/${{ github.repository }} tags: | @@ -42,7 +42,7 @@ jobs: type=semver,pattern={{version}} type=semver,pattern={{major}}.{{minor}} type=sha,prefix=sha- - - uses: docker/build-push-action@v6 + - uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0 with: context: . platforms: ${{ github.event_name == 'push' && 'linux/amd64,linux/arm64' || 'linux/amd64' }} From 8ed43e485da203f91aad14dc46286b2c0965e5b9 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 16:47:18 +0200 Subject: [PATCH 15/21] fix(resource-aware): name the SK twin's budget soft too Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- ResourceAwareOptimization.SemanticKernel/Program.cs | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/ResourceAwareOptimization.SemanticKernel/Program.cs b/ResourceAwareOptimization.SemanticKernel/Program.cs index 0695e3b..ad352ca 100644 --- a/ResourceAwareOptimization.SemanticKernel/Program.cs +++ b/ResourceAwareOptimization.SemanticKernel/Program.cs @@ -31,7 +31,7 @@ var kernel = builder.Build(); // 2¢ budget — deliberately small so the demo actually trips it on the reasoning query -var budgetTracker = new BudgetTracker(2); +var softBudgetTracker = new BudgetTracker(2); // Maps a service id back to its model for cost lookup var modelForService = new Dictionary @@ -96,7 +96,7 @@ async Task HandleQueryAsync(string userQuery) var response = await chatService.GetChatMessageContentAsync( history, settings, kernel); - budgetTracker.Record(modelForService[sid], response); + softBudgetTracker.Record(modelForService[sid], response); Console.WriteLine($" [Router] Success with: {sid}"); return response.Content ?? "No response generated."; @@ -124,11 +124,11 @@ async Task HandleQueryAsync(string userQuery) var answer = await HandleQueryAsync(query); Console.WriteLine($"Agent: {answer}"); - if (budgetTracker.BudgetExceeded) + if (softBudgetTracker.BudgetExceeded) { Console.WriteLine("\n?Budget limit reached. Skipping remaining queries."); break; } } -Console.WriteLine($"\nTotal estimated cost: {budgetTracker.TotalCostCents:F2}¢"); \ No newline at end of file +Console.WriteLine($"\nTotal estimated cost: {softBudgetTracker.TotalCostCents:F2}¢"); \ No newline at end of file From c3c8f957427bddbcb385c69fc7af78b4bf073757 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 17:21:14 +0200 Subject: [PATCH 16/21] fix(react): enforce the tool-call budget with Terminate, not an exception MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ToolCallBudgetFilter was an IFunctionInvocationFilter that threw. SK 1.79's FunctionCallsProcessor wraps every auto-invoked call in a catch-all that turns any exception into a tool-result error message and keeps looping, so the throw blocked the tool body but not the loop, and handed the model the budget refusal as tool output to paraphrase — the exact failure mode task 4.3 removed from ToolAuthorization. Measured against real SK 1.79 with a stub endpoint that always requests a tool: before (IFunctionInvocationFilter + throw): inner 10, model calls 129, 118 tool messages reading "Error: ... Tool-call budget of 10 exhausted." after (IAutoFunctionInvocationFilter + context.Terminate = true): inner 10, model calls 11, no refusal text reaches the model Same mechanism GoalMonitoringFilter already used two projects away. SK returns normally after Terminate, so the stop leaves via a BudgetExhausted flag that Program.cs reads for its PARTIAL block; ToolCallBudgetExceededException and the now-unreachable catch are deleted. The filter instance is registered per run, so the counter's lifetime matches the per-run budget the doc claims. AutoFunctionInvocationContext has a public 5-arg constructor, so the tests now drive OnAutoFunctionInvocationAsync directly against the type the runtime uses; the GuardAsync seam and its ponytail comment are gone. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- .../ToolCallBudgetFilterTests.cs | 101 ++++++++++++------ .../patterns/ReasoningAndActing.md | 30 ++++-- ReasoningAndActing/Program.cs | 33 +++--- .../ToolCallBudgetExceededException.cs | 10 -- ReasoningAndActing/ToolCallBudgetFilter.cs | 48 ++++++--- 5 files changed, 138 insertions(+), 84 deletions(-) delete mode 100644 ReasoningAndActing/ToolCallBudgetExceededException.cs diff --git a/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs b/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs index 849ae57..467b7a8 100644 --- a/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs +++ b/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs @@ -1,3 +1,5 @@ +using Microsoft.SemanticKernel; +using Microsoft.SemanticKernel.ChatCompletion; using ReasoningAndActing; using Xunit; @@ -5,61 +7,92 @@ namespace AgenticPatterns.Tests; public class ToolCallBudgetFilterTests { + // A real AutoFunctionInvocationContext — the exact type SK 1.79 hands the filter at runtime. + // Its 5-argument constructor is public, so no test seam is needed: these tests drive + // OnAutoFunctionInvocationAsync itself, which is the only method the runtime ever calls. + private static AutoFunctionInvocationContext NewContext() + { + var kernel = new Kernel(); + var function = KernelFunctionFactory.CreateFromMethod(() => "tool ran", "Tool"); + return new AutoFunctionInvocationContext( + kernel, function, new FunctionResult(function), new ChatHistory(), + new ChatMessageContent(AuthorRole.Assistant, "calling a tool")); + } + + private static async Task<(bool InnerRan, AutoFunctionInvocationContext Context)> InvokeAsync( + ToolCallBudgetFilter filter) + { + var context = NewContext(); + var innerRan = false; + await filter.OnAutoFunctionInvocationAsync(context, _ => + { + innerRan = true; + return Task.CompletedTask; + }); + return (innerRan, context); + } + [Fact] - public async Task TenthCallIsAllowed_EleventhThrows() + public async Task TenthCallRuns_EleventhIsRefused() { var filter = new ToolCallBudgetFilter(); - var calls = 0; for (var i = 0; i < ToolCallBudgetFilter.MaxToolCalls; i++) - await filter.GuardAsync(() => - { - calls++; - return Task.CompletedTask; - }); - - Assert.Equal(ToolCallBudgetFilter.MaxToolCalls, calls); + { + var (ran, ctx) = await InvokeAsync(filter); + Assert.True(ran); + Assert.False(ctx.Terminate); + Assert.False(filter.BudgetExhausted); + } - await Assert.ThrowsAsync(() => - filter.GuardAsync(() => - { - calls++; - return Task.CompletedTask; - })); + var (eleventhRan, eleventhCtx) = await InvokeAsync(filter); - // The 11th call must not have reached the inner delegate. - Assert.Equal(ToolCallBudgetFilter.MaxToolCalls, calls); + // The 11th tool body must not run... + Assert.False(eleventhRan); + // ...and the auto-invocation loop must stop, rather than the model being handed an error + // to paraphrase and carrying on. Terminate is the only stop SK 1.79 honours. + Assert.True(eleventhCtx.Terminate); + Assert.True(filter.BudgetExhausted); } [Fact] - public async Task OverBudgetCall_DoesNotInvokeInnerDelegate() + public async Task OverBudgetCall_DoesNotThrow() { + // Throwing is what the runtime swallows: SK converts the exception into a tool-result + // error message and keeps looping. The filter must return normally instead. var filter = new ToolCallBudgetFilter(); for (var i = 0; i < ToolCallBudgetFilter.MaxToolCalls; i++) - await filter.GuardAsync(() => Task.CompletedTask); + await InvokeAsync(filter); - var innerInvoked = false; + var exception = await Record.ExceptionAsync(() => InvokeAsync(filter)); - await Assert.ThrowsAsync(() => - filter.GuardAsync(() => - { - innerInvoked = true; - return Task.CompletedTask; - })); - - Assert.False(innerInvoked); + Assert.Null(exception); } [Fact] - public async Task ExceptionMessage_NamesTheBudget() + public async Task StopReason_NamesTheBudget() { var filter = new ToolCallBudgetFilter(); - for (var i = 0; i < ToolCallBudgetFilter.MaxToolCalls; i++) - await filter.GuardAsync(() => Task.CompletedTask); + for (var i = 0; i <= ToolCallBudgetFilter.MaxToolCalls; i++) + await InvokeAsync(filter); + + Assert.True(filter.BudgetExhausted); + Assert.Contains(ToolCallBudgetFilter.MaxToolCalls.ToString(), filter.StopReason); + } + + [Fact] + public async Task BudgetIsPerInstance_NotPerProcess() + { + // Program.cs registers one instance per run; a second instance is a second run and must + // start with a full budget. + var exhausted = new ToolCallBudgetFilter(); + for (var i = 0; i <= ToolCallBudgetFilter.MaxToolCalls; i++) + await InvokeAsync(exhausted); + Assert.True(exhausted.BudgetExhausted); - var ex = await Assert.ThrowsAsync(() => - filter.GuardAsync(() => Task.CompletedTask)); + var (ran, ctx) = await InvokeAsync(new ToolCallBudgetFilter()); - Assert.Contains(ToolCallBudgetFilter.MaxToolCalls.ToString(), ex.Message); + Assert.True(ran); + Assert.False(ctx.Terminate); } } diff --git a/PatternExplorer/patterns/ReasoningAndActing.md b/PatternExplorer/patterns/ReasoningAndActing.md index 7583897..b63ba23 100644 --- a/PatternExplorer/patterns/ReasoningAndActing.md +++ b/PatternExplorer/patterns/ReasoningAndActing.md @@ -57,10 +57,20 @@ flowchart LR SK 1.79 exposes no max-auto-invoke setting on `FunctionChoiceBehavior`, so the system prompt still carries *"Use at most 10 tool calls before giving your final answer."* — but that sentence is only a hint the model can ignore. The actual control is `ToolCallBudgetFilter`, an -`IFunctionInvocationFilter` registered on the local kernel: it counts every tool call and throws -once the 11th would run, so the call that would exceed the budget never executes. The top-level -`try`/`catch` turns that exception into a `PARTIAL` result — the same shape **Bounded Execution** -uses for its own hard stops — instead of letting the process crash mid-answer. +`IAutoFunctionInvocationFilter` registered on the local kernel as a single instance created for +this run — the counter lives on the instance, so the budget is per run, not per process. It counts +every auto-invoked call; the 10th runs, and the 11th never reaches the tool because the filter sets +`context.Terminate = true`, which ends SK's auto-invocation loop. + +Throwing from a filter does *not* end it. SK 1.79 wraps every auto-invoked call in a catch-all that +converts any exception into a tool-result error message and keeps looping, so a throwing filter +blocks the tool body, hands the model its own budget refusal as tool output to paraphrase, and lets +the loop run on to SK's internal 128-round ceiling — measured at 129 model calls against a stub that +always requests a tool, versus 11 with `Terminate`. `Terminate` is the stop SK honours, and it is +the same mechanism **Goal Settings and Monitoring**'s `GoalMonitoringFilter` uses for its own +max-iteration bound. Because SK returns normally after terminating (with empty content), the filter +exposes the stop as a `BudgetExhausted` flag rather than an exception; `Program.cs` reads it and +prints a `PARTIAL` result — the same shape **Bounded Execution** uses for its own hard stops. ## Key APIs @@ -70,16 +80,18 @@ uses for its own hard stops — instead of letting the process crash mid-answer. - `new OpenAIPromptExecutionSettings { FunctionChoiceBehavior = FunctionChoiceBehavior.Auto() }`. - `chatService.GetChatMessageContentAsync(history, settings, kernel)` — the kernel argument is what makes auto-invocation possible. -- `ToolCallBudgetFilter : IFunctionInvocationFilter` — the host-enforced call cap; see - **Bounded Execution** for the fuller pattern of hard, host-enforced run limits. +- `ToolCallBudgetFilter : IAutoFunctionInvocationFilter` — the host-enforced call cap, stopping + the loop with `context.Terminate = true`; see **Bounded Execution** for the fuller pattern of + hard, host-enforced run limits. ## What to watch in the output The demo prints a single block prefixed `ReAct Agent:`. The tool calls themselves are not logged, so the tell is in the content: the answer should quote the exact simulated figures — 40.1 million and 26.5 million — and a ratio near 1.5, none of which the model could produce without calling -the plugin. If the model instead wanders past the tool-call budget, the output switches to -`Result status: PARTIAL` with a `Stop reason:` line naming the exhausted budget — proof the bound -stopped the loop rather than the model choosing to stop. **Tool Use** is the single-call +the plugin. If the model instead wanders past the tool-call budget, the answer block is +replaced by `Result status: PARTIAL`, a `Stop reason:` line naming the exhausted budget, and an +explicit incomplete label — proof the bound stopped the loop rather than the model choosing to +stop. **Tool Use** is the single-call foundation this loops over; **Middleware** shows how to log every invocation so the reasoning-acting alternation becomes visible. diff --git a/ReasoningAndActing/Program.cs b/ReasoningAndActing/Program.cs index 5a68b13..c5ef453 100644 --- a/ReasoningAndActing/Program.cs +++ b/ReasoningAndActing/Program.cs @@ -7,7 +7,10 @@ // Local kernel so the shared Settings.Kernel singleton stays unmodified var reactBuilder = Settings.CreateKernelBuilder(); -reactBuilder.Services.AddSingleton(); +// One filter instance for this one run: the counter lives on the instance, so registering the +// instance (not the type) keeps the budget per-run rather than per-process. +var toolCallBudget = new ToolCallBudgetFilter(); +reactBuilder.Services.AddSingleton(toolCallBudget); var reactKernel = reactBuilder.Build(); // Register tools the agent can use mid-reasoning @@ -37,25 +40,27 @@ Use at most 10 tool calls before giving your final answer. { // SK 1.79 exposes no max-auto-invoke option on FunctionChoiceBehavior/FunctionChoiceBehaviorOptions, // so the "at most 10 tool calls" instruction above is only a hint to the model. ToolCallBudgetFilter, - // registered on reactKernel above, is the actual control: it throws once the budget is exhausted, - // stopping the loop regardless of what the model intends to do next. + // registered on reactKernel above, is the actual control: the 11th call is refused and the + // auto-invocation loop is terminated, regardless of what the model intends to do next. FunctionChoiceBehavior = FunctionChoiceBehavior.Auto() }; // The agent will autonomously call GetPopulation() for each country, // then reason about the results to compute the ratio. -try -{ - var reactResponse = await reactService.GetChatMessageContentAsync( - reactHistory, reactSettings, reactKernel); - Console.WriteLine($"ReAct Agent:\n{reactResponse.Content}\n"); -} -catch (ToolCallBudgetExceededException ex) +var reactResponse = await reactService.GetChatMessageContentAsync( + reactHistory, reactSettings, reactKernel); + +if (toolCallBudget.BudgetExhausted) { // Same shape as BoundedExecution: a PARTIAL result, the stop reason, and an explicit - // incomplete label rather than silently truncated output. + // incomplete label rather than silently truncated output. SK returns normally after + // Terminate (with empty content), so the filter's flag — not an exception — is the signal. Console.WriteLine("Result status: PARTIAL"); - Console.WriteLine($"Stop reason: {ex.Message}"); + Console.WriteLine($"Stop reason: {toolCallBudget.StopReason}"); Console.WriteLine("ReAct Agent:\nStopped at the tool-call budget before a final answer was reached; " + - "any reasoning gathered so far is incomplete.\n"); -} \ No newline at end of file + "any reasoning gathered so far is incomplete.\n"); +} +else +{ + Console.WriteLine($"ReAct Agent:\n{reactResponse.Content}\n"); +} diff --git a/ReasoningAndActing/ToolCallBudgetExceededException.cs b/ReasoningAndActing/ToolCallBudgetExceededException.cs deleted file mode 100644 index f22ed3d..0000000 --- a/ReasoningAndActing/ToolCallBudgetExceededException.cs +++ /dev/null @@ -1,10 +0,0 @@ -namespace ReasoningAndActing; - -/// -/// Thrown by once the tool-call budget is exhausted. A typed -/// exception, not a message-substring match, so the catch site in Program.cs can't go silently -/// dead if the message text is ever reworded — see BoundedExecution.AgentFramework's -/// BudgetExceededException for the same shape. -/// -public sealed class ToolCallBudgetExceededException(int maxToolCalls) - : InvalidOperationException($"Tool-call budget of {maxToolCalls} exhausted."); diff --git a/ReasoningAndActing/ToolCallBudgetFilter.cs b/ReasoningAndActing/ToolCallBudgetFilter.cs index 509e60a..7736b18 100644 --- a/ReasoningAndActing/ToolCallBudgetFilter.cs +++ b/ReasoningAndActing/ToolCallBudgetFilter.cs @@ -1,35 +1,49 @@ -using Microsoft.SemanticKernel; +using Microsoft.SemanticKernel; namespace ReasoningAndActing; /// /// Hard bound on tool calls: the prompt asks the model to stop at 10, this enforces it. /// -// ponytail: GuardAsync(Func) is a context-free stand-in for OnFunctionInvocationAsync so -// the counting/throwing logic is unit-testable. Upgrade path: SemanticKernel's -// FunctionInvocationContext constructor is internal, so a test can't build a real one - if a -// future SK version makes it public, this seam can be dropped and the test can drive -// OnFunctionInvocationAsync directly. -public class ToolCallBudgetFilter : IFunctionInvocationFilter +/// +/// This must be an setting context.Terminate, +/// not an that throws. SK 1.79's +/// FunctionCallsProcessor.ExecuteFunctionCallAsync wraps every auto-invoked call in a +/// catch-all that turns any exception into a tool-result error message and keeps looping, so a +/// throwing filter blocks the tool body but not the loop — and hands the model the refusal to +/// paraphrase. Terminate is the only stop SK honours; see +/// GoalSettingsAndMonitoring.SemanticKernel.GoalMonitoringFilter for the same mechanism. +/// +/// The counter lives on the instance, so the budget is per filter instance: register one instance +/// per run (Program.cs does), not the type as a process-wide singleton. +/// +public class ToolCallBudgetFilter : IAutoFunctionInvocationFilter { public const int MaxToolCalls = 10; private int _toolCalls; + /// True once a call was refused because the budget was exhausted. SK returns + /// normally after Terminate, so this flag is the only route the stop has out. + public bool BudgetExhausted { get; private set; } + + public string StopReason => $"Tool-call budget of {MaxToolCalls} exhausted."; + /// - /// Increments the call counter and invokes only while under budget. - /// Throws before calling once the budget is exhausted, so the call - /// that would exceed it never runs. + /// Counts every auto-invoked tool call and passes it through while under budget. The call that + /// would exceed the budget never reaches , and terminates the + /// auto-invocation loop instead of feeding the model an error to reason about. /// - public async Task GuardAsync(Func next) + public async Task OnAutoFunctionInvocationAsync( + AutoFunctionInvocationContext context, Func next) { if (Interlocked.Increment(ref _toolCalls) > MaxToolCalls) - throw new ToolCallBudgetExceededException(MaxToolCalls); + { + BudgetExhausted = true; + context.Terminate = true; + return; + } - await next(); + await next(context); } - - public Task OnFunctionInvocationAsync( - FunctionInvocationContext context, Func next) => - GuardAsync(() => next(context)); } From f9b000045f51fbf258c17127638c921f054eed55 Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 17:25:30 +0200 Subject: [PATCH 17/21] fix(judge): an unreadable rubric verdict is Indeterminate, not a zero MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RubricJudgeEvaluator deserialized the judge's reply with no try/catch: "" and "not json" threw JsonException and crashed the whole LLMAsJudge run before the pairwise probe ran, and "{}" produced a NumericMetric of 0 — below the rubric's own floor of 1, so an unparseable verdict was recorded as worse than the worst possible answer. Only the literal "null" ever reached the ?? fallback. ParseVerdict now catches JsonException and rejects any score outside 1-5, returning null; the evaluator turns that into a value-less NumericMetric with an Indeterminate reason, and Program.cs prints INDETERMINATE rather than a blank column that reads like a zero. Same ruling as JudgeParsing.Parse in task 4.5. Tests drive the real evaluator through EvaluateAsync with a scripted judge client, covering all four measured inputs plus whitespace, wrong-shaped JSON, and out-of-range scores. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- AgenticPatterns.Tests/LlmAsJudgeTests.cs | 55 +++++++++++++++++++ LLMAsJudge.AgentFramework/Program.cs | 4 +- .../RubricJudgeEvaluator.cs | 42 ++++++++++++-- PatternExplorer/patterns/LLMAsJudge.md | 10 +++- 4 files changed, 104 insertions(+), 7 deletions(-) diff --git a/AgenticPatterns.Tests/LlmAsJudgeTests.cs b/AgenticPatterns.Tests/LlmAsJudgeTests.cs index 089cbdb..28e79f9 100644 --- a/AgenticPatterns.Tests/LlmAsJudgeTests.cs +++ b/AgenticPatterns.Tests/LlmAsJudgeTests.cs @@ -1,4 +1,6 @@ using LLMAsJudge.AgentFramework; +using Microsoft.Extensions.AI; +using Microsoft.Extensions.AI.Evaluation; using Xunit; namespace AgenticPatterns.Tests; @@ -63,3 +65,56 @@ public void SummarizeBiasRateIsZeroWhenAllIndeterminate() Assert.Equal(0.0, report.PositionBiasRate); } } + +// Drives the real RubricJudgeEvaluator end to end - a scripted judge reply through +// EvaluateAsync - because the defect lived in how the evaluator reads that reply. +public class RubricJudgeEvaluatorTests +{ + private static async Task JudgeSaysAsync(string judgeReply) + { + var client = new ScriptedChatClient( + new ChatResponse(new ChatMessage(ChatRole.Assistant, judgeReply))); + + var result = await new RubricJudgeEvaluator().EvaluateAsync( + [new ChatMessage(ChatRole.User, "What warranty do the laptops come with?")], + new ChatResponse(new ChatMessage(ChatRole.Assistant, "Two years.")), + new ChatConfiguration(client)); + + return result.Get(RubricJudgeEvaluator.RubricScoreMetricName); + } + + [Theory] + [InlineData("")] // truncated to nothing + [InlineData("not json")] // prose instead of JSON + [InlineData("{}")] // JSON, but no score + [InlineData("null")] // literal JSON null + [InlineData(" ")] // whitespace only + [InlineData("[1,2,3]")] // JSON of the wrong shape + [InlineData("{\"score\":0,\"justification\":\"x\"}")] // below the rubric floor + [InlineData("{\"score\":9,\"justification\":\"x\"}")] // above the rubric ceiling + public async Task UnreadableVerdictIsIndeterminate_NeverThrows_NeverANumber(string judgeReply) + { + var metric = await JudgeSaysAsync(judgeReply); + + // Indeterminate is "no value", not 0: 0 is below the rubric's own floor of 1, so scoring + // an unreadable verdict as a number would rank it worse than the worst possible answer. + Assert.Null(metric.Value); + Assert.Contains("Indeterminate", metric.Reason); + } + + [Fact] + public async Task ParseableVerdictKeepsScoreAndJustification() + { + var metric = await JudgeSaysAsync("{\"score\":4,\"justification\":\"Accurate but terse.\"}"); + + Assert.Equal(4, metric.Value); + Assert.Equal("Accurate but terse.", metric.Reason); + } + + [Fact] + public async Task ScoreIsNeverBelowTheRubricFloor() + { + foreach (var reply in new[] { "", "not json", "{}", "null", "{\"score\":-3}" }) + Assert.True(await JudgeSaysAsync(reply) is { Value: null or >= 1 }); + } +} diff --git a/LLMAsJudge.AgentFramework/Program.cs b/LLMAsJudge.AgentFramework/Program.cs index 0305862..a613d69 100644 --- a/LLMAsJudge.AgentFramework/Program.cs +++ b/LLMAsJudge.AgentFramework/Program.cs @@ -37,7 +37,9 @@ var result = await eval.EvaluateAsync(conversation, response, chatConfig, ctx is null ? null : [ctx]); var metric = result.Get(result.Metrics.Keys.First()); - Console.WriteLine($" {name,-14}: {metric.Value} ({metric.Reason})"); + // A null value means the judge's verdict could not be read — print that plainly rather + // than a blank column that reads like a zero. + Console.WriteLine($" {name,-14}: {metric.Value?.ToString() ?? "INDETERMINATE"} ({metric.Reason})"); } Console.WriteLine(); } diff --git a/LLMAsJudge.AgentFramework/RubricJudgeEvaluator.cs b/LLMAsJudge.AgentFramework/RubricJudgeEvaluator.cs index b77e72b..c954b99 100644 --- a/LLMAsJudge.AgentFramework/RubricJudgeEvaluator.cs +++ b/LLMAsJudge.AgentFramework/RubricJudgeEvaluator.cs @@ -12,7 +12,36 @@ public sealed class RubricJudgeEvaluator : IEvaluator public const string RubricScoreMetricName = "Rubric Score"; public IReadOnlyCollection EvaluationMetricNames => [RubricScoreMetricName]; - private sealed record Verdict(int Score, string Justification); + // The rubric's own floor and ceiling. A score outside them is not a verdict. + private const int MinScore = 1; + private const int MaxScore = 5; + + private sealed record Verdict(int Score, string? Justification); + + private static readonly JsonSerializerOptions JsonOptions = new(JsonSerializerDefaults.Web); + + /// + /// Never throws, for any input. An empty, truncated, non-JSON, or score-less judge reply is + /// null — indeterminate — and not a Verdict(0, …): 0 sits below the rubric's own + /// floor of 1, so recording an unreadable verdict as a number would score it worse than the + /// worst possible answer instead of admitting the judge was not understood. Empty and + /// whitespace input arrive here as a like any other malformed + /// reply, and "null" deserializes to a null Verdict. + /// + private static Verdict? ParseVerdict(string text) + { + Verdict? verdict; + try + { + verdict = JsonSerializer.Deserialize(text, JsonOptions); + } + catch (JsonException) + { + return null; + } + + return verdict is { Score: >= MinScore and <= MaxScore } ? verdict : null; + } public async ValueTask EvaluateAsync( IEnumerable messages, @@ -38,11 +67,14 @@ [new ChatMessage(ChatRole.User, prompt)], new ChatOptions { Temperature = 0f, ResponseFormat = ChatResponseFormat.Json }, cancellationToken); - var verdict = JsonSerializer.Deserialize(response.Text, - new JsonSerializerOptions(JsonSerializerDefaults.Web)) - ?? new Verdict(0, "Judge returned unparseable output."); + // An unreadable judge reply is reported as a metric with no value at all, so downstream + // averages and gates see "not measured" rather than a number the judge never gave. + var metric = ParseVerdict(response.Text) is { } verdict + ? new NumericMetric(RubricScoreMetricName, verdict.Score, verdict.Justification) + : new NumericMetric(RubricScoreMetricName, value: null, + reason: $"Indeterminate: the judge's reply was not a parseable " + + $"{MinScore}-{MaxScore} rubric verdict."); - var metric = new NumericMetric(RubricScoreMetricName, verdict.Score, verdict.Justification); return new EvaluationResult(metric); } } diff --git a/PatternExplorer/patterns/LLMAsJudge.md b/PatternExplorer/patterns/LLMAsJudge.md index 10f6840..812af08 100644 --- a/PatternExplorer/patterns/LLMAsJudge.md +++ b/PatternExplorer/patterns/LLMAsJudge.md @@ -60,6 +60,12 @@ Each answer is scored by the three quality evaluators plus `RubricJudgeEvaluator as a `NumericMetric`. `GroundednessEvaluator` receives the policy text as its context so it can check the answer against the source rather than against the model's own memory. +The rubric judge never throws and never invents a number. A reply that is empty, truncated, not +JSON, missing a score, or scored outside 1–5 yields a `NumericMetric` with **no value** and an +`Indeterminate` reason, printed as `INDETERMINATE`. A numeric `0` would be worse than that: it sits +below the rubric's own floor of 1, so an unreadable verdict would be recorded as *worse than the +worst possible answer* and would drag any average computed over the metric. + ```mermaid flowchart LR A[SupportAgent answer] --> E[Quality evaluators
Relevance, Coherence, Groundedness] @@ -85,6 +91,7 @@ precise answer regardless of slot, so any determinate flip is the bias signal. | `GroundednessEvaluatorContext(policy)` | Supplies grounding source to the evaluator | | `IEvaluator.EvaluateAsync(messages, response, chatConfig, contexts)` | The evaluator contract | | `result.Get(name)` | Reads a score back out of the `EvaluationResult` | +| `RubricJudgeEvaluator` | Custom 1–5 rubric judge; an unreadable verdict becomes a value-less metric, never a 0 | | `JudgeParsing.Parse(json)` | Strict verdict parse: `A` / `B` / `Indeterminate`, never throws | | `JudgeParsing.Summarize(picks)` | Win/loss/indeterminate counts plus a bias rate that excludes indeterminates | @@ -96,7 +103,8 @@ dotnet run --project LLMAsJudge.AgentFramework -- --selfcheck # offline bias-l ## What to watch in the output The first block prints each answer with four scored lines (`Relevance`, `Coherence`, -`Groundedness`, `RubricScore`) and the judge's reason per metric. The second block runs five +`Groundedness`, `RubricScore`) and the judge's reason per metric; a line reading `INDETERMINATE` +means that judge's reply could not be read, not that the answer scored badly. The second block runs five randomized orderings and prints the win/loss/indeterminate counts, the position-bias rate (of determinate verdicts only), and a `► Position bias` verdict. A well-behaved judge on a clear-cut pair should pick the precise answer regardless of slot — if it flips, you have just measured your From 79aa025448bc34087c7726219ee8403a0c68cc6a Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 17:30:12 +0200 Subject: [PATCH 18/21] fix(judge): measure position bias per slot instead of folding the slot away MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolve() collapsed "which slot the reference candidate sat in" into "did it win" before the statistic was formed, so five trials drawn in one slot produced the same number as five alternating ones and the randomisation did no work. A judge that is simply wrong — it prefers the vague answer in both slots — reported "Position bias DETECTED: 100%". Summarize now takes Trial(slot, verdict), computes the reference candidate's win rate within each slot, and reports the absolute difference as PositionSwing. Always-picks-slot-A swings 1; always-wrong swings 0; a slot with no determinate verdict makes the swing null (not measurable) rather than 0. The five orderings are a balanced 3/2 shuffle instead of five coin flips, which also removes the 6.25% all-one-slot case. Doc frontmatter summary, body, README row and console labels all say swing now. --selfcheck still runs the real Summarize offline against fixed inputs. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- AgenticPatterns.Tests/LlmAsJudgeTests.cs | 92 +++++++++++++++++++++-- LLMAsJudge.AgentFramework/JudgeParsing.cs | 49 +++++++++--- LLMAsJudge.AgentFramework/Program.cs | 62 +++++++++++---- PatternExplorer/patterns/LLMAsJudge.md | 53 ++++++++----- README.md | 2 +- 5 files changed, 206 insertions(+), 52 deletions(-) diff --git a/AgenticPatterns.Tests/LlmAsJudgeTests.cs b/AgenticPatterns.Tests/LlmAsJudgeTests.cs index 28e79f9..fa52581 100644 --- a/AgenticPatterns.Tests/LlmAsJudgeTests.cs +++ b/AgenticPatterns.Tests/LlmAsJudgeTests.cs @@ -37,10 +37,19 @@ public void ResolveTranslatesVerdictAndPositionIntoReferenceWin( public void ResolveIsNullForIndeterminate() => Assert.Null(JudgeParsing.Resolve(Preference.Indeterminate, referenceInPositionA: true)); + private static Trial InA(Preference verdict) => new(ReferenceInPositionA: true, verdict); + private static Trial InB(Preference verdict) => new(ReferenceInPositionA: false, verdict); + [Fact] public void SummarizeCountsWinsAndIndeterminates() { - var report = JudgeParsing.Summarize([true, false, null, true, true]); + var report = JudgeParsing.Summarize([ + InA(Preference.A), // reference in A, judge picks A -> reference wins + InA(Preference.B), // reference in A, judge picks B -> other wins + InA(Preference.Indeterminate), + InB(Preference.B), // reference in B, judge picks B -> reference wins + InB(Preference.B) + ]); Assert.Equal(3, report.ReferenceWins); Assert.Equal(1, report.OtherWins); @@ -48,21 +57,88 @@ public void SummarizeCountsWinsAndIndeterminates() } [Fact] - public void SummarizeExcludesIndeterminateFromBiasRate() + public void JudgeThatAlwaysPicksTheSameSlot_IsFullPositionBias() + { + var report = JudgeParsing.Summarize([ + InA(Preference.A), InA(Preference.A), InB(Preference.A), InB(Preference.A) + ]); + + // Reference wins 100% of the time it sits in A, 0% of the time it sits in B. + Assert.Equal(1.0, report.PositionSwing); + } + + [Fact] + public void JudgeThatIsSimplyWrong_IsNotPositionBias() { - // 4 determinate results, 1 of them a flip away from the reference: rate is 1/4, not 1/5. - var report = JudgeParsing.Summarize([true, true, true, false, null]); + // The judge prefers the weaker candidate every single time, in both slots. That is a bad + // judge, not a position-dependent one - the pre-fix rate reported this as 100% bias + // because it folded the slot away before forming the statistic. + var report = JudgeParsing.Summarize([ + InA(Preference.B), InA(Preference.B), InB(Preference.A), InB(Preference.A) + ]); + + Assert.Equal(4, report.OtherWins); + Assert.Equal(0.0, report.PositionSwing); + } + + [Fact] + public void ConsistentJudge_HasNoPositionSwing() + { + var report = JudgeParsing.Summarize([ + InA(Preference.A), InA(Preference.A), InB(Preference.B), InB(Preference.B) + ]); + + Assert.Equal(0.0, report.PositionSwing); + } - Assert.Equal(0.25, report.PositionBiasRate); + [Fact] + public void IdenticalOutcomesInDifferentSlotsProduceDifferentStatistics() + { + // The reference candidate wins every trial in both runs - identical outcomes - but one run + // never left slot A. The pre-fix rate folded the slot away before forming the number and + // so reported the same value for both, making the randomisation do no work at all. + var oneSlot = JudgeParsing.Summarize([InA(Preference.A), InA(Preference.A), InA(Preference.A)]); + var bothSlots = JudgeParsing.Summarize([InA(Preference.A), InB(Preference.B), InA(Preference.A)]); + + Assert.Equal(oneSlot.ReferenceWins, bothSlots.ReferenceWins); + Assert.NotEqual(oneSlot.PositionSwing, bothSlots.PositionSwing); + } + + [Fact] + public void SwingIsUnmeasurableWhenOnlyOneSlotWasSampled() + { + // Five coin flips land all five trials in one slot 6.25% of the time; Program.cs uses a + // balanced 3/2 shuffle so this cannot happen there, but the statistic still refuses to + // invent a position measurement from a single slot. + var report = JudgeParsing.Summarize([InA(Preference.A), InA(Preference.B), InA(Preference.A)]); + + Assert.Null(report.PositionSwing); + } + + [Fact] + public void IndeterminateVerdictsAreExcludedFromTheSwing() + { + // Both slots: every determinate verdict is a reference win, so both rates are 1 and the + // swing is 0. The unparseable verdicts are piled onto slot B only - count them in the + // denominator and slot B's rate drops to 1/3, inventing a swing out of noise. + var report = JudgeParsing.Summarize([ + InA(Preference.A), + InB(Preference.B), InB(Preference.Indeterminate), InB(Preference.Indeterminate) + ]); + + Assert.Equal(2, report.Indeterminate); + Assert.Equal(0.0, report.PositionSwing); } [Fact] - public void SummarizeBiasRateIsZeroWhenAllIndeterminate() + public void SwingIsUnmeasurableWhenEverythingIsIndeterminate() { - var report = JudgeParsing.Summarize([null, null, null]); + var report = JudgeParsing.Summarize([ + InA(Preference.Indeterminate), InB(Preference.Indeterminate), InA(Preference.Indeterminate) + ]); Assert.Equal(3, report.Indeterminate); - Assert.Equal(0.0, report.PositionBiasRate); + Assert.Null(report.PositionSwing); } } diff --git a/LLMAsJudge.AgentFramework/JudgeParsing.cs b/LLMAsJudge.AgentFramework/JudgeParsing.cs index eea4aa3..aa6fa59 100644 --- a/LLMAsJudge.AgentFramework/JudgeParsing.cs +++ b/LLMAsJudge.AgentFramework/JudgeParsing.cs @@ -4,7 +4,12 @@ namespace LLMAsJudge.AgentFramework; public enum Preference { A, B, Indeterminate } -// Parses judge verdicts and summarizes them across randomized position-bias orderings. +/// One pairwise trial: which slot the reference candidate sat in, and what the judge said. +/// The slot must survive into the statistic — fold it away early and the randomisation stops +/// measuring anything. +public readonly record struct Trial(bool ReferenceInPositionA, Preference Verdict); + +// Parses judge verdicts and summarizes them across balanced position orderings. public static class JudgeParsing { public static Preference Parse(string? json) @@ -41,17 +46,41 @@ public static Preference Parse(string? json) _ => null }; + /// How much the reference candidate's win rate moved when it + /// changed slots: |win rate with the reference in A − win rate with it in B|. 0 means the + /// verdict did not depend on position; 1 means it depended on nothing else. null when + /// either slot produced no determinate verdict — one slot cannot measure position bias. public readonly record struct PreferenceReport( - int ReferenceWins, int OtherWins, int Indeterminate, double PositionBiasRate); + int ReferenceWins, int OtherWins, int Indeterminate, double? PositionSwing); - // Indeterminate results (null) are excluded from the bias rate but counted separately. - public static PreferenceReport Summarize(IReadOnlyList results) + // Indeterminate verdicts are excluded from the swing but counted separately. + public static PreferenceReport Summarize(IReadOnlyList trials) { - var referenceWins = results.Count(r => r == true); - var otherWins = results.Count(r => r == false); - var indeterminate = results.Count(r => r is null); - var determinate = referenceWins + otherWins; - var biasRate = determinate == 0 ? 0.0 : (double)otherWins / determinate; - return new PreferenceReport(referenceWins, otherWins, indeterminate, biasRate); + var outcomes = trials.Select(t => Resolve(t.Verdict, t.ReferenceInPositionA)).ToList(); + + var inA = ReferenceWinRate(trials, referenceInPositionA: true); + var inB = ReferenceWinRate(trials, referenceInPositionA: false); + + return new PreferenceReport( + outcomes.Count(r => r == true), + outcomes.Count(r => r == false), + outcomes.Count(r => r is null), + inA is null || inB is null ? null : Math.Abs(inA.Value - inB.Value)); + } + + // A judge that is simply wrong — it prefers the same candidate in both slots — has a win rate + // of 0 in both and so a swing of 0: wrong is not the same defect as position-dependent, and + // this is what folding the slot away before the statistic used to hide. + private static double? ReferenceWinRate(IReadOnlyList trials, bool referenceInPositionA) + { + var determinate = trials + .Where(t => t.ReferenceInPositionA == referenceInPositionA) + .Select(t => Resolve(t.Verdict, referenceInPositionA)) + .Where(r => r is not null) + .ToList(); + + return determinate.Count == 0 + ? null + : (double)determinate.Count(r => r == true) / determinate.Count; } } diff --git a/LLMAsJudge.AgentFramework/Program.cs b/LLMAsJudge.AgentFramework/Program.cs index a613d69..9ec1062 100644 --- a/LLMAsJudge.AgentFramework/Program.cs +++ b/LLMAsJudge.AgentFramework/Program.cs @@ -50,24 +50,34 @@ var good = "TechCorp laptops come with a two-year limited warranty."; var vague = "TechCorp offers a warranty on its laptops for a period of time."; -const int orderings = 5; -var picks = new List(); -for (var i = 0; i < orderings; i++) +// A balanced 3/2 split, shuffled: both slots are always sampled, so the position statistic is +// always defined. Five independent coin flips land every trial in one slot 6.25% of the time, +// and a statistic conditioned on position cannot be computed from a single slot. +var slots = new[] { true, true, true, false, false }; +Random.Shared.Shuffle(slots); + +var trials = new List(); +foreach (var goodInPositionA in slots) { - var goodInPositionA = Random.Shared.Next(2) == 0; var candidateA = goodInPositionA ? good : vague; var candidateB = goodInPositionA ? vague : good; var verdict = await PairwiseWinnerAsync(pairwiseQuestion, candidateA, candidateB); - picks.Add(JudgeParsing.Resolve(verdict, goodInPositionA)); + trials.Add(new Trial(goodInPositionA, verdict)); } -var report = JudgeParsing.Summarize(picks); +var report = JudgeParsing.Summarize(trials); Console.WriteLine( $"Good wins: {report.ReferenceWins} Vague wins: {report.OtherWins} Indeterminate: {report.Indeterminate}"); -Console.WriteLine($"Position-bias rate (of determinate verdicts): {report.PositionBiasRate:P0}"); -Console.WriteLine(report.PositionBiasRate > 0 - ? "► Position bias DETECTED: the weaker candidate won at least once across randomized orderings." - : "► Consistent verdict across randomized orderings."); +Console.WriteLine(report.PositionSwing is { } swing + ? $"Position swing (good answer's win rate in slot A vs slot B): {swing:P0}" + : "Position swing: not measurable — one slot produced no determinate verdict."); +Console.WriteLine(report.PositionSwing switch +{ + > 0 => "► Position bias DETECTED: the same pair got a different verdict depending on which slot " + + "the better answer sat in.", + 0 => "► No position bias: the verdict did not change when the candidates swapped slots.", + _ => "► Position bias not measured this run." +}); IEnumerable<(string, IEvaluator, EvaluationContext?)> Evaluators() => [ @@ -106,11 +116,33 @@ static void SelfCheck() if (JudgeParsing.Resolve(Preference.B, referenceInPositionA: true) != false) throw new Exception("resolve B/A wrong"); if (JudgeParsing.Resolve(Preference.Indeterminate, referenceInPositionA: true) is not null) throw new Exception("resolve indeterminate wrong"); - // JudgeParsing.Summarize: same candidate wins every time -> no bias; a flip -> bias. - if (JudgeParsing.Summarize([true, true, true, true, true]).PositionBiasRate != 0) throw new Exception("false positive bias"); - if (JudgeParsing.Summarize([true, false, true, true, true]).PositionBiasRate <= 0) throw new Exception("missed bias"); - var allIndeterminate = JudgeParsing.Summarize([null, null, null]); - if (allIndeterminate.Indeterminate != 3 || allIndeterminate.PositionBiasRate != 0) throw new Exception("indeterminate handling wrong"); + // JudgeParsing.Summarize: the swing is position-conditioned, so it must separate "the judge + // depends on the slot" from "the judge is simply wrong". Fixed inputs, no randomness. + Trial InA(Preference v) => new(ReferenceInPositionA: true, v); + Trial InB(Preference v) => new(ReferenceInPositionA: false, v); + + // Judge always picks the reference candidate, whichever slot it is in: no position dependence. + if (JudgeParsing.Summarize([InA(Preference.A), InA(Preference.A), InB(Preference.B), InB(Preference.B)]) + .PositionSwing != 0) throw new Exception("false positive bias"); + + // Judge is simply wrong — always picks the other candidate, in both slots. Still no position + // dependence: the old rate reported 100% bias here, which is the defect this replaces. + var alwaysWrong = JudgeParsing.Summarize( + [InA(Preference.B), InA(Preference.B), InB(Preference.A), InB(Preference.A)]); + if (alwaysWrong.PositionSwing != 0) throw new Exception("wrongness misreported as position bias"); + if (alwaysWrong.OtherWins != 4) throw new Exception("win counting wrong"); + + // Judge always picks whatever sits in slot A: total position dependence. + if (JudgeParsing.Summarize([InA(Preference.A), InA(Preference.A), InB(Preference.A), InB(Preference.A)]) + .PositionSwing != 1) throw new Exception("missed bias"); + + // Five verdicts drawn in the same slot say nothing about position, however lopsided. + if (JudgeParsing.Summarize([InA(Preference.A), InA(Preference.B), InA(Preference.A)]) + .PositionSwing is not null) throw new Exception("one-slot swing should be unmeasurable"); + + var allIndeterminate = JudgeParsing.Summarize( + [InA(Preference.Indeterminate), InB(Preference.Indeterminate), InA(Preference.Indeterminate)]); + if (allIndeterminate.Indeterminate != 3 || allIndeterminate.PositionSwing is not null) throw new Exception("indeterminate handling wrong"); Console.WriteLine("selfcheck ok"); } diff --git a/PatternExplorer/patterns/LLMAsJudge.md b/PatternExplorer/patterns/LLMAsJudge.md index 812af08..dd844d6 100644 --- a/PatternExplorer/patterns/LLMAsJudge.md +++ b/PatternExplorer/patterns/LLMAsJudge.md @@ -1,7 +1,7 @@ --- { "title": "LLM as Judge", - "summary": "Score answers with a judge model against a rubric, and probe the judge's own position bias across randomized orderings, discarding verdicts it cannot parse.", + "summary": "Score answers with a judge model against a rubric, and measure the judge's own position bias by comparing its verdicts across balanced candidate orderings, discarding verdicts it cannot parse.", "category": "Evaluation", "projects": [ { "flavor": "AgentFramework", "path": "LLMAsJudge.AgentFramework" } @@ -19,10 +19,17 @@ sample scores answers three ways: built-in **quality evaluators** (`Relevance`, written directly against `IEvaluator`, and a **pairwise comparison** that picks the better of two answers. -The judge is not a neutral instrument, and the pairwise step exists to prove it: the same two -candidates are judged across five randomized position orderings. If the same candidate does not -win regardless of slot, the judge has **position bias** — one of its documented failure modes -alongside verbosity bias (longer looks better) and self-preference (its own style looks better). +The judge is not a neutral instrument, and the pairwise step exists to measure it: the same two +candidates are judged five times, with the better answer placed in slot A three times and slot B +twice, shuffled. The statistic is the **position swing** — how far the better answer's win rate +moves when it changes slots. A judge whose verdict does not depend on position swings 0; a judge +that just picks whatever sits in slot A swings 1. That is **position bias** — one of its documented +failure modes alongside verbosity bias (longer looks better) and self-preference (its own style +looks better). + +Crucially, a judge that is simply *wrong* — it prefers the vague answer in both slots — also swings +0. Wrongness and position-dependence are different defects, and a statistic that cannot tell them +apart is not measuring position. A judge is also not a reliable narrator of its own output format: `JudgeParsing.Parse` treats any verdict it cannot parse as its own contract — `{"winner": "A"}` or `{"winner": "B"}`, matched exactly — as `Indeterminate` rather than silently counting it as a win for either side. @@ -72,15 +79,23 @@ flowchart LR A --> R[RubricJudgeEvaluator
1-5 + justification] P[Two candidate answers] --> J[Pairwise judge x5
randomized position] J --> Parse[JudgeParsing.Parse
A / B / Indeterminate] - Parse --> B[Preference distribution
+ position-bias rate] + Parse --> B[Preference distribution
+ position swing across slots] ``` -The pairwise section pits a precise answer against a vague one across five randomized orderings, -parses each verdict with `JudgeParsing.Parse` (which returns `Indeterminate` — never a default -winner — for anything that doesn't match the `{"winner": "A"}` / `{"winner": "B"}` contract), and -reports the win distribution. `JudgeParsing.Summarize` excludes `Indeterminate` verdicts from the -position-bias rate and reports their count separately; a well-behaved judge should pick the -precise answer regardless of slot, so any determinate flip is the bias signal. +The pairwise section pits a precise answer against a vague one across five orderings, parses each +verdict with `JudgeParsing.Parse` (which returns `Indeterminate` — never a default winner — for +anything that doesn't match the `{"winner": "A"}` / `{"winner": "B"}` contract), and records each +result as a `Trial` that keeps **which slot the precise answer occupied**. The slot has to survive +into the statistic: fold it away first and five trials drawn in one slot yield the same number as +five alternating ones, and the randomisation measures nothing. + +`JudgeParsing.Summarize` partitions the trials by slot, computes the precise answer's win rate +within each, and reports the absolute difference as `PositionSwing`. `Indeterminate` verdicts are +excluded from both rates and counted separately. If either slot produced no determinate verdict the +swing is `null` — *not measurable* rather than zero, because one slot cannot say anything about +position. The five orderings are a balanced 3/2 split, shuffled, rather than five coin flips: five +flips land every trial in one slot 6.25% of the time, and a balanced split costs nothing and rules +that out. ## Key APIs @@ -93,7 +108,8 @@ precise answer regardless of slot, so any determinate flip is the bias signal. | `result.Get(name)` | Reads a score back out of the `EvaluationResult` | | `RubricJudgeEvaluator` | Custom 1–5 rubric judge; an unreadable verdict becomes a value-less metric, never a 0 | | `JudgeParsing.Parse(json)` | Strict verdict parse: `A` / `B` / `Indeterminate`, never throws | -| `JudgeParsing.Summarize(picks)` | Win/loss/indeterminate counts plus a bias rate that excludes indeterminates | +| `Trial(referenceInPositionA, verdict)` | One pairwise result with its slot kept, not folded away | +| `JudgeParsing.Summarize(trials)` | Win/loss/indeterminate counts plus the position swing between slots | ```bash dotnet run --project LLMAsJudge.AgentFramework @@ -104,9 +120,10 @@ dotnet run --project LLMAsJudge.AgentFramework -- --selfcheck # offline bias-l The first block prints each answer with four scored lines (`Relevance`, `Coherence`, `Groundedness`, `RubricScore`) and the judge's reason per metric; a line reading `INDETERMINATE` -means that judge's reply could not be read, not that the answer scored badly. The second block runs five -randomized orderings and prints the win/loss/indeterminate counts, the position-bias rate (of -determinate verdicts only), and a `► Position bias` verdict. A well-behaved judge on a clear-cut -pair should pick the precise answer regardless of slot — if it flips, you have just measured your -instrument, not your agent. **RegressionEvals** builds a gate on top of these evaluators, and +means that judge's reply could not be read, not that the answer scored badly. The second block runs five balanced +orderings and prints the win/loss/indeterminate counts, the position swing between the two slots +(computed from determinate verdicts only), and a `► Position bias` verdict. A well-behaved judge on +a clear-cut pair picks the precise answer regardless of slot and swings 0 — if the swing is above +0, you have just measured your instrument, not your agent. A judge that picks the vague answer in +both slots also swings 0: that shows up in the win counts, which is where wrongness belongs. **RegressionEvals** builds a gate on top of these evaluators, and **EvaluationAndMonitoring** tracks the token cost of running a judge on every answer. diff --git a/README.md b/README.md index 4d7c47d..7245737 100644 --- a/README.md +++ b/README.md @@ -158,7 +158,7 @@ the catalog together; each result states its scope limits and cites a primary so | Pattern | What it demonstrates | |---|---| -| LLMAsJudge | Judge-model rubric scoring plus a randomized-ordering position-bias probe | +| LLMAsJudge | Judge-model rubric scoring plus a position-bias probe that compares verdicts across balanced candidate orderings | | RedTeaming | Deterministic leak checks first, judge second, against the real GuardRails filter | | RegressionEvals | Golden-dataset suite of reviewed cases with tiered assertions, cached as a CI gate | | TrajectoryEvaluation | Scoring the agent's tool-use path with agent evaluators | From c97edd92e4044d1de0be2f3faa12eb982eb924ce Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 17:31:40 +0200 Subject: [PATCH 19/21] fix(twins,docs): narrow the SK fallback catch; draw the abstain/partial branches MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ResourceAwareOptimization.SemanticKernel caught bare Exception and fell through to the next tier on anything at all, while its AgentFramework twin was narrowed in an earlier PR. Narrowed to HttpOperationException / ClientResultException / HttpRequestException / TaskCanceledException. The twin's extra `!cancellationToken.IsCancellationRequested` guard has no counterpart here — the SK sample threads no token, so a TaskCanceledException can only be a timeout; the comment names the guard to add if a token is ever introduced. OrchestratorWorkers.md's mermaid ended at W1/W2/W3 -> S while the prose beside it described WorkerRegistry.Assess labelling runs Complete / Partial / Abstained. The diagram now shows the assess step, the all-failed abstain that skips synthesis, and the partial branch that flags the answer incomplete. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- PatternExplorer/patterns/OrchestratorWorkers.md | 11 ++++++++--- ResourceAwareOptimization.SemanticKernel/Program.cs | 9 ++++++++- 2 files changed, 16 insertions(+), 4 deletions(-) diff --git a/PatternExplorer/patterns/OrchestratorWorkers.md b/PatternExplorer/patterns/OrchestratorWorkers.md index cf86fc6..b1032c6 100644 --- a/PatternExplorer/patterns/OrchestratorWorkers.md +++ b/PatternExplorer/patterns/OrchestratorWorkers.md @@ -45,9 +45,14 @@ flowchart LR R -->|bounded concurrency| W1[Market] R --> W2[Competition] R --> W3[Regulation] - W1 --> S[Synthesizer] - W2 --> S - W3 --> S + W1 --> A{WorkerRegistry.Assess} + W2 --> A + W3 --> A + A -->|complete| S[Synthesizer] + A -->|partial: some tasks FAILED| S + A -->|all failed| X[Abstain: no synthesis] + S -->|partial| P[Answer flagged incomplete] + S -->|complete| R[Answer] V -->|invalid| D[Reject before execution] ``` diff --git a/ResourceAwareOptimization.SemanticKernel/Program.cs b/ResourceAwareOptimization.SemanticKernel/Program.cs index ad352ca..2bb98e2 100644 --- a/ResourceAwareOptimization.SemanticKernel/Program.cs +++ b/ResourceAwareOptimization.SemanticKernel/Program.cs @@ -1,3 +1,4 @@ +using System.ClientModel; using Microsoft.Extensions.DependencyInjection; using Microsoft.SemanticKernel; using Microsoft.SemanticKernel.ChatCompletion; @@ -101,7 +102,13 @@ async Task HandleQueryAsync(string userQuery) Console.WriteLine($" [Router] Success with: {sid}"); return response.Content ?? "No response generated."; } - catch (Exception ex) + // Only transient failures (HTTP errors, timeouts, network) should trigger the fallback + // tier — same narrowing as the AgentFramework twin. That twin also excludes caller + // cancellation; this sample threads no CancellationToken through, so a TaskCanceledException + // here can only be a request timeout. Add the `&& !cancellationToken.IsCancellationRequested` + // guard alongside a token if one is ever introduced. + catch (Exception ex) when (ex is HttpOperationException or ClientResultException + or HttpRequestException or TaskCanceledException) { Console.WriteLine($" [Fallback] {sid} failed: {ex.Message}"); } From c0bb6ef69f7ce9f6da9ab021c1bf77c94481e4aa Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 17:47:38 +0200 Subject: [PATCH 20/21] docs: stop the new mermaid branch from relabelling the worker registry `S -->|complete| R[Answer]` reused the node id already bound to the registry, so the rendered graph lost the registry and gave "Answer" its inbound and outbound edges. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- PatternExplorer/patterns/LLMAsJudge.md | 1 + PatternExplorer/patterns/OrchestratorWorkers.md | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/PatternExplorer/patterns/LLMAsJudge.md b/PatternExplorer/patterns/LLMAsJudge.md index dd844d6..672c26b 100644 --- a/PatternExplorer/patterns/LLMAsJudge.md +++ b/PatternExplorer/patterns/LLMAsJudge.md @@ -30,6 +30,7 @@ looks better). Crucially, a judge that is simply *wrong* — it prefers the vague answer in both slots — also swings 0. Wrongness and position-dependence are different defects, and a statistic that cannot tell them apart is not measuring position. + A judge is also not a reliable narrator of its own output format: `JudgeParsing.Parse` treats any verdict it cannot parse as its own contract — `{"winner": "A"}` or `{"winner": "B"}`, matched exactly — as `Indeterminate` rather than silently counting it as a win for either side. diff --git a/PatternExplorer/patterns/OrchestratorWorkers.md b/PatternExplorer/patterns/OrchestratorWorkers.md index b1032c6..1b287b5 100644 --- a/PatternExplorer/patterns/OrchestratorWorkers.md +++ b/PatternExplorer/patterns/OrchestratorWorkers.md @@ -52,7 +52,7 @@ flowchart LR A -->|partial: some tasks FAILED| S A -->|all failed| X[Abstain: no synthesis] S -->|partial| P[Answer flagged incomplete] - S -->|complete| R[Answer] + S -->|complete| Ans[Answer] V -->|invalid| D[Reject before execution] ``` From 550f4f76d7a9c57fb5d2646e8d21057331bcf63a Mon Sep 17 00:00:00 2001 From: arst Date: Wed, 26 Aug 2026 17:59:07 +0200 Subject: [PATCH 21/21] test(react): pin the auto-invocation loop to Terminate, not just the filter MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ToolCallBudgetFilterTests already covers OnAutoFunctionInvocationAsync directly, but nothing in the suite drove real Semantic Kernel's auto-invocation loop — the seam where the original throwing-filter defect actually lived (129 model calls instead of 11, budget refusal handed to the model as tool output). Add ToolCallBudgetFilterRealLoopTests: a real Kernel wired the way Program.cs wires it, pointed at a stub HttpMessageHandler (ScriptedToolCallHttpHandler, added to Fakes.cs) that always answers with another tool call. Verified red under the original throwing shape (129 model calls, matching the measured figure) and under a >= boundary mutation (9 tool calls instead of 10); green restored on both. Also softens ReasoningAndActing.md's bare "128-round"/"129" SK-internals numbers to point at this test instead. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LB4jjPp7i2pxV55Vpe6tpc --- AgenticPatterns.Tests/Fakes.cs | 65 ++++++++++ .../ToolCallBudgetFilterRealLoopTests.cs | 111 ++++++++++++++++++ .../patterns/ReasoningAndActing.md | 16 ++- 3 files changed, 186 insertions(+), 6 deletions(-) create mode 100644 AgenticPatterns.Tests/ToolCallBudgetFilterRealLoopTests.cs diff --git a/AgenticPatterns.Tests/Fakes.cs b/AgenticPatterns.Tests/Fakes.cs index 780a96e..35c4845 100644 --- a/AgenticPatterns.Tests/Fakes.cs +++ b/AgenticPatterns.Tests/Fakes.cs @@ -1,3 +1,6 @@ +using System.Net; +using System.Text; +using System.Text.Json; using CodeAct.AgentFramework.Execution; using Microsoft.Extensions.AI; @@ -77,3 +80,65 @@ public static T WithEnvironmentVariable(string name, string? value, Func b finally { Environment.SetEnvironmentVariable(name, original); } } } + +/// Stands in for the OpenAI endpoint at the seam: no +/// port, no socket, no real network call. Answers every chat-completion request with an assistant +/// message that calls back whichever function the request offered, repeated +/// times per response — so a Semantic Kernel auto-invocation +/// loop driven against this handler only ever stops if something (a filter) stops it. +internal sealed class ScriptedToolCallHttpHandler(int toolCallsPerTurn = 1) : HttpMessageHandler +{ + private int _requestCount; + + public int RequestCount => _requestCount; + + /// Every request body this handler has answered, in order — lets a test assert the + /// model was never fed a budget-refusal message to paraphrase. + public List RequestBodies { get; } = []; + + protected override async Task SendAsync( + HttpRequestMessage request, CancellationToken cancellationToken) + { + var body = request.Content is null + ? "" + : await request.Content.ReadAsStringAsync(cancellationToken); + lock (RequestBodies) RequestBodies.Add(body); + var requestNumber = Interlocked.Increment(ref _requestCount); + + using var doc = JsonDocument.Parse(body); + var toolName = doc.RootElement.GetProperty("tools")[0].GetProperty("function") + .GetProperty("name").GetString(); + + // Built via object graph + JsonSerializer, not a hand-assembled string: OpenAI's + // chat-completion response has enough nested braces that a raw string literal fights + // its own interpolation syntax. + var toolCalls = Enumerable.Range(0, toolCallsPerTurn).Select(i => new + { + id = $"call_{requestNumber}_{i}", + type = "function", + function = new { name = toolName, arguments = "{}" } + }); + var responseBody = new + { + id = $"chatcmpl-{requestNumber}", + @object = "chat.completion", + created = 0, + model = "stub-model", + choices = new[] + { + new + { + index = 0, + message = new { role = "assistant", content = (string?)null, tool_calls = toolCalls }, + finish_reason = "tool_calls" + } + }, + usage = new { prompt_tokens = 1, completion_tokens = 1, total_tokens = 2 } + }; + + return new HttpResponseMessage(HttpStatusCode.OK) + { + Content = new StringContent(JsonSerializer.Serialize(responseBody), Encoding.UTF8, "application/json") + }; + } +} diff --git a/AgenticPatterns.Tests/ToolCallBudgetFilterRealLoopTests.cs b/AgenticPatterns.Tests/ToolCallBudgetFilterRealLoopTests.cs new file mode 100644 index 0000000..a5efb8d --- /dev/null +++ b/AgenticPatterns.Tests/ToolCallBudgetFilterRealLoopTests.cs @@ -0,0 +1,111 @@ +using System.ComponentModel; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.SemanticKernel; +using Microsoft.SemanticKernel.ChatCompletion; +using Microsoft.SemanticKernel.Connectors.OpenAI; +using ReasoningAndActing; +using Xunit; + +namespace AgenticPatterns.Tests; + +/// +/// drives OnAutoFunctionInvocationAsync directly — +/// that proves the filter's own logic, but proves nothing about whether Semantic Kernel's real +/// auto-invocation loop actually honours context.Terminate. The first cut of this control +/// was an IFunctionInvocationFilter that threw; that passed its own unit tests but, run +/// against real SK 1.79, made 129 model calls instead of 11 because +/// FunctionCallsProcessor.ExecuteFunctionCallAsync swallows the exception into a tool-result +/// error and keeps looping. These tests close that gap: they build a real kernel, wire +/// in the same shape as +/// ReasoningAndActing/Program.cs, and drive IChatCompletionService with +/// FunctionChoiceBehavior.Auto() against a stub HTTP handler +/// () that answers every completion request with another +/// tool call, forever. The loop only stops if Terminate actually stops it. +/// +public class ToolCallBudgetFilterRealLoopTests +{ + /// The one tool the stub loop calls; counts how many times its body actually ran. + private sealed class CountingTool + { + public int Calls { get; private set; } + + [KernelFunction] + [Description("A tool that can be called any number of times.")] + public string Invoke() + { + Calls++; + return "ok"; + } + } + + private static (Kernel Kernel, ToolCallBudgetFilter Filter, CountingTool Tool, ScriptedToolCallHttpHandler Handler) + BuildKernel(int toolCallsPerTurn) + { + var handler = new ScriptedToolCallHttpHandler(toolCallsPerTurn); + var httpClient = new HttpClient(handler); + + var builder = Kernel.CreateBuilder(); + // Same connector Program.cs uses (Connectors.OpenAI, transitively via + // Connectors.AzureOpenAI); httpClient points every request at the stub handler above + // instead of a real endpoint — no port, no socket, no Settings/credentials needed. + builder.AddOpenAIChatCompletion(modelId: "stub-model", apiKey: "stub-key", httpClient: httpClient); + + // Same registration shape as Program.cs: one filter instance per run. + var filter = new ToolCallBudgetFilter(); + builder.Services.AddSingleton(filter); + + var kernel = builder.Build(); + var tool = new CountingTool(); + kernel.Plugins.AddFromObject(tool, "Tool"); + + return (kernel, filter, tool, handler); + } + + private static async Task RunLoopAsync(Kernel kernel) + { + var service = kernel.GetRequiredService(); + var history = new ChatHistory(); + history.AddUserMessage("Call the tool as many times as you like."); + var settings = new OpenAIPromptExecutionSettings { FunctionChoiceBehavior = FunctionChoiceBehavior.Auto() }; + return await service.GetChatMessageContentAsync(history, settings, kernel); + } + + [Fact] + public async Task RealAutoInvocationLoop_StopsExactlyAtTheBudget() + { + var (kernel, filter, tool, handler) = BuildKernel(toolCallsPerTurn: 1); + + await RunLoopAsync(kernel); + + // Boundary is exact: the 10th tool body runs, the 11th never does. + Assert.Equal(ToolCallBudgetFilter.MaxToolCalls, tool.Calls); + Assert.True(filter.BudgetExhausted); + + // Model-call count: one call per round in this single-call-per-turn shape, so it is exactly + // MaxToolCalls (10 rounds that each run their tool) + 1 (the round whose tool call is + // refused and terminates the loop before a 12th model call is ever made). Asserted exact, + // not as a bound: the stub is fully synchronous and deterministic, and the reviewer's own + // run also landed on exactly 11 — a bound would hide a regression that shaves rounds off. + Assert.Equal(ToolCallBudgetFilter.MaxToolCalls + 1, handler.RequestCount); + + // The refusal must never be handed to the model as tool output to paraphrase — Terminate + // ends the loop before any further request is built, so no request body the model saw + // should mention the budget at all. + Assert.DoesNotContain(handler.RequestBodies, body => body.Contains(filter.StopReason)); + } + + [Fact] + public async Task RealAutoInvocationLoop_BatchedToolCalls_StopsExactlyAtTheBudget() + { + // Same run, but the stub packs 3 tool calls into every assistant turn. A counter checked + // once per turn (rather than once per individual call) would let the whole batch that + // crosses the boundary run, overshooting past 10. ToolCallBudgetFilter counts per call, so + // the 10th call in the middle of a batch still runs and the 11th still doesn't. + var (kernel, filter, tool, _) = BuildKernel(toolCallsPerTurn: 3); + + await RunLoopAsync(kernel); + + Assert.Equal(ToolCallBudgetFilter.MaxToolCalls, tool.Calls); + Assert.True(filter.BudgetExhausted); + } +} diff --git a/PatternExplorer/patterns/ReasoningAndActing.md b/PatternExplorer/patterns/ReasoningAndActing.md index b63ba23..47ae28e 100644 --- a/PatternExplorer/patterns/ReasoningAndActing.md +++ b/PatternExplorer/patterns/ReasoningAndActing.md @@ -65,12 +65,16 @@ every auto-invoked call; the 10th runs, and the 11th never reaches the tool beca Throwing from a filter does *not* end it. SK 1.79 wraps every auto-invoked call in a catch-all that converts any exception into a tool-result error message and keeps looping, so a throwing filter blocks the tool body, hands the model its own budget refusal as tool output to paraphrase, and lets -the loop run on to SK's internal 128-round ceiling — measured at 129 model calls against a stub that -always requests a tool, versus 11 with `Terminate`. `Terminate` is the stop SK honours, and it is -the same mechanism **Goal Settings and Monitoring**'s `GoalMonitoringFilter` uses for its own -max-iteration bound. Because SK returns normally after terminating (with empty content), the filter -exposes the stop as a `BudgetExhausted` flag rather than an exception; `Program.cs` reads it and -prints a `PARTIAL` result — the same shape **Bounded Execution** uses for its own hard stops. +the loop run on to SK's internal auto-invoke ceiling instead of stopping at 10. `Terminate` is the +stop SK honours, and it is the same mechanism **Goal Settings and Monitoring**'s +`GoalMonitoringFilter` uses for its own max-iteration bound. Because SK returns normally after +terminating (with empty content), the filter exposes the stop as a `BudgetExhausted` flag rather +than an exception; `Program.cs` reads it and prints a `PARTIAL` result — the same shape +**Bounded Execution** uses for its own hard stops. +`AgenticPatterns.Tests.ToolCallBudgetFilterRealLoopTests` drives this against a real SK +auto-invocation loop (a stubbed HTTP handler that always returns a tool call, no network) and pins +both the exact `Terminate` behaviour and the fact that reverting to a throwing filter blows past the +budget by more than an order of magnitude. ## Key APIs