diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml
index 61e6ca0..b3c1124 100644
--- a/.github/workflows/build.yml
+++ b/.github/workflows/build.yml
@@ -10,8 +10,8 @@ jobs:
build:
runs-on: ubuntu-latest
steps:
- - uses: actions/checkout@v4
- - uses: actions/setup-dotnet@v4
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
+ - uses: actions/setup-dotnet@a98b56852c35b8e3190ac28c8c2271da59106c68 # v6.0.0
with:
dotnet-version: 10.0.x
- run: dotnet build "Agentic Patterns.slnx" --configuration Release
@@ -24,17 +24,17 @@ jobs:
contents: read
packages: write
steps:
- - uses: actions/checkout@v4
- - uses: docker/setup-qemu-action@v3
- - uses: docker/setup-buildx-action@v3
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
+ - uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4.2.0
+ - uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0
- if: github.event_name == 'push'
- uses: docker/login-action@v3
+ uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- id: meta
- uses: docker/metadata-action@v5
+ uses: docker/metadata-action@dc802804100637a589fabce1cb79ff13a1411302 # v6.2.0
with:
images: ghcr.io/${{ github.repository }}
tags: |
@@ -42,7 +42,7 @@ jobs:
type=semver,pattern={{version}}
type=semver,pattern={{major}}.{{minor}}
type=sha,prefix=sha-
- - uses: docker/build-push-action@v6
+ - uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0
with:
context: .
platforms: ${{ github.event_name == 'push' && 'linux/amd64,linux/arm64' || 'linux/amd64' }}
diff --git a/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj b/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj
index 8851ea2..50acb13 100644
--- a/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj
+++ b/AgenticPatterns.Tests/AgenticPatterns.Tests.csproj
@@ -26,6 +26,7 @@
(and Program.cs's top-level-statement type) are internal and invisible across the
assembly boundary, so referencing it too would add nothing to test. -->
+
+
+
diff --git a/AgenticPatterns.Tests/CasePartitionTests.cs b/AgenticPatterns.Tests/CasePartitionTests.cs
new file mode 100644
index 0000000..5a1a84d
--- /dev/null
+++ b/AgenticPatterns.Tests/CasePartitionTests.cs
@@ -0,0 +1,67 @@
+using RegressionEvals.AgentFramework;
+using Xunit;
+
+namespace AgenticPatterns.Tests;
+
+public class CasePartitionTests
+{
+ private static GoldenCase Case(string id, string reviewedBy) =>
+ new(id, "Q?", "A.", "contains", reviewedBy);
+
+ [Fact]
+ public void ReviewedCaseIsEvaluated()
+ {
+ var (evaluated, awaitingReview) = CasePartition.Partition([Case("reviewed", "alex")]);
+
+ Assert.Equal(["reviewed"], evaluated.Select(c => c.Id));
+ Assert.Empty(awaitingReview);
+ }
+
+ // This is the guarantee this task exists to pin: a case with no reviewer - exactly the shape
+ // ExtractTraceCase used to hand straight to the evaluator - must be EXCLUDED from evaluation,
+ // not merely present somewhere.
+ [Fact]
+ public void UnreviewedCaseIsExcludedFromEvaluationAndReportedAsAwaitingReview()
+ {
+ var (evaluated, awaitingReview) = CasePartition.Partition([Case("from-trace", reviewedBy: null!)]);
+
+ Assert.Empty(evaluated);
+ Assert.Equal(["from-trace"], awaitingReview.Select(c => c.Id));
+ }
+
+ [Fact]
+ public void EmptyReviewedByIsAlsoAwaitingReview()
+ {
+ var (evaluated, awaitingReview) = CasePartition.Partition([Case("blank", reviewedBy: "")]);
+
+ Assert.Empty(evaluated);
+ Assert.Single(awaitingReview);
+ }
+
+ [Fact]
+ public void MixedCorpusSplitsCorrectly()
+ {
+ var (evaluated, awaitingReview) = CasePartition.Partition(
+ [
+ Case("a", "alex"),
+ Case("b", null!),
+ Case("c", "jamie"),
+ Case("d", "")
+ ]);
+
+ Assert.Equal(["a", "c"], evaluated.Select(c => c.Id));
+ Assert.Equal(["b", "d"], awaitingReview.Select(c => c.Id));
+ }
+
+ [Fact]
+ public void GateFailsWhenNothingWasEvaluatedEvenWithZeroFailures() =>
+ Assert.Equal(1, CasePartition.GateExitCode(evaluatedCount: 0, failureCount: 0));
+
+ [Fact]
+ public void GatePassesWhenSomethingWasEvaluatedAndNothingFailed() =>
+ Assert.Equal(0, CasePartition.GateExitCode(evaluatedCount: 3, failureCount: 0));
+
+ [Fact]
+ public void GateFailsOnAnyFailure() =>
+ Assert.Equal(1, CasePartition.GateExitCode(evaluatedCount: 3, failureCount: 1));
+}
diff --git a/AgenticPatterns.Tests/Fakes.cs b/AgenticPatterns.Tests/Fakes.cs
index 780a96e..35c4845 100644
--- a/AgenticPatterns.Tests/Fakes.cs
+++ b/AgenticPatterns.Tests/Fakes.cs
@@ -1,3 +1,6 @@
+using System.Net;
+using System.Text;
+using System.Text.Json;
using CodeAct.AgentFramework.Execution;
using Microsoft.Extensions.AI;
@@ -77,3 +80,65 @@ public static T WithEnvironmentVariable(string name, string? value, Func b
finally { Environment.SetEnvironmentVariable(name, original); }
}
}
+
+/// Stands in for the OpenAI endpoint at the seam: no
+/// port, no socket, no real network call. Answers every chat-completion request with an assistant
+/// message that calls back whichever function the request offered, repeated
+/// times per response — so a Semantic Kernel auto-invocation
+/// loop driven against this handler only ever stops if something (a filter) stops it.
+internal sealed class ScriptedToolCallHttpHandler(int toolCallsPerTurn = 1) : HttpMessageHandler
+{
+ private int _requestCount;
+
+ public int RequestCount => _requestCount;
+
+ /// Every request body this handler has answered, in order — lets a test assert the
+ /// model was never fed a budget-refusal message to paraphrase.
+ public List RequestBodies { get; } = [];
+
+ protected override async Task SendAsync(
+ HttpRequestMessage request, CancellationToken cancellationToken)
+ {
+ var body = request.Content is null
+ ? ""
+ : await request.Content.ReadAsStringAsync(cancellationToken);
+ lock (RequestBodies) RequestBodies.Add(body);
+ var requestNumber = Interlocked.Increment(ref _requestCount);
+
+ using var doc = JsonDocument.Parse(body);
+ var toolName = doc.RootElement.GetProperty("tools")[0].GetProperty("function")
+ .GetProperty("name").GetString();
+
+ // Built via object graph + JsonSerializer, not a hand-assembled string: OpenAI's
+ // chat-completion response has enough nested braces that a raw string literal fights
+ // its own interpolation syntax.
+ var toolCalls = Enumerable.Range(0, toolCallsPerTurn).Select(i => new
+ {
+ id = $"call_{requestNumber}_{i}",
+ type = "function",
+ function = new { name = toolName, arguments = "{}" }
+ });
+ var responseBody = new
+ {
+ id = $"chatcmpl-{requestNumber}",
+ @object = "chat.completion",
+ created = 0,
+ model = "stub-model",
+ choices = new[]
+ {
+ new
+ {
+ index = 0,
+ message = new { role = "assistant", content = (string?)null, tool_calls = toolCalls },
+ finish_reason = "tool_calls"
+ }
+ },
+ usage = new { prompt_tokens = 1, completion_tokens = 1, total_tokens = 2 }
+ };
+
+ return new HttpResponseMessage(HttpStatusCode.OK)
+ {
+ Content = new StringContent(JsonSerializer.Serialize(responseBody), Encoding.UTF8, "application/json")
+ };
+ }
+}
diff --git a/AgenticPatterns.Tests/LlmAsJudgeTests.cs b/AgenticPatterns.Tests/LlmAsJudgeTests.cs
new file mode 100644
index 0000000..fa52581
--- /dev/null
+++ b/AgenticPatterns.Tests/LlmAsJudgeTests.cs
@@ -0,0 +1,196 @@
+using LLMAsJudge.AgentFramework;
+using Microsoft.Extensions.AI;
+using Microsoft.Extensions.AI.Evaluation;
+using Xunit;
+
+namespace AgenticPatterns.Tests;
+
+public class LlmAsJudgeTests
+{
+ [Theory]
+ [InlineData(null)] [InlineData("")] [InlineData("garbage")] [InlineData("{}")]
+ [InlineData("{\"winner\":\"a\"}")] [InlineData("{\"winner\":\"C\"}")]
+ public void AnythingUnexpectedIsIndeterminate(string? json) =>
+ Assert.Equal(Preference.Indeterminate, JudgeParsing.Parse(json));
+
+ [Fact] public void AParses() => Assert.Equal(Preference.A, JudgeParsing.Parse("{\"winner\":\"A\"}"));
+ [Fact] public void BParses() => Assert.Equal(Preference.B, JudgeParsing.Parse("{\"winner\":\"B\"}"));
+
+ // Malformed JSON that still throws on Deserialize (not just "returns null") - the brief's
+ // controller ruling: Parse must catch JsonException, not just handle null/missing keys.
+ [Theory]
+ [InlineData("[1,2,3]")]
+ [InlineData("not json at all {{{")]
+ public void MalformedJsonDoesNotThrow(string json) =>
+ Assert.Equal(Preference.Indeterminate, JudgeParsing.Parse(json));
+
+ [Theory]
+ [InlineData(Preference.A, true, true)]
+ [InlineData(Preference.B, true, false)]
+ [InlineData(Preference.A, false, false)]
+ [InlineData(Preference.B, false, true)]
+ public void ResolveTranslatesVerdictAndPositionIntoReferenceWin(
+ Preference verdict, bool referenceInPositionA, bool expectedReferenceWon) =>
+ Assert.Equal(expectedReferenceWon, JudgeParsing.Resolve(verdict, referenceInPositionA));
+
+ [Fact]
+ public void ResolveIsNullForIndeterminate() =>
+ Assert.Null(JudgeParsing.Resolve(Preference.Indeterminate, referenceInPositionA: true));
+
+ private static Trial InA(Preference verdict) => new(ReferenceInPositionA: true, verdict);
+ private static Trial InB(Preference verdict) => new(ReferenceInPositionA: false, verdict);
+
+ [Fact]
+ public void SummarizeCountsWinsAndIndeterminates()
+ {
+ var report = JudgeParsing.Summarize([
+ InA(Preference.A), // reference in A, judge picks A -> reference wins
+ InA(Preference.B), // reference in A, judge picks B -> other wins
+ InA(Preference.Indeterminate),
+ InB(Preference.B), // reference in B, judge picks B -> reference wins
+ InB(Preference.B)
+ ]);
+
+ Assert.Equal(3, report.ReferenceWins);
+ Assert.Equal(1, report.OtherWins);
+ Assert.Equal(1, report.Indeterminate);
+ }
+
+ [Fact]
+ public void JudgeThatAlwaysPicksTheSameSlot_IsFullPositionBias()
+ {
+ var report = JudgeParsing.Summarize([
+ InA(Preference.A), InA(Preference.A), InB(Preference.A), InB(Preference.A)
+ ]);
+
+ // Reference wins 100% of the time it sits in A, 0% of the time it sits in B.
+ Assert.Equal(1.0, report.PositionSwing);
+ }
+
+ [Fact]
+ public void JudgeThatIsSimplyWrong_IsNotPositionBias()
+ {
+ // The judge prefers the weaker candidate every single time, in both slots. That is a bad
+ // judge, not a position-dependent one - the pre-fix rate reported this as 100% bias
+ // because it folded the slot away before forming the statistic.
+ var report = JudgeParsing.Summarize([
+ InA(Preference.B), InA(Preference.B), InB(Preference.A), InB(Preference.A)
+ ]);
+
+ Assert.Equal(4, report.OtherWins);
+ Assert.Equal(0.0, report.PositionSwing);
+ }
+
+ [Fact]
+ public void ConsistentJudge_HasNoPositionSwing()
+ {
+ var report = JudgeParsing.Summarize([
+ InA(Preference.A), InA(Preference.A), InB(Preference.B), InB(Preference.B)
+ ]);
+
+ Assert.Equal(0.0, report.PositionSwing);
+ }
+
+ [Fact]
+ public void IdenticalOutcomesInDifferentSlotsProduceDifferentStatistics()
+ {
+ // The reference candidate wins every trial in both runs - identical outcomes - but one run
+ // never left slot A. The pre-fix rate folded the slot away before forming the number and
+ // so reported the same value for both, making the randomisation do no work at all.
+ var oneSlot = JudgeParsing.Summarize([InA(Preference.A), InA(Preference.A), InA(Preference.A)]);
+ var bothSlots = JudgeParsing.Summarize([InA(Preference.A), InB(Preference.B), InA(Preference.A)]);
+
+ Assert.Equal(oneSlot.ReferenceWins, bothSlots.ReferenceWins);
+ Assert.NotEqual(oneSlot.PositionSwing, bothSlots.PositionSwing);
+ }
+
+ [Fact]
+ public void SwingIsUnmeasurableWhenOnlyOneSlotWasSampled()
+ {
+ // Five coin flips land all five trials in one slot 6.25% of the time; Program.cs uses a
+ // balanced 3/2 shuffle so this cannot happen there, but the statistic still refuses to
+ // invent a position measurement from a single slot.
+ var report = JudgeParsing.Summarize([InA(Preference.A), InA(Preference.B), InA(Preference.A)]);
+
+ Assert.Null(report.PositionSwing);
+ }
+
+ [Fact]
+ public void IndeterminateVerdictsAreExcludedFromTheSwing()
+ {
+ // Both slots: every determinate verdict is a reference win, so both rates are 1 and the
+ // swing is 0. The unparseable verdicts are piled onto slot B only - count them in the
+ // denominator and slot B's rate drops to 1/3, inventing a swing out of noise.
+ var report = JudgeParsing.Summarize([
+ InA(Preference.A),
+ InB(Preference.B), InB(Preference.Indeterminate), InB(Preference.Indeterminate)
+ ]);
+
+ Assert.Equal(2, report.Indeterminate);
+ Assert.Equal(0.0, report.PositionSwing);
+ }
+
+ [Fact]
+ public void SwingIsUnmeasurableWhenEverythingIsIndeterminate()
+ {
+ var report = JudgeParsing.Summarize([
+ InA(Preference.Indeterminate), InB(Preference.Indeterminate), InA(Preference.Indeterminate)
+ ]);
+
+ Assert.Equal(3, report.Indeterminate);
+ Assert.Null(report.PositionSwing);
+ }
+}
+
+// Drives the real RubricJudgeEvaluator end to end - a scripted judge reply through
+// EvaluateAsync - because the defect lived in how the evaluator reads that reply.
+public class RubricJudgeEvaluatorTests
+{
+ private static async Task JudgeSaysAsync(string judgeReply)
+ {
+ var client = new ScriptedChatClient(
+ new ChatResponse(new ChatMessage(ChatRole.Assistant, judgeReply)));
+
+ var result = await new RubricJudgeEvaluator().EvaluateAsync(
+ [new ChatMessage(ChatRole.User, "What warranty do the laptops come with?")],
+ new ChatResponse(new ChatMessage(ChatRole.Assistant, "Two years.")),
+ new ChatConfiguration(client));
+
+ return result.Get(RubricJudgeEvaluator.RubricScoreMetricName);
+ }
+
+ [Theory]
+ [InlineData("")] // truncated to nothing
+ [InlineData("not json")] // prose instead of JSON
+ [InlineData("{}")] // JSON, but no score
+ [InlineData("null")] // literal JSON null
+ [InlineData(" ")] // whitespace only
+ [InlineData("[1,2,3]")] // JSON of the wrong shape
+ [InlineData("{\"score\":0,\"justification\":\"x\"}")] // below the rubric floor
+ [InlineData("{\"score\":9,\"justification\":\"x\"}")] // above the rubric ceiling
+ public async Task UnreadableVerdictIsIndeterminate_NeverThrows_NeverANumber(string judgeReply)
+ {
+ var metric = await JudgeSaysAsync(judgeReply);
+
+ // Indeterminate is "no value", not 0: 0 is below the rubric's own floor of 1, so scoring
+ // an unreadable verdict as a number would rank it worse than the worst possible answer.
+ Assert.Null(metric.Value);
+ Assert.Contains("Indeterminate", metric.Reason);
+ }
+
+ [Fact]
+ public async Task ParseableVerdictKeepsScoreAndJustification()
+ {
+ var metric = await JudgeSaysAsync("{\"score\":4,\"justification\":\"Accurate but terse.\"}");
+
+ Assert.Equal(4, metric.Value);
+ Assert.Equal("Accurate but terse.", metric.Reason);
+ }
+
+ [Fact]
+ public async Task ScoreIsNeverBelowTheRubricFloor()
+ {
+ foreach (var reply in new[] { "", "not json", "{}", "null", "{\"score\":-3}" })
+ Assert.True(await JudgeSaysAsync(reply) is { Value: null or >= 1 });
+ }
+}
diff --git a/AgenticPatterns.Tests/ProductionControlTests.cs b/AgenticPatterns.Tests/ProductionControlTests.cs
index 736b0d4..94e2d6f 100644
--- a/AgenticPatterns.Tests/ProductionControlTests.cs
+++ b/AgenticPatterns.Tests/ProductionControlTests.cs
@@ -206,6 +206,159 @@ public void OneTimeGrantCannotBeReplayed()
Assert.Equal(AuthorizationOutcome.Allowed, policy.Authorize(Principal, capability, "GetOrder", args).Outcome);
Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "GetOrder", args).Outcome);
}
+
+ [Fact]
+ public void AVerifiedPreEffectFailureReleasesTheOneTimeCapability()
+ {
+ var policy = new ToolAuthorizationPolicy(Owners);
+ var capability = Grant("IssueRefund", maximum: 50m, oneTime: true);
+ var args = new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 25m };
+
+ Assert.Equal(AuthorizationOutcome.Allowed, policy.Authorize(Principal, capability, "IssueRefund", args).Outcome);
+ policy.Release(capability.Nonce); // the tool threw before doing anything, and the caller verified that
+ Assert.Equal(AuthorizationOutcome.Allowed, policy.Authorize(Principal, capability, "IssueRefund", args).Outcome);
+ }
+
+ [Fact]
+ public void ACommittedOneTimeCapabilityCannotBeReused()
+ {
+ var policy = new ToolAuthorizationPolicy(Owners);
+ var capability = Grant("IssueRefund", maximum: 50m, oneTime: true);
+ var args = new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 25m };
+
+ Assert.Equal(AuthorizationOutcome.Allowed, policy.Authorize(Principal, capability, "IssueRefund", args).Outcome);
+ policy.Commit(capability.Nonce);
+ policy.Release(capability.Nonce); // a late release must not resurrect a committed capability
+ Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "IssueRefund", args).Outcome);
+ }
+
+ [Fact]
+ public void ARefundWithoutAConfiguredMaximumStillRequiresAPositiveAmount()
+ {
+ var policy = new ToolAuthorizationPolicy(Owners);
+ var capability = Grant("IssueRefund", maximum: null);
+
+ Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "IssueRefund",
+ new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = -5m }).Outcome);
+ Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "IssueRefund",
+ new AIFunctionArguments { ["orderId"] = "ORD-100" }).Outcome);
+ Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "IssueRefund",
+ new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = "not-money" }).Outcome);
+ }
+
+ [Fact]
+ public void ApprovalIsAPendingRequestNotToolOutput()
+ {
+ var decision = new ToolAuthorizationPolicy(Owners).Authorize(Principal, Grant("IssueRefund", 50m),
+ "IssueRefund", new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 10_000m });
+
+ Assert.NotNull(decision.PendingApproval);
+ Assert.Equal("IssueRefund", decision.PendingApproval!.ToolName);
+ Assert.Equal(10_000m, decision.PendingApproval.Arguments["amount"]);
+ }
+
+ [Fact]
+ public void PendingApprovalArgumentsAreSnapshottedNotAliased()
+ {
+ var arguments = new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 10_000m };
+ var decision = new ToolAuthorizationPolicy(Owners).Authorize(Principal, Grant("IssueRefund", 50m),
+ "IssueRefund", arguments);
+ arguments["amount"] = 1m;
+
+ Assert.Equal(10_000m, decision.PendingApproval!.Arguments["amount"]);
+ }
+
+ [Fact]
+ public void ConcurrentAuthorizeReservesAOneTimeNonceExactlyOnce()
+ {
+ var policy = new ToolAuthorizationPolicy(Owners);
+
+ // One trial is not a gate: a read-then-write reserve wins the race the overwhelming
+ // majority of the time, and is *most* likely to win on a loaded machine, which is exactly
+ // how xUnit runs this. Repeat with a fresh nonce per trial so one lost race fails the test.
+ for (var trial = 0; trial < 5000; trial++)
+ {
+ var capability = Grant("GetOrder", oneTime: true);
+ var allowed = 0;
+
+ Parallel.For(0, 4, _ =>
+ {
+ if (policy.Authorize(Principal, capability, "GetOrder",
+ new AIFunctionArguments { ["orderId"] = "ORD-100" }).Outcome == AuthorizationOutcome.Allowed)
+ Interlocked.Increment(ref allowed);
+ });
+
+ Assert.Equal(1, allowed);
+ }
+ }
+
+ [Fact]
+ public void ARefusedInvocationDoesNotBurnTheOneTimeCapability()
+ {
+ var policy = new ToolAuthorizationPolicy(Owners);
+ var capability = Grant("IssueRefund", maximum: 50m, oneTime: true);
+
+ Assert.Equal(AuthorizationOutcome.Denied, policy.Authorize(Principal, capability, "IssueRefund",
+ new AIFunctionArguments { ["orderId"] = "ORD-OTHER-TENANT", ["amount"] = 25m }).Outcome);
+ Assert.Equal(AuthorizationOutcome.ApprovalRequired, policy.Authorize(Principal, capability, "IssueRefund",
+ new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 500m }).Outcome);
+
+ // The reserve happens after every check, so neither refusal cost the caller the capability:
+ // the approved retry uses the very same grant.
+ Assert.Equal(AuthorizationOutcome.Allowed, policy.Authorize(Principal, capability, "IssueRefund",
+ new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 25m }).Outcome);
+ }
+
+ [Fact]
+ public async Task TheWrapperCommitsTheReservationOnlyAfterTheToolReturns()
+ {
+ var policy = new ToolAuthorizationPolicy(Owners);
+ var capability = Grant("GetOrder", oneTime: true);
+ var args = new AIFunctionArguments { ["orderId"] = "ORD-100" };
+ var tool = new AuthorizedAIFunction(
+ AIFunctionFactory.Create((string orderId) => $"{orderId}: ok", "GetOrder"), Principal, capability, policy);
+
+ Assert.Equal("ORD-100: ok", (await tool.InvokeAsync(args))?.ToString());
+ policy.Release(capability.Nonce); // committed already, so this cannot hand the capability back
+ await Assert.ThrowsAsync(() => tool.InvokeAsync(args).AsTask());
+ }
+
+ [Fact]
+ public async Task AnUnverifiedToolFailureDoesNotReleaseTheReservation()
+ {
+ var policy = new ToolAuthorizationPolicy(Owners);
+ var capability = Grant("GetOrder", oneTime: true);
+ var args = new AIFunctionArguments { ["orderId"] = "ORD-100" };
+ var tool = new AuthorizedAIFunction(
+ AIFunctionFactory.Create((Func)Boom, "GetOrder"), Principal, capability, policy);
+
+ await Assert.ThrowsAsync(() => tool.InvokeAsync(args).AsTask());
+ // Failing closed: from inside the wrapper the failure is unverified, so the reservation stands.
+ await Assert.ThrowsAsync(() => tool.InvokeAsync(args).AsTask());
+
+ // ...and it is still merely Reserved, never Consumed. This is what pins Commit to its
+ // position *after* the inner call: committing first would deny identically here while
+ // leaving a capability that threw pre-effect permanently unreleasable.
+ policy.Release(capability.Nonce);
+ Assert.Equal(AuthorizationOutcome.Allowed,
+ policy.Authorize(Principal, capability, "GetOrder", args).Outcome);
+ static string Boom(string orderId) => throw new InvalidOperationException("simulated tool failure");
+ }
+
+ [Fact]
+ public async Task TheWrapperNeverHandsAnApprovalRequestBackAsToolOutput()
+ {
+ var policy = new ToolAuthorizationPolicy(Owners);
+ var tool = new AuthorizedAIFunction(
+ AIFunctionFactory.Create((string orderId, decimal amount) => "refunded", "IssueRefund"),
+ Principal, Grant("IssueRefund", 50m), policy);
+
+ var failure = await Assert.ThrowsAsync(() =>
+ tool.InvokeAsync(new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 500m }).AsTask());
+
+ Assert.Equal(AuthorizationOutcome.ApprovalRequired, failure.Decision.Outcome);
+ Assert.NotNull(failure.Decision.PendingApproval);
+ }
}
public class IdempotentToolCallTests
@@ -319,8 +472,43 @@ public async Task ExecutionHonorsConcurrencyAndPreservesPartialSuccess()
var synthesis = WorkerRegistry.BuildSynthesisInput(results);
Assert.Contains("one", synthesis);
Assert.Contains("three", synthesis);
- Assert.DoesNotContain("simulated failure", synthesis);
+ Assert.Contains("simulated failure", synthesis);
+ Assert.Contains("bad", synthesis);
+ }
+
+ [Fact]
+ public void SynthesisInputNamesTheFailures()
+ {
+ var input = WorkerRegistry.BuildSynthesisInput([
+ new WorkerResult("t1", "research", "found A", null),
+ new WorkerResult("t2", "research", null, "timed out")]);
+ Assert.Contains("t2", input);
+ Assert.Contains("FAILED", input);
+ Assert.Contains("timed out", input);
}
+
+ [Fact]
+ public void EveryWorkerFailingAbstainsInsteadOfSynthesising() =>
+ Assert.Equal(RunCompleteness.Abstained,
+ WorkerRegistry.Assess([new WorkerResult("t1", "r", null, "boom")], requiredQuorum: 1));
+
+ [Fact]
+ public void PartialSuccessIsLabelledPartial() =>
+ Assert.Equal(RunCompleteness.Partial, WorkerRegistry.Assess(
+ [new WorkerResult("t1", "r", "ok", null), new WorkerResult("t2", "r", null, "boom")],
+ requiredQuorum: 1));
+
+ [Fact]
+ public void AllSucceedingAndMeetingQuorumIsComplete() =>
+ Assert.Equal(RunCompleteness.Complete, WorkerRegistry.Assess(
+ [new WorkerResult("t1", "r", "ok", null), new WorkerResult("t2", "r", "ok", null)],
+ requiredQuorum: 2));
+
+ [Fact]
+ public void AllSucceedingButBelowQuorumIsPartial() =>
+ Assert.Equal(RunCompleteness.Partial, WorkerRegistry.Assess(
+ [new WorkerResult("t1", "r", "ok", null), new WorkerResult("t2", "r", "ok", null)],
+ requiredQuorum: 3));
}
public class GuardRailsTests
diff --git a/AgenticPatterns.Tests/RetryTests.cs b/AgenticPatterns.Tests/RetryTests.cs
index 550edfd..2c4efa8 100644
--- a/AgenticPatterns.Tests/RetryTests.cs
+++ b/AgenticPatterns.Tests/RetryTests.cs
@@ -62,6 +62,15 @@ public async Task RecoveryOnSecondAttempt_ReturnsResponse()
Assert.Equal(2, attempts);
}
+ [Fact]
+ public async Task CallerCancellationIsNotTurnedIntoAFallback()
+ {
+ using var cts = new CancellationTokenSource();
+ await cts.CancelAsync();
+ await Assert.ThrowsAnyAsync(() =>
+ Retry.RunAsync(_ => Task.FromCanceled(cts.Token), maxRetries: 3, NoBackoff));
+ }
+
[Fact]
public void HasToolError_DetectsExceptionAndErrorString()
{
diff --git a/AgenticPatterns.Tests/ToolCallBudgetFilterRealLoopTests.cs b/AgenticPatterns.Tests/ToolCallBudgetFilterRealLoopTests.cs
new file mode 100644
index 0000000..a5efb8d
--- /dev/null
+++ b/AgenticPatterns.Tests/ToolCallBudgetFilterRealLoopTests.cs
@@ -0,0 +1,111 @@
+using System.ComponentModel;
+using Microsoft.Extensions.DependencyInjection;
+using Microsoft.SemanticKernel;
+using Microsoft.SemanticKernel.ChatCompletion;
+using Microsoft.SemanticKernel.Connectors.OpenAI;
+using ReasoningAndActing;
+using Xunit;
+
+namespace AgenticPatterns.Tests;
+
+///
+/// drives OnAutoFunctionInvocationAsync directly —
+/// that proves the filter's own logic, but proves nothing about whether Semantic Kernel's real
+/// auto-invocation loop actually honours context.Terminate. The first cut of this control
+/// was an IFunctionInvocationFilter that threw; that passed its own unit tests but, run
+/// against real SK 1.79, made 129 model calls instead of 11 because
+/// FunctionCallsProcessor.ExecuteFunctionCallAsync swallows the exception into a tool-result
+/// error and keeps looping. These tests close that gap: they build a real kernel, wire
+/// in the same shape as
+/// ReasoningAndActing/Program.cs, and drive IChatCompletionService with
+/// FunctionChoiceBehavior.Auto() against a stub HTTP handler
+/// () that answers every completion request with another
+/// tool call, forever. The loop only stops if Terminate actually stops it.
+///
+public class ToolCallBudgetFilterRealLoopTests
+{
+ /// The one tool the stub loop calls; counts how many times its body actually ran.
+ private sealed class CountingTool
+ {
+ public int Calls { get; private set; }
+
+ [KernelFunction]
+ [Description("A tool that can be called any number of times.")]
+ public string Invoke()
+ {
+ Calls++;
+ return "ok";
+ }
+ }
+
+ private static (Kernel Kernel, ToolCallBudgetFilter Filter, CountingTool Tool, ScriptedToolCallHttpHandler Handler)
+ BuildKernel(int toolCallsPerTurn)
+ {
+ var handler = new ScriptedToolCallHttpHandler(toolCallsPerTurn);
+ var httpClient = new HttpClient(handler);
+
+ var builder = Kernel.CreateBuilder();
+ // Same connector Program.cs uses (Connectors.OpenAI, transitively via
+ // Connectors.AzureOpenAI); httpClient points every request at the stub handler above
+ // instead of a real endpoint — no port, no socket, no Settings/credentials needed.
+ builder.AddOpenAIChatCompletion(modelId: "stub-model", apiKey: "stub-key", httpClient: httpClient);
+
+ // Same registration shape as Program.cs: one filter instance per run.
+ var filter = new ToolCallBudgetFilter();
+ builder.Services.AddSingleton(filter);
+
+ var kernel = builder.Build();
+ var tool = new CountingTool();
+ kernel.Plugins.AddFromObject(tool, "Tool");
+
+ return (kernel, filter, tool, handler);
+ }
+
+ private static async Task RunLoopAsync(Kernel kernel)
+ {
+ var service = kernel.GetRequiredService();
+ var history = new ChatHistory();
+ history.AddUserMessage("Call the tool as many times as you like.");
+ var settings = new OpenAIPromptExecutionSettings { FunctionChoiceBehavior = FunctionChoiceBehavior.Auto() };
+ return await service.GetChatMessageContentAsync(history, settings, kernel);
+ }
+
+ [Fact]
+ public async Task RealAutoInvocationLoop_StopsExactlyAtTheBudget()
+ {
+ var (kernel, filter, tool, handler) = BuildKernel(toolCallsPerTurn: 1);
+
+ await RunLoopAsync(kernel);
+
+ // Boundary is exact: the 10th tool body runs, the 11th never does.
+ Assert.Equal(ToolCallBudgetFilter.MaxToolCalls, tool.Calls);
+ Assert.True(filter.BudgetExhausted);
+
+ // Model-call count: one call per round in this single-call-per-turn shape, so it is exactly
+ // MaxToolCalls (10 rounds that each run their tool) + 1 (the round whose tool call is
+ // refused and terminates the loop before a 12th model call is ever made). Asserted exact,
+ // not as a bound: the stub is fully synchronous and deterministic, and the reviewer's own
+ // run also landed on exactly 11 — a bound would hide a regression that shaves rounds off.
+ Assert.Equal(ToolCallBudgetFilter.MaxToolCalls + 1, handler.RequestCount);
+
+ // The refusal must never be handed to the model as tool output to paraphrase — Terminate
+ // ends the loop before any further request is built, so no request body the model saw
+ // should mention the budget at all.
+ Assert.DoesNotContain(handler.RequestBodies, body => body.Contains(filter.StopReason));
+ }
+
+ [Fact]
+ public async Task RealAutoInvocationLoop_BatchedToolCalls_StopsExactlyAtTheBudget()
+ {
+ // Same run, but the stub packs 3 tool calls into every assistant turn. A counter checked
+ // once per turn (rather than once per individual call) would let the whole batch that
+ // crosses the boundary run, overshooting past 10. ToolCallBudgetFilter counts per call, so
+ // the 10th call in the middle of a batch still runs and the 11th still doesn't.
+ var (kernel, filter, tool, _) = BuildKernel(toolCallsPerTurn: 3);
+
+ await RunLoopAsync(kernel);
+
+ Assert.Equal(ToolCallBudgetFilter.MaxToolCalls, tool.Calls);
+ Assert.True(filter.BudgetExhausted);
+ }
+}
diff --git a/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs b/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs
new file mode 100644
index 0000000..467b7a8
--- /dev/null
+++ b/AgenticPatterns.Tests/ToolCallBudgetFilterTests.cs
@@ -0,0 +1,98 @@
+using Microsoft.SemanticKernel;
+using Microsoft.SemanticKernel.ChatCompletion;
+using ReasoningAndActing;
+using Xunit;
+
+namespace AgenticPatterns.Tests;
+
+public class ToolCallBudgetFilterTests
+{
+ // A real AutoFunctionInvocationContext — the exact type SK 1.79 hands the filter at runtime.
+ // Its 5-argument constructor is public, so no test seam is needed: these tests drive
+ // OnAutoFunctionInvocationAsync itself, which is the only method the runtime ever calls.
+ private static AutoFunctionInvocationContext NewContext()
+ {
+ var kernel = new Kernel();
+ var function = KernelFunctionFactory.CreateFromMethod(() => "tool ran", "Tool");
+ return new AutoFunctionInvocationContext(
+ kernel, function, new FunctionResult(function), new ChatHistory(),
+ new ChatMessageContent(AuthorRole.Assistant, "calling a tool"));
+ }
+
+ private static async Task<(bool InnerRan, AutoFunctionInvocationContext Context)> InvokeAsync(
+ ToolCallBudgetFilter filter)
+ {
+ var context = NewContext();
+ var innerRan = false;
+ await filter.OnAutoFunctionInvocationAsync(context, _ =>
+ {
+ innerRan = true;
+ return Task.CompletedTask;
+ });
+ return (innerRan, context);
+ }
+
+ [Fact]
+ public async Task TenthCallRuns_EleventhIsRefused()
+ {
+ var filter = new ToolCallBudgetFilter();
+
+ for (var i = 0; i < ToolCallBudgetFilter.MaxToolCalls; i++)
+ {
+ var (ran, ctx) = await InvokeAsync(filter);
+ Assert.True(ran);
+ Assert.False(ctx.Terminate);
+ Assert.False(filter.BudgetExhausted);
+ }
+
+ var (eleventhRan, eleventhCtx) = await InvokeAsync(filter);
+
+ // The 11th tool body must not run...
+ Assert.False(eleventhRan);
+ // ...and the auto-invocation loop must stop, rather than the model being handed an error
+ // to paraphrase and carrying on. Terminate is the only stop SK 1.79 honours.
+ Assert.True(eleventhCtx.Terminate);
+ Assert.True(filter.BudgetExhausted);
+ }
+
+ [Fact]
+ public async Task OverBudgetCall_DoesNotThrow()
+ {
+ // Throwing is what the runtime swallows: SK converts the exception into a tool-result
+ // error message and keeps looping. The filter must return normally instead.
+ var filter = new ToolCallBudgetFilter();
+ for (var i = 0; i < ToolCallBudgetFilter.MaxToolCalls; i++)
+ await InvokeAsync(filter);
+
+ var exception = await Record.ExceptionAsync(() => InvokeAsync(filter));
+
+ Assert.Null(exception);
+ }
+
+ [Fact]
+ public async Task StopReason_NamesTheBudget()
+ {
+ var filter = new ToolCallBudgetFilter();
+ for (var i = 0; i <= ToolCallBudgetFilter.MaxToolCalls; i++)
+ await InvokeAsync(filter);
+
+ Assert.True(filter.BudgetExhausted);
+ Assert.Contains(ToolCallBudgetFilter.MaxToolCalls.ToString(), filter.StopReason);
+ }
+
+ [Fact]
+ public async Task BudgetIsPerInstance_NotPerProcess()
+ {
+ // Program.cs registers one instance per run; a second instance is a second run and must
+ // start with a full budget.
+ var exhausted = new ToolCallBudgetFilter();
+ for (var i = 0; i <= ToolCallBudgetFilter.MaxToolCalls; i++)
+ await InvokeAsync(exhausted);
+ Assert.True(exhausted.BudgetExhausted);
+
+ var (ran, ctx) = await InvokeAsync(new ToolCallBudgetFilter());
+
+ Assert.True(ran);
+ Assert.False(ctx.Terminate);
+ }
+}
diff --git a/ExceptionHandlingAndRecovery.AgentFramework/Retry.cs b/ExceptionHandlingAndRecovery.AgentFramework/Retry.cs
index d222098..93f6d6d 100644
--- a/ExceptionHandlingAndRecovery.AgentFramework/Retry.cs
+++ b/ExceptionHandlingAndRecovery.AgentFramework/Retry.cs
@@ -17,6 +17,18 @@ public static bool HasToolError(AgentResponse response) =>
/// Runs attempt() up to maxRetries times. Returns (null, lastError) when every attempt threw
/// OR the final attempt still reported a tool error — a persistent tool failure is never a success.
///
+ ///
+ /// Whole-turn retry replays every tool call the turn made. Only use it for turns whose tools are
+ /// read-only or idempotent; a turn that issues a refund must retry at the tool boundary with an
+ /// idempotency key instead (see IdempotentToolCalls).
+ ///
+ /// is always rethrown, never retried. RunAsync takes no
+ /// of its own, so it has nothing to test the exception against
+ /// (the caller's token lives inside ) — there is no way to tell "the
+ /// caller cancelled" apart from "the tool timed out internally" here. One consequence: a tool that
+ /// raises for its own internal timeout will no longer be
+ /// retried by this helper either.
+ ///
public static async Task<(AgentResponse? Response, Exception? LastError)> RunAsync(
Func> attempt, int maxRetries, Func backoff)
{
@@ -34,6 +46,10 @@ public static bool HasToolError(AgentResponse response) =>
if (attemptNumber < maxRetries)
await backoff(attemptNumber);
}
+ catch (OperationCanceledException)
+ {
+ throw; // the caller asked to stop; retrying is not recovery
+ }
catch (Exception ex)
{
lastError = ex;
diff --git a/ExceptionHandlingAndRecovery.SemanticKernel/RetryAndFallbackFilter.cs b/ExceptionHandlingAndRecovery.SemanticKernel/RetryAndFallbackFilter.cs
index 0d7d9ab..ab145e1 100644
--- a/ExceptionHandlingAndRecovery.SemanticKernel/RetryAndFallbackFilter.cs
+++ b/ExceptionHandlingAndRecovery.SemanticKernel/RetryAndFallbackFilter.cs
@@ -13,6 +13,11 @@ public RetryAndFallbackFilter(ILogger logger)
_logger = logger;
}
+ ///
+ /// Whole-turn retry replays every tool call the turn made. Only use it for turns whose tools are
+ /// read-only or idempotent; a turn that issues a refund must retry at the tool boundary with an
+ /// idempotency key instead (see IdempotentToolCalls).
+ ///
public async Task OnFunctionInvocationAsync(
FunctionInvocationContext context, Func next)
{
@@ -38,6 +43,12 @@ public async Task OnFunctionInvocationAsync(
_logger.LogInformation("[Recovery] {Function} succeeded on attempt {Attempt}", functionName, attempt);
return;
}
+ catch (OperationCanceledException) when (context.CancellationToken.IsCancellationRequested)
+ {
+ // the caller asked to stop; retrying (and then falling back to a different
+ // function) is not recovery
+ throw;
+ }
catch (Exception ex)
{
_logger.LogWarning(
diff --git a/LLMAsJudge.AgentFramework/JudgeParsing.cs b/LLMAsJudge.AgentFramework/JudgeParsing.cs
new file mode 100644
index 0000000..aa6fa59
--- /dev/null
+++ b/LLMAsJudge.AgentFramework/JudgeParsing.cs
@@ -0,0 +1,86 @@
+using System.Text.Json;
+
+namespace LLMAsJudge.AgentFramework;
+
+public enum Preference { A, B, Indeterminate }
+
+/// One pairwise trial: which slot the reference candidate sat in, and what the judge said.
+/// The slot must survive into the statistic — fold it away early and the randomisation stops
+/// measuring anything.
+public readonly record struct Trial(bool ReferenceInPositionA, Preference Verdict);
+
+// Parses judge verdicts and summarizes them across balanced position orderings.
+public static class JudgeParsing
+{
+ public static Preference Parse(string? json)
+ {
+ if (string.IsNullOrEmpty(json)) return Preference.Indeterminate;
+
+ string? winner;
+ try
+ {
+ winner = JsonSerializer.Deserialize>(json,
+ new JsonSerializerOptions(JsonSerializerDefaults.Web))?.GetValueOrDefault("winner");
+ }
+ catch (JsonException)
+ {
+ return Preference.Indeterminate;
+ }
+
+ // Exact-case match only: a judge that writes "a" instead of "A" has not followed the
+ // output contract, and its verdict should not be counted. Do not loosen to OrdinalIgnoreCase.
+ return winner switch
+ {
+ "A" => Preference.A,
+ "B" => Preference.B,
+ _ => Preference.Indeterminate
+ };
+ }
+
+ // Translates a verdict plus which slot the "reference" candidate occupied into whether the
+ // reference candidate won. Indeterminate verdicts carry no position information.
+ public static bool? Resolve(Preference verdict, bool referenceInPositionA) => verdict switch
+ {
+ Preference.A => referenceInPositionA,
+ Preference.B => !referenceInPositionA,
+ _ => null
+ };
+
+ /// How much the reference candidate's win rate moved when it
+ /// changed slots: |win rate with the reference in A − win rate with it in B|. 0 means the
+ /// verdict did not depend on position; 1 means it depended on nothing else. null when
+ /// either slot produced no determinate verdict — one slot cannot measure position bias.
+ public readonly record struct PreferenceReport(
+ int ReferenceWins, int OtherWins, int Indeterminate, double? PositionSwing);
+
+ // Indeterminate verdicts are excluded from the swing but counted separately.
+ public static PreferenceReport Summarize(IReadOnlyList trials)
+ {
+ var outcomes = trials.Select(t => Resolve(t.Verdict, t.ReferenceInPositionA)).ToList();
+
+ var inA = ReferenceWinRate(trials, referenceInPositionA: true);
+ var inB = ReferenceWinRate(trials, referenceInPositionA: false);
+
+ return new PreferenceReport(
+ outcomes.Count(r => r == true),
+ outcomes.Count(r => r == false),
+ outcomes.Count(r => r is null),
+ inA is null || inB is null ? null : Math.Abs(inA.Value - inB.Value));
+ }
+
+ // A judge that is simply wrong — it prefers the same candidate in both slots — has a win rate
+ // of 0 in both and so a swing of 0: wrong is not the same defect as position-dependent, and
+ // this is what folding the slot away before the statistic used to hide.
+ private static double? ReferenceWinRate(IReadOnlyList trials, bool referenceInPositionA)
+ {
+ var determinate = trials
+ .Where(t => t.ReferenceInPositionA == referenceInPositionA)
+ .Select(t => Resolve(t.Verdict, referenceInPositionA))
+ .Where(r => r is not null)
+ .ToList();
+
+ return determinate.Count == 0
+ ? null
+ : (double)determinate.Count(r => r == true) / determinate.Count;
+ }
+}
diff --git a/LLMAsJudge.AgentFramework/Program.cs b/LLMAsJudge.AgentFramework/Program.cs
index 646f079..9ec1062 100644
--- a/LLMAsJudge.AgentFramework/Program.cs
+++ b/LLMAsJudge.AgentFramework/Program.cs
@@ -1,4 +1,3 @@
-using System.Text.Json;
using LLMAsJudge.AgentFramework;
using Microsoft.Agents.AI;
using Microsoft.Extensions.AI;
@@ -38,27 +37,47 @@
var result = await eval.EvaluateAsync(conversation, response, chatConfig,
ctx is null ? null : [ctx]);
var metric = result.Get(result.Metrics.Keys.First());
- Console.WriteLine($" {name,-14}: {metric.Value} ({metric.Reason})");
+ // A null value means the judge's verdict could not be read — print that plainly rather
+ // than a blank column that reads like a zero.
+ Console.WriteLine($" {name,-14}: {metric.Value?.ToString() ?? "INDETERMINATE"} ({metric.Reason})");
}
Console.WriteLine();
}
-// ---- Pairwise comparison with position swap ----
+// ---- Pairwise comparison across randomized position orderings ----
Console.WriteLine("==== Pairwise comparison (position-bias probe) ====\n");
const string pairwiseQuestion = "What warranty do TechCorp laptops come with?";
var good = "TechCorp laptops come with a two-year limited warranty.";
var vague = "TechCorp offers a warranty on its laptops for a period of time.";
-var firstWins = await PairwiseWinnerAsync(pairwiseQuestion, good, vague); // good in position A
-var swappedWins = await PairwiseWinnerAsync(pairwiseQuestion, vague, good); // good in position B
-// Winner is reported as "A" or "B"; translate to the candidate identity.
-var pick1 = firstWins == "A" ? "good" : "vague";
-var pick2 = swappedWins == "A" ? "vague" : "good";
-Console.WriteLine($"Original order picked: {pick1}");
-Console.WriteLine($"Swapped order picked: {pick2}");
-Console.WriteLine(PositionBiasDetected(pick1, pick2)
- ? "► Position bias DETECTED: verdict flipped when candidates were swapped."
- : "► Consistent verdict across positions.");
+// A balanced 3/2 split, shuffled: both slots are always sampled, so the position statistic is
+// always defined. Five independent coin flips land every trial in one slot 6.25% of the time,
+// and a statistic conditioned on position cannot be computed from a single slot.
+var slots = new[] { true, true, true, false, false };
+Random.Shared.Shuffle(slots);
+
+var trials = new List();
+foreach (var goodInPositionA in slots)
+{
+ var candidateA = goodInPositionA ? good : vague;
+ var candidateB = goodInPositionA ? vague : good;
+ var verdict = await PairwiseWinnerAsync(pairwiseQuestion, candidateA, candidateB);
+ trials.Add(new Trial(goodInPositionA, verdict));
+}
+
+var report = JudgeParsing.Summarize(trials);
+Console.WriteLine(
+ $"Good wins: {report.ReferenceWins} Vague wins: {report.OtherWins} Indeterminate: {report.Indeterminate}");
+Console.WriteLine(report.PositionSwing is { } swing
+ ? $"Position swing (good answer's win rate in slot A vs slot B): {swing:P0}"
+ : "Position swing: not measurable — one slot produced no determinate verdict.");
+Console.WriteLine(report.PositionSwing switch
+{
+ > 0 => "► Position bias DETECTED: the same pair got a different verdict depending on which slot "
+ + "the better answer sat in.",
+ 0 => "► No position bias: the verdict did not change when the candidates swapped slots.",
+ _ => "► Position bias not measured this run."
+});
IEnumerable<(string, IEvaluator, EvaluationContext?)> Evaluators() =>
[
@@ -68,7 +87,7 @@
("RubricScore", new RubricJudgeEvaluator(), null)
];
-async Task PairwiseWinnerAsync(string q, string candidateA, string candidateB)
+async Task PairwiseWinnerAsync(string q, string candidateA, string candidateB)
{
var prompt =
$$"""
@@ -79,19 +98,51 @@ async Task PairwiseWinnerAsync(string q, string candidateA, string candi
""";
var r = await chatClient.GetResponseAsync([new ChatMessage(ChatRole.User, prompt)],
new ChatOptions { Temperature = 0f, ResponseFormat = ChatResponseFormat.Json });
- var winner = JsonSerializer.Deserialize>(r.Text,
- new JsonSerializerOptions(JsonSerializerDefaults.Web))?.GetValueOrDefault("winner");
- return winner == "B" ? "B" : "A";
+ return JudgeParsing.Parse(r.Text);
}
-// Bias is present when the same candidate does NOT win regardless of its slot.
-static bool PositionBiasDetected(string pickOriginal, string pickSwapped) =>
- pickOriginal != pickSwapped;
-
static void SelfCheck()
{
- // Same candidate wins in both orders -> no bias. Different -> bias.
- if (PositionBiasDetected("good", "good")) throw new Exception("false positive");
- if (!PositionBiasDetected("good", "vague")) throw new Exception("missed flip");
+ // JudgeParsing.Parse: strict verdicts, Indeterminate on anything unrecognised or unparseable.
+ if (JudgeParsing.Parse("{\"winner\":\"A\"}") != Preference.A) throw new Exception("A misparsed");
+ if (JudgeParsing.Parse("{\"winner\":\"B\"}") != Preference.B) throw new Exception("B misparsed");
+ if (JudgeParsing.Parse("{\"winner\":\"a\"}") != Preference.Indeterminate) throw new Exception("lowercase not indeterminate");
+ if (JudgeParsing.Parse("garbage") != Preference.Indeterminate) throw new Exception("garbage not indeterminate");
+ if (JudgeParsing.Parse(null) != Preference.Indeterminate) throw new Exception("null not indeterminate");
+ if (JudgeParsing.Parse("") != Preference.Indeterminate) throw new Exception("empty not indeterminate");
+
+ // JudgeParsing.Resolve: verdict + slot -> did the reference candidate win?
+ if (JudgeParsing.Resolve(Preference.A, referenceInPositionA: true) != true) throw new Exception("resolve A/A wrong");
+ if (JudgeParsing.Resolve(Preference.B, referenceInPositionA: true) != false) throw new Exception("resolve B/A wrong");
+ if (JudgeParsing.Resolve(Preference.Indeterminate, referenceInPositionA: true) is not null) throw new Exception("resolve indeterminate wrong");
+
+ // JudgeParsing.Summarize: the swing is position-conditioned, so it must separate "the judge
+ // depends on the slot" from "the judge is simply wrong". Fixed inputs, no randomness.
+ Trial InA(Preference v) => new(ReferenceInPositionA: true, v);
+ Trial InB(Preference v) => new(ReferenceInPositionA: false, v);
+
+ // Judge always picks the reference candidate, whichever slot it is in: no position dependence.
+ if (JudgeParsing.Summarize([InA(Preference.A), InA(Preference.A), InB(Preference.B), InB(Preference.B)])
+ .PositionSwing != 0) throw new Exception("false positive bias");
+
+ // Judge is simply wrong — always picks the other candidate, in both slots. Still no position
+ // dependence: the old rate reported 100% bias here, which is the defect this replaces.
+ var alwaysWrong = JudgeParsing.Summarize(
+ [InA(Preference.B), InA(Preference.B), InB(Preference.A), InB(Preference.A)]);
+ if (alwaysWrong.PositionSwing != 0) throw new Exception("wrongness misreported as position bias");
+ if (alwaysWrong.OtherWins != 4) throw new Exception("win counting wrong");
+
+ // Judge always picks whatever sits in slot A: total position dependence.
+ if (JudgeParsing.Summarize([InA(Preference.A), InA(Preference.A), InB(Preference.A), InB(Preference.A)])
+ .PositionSwing != 1) throw new Exception("missed bias");
+
+ // Five verdicts drawn in the same slot say nothing about position, however lopsided.
+ if (JudgeParsing.Summarize([InA(Preference.A), InA(Preference.B), InA(Preference.A)])
+ .PositionSwing is not null) throw new Exception("one-slot swing should be unmeasurable");
+
+ var allIndeterminate = JudgeParsing.Summarize(
+ [InA(Preference.Indeterminate), InB(Preference.Indeterminate), InA(Preference.Indeterminate)]);
+ if (allIndeterminate.Indeterminate != 3 || allIndeterminate.PositionSwing is not null) throw new Exception("indeterminate handling wrong");
+
Console.WriteLine("selfcheck ok");
}
diff --git a/LLMAsJudge.AgentFramework/RubricJudgeEvaluator.cs b/LLMAsJudge.AgentFramework/RubricJudgeEvaluator.cs
index b77e72b..c954b99 100644
--- a/LLMAsJudge.AgentFramework/RubricJudgeEvaluator.cs
+++ b/LLMAsJudge.AgentFramework/RubricJudgeEvaluator.cs
@@ -12,7 +12,36 @@ public sealed class RubricJudgeEvaluator : IEvaluator
public const string RubricScoreMetricName = "Rubric Score";
public IReadOnlyCollection EvaluationMetricNames => [RubricScoreMetricName];
- private sealed record Verdict(int Score, string Justification);
+ // The rubric's own floor and ceiling. A score outside them is not a verdict.
+ private const int MinScore = 1;
+ private const int MaxScore = 5;
+
+ private sealed record Verdict(int Score, string? Justification);
+
+ private static readonly JsonSerializerOptions JsonOptions = new(JsonSerializerDefaults.Web);
+
+ ///
+ /// Never throws, for any input. An empty, truncated, non-JSON, or score-less judge reply is
+ /// null — indeterminate — and not a Verdict(0, …): 0 sits below the rubric's own
+ /// floor of 1, so recording an unreadable verdict as a number would score it worse than the
+ /// worst possible answer instead of admitting the judge was not understood. Empty and
+ /// whitespace input arrive here as a like any other malformed
+ /// reply, and "null" deserializes to a null Verdict.
+ ///
+ private static Verdict? ParseVerdict(string text)
+ {
+ Verdict? verdict;
+ try
+ {
+ verdict = JsonSerializer.Deserialize(text, JsonOptions);
+ }
+ catch (JsonException)
+ {
+ return null;
+ }
+
+ return verdict is { Score: >= MinScore and <= MaxScore } ? verdict : null;
+ }
public async ValueTask EvaluateAsync(
IEnumerable messages,
@@ -38,11 +67,14 @@ [new ChatMessage(ChatRole.User, prompt)],
new ChatOptions { Temperature = 0f, ResponseFormat = ChatResponseFormat.Json },
cancellationToken);
- var verdict = JsonSerializer.Deserialize(response.Text,
- new JsonSerializerOptions(JsonSerializerDefaults.Web))
- ?? new Verdict(0, "Judge returned unparseable output.");
+ // An unreadable judge reply is reported as a metric with no value at all, so downstream
+ // averages and gates see "not measured" rather than a number the judge never gave.
+ var metric = ParseVerdict(response.Text) is { } verdict
+ ? new NumericMetric(RubricScoreMetricName, verdict.Score, verdict.Justification)
+ : new NumericMetric(RubricScoreMetricName, value: null,
+ reason: $"Indeterminate: the judge's reply was not a parseable "
+ + $"{MinScore}-{MaxScore} rubric verdict.");
- var metric = new NumericMetric(RubricScoreMetricName, verdict.Score, verdict.Justification);
return new EvaluationResult(metric);
}
}
diff --git a/OrchestratorWorkers.AgentFramework/Program.cs b/OrchestratorWorkers.AgentFramework/Program.cs
index f4ebba8..7a540d8 100644
--- a/OrchestratorWorkers.AgentFramework/Program.cs
+++ b/OrchestratorWorkers.AgentFramework/Program.cs
@@ -29,11 +29,23 @@
foreach (var result in results)
Console.WriteLine($"\n[{result.TaskId}/{result.Worker}] {(result.Succeeded ? result.Output : "FAILED: " + result.Error)}");
+// ponytail: this demo demands every planned task, so the quorum term can never be the
+// deciding one here; pass a real minimum-viable subset when a caller can act on less.
+var completeness = WorkerRegistry.Assess(results, requiredQuorum: plan.Tasks.Count);
+if (completeness == RunCompleteness.Abstained)
+{
+ Console.WriteLine("\n=== Synthesis (ABSTAINED) ===\nEvery worker failed, so there is no evidence to synthesize.");
+ return;
+}
+
+var instruction = completeness == RunCompleteness.Partial
+ ? "Some tasks failed. State clearly which conclusions are unsupported; do not fill the gaps. " +
+ "Synthesize the worker reports into a concise recommendation."
+ : "Synthesize the worker reports into a concise recommendation.";
var evidence = WorkerRegistry.BuildSynthesisInput(results);
-var synthesizer = new ChatClientAgent(client,
- "Synthesize the successful worker reports into a concise recommendation.",
- "Synthesizer");
-Console.WriteLine($"\n=== Synthesis ===\n{await synthesizer.RunAsync($"Request: {request}\n\nWorker reports:\n{evidence}")}");
+var synthesizer = new ChatClientAgent(client, instruction, "Synthesizer");
+var label = completeness == RunCompleteness.Partial ? "PARTIAL" : "COMPLETE";
+Console.WriteLine($"\n=== Synthesis ({label}) ===\n{await synthesizer.RunAsync($"Request: {request}\n\nWorker reports:\n{evidence}")}");
return;
Func> Run(string roleInstruction) => async (task, cancellationToken) =>
diff --git a/OrchestratorWorkers.AgentFramework/WorkerRegistry.cs b/OrchestratorWorkers.AgentFramework/WorkerRegistry.cs
index 1797ec5..7dac1bf 100644
--- a/OrchestratorWorkers.AgentFramework/WorkerRegistry.cs
+++ b/OrchestratorWorkers.AgentFramework/WorkerRegistry.cs
@@ -44,6 +44,18 @@ public async Task> ExecuteAsync(WorkPlan plan, int m
}
public static string BuildSynthesisInput(IEnumerable results) =>
- string.Join("\n\n", results.Where(r => r.Succeeded)
- .Select(r => $"## {r.TaskId} ({r.Worker})\n{r.Output}"));
+ string.Join("\n\n", results.Select(r => r.Succeeded
+ ? $"## {r.TaskId} ({r.Worker})\nSTATUS: OK\n{r.Output}"
+ : $"## {r.TaskId} ({r.Worker})\nSTATUS: FAILED\nERROR: {r.Error}\nNO OUTPUT - do not infer one."));
+
+ public static RunCompleteness Assess(IReadOnlyList results, int requiredQuorum)
+ {
+ var succeeded = results.Count(r => r.Succeeded);
+ if (succeeded == 0) return RunCompleteness.Abstained;
+ return succeeded == results.Count && succeeded >= requiredQuorum
+ ? RunCompleteness.Complete
+ : RunCompleteness.Partial;
+ }
}
+
+public enum RunCompleteness { Complete, Partial, Abstained }
diff --git a/PatternExplorer/patterns/EvaluationAndMonitoring.md b/PatternExplorer/patterns/EvaluationAndMonitoring.md
index 4783b1a..521ce64 100644
--- a/PatternExplorer/patterns/EvaluationAndMonitoring.md
+++ b/PatternExplorer/patterns/EvaluationAndMonitoring.md
@@ -128,5 +128,7 @@ signals to an answer where this pattern measures cost and speed.
This pattern measures cost and speed and records ground truth; the **Evaluation** category
(**LLMAsJudge**, **RegressionEvals**, **TrajectoryEvaluation**, **RedTeaming**) judges quality —
-**RegressionEvals** in particular turns the traces recorded here into a regression gate, and
-**TrajectoryEvaluation** judges whether the tool calls this pattern counts were the right ones.
+**RegressionEvals** in particular turns the traces recorded here into review candidates for its
+regression gate (a trace records what happened, not what should have happened, so a human still
+promotes each one), and **TrajectoryEvaluation** judges whether the tool calls this pattern counts
+were the right ones.
diff --git a/PatternExplorer/patterns/ExceptionHandlingAndRecovery.md b/PatternExplorer/patterns/ExceptionHandlingAndRecovery.md
index cab34a0..c1b41f1 100644
--- a/PatternExplorer/patterns/ExceptionHandlingAndRecovery.md
+++ b/PatternExplorer/patterns/ExceptionHandlingAndRecovery.md
@@ -1,7 +1,7 @@
---
{
"title": "Exception Handling, Recovery, and Circuit Breaker",
- "summary": "Retry transient faults, fail fast through an open dependency circuit, then degrade safely.",
+ "summary": "Retry transient faults on idempotent turns, fail fast through an open dependency circuit, then degrade safely — caller cancellation is never retried.",
"category": "Production controls",
"projects": [
{ "flavor": "AgentFramework", "path": "ExceptionHandlingAndRecovery.AgentFramework" },
@@ -22,6 +22,13 @@ dependency, and degrade gracefully. A circuit breaker is distinct from **Bounded
the breaker shares dependency health across calls through closed, open, and half-open states;
a run budget limits one agent execution.
+Both flavors retry the *whole turn*, which replays every tool call the turn made. That is only
+safe when the turn's tools are read-only or idempotent, as `GetPreciseLocation` is here. A turn
+that issues a refund must not be retried this way — retry at the tool boundary instead, with an
+idempotency key, as **IdempotentToolCalls** does. And in both flavors, caller cancellation
+(`OperationCanceledException`) is always rethrown, never retried or turned into a fallback —
+the caller asked to stop, and retrying is not recovery.
+
## When to use it
- A tool depends on a network service, a rate-limited API, or anything with transient faults.
diff --git a/PatternExplorer/patterns/LLMAsJudge.md b/PatternExplorer/patterns/LLMAsJudge.md
index cc17aad..672c26b 100644
--- a/PatternExplorer/patterns/LLMAsJudge.md
+++ b/PatternExplorer/patterns/LLMAsJudge.md
@@ -1,7 +1,7 @@
---
{
"title": "LLM as Judge",
- "summary": "Score answers with a judge model against a rubric, and probe the judge's own position bias.",
+ "summary": "Score answers with a judge model against a rubric, and measure the judge's own position bias by comparing its verdicts across balanced candidate orderings, discarding verdicts it cannot parse.",
"category": "Evaluation",
"projects": [
{ "flavor": "AgentFramework", "path": "LLMAsJudge.AgentFramework" }
@@ -19,10 +19,21 @@ sample scores answers three ways: built-in **quality evaluators** (`Relevance`,
written directly against `IEvaluator`, and a **pairwise comparison** that picks the better of two
answers.
-The judge is not a neutral instrument, and the pairwise step exists to prove it: the same two
-candidates are judged twice with their positions swapped. If the verdict flips, the judge has
-**position bias** — one of its documented failure modes alongside verbosity bias (longer looks
-better) and self-preference (its own style looks better).
+The judge is not a neutral instrument, and the pairwise step exists to measure it: the same two
+candidates are judged five times, with the better answer placed in slot A three times and slot B
+twice, shuffled. The statistic is the **position swing** — how far the better answer's win rate
+moves when it changes slots. A judge whose verdict does not depend on position swings 0; a judge
+that just picks whatever sits in slot A swings 1. That is **position bias** — one of its documented
+failure modes alongside verbosity bias (longer looks better) and self-preference (its own style
+looks better).
+
+Crucially, a judge that is simply *wrong* — it prefers the vague answer in both slots — also swings
+0. Wrongness and position-dependence are different defects, and a statistic that cannot tell them
+apart is not measuring position.
+
+A judge is also not a reliable narrator of its own output format: `JudgeParsing.Parse` treats any
+verdict it cannot parse as its own contract — `{"winner": "A"}` or `{"winner": "B"}`, matched
+exactly — as `Indeterminate` rather than silently counting it as a win for either side.
**Builds on:** **SelfCorrectionLoop** and **Debate** use a judge *inline* to drive generation;
this pattern is the same judge as a standalone measurement, plus its failure modes.
@@ -33,7 +44,7 @@ where this is what a second model says about the answer.
A judge score is a proxy, and Goodhart's law applies the moment you optimize against it (see
`docs/coordination-physics.md`): tune a prompt to please the judge and you may buy judge points
-without buying quality. The position-swap probe is the honest correction — it measures the
+without buying quality. The randomized-ordering probe is the honest correction — it measures the
*instrument's* noise floor before you trust its readings. A judge that flips on position is not
measuring answer quality; it is measuring slot order, and any score it emits is that much less
information about the thing you actually care about.
@@ -57,18 +68,35 @@ Each answer is scored by the three quality evaluators plus `RubricJudgeEvaluator
as a `NumericMetric`. `GroundednessEvaluator` receives the policy text as its context so it can
check the answer against the source rather than against the model's own memory.
+The rubric judge never throws and never invents a number. A reply that is empty, truncated, not
+JSON, missing a score, or scored outside 1–5 yields a `NumericMetric` with **no value** and an
+`Indeterminate` reason, printed as `INDETERMINATE`. A numeric `0` would be worse than that: it sits
+below the rubric's own floor of 1, so an unreadable verdict would be recorded as *worse than the
+worst possible answer* and would drag any average computed over the metric.
+
```mermaid
flowchart LR
A[SupportAgent answer] --> E[Quality evaluators
Relevance, Coherence, Groundedness]
A --> R[RubricJudgeEvaluator
1-5 + justification]
- P[Two candidate answers] --> J[Pairwise judge]
- J -->|swap positions| J
- J --> B[Position-bias verdict]
+ P[Two candidate answers] --> J[Pairwise judge x5
randomized position]
+ J --> Parse[JudgeParsing.Parse
A / B / Indeterminate]
+ Parse --> B[Preference distribution
+ position swing across slots]
```
-The pairwise section pits a precise answer against a vague one, judges them in both orders, and
-reports whether the same candidate won regardless of slot. `PositionBiasDetected` is the whole
-verdict: original pick ≠ swapped pick means bias.
+The pairwise section pits a precise answer against a vague one across five orderings, parses each
+verdict with `JudgeParsing.Parse` (which returns `Indeterminate` — never a default winner — for
+anything that doesn't match the `{"winner": "A"}` / `{"winner": "B"}` contract), and records each
+result as a `Trial` that keeps **which slot the precise answer occupied**. The slot has to survive
+into the statistic: fold it away first and five trials drawn in one slot yield the same number as
+five alternating ones, and the randomisation measures nothing.
+
+`JudgeParsing.Summarize` partitions the trials by slot, computes the precise answer's win rate
+within each, and reports the absolute difference as `PositionSwing`. `Indeterminate` verdicts are
+excluded from both rates and counted separately. If either slot produced no determinate verdict the
+swing is `null` — *not measurable* rather than zero, because one slot cannot say anything about
+position. The five orderings are a balanced 3/2 split, shuffled, rather than five coin flips: five
+flips land every trial in one slot 6.25% of the time, and a balanced split costs nothing and rules
+that out.
## Key APIs
@@ -79,6 +107,10 @@ verdict: original pick ≠ swapped pick means bias.
| `GroundednessEvaluatorContext(policy)` | Supplies grounding source to the evaluator |
| `IEvaluator.EvaluateAsync(messages, response, chatConfig, contexts)` | The evaluator contract |
| `result.Get(name)` | Reads a score back out of the `EvaluationResult` |
+| `RubricJudgeEvaluator` | Custom 1–5 rubric judge; an unreadable verdict becomes a value-less metric, never a 0 |
+| `JudgeParsing.Parse(json)` | Strict verdict parse: `A` / `B` / `Indeterminate`, never throws |
+| `Trial(referenceInPositionA, verdict)` | One pairwise result with its slot kept, not folded away |
+| `JudgeParsing.Summarize(trials)` | Win/loss/indeterminate counts plus the position swing between slots |
```bash
dotnet run --project LLMAsJudge.AgentFramework
@@ -88,9 +120,11 @@ dotnet run --project LLMAsJudge.AgentFramework -- --selfcheck # offline bias-l
## What to watch in the output
The first block prints each answer with four scored lines (`Relevance`, `Coherence`,
-`Groundedness`, `RubricScore`) and the judge's reason per metric. The second block prints the
-original-order pick, the swapped-order pick, and a `► Position bias` verdict. A well-behaved
-judge on a clear-cut pair should pick the precise answer both times — if it flips, you have just
-measured your instrument, not your agent. **RegressionEvals** builds a gate on top of these
-evaluators, and **EvaluationAndMonitoring** tracks the token cost of running a judge on every
-answer.
+`Groundedness`, `RubricScore`) and the judge's reason per metric; a line reading `INDETERMINATE`
+means that judge's reply could not be read, not that the answer scored badly. The second block runs five balanced
+orderings and prints the win/loss/indeterminate counts, the position swing between the two slots
+(computed from determinate verdicts only), and a `► Position bias` verdict. A well-behaved judge on
+a clear-cut pair picks the precise answer regardless of slot and swings 0 — if the swing is above
+0, you have just measured your instrument, not your agent. A judge that picks the vague answer in
+both slots also swings 0: that shows up in the win counts, which is where wrongness belongs. **RegressionEvals** builds a gate on top of these evaluators, and
+**EvaluationAndMonitoring** tracks the token cost of running a judge on every answer.
diff --git a/PatternExplorer/patterns/OrchestratorWorkers.md b/PatternExplorer/patterns/OrchestratorWorkers.md
index 91b9fe4..1b287b5 100644
--- a/PatternExplorer/patterns/OrchestratorWorkers.md
+++ b/PatternExplorer/patterns/OrchestratorWorkers.md
@@ -1,7 +1,7 @@
---
{
"title": "Orchestrator-Workers",
- "summary": "Let an orchestrator choose request-specific tasks, validate them, run fixed workers with bounded concurrency, and synthesize the results.",
+ "summary": "Let an orchestrator choose request-specific tasks, validate them, run fixed workers with bounded concurrency, and synthesize the results while naming any that failed.",
"category": "Orchestration",
"projects": [ { "flavor": "AgentFramework", "path": "OrchestratorWorkers.AgentFramework" } ]
}
@@ -11,7 +11,9 @@
Orchestrator-Workers uses a central model to decide which independent subtasks a particular
request needs. The host validates that typed plan, dispatches each task to a fixed worker registry,
-and asks a synthesizer to combine successful results.
+and asks a synthesizer to combine the results — successes and failures alike. A failed task is
+carried in with its error and an explicit instruction not to invent an output, so the gap is at
+least visible to the synthesizer; the host abstains outright when every worker fails.
This is dynamic fan-out, but not an open-ended team protocol:
@@ -43,14 +45,23 @@ flowchart LR
R -->|bounded concurrency| W1[Market]
R --> W2[Competition]
R --> W3[Regulation]
- W1 --> S[Synthesizer]
- W2 --> S
- W3 --> S
+ W1 --> A{WorkerRegistry.Assess}
+ W2 --> A
+ W3 --> A
+ A -->|complete| S[Synthesizer]
+ A -->|partial: some tasks FAILED| S
+ A -->|all failed| X[Abstain: no synthesis]
+ S -->|partial| P[Answer flagged incomplete]
+ S -->|complete| Ans[Answer]
V -->|invalid| D[Reject before execution]
```
`WorkerRegistry` uses `SemaphoreSlim` to cap concurrency and captures per-task failures without
-discarding successful reports. Only successful outputs enter the synthesis evidence.
+discarding successful reports. Every result enters the synthesis evidence, failed tasks included,
+each tagged `STATUS: FAILED` with its error so the synthesizer cannot silently paper over a gap.
+`WorkerRegistry.Assess` labels the run `Complete`, `Partial`, or `Abstained`; an all-failed run
+skips synthesis entirely, and a partial run tells the synthesizer to call out unsupported
+conclusions instead of inferring them.
## Key APIs
@@ -58,9 +69,12 @@ discarding successful reports. Only successful outputs enter the synthesis evide
- `PlanValidator.Validate(...)` — trusted-host validation before dispatch.
- `WorkerRegistry` — fixed role-to-executor mapping; the model cannot create arbitrary workers.
- `Task.WhenAll(...)` + `SemaphoreSlim` — concurrent execution with a hard concurrency ceiling.
+- `WorkerRegistry.Assess(...)` — labels a run `Complete`, `Partial`, or `Abstained` from its results.
## What to watch in the output
First inspect the serialized validated plan: its tasks should reflect the request rather than a
-hard-coded fan-out. Worker outputs follow, then one synthesis. A worker failure becomes a failed
-`WorkerResult`; it does not erase independent successful evidence.
+hard-coded fan-out. Worker outputs follow, then one synthesis, labelled `COMPLETE` or `PARTIAL` —
+or, if every worker failed, an abstention with no synthesis at all. A worker failure becomes a
+failed `WorkerResult`; it does not erase independent successful evidence, and it is not hidden
+from the synthesizer either.
diff --git a/PatternExplorer/patterns/ReasoningAndActing.md b/PatternExplorer/patterns/ReasoningAndActing.md
index 21f752d..47ae28e 100644
--- a/PatternExplorer/patterns/ReasoningAndActing.md
+++ b/PatternExplorer/patterns/ReasoningAndActing.md
@@ -54,9 +54,27 @@ flowchart LR
A --> R[Final answer with ratio]
```
-There is one honest wart worth knowing: SK 1.79 exposes no max-auto-invoke setting on
-`FunctionChoiceBehavior`, so the loop is capped by a sentence in the system prompt — *"Use at
-most 10 tool calls before giving your final answer."* A prompt is a request, not a guardrail.
+SK 1.79 exposes no max-auto-invoke setting on `FunctionChoiceBehavior`, so the system prompt still
+carries *"Use at most 10 tool calls before giving your final answer."* — but that sentence is only
+a hint the model can ignore. The actual control is `ToolCallBudgetFilter`, an
+`IAutoFunctionInvocationFilter` registered on the local kernel as a single instance created for
+this run — the counter lives on the instance, so the budget is per run, not per process. It counts
+every auto-invoked call; the 10th runs, and the 11th never reaches the tool because the filter sets
+`context.Terminate = true`, which ends SK's auto-invocation loop.
+
+Throwing from a filter does *not* end it. SK 1.79 wraps every auto-invoked call in a catch-all that
+converts any exception into a tool-result error message and keeps looping, so a throwing filter
+blocks the tool body, hands the model its own budget refusal as tool output to paraphrase, and lets
+the loop run on to SK's internal auto-invoke ceiling instead of stopping at 10. `Terminate` is the
+stop SK honours, and it is the same mechanism **Goal Settings and Monitoring**'s
+`GoalMonitoringFilter` uses for its own max-iteration bound. Because SK returns normally after
+terminating (with empty content), the filter exposes the stop as a `BudgetExhausted` flag rather
+than an exception; `Program.cs` reads it and prints a `PARTIAL` result — the same shape
+**Bounded Execution** uses for its own hard stops.
+`AgenticPatterns.Tests.ToolCallBudgetFilterRealLoopTests` drives this against a real SK
+auto-invocation loop (a stubbed HTTP handler that always returns a tool call, no network) and pins
+both the exact `Terminate` behaviour and the fact that reverting to a throwing filter blows past the
+budget by more than an order of magnitude.
## Key APIs
@@ -66,11 +84,18 @@ most 10 tool calls before giving your final answer."* A prompt is a request, not
- `new OpenAIPromptExecutionSettings { FunctionChoiceBehavior = FunctionChoiceBehavior.Auto() }`.
- `chatService.GetChatMessageContentAsync(history, settings, kernel)` — the kernel argument is what
makes auto-invocation possible.
+- `ToolCallBudgetFilter : IAutoFunctionInvocationFilter` — the host-enforced call cap, stopping
+ the loop with `context.Terminate = true`; see **Bounded Execution** for the fuller pattern of
+ hard, host-enforced run limits.
## What to watch in the output
The demo prints a single block prefixed `ReAct Agent:`. The tool calls themselves are not logged,
so the tell is in the content: the answer should quote the exact simulated figures — 40.1 million
and 26.5 million — and a ratio near 1.5, none of which the model could produce without calling
-the plugin. **Tool Use** is the single-call foundation this loops over; **Middleware** shows how
-to log every invocation so the reasoning-acting alternation becomes visible.
+the plugin. If the model instead wanders past the tool-call budget, the answer block is
+replaced by `Result status: PARTIAL`, a `Stop reason:` line naming the exhausted budget, and an
+explicit incomplete label — proof the bound stopped the loop rather than the model choosing to
+stop. **Tool Use** is the single-call
+foundation this loops over; **Middleware** shows how to log every invocation so the
+reasoning-acting alternation becomes visible.
diff --git a/PatternExplorer/patterns/RegressionEvals.md b/PatternExplorer/patterns/RegressionEvals.md
index 2f248f3..6598eb8 100644
--- a/PatternExplorer/patterns/RegressionEvals.md
+++ b/PatternExplorer/patterns/RegressionEvals.md
@@ -1,7 +1,7 @@
---
{
"title": "Regression Evals",
- "summary": "A golden dataset with tiered assertions (string, NLP, judge) run as a gate, cached for CI.",
+ "summary": "A golden dataset of reviewed cases with tiered assertions (contains, NLP, judge) run as a gate, cached for CI.",
"category": "Evaluation",
"projects": [
{ "flavor": "AgentFramework", "path": "RegressionEvals.AgentFramework" }
@@ -21,24 +21,34 @@ tool renders as an HTML report.
The assertions are **tiered cheapest-first**, because not every check needs a model:
-- **exact** — a plain `Contains` string check. No model call.
+- **contains** — a plain `Contains` string check. No model call.
- **nlp** — `F1Evaluator` token overlap against the golden answer. No model call.
- **judge** — `EquivalenceEvaluator`, an LLM-as-judge, for semantic equality. One model call.
+Only **reviewed** cases run: a `GoldenCase` with a non-empty `ReviewedBy` is evaluated, everything
+else is reported as awaiting review and skipped. The gate itself fails if nothing was evaluated —
+a green run over zero reviewed cases is the same class of false confidence as a red regression
+that ships anyway.
+
**Builds on:** **EvaluationAndMonitoring** records the trajectories this pattern turns into
-golden cases — one case here is extracted straight from a recorded `run-trace.json`, the
-canonical *production trace → eval case* pipeline. Where **SelfCorrectionLoop** runs an evaluator
-inline to fix a single answer, this runs the same class of evaluators as a pre-merge gate over
-many. The judge tier is **LLMAsJudge** embedded in a suite.
+review candidates — one is extracted straight from a recorded `run-trace.json` into `candidates/`
+as the pipeline `production trace → candidate case → reviewer supplies/verifies the expected
+result → promoted to the golden set`. A trace is ground truth about what *happened*, never about
+what *should have* happened, so extraction alone never gates anything. Where **SelfCorrectionLoop**
+runs an evaluator inline to fix a single answer, this runs the same class of evaluators as a
+pre-merge gate over many. The judge tier is **LLMAsJudge** embedded in a suite.
## Information-theoretic view
An eval suite is a proxy for product quality, and a passing suite alongside a failing product
means the proxy has drifted (see `docs/coordination-physics.md`). The working posture is that the
suite is a hypothesis incidents revise: when a real regression ships green, the fix is a new
-golden case, not a shrug. This is why the trace-sourced case matters — a captured trajectory is
-ground truth about what actually happened, so growing the suite from real traces keeps the proxy
-anchored to reality instead of to whatever the author imagined.
+golden case, not a shrug. This is why the trace-sourced candidate matters — a captured trajectory
+is ground truth about what actually happened, so growing the suite from real traces keeps the
+proxy anchored to reality instead of to whatever the author imagined. But a trace only records
+what happened, not what *should* have happened: a reviewer must supply or confirm the expected
+result before a trace-derived case can gate anything, or the suite would simply freeze the
+model's own historical mistakes into its own ground truth.
## When to use it
@@ -51,28 +61,39 @@ the honest amount of machinery for a question you will not ask twice.
## How the demo works
-Five golden cases live in `golden-cases.json`; a sixth is extracted at runtime from
-`sample-run-trace.json` (a minimal `RunTrace` in EvaluationAndMonitoring's format) by reading its
-first question and answer with `JsonDocument`. Each case runs through a `ScenarioRun` created from
-a cache-enabled `DiskBasedReportingConfiguration`; the tier's assertion decides pass/fail; any
-failure sets the process exit code to 1.
+Five golden cases live in `golden-cases.json`, each already carrying a `reviewedBy`. At runtime a
+sixth candidate is extracted from `sample-run-trace.json` (a minimal `RunTrace` in
+EvaluationAndMonitoring's format) by reading its first question and observed answer with
+`JsonDocument`, and written to `candidates/from-trace.json` with `reviewedBy: null` — it never
+joins the evaluated set. `CasePartition.Partition` splits `golden-cases.json` into cases with a
+reviewer (evaluated) and cases without one (reported and skipped). The run prints two separate
+counts, because the two states are not the same thing: `N golden case(s) awaiting sign-off` for
+`golden-cases.json` rows that already have an expected answer and tier but no reviewer yet, and
+`1 candidate case(s) awaiting review` for the trace-extracted `CandidateCase`, which has neither
+and needs a reviewer to write the expected answer from scratch. Each evaluated case runs through
+a `ScenarioRun` created from a cache-enabled
+`DiskBasedReportingConfiguration`; the tier's assertion decides pass/fail; any failure, or an
+empty evaluated set, sets the process exit code to 1.
```mermaid
flowchart LR
- G[golden-cases.json] --> S[Suite runner]
- T[sample-run-trace.json] -->|extract Q and A| S
+ T[sample-run-trace.json] -->|extract Q and observed A| C[candidate case in candidates/]
+ C -->|reviewer supplies/verifies the expected result| G[golden-cases.json]
+ G --> P{CasePartition: reviewedBy?}
+ P -->|non-empty| S[Suite runner]
+ P -->|empty/missing| W[awaiting sign-off — not evaluated]
S --> A{tier}
- A -->|exact| X[Contains check]
+ A -->|contains| X[Contains check]
A -->|nlp| F[F1Evaluator]
A -->|judge| E[EquivalenceEvaluator]
X --> R[pass/fail + exit code]
F --> R
E --> R
- S -->|via ReportingConfiguration| C[(response cache + result store)]
+ S -->|via ReportingConfiguration| Cache[(response cache + result store)]
```
-The `exact` and `nlp` tiers never call the model. The `judge` tier does, but its result is cached
-by the reporting configuration, so a second run of an unchanged suite is free.
+The `contains` and `nlp` tiers never call the model. The `judge` tier does, but its result is
+cached by the reporting configuration, so a second run of an unchanged suite is free.
## Key APIs
@@ -85,16 +106,19 @@ by the reporting configuration, so a second run of an unchanged suite is free.
| `dotnet tool run aieval report --path --output report.html` | Renders the HTML report |
```bash
-dotnet run --project RegressionEvals.AgentFramework # run the suite (exit 1 on failure)
+dotnet run --project RegressionEvals.AgentFramework # run the suite (exit 1 on failure or 0 reviewed)
dotnet run --project RegressionEvals.AgentFramework # re-run: judged case served from cache
dotnet run --project RegressionEvals.AgentFramework -- --selfcheck # offline trace-extraction check
```
## What to watch in the output
-Each case prints a `[PASS]`/`[FAIL]` line with its tier and the assertion detail, then the
-answer. The summary line reports the pass count and the `aieval` command to render the report;
-the process exits non-zero if anything failed — that is the CI gate. Run it twice: the second run
-returns the judged case from the response cache without a model call. **EvaluationAndMonitoring**
-is where the trace-sourced case comes from, and **LLMAsJudge** documents the judge tier's own
-failure modes.
+The first two lines report what was skipped, split by state: unsigned `golden-cases.json` rows
+(fully specified, just need a reviewer's sign-off) and the trace-derived candidate (no expected
+answer yet, needs one written from scratch). Each evaluated case then prints a `[PASS]`/`[FAIL]`
+line with its tier and the assertion detail,
+then the answer. The summary line reports the pass count and the `aieval` command to render the
+report; the process exits non-zero if anything failed *or* if nothing was evaluated at all — both
+are the CI gate. Run it twice: the second run returns the judged case from the response cache
+without a model call. **EvaluationAndMonitoring** is where the trace-sourced candidate comes
+from, and **LLMAsJudge** documents the judge tier's own failure modes.
diff --git a/PatternExplorer/patterns/ResourceAwareOptimization.md b/PatternExplorer/patterns/ResourceAwareOptimization.md
index c5a8d24..9671845 100644
--- a/PatternExplorer/patterns/ResourceAwareOptimization.md
+++ b/PatternExplorer/patterns/ResourceAwareOptimization.md
@@ -1,7 +1,7 @@
---
{
"title": "Resource-Aware Optimization",
- "summary": "Route each query to the cheapest model that can answer it, and stop spending when the budget runs out.",
+ "summary": "Route each query to the cheapest model that can answer it, and degrade — forcing the cheap tier or refusing reasoning-tier work — once a softly-enforced budget is crossed.",
"category": "Production controls",
"projects": [
{ "flavor": "AgentFramework", "path": "ResourceAwareOptimization.AgentFramework" },
@@ -14,9 +14,12 @@
Not every question deserves the expensive model. Resource-aware optimization puts a cheap
classifier in front of several model tiers, sends each request to the smallest tier that can
-handle it, and tracks what the answers cost. When the running total crosses a budget, the system
-degrades deliberately — forcing the cheap tier or refusing the work — instead of silently burning
-money.
+handle it, and tracks what the answers cost. The budget is **soft**: routing and refusal
+decisions are made from the running total observed *after* each call returns, not reserved
+before dispatch, so the call that pushes the total over the cap has already been paid for by the
+time anything reacts to it. Once the prior total is over budget, the system degrades
+deliberately — forcing the cheap tier for simple queries, refusing reasoning-tier work — instead
+of silently burning money. For a hard, pre-call ceiling instead, see **Bounded Execution**.
Two mechanisms combine: **tiered routing** (pick a model per request, with a fallback chain if a
tier errors) and **budget enforcement** (measure token usage, convert to cents, gate on the total).
@@ -24,7 +27,8 @@ tier errors) and **budget enforcement** (measure token usage, convert to cents,
## When to use it
- Mixed traffic where most requests are trivial and a minority genuinely need a reasoning model.
-- Hard cost ceilings per session, tenant, or user.
+- Soft cost ceilings per session, tenant, or user, where degrading gracefully is acceptable and
+ a hard pre-call cap isn't required.
- You want graceful degradation under provider outages: fall through to the next tier instead of
failing the request.
@@ -42,21 +46,24 @@ and each tier has a fallback chain if the call throws.
```mermaid
flowchart LR
- Q[User query] --> C[ClassifyQuery heuristic]
+ Q[User query] --> Chk[Check running total
from prior calls]
+ Chk -->|over budget, needs reasoning| G[Refuse, or force fast tier]
+ Chk -->|under budget, or simple query| C[ClassifyQuery heuristic]
C -->|simple| F[Fast tier
gpt-4o-mini]
C -->|reasoning| R[Reasoning tier
o4-mini]
- F --> B[Budget tracker
tokens to cents]
+ F --> B[Record actual usage
after the call returns]
R --> B
- B -->|under budget| A[Answer]
- B -->|over budget| G[Degrade to fast tier
or refuse]
+ B -.->|updates the total
for the next query| Chk
```
The flavors differ in where the logic lives. Agent Framework implements it as **two middleware
layers**: `RoutingMiddleware` on the `IChatClient` picks the tier and calls the chosen client
-directly instead of `next`, while `BudgetEnforcementMiddleware` on the agent short-circuits the
-whole run once `BudgetState.Exceeded` is true, returning a canned "I've reached my processing
-budget" message. Semantic Kernel does it with **keyed services**: three chat-completion services
-registered under the ids `fast`, `reasoning`, and `default`, resolved per query via
+directly instead of `next`, while `BudgetEnforcementMiddleware` on the agent short-circuits
+reasoning-tier queries once `BudgetState.Exceeded` is true — via `QueryRouter.RefuseForBudget`,
+so simple queries still get answered on the fast tier — returning a canned "I've reached my
+processing budget" message. Semantic Kernel does it with **keyed services**: three
+chat-completion services registered under the ids `fast`, `reasoning`, and `default`, resolved
+per query via
`GetRequiredKeyedService(sid)`, with the loop breaking out entirely once
`BudgetTracker.BudgetExceeded` flips. Its budget is a deliberately tiny 2¢ so the reasoning query
actually trips it; the Agent Framework budget is 50¢.
@@ -77,10 +84,13 @@ Both cost models are approximations hard-coded per 1K tokens: `gpt-4o-mini` 0.01
## What to watch in the output
Each query prints `[Router] Classified as: simple|reasoning`, then `[Router] Trying: ` and
-`[Router] Success with: `; a failed tier prints `[Fallback] failed: ...`. Every
-completed call prints a `[Budget]` line with token count, incremental cost, and the running total
-against the cap, and the run ends with `Total estimated cost: N.NN¢`. Watch for
-`[Router] Budget exceeded — forcing fast tier.` and `[BudgetMiddleware] Budget exceeded. Returning
-early.` in the Agent Framework flavor, and `Budget limit reached. Skipping remaining queries.` in
-Semantic Kernel. **Routing** shows the same classify-then-dispatch idea without the cost angle,
-and **Middleware** explains the interception layers this sample builds on.
+`[Router] Success with: `; a failed tier prints `[Fallback] failed: ...`. If every
+tier in the chain fails, Agent Framework prints `[Fallback] All tier models failed. Trying
+original pipeline.` and makes one more call against the underlying client — that call's cost is
+recorded too, so it is folded into the same running total. Every completed call, including that
+last-resort one, prints a `[Budget]` line with token count, incremental cost, and the running
+total against the cap, and the run ends with `Total estimated cost: N.NN¢`. Watch for
+`[Router] Budget exceeded — forcing fast tier.` and `[BudgetMiddleware] Budget exceeded. Refusing
+expensive-tier work.` in the Agent Framework flavor, and `Budget limit reached. Skipping remaining
+queries.` in Semantic Kernel. **Routing** shows the same classify-then-dispatch idea without the
+cost angle, and **Middleware** explains the interception layers this sample builds on.
diff --git a/PatternExplorer/patterns/ToolAuthorization.md b/PatternExplorer/patterns/ToolAuthorization.md
index 6acb2e9..ff812b2 100644
--- a/PatternExplorer/patterns/ToolAuthorization.md
+++ b/PatternExplorer/patterns/ToolAuthorization.md
@@ -1,7 +1,7 @@
---
{
"title": "Tool Authorization / Capability Scoping",
- "summary": "Authorize each concrete tool invocation against caller, tenant, resource, amount, expiry, and replay constraints.",
+ "summary": "Authorize each concrete tool invocation against caller, tenant, resource, amount, expiry, and replay constraints, reserving a one-time capability until the effect is committed.",
"category": "Production controls",
"projects": [ { "flavor": "AgentFramework", "path": "ToolAuthorization.AgentFramework" } ]
}
@@ -34,7 +34,27 @@ A customer-support principal receives separate short-lived capabilities for `Get
`IssueRefund`. `AuthorizedAIFunction` intercepts the concrete arguments before calling the inner
function. The policy normalizes order IDs, checks tenant and ownership, enforces the exact tool
name, denies missing or malformed authorization inputs, turns refunds over €50 into
-`ApprovalRequired`, and can consume one-time nonces.
+`ApprovalRequired`, and reserves one-time nonces.
+
+Two design choices carry as much weight as the checks themselves.
+
+**A refusal is not a tool result.** `AuthorizedAIFunction` throws `ToolAuthorizationException`
+rather than returning `"ApprovalRequired: ..."` as the function's output. Handing a refusal back on
+the tool channel gives the model a sentence to paraphrase — frequently into a claim that the work
+was done — and puts an approval request somewhere the model can answer. The host catches the
+exception, routes `decision.PendingApproval` to a human, and decides what the model is told.
+
+Throwing is necessary but not by itself sufficient: it keeps the refusal off the tool channel only
+because this host invokes the function directly. Run the same wrapper under
+`FunctionInvokingChatClient` — the loop the diagram above implies — and the framework catches the
+function's exception and feeds the model a generic error, discarding the `PendingApproval` entirely.
+A host on that path has to intercept the exception before the invocation loop does. The
+`PendingApproval` carries a *snapshot* of the arguments, not the caller's live dictionary, so the
+approver judges the values that were actually authorized.
+
+**The amount check does not depend on the grant carrying a ceiling.** `IssueRefund` is a
+money-moving tool: an absent, negative, or unparseable `amount` is refused whether or not
+`MaximumAmount` is set. A configured maximum only adds the ceiling on top of that floor.
```mermaid
flowchart LR
@@ -43,7 +63,7 @@ flowchart LR
C[Host-created capability] --> A
W --> A
A -->|allowed| T[Real tool]
- A -->|high amount| H[Approval required]
+ A -->|high amount| H[PendingApproval → human channel]
A -->|wrong tenant/resource/tool
expired or replayed| D[Denied]
```
@@ -51,15 +71,53 @@ flowchart LR
capability cannot authorize another tool. Those are separate from the argument-level denial when
the current customer asks for someone else's order.
+## One-time capabilities: reserve, then commit
+
+This sample demonstrates **reserve/commit**, not the idempotency-key alternative — the two solve
+the same replay problem and shipping both would just be two half-enforced ledgers.
+(`IdempotentToolCalls` is where the idempotency-key design lives, and it keeps its dedup record
+with the side effect, which is the right home for it.)
+
+`Authorize` moves a one-time nonce `Available -> Reserved`. `Commit` moves it `Reserved ->
+Consumed` once the effect is durable; `Release` moves it back to `Available` after a *verified*
+pre-effect failure. Burning the nonce inside `Authorize`, as an earlier version did, destroyed a
+valid capability whenever the tool failed before doing anything.
+
+`AuthorizedAIFunction` commits after the inner call returns and deliberately does **not** release
+when it throws: from inside the wrapper the failure is unverified — the effect may well have
+happened — so the reservation stands and the capability fails closed. `Release` stays a caller-
+driven act for a failure the caller has confirmed was pre-effect.
+
+The honest gap: **nothing here resolves a reservation whose commit never arrives.** If the host
+crashes between reserve and commit, the capability is stranded in `Reserved` forever, and because
+the ledger is an in-process `ConcurrentDictionary` a restart forgets it instead. A real system
+gives a reservation a lease with an expiry and a sweeper that decides — by asking the downstream
+system whether the effect landed, not by guessing — whether an expired reservation becomes
+`Consumed` or `Available`. Better still, it stores the three states in the same transactional store
+that owns the side effect, so the commit is atomic with the effect and the question never arises.
+
## Key APIs
- `DelegatingAIFunction.InvokeCoreAsync(...)` — the final host-side enforcement point.
- `ToolCapability` — immutable subject, tenant, exact tool, resource, amount, expiry, and nonce.
-- `ToolAuthorizationPolicy.Authorize(...)` — returns `Allowed`, `Denied`, or `ApprovalRequired`.
+- `ToolAuthorizationPolicy.Authorize(...)` — returns `Allowed`, `Denied`, or `ApprovalRequired`, and
+ reserves a one-time capability rather than consuming it.
+- `ToolAuthorizationPolicy.Commit(nonce)` / `Release(nonce)` — the two ends of the reservation.
+- `CapabilityState` — `Available`, `Reserved`, `Consumed`.
+- `AuthorizationDecision.PendingApproval` — tool name, argument snapshot, and reason for the human
+ channel.
+- `ToolAuthorizationException` — how a refusal leaves the tool-result channel.
- `RunPrincipal` — authenticated identity supplied by the application, never by the prompt.
## What to watch in the output
-The five probes show: own order allowed, another customer's order denied, €25 refund allowed,
-€500 refund escalated, and a tool absent from the grant denied. Each decision is printed before
-the underlying function can run.
+The probes show: own order allowed, another customer's order refused, €25 refund allowed, and a
+€500 refund escalated — printed as an out-of-band approval request, then executed only after an
+approver mints a *new* single-use capability sized to that exact request. The original capability
+is never widened, and the model plays no part in producing the new grant. A one-time `GetOrder`
+capability then succeeds once and is refused on replay, and a tool absent from the grant is
+refused. Each decision is printed before the underlying function can run.
+
+The approver's answer in the sample is a constant, marked `// ponytail:` — the sample must run
+unattended, so it cannot block on `Console.ReadLine`. `DurableHumanInTheLoop` is where the real
+version of that wait lives.
diff --git a/README.md b/README.md
index 84e695d..7245737 100644
--- a/README.md
+++ b/README.md
@@ -151,16 +151,16 @@ the catalog together; each result states its scope limits and cites a primary so
| HumanInTheLoop | Tool-call approval gates |
| IdempotentToolCalls | Retry-safe side effects: the dedup record lives with the side effect, not the caller |
| Middleware | Agent-run and function-invocation middleware (logging, latency, tool guards) |
-| ResourceAwareOptimization | Model routing under a cost budget |
-| ToolAuthorization | Capability-scoped, argument-level authorization before tool execution |
+| ResourceAwareOptimization | Model routing under a soft, post-call cost budget |
+| ToolAuthorization | Capability-scoped, argument-level authorization before tool execution; one-time grants are reserved, then committed after the effect |
### Evaluation
| Pattern | What it demonstrates |
|---|---|
-| LLMAsJudge | Judge-model rubric scoring plus a position-bias probe |
+| LLMAsJudge | Judge-model rubric scoring plus a position-bias probe that compares verdicts across balanced candidate orderings |
| RedTeaming | Deterministic leak checks first, judge second, against the real GuardRails filter |
-| RegressionEvals | Golden-dataset suite with tiered assertions, cached as a CI gate |
+| RegressionEvals | Golden-dataset suite of reviewed cases with tiered assertions, cached as a CI gate |
| TrajectoryEvaluation | Scoring the agent's tool-use path with agent evaluators |
## Setup
diff --git a/ReasoningAndActing/Program.cs b/ReasoningAndActing/Program.cs
index 8499081..c5ef453 100644
--- a/ReasoningAndActing/Program.cs
+++ b/ReasoningAndActing/Program.cs
@@ -1,3 +1,4 @@
+using Microsoft.Extensions.DependencyInjection;
using Microsoft.SemanticKernel;
using Microsoft.SemanticKernel.ChatCompletion;
using Microsoft.SemanticKernel.Connectors.OpenAI;
@@ -5,7 +6,12 @@
using Shared;
// Local kernel so the shared Settings.Kernel singleton stays unmodified
-var reactKernel = Settings.CreateKernelBuilder().Build();
+var reactBuilder = Settings.CreateKernelBuilder();
+// One filter instance for this one run: the counter lives on the instance, so registering the
+// instance (not the type) keeps the budget per-run rather than per-process.
+var toolCallBudget = new ToolCallBudgetFilter();
+reactBuilder.Services.AddSingleton(toolCallBudget);
+var reactKernel = reactBuilder.Build();
// Register tools the agent can use mid-reasoning
reactKernel.Plugins.AddFromType();
@@ -33,7 +39,9 @@ Use at most 10 tool calls before giving your final answer.
var reactSettings = new OpenAIPromptExecutionSettings
{
// SK 1.79 exposes no max-auto-invoke option on FunctionChoiceBehavior/FunctionChoiceBehaviorOptions,
- // so the tool-call loop is capped via the "at most 10 tool calls" instruction above.
+ // so the "at most 10 tool calls" instruction above is only a hint to the model. ToolCallBudgetFilter,
+ // registered on reactKernel above, is the actual control: the 11th call is refused and the
+ // auto-invocation loop is terminated, regardless of what the model intends to do next.
FunctionChoiceBehavior = FunctionChoiceBehavior.Auto()
};
@@ -41,4 +49,18 @@ Use at most 10 tool calls before giving your final answer.
// then reason about the results to compute the ratio.
var reactResponse = await reactService.GetChatMessageContentAsync(
reactHistory, reactSettings, reactKernel);
-Console.WriteLine($"ReAct Agent:\n{reactResponse.Content}\n");
\ No newline at end of file
+
+if (toolCallBudget.BudgetExhausted)
+{
+ // Same shape as BoundedExecution: a PARTIAL result, the stop reason, and an explicit
+ // incomplete label rather than silently truncated output. SK returns normally after
+ // Terminate (with empty content), so the filter's flag — not an exception — is the signal.
+ Console.WriteLine("Result status: PARTIAL");
+ Console.WriteLine($"Stop reason: {toolCallBudget.StopReason}");
+ Console.WriteLine("ReAct Agent:\nStopped at the tool-call budget before a final answer was reached; " +
+ "any reasoning gathered so far is incomplete.\n");
+}
+else
+{
+ Console.WriteLine($"ReAct Agent:\n{reactResponse.Content}\n");
+}
diff --git a/ReasoningAndActing/ToolCallBudgetFilter.cs b/ReasoningAndActing/ToolCallBudgetFilter.cs
new file mode 100644
index 0000000..7736b18
--- /dev/null
+++ b/ReasoningAndActing/ToolCallBudgetFilter.cs
@@ -0,0 +1,49 @@
+using Microsoft.SemanticKernel;
+
+namespace ReasoningAndActing;
+
+///
+/// Hard bound on tool calls: the prompt asks the model to stop at 10, this enforces it.
+///
+///
+/// This must be an setting context.Terminate,
+/// not an that throws. SK 1.79's
+/// FunctionCallsProcessor.ExecuteFunctionCallAsync wraps every auto-invoked call in a
+/// catch-all that turns any exception into a tool-result error message and keeps looping, so a
+/// throwing filter blocks the tool body but not the loop — and hands the model the refusal to
+/// paraphrase. Terminate is the only stop SK honours; see
+/// GoalSettingsAndMonitoring.SemanticKernel.GoalMonitoringFilter for the same mechanism.
+///
+/// The counter lives on the instance, so the budget is per filter instance: register one instance
+/// per run (Program.cs does), not the type as a process-wide singleton.
+///
+public class ToolCallBudgetFilter : IAutoFunctionInvocationFilter
+{
+ public const int MaxToolCalls = 10;
+
+ private int _toolCalls;
+
+ /// True once a call was refused because the budget was exhausted. SK returns
+ /// normally after Terminate, so this flag is the only route the stop has out.
+ public bool BudgetExhausted { get; private set; }
+
+ public string StopReason => $"Tool-call budget of {MaxToolCalls} exhausted.";
+
+ ///
+ /// Counts every auto-invoked tool call and passes it through while under budget. The call that
+ /// would exceed the budget never reaches , and terminates the
+ /// auto-invocation loop instead of feeding the model an error to reason about.
+ ///
+ public async Task OnAutoFunctionInvocationAsync(
+ AutoFunctionInvocationContext context, Func next)
+ {
+ if (Interlocked.Increment(ref _toolCalls) > MaxToolCalls)
+ {
+ BudgetExhausted = true;
+ context.Terminate = true;
+ return;
+ }
+
+ await next(context);
+ }
+}
diff --git a/RegressionEvals.AgentFramework/CasePartition.cs b/RegressionEvals.AgentFramework/CasePartition.cs
new file mode 100644
index 0000000..6846206
--- /dev/null
+++ b/RegressionEvals.AgentFramework/CasePartition.cs
@@ -0,0 +1,24 @@
+namespace RegressionEvals.AgentFramework;
+
+// The rule that decides which golden cases are trustworthy enough to gate a release: a case is
+// evaluated only once a human has reviewed it (a non-empty ReviewedBy); everything else is
+// awaiting review and must never be silently evaluated. This is what stops a trace-derived
+// candidate - or any hand-added row missing a reviewer - from freezing an unreviewed answer into
+// the suite.
+public static class CasePartition
+{
+ public static (IReadOnlyList Evaluated, IReadOnlyList AwaitingReview) Partition(
+ IEnumerable cases)
+ {
+ var evaluated = new List();
+ var awaitingReview = new List();
+ foreach (var c in cases)
+ (string.IsNullOrEmpty(c.ReviewedBy) ? awaitingReview : evaluated).Add(c);
+ return (evaluated, awaitingReview);
+ }
+
+ // A suite that evaluated zero cases is not a passing suite, even if nothing failed - the same
+ // class of defect as freezing an unreviewed trace answer into the gate.
+ public static int GateExitCode(int evaluatedCount, int failureCount) =>
+ evaluatedCount == 0 || failureCount > 0 ? 1 : 0;
+}
diff --git a/RegressionEvals.AgentFramework/GoldenCase.cs b/RegressionEvals.AgentFramework/GoldenCase.cs
new file mode 100644
index 0000000..ca9a56b
--- /dev/null
+++ b/RegressionEvals.AgentFramework/GoldenCase.cs
@@ -0,0 +1,7 @@
+namespace RegressionEvals.AgentFramework;
+
+// A trace is ground truth about what HAPPENED, never about what SHOULD have happened. Extracting
+// one gives a candidate; a reviewer supplies or confirms the expected result before it can gate a
+// release. Promoting the model's own historical answer freezes its mistakes into the suite.
+public record CandidateCase(string Id, string Question, string ObservedAnswer, string SourceTrace);
+public record GoldenCase(string Id, string Question, string ExpectedAnswer, string Tier, string ReviewedBy);
diff --git a/RegressionEvals.AgentFramework/Program.cs b/RegressionEvals.AgentFramework/Program.cs
index 0943b82..4dc9734 100644
--- a/RegressionEvals.AgentFramework/Program.cs
+++ b/RegressionEvals.AgentFramework/Program.cs
@@ -6,6 +6,7 @@
using Microsoft.Extensions.AI.Evaluation.Quality;
using Microsoft.Extensions.AI.Evaluation.Reporting;
using Microsoft.Extensions.AI.Evaluation.Reporting.Storage;
+using RegressionEvals.AgentFramework;
using Shared;
if (args.FirstOrDefault() == "--selfcheck") { SelfCheck(); return; }
@@ -14,11 +15,34 @@
var baseDir = AppContext.BaseDirectory;
var web = new JsonSerializerOptions(JsonSerializerDefaults.Web);
-var cases = JsonSerializer.Deserialize>(
+var allCases = JsonSerializer.Deserialize>(
File.ReadAllText(Path.Combine(baseDir, "golden-cases.json")), web)!;
-// The canonical "production trace -> eval case" pipeline: one case comes straight from a
-// recorded EvaluationAndMonitoring trajectory rather than being hand-written.
-cases.Add(ExtractTraceCase(Path.Combine(baseDir, "sample-run-trace.json")));
+var (cases, awaitingReview) = CasePartition.Partition(allCases);
+
+// The canonical "production trace -> candidate case" pipeline: extraction pulls the question and
+// the OBSERVED answer straight from a recorded EvaluationAndMonitoring trajectory. A trace is
+// ground truth about what HAPPENED, never about what SHOULD have happened, so it lands in
+// candidates/ for a reviewer to supply/confirm the expected answer - it never joins `cases` above.
+List candidates = [ExtractTraceCase(Path.Combine(baseDir, "sample-run-trace.json"))];
+var candidatesDir = Path.Combine(baseDir, "candidates");
+Directory.CreateDirectory(candidatesDir);
+// ponytail: candidates/ lives under bin/, build output wiped by `dotnet clean`, so a candidate
+// does not survive here to actually be reviewed. A real repo commits candidates beside
+// golden-cases.json and reviews the promotion (filling in expectedAnswer/tier/reviewedBy) in a PR.
+foreach (var candidate in candidates)
+ File.WriteAllText(Path.Combine(candidatesDir, $"{candidate.Id}.json"), JsonSerializer.Serialize(new
+ {
+ candidate.Id, candidate.Question, candidate.ObservedAnswer, candidate.SourceTrace,
+ reviewedBy = (string?)null
+ }, web));
+
+// Two different states, two different counts: an awaiting-review GoldenCase already has an
+// ExpectedAnswer and Tier and just needs sign-off, while a CandidateCase has neither and needs a
+// reviewer to write the expected answer from scratch. Folding them into one number would call a
+// fully-specified unsigned golden case a "candidate", contradicting the distinction this task exists
+// to enforce.
+Console.WriteLine($"{awaitingReview.Count} golden case(s) awaiting sign-off - not evaluated.");
+Console.WriteLine($"{candidates.Count} candidate case(s) awaiting review - not evaluated.\n");
const string policy =
"TechCorp laptops include a two-year limited warranty. Defective products may be " +
@@ -42,7 +66,7 @@
var (passed, detail) = c.Tier switch
{
- "exact" => (answer.Contains(c.ExpectedAnswer, StringComparison.OrdinalIgnoreCase),
+ "contains" => (answer.Contains(c.ExpectedAnswer, StringComparison.OrdinalIgnoreCase),
$"contains \"{c.ExpectedAnswer}\""),
"nlp" => await NlpTierAsync(run, c, answer),
"judge" => await JudgeTierAsync(run, c, answer),
@@ -57,7 +81,7 @@
Console.WriteLine($"\n{cases.Count - failures}/{cases.Count} passed. " +
"Re-run to see cached (zero-call) evaluation; generate an HTML report with:");
Console.WriteLine($" dotnet tool run aieval report --path {Path.Combine(baseDir, "eval-results")} --output report.html");
-Environment.Exit(failures == 0 ? 0 : 1);
+Environment.Exit(CasePartition.GateExitCode(cases.Count, failures));
async Task<(bool, string)> NlpTierAsync(ScenarioRun run, GoldenCase c, string answer)
{
@@ -81,13 +105,14 @@ [new ChatMessage(ChatRole.User, c.Question)],
return (pass, $"equivalence={eq.Value} ({eq.Interpretation?.Rating})");
}
-GoldenCase ExtractTraceCase(string tracePath)
+CandidateCase ExtractTraceCase(string tracePath)
{
using var doc = JsonDocument.Parse(File.ReadAllText(tracePath));
var firstCall = doc.RootElement.GetProperty("modelCalls")[0];
string TextOf(string arrayName) => firstCall.GetProperty(arrayName)[0]
.GetProperty("contents")[0].GetProperty("payload").GetProperty("value").GetString()!;
- return new GoldenCase("from-trace", TextOf("messages"), TextOf("responseMessages"), "judge");
+ return new CandidateCase(
+ "from-trace", TextOf("messages"), TextOf("responseMessages"), Path.GetFileName(tracePath));
}
static void SelfCheck()
@@ -103,5 +128,3 @@ static void SelfCheck()
if (q != "Q?") throw new Exception("extraction broken");
Console.WriteLine("selfcheck ok");
}
-
-record GoldenCase(string Id, string Question, string ExpectedAnswer, string Tier);
diff --git a/RegressionEvals.AgentFramework/golden-cases.json b/RegressionEvals.AgentFramework/golden-cases.json
index 24aba0a..f645ae4 100644
--- a/RegressionEvals.AgentFramework/golden-cases.json
+++ b/RegressionEvals.AgentFramework/golden-cases.json
@@ -1,7 +1,7 @@
[
- { "id": "warranty-length", "question": "How long is the TechCorp laptop warranty?", "expectedAnswer": "two-year limited warranty", "tier": "exact" },
- { "id": "return-window", "question": "How many days do I have to return a defective product?", "expectedAnswer": "30 days", "tier": "exact" },
- { "id": "warranty-phrasing", "question": "What warranty comes with TechCorp laptops?", "expectedAnswer": "TechCorp laptops include a two-year limited warranty.", "tier": "nlp" },
- { "id": "returns-phrasing", "question": "How do I return a defective item?", "expectedAnswer": "Return defective products within 30 days with your order number.", "tier": "nlp" },
- { "id": "warranty-semantic", "question": "Tell me about laptop warranty coverage.", "expectedAnswer": "TechCorp laptops have a two-year limited warranty.", "tier": "judge" }
+ { "id": "warranty-length", "question": "How long is the TechCorp laptop warranty?", "expectedAnswer": "two-year limited warranty", "tier": "contains", "reviewedBy": "maintainer" },
+ { "id": "return-window", "question": "How many days do I have to return a defective product?", "expectedAnswer": "30 days", "tier": "contains", "reviewedBy": "maintainer" },
+ { "id": "warranty-phrasing", "question": "What warranty comes with TechCorp laptops?", "expectedAnswer": "TechCorp laptops include a two-year limited warranty.", "tier": "nlp", "reviewedBy": "maintainer" },
+ { "id": "returns-phrasing", "question": "How do I return a defective item?", "expectedAnswer": "Return defective products within 30 days with your order number.", "tier": "nlp", "reviewedBy": "maintainer" },
+ { "id": "warranty-semantic", "question": "Tell me about laptop warranty coverage.", "expectedAnswer": "TechCorp laptops have a two-year limited warranty.", "tier": "judge", "reviewedBy": "maintainer" }
]
diff --git a/ResourceAwareOptimization.AgentFramework/Program.cs b/ResourceAwareOptimization.AgentFramework/Program.cs
index 1826e4f..c449b5e 100644
--- a/ResourceAwareOptimization.AgentFramework/Program.cs
+++ b/ResourceAwareOptimization.AgentFramework/Program.cs
@@ -17,7 +17,7 @@
var reasoningModel = (Client: azureClient.GetChatClient(ReasoningModelDeployment).AsIChatClient(),
ModelId: ReasoningModelDeployment);
-var budget = new BudgetState(50);
+var softBudget = new BudgetState(50);
async Task RoutingMiddleware(
IEnumerable messages,
@@ -34,7 +34,7 @@ async Task RoutingMiddleware(
: new[] { fastModel, reasoningModel };
// If budget is exceeded, force the cheapest model
- if (budget.Exceeded)
+ if (softBudget.Exceeded)
{
Console.WriteLine(" [Router] Budget exceeded — forcing fast tier.");
chain = [fastModel];
@@ -50,7 +50,7 @@ async Task RoutingMiddleware(
var response = await client.GetResponseAsync(
messages.ToList(), options, cancellationToken);
- budget.RecordUsage(modelId, response);
+ softBudget.RecordUsage(modelId, response);
Console.WriteLine($" [Router] Success with: {modelId}");
return response;
@@ -66,7 +66,11 @@ async Task RoutingMiddleware(
// All models failed — call the original pipeline as last resort
Console.WriteLine(" [Fallback] All tier models failed. Trying original pipeline.");
- return await chatClient.GetResponseAsync(messages, options, cancellationToken);
+ var fallbackResponse = await chatClient.GetResponseAsync(messages, options, cancellationToken);
+ // chatClient wraps fastModel.Client (see the pipeline built below), so the cost belongs to
+ // fastModel.ModelId — not to whichever tier was last attempted in the chain above.
+ softBudget.RecordUsage(fastModel.ModelId, fallbackResponse);
+ return fallbackResponse;
}
async Task BudgetEnforcementMiddleware(
@@ -78,7 +82,7 @@ async Task BudgetEnforcementMiddleware(
{
// Soft budget: only refuse work that would need the expensive tier —
// simple queries still run and RoutingMiddleware forces the fast tier for them.
- if (QueryRouter.RefuseForBudget(messages, budget.Exceeded))
+ if (QueryRouter.RefuseForBudget(messages, softBudget.Exceeded))
{
Console.WriteLine(" [BudgetMiddleware] Budget exceeded. Refusing expensive-tier work.");
return new AgentResponse([
@@ -125,4 +129,4 @@ You are a helpful support agent. Answer user questions concisely and accurately.
Console.WriteLine($"Agent: {result}");
}
-Console.WriteLine($"\nTotal estimated cost: {budget.TotalCostCents:F2}¢");
\ No newline at end of file
+Console.WriteLine($"\nTotal estimated cost: {softBudget.TotalCostCents:F2}¢");
\ No newline at end of file
diff --git a/ResourceAwareOptimization.SemanticKernel/Program.cs b/ResourceAwareOptimization.SemanticKernel/Program.cs
index 0695e3b..2bb98e2 100644
--- a/ResourceAwareOptimization.SemanticKernel/Program.cs
+++ b/ResourceAwareOptimization.SemanticKernel/Program.cs
@@ -1,3 +1,4 @@
+using System.ClientModel;
using Microsoft.Extensions.DependencyInjection;
using Microsoft.SemanticKernel;
using Microsoft.SemanticKernel.ChatCompletion;
@@ -31,7 +32,7 @@
var kernel = builder.Build();
// 2¢ budget — deliberately small so the demo actually trips it on the reasoning query
-var budgetTracker = new BudgetTracker(2);
+var softBudgetTracker = new BudgetTracker(2);
// Maps a service id back to its model for cost lookup
var modelForService = new Dictionary
@@ -96,12 +97,18 @@ async Task HandleQueryAsync(string userQuery)
var response = await chatService.GetChatMessageContentAsync(
history, settings, kernel);
- budgetTracker.Record(modelForService[sid], response);
+ softBudgetTracker.Record(modelForService[sid], response);
Console.WriteLine($" [Router] Success with: {sid}");
return response.Content ?? "No response generated.";
}
- catch (Exception ex)
+ // Only transient failures (HTTP errors, timeouts, network) should trigger the fallback
+ // tier — same narrowing as the AgentFramework twin. That twin also excludes caller
+ // cancellation; this sample threads no CancellationToken through, so a TaskCanceledException
+ // here can only be a request timeout. Add the `&& !cancellationToken.IsCancellationRequested`
+ // guard alongside a token if one is ever introduced.
+ catch (Exception ex) when (ex is HttpOperationException or ClientResultException
+ or HttpRequestException or TaskCanceledException)
{
Console.WriteLine($" [Fallback] {sid} failed: {ex.Message}");
}
@@ -124,11 +131,11 @@ async Task HandleQueryAsync(string userQuery)
var answer = await HandleQueryAsync(query);
Console.WriteLine($"Agent: {answer}");
- if (budgetTracker.BudgetExceeded)
+ if (softBudgetTracker.BudgetExceeded)
{
Console.WriteLine("\n?Budget limit reached. Skipping remaining queries.");
break;
}
}
-Console.WriteLine($"\nTotal estimated cost: {budgetTracker.TotalCostCents:F2}¢");
\ No newline at end of file
+Console.WriteLine($"\nTotal estimated cost: {softBudgetTracker.TotalCostCents:F2}¢");
\ No newline at end of file
diff --git a/ToolAuthorization.AgentFramework/AuthorizationDecision.cs b/ToolAuthorization.AgentFramework/AuthorizationDecision.cs
index 45bbeca..d0d4422 100644
--- a/ToolAuthorization.AgentFramework/AuthorizationDecision.cs
+++ b/ToolAuthorization.AgentFramework/AuthorizationDecision.cs
@@ -1,10 +1,49 @@
+using System.Collections.Frozen;
+
namespace ToolAuthorization.AgentFramework;
public enum AuthorizationOutcome { Allowed, Denied, ApprovalRequired }
+///
+/// A request for a human decision. It travels on the host's approval channel, never back to the
+/// model as tool output. is a snapshot: the caller's live argument
+/// dictionary stays mutable, so an approver must see the values that were actually judged.
+///
+public sealed record PendingApproval
+{
+ public PendingApproval(string toolName, IEnumerable> arguments, string reason)
+ {
+ ToolName = toolName;
+ Arguments = arguments.ToFrozenDictionary(StringComparer.Ordinal);
+ Reason = reason;
+ }
+
+ public string ToolName { get; }
+ public IReadOnlyDictionary Arguments { get; }
+ public string Reason { get; }
+}
+
public sealed record AuthorizationDecision(AuthorizationOutcome Outcome, string Reason)
{
+ /// Set only when is .
+ public PendingApproval? PendingApproval { get; init; }
+
public static AuthorizationDecision Allow() => new(AuthorizationOutcome.Allowed, "Capability permits this invocation.");
public static AuthorizationDecision Deny(string reason) => new(AuthorizationOutcome.Denied, reason);
public static AuthorizationDecision RequireApproval(string reason) => new(AuthorizationOutcome.ApprovalRequired, reason);
+
+ public static AuthorizationDecision RequireApproval(string reason, string toolName,
+ IEnumerable> arguments) =>
+ new(AuthorizationOutcome.ApprovalRequired, reason) { PendingApproval = new(toolName, arguments, reason) };
+}
+
+///
+/// Thrown when a tool invocation is not allowed. Refusals leave the tool-result channel entirely:
+/// the host decides what, if anything, the model is told, and an approval request reaches a human
+/// instead of becoming text the model can argue with or paraphrase into a false success.
+///
+public sealed class ToolAuthorizationException(AuthorizationDecision decision)
+ : InvalidOperationException($"{decision.Outcome}: {decision.Reason}")
+{
+ public AuthorizationDecision Decision { get; } = decision;
}
diff --git a/ToolAuthorization.AgentFramework/AuthorizedAIFunction.cs b/ToolAuthorization.AgentFramework/AuthorizedAIFunction.cs
index 246234b..1b0ce6b 100644
--- a/ToolAuthorization.AgentFramework/AuthorizedAIFunction.cs
+++ b/ToolAuthorization.AgentFramework/AuthorizedAIFunction.cs
@@ -13,8 +13,20 @@ public sealed class AuthorizedAIFunction(
{
var decision = policy.Authorize(principal, capability, Name, arguments);
Console.WriteLine($" [authorization] {Name}: {decision.Outcome} — {decision.Reason}");
- return decision.Outcome == AuthorizationOutcome.Allowed
- ? await base.InvokeCoreAsync(arguments, cancellationToken)
- : $"{decision.Outcome}: {decision.Reason}";
+
+ // A refusal is not a tool result. Returning "ApprovalRequired: ..." as text hands the model
+ // a sentence to paraphrase — often into a claim that the work was done — and puts the
+ // approval request on the channel the model controls. The host gets an exception instead.
+ if (decision.Outcome != AuthorizationOutcome.Allowed)
+ throw new ToolAuthorizationException(decision);
+
+ var result = await base.InvokeCoreAsync(arguments, cancellationToken);
+
+ // Commit only after the inner call returns. There is deliberately no Release on the
+ // exception path: from inside this wrapper the failure is unverified — the effect may
+ // already have happened — so the reservation stands and the capability fails closed.
+ // Release is a caller-driven act for a failure the caller has verified was pre-effect.
+ policy.Commit(capability.Nonce);
+ return result;
}
}
diff --git a/ToolAuthorization.AgentFramework/Program.cs b/ToolAuthorization.AgentFramework/Program.cs
index 54df746..27567f4 100644
--- a/ToolAuthorization.AgentFramework/Program.cs
+++ b/ToolAuthorization.AgentFramework/Program.cs
@@ -1,3 +1,4 @@
+using System.Globalization;
using Microsoft.Extensions.AI;
using ToolAuthorization.AgentFramework;
@@ -9,21 +10,74 @@
};
var policy = new ToolAuthorizationPolicy(orderOwners);
-ToolCapability Grant(string tool, decimal? maximumAmount = null) => new(
+ToolCapability Grant(string tool, decimal? maximumAmount = null, bool oneTime = false) => new(
principal.SubjectId, principal.TenantId, tool, new Dictionary(), maximumAmount,
- DateTimeOffset.UtcNow.AddMinutes(5), Guid.NewGuid().ToString("N"));
+ DateTimeOffset.UtcNow.AddMinutes(5), Guid.NewGuid().ToString("N"), oneTime);
-var getOrder = new AuthorizedAIFunction(
- AIFunctionFactory.Create((string orderId) => $"{orderId}: paid, awaiting shipment", "GetOrder"),
- principal, Grant("GetOrder"), policy);
-var issueRefund = new AuthorizedAIFunction(
- AIFunctionFactory.Create((string orderId, decimal amount) => $"Refunded €{amount:F2} for {orderId}", "IssueRefund"),
- principal, Grant("IssueRefund", maximumAmount: 50m), policy);
+var getOrderFunction = AIFunctionFactory.Create(
+ (string orderId) => $"{orderId}: paid, awaiting shipment", "GetOrder");
+var refundFunction = AIFunctionFactory.Create(
+ (string orderId, decimal amount) => $"Refunded €{amount:F2} for {orderId}", "IssueRefund");
+
+var getOrder = new AuthorizedAIFunction(getOrderFunction, principal, Grant("GetOrder"), policy);
+var issueRefund = new AuthorizedAIFunction(refundFunction, principal, Grant("IssueRefund", maximumAmount: 50m), policy);
async Task Show(AIFunction tool, AIFunctionArguments arguments)
{
Console.WriteLine($"\n{tool.Name}({string.Join(", ", arguments.Select(a => $"{a.Key}={a.Value}"))})");
- Console.WriteLine($" result: {await tool.InvokeAsync(arguments)}");
+ try
+ {
+ Console.WriteLine($" result: {await tool.InvokeAsync(arguments)}");
+ }
+ catch (ToolAuthorizationException ex)
+ {
+ // The model never sees any of this — because this host invokes the function directly.
+ // Throwing alone does not guarantee it: under FunctionInvokingChatClient the framework
+ // catches a function exception and feeds the model a generic error, discarding the
+ // PendingApproval. A host on that path must intercept before the loop does.
+ if (ex.Decision.PendingApproval is not { } pending)
+ {
+ Console.WriteLine($" refused: {ex.Decision.Reason}");
+ return;
+ }
+ await Escalate(pending);
+ }
+}
+
+async Task Escalate(PendingApproval pending)
+{
+ Console.WriteLine($" approval required: {pending.Reason}");
+ Console.WriteLine(" → sent to the approver's channel, not returned to the model:");
+ Console.WriteLine($" {pending.ToolName}({string.Join(", ", pending.Arguments.Select(a => $"{a.Key}={a.Value}"))})");
+
+ // ponytail: the approver's answer is a constant so the sample runs with no TTY and no
+ // credentials. Upgrade path: await a durable approval record (see DurableHumanInTheLoop) and
+ // resume from it — never Console.ReadLine, which would hang an unattended run.
+ var approverApproved = true;
+ if (!approverApproved)
+ {
+ Console.WriteLine(" approver declined — the tool is never invoked.");
+ return;
+ }
+
+ // The approver issues a *new* capability, single-use and sized to this one request — the
+ // ceiling is read off the snapshot the approver saw, not hard-coded, so it stays correct when
+ // the probe changes. The original capability is never widened, and the model had no part in
+ // producing this grant. The sample has exactly one escalating tool, so the inner function is
+ // named directly.
+ var requested = Convert.ToDecimal(pending.Arguments["amount"], CultureInfo.InvariantCulture);
+ var approved = new AuthorizedAIFunction(refundFunction, principal,
+ Grant(pending.ToolName, maximumAmount: requested, oneTime: true), policy);
+ Console.WriteLine($" approver granted a single-use capability capped at €{requested:F2}.");
+ try
+ {
+ Console.WriteLine($" result: {await approved.InvokeAsync(new AIFunctionArguments(pending.Arguments.ToDictionary()))}");
+ }
+ catch (ToolAuthorizationException ex)
+ {
+ // The approver's own grant is enforced too, and the sample must run unattended to the end.
+ Console.WriteLine($" approved call still refused: {ex.Decision.Reason}");
+ }
}
Console.WriteLine("=== Capability-scoped tools ===");
@@ -32,6 +86,11 @@ async Task Show(AIFunction tool, AIFunctionArguments arguments)
await Show(issueRefund, new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 25m });
await Show(issueRefund, new AIFunctionArguments { ["orderId"] = "ORD-100", ["amount"] = 500m });
+Console.WriteLine("\n=== One-time capability: reserve, then commit ===");
+var oneTimeGetOrder = new AuthorizedAIFunction(getOrderFunction, principal, Grant("GetOrder", oneTime: true), policy);
+await Show(oneTimeGetOrder, new AIFunctionArguments { ["orderId"] = "ORD-100" }); // reserved, then committed
+await Show(oneTimeGetOrder, new AIFunctionArguments { ["orderId"] = "ORD-100" }); // refused: already consumed
+
var missingGrant = policy.Authorize(principal, Grant("GetOrder"), "GetInternalFraudScore", []);
Console.WriteLine($"\nGetInternalFraudScore: {missingGrant.Outcome} — {missingGrant.Reason}");
Console.WriteLine("DeleteCustomer: not registered; the model never receives its definition.");
diff --git a/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs b/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs
index 2a88deb..b054e60 100644
--- a/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs
+++ b/ToolAuthorization.AgentFramework/ToolAuthorizationPolicy.cs
@@ -5,12 +5,30 @@
namespace ToolAuthorization.AgentFramework;
+///
+/// The lifecycle of a one-time capability. reserves; the host commits once
+/// the effect is durable, or releases after a verified pre-effect failure.
+///
+public enum CapabilityState { Available, Reserved, Consumed }
+
public sealed class ToolAuthorizationPolicy(
IReadOnlyDictionary orderOwners,
TimeProvider? timeProvider = null)
{
+ /// Tools that move money always need a present, positive, parseable amount.
+ // ponytail: "which tools move money" is a hand-maintained set in the policy rather than a
+ // property of the tool registration. The ceiling is a silent one: register `IssueCredit`, grant
+ // it without a MaximumAmount, forget this line, and it gets no amount floor at all — absent and
+ // negative amounts authorize. Upgrade path: move the flag onto the capability/tool registration
+ // so adding a money tool cannot compile without declaring it.
+ private static readonly HashSet MoneyMovingTools = new(StringComparer.Ordinal) { "IssueRefund" };
+
private readonly TimeProvider _timeProvider = timeProvider ?? TimeProvider.System;
- private readonly ConcurrentDictionary _usedNonces = new(StringComparer.Ordinal);
+
+ // ponytail: in-process nonce ledger, so a restart forgets every reservation and a crashed host
+ // leaks one. Upgrade path: the same three states in the transactional store that already owns
+ // the side effect, which is where the commit becomes atomic with the effect.
+ private readonly ConcurrentDictionary _nonceStates = new(StringComparer.Ordinal);
public AuthorizationDecision Authorize(RunPrincipal principal, ToolCapability capability,
string toolName, AIFunctionArguments arguments)
@@ -33,7 +51,10 @@ public AuthorizationDecision Authorize(RunPrincipal principal, ToolCapability ca
return AuthorizationDecision.Deny("Order is outside the capability resource scope.");
}
- if (capability.MaximumAmount is { } maximum)
+ // The amount is validated whenever the tool moves money, not only when the grant happens to
+ // carry a ceiling: an absent, negative, or unparseable amount is never authorizable. A
+ // configured maximum only adds the ceiling on top of that.
+ if (MoneyMovingTools.Contains(toolName) || capability.MaximumAmount is not null)
{
decimal? amount;
try { amount = ReadDecimal(arguments, "amount"); }
@@ -43,16 +64,32 @@ public AuthorizationDecision Authorize(RunPrincipal principal, ToolCapability ca
}
if (amount is null or <= 0)
return AuthorizationDecision.Deny("A valid amount is required for authorization.");
- if (amount > maximum)
- return AuthorizationDecision.RequireApproval($"Amount €{amount:F2} exceeds the capability limit of €{maximum:F2}.");
+ if (capability.MaximumAmount is { } maximum && amount > maximum)
+ return AuthorizationDecision.RequireApproval(
+ $"Amount €{amount:F2} exceeds the capability limit of €{maximum:F2}.", toolName, arguments);
}
- if (capability.OneTimeUse && !_usedNonces.TryAdd(capability.Nonce, 0))
- return AuthorizationDecision.Deny("One-time capability has already been used.");
+ // Reserve last, so a refusal above never costs the caller a one-time capability.
+ if (capability.OneTimeUse && !TryReserve(capability.Nonce))
+ return AuthorizationDecision.Deny("One-time capability is already reserved or consumed.");
return AuthorizationDecision.Allow();
}
+ /// Call once the effect is durable. Moves Reserved -> Consumed.
+ public void Commit(string nonce) => _nonceStates.TryUpdate(nonce, CapabilityState.Consumed, CapabilityState.Reserved);
+
+ ///
+ /// Call after a verified pre-effect failure — the caller must know the side effect did
+ /// not happen. Moves Reserved -> Available; a committed capability stays consumed.
+ ///
+ public void Release(string nonce) => _nonceStates.TryUpdate(nonce, CapabilityState.Available, CapabilityState.Reserved);
+
+ /// Atomic Available -> Reserved; two racing callers cannot both win.
+ private bool TryReserve(string nonce) =>
+ _nonceStates.TryAdd(nonce, CapabilityState.Reserved) ||
+ _nonceStates.TryUpdate(nonce, CapabilityState.Reserved, CapabilityState.Available);
+
private static string? ReadString(AIFunctionArguments arguments, string name) =>
arguments.TryGetValue(name, out var value) ? value switch
{