diff --git a/AgenticPatterns.Tests/NewOrchestrationPatternTests.cs b/AgenticPatterns.Tests/NewOrchestrationPatternTests.cs index 070595a..d771484 100644 --- a/AgenticPatterns.Tests/NewOrchestrationPatternTests.cs +++ b/AgenticPatterns.Tests/NewOrchestrationPatternTests.cs @@ -100,6 +100,7 @@ public async Task TwoHandlersFeedingEachOtherAreStoppedByTheGenerationCap() public void AnEventNobodySubscribesToIsRecordedNotDropped() { var bus = new EventBus(maxEvents: 10, maxGeneration: 5); + bus.RegisterTerminal("nobody-listens"); // Not queued, but not lost either. Which list it lands in is asserted by // EventBusTaxonomyTests - a terminal event is a workflow output, not a delivery failure. diff --git a/AgenticPatterns.Tests/ReviewFollowupTests.cs b/AgenticPatterns.Tests/ReviewFollowupTests.cs index 3bfee32..9bf7e93 100644 --- a/AgenticPatterns.Tests/ReviewFollowupTests.cs +++ b/AgenticPatterns.Tests/ReviewFollowupTests.cs @@ -106,16 +106,30 @@ public class EventBusTaxonomyTests static AgentEvent Event(string topic, int generation = 0) => new(topic, "payload", "test", generation); [Fact] - public void AnEventNobodySubscribesToIsTerminalNotADeadLetter() + public void ADeclaredTerminalTopicIsAnOutcomeNotADeadLetter() { var bus = new EventBus(maxEvents: 10, maxGeneration: 5); + bus.RegisterTerminal("workflow-finished"); - bus.Publish(Event("nobody-listens")); + bus.Publish(Event("workflow-finished")); Assert.Single(bus.TerminalEvents); Assert.Empty(bus.DeadLetters); } + [Fact] + public void AnUndeclaredTopicIsADeliveryFailure() + { + var bus = new EventBus(maxEvents: 10, maxGeneration: 5); + + // The typo case: neither subscribed nor registered as terminal. Inferring "terminal" from + // an empty handler list would make this a successful outcome. + bus.Publish(Event("DecisionMdae")); + + Assert.Empty(bus.TerminalEvents); + Assert.Equal(Refusal.NoSubscriber, bus.DeadLetters.Single().Reason); + } + [Fact] public async Task AGenerationCapProducesADeadLetterWithThatReason() { diff --git a/ChainOfVerification.AgentFramework/Program.cs b/ChainOfVerification.AgentFramework/Program.cs index 4715886..695d150 100644 --- a/ChainOfVerification.AgentFramework/Program.cs +++ b/ChainOfVerification.AgentFramework/Program.cs @@ -15,6 +15,13 @@ // Roman founding dates that is not a remote possibility. What you get is a blind cross-check: // strong evidence when it disagrees, weak evidence when it agrees. Real independence needs a // different source - retrieval, a tool, a second model - which is what **AgenticRAG** brings. +// +// Which is why a disagreement here is a FLAG, not a correction. No new fact entered the system +// between the draft and the check; preferring the check would just be preferring the model's +// later guess to its earlier one. And the checker's own "CONFIDENT:" is a self-report, not +// evidence - a label the same weights wrote about themselves cannot promote the second guess +// into an authority. So disagreement resolves to contested, and settling it is somebody else's +// job: retrieval, a calculator, an authoritative record. var client = Settings.ChatClient; var lowTemp = new ChatClientAgentRunOptions(new ChatOptions { Temperature = 0.2f }); @@ -70,6 +77,8 @@ Return at most 8 claims. // ── 3. Answer each check blind ─────────────────────────────────────────────── // A fresh stateless agent, one question per run, no session, no draft in context. // This is the structural difference from a self-critique loop. +// The CONFIDENT:/UNCERTAIN: prefix is worth having, and worth being clear about: it tells a +// reader how firmly the checker holds its answer. It does not decide which value wins. var verifier = new ChatClientAgent(client, name: "Verifier", instructions: "Answer the single factual question as precisely as you can. Begin your reply " + "with CONFIDENT: or UNCERTAIN: — uncertainty is a useful answer and a guess " + @@ -86,26 +95,29 @@ Return at most 8 claims. Console.WriteLine($" [{claim.Id}] {question}\n → {answer.ReplaceLineEndings(" ")}\n"); // ── 4. Revise ──────────────────────────────────────────────────────────────── -// The reviser sees the draft and the independent answers side by side, and is told which one -// wins when they disagree. Without that instruction the model tends to defend its own draft. +// The reviser sees the draft and the blind answers side by side, and is told that neither wins. +// Without an explicit rule the model either defends its own draft or capitulates to the check, +// and both are the same mistake: treating one of two same-weights guesses as the authority. var reviser = new ChatClientAgent(client, name: "Reviser", instructions: """ You are given a draft answer and a set of blind cross-checks: the same model answering each factual question with the draft out of sight. - A cross-check is not an authority. Resolve each disagreement into one of three - outcomes, and never silently keep the draft: + A cross-check is not an authority, and neither is the draft. Resolve every + claim into one of two outcomes, and never silently keep a disagreement: - - check CONFIDENT and disagrees -> correct the draft to the check. - - check UNCERTAIN and disagrees -> mark the claim contested: state both - values and that they could not be settled. Do not pick one. - - check agrees -> leave the claim as it is. Agreement between - a model and itself is weak evidence, so do not upgrade the wording. + - check agrees -> leave the claim as it is. Agreement between a model and + itself is weak evidence, so do not upgrade the wording. + - check disagrees -> mark the claim contested: state both values and that + nothing here could settle them. Do NOT pick one, and do not let the + check's own CONFIDENT:/UNCERTAIN: label pick for you — that label is the + checker describing itself, not evidence about the world. Report it as + what it is: "the blind check said X (self-reported confident)". Do not add new claims. - Output the corrected answer, then "Changes:" listing corrections and contested - claims separately. + Output the answer with contested claims marked inline, then "Changes:" + listing every contested claim and what would settle it. """); var evidence = string.Join("\n", answers.Select(a => $"Q: {a.Question}\nA: {a.Answer}")); @@ -141,7 +153,9 @@ claims separately. : "\nEvery claim in the draft was extracted and cross-checked."); Console.WriteLine("Cross-checked is not verified: the checker shares the drafter's weights, so " - + "agreement rules out anchoring on the draft, not a shared misconception."); + + "agreement rules out anchoring on the draft, not a shared misconception — and " + + "a disagreement is a flag, not a correction. Nothing here can settle one; that " + + "needs a source outside the model."); // Structured-output shape for the planning call. internal sealed record PlannedClaim(int Id, string Text, string Value, string Question); diff --git a/DualLlm.AgentFramework/Program.cs b/DualLlm.AgentFramework/Program.cs index 482119c..6886ed1 100644 --- a/DualLlm.AgentFramework/Program.cs +++ b/DualLlm.AgentFramework/Program.cs @@ -147,10 +147,11 @@ You extract one value from a document. You have no tools and no ability to act. // ── What taint does NOT buy ────────────────────────────────────────────────── // The run above depends on the quarantined model reporting the real total. Suppose it had -// complied with the injection instead and returned 48000.00: that is a well-formed decimal, -// inside the range bound, and it would file. Control flow is still intact - no new step, no new -// tool - and the expense is still wrong. Only the value policy stops it, and it is worth seeing -// that stop happen rather than trusting that it would. +// complied with the injection instead and returned 48000.00: that is a well-formed decimal, so +// it passes the type gate untouched. Control flow is still intact - no new step, no new tool - +// and the expense is still wrong. What stops it is the second, separate gate: the unattended +// value policy holds it for a person. Worth watching both gates run rather than trusting that +// they would. var injected = new Value("invoice_total", "decimal", "48000.00", Tainted: true); Console.WriteLine($"\n=== If the quarantined model had returned the injected figure ==="); Console.WriteLine($" coerces to a valid decimal: {DataFlowPlan.TryCoerce(injected, "decimal", out _)}"); diff --git a/EventDrivenAgents.AgentFramework/EventBus.cs b/EventDrivenAgents.AgentFramework/EventBus.cs index 6c9823a..493b7ea 100644 --- a/EventDrivenAgents.AgentFramework/EventBus.cs +++ b/EventDrivenAgents.AgentFramework/EventBus.cs @@ -24,11 +24,17 @@ public sealed record DeadLetter(AgentEvent Event, Refusal Reason); /// /// Refused events are kept with a reason rather than dropped: a silent drop looks exactly like a /// handler that never fired. +/// +/// Terminal topics are declared, not inferred. "Nobody subscribes" is ambiguous - it is either +/// the workflow finishing or a topic name nobody will ever match - and in a system whose wiring +/// IS the subscription table, a typo is the likeliest wiring bug there is. Inferring terminal +/// from an empty handler list makes `DecisionMdae` a successful outcome. public sealed class EventBus(int maxEvents, int maxGeneration) { readonly Channel channel = Channel.CreateUnbounded(); readonly Dictionary>>>> handlers = new(StringComparer.OrdinalIgnoreCase); + readonly HashSet terminalTopics = new(StringComparer.OrdinalIgnoreCase); /// Events refused by the budget or the generation cap. A dead letter is a FAILURE - something /// that could not be processed. @@ -43,6 +49,9 @@ public sealed class EventBus(int maxEvents, int maxGeneration) public int Published { get; private set; } + /// Declares a topic as an outcome of the run: legal to publish, nothing reacts to it. + public void RegisterTerminal(string topic) => terminalTopics.Add(topic); + public void Subscribe(string topic, Func>> handler) { if (!handlers.TryGetValue(topic, out var list)) handlers[topic] = list = []; @@ -50,7 +59,8 @@ public void Subscribe(string topic, Func= maxEvents) @@ -61,7 +71,11 @@ public bool Publish(AgentEvent @event) if (!handlers.ContainsKey(@event.Topic)) { - // Nothing left to react to. That is the workflow ending, not a delivery failing. + // A declared terminal topic is the workflow ending, not a delivery failing. An + // undeclared one is a message addressed to nobody - which is what a misspelled topic + // looks like, and it should be loud. + if (!terminalTopics.Contains(@event.Topic)) return Refuse(@event, Refusal.NoSubscriber); + TerminalEvents.Add(@event); return false; } diff --git a/EventDrivenAgents.AgentFramework/Program.cs b/EventDrivenAgents.AgentFramework/Program.cs index c12ac5f..0e3a6f1 100644 --- a/EventDrivenAgents.AgentFramework/Program.cs +++ b/EventDrivenAgents.AgentFramework/Program.cs @@ -46,9 +46,11 @@ "Approver", 0) ]); -// Nothing subscribes to DecisionMade. That makes it a TERMINAL event - the workflow finished - -// which the bus records separately from dead letters. A workflow output filed as a delivery -// failure makes the dead-letter queue useless as an alarm. +// Nothing subscribes to DecisionMade, and that is declared rather than inferred: it is a TERMINAL +// topic, an outcome of the run, which the bus records separately from dead letters. A workflow +// output filed as a delivery failure makes the dead-letter queue useless as an alarm - and, +// inferred the other way round, a topic nobody can match becomes a silent success. +bus.RegisterTerminal("DecisionMade"); bus.Publish(new AgentEvent("PurchaseRequested", "Purchase request: 3-year contract with a Norwegian logistics SaaS vendor, EUR 84,000/year, " + @@ -60,10 +62,20 @@ await bus.RunToCompletionAsync(e => Console.WriteLine($"\n=== Done: {bus.Published} events dispatched ==="); foreach (var terminal in bus.TerminalEvents) Console.WriteLine($" terminal: {terminal.Topic} (gen {terminal.Generation}) from {terminal.Source} " + - "— nothing subscribes, the workflow ends here"); + $"— registered as an outcome, the workflow ends here\n {terminal.Payload}"); foreach (var dead in bus.DeadLetters) Console.WriteLine($" dead-letter: {dead.Event.Topic} (gen {dead.Event.Generation}) " + $"from {dead.Event.Source} — {dead.Reason}"); if (bus.DeadLetters.Count == 0) Console.WriteLine(" no dead letters: nothing hit the event budget or the generation cap."); + +// ── What "the wiring is the subscription table" costs ──────────────────────── +// There is no compiler for a topic name. A handler that publishes "DecisionMdae" is not a broken +// handler you can find by reading it - it is a message addressed to nobody, and the only reason +// it is visible here is that the bus refuses topics it was never told about. +Console.WriteLine("\n=== A misspelled topic, published by nobody's mistake in particular ==="); +bus.Publish(new AgentEvent("DecisionMdae", "APPROVE", "Approver", 3)); +var typo = bus.DeadLetters[^1]; +Console.WriteLine($" dead-letter: {typo.Event.Topic} — {typo.Reason}. Neither subscribed nor " + + "registered as terminal, so it is a delivery failure and says so."); diff --git a/GraphRAG.AgentFramework/Program.cs b/GraphRAG.AgentFramework/Program.cs index eff8aa5..391bf47 100644 --- a/GraphRAG.AgentFramework/Program.cs +++ b/GraphRAG.AgentFramework/Program.cs @@ -88,22 +88,40 @@ cluster is about and what recurs in it. // believing them. Without this the provenance chain breaks exactly here: documents carry ids, // relations carry ids, and then a free-text summary carries whatever the model happened to // retain - after which the final answerer is asked to "cite the incident ids" and can only - // repeat, or invent, what reached it. An id the summariser names that is not in the community - // is a fabrication, and it is cheap to catch because the truth is a set the host already has. + // repeat, or invent, what reached it. + // + // The citation is the INTERSECTION, and both halves of that matter. An id the summariser + // names that is not in the community is a fabrication. An id in the community that the + // summariser did not use is not support for this summary - attaching the whole community + // would trade fabricated provenance for inflated provenance, which reads exactly as + // convincing and is just as untrue. The host can only vouch for what the model claimed AND + // the graph contains. var actual = community.Select(r => r.SourceDoc).Distinct(StringComparer.OrdinalIgnoreCase).Order().ToArray(); var claimed = summarised.SourceDocumentIds ?? []; + var validated = claimed.Intersect(actual, StringComparer.OrdinalIgnoreCase) + .Distinct(StringComparer.OrdinalIgnoreCase).Order().ToArray(); var fabricated = claimed.Except(actual, StringComparer.OrdinalIgnoreCase).ToArray(); - - // Cite what the community actually contains - the host's set, not the model's recollection. - summaries.Add($"Community {index + 1} [sources: {string.Join(", ", actual)}]: {summarised.Summary}"); + var unused = actual.Except(validated, StringComparer.OrdinalIgnoreCase).ToArray(); Console.WriteLine($"\n Community {index + 1} ({community.Count} relations, " + $"{community.SelectMany(r => new[] { r.From, r.To }).Distinct(StringComparer.OrdinalIgnoreCase).Count()} entities)"); Console.WriteLine($" {summarised.Summary}"); - Console.WriteLine($" sources (from the graph): {string.Join(", ", actual)}"); + Console.WriteLine($" validated sources: {(validated.Length == 0 ? "(none)" : string.Join(", ", validated))}"); if (fabricated.Length > 0) - Console.WriteLine($" [provenance] summariser also claimed {string.Join(", ", fabricated)} — " + - "not in this community, dropped"); + Console.WriteLine($" [provenance] claimed {string.Join(", ", fabricated)} — not in this " + + "community, dropped as fabricated"); + if (unused.Length > 0) + Console.WriteLine($" [provenance] {string.Join(", ", unused)} are in this community but " + + "were not claimed as used — not cited as support"); + + if (validated.Length == 0) + { + Console.WriteLine(" [provenance] no claimed id survives — the summary is not carried " + + "to the global answer. A claim nothing can be traced to is not evidence."); + continue; + } + + summaries.Add($"Community {index + 1} [sources: {string.Join(", ", validated)}]: {summarised.Summary}"); } var answerer = new ChatClientAgent(client, name: "Answerer", @@ -114,9 +132,12 @@ cluster is about and what recurs in it. // ── 3a. Global question: answered from community summaries ─────────────────── Console.WriteLine("\n=== Global question ==="); Console.WriteLine("Q: What is the recurring systemic problem across these incidents?\n"); -Console.WriteLine(await answerer.RunAsync( - $"Community summaries:\n{string.Join("\n", summaries)}\n\n" + - "Q: What is the recurring systemic problem across these incidents?", options: precise)); +Console.WriteLine(summaries.Count == 0 + ? " (no summary kept its provenance, so there is no evidence to answer from. Saying that is\n" + + " the answer; synthesising one from unattributable text is the failure this guards.)" + : (await answerer.RunAsync( + $"Community summaries:\n{string.Join("\n", summaries)}\n\n" + + "Q: What is the recurring systemic problem across these incidents?", options: precise)).Text); // ── 3b. Local question: answered from a neighbourhood ──────────────────────── var neighbourhood = graph.Neighbourhood("Team Atlas", hops: 2); diff --git a/LeastToMost.AgentFramework/Program.cs b/LeastToMost.AgentFramework/Program.cs index 47ce7fb..1acf3bb 100644 --- a/LeastToMost.AgentFramework/Program.cs +++ b/LeastToMost.AgentFramework/Program.cs @@ -16,6 +16,11 @@ // state does not make the chain safer by itself - it makes it CHECKABLE, which is only a benefit // if something checks. So the host attaches a deterministic verifier where one exists, and says // plainly where one does not. +// +// And a verifier that cannot refuse is only telemetry. Once a deterministic check has proved a +// value wrong, that value does not become an established fact for the steps after it, and does +// not become the run's answer either. The run ends CONTESTED instead - a worse-looking output +// and a far better one than a number the host already knows is false. var client = Settings.ChatClient; var precise = new ChatClientAgentRunOptions(new ChatOptions { Temperature = 0.1f }); @@ -57,7 +62,7 @@ answers to every earlier subproblem - treat those answers as established facts start: new DateOnly(2025, 3, 3), upgradeEffective: new DateOnly(2025, 7, 3), cancelled: new DateOnly(2025, 10, 15), beforeUpgrade: 14m, afterUpgrade: 22m); -var solved = new List<(SubProblem Step, string Answer)>(); +var solved = new List(); foreach (var step in steps) { var known = solved.Count == 0 @@ -83,33 +88,56 @@ answers to every earlier subproblem - treat those answers as established facts // subproblems in most chains do not. Where one exists, a failed check is caught before the // answer becomes an "established fact" that every later step cites. var isFinal = step.Order == steps.Count; - if (isFinal) + if (!isFinal) { - var check = StepChecks.AgainstTotal(answer, expectedTotal); - Console.WriteLine($"\n[{step.Order}] {step.Question}\n → {answer.ReplaceLineEndings(" ")}"); - Console.WriteLine($" [check] {check.Detail}"); - - if (!check.Passed) - { - // One retry, with the discrepancy named. Not a loop: an unbounded "try again" on a - // criterion the model cannot see is how a sample becomes a hang. - answer = (await solver.RunAsync( - $"{prompt}\n\nA deterministic check of your answer failed: {check.Detail}. " + - "Recompute carefully, listing each billing date and its charge.", options: precise)).Text.Trim(); + solved.Add(new SolvedStep(step, answer, StepStatus.Unverified)); + Console.WriteLine($"\n[{step.Order}] {step.Question}\n → {answer.ReplaceLineEndings(" ")} [no verifier for this step]"); + continue; + } - var recheck = StepChecks.AgainstTotal(answer, expectedTotal); - Console.WriteLine($" → retry: {answer.ReplaceLineEndings(" ")}"); - Console.WriteLine($" [check] {(recheck.Passed ? recheck.Detail : recheck.Detail + " — CONTESTED, not settled")}"); - } + var check = StepChecks.AgainstTotal(answer, expectedTotal); + Console.WriteLine($"\n[{step.Order}] {step.Question}\n → {answer.ReplaceLineEndings(" ")}"); + Console.WriteLine($" [check] {check.Detail}"); - solved.Add((step, answer)); - continue; + if (!check.Passed) + { + // One retry, with the discrepancy named. Not a loop: an unbounded "try again" on a + // criterion the model cannot see is how a sample becomes a hang. + answer = (await solver.RunAsync( + $"{prompt}\n\nA deterministic check of your answer failed: {check.Detail}. " + + "Recompute carefully, listing each billing date and its charge.", options: precise)).Text.Trim(); + + check = StepChecks.AgainstTotal(answer, expectedTotal); + Console.WriteLine($" → retry: {answer.ReplaceLineEndings(" ")}"); + Console.WriteLine($" [check] {check.Detail}"); } - solved.Add((step, answer)); - Console.WriteLine($"\n[{step.Order}] {step.Question}\n → {answer.ReplaceLineEndings(" ")} [no verifier for this step]"); + solved.Add(new SolvedStep(step, answer, check.Passed ? StepStatus.Accepted : StepStatus.Contested)); + + // A step the host has proved wrong is not handed to the steps after it. Here it is the last + // step, so there are none - but the rule is the point, not the arithmetic. + if (!check.Passed) break; } -Console.WriteLine($"\n=== Final answer ===\n{solved[^1].Answer}"); +var last = solved[^1]; +Console.WriteLine(last.Status switch +{ + StepStatus.Contested => $""" + + === Run stopped: the final answer failed a deterministic check === + candidate: {last.Answer.ReplaceLineEndings(" ")} + host rule: EUR {expectedTotal:F2} + status: CONTESTED + + The host will not hand on a value it has already proved wrong. A + chain that prints this is working; one that prints the number anyway + was only ever logging its verifier. + """, + _ => $"\n=== Final answer (checked against the host's schedule) ===\n{last.Answer}" +}); + +internal enum StepStatus { Accepted, Unverified, Contested } + +internal sealed record SolvedStep(SubProblem Step, string Answer, StepStatus Status); internal sealed record ProposedSteps(string[] Steps); diff --git a/PatternExplorer/patterns/ChainOfVerification.md b/PatternExplorer/patterns/ChainOfVerification.md index 0ee198e..2653444 100644 --- a/PatternExplorer/patterns/ChainOfVerification.md +++ b/PatternExplorer/patterns/ChainOfVerification.md @@ -70,13 +70,18 @@ Four stages, of which only the third is unusual: stateless run — no session, no draft, no siblings. The questions run concurrently because they are genuinely independent; that independence is the point, and the parallelism is a free consequence of it. -4. **Revise, into three outcomes.** The reviser sees the draft and the cross-checks together, and - is explicitly told that a cross-check is *not an authority*. Each disagreement resolves as: - check confident and disagrees → correct the draft; check uncertain and disagrees → mark the - claim **contested**, stating both values without picking one; check agrees → leave it alone and - do **not** upgrade the wording, because agreement between a model and itself is weak evidence. - The verifier is asked to prefix its answers `CONFIDENT:` or `UNCERTAIN:` so that split is - available to act on. +4. **Revise, into two outcomes.** The reviser sees the draft and the cross-checks together, and is + explicitly told that neither is an authority. Check agrees → leave it alone and do **not** + upgrade the wording, because agreement between a model and itself is weak evidence. Check + disagrees → mark the claim **contested**, stating both values without picking one. + + *A disagreement is a flag, not a correction.* No new fact entered the system between the draft + and the check — preferring the check would only be preferring the model's later guess to its + earlier one. The verifier's `CONFIDENT:`/`UNCERTAIN:` prefix is worth having and worth being + precise about: it is the checker describing itself, not evidence about the world, and letting + that word decide which value wins promotes a self-report into an authority. It is reported, not + obeyed. Settling a contested claim needs a source outside the model, which is where + **AgenticRAG** or a deterministic tool comes in. Between 2 and 3 sits the host's contribution, `VerificationGate`. Models drift toward leading questions — it is the natural way to phrase a check — and a question containing the drafted @@ -129,12 +134,12 @@ the gate ===` lists each question next to what the draft claimed, which is the c what is about to be tested. The section worth reading closely is `=== Blind cross-checks ===`. Compare each to the -`draft says:` value above it, and note the `CONFIDENT:`/`UNCERTAIN:` prefix — that is what decides -whether a disagreement becomes a correction or a contested claim. +`draft says:` value above it. The `CONFIDENT:`/`UNCERTAIN:` prefix tells you how firmly the checker +holds its answer and nothing more — a disagreement is contested either way. -Then `=== Cross-checked answer ===`, whose `Changes:` list is the deliverable: corrections and -contested claims, listed separately. An empty change list means the draft and the check agreed, -which — per above — is the weaker of the two possible results, not a clean bill of health. +Then `=== Cross-checked answer ===`, whose `Changes:` list is the deliverable: every contested +claim and what would settle it. An empty change list means the draft and the check agreed, which — +per above — is the weaker of the two possible results, not a clean bill of health. Finally `=== Coverage ===`. If it says `never checked: 2`, two of the draft's specifics carry the draft's confidence and nothing more, and the run says so rather than letting the header imply diff --git a/PatternExplorer/patterns/DualLlm.md b/PatternExplorer/patterns/DualLlm.md index 4de14ae..6c1e4fd 100644 --- a/PatternExplorer/patterns/DualLlm.md +++ b/PatternExplorer/patterns/DualLlm.md @@ -84,7 +84,7 @@ It said nothing about whether the value is *true* — and that distinction is th most often over-read. Suppose the quarantined model had complied with the injection and returned `48000.00`. That is a -well-formed decimal, inside the range bound, and it files. Control flow was never subverted — no +well-formed decimal, so it passes the type gate untouched. Control flow was never subverted — no new step, no new tool — and the expense is still wrong. Taint stopped untrusted content from becoming an *instruction*; it did nothing to make it a *fact*. diff --git a/PatternExplorer/patterns/EventDrivenAgents.md b/PatternExplorer/patterns/EventDrivenAgents.md index 0bcfcdd..c911858 100644 --- a/PatternExplorer/patterns/EventDrivenAgents.md +++ b/PatternExplorer/patterns/EventDrivenAgents.md @@ -50,10 +50,10 @@ subscribed: - `FindingsProduced` → **Risk** → publishes `RiskAssessed` - `RiskAssessed` → **Approver** → publishes `DecisionMade` -Nothing subscribes to `DecisionMade`. That is deliberate: it lands in `DeadLetters` and the run -reports it. An unroutable event that is *dropped* looks exactly like a handler that never fired, -which is the debugging experience event-driven systems are notorious for; keeping it makes the -terminal event visible instead of missing. +Nothing subscribes to `DecisionMade`, and the host *declares* that: `bus.RegisterTerminal( +"DecisionMade")`. It lands in `TerminalEvents` and the run reports it. An unroutable event that is +*dropped* looks exactly like a handler that never fired, which is the debugging experience +event-driven systems are notorious for; keeping it makes the outcome visible instead of missing. `EventBus` is a `Channel` plus a subscription dictionary, with every limit checked in `Publish`. `RunToCompletionAsync` drains the channel and republishes each handler's output at @@ -66,13 +66,20 @@ and its overflow modes either block a producer or silently drop. The quantity ne is different — how many events the run may *accept*, and how deep a chain may go — and both are counted before anything is queued. -**Terminal events are not dead letters.** An event nobody subscribes to has finished the workflow; -an event refused by the generation cap or the run budget has failed. Filing both in one list makes -the dead-letter queue useless as an alarm, which matters the moment this bus is composed with +**Terminal events are not dead letters.** An event that finished the workflow is an output; an +event refused by the generation cap or the run budget has failed. Filing both in one list makes the +dead-letter queue useless as an alarm, which matters the moment this bus is composed with **AgentCommunicationFaultTolerance**, where a dead letter means *requeue or escalate*. So the bus keeps `TerminalEvents` and `DeadLetters` separately, and every dead letter carries a `Refusal` reason: `NoSubscriber`, `GenerationLimit`, or `RunBudgetExceeded`. +**And terminal is declared, not inferred.** "Nobody subscribes" is ambiguous: it is either the +workflow finishing or a topic name nothing will ever match. In a system whose wiring *is* the +subscription table there is no compiler for a topic string, so a typo is the likeliest bug there +is — and inferring terminal from an empty handler list turns `DecisionMdae` into a successful +outcome. Terminal topics are registered; anything else with no subscriber is a `NoSubscriber` dead +letter. The run publishes one misspelled event at the end to show it. + ```mermaid flowchart TB I[PurchaseRequested gen 0] --> B{EventBus
budget + generation cap} @@ -82,7 +89,8 @@ flowchart TB K -->|RiskAssessed gen 2| B B --> A[Approver] A -->|DecisionMade gen 3| B - B -->|no subscriber| D[Dead letters] + B -->|registered terminal| T[Terminal events] + B -->|unknown topic, cap, budget| D[Dead letters] ``` ## Key APIs @@ -94,7 +102,9 @@ flowchart TB budget; publishing directly would let a handler bypass both. - `EventBus.Publish` returning `bool` — not queued is a normal outcome with a visible record, not an exception. -- `bus.TerminalEvents` — workflow outputs nobody subscribes to. +- `EventBus.RegisterTerminal(topic)` — declares a topic as an outcome of the run. Without it, an + unsubscribed topic is a delivery failure, which is what a misspelled one actually is. +- `bus.TerminalEvents` — declared workflow outputs. - `bus.DeadLetters` — `DeadLetter(Event, Refusal)`, so the report says *why*, not just *that*. ## What to watch in the output @@ -103,9 +113,14 @@ Each dispatch prints `── Topic (gen N, from Source) ──` followed by the generation counter climb: it is the depth of the reaction chain, and it is what the cap acts on. At the end, `=== Done: N events dispatched ===` followed by two separate lists. `DecisionMade` -appears as `terminal: … nothing subscribes, the workflow ends here` — an output, not a failure — -and on a clean run the dead-letter list is explicitly empty. That separation is the point: if a -dead letter ever appears, something was genuinely refused, and the `Refusal` says which limit. +appears as `terminal: … registered as an outcome, the workflow ends here`, with the approver's +decision underneath it — an output, not a failure — +and up to that point the dead-letter list is explicitly empty. That separation is the point: if a +dead letter appears, something was genuinely refused, and the `Refusal` says which limit. + +Then the last block publishes `DecisionMdae` — one transposition away from the real topic — and it +comes back as `dead-letter: DecisionMdae — NoSubscriber`. That is the whole argument for declaring +terminal topics: the same event, under the inferred rule, would have been filed as a success. To see the mechanism that matters, add a subscription from `DecisionMade` back to `PurchaseRequested` and re-run. Without the generation cap that is an infinite billed loop; with it diff --git a/PatternExplorer/patterns/GraphRAG.md b/PatternExplorer/patterns/GraphRAG.md index 207929f..aeca211 100644 --- a/PatternExplorer/patterns/GraphRAG.md +++ b/PatternExplorer/patterns/GraphRAG.md @@ -71,10 +71,14 @@ then a free-text summary carries whatever the model happened to retain. The fina to "cite the incident ids", and can only repeat what reached it, or invent. So the summariser returns a structured `CommunitySummary(Summary, SourceDocumentIds)`, **and the -host checks those ids against the graph rather than believing them.** Ids the model names that are -not in the community are reported and dropped; what gets attached to the summary is the set the -host already knows. Verifying is cheap here precisely because the truth is a set the host holds — -which is the difference between GraphRAG and summarising some graph-shaped text. +host checks those ids against the graph rather than believing them.** What gets attached is the +*intersection*, and both halves of that matter. An id the model names that the community does not +contain is a fabrication, reported and dropped. An id the community contains that the model did not +use is not support for this summary — attaching the whole community would trade fabricated +provenance for **inflated** provenance, which reads exactly as convincing and is just as untrue. A +summary with no id surviving is not carried to the global answer at all: a claim nothing can be +traced to is not evidence. Verifying is cheap here precisely because the truth is a set the host +holds — which is the difference between GraphRAG and summarising some graph-shaped text. **5. Answer, two ways.** - *Global:* "what is the recurring systemic problem" — answered from community summaries alone. @@ -101,8 +105,8 @@ flowchart TB - `KnowledgeGraph.Communities()` — union-find over the relations, groups ordered largest first. - `KnowledgeGraph.Neighbourhood(entity, hops)` — breadth-limited traversal for local questions. - `Relation.SourceDoc` — every edge remembers its document, so answers can cite incident ids. -- `agent.RunAsync(...)` — summary plus claimed source ids, checked against the - community's actual ids before either is used. +- `agent.RunAsync(...)` — summary plus claimed source ids. The citation attached + to the summary is `claimed ∩ actual`: fabricated ids dropped, unclaimed ids not inflated in. ## What to watch in the output @@ -118,10 +122,12 @@ everything else to join into one component through the shared Postgres cluster a chain. If you see four or five tiny communities instead, extraction drifted on entity names; that is the failure this pipeline has, and the relation list above is where you diagnose it. -Each community also prints `sources (from the graph): INC-…` — the provenance the answerer will -cite, taken from the host's set rather than the summariser's memory. A -`[provenance] summariser also claimed …` line means the model named an id the community does not -contain; it is dropped, and seeing it occasionally is the check earning its place. +Each community also prints `validated sources: INC-…` — the provenance the answerer will cite, +which is what the summariser claimed *and* the graph confirms. Two `[provenance]` lines can follow. +`claimed … not in this community` means the model named an id that does not exist here; it is +dropped as fabricated. `… are in this community but were not claimed as used` means the reverse: +real documents that this particular summary does not rest on, so they are not cited as support for +it. Seeing either occasionally is the check earning its place. Then the two answers. The global one should name weak change management around shared infrastructure, citing manual rollbacks and the shared Postgres cluster — a claim no single report diff --git a/PatternExplorer/patterns/LeastToMost.md b/PatternExplorer/patterns/LeastToMost.md index d400c02..a5a4ef4 100644 --- a/PatternExplorer/patterns/LeastToMost.md +++ b/PatternExplorer/patterns/LeastToMost.md @@ -71,8 +71,13 @@ intermediate state inspectable does not make it correct. The actual benefit is one step further along: externalised state can be **checked**, if something checks it. So the host attaches a deterministic verifier where one exists — here `StepChecks` recomputes the billing schedule from the problem's own rules and compares it against the final -answer's stated total. A failure gets one retry with the discrepancy named; a second failure is -reported as contested rather than quietly accepted. +answer's stated total. A failure gets one retry with the discrepancy named. + +**And a verifier that cannot refuse is only telemetry.** A second failure does not become the run's +answer: the step is recorded `Contested`, the chain stops rather than handing a value to steps that +would cite it as established, and the run ends with the candidate, the host's figure, and the word +`CONTESTED` instead of a total. Printing the number anyway — with the check's own failure logged +directly above it — is the version of this pattern that looks rigorous and is not. Most steps have no verifier, and the run says so with `[no verifier for this step]` rather than implying coverage it does not have. That is the honest situation in most chains, and it is why the @@ -118,8 +123,11 @@ for work twice. Most steps end `[no verifier for this step]`. The final one ends `[check] EUR 144.00 matches the schedule computed by the host` — the only claim in the whole run that anything actually tested. If -it ever prints a mismatch, watch the retry: the model is handed the discrepancy and recomputes, and -if it still disagrees the run says **CONTESTED** rather than shipping the number. +it ever prints a mismatch, watch the retry: the model is handed the discrepancy and recomputes. If +it still disagrees, the run does not print a final answer at all — it prints +`=== Run stopped: the final answer failed a deterministic check ===` with the candidate beside the +host's figure. A `SolvedStep` carries `Accepted` / `Unverified` / `Contested`, and only the first +two are answers. **ChainofThoughts** is the single-call version; **Planning** turns the decomposition into a validated tool plan rather than a question chain; **SelfNote** is the same "prepare, then answer" diff --git a/PatternExplorer/patterns/ProactiveClarification.md b/PatternExplorer/patterns/ProactiveClarification.md index 9bbdedf..8675de2 100644 --- a/PatternExplorer/patterns/ProactiveClarification.md +++ b/PatternExplorer/patterns/ProactiveClarification.md @@ -96,9 +96,12 @@ flowchart TB - `agent.RunAsync(...)` — structured output splits "what the request said" from "what I want to ask", so the host can screen the second against the first. -- `ClarificationGate.Screen(slots, filled, questions, maxQuestions)` — returns every question - with a rejection reason or `null`, so the run can print what it chose not to ask. Deciding by - *slot* rather than by question text is what makes "one question per slot" enforceable. +- `ClarificationGate.Screen(slots, filled, questions, maxQuestions)` — returns every question with + the slot it resolved to and a rejection reason or `null`, so the run can print what it chose not + to ask. Deciding by *slot* rather than by question text is what makes "one question per slot" + enforceable — and `ScreenedQuestion.TargetSlot` carries that decision forward to the merge, so + the lexical match happens once and the two stages cannot drift into disagreeing about what a + question was asking. - `parser.RunAsync(...)` — splits one free-text reply into per-slot answers. Model-parsed, therefore untrusted, therefore gated. - `ClarificationGate.Merge(filled, knownSlots, askedSlots, answers)` — writes answers into slot diff --git a/ProactiveClarification.AgentFramework/ClarificationGate.cs b/ProactiveClarification.AgentFramework/ClarificationGate.cs index fc329e9..c368975 100644 --- a/ProactiveClarification.AgentFramework/ClarificationGate.cs +++ b/ProactiveClarification.AgentFramework/ClarificationGate.cs @@ -5,7 +5,10 @@ namespace ProactiveClarification.AgentFramework; /// decides which ones are worth a human's attention. public sealed record Slot(string Name, string[] Keywords); -public sealed record ScreenedQuestion(string Question, string? RejectedBecause) +/// `TargetSlot` is the slot the screen resolved this question to - carried out of `Screen` +/// rather than re-derived from the question text later. The lexical match happens once, so the +/// screen and the merge cannot drift into disagreeing about what a question was asking. +public sealed record ScreenedQuestion(string Question, string? TargetSlot, string? RejectedBecause) { public bool Allowed => RejectedBecause is null; } @@ -75,7 +78,7 @@ public static IReadOnlyList Screen( var reason = Reject(target); if (reason is null) asked.Add(target!.Name); - screened.Add(new ScreenedQuestion(question, reason)); + screened.Add(new ScreenedQuestion(question, target?.Name, reason)); } return screened; diff --git a/ProactiveClarification.AgentFramework/Program.cs b/ProactiveClarification.AgentFramework/Program.cs index b26ef63..e3bfc71 100644 --- a/ProactiveClarification.AgentFramework/Program.cs +++ b/ProactiveClarification.AgentFramework/Program.cs @@ -79,8 +79,7 @@ Never ask about a slot you listed as filled. // told, and then "assumes" the thing it was just told. var askedSlots = screened .Where(q => q.Allowed) - .Select(q => slots.First(s => - s.Keywords.Any(k => q.Question.Contains(k, StringComparison.OrdinalIgnoreCase))).Name) + .Select(q => q.TargetSlot!) .ToHashSet(StringComparer.OrdinalIgnoreCase); if (!string.IsNullOrWhiteSpace(reply))