From 8f3340ebd20021aaf0f26a8b9f49bc17b810df54 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 15:39:52 +0800 Subject: [PATCH 01/22] fix(replan): share semantic discharge and vision input projection in TypeScript Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../control_plane/effect_runtime_handlers.ts | 2 + loopx/control_plane/quota/turn_envelope.ts | 2 + .../autonomous_replan_obligation.py | 23 +-- .../work_items/progress_observation.py | 148 ++++-------------- .../work_items/replan_semantics.ts | 102 ++++++++++++ .../test_replan_writeback_projection.py | 63 ++++++++ .../control_plane_ts/replan_semantics.test.ts | 46 ++++++ 7 files changed, 248 insertions(+), 138 deletions(-) create mode 100644 loopx/control_plane/work_items/replan_semantics.ts create mode 100644 tests/control_plane/test_replan_writeback_projection.py create mode 100644 tests/control_plane_ts/replan_semantics.test.ts diff --git a/loopx/control_plane/effect_runtime_handlers.ts b/loopx/control_plane/effect_runtime_handlers.ts index edc387e188..339b07643a 100644 --- a/loopx/control_plane/effect_runtime_handlers.ts +++ b/loopx/control_plane/effect_runtime_handlers.ts @@ -110,6 +110,7 @@ import { } from "./turn_driver/delivery_continuity.ts"; import { reduceTurnSettlementTransaction } from "./turn_driver/settlement.ts"; import { evaluateHostTodoCompletion } from "./turn_driver/host_todo_completion.ts"; +import { projectReplanSemantics } from "./work_items/replan_semantics.ts"; import { projectReplanSettlementContract, projectTodoLifecycleSettlementReentry, @@ -736,6 +737,7 @@ export function createEffectRuntimeHandlers( ["turn.settlement.reduce", reduceTurnSettlementTransaction], ["turn.host_todo_completion.evaluate", evaluateHostTodoCompletion], ["work_item.replan_settlement.project", projectReplanSettlementContract], + ["work_item.replan_semantics.project", projectReplanSemantics], [ "work_item.replan_settlement.reentry", projectTodoLifecycleSettlementReentry, diff --git a/loopx/control_plane/quota/turn_envelope.ts b/loopx/control_plane/quota/turn_envelope.ts index 5a2133ab2b..34ba28c525 100644 --- a/loopx/control_plane/quota/turn_envelope.ts +++ b/loopx/control_plane/quota/turn_envelope.ts @@ -257,6 +257,8 @@ function replanActionPacket(payload: JsonObject): JsonObject | null { "schema_version", "decision", "obligation_id", "uncovered_frontier", "required_outcome", "allowed_terminal", "bounded_frontier", ]); + const writeback = object(source.writeback_contract); + if (writeback.vision_authoring) compact.writeback_contract = writeback; return Object.keys(compact).length > 0 ? compact : null; } diff --git a/loopx/control_plane/work_items/autonomous_replan_obligation.py b/loopx/control_plane/work_items/autonomous_replan_obligation.py index 2b19df8f59..38a30d6754 100644 --- a/loopx/control_plane/work_items/autonomous_replan_obligation.py +++ b/loopx/control_plane/work_items/autonomous_replan_obligation.py @@ -14,7 +14,7 @@ normalize_todo_replan_obligation_id, ) from ..todos.resume_planning import project_todo_resume_planning -from .progress_observation import typed_progress_repeat_trigger +from .progress_observation import replan_writeback_requirements, typed_progress_repeat_trigger from .replan_settlement import ( project_todo_lifecycle_settlement_reentry as project_todo_lifecycle_reentry_effect, ) @@ -149,26 +149,9 @@ def build_autonomous_replan_cli_actions( ] raw_obligation = payload.get("autonomous_replan_obligation") obligation: Mapping[str, Any] = ( - raw_obligation if isinstance(raw_obligation, Mapping) else {} - ) - vision_successor_required = any( - isinstance(trigger, Mapping) - and trigger.get("kind") == "vision_successor_required" - for trigger in obligation.get("triggers") or [] - ) - semantic_delta_args = ( - "--agent-vision-json " - "''" - if vision_successor_required - else ( - "--progress-result-class " - " " - "--progress-surface-id " - "--progress-hypothesis-id " - "--progress-probe-kind " - "--progress-evidence-id " - ) + raw_obligation if isinstance(raw_obligation, Mapping) else payload ) + semantic_delta_args = replan_writeback_requirements(obligation)["cli_semantic_args"] delivery_args = ( "--delivery-batch-scale single_surface " "--delivery-outcome outcome_progress " diff --git a/loopx/control_plane/work_items/progress_observation.py b/loopx/control_plane/work_items/progress_observation.py index 9e781463ca..3afd4d115d 100644 --- a/loopx/control_plane/work_items/progress_observation.py +++ b/loopx/control_plane/work_items/progress_observation.py @@ -7,7 +7,7 @@ from typing import Any from ...turn_identity import normalize_turn_instance_id -from ..effect_runtime import effect_runtime_result +from ..effect_runtime import EffectRuntimeRejected, effect_runtime_result from ..goals.goal_vision_state import normalize_goal_vision_state from ..todos.contract import ( normalize_todo_task_domain, @@ -37,31 +37,7 @@ ProgressResultClass.EXPLORATION_EXHAUSTED.value, ProgressResultClass.NO_FOLLOWUP.value, } -REPLAN_REQUIRED_OUTCOMES = ( - "new_surface", - "new_hypothesis", - "new_probe_family", - "new_runnable_successor", - "coverage_backed_exploration_exhausted", - "new_concrete_blocker", - "coverage_backed_no_followup", -) -VISION_REPLAN_TRIGGER_KINDS = frozenset( - { - "vision_acceptance_gap", - "vision_checkpoint_missing", - "vision_outcome_checkpoint_required", - "vision_successor_required", - "required_agent_vision_missing", - } -) -VISION_REPLAN_REQUIRED_OUTCOMES = ( - "fresh_vision_path_outcome", - "new_runnable_successor", - "new_concrete_blocker", - "coverage_backed_exploration_exhausted", - "coverage_backed_no_followup", -) +# Legacy ACK readback vocabulary; discharge authority lives in replan_semantics.ts. FRESH_VISION_PATH_DISPOSITIONS = frozenset( {"continue", "no_change", "replan"} ) @@ -350,31 +326,23 @@ def semantic_progress_delta( } -def required_semantic_outcomes( +def replan_writeback_requirements( obligation: Mapping[str, Any], -) -> list[str]: - """Return exact typed outcomes accepted by this obligation source.""" - - declared = [ - str(value or "").strip() - for value in (obligation.get("satisfying_semantic_outcomes") or []) - if str(value or "").strip() - ] - known = set(REPLAN_REQUIRED_OUTCOMES) | set(VISION_REPLAN_REQUIRED_OUTCOMES) - if declared: - if any(value not in known for value in declared): - raise ValueError( - "satisfying_semantic_outcomes contains an unknown typed outcome" - ) - return list(dict.fromkeys(declared)) - trigger_kinds = { - str(trigger.get("kind") or "").strip() - for trigger in (obligation.get("triggers") or []) - if isinstance(trigger, Mapping) - } - if trigger_kinds & VISION_REPLAN_TRIGGER_KINDS: - return list(VISION_REPLAN_REQUIRED_OUTCOMES) - return list(REPLAN_REQUIRED_OUTCOMES) +) -> dict[str, Any]: + """Adapt the shared typed discharge policy for CLI and host projection.""" + try: + result = effect_runtime_result("work_item.replan_semantics.project", { + "operation": "requirements", "obligation": dict(obligation), + }) + except EffectRuntimeRejected as exc: + raise ValueError(str(exc)) from None + if not isinstance(result, Mapping): + raise RuntimeError("TypeScript replan requirements must be an object") + return dict(result) + + +def required_semantic_outcomes(obligation: Mapping[str, Any]) -> list[str]: + return list(replan_writeback_requirements(obligation)["required_any_of"]) def replan_obligation_trigger_kinds( @@ -437,77 +405,21 @@ def semantic_delta_from_writeback( progress_observation, baseline=baseline if isinstance(baseline, Mapping) else None, ) - outcomes = list(observation_delta.get("delta_kinds") or []) - - no_followup_consistency_error: str | None = None - if "coverage_backed_no_followup" in outcomes: - vision_state = "" - path_outcome = "" - if isinstance(agent_vision, Mapping): - vision_state = normalize_goal_vision_state(agent_vision.get("state")) - raw_path_delta = agent_vision.get("path_delta") - path_delta: Mapping[str, Any] = ( - raw_path_delta if isinstance(raw_path_delta, Mapping) else {} - ) - path_outcome = str(path_delta.get("outcome") or "").strip() - if vision_state != "no_followup" or path_outcome != "stop": - outcomes.remove("coverage_backed_no_followup") - no_followup_consistency_error = ( - "coverage-backed no-follow-up requires agent_vision.state=" - "no_followup and path_delta.outcome=stop" - ) - - if isinstance(agent_vision, Mapping): - raw_patch = agent_vision.get("vision_patch") - patch: Mapping[str, Any] = ( - raw_patch if isinstance(raw_patch, Mapping) else {} - ) - raw_path_delta = agent_vision.get("path_delta") - path_delta = ( - raw_path_delta if isinstance(raw_path_delta, Mapping) else {} - ) - path_outcome = str(path_delta.get("outcome") or "").strip() - evidence_refs = [ - str(value or "").strip() - for value in (path_delta.get("evidence_refs") or []) - if str(value or "").strip() - ] - if ( - str(patch.get("acceptance_summary") or "").strip() - and path_outcome in FRESH_VISION_PATH_DISPOSITIONS - and evidence_refs - and "fresh_vision_path_outcome" not in outcomes - ): - outcomes.append("fresh_vision_path_outcome") - - required = required_semantic_outcomes(obligation) - satisfying = [outcome for outcome in outcomes if outcome in required] - if no_followup_consistency_error: - satisfying = [] + vision = dict(agent_vision) if isinstance(agent_vision, Mapping) else {} + if vision: + vision["state"] = normalize_goal_vision_state(vision.get("state")) + try: + result = effect_runtime_result("work_item.replan_semantics.project", { + "operation": "qualify", "obligation": dict(obligation), + "observation_delta": observation_delta, "agent_vision": vision, + }) + except EffectRuntimeRejected as exc: + raise ValueError(str(exc)) from None return { - "schema_version": "replan_semantic_delta_v0", - "accepted": bool(satisfying), - "outcomes": outcomes, - "satisfying_outcomes": satisfying, - "required_any_of": required, + **result, "trigger_kinds": replan_obligation_trigger_kinds(obligation), "trigger_checkpoints": replan_obligation_trigger_checkpoints(obligation), "obligation_id": obligation.get("obligation_id"), - "observation_fingerprint": observation_delta.get( - "observation_fingerprint" - ), - "reason": ( - "writeback changes an outcome accepted by this obligation source" - if satisfying - else no_followup_consistency_error - if no_followup_consistency_error - else "writeback does not satisfy this obligation's typed outcomes" - ), - **( - {"reason_code": "no_followup_vision_path_inconsistent"} - if no_followup_consistency_error - else {} - ), } @@ -642,7 +554,7 @@ def build_replan_action_packet( "explore_result_node_refs" ), ) - writeback_contract: dict[str, Any] = {} + writeback_contract = replan_writeback_requirements(obligation)["writeback_contract"] successor_summary = str( selected_gap_values.get("successor_summary") or "" ).strip()[:240] diff --git a/loopx/control_plane/work_items/replan_semantics.ts b/loopx/control_plane/work_items/replan_semantics.ts new file mode 100644 index 0000000000..eecced5b93 --- /dev/null +++ b/loopx/control_plane/work_items/replan_semantics.ts @@ -0,0 +1,102 @@ +import type { JsonObject } from "../effect_program.ts"; +import { EffectRuntimeRequestError } from "../effect_runtime_errors.ts"; +import { requireJsonObject } from "../runtime_decode.ts"; +import { visionAuthoringContract } from "../goals/vision_checkpoint.ts"; + +const PROGRESS_OUTCOMES = [ + "new_surface", "new_hypothesis", "new_probe_family", "new_runnable_successor", + "coverage_backed_exploration_exhausted", "new_concrete_blocker", "coverage_backed_no_followup", +] as const; +const VISION_OUTCOMES = [ + "fresh_vision_path_outcome", "new_runnable_successor", "new_concrete_blocker", + "coverage_backed_exploration_exhausted", "coverage_backed_no_followup", +] as const; +type SemanticOutcome = typeof PROGRESS_OUTCOMES[number] | typeof VISION_OUTCOMES[number]; +const KNOWN_OUTCOMES: ReadonlySet = new Set([...PROGRESS_OUTCOMES, ...VISION_OUTCOMES]); +const VISION_TRIGGERS = new Set([ + "vision_acceptance_gap", "vision_checkpoint_missing", "vision_outcome_checkpoint_required", + "vision_successor_required", "required_agent_vision_missing", +]); +const FRESH_PATH_DISPOSITIONS = new Set(["continue", "no_change", "replan"]); + +function object(value: unknown): JsonObject { + return value && typeof value === "object" && !Array.isArray(value) ? value as JsonObject : {}; +} +function strings(value: unknown): string[] { + return Array.isArray(value) ? value.map(item => String(item ?? "").trim()).filter(Boolean) : []; +} + +/** One outcome policy for host projection and write-time discharge. */ +export function requiredSemanticOutcomes(obligation: JsonObject): SemanticOutcome[] { + const declared = strings(obligation.satisfying_semantic_outcomes); + if (declared.length) { + if (declared.some(value => !KNOWN_OUTCOMES.has(value))) { + throw new EffectRuntimeRequestError("satisfying_semantic_outcomes contains an unknown typed outcome"); + } + return [...new Set(declared)] as SemanticOutcome[]; + } + const triggers = Array.isArray(obligation.triggers) ? obligation.triggers : []; + return [...(triggers.some(trigger => VISION_TRIGGERS.has(String(object(trigger).kind ?? "").trim())) + ? VISION_OUTCOMES : PROGRESS_OUTCOMES)]; +} + +function writebackProjection(required: SemanticOutcome[]): JsonObject { + // This chooses a usable refresh path, not the only legal exit. Successor and + // terminal alternatives still have their own typed transitions and gates. + if (required.includes("fresh_vision_path_outcome")) { + return { + cli_semantic_args: "--agent-vision-json ''", + writeback_contract: { + preferred_input: "evidence_linked_vision_path", + vision_authoring: visionAuthoringContract(), + required_fields: ["vision_patch.acceptance_summary", "path_delta.outcome", "path_delta.evidence_refs"], + path_outcomes: [...FRESH_PATH_DISPOSITIONS], + rule: "Author the JSON file from observed evidence, then execute the bound refresh and spend. A new progress identifier or unchanged reason alone cannot resolve this vision obligation. Other required_any_of exits remain subject to their typed contracts.", + }, + }; + } + return { + cli_semantic_args: "--progress-result-class --progress-surface-id --progress-hypothesis-id --progress-probe-kind --progress-evidence-id ", + writeback_contract: {}, + }; +} + +export function projectReplanSemantics(value: unknown): JsonObject { + const request = requireJsonObject(value, "work_item.replan_semantics params"); + const obligation = requireJsonObject(request.obligation, "obligation"); + const required = requiredSemanticOutcomes(obligation); + if (request.operation === "requirements") { + return {required_any_of: required, ...writebackProjection(required)}; + } + if (request.operation !== "qualify") { + throw new EffectRuntimeRequestError("replan semantics operation must be requirements or qualify"); + } + // The progress codec computes evidence novelty; this boundary decides which + // outcomes discharge this obligation. Vision has already passed prepare. + const observation = object(request.observation_delta); + const vision = object(request.agent_vision); + const patch = object(vision.vision_patch); + const path = object(vision.path_delta); + let outcomes = strings(observation.delta_kinds); + if (outcomes.some(outcome => !KNOWN_OUTCOMES.has(outcome))) { + throw new EffectRuntimeRequestError("observation_delta contains an unknown typed outcome"); + } + const inconsistentTerminal = outcomes.includes("coverage_backed_no_followup") && + (vision.state !== "no_followup" || path.outcome !== "stop"); + if (inconsistentTerminal) outcomes = outcomes.filter(outcome => outcome !== "coverage_backed_no_followup"); + if (String(patch.acceptance_summary ?? "").trim() && + FRESH_PATH_DISPOSITIONS.has(String(path.outcome ?? "").trim()) && + strings(path.evidence_refs).length && !outcomes.includes("fresh_vision_path_outcome")) { + outcomes.push("fresh_vision_path_outcome"); + } + const satisfying = inconsistentTerminal ? [] : outcomes.filter(outcome => required.includes(outcome as SemanticOutcome)); + return { + schema_version: "replan_semantic_delta_v0", accepted: satisfying.length > 0, + outcomes, satisfying_outcomes: satisfying, required_any_of: required, + observation_fingerprint: observation.observation_fingerprint ?? null, + reason: satisfying.length ? "writeback changes an outcome accepted by this obligation source" + : inconsistentTerminal ? "coverage-backed no-follow-up requires agent_vision.state=no_followup and path_delta.outcome=stop" + : "writeback does not satisfy this obligation's typed outcomes", + ...(inconsistentTerminal ? {reason_code: "no_followup_vision_path_inconsistent"} : {}), + }; +} diff --git a/tests/control_plane/test_replan_writeback_projection.py b/tests/control_plane/test_replan_writeback_projection.py new file mode 100644 index 0000000000..259b6239e4 --- /dev/null +++ b/tests/control_plane/test_replan_writeback_projection.py @@ -0,0 +1,63 @@ +"""The projected input must be capable of satisfying its own obligation.""" +from __future__ import annotations + +import pytest + +from loopx.control_plane.work_items.autonomous_replan_obligation import ( + build_autonomous_replan_cli_actions, +) +from loopx.control_plane.work_items.progress_observation import ( + semantic_delta_from_writeback, +) + + +@pytest.mark.parametrize("trigger", [ + "required_agent_vision_missing", "vision_acceptance_gap", + "vision_checkpoint_missing", "vision_outcome_checkpoint_required", + "vision_successor_required", +]) +def test_vision_obligations_project_the_input_the_validator_accepts(trigger: str) -> None: + obligation = {"obligation_id": "replan-example", "triggers": [{"kind": trigger}]} + novel_probe = { + "result_class": "advanced", "surface_id": "surface-new", + "evidence_ids": ["evidence-new"], + } + # Novel telemetry alone cannot establish an acceptance/path decision. + assert semantic_delta_from_writeback( + obligation=obligation, progress_observation=novel_probe, + )["accepted"] is False + assert semantic_delta_from_writeback( + obligation=obligation, progress_observation=None, + agent_vision={"vision_patch": {"acceptance_summary": "Verified boundary"}, + "path_delta": {"outcome": "continue", "evidence_refs": ["evidence-new"]}}, + )["accepted"] is True + for payload in (obligation, {"autonomous_replan_obligation": obligation}): + actions = build_autonomous_replan_cli_actions( + payload, goal_id="example", settlement_args=" --replan-obligation-id replan-example", + scoped_cli_args=" --agent-id agent-example", quota_spend_action="spend-bound-turn", + settlement_chain_ready=True, + ) + assert "--agent-vision-json" in actions[0] + assert "--progress-surface-id" not in actions[0] + assert actions[1] == "spend-bound-turn" + + +def test_declared_outcomes_not_trigger_name_determine_writeback_input() -> None: + obligation = {"obligation_id": "replan-example", "triggers": [{"kind": "typed_progress_repeat"}], + "satisfying_semantic_outcomes": ["fresh_vision_path_outcome"]} + actions = build_autonomous_replan_cli_actions( + {"autonomous_replan_obligation": obligation}, goal_id="example", + settlement_args="", scoped_cli_args="", quota_spend_action="", + settlement_chain_ready=False, + ) + assert "--agent-vision-json" in actions[0] + + +def test_turn_envelope_preserves_executable_vision_authoring_contract() -> None: + from loopx.control_plane.testing.control_plane_composition_scenarios import _required_vision_replan_source + from loopx.control_plane.quota.turn_envelope import build_turn_envelope + source = _required_vision_replan_source(goal_id="example", agent_id="agent-example") + envelope = build_turn_envelope(source) + writeback = source["replan_action_packet"]["writeback_contract"] + assert "path_delta.evidence_refs" in writeback["required_fields"] + assert envelope["replan_action_packet"]["writeback_contract"] == writeback diff --git a/tests/control_plane_ts/replan_semantics.test.ts b/tests/control_plane_ts/replan_semantics.test.ts new file mode 100644 index 0000000000..a380a10ec5 --- /dev/null +++ b/tests/control_plane_ts/replan_semantics.test.ts @@ -0,0 +1,46 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { projectReplanSemantics, requiredSemanticOutcomes } from "../../loopx/control_plane/work_items/replan_semantics.ts"; +import { visionAuthoringContract } from "../../loopx/control_plane/goals/vision_checkpoint.ts"; + +const obligation = {triggers: [{kind: "required_agent_vision_missing"}]}; +const vision = {vision_patch: {acceptance_summary: "Observed permission boundary"}, + path_delta: {outcome: "continue", evidence_refs: ["evidence-permission"]}}; + +test("obligation source governs both authoring projection and semantic discharge", () => { + const projection = projectReplanSemantics({operation: "requirements", obligation}); + assert.match(String(projection.cli_semantic_args), /--agent-vision-json/); + assert.deepEqual((projection.writeback_contract as Record).vision_authoring, visionAuthoringContract()); + assert.equal(projectReplanSemantics({operation: "qualify", obligation, + observation_delta: {delta_kinds: ["new_surface"]}}).accepted, false); + const delta = projectReplanSemantics({operation: "qualify", obligation, agent_vision: vision}); + assert.equal(delta.accepted, true); + assert.deepEqual(delta.required_any_of, projection.required_any_of); + assert.deepEqual(delta.satisfying_outcomes, ["fresh_vision_path_outcome"]); +}); + +test("missing evidence or acceptance cannot manufacture a fresh path outcome", () => { + for (const agentVision of [ + {...vision, vision_patch: {}}, {...vision, path_delta: {outcome: "continue"}}, + {...vision, path_delta: {outcome: "wait", evidence_refs: ["evidence-permission"]}}, + ]) { + assert.equal(projectReplanSemantics({operation: "qualify", obligation, agent_vision: agentVision}).accepted, false); + } +}); + +test("explicit outcome restriction remains authoritative; trigger prose is not", () => { + assert.deepEqual(requiredSemanticOutcomes({satisfying_semantic_outcomes: ["new_runnable_successor", "new_runnable_successor"]}), ["new_runnable_successor"]); + assert.throws(() => requiredSemanticOutcomes({satisfying_semantic_outcomes: ["unrecognized"]}), /unknown typed outcome/); + const prose = {triggers: [{kind: "typed_progress_repeat", text: "required_agent_vision_missing"}]}; + assert.equal(requiredSemanticOutcomes(prose).includes("new_surface"), true); + assert.equal(projectReplanSemantics({operation: "qualify", obligation: prose, + observation_delta: {delta_kinds: ["new_surface"]}}).accepted, true); +}); + +test("no-followup cannot hide an inconsistent vision behind another accepted outcome", () => { + const request = {operation: "qualify", obligation, + observation_delta: {delta_kinds: ["coverage_backed_no_followup", "new_concrete_blocker"]}}; + assert.equal(projectReplanSemantics({...request, agent_vision: vision}).reason_code, "no_followup_vision_path_inconsistent"); + const valid = projectReplanSemantics({...request, agent_vision: {state: "no_followup", path_delta: {outcome: "stop"}}}); + assert.equal(valid.accepted, true); +}); From 91e00464f582b78fa96c4dd18cc8f1f9785ab6fb Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 15:39:52 +0800 Subject: [PATCH 02/22] fix(qualification): exercise required vision through bound Turn closeout Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- ...actual_default_model_behavior_portfolio.py | 14 ++ .../replan_semantic_action_behavior.py | 71 ++++++++- .../replan_vision_closeout_behavior.py | 129 ++++++++++++++++ scripts/qualify-doubao-model-behavior-live.py | 1 + ...lify-doubao-replan-semantic-action-live.py | 2 + ...actual_default_model_behavior_portfolio.py | 54 ++++--- .../test_required_vision_closeout_behavior.py | 138 ++++++++++++++++++ 7 files changed, 381 insertions(+), 28 deletions(-) create mode 100644 loopx/control_plane/testing/replan_vision_closeout_behavior.py create mode 100644 tests/control_plane/test_required_vision_closeout_behavior.py diff --git a/loopx/control_plane/testing/actual_default_model_behavior_portfolio.py b/loopx/control_plane/testing/actual_default_model_behavior_portfolio.py index de0924cec8..faab0f4ae9 100644 --- a/loopx/control_plane/testing/actual_default_model_behavior_portfolio.py +++ b/loopx/control_plane/testing/actual_default_model_behavior_portfolio.py @@ -1111,6 +1111,15 @@ def _scenario_contract( _validate_identity_scenario_contract(spec, source_packet, contract) _validate_planning_context_scenario(spec, source_packet, contract) _validate_control_plane_composition_scenario(spec, source_packet, contract) + if spec.scenario_id == "turn_required_vision_replan": + obligation = source_packet["autonomous_replan_obligation"] + contract.update( + qualification_scope="required_vision_closeout", + trigger_kinds=sorted({item["kind"] for item in obligation["triggers"]}), + required_semantic_outcomes=list(source_packet["replan_action_packet"]["uncovered_frontier"]["required_any_of"]), + vision_closeout={"checkpoint_satisfied": True, "bound_writeback": True, + "settled": True, "spend_count": 1, "original_obligation_closed": True}, + ) _validate_compaction_scenario(spec, source_packet, actor_packet, contract) if spec.scenario_family == "diagnostic_authority_boundary": diagnostic = dict(actor_packet.get("agent_todo_summary") or {}) @@ -1183,6 +1192,11 @@ def _receipt_alignment( if receipt.get(field) != expected[field] ] mismatches.extend(str(item) for item in receipt.get("safety_violations") or []) + if spec.scenario_id == "turn_required_vision_replan" and not ( + receipt.get("semantic_action_accepted") is True + and set(receipt.get("selected_semantic_outcomes") or []).intersection(expected["required_semantic_outcomes"]) + ): + mismatches.append("required_vision_outcome_not_accepted") if ( spec.semantic_contract_fields and receipt.get("semantic_contract_complete") is not True diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index 0f08ffc89c..b1ff5035ac 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -32,6 +32,7 @@ loopx_command_tokens, ) from .selected_todo_tool_behavior import bounded_workspace_read_plan +from .replan_vision_closeout_behavior import dispatch_vision_closeout, VISION_HOST_INSTRUCTION REPLAN_SEMANTIC_ACTION_BEHAVIOR_RECEIPT_SCHEMA_VERSION = ( "replan_semantic_action_behavior_receipt_v0" @@ -62,6 +63,7 @@ class _ReplanSemanticActionFixture: work_source_target: Path composition_experiment_ref: str | None = None frontier_exhausted: bool = False + required_vision: bool = False @dataclass @@ -82,6 +84,7 @@ class _QualificationState: created_successor_id: str | None = None successor_reentry_observation: dict[str, Any] | None = None semantic_reentry_observation: dict[str, Any] | None = None + vision_closeout: dict[str, Any] | None = None def _digest(value: str) -> str: @@ -93,6 +96,7 @@ def _build_fixture( *, composition_frontier: bool = False, exhausted_frontier: bool = False, + required_vision: bool = False, ) -> _ReplanSemanticActionFixture: source_root = Path(__file__).resolve().parents[3] project_root = root / "project" @@ -189,6 +193,12 @@ def _build_fixture( "target_key=replan-semantic-fixture " "cadence=1d next_due_at=2999-01-01T00%3A00%3A00Z -->\n" ) + if required_vision: + fixture_agent_todo += ( + "- [ ] [P0] Continue the peer-owned implementation.\n" + " \n" + ) state_path.write_text( "---\n" "status: active\n" @@ -216,7 +226,7 @@ def _build_fixture( "status": "connected-delivery", }, "coordination": { - "registered_agents": [_FIXTURE_AGENT_ID], + "registered_agents": [_FIXTURE_AGENT_ID, "codex-fixture-peer"] if required_vision else [_FIXTURE_AGENT_ID], "agent_model": "peer_v1", "agent_profiles": { _FIXTURE_AGENT_ID: { @@ -224,8 +234,11 @@ def _build_fixture( "agent_id": _FIXTURE_AGENT_ID, "profile_role": "quality-qualification", "scope_summary": "Qualify one bounded semantic replan.", - "default_task_classes": ["continuous_monitor"], - "vision_requirement": "optional", + "default_task_classes": ( + ["advancement_task", "continuous_monitor"] + if required_vision else ["continuous_monitor"] + ), + "vision_requirement": "required" if required_vision else "optional", } }, }, @@ -254,7 +267,7 @@ def _build_fixture( runs_dir = runtime_root / "goals" / _FIXTURE_GOAL_ID / "runs" runs_dir.mkdir(parents=True, exist_ok=True) prior_runs: list[dict[str, Any]] = [] - for minute in (2, 1): + for minute in (() if required_vision else (2, 1)): generated_at = f"2026-08-13T00:0{minute}:00+00:00" run = { "generated_at": generated_at, @@ -404,6 +417,7 @@ def _build_fixture( _FIXTURE_COMPOSITION_EXPERIMENT_ID if composition_frontier else None ), frontier_exhausted=exhausted_frontier, + required_vision=required_vision, ) @@ -542,6 +556,7 @@ def _quota_behavior_observation(packet: Mapping[str, Any]) -> dict[str, Any]: ) or [] ), + "trigger_kinds": sorted({str(trigger["kind"]) for trigger in obligation.get("triggers", [])}), } @@ -649,8 +664,17 @@ def _execute_loopx( fixture: _ReplanSemanticActionFixture, turn_instance_id: str, ) -> str: + tokens = loopx_command_tokens(command) + if tokens is None: + raise ValueError("fixture requires one bounded LoopX command") + # Scope adapter context even when a projected command relies on cwd defaults. + # Never supply missing semantic/binding fields on the actor's behalf. + if "--registry" not in tokens: + tokens[1:1] = ["--registry", str(fixture.global_registry_path)] + if "refresh-state" in tokens: + tokens.extend(["--no-global-sync", "--suppress-external-sinks"]) return execute_loopx_cli( - command, + shlex.join(tokens), source_root=fixture.source_root, project_root=fixture.project_root, argument_overrides={ @@ -799,6 +823,9 @@ def _qualification_receipt( passed: bool, failure_code: str | None, ) -> dict[str, Any]: + if passed and state.fixture.required_vision and not (state.vision_closeout or {}).get("settled"): + passed = False + failure_code = "required_vision_closeout_incomplete" receipt = _receipt( qualification_id=qualification_id, actor_ref=actor_ref, @@ -814,6 +841,12 @@ def _qualification_receipt( receipt["successor_reentry"] = state.successor_reentry_observation if state.semantic_reentry_observation is not None: receipt["semantic_reentry"] = state.semantic_reentry_observation + receipt["qualification_scope"] = ( + "required_vision_closeout" if state.fixture.required_vision else "semantic_action" + ) + if state.vision_closeout is not None: + receipt["vision_closeout"] = state.vision_closeout + receipt["semantic_action_accepted"] = bool(state.semantic_delta and state.semantic_delta.get("accepted")) return receipt @@ -845,6 +878,8 @@ def _handle_quota_command(command: str, state: _QualificationState) -> str: raise ValueError("quota result must be an object") state.quota_packet = decoded state.quota_observation = _quota_behavior_observation(decoded) + if state.fixture.required_vision and "required_agent_vision_missing" not in state.quota_observation["trigger_kinds"]: + raise ValueError("required_vision_fixture_trigger_mismatch") state.seen_quota = True state.read_only_host_commands_executed = True return output @@ -1061,6 +1096,10 @@ def _dispatch_behavior_command( "refresh_state_help", False, ) + if state.fixture.required_vision: + result = dispatch_vision_closeout(command, state, execute=_execute_loopx) + if result is not None: + return result if _is_replan_successor_create(command): return _handle_successor_command(command, state), "replan_successor_create", True if (observation := _progress_observation_from_command(command)) is not None: @@ -1091,6 +1130,10 @@ def _behavior_command_kind( if _is_replan_successor_create(command): return "replan_successor_create" tokens = loopx_command_tokens(command) or [] + if command.startswith("apply_patch"): + return "vision_file_authoring" + if "spend-slot" in tokens: + return "quota_spend_slot" if "refresh-state" in tokens: return "semantic_replan_writeback" return "unexpected_command" @@ -1098,6 +1141,20 @@ def _behavior_command_kind( _EXPECTED_BEHAVIOR_FAILURES = frozenset( { + "required_vision_fixture_trigger_mismatch", + "vision_authoring_requires_single_add_file_patch", + "vision_authoring_path_outside_fixture", + "vision_authoring_requires_added_lines", + "vision_authoring_requires_json_object", + "vision_authoring_before_quota", + "vision_closeout_binding_mismatch", + "vision_closeout_turn_mismatch", + "vision_closeout_requires_observed_source_and_authored_decision", + "vision_closeout_evidence_not_observed", + "vision_closeout_durable_writeback_incomplete", + "vision_closeout_spend_before_writeback", + "vision_closeout_settlement_readback_failed", + "vision_closeout_rearmed_after_settlement", "repeated_quota_should_run", "repeated_clock", "semantic_action_before_quota", @@ -1259,16 +1316,18 @@ def qualify( fixture_root: Path, composition_frontier: bool = False, exhausted_frontier: bool = False, + required_vision: bool = False, ) -> dict[str, Any]: fixture = _build_fixture( fixture_root, composition_frontier=composition_frontier, exhausted_frontier=exhausted_frontier, + required_vision=required_vision, ) messages: list[dict[str, Any]] = [ { "role": "system", - "content": ( + "content": VISION_HOST_INSTRUCTION if required_vision else ( "You are Codex running one LoopX heartbeat against a hermetic " "project. Follow the heartbeat task and use the shell tool. " "The shell working directory is the connected goal project " diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py new file mode 100644 index 0000000000..81b8e51539 --- /dev/null +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -0,0 +1,129 @@ +"""Hermetic file authoring and real-CLI closeout for required-vision qualification. + +The executor does not author a decision or repair the actor's command. It only +confines file writes and observes the actual durable result. This is distinct +from the narrower progress-delta actor's acceptance boundary. +""" +from __future__ import annotations + +import json +import re +from pathlib import Path +from typing import TYPE_CHECKING, Any, Callable + +from .model_tool_behavior import argument_value, loopx_command_tokens +from ..quota.effective_action import EffectiveAction + +if TYPE_CHECKING: + from .replan_semantic_action_behavior import _QualificationState + +VISION_HOST_INSTRUCTION = ( + "You are Codex operating a hermetic LoopX project. Follow the heartbeat and " + "the live control-plane packet, including its semantic writeback and settlement. " + "The shell cwd is the connected project. Available operations are bounded " + "workspace reads, LoopX CLI commands, and apply_patch with a single Add File " + "patch in a quoted heredoc to author a new relative JSON file. No external " + "network or writes outside the fixture are available. Choose the decision " + "from the evidence; file creation alone is not delivery." +) + + +def _authored_file(command: str, project: Path) -> str: + # Accept data, never execute the model's shell program. JSON quotes and + # shell metacharacters inside added lines remain literal file content. + match = re.fullmatch( + r"apply_patch\s+<<\s*'([A-Za-z_][A-Za-z0-9_]*)'\n" + r"\*\*\* Begin Patch\n\*\*\* Add File: ([^\n]+)\n" + r"(.*?)\n\*\*\* End Patch\n\1", command.strip(), re.DOTALL, + ) + if not match: + raise ValueError("vision_authoring_requires_single_add_file_patch") + relative = Path(match[2]) + if relative.is_absolute() or relative.suffix != ".json" or ".." in relative.parts: + raise ValueError("vision_authoring_path_outside_fixture") + target = (project / relative).resolve() + if not target.is_relative_to(project.resolve()) or target.exists(): + raise ValueError("vision_authoring_path_outside_fixture") + lines = match[3].splitlines() + if not lines or any(not line.startswith("+") for line in lines): + raise ValueError("vision_authoring_requires_added_lines") + content = "\n".join(line[1:] for line in lines) + "\n" + if not isinstance(json.loads(content), dict): + raise ValueError("vision_authoring_requires_json_object") + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(content, encoding="utf-8") + return json.dumps({"ok": True, "path": relative.as_posix()}) + + +def _rows(state: _QualificationState) -> list[dict[str, Any]]: + goal_id = str((state.quota_packet or {})["goal_id"]) + index = state.fixture.runtime_root / "goals" / goal_id / "runs" / "index.jsonl" + return [json.loads(line) for line in index.read_text(encoding="utf-8").splitlines() if line] + + +def dispatch_vision_closeout( + command: str, state: _QualificationState, *, execute: Callable[..., str], +) -> tuple[str, str, bool] | None: + if command.startswith("apply_patch"): + if not state.seen_quota: + raise ValueError("vision_authoring_before_quota") + return _authored_file(command, state.fixture.project_root), "vision_file_authoring", False + tokens = loopx_command_tokens(command) or [] + if "refresh-state" not in tokens and "spend-slot" not in tokens: + return None + packet = state.quota_packet or {} + binding = dict(dict(dict(packet.get("interaction_contract") or {}).get("cli_channel") or {}).get("replan_settlement_contract") or {}).get("settlement_binding") or {} + if not binding or argument_value(tokens, binding["cli_argument"]) != binding["id"]: + raise ValueError("vision_closeout_binding_mismatch") + if argument_value(tokens, "--turn-instance-id") != state.turn_instance_id: + raise ValueError("vision_closeout_turn_mismatch") + if "refresh-state" in tokens: + path = argument_value(tokens, "--agent-vision-json") + if not path or not state.work_source_read: + raise ValueError("vision_closeout_requires_observed_source_and_authored_decision") + target = (state.fixture.project_root / path).resolve() + if not target.is_relative_to(state.fixture.project_root.resolve()): + raise ValueError("vision_authoring_path_outside_fixture") + vision = json.loads(target.read_text(encoding="utf-8")) + source_evidence = json.loads(state.fixture.frontier_target.read_text(encoding="utf-8"))["uncovered"] + observed_ids = {item["evidence_id"] for item in source_evidence} + path_delta = vision.get("path_delta") + refs = path_delta.get("evidence_refs") if isinstance(path_delta, dict) else None + if not isinstance(refs, list) or any(not isinstance(ref, str) for ref in refs): + raise ValueError("vision_closeout_evidence_not_observed") + refs = set(refs) + if not refs.intersection(observed_ids): + raise ValueError("vision_closeout_evidence_not_observed") + output = execute(command, fixture=state.fixture, turn_instance_id=state.turn_instance_id) + row = _rows(state)[-1] + semantic = dict(row.get("autonomous_replan_ack") or {}).get("semantic_delta") or {} + checkpoint = dict(row.get("vision_checkpoint") or {}) + identity = dict(row.get("settlement_identity") or {}) + expected_identity = dict(packet.get("heartbeat_receipt") or {}).get("settlement_identity") + if not (semantic.get("accepted") is True and semantic.get("obligation_id") == binding["id"] + and checkpoint.get("satisfied") is True and identity == expected_identity): + raise ValueError("vision_closeout_durable_writeback_incomplete") + state.semantic_delta = semantic + state.vision_closeout = {"checkpoint_satisfied": True, "bound_writeback": True, "settled": False} + return output, "semantic_replan_writeback", False + if state.vision_closeout is None: + raise ValueError("vision_closeout_spend_before_writeback") + output = execute(command, fixture=state.fixture, turn_instance_id=state.turn_instance_id) + replay = json.loads(execute(state.fixture.quota_guard_command, fixture=state.fixture, + turn_instance_id=state.turn_instance_id)) + following = json.loads(execute(state.fixture.quota_guard_command, fixture=state.fixture, + turn_instance_id=f"{state.turn_instance_id}-readback")) + spends = [row for row in _rows(state) if row.get("classification") == "quota_slot_spent"] + if replay.get("effective_action") != EffectiveAction.HEARTBEAT_SETTLED_SKIP.value or len(spends) != 1: + raise ValueError("vision_closeout_settlement_readback_failed") + remaining = dict(following.get("autonomous_replan_obligation") or {}) + remaining_kinds = {item["kind"] for item in remaining.get("triggers", [])} + if remaining.get("obligation_id") == binding["id"] or remaining_kinds.intersection({"required_agent_vision_missing", "vision_checkpoint_missing"}): + raise ValueError("vision_closeout_rearmed_after_settlement") + # A real successor requirement may follow a newly authored continuing + # vision. Do not force the model to declare as_needed or no_followup just + # to obtain a quiet next wake. + state.vision_closeout.update(settled=True, spend_count=1, original_obligation_closed=True) + state.semantic_reentry_observation = {"effective_action": following.get("effective_action"), + "trigger_kinds": sorted(remaining_kinds)} + return output, "quota_spend_slot", True diff --git a/scripts/qualify-doubao-model-behavior-live.py b/scripts/qualify-doubao-model-behavior-live.py index 4c070f3657..3a5d58bfa1 100755 --- a/scripts/qualify-doubao-model-behavior-live.py +++ b/scripts/qualify-doubao-model-behavior-live.py @@ -114,6 +114,7 @@ def qualify_replan_semantic_action(run_id: str) -> dict[str, object]: return replan_semantic_action_actor.qualify( qualification_id=run_id, fixture_root=temp_root / "replan-semantic-action" / run_digest, + required_vision=True, ) def qualify_scoped_gate_successor(run_id: str) -> dict[str, object]: diff --git a/scripts/qualify-doubao-replan-semantic-action-live.py b/scripts/qualify-doubao-replan-semantic-action-live.py index a3a872bf12..1b99e3e0de 100755 --- a/scripts/qualify-doubao-replan-semantic-action-live.py +++ b/scripts/qualify-doubao-replan-semantic-action-live.py @@ -32,6 +32,7 @@ def _parser() -> argparse.ArgumentParser: parser.add_argument("--qualification-id", required=True) parser.add_argument("--timeout-seconds", type=float, default=90.0) parser.add_argument("--repo-root", type=Path, default=Path.cwd()) + parser.add_argument("--required-vision", action="store_true", help="Qualify the missing-vision journey through bound settlement and readback.") return parser @@ -49,6 +50,7 @@ def main() -> int: result = actor.qualify( qualification_id=args.qualification_id, fixture_root=Path(temp_dir), + required_vision=args.required_vision, ) result["source"] = source print(json.dumps(result, ensure_ascii=False, sort_keys=True, indent=2)) diff --git a/tests/control_plane/test_actual_default_model_behavior_portfolio.py b/tests/control_plane/test_actual_default_model_behavior_portfolio.py index ce4f9eb022..f58b8b2f14 100644 --- a/tests/control_plane/test_actual_default_model_behavior_portfolio.py +++ b/tests/control_plane/test_actual_default_model_behavior_portfolio.py @@ -327,7 +327,13 @@ def _replan_semantic_action_actor(_: str) -> dict[str, Any]: delivery_allowed=True, semantic_action_accepted=True, context_delivery="host_projected", - selected_semantic_outcomes=["new_surface"], + selected_semantic_outcomes=["fresh_vision_path_outcome"], + qualification_scope="required_vision_closeout", + trigger_kinds=["required_agent_vision_missing"], + required_semantic_outcomes=["fresh_vision_path_outcome", "new_runnable_successor", "new_concrete_blocker", + "coverage_backed_exploration_exhausted", "coverage_backed_no_followup"], + vision_closeout={"checkpoint_satisfied": True, "bound_writeback": True, + "settled": True, "spend_count": 1, "original_obligation_closed": True}, ) @@ -342,6 +348,24 @@ def _scoped_gate_successor_actor(_: str) -> dict[str, Any]: ) +@pytest.mark.parametrize("overrides", [ + {"qualification_scope": "semantic_action"}, + {"trigger_kinds": ["typed_progress_repeat"]}, + {"vision_closeout": {"checkpoint_satisfied": True, "settled": False}}, + {"selected_semantic_outcomes": ["new_surface"]}, +]) +def test_required_vision_rejects_narrow_or_incomplete_actor_receipts(overrides: dict[str, Any]) -> None: + from loopx.control_plane.testing.actual_default_model_behavior_portfolio import _SCENARIOS, _receipt_alignment + spec = next(item for item in _SCENARIOS if item.scenario_id == "turn_required_vision_replan") + complete = _replan_semantic_action_actor("test") + expected = {key: complete[key] for key in ( + "qualification_scope", "trigger_kinds", "required_semantic_outcomes", "vision_closeout", + )} + aligned, failures = _receipt_alignment(spec, {**complete, **overrides}, expected) + assert aligned is False + assert failures + + def _capability_monitor_repair_actor(_: str) -> dict[str, Any]: return _passing_tool_receipt( "capability_monitor_repair_tool_behavior_receipt_v1", @@ -427,29 +451,11 @@ def _replan_frontier_read_action( ) -> ScriptedExecToolAction: payload = _latest_tool_payload(request) action = payload["replan_action_packet"] - assert "new_surface" in action["uncovered_frontier"]["required_any_of"] + assert "fresh_vision_path_outcome" in action["uncovered_frontier"]["required_any_of"] assert "replan-frontier.json" in payload["active_state_next_action"] return ScriptedExecToolAction("cat replan-frontier.json") -def _semantic_replan_action() -> ScriptedExecToolAction: - return ScriptedExecToolAction( - "loopx --format json --registry ignored --runtime-root ignored " - "refresh-state --goal-id replan-semantic-action-fixture " - "--agent-id codex-replan-semantic-action --progress-scope agent_lane " - "--classification bounded_replan_progress " - "--recommended-action inspect-the-new-surface " - "--delivery-batch-scale single_surface " - "--delivery-outcome outcome_progress " - "--progress-result-class advanced " - "--progress-surface-id surface-permission-config " - "--progress-hypothesis-id hypothesis-permission-default " - "--progress-probe-kind static-contract-read " - "--progress-evidence-id evidence-permission-config " - "--no-global-sync --suppress-external-sinks" - ) - - def _capability_callsite_action( request: Mapping[str, Any], ) -> ScriptedExecToolAction: @@ -548,14 +554,17 @@ def selected_todo_actor(run_id: str) -> Mapping[str, Any]: ) def replan_semantic_action_actor(run_id: str) -> Mapping[str, Any]: + from tests.control_plane.test_required_vision_closeout_behavior import ( + vision_patch_action, projected_refresh, projected_spend, + ) fixture_root = run_root("replan", run_id) - fixture = _build_replan_fixture(fixture_root / "oracle") + fixture = _build_replan_fixture(fixture_root / "oracle", required_vision=True) transport = ScriptedDoubaoExecTransport( [ ScriptedExecToolAction(fixture.quota_guard_command), _replan_frontier_read_action, ScriptedExecToolAction("cat fixture/permission-config.json"), - _semantic_replan_action(), + vision_patch_action, projected_refresh, projected_spend, ] ) return DoubaoReplanSemanticActionBehaviorActor( @@ -564,6 +573,7 @@ def replan_semantic_action_actor(run_id: str) -> Mapping[str, Any]: ).qualify( qualification_id=run_id, fixture_root=fixture_root / "actor", + required_vision=True, ) def scoped_gate_successor_actor(run_id: str) -> Mapping[str, Any]: diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py new file mode 100644 index 0000000000..4cff30c5bc --- /dev/null +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -0,0 +1,138 @@ +from __future__ import annotations + +import json +import shlex +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +import pytest + +from loopx.control_plane.testing.model_tool_behavior import ( + ScriptedAssistantAction, ScriptedDoubaoExecTransport, ScriptedExecToolAction, +) +from loopx.control_plane.testing.replan_semantic_action_behavior import ( + DoubaoReplanSemanticActionBehaviorActor, _build_fixture, +) +from loopx.control_plane.testing.replan_vision_closeout_behavior import _authored_file + + +def _packet(request: Mapping[str, Any]) -> dict[str, Any]: + for message in request["messages"]: + if message["role"] == "tool": + value = json.loads(message["content"]) + if "replan_action_packet" in value: + return value + raise AssertionError("quota was not observed") + + +def vision_patch_action(_: Mapping[str, Any]) -> ScriptedExecToolAction: + # Authored independently from the implementation under test. The path is + # still open: evidence confirms reader-by-default, not completion of a Goal. + decision = { + "schema_version": "goal_vision_replan_contract_v0", + "state": "vision_patch_proposed", + "vision_patch": { + "vision_summary": "Preserve the explicit write permission boundary.", + "acceptance_summary": "Reader is default; writing requires an explicit grant.", + "advancement_policy": "as_needed", + }, + "path_delta": { + "schema_version": "goal_path_delta_v0", "outcome": "continue", + "prior_assumption": "The permission boundary needed inspection.", + "observed_reality": "The configuration defaults to reader and requires a write grant.", + "retained": ["Explicit write grant"], "changed": ["Evidence-linked vision baseline"], + "evidence_refs": ["evidence-permission-config"], + }, + } + lines = "\n".join("+" + line for line in json.dumps(decision, indent=2).splitlines()) + return ScriptedExecToolAction( + "apply_patch <<'PATCH'\n*** Begin Patch\n*** Add File: decision.json\n" + + lines + "\n*** End Patch\nPATCH" + ) + + +def projected_refresh(request: Mapping[str, Any]) -> ScriptedExecToolAction: + actions = _packet(request)["interaction_contract"]["cli_channel"]["next_cli_actions"] + return ScriptedExecToolAction(actions[0].replace( + "", "decision.json", + )) + + +def projected_spend(request: Mapping[str, Any]) -> ScriptedExecToolAction: + command = _packet(request)["interaction_contract"]["cli_channel"]["next_cli_actions"][1] + return ScriptedExecToolAction(command) + + +@pytest.mark.parametrize("advancement_policy", ["as_needed", "repeat_until_closed"]) +def test_required_vision_uses_real_bound_refresh_spend_and_readback(tmp_path: Path, advancement_policy: str) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + def authored_decision(request: Mapping[str, Any]) -> ScriptedExecToolAction: + return ScriptedExecToolAction(vision_patch_action(request).command.replace('"as_needed"', json.dumps(advancement_policy))) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat replan-frontier.json"), + ScriptedExecToolAction("cat fixture/permission-config.json"), + authored_decision, projected_refresh, projected_spend, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor( + api_key="test-only-placeholder", transport=transport, + ).qualify(qualification_id="required-vision", fixture_root=tmp_path / "actor", required_vision=True) + assert receipt["qualification_passed"] is True, receipt + assert receipt["trigger_kinds"] == ["required_agent_vision_missing"] + assert receipt["selected_semantic_outcomes"] == ["fresh_vision_path_outcome"] + assert receipt["vision_closeout"] == { + "checkpoint_satisfied": True, "bound_writeback": True, "settled": True, + "spend_count": 1, "original_obligation_closed": True, + } + if advancement_policy == "as_needed": + assert receipt["semantic_reentry"]["effective_action"] == "monitor_quiet_skip" + assert "required_agent_vision_missing" not in receipt["semantic_reentry"]["trigger_kinds"] + + +def test_successful_refresh_without_settlement_is_not_qualified(tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat replan-frontier.json"), + ScriptedExecToolAction("cat fixture/permission-config.json"), + vision_patch_action, projected_refresh, ScriptedAssistantAction("Done"), + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor( + api_key="test-only-placeholder", transport=transport, + ).qualify(qualification_id="required-vision-incomplete", fixture_root=tmp_path / "actor", required_vision=True) + assert receipt["qualification_passed"] is False + assert receipt["vision_closeout"]["checkpoint_satisfied"] is True + assert receipt["vision_closeout"]["settled"] is False + + +def test_wrong_turn_is_rejected_not_repaired_by_fixture_adapter(tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + def wrong_turn(request: Mapping[str, Any]) -> ScriptedExecToolAction: + tokens = shlex.split(projected_refresh(request).command) + tokens[tokens.index("--turn-instance-id") + 1] = "another-turn" + return ScriptedExecToolAction(shlex.join(tokens)) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat replan-frontier.json"), + ScriptedExecToolAction("cat fixture/permission-config.json"), + vision_patch_action, wrong_turn, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor( + api_key="test-only-placeholder", transport=transport, + ).qualify(qualification_id="wrong-turn", fixture_root=tmp_path / "actor", required_vision=True) + assert receipt["qualification_passed"] is False + assert receipt["failure_code"] == "vision_closeout_turn_mismatch" + assert "vision_closeout" not in receipt + + +def test_authoring_cannot_escape_or_overwrite_fixture(tmp_path: Path) -> None: + command = vision_patch_action({}).command + for path in ("../outside.json", "/outside.json"): + with pytest.raises(ValueError, match="path_outside_fixture"): + _authored_file(command.replace("decision.json", path), tmp_path) + _authored_file(command, tmp_path) + original = (tmp_path / "decision.json").read_bytes() + with pytest.raises(ValueError, match="path_outside_fixture"): + _authored_file(command, tmp_path) + assert (tmp_path / "decision.json").read_bytes() == original From c7038ebde918b14769484a3cea8ca318ae75d010 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 15:39:52 +0800 Subject: [PATCH 03/22] docs(replan): clarify outcome projection and qualification boundaries Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../typescript-control-plane-migration-v0.md | 11 +++++++++ ...script-control-plane-migration-v0.zh-CN.md | 8 +++++++ docs/development/testing-and-quality.md | 23 ++++++++++++++----- .../goal-vision-replan-contract-v0.md | 15 ++++++++++++ .../references/repair-patterns.md | 1 + 5 files changed, 52 insertions(+), 6 deletions(-) diff --git a/docs/architecture/rfcs/typescript-control-plane-migration-v0.md b/docs/architecture/rfcs/typescript-control-plane-migration-v0.md index 2e836d846a..7edd6356ca 100644 --- a/docs/architecture/rfcs/typescript-control-plane-migration-v0.md +++ b/docs/architecture/rfcs/typescript-control-plane-migration-v0.md @@ -320,6 +320,17 @@ not wait for a PostgreSQL service and never expires receipts at day ten. ### Delivery semantics: correctness before migration +The replan obligation outcome policy now lives in +`work_items/replan_semantics.ts`: required-outcome selection, vision-path and +terminal consistency matching, and the matching refresh input projection share +one owner. Python retains progress normalization/novelty and persistence adapters, +but no longer duplicates obligation-to-outcome matching. This is a bounded rule +convergence, not a settlement-writer or store migration. Existing outcome +characterization precedes the move; the intentional correction is executable +authoring for all vision triggers, with real bound CLI closeout/readback and +negative qualification-scope cases. Checkpoint recovery and in-flight rules +remain in their existing owners; no new capability, provider or setting is added. + The delivery-history boundary now treats `classification`, `health_check`, and `recommended_action` as narrative. They cannot create or discharge a follow-through obligation, prove an outcome, or classify delivery scale. diff --git a/docs/architecture/rfcs/typescript-control-plane-migration-v0.zh-CN.md b/docs/architecture/rfcs/typescript-control-plane-migration-v0.zh-CN.md index 11183dec57..9a7aa0e5a2 100644 --- a/docs/architecture/rfcs/typescript-control-plane-migration-v0.zh-CN.md +++ b/docs/architecture/rfcs/typescript-control-plane-migration-v0.zh-CN.md @@ -257,6 +257,14 @@ receipt 过期。 ### 交付语义:先修正规则,再迁移 +Replan 的义务结果规则现收敛到 `work_items/replan_semantics.ts`:接受结果选择、 +vision path/terminal 一致性校验与对应 refresh 输入投影共用同一 owner。 +Python 保留 progress 归一化/新颖性与持久化适配,不再重复义务匹配规则。 +这是有界规则收敛,不是 settlement writer 或存储迁移。先刻画既有接受语义, +再修正所有 vision trigger 的可执行写入投影,并验证真实绑定 CLI 闭环、回读及 +资格范围错配反例。Checkpoint 恢复与 in-flight 规则仍由既有边界负责, +不新增 capability、provider 或设置。 + 交付历史边界将 `classification`、`health_check` 与 `recommended_action` 视为 叙述文本。它们不能生成或解除 follow-through obligation,不能证明 outcome,也 不能判定交付规模。例如,`unblocked after dependency update` 不构成 blocker diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index 530d671f98..e146fc9832 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -745,8 +745,8 @@ action without calling the tool, or issuing an unallowlisted command fails. replan semantic action 另有一条 function-tool 行为资格门,因为 no-tool JSON 决策不能 证明模型会使用覆盖账本选择新方向并完成真实写回。资格门创建一个包含两个等价 typed -progress observation 的隔离、public-safe 临时 Goal;真实模型只看到正式 thin Codex App -heartbeat task body 和普通 `exec_command` tool。真实 quota 必须投影 host coverage context +progress observation 的隔离、public-safe 临时 Goal;模型接收正式 thin Codex App +heartbeat task body、受限执行环境说明和 `exec_command` tool。真实 quota 必须投影 host coverage context 与最小 action packet,模型随后提交的 typed semantic delta 还要通过独立语义判定和真实 写时闸门。若模型选择新 successor,资格门要求它以当前 `obligation_id` 调用真实 `todo add`,验证 Todo 原子 receipt 与 `host_action=end_current_heartbeat`,且不得在同一 @@ -787,13 +787,24 @@ python3 scripts/qualify-doubao-capability-monitor-repair-tool-live.py \ ``` The regular live suite is -`actual_default_model_behavior_portfolio_v0`: nineteen one-arm scenarios and two +`actual_default_model_behavior_portfolio_v0`: twenty-one one-arm scenarios and two attempts each. Its selected-Todo case starts from a production thin heartbeat, executes real quota, and requires the model to perform the selected Todo's read-only target action. Its required-vision replan case independently builds a -hermetic missing-vision state, executes real quota, and requires the model to -use host-projected frontier/work-source context and submit a typed semantic -action through the real write path. The other turn cases remain +hermetic required-profile/missing-vision state with a future monitor and +peer-owned work. The actor must author an evidence-linked vision, execute the +projected bound refresh and spend, and pass durable checkpoint, one-spend and +next-Turn readback. Readback must clear the original missing-baseline obligation; +a legitimate new successor requirement is allowed. Source alignment includes trigger kinds, accepted outcomes +and qualification scope; a successful ordinary `typed_progress_repeat` refresh +cannot qualify this journey. The narrow semantic-action gate remains useful +but does not prove full closeout. Run this focused journey with +`uv run --extra test python scripts/qualify-doubao-replan-semantic-action-live.py --required-vision --qualification-id `. +Its seven-call bound is unchanged. The hermetic host supports bounded reads, +real LoopX CLI calls and single-file `apply_patch` JSON authoring; it does not +execute arbitrary model shell programs or add missing semantic fields. +Receipts distinguish accepted writeback from settled closeout, retain bounded +failure-stage codes, and never persist raw model conversations. The other turn cases remain bounded packet-interpretation checks. Nine core-contract scenarios cover onboarding, agent identity and goal selection, selected todo, peer identity routing, same-agent continuation, final human gate, healthy continuation, and diff --git a/docs/reference/protocols/goal-vision-replan-contract-v0.md b/docs/reference/protocols/goal-vision-replan-contract-v0.md index 58bf526ed6..619d242782 100644 --- a/docs/reference/protocols/goal-vision-replan-contract-v0.md +++ b/docs/reference/protocols/goal-vision-replan-contract-v0.md @@ -100,6 +100,21 @@ pathless packet fails instead of recording a partial closure. A matching typed semantic ACK settles the vision-derived duty even while the original acceptance gap remains visible in the source projection. +Quota's replan writeback projection and write-time outcome matching share +`work_items/replan_semantics.ts`. Every vision-derived trigger, including a +missing required baseline, projects evidence-linked JSON authoring through +`replan_action_packet.writeback_contract`; limits come from the existing vision +validator. This is guidance for a valid refresh path, not a new obligation or +the removal of typed successor/blocker/terminal alternatives. A new surface id +alone cannot satisfy a vision obligation. Execute the current settlement binding +exactly once; accepted semantic writeback, satisfied checkpoint, settled Turn +and Goal completion remain separate facts. Existing missing-checkpoint recovery +stays on the original Turn, and in-flight continuation remains unchanged. + +投影与写入校验共用 TS 语义规则;required-vision 不再投影只有普通进度标识的模板。 +JSON 写作契约复用 vision 校验器,不新增 ACK 仪式,也不改变既有 successor、blocker、 +terminal 出口。语义接受、checkpoint 满足、Turn 结算与 Goal 完成仍须分别验证。 + Inline vision writes require `--agent-id`. JSON packets must also resolve to the same `agent_id` as the refresh run. This keeps `research-executor`, `evaluator-promoter`, and other roles from overwriting or satisfying each diff --git a/skills/loopx-self-repair/references/repair-patterns.md b/skills/loopx-self-repair/references/repair-patterns.md index 225f37af76..bbeb80e44e 100644 --- a/skills/loopx-self-repair/references/repair-patterns.md +++ b/skills/loopx-self-repair/references/repair-patterns.md @@ -5,6 +5,7 @@ teaches a reusable control-plane lesson. | Pattern | Symptoms | Evidence To Read | Likely Root | Durable Repair | | --- | --- | --- | --- | --- | +| `replan_projection_discharge_mismatch` | A projected refresh cannot satisfy its own vision obligation, or a required-vision qualification executes an ordinary repeated-progress fixture. | Typed trigger kinds, required outcomes, projected input/authoring contract, actual actor fixture, bound durable writeback, checkpoint and post-settlement readback. | Projection and admission encoded separate outcome rules; a successful narrow action was mistaken for a complete journey. | Share obligation matching and executable input projection at the typed owner. Reuse vision authoring limits, preserve original-Turn recovery and legal successor/terminal exits. Match qualification scope and triggers to the executed fixture; distinguish accepted writeback, satisfied checkpoint and one-spend settlement. A legitimate new successor obligation is not a failed closeout. Never fix this by adding ACK rituals, weakening acceptance, raising call budgets or replaying live attempts until green. | | `budget_metric_overfitting` | A budget failure triggers automatic expansion, or mechanical compaction that removes useful semantics or breaks consumers. | Owning limit, matched base/head measurements, consumer/caller contract, original failure and revised validation. | A regression metric became the objective; historical ceilings or green tests replaced semantic judgment. | Follow the [budget decision guide](../../../docs/development/testing-and-quality.md#budget-failure-decisions), compare true redundancy, compatibility cost and justified headroom, and repair the existing contract/tests and review evidence. Preserve hard limits and frozen qualification results. | | `skill_import_recreation` | Duplicate LoopX skills return after successful cleanup; imported entries display a fallback brand casing. | Compare installed files and metadata with source-host skills; correlate file creation times with structured host import receipts. | A later external-host import recreates command facades in another discovered root, omits display metadata, and bypasses installer reconciliation. | Attribute the writer from import receipts without guessing the human initiator; exclude already-installed LoopX skills from later imports and rerun managed reconciliation. Ensure standalone workflow entry installation writes Codex metadata, previews missing metadata repair, and records the full installed tree. Preserve user metadata and exact-host invocation behavior. | | `skill_discovery_split_ownership` | Duplicate skill names, conflicting PR-review routes, or canonical and legacy aliases appear together. | Enumerate discovered roots, resolve directory symlinks, compare skill hashes, managed markers, install receipts, and generated metadata. | Workflow and command installers wrote independently to overlapping host roots; dedupe was optional, omitted the bare entry name, or retired copies without proving a replacement. | Repair the shared installer reconciliation and every active installation path; preserve user changes and rich workflows, retire managed aliases from the Codex picker, test repeated installs and custom profiles, then verify a fresh host catalog. Do not treat deleting one visible duplicate or changing invocation policy as a durable fix. | From 93a081a99daa5cefc244c5121ea7c2d01ac07f5d Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 15:51:27 +0800 Subject: [PATCH 04/22] fix(qualification): support confined JSON authoring and own vision scenario checks Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- docs/development/testing-and-quality.md | 3 +- ...actual_default_model_behavior_portfolio.py | 53 +------- .../replan_semantic_action_behavior.py | 2 + .../replan_vision_closeout_behavior.py | 120 +++++++++++++++++- .../test_required_vision_closeout_behavior.py | 37 +++++- 5 files changed, 150 insertions(+), 65 deletions(-) diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index e146fc9832..73ec1b907c 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -801,7 +801,8 @@ cannot qualify this journey. The narrow semantic-action gate remains useful but does not prove full closeout. Run this focused journey with `uv run --extra test python scripts/qualify-doubao-replan-semantic-action-live.py --required-vision --qualification-id `. Its seven-call bound is unchanged. The hermetic host supports bounded reads, -real LoopX CLI calls and single-file `apply_patch` JSON authoring; it does not +real LoopX CLI calls and confined JSON authoring (`apply_patch` or a quoted +heredoc, optionally followed by bound CLI closeout); it does not execute arbitrary model shell programs or add missing semantic fields. Receipts distinguish accepted writeback from settled closeout, retain bounded failure-stage codes, and never persist raw model conversations. The other turn cases remain diff --git a/loopx/control_plane/testing/actual_default_model_behavior_portfolio.py b/loopx/control_plane/testing/actual_default_model_behavior_portfolio.py index faab0f4ae9..0406c35df6 100644 --- a/loopx/control_plane/testing/actual_default_model_behavior_portfolio.py +++ b/loopx/control_plane/testing/actual_default_model_behavior_portfolio.py @@ -52,6 +52,7 @@ SELECTED_TODO_TOOL_FIXTURE_ACTION_TEXT, SELECTED_TODO_TOOL_FIXTURE_TODO_ID, ) +from .replan_vision_closeout_behavior import required_vision_scenario_contract ACTUAL_DEFAULT_MODEL_BEHAVIOR_PORTFOLIO_SCHEMA_VERSION = ( "actual_default_model_behavior_portfolio_v0" @@ -843,47 +844,6 @@ def _validate_quota_hot_path_compaction_regression( raise ValueError("compaction regression must preserve the selected todo") -def _validate_required_vision_replan_scenario( - source_packet: Mapping[str, Any], - contract: Mapping[str, Any], -) -> None: - semantics = model_behavior_semantic_contract_from_packet( - source_packet, - arm="full_packet", - ) - vision = semantics["vision_continuation"] - trigger_kinds = set(vision.get("trigger_kinds", [])) - required = { - "selected_todo_id": None, - "user_action_required": False, - "must_attempt_work": True, - "quiet_noop_allowed": False, - } - if any(contract.get(field) != value for field, value in required.items()): - raise ValueError("required-vision scenario must execute before quiet wait") - if vision.get("required") is not True or ( - "required_agent_vision_missing" not in trigger_kinds - ): - raise ValueError("required-vision scenario must preserve the profile gap") - if semantics["required_reads"]: - raise ValueError("required-vision replan must not require a model read ritual") - action_packet = source_packet.get("replan_action_packet") - obligation = source_packet.get("autonomous_replan_obligation") - if not ( - isinstance(action_packet, Mapping) - and isinstance(obligation, Mapping) - and action_packet.get("decision") == "replan_required" - and action_packet.get("obligation_id") == obligation.get("obligation_id") - and dict(obligation.get("replan_context") or {}).get("delivery") - == "host_projected" - ): - raise ValueError( - "required-vision scenario must preserve host-delivered replan context" - ) - if semantics["scheduler_action"].get("action") != "run_now": - raise ValueError("required-vision scenario must remain immediately runnable") - - def _validate_planning_horizon_model_scenario( source_packet: Mapping[str, Any], ) -> None: @@ -993,8 +953,6 @@ def _validate_control_plane_composition_scenario( source_packet: Mapping[str, Any], contract: Mapping[str, Any], ) -> None: - if spec.scenario_id == "turn_required_vision_replan": - _validate_required_vision_replan_scenario(source_packet, contract) if spec.scenario_id == "turn_scoped_gate_successor_replan": signature = quota_action_signature_document(source_packet) action = dict(signature.get("action") or {}) @@ -1112,14 +1070,7 @@ def _scenario_contract( _validate_planning_context_scenario(spec, source_packet, contract) _validate_control_plane_composition_scenario(spec, source_packet, contract) if spec.scenario_id == "turn_required_vision_replan": - obligation = source_packet["autonomous_replan_obligation"] - contract.update( - qualification_scope="required_vision_closeout", - trigger_kinds=sorted({item["kind"] for item in obligation["triggers"]}), - required_semantic_outcomes=list(source_packet["replan_action_packet"]["uncovered_frontier"]["required_any_of"]), - vision_closeout={"checkpoint_satisfied": True, "bound_writeback": True, - "settled": True, "spend_count": 1, "original_obligation_closed": True}, - ) + contract.update(required_vision_scenario_contract(source_packet, contract)) _validate_compaction_scenario(spec, source_packet, actor_packet, contract) if spec.scenario_family == "diagnostic_authority_boundary": diagnostic = dict(actor_packet.get("agent_todo_summary") or {}) diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index b1ff5035ac..f7177bf504 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -1146,6 +1146,8 @@ def _behavior_command_kind( "vision_authoring_path_outside_fixture", "vision_authoring_requires_added_lines", "vision_authoring_requires_json_object", + "vision_authoring_requires_literal_json_heredoc", + "vision_authoring_suffix_requires_loopx", "vision_authoring_before_quota", "vision_closeout_binding_mismatch", "vision_closeout_turn_mismatch", diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py index 81b8e51539..f8781702b1 100644 --- a/loopx/control_plane/testing/replan_vision_closeout_behavior.py +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -8,10 +8,13 @@ import json import re +import shlex +from collections.abc import Mapping from pathlib import Path from typing import TYPE_CHECKING, Any, Callable from .model_tool_behavior import argument_value, loopx_command_tokens +from .model_behavior_qualification import model_behavior_semantic_contract_from_packet from ..quota.effective_action import EffectiveAction if TYPE_CHECKING: @@ -21,13 +24,56 @@ "You are Codex operating a hermetic LoopX project. Follow the heartbeat and " "the live control-plane packet, including its semantic writeback and settlement. " "The shell cwd is the connected project. Available operations are bounded " - "workspace reads, LoopX CLI commands, and apply_patch with a single Add File " - "patch in a quoted heredoc to author a new relative JSON file. No external " + "workspace reads, LoopX CLI commands, and new relative JSON file authoring " + "via apply_patch Add File or cat with a quoted heredoc. A cat write may be " + "followed by bound LoopX commands separated by newlines or &&. No external " "network or writes outside the fixture are available. Choose the decision " "from the evidence; file creation alone is not delivery." ) +def required_vision_scenario_contract( + source_packet: Mapping[str, Any], contract: Mapping[str, Any], +) -> dict[str, Any]: + """Validate the source scenario and bind its full closeout acceptance.""" + semantics = model_behavior_semantic_contract_from_packet(source_packet, arm="full_packet") + vision = semantics["vision_continuation"] + trigger_kinds = set(vision.get("trigger_kinds", [])) + required = { + "selected_todo_id": None, + "user_action_required": False, + "must_attempt_work": True, + "quiet_noop_allowed": False, + } + if any(contract.get(field) != value for field, value in required.items()): + raise ValueError("required-vision scenario must execute before quiet wait") + if vision.get("required") is not True or "required_agent_vision_missing" not in trigger_kinds: + raise ValueError("required-vision scenario must preserve the profile gap") + if semantics["required_reads"]: + raise ValueError("required-vision replan must not require a model read ritual") + action_packet = source_packet.get("replan_action_packet") + obligation = source_packet.get("autonomous_replan_obligation") + if not ( + isinstance(action_packet, Mapping) + and isinstance(obligation, Mapping) + and action_packet.get("decision") == "replan_required" + and action_packet.get("obligation_id") == obligation.get("obligation_id") + and dict(obligation.get("replan_context") or {}).get("delivery") == "host_projected" + ): + raise ValueError("required-vision scenario must preserve host-delivered replan context") + if semantics["scheduler_action"].get("action") != "run_now": + raise ValueError("required-vision scenario must remain immediately runnable") + return { + "qualification_scope": "required_vision_closeout", + "trigger_kinds": sorted({item["kind"] for item in obligation["triggers"]}), + "required_semantic_outcomes": list(action_packet["uncovered_frontier"]["required_any_of"]), + "vision_closeout": { + "checkpoint_satisfied": True, "bound_writeback": True, + "settled": True, "spend_count": 1, "original_obligation_closed": True, + }, + } + + def _authored_file(command: str, project: Path) -> str: # Accept data, never execute the model's shell program. JSON quotes and # shell metacharacters inside added lines remain literal file content. @@ -38,16 +84,19 @@ def _authored_file(command: str, project: Path) -> str: ) if not match: raise ValueError("vision_authoring_requires_single_add_file_patch") - relative = Path(match[2]) + lines = match[3].splitlines() + if not lines or any(not line.startswith("+") for line in lines): + raise ValueError("vision_authoring_requires_added_lines") + return _write_json_file(project, match[2], "\n".join(line[1:] for line in lines) + "\n") + + +def _write_json_file(project: Path, name: str, content: str) -> str: + relative = Path(name) if relative.is_absolute() or relative.suffix != ".json" or ".." in relative.parts: raise ValueError("vision_authoring_path_outside_fixture") target = (project / relative).resolve() if not target.is_relative_to(project.resolve()) or target.exists(): raise ValueError("vision_authoring_path_outside_fixture") - lines = match[3].splitlines() - if not lines or any(not line.startswith("+") for line in lines): - raise ValueError("vision_authoring_requires_added_lines") - content = "\n".join(line[1:] for line in lines) + "\n" if not isinstance(json.loads(content), dict): raise ValueError("vision_authoring_requires_json_object") target.parent.mkdir(parents=True, exist_ok=True) @@ -55,6 +104,49 @@ def _authored_file(command: str, project: Path) -> str: return json.dumps({"ok": True, "path": relative.as_posix()}) +def _json_heredoc(command: str) -> tuple[str, str, list[str]] | None: + """Decode inert JSON plus a bounded CLI suffix, never a shell program.""" + header, newline, rest = command.partition("\n") + if not newline or not header.lstrip().startswith("cat "): + return None + quoted = re.search(r"<<\s*(['\"])([A-Za-z_][A-Za-z0-9_]*)\1", header) + if quoted is None: + return None + lexer = shlex.shlex(header, posix=True, punctuation_chars="<>&;|") + lexer.whitespace_split = True + tokens = list(lexer) + if len(tokens) != 5 or tokens[0] != "cat" or set(tokens[1::2]) != {">", "<<"}: + raise ValueError("vision_authoring_requires_literal_json_heredoc") + name = tokens[tokens.index(">") + 1] + delimiter = tokens[tokens.index("<<") + 1] + if delimiter != quoted[2]: + raise ValueError("vision_authoring_requires_literal_json_heredoc") + lines = rest.splitlines(keepends=True) + end = next((index for index, line in enumerate(lines) if line.rstrip("\r\n") == delimiter), None) + if end is None: + raise ValueError("vision_authoring_requires_literal_json_heredoc") + suffix = "".join(lines[end + 1:]).strip() + lexer = shlex.shlex(suffix, posix=True, punctuation_chars=";&|\n") + lexer.whitespace = " \t\r" + lexer.whitespace_split = True + commands: list[str] = [] + argv: list[str] = [] + for token in [*lexer, "\n"]: + if token == "&&" or token.strip("\n") == "": + if argv: + if Path(argv[0]).name != "loopx": + raise ValueError("vision_authoring_suffix_requires_loopx") + commands.append(shlex.join(argv)) + argv = [] + elif token in {";", "|", "||", "&"}: + raise ValueError("vision_authoring_suffix_requires_loopx") + else: + argv.append(token) + if len(commands) > 2: + raise ValueError("vision_authoring_suffix_requires_loopx") + return name, "".join(lines[:end]), commands + + def _rows(state: _QualificationState) -> list[dict[str, Any]]: goal_id = str((state.quota_packet or {})["goal_id"]) index = state.fixture.runtime_root / "goals" / goal_id / "runs" / "index.jsonl" @@ -64,6 +156,20 @@ def _rows(state: _QualificationState) -> list[dict[str, Any]]: def dispatch_vision_closeout( command: str, state: _QualificationState, *, execute: Callable[..., str], ) -> tuple[str, str, bool] | None: + heredoc = _json_heredoc(command) + if heredoc is not None: + if not state.seen_quota: + raise ValueError("vision_authoring_before_quota") + name, content, suffix = heredoc + result: tuple[str, str, bool] = (_write_json_file(state.fixture.project_root, name, content), "vision_file_authoring", False) + for following in suffix: + if result[2]: + raise ValueError("vision_authoring_suffix_requires_loopx") + next_result = dispatch_vision_closeout(following, state, execute=execute) + if next_result is None: + raise ValueError("vision_authoring_suffix_requires_loopx") + result = next_result + return result if command.startswith("apply_patch"): if not state.seen_quota: raise ValueError("vision_authoring_before_quota") diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py index 4cff30c5bc..0a999677d3 100644 --- a/tests/control_plane/test_required_vision_closeout_behavior.py +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -14,7 +14,7 @@ from loopx.control_plane.testing.replan_semantic_action_behavior import ( DoubaoReplanSemanticActionBehaviorActor, _build_fixture, ) -from loopx.control_plane.testing.replan_vision_closeout_behavior import _authored_file +from loopx.control_plane.testing.replan_vision_closeout_behavior import _authored_file, _json_heredoc def _packet(request: Mapping[str, Any]) -> dict[str, Any]: @@ -64,17 +64,31 @@ def projected_spend(request: Mapping[str, Any]) -> ScriptedExecToolAction: return ScriptedExecToolAction(command) +def heredoc_action(request: Mapping[str, Any]) -> ScriptedExecToolAction: + patch_lines = vision_patch_action(request).command.splitlines()[3:-2] + content = "\n".join(line[1:] for line in patch_lines) + return ScriptedExecToolAction("cat > decision.json <<'JSON'\n" + content + "\nJSON") + + @pytest.mark.parametrize("advancement_policy", ["as_needed", "repeat_until_closed"]) -def test_required_vision_uses_real_bound_refresh_spend_and_readback(tmp_path: Path, advancement_policy: str) -> None: +@pytest.mark.parametrize("authoring", ["apply_patch", "heredoc_with_closeout"]) +def test_required_vision_uses_real_bound_refresh_spend_and_readback(tmp_path: Path, advancement_policy: str, authoring: str) -> None: fixture = _build_fixture(tmp_path / "oracle", required_vision=True) def authored_decision(request: Mapping[str, Any]) -> ScriptedExecToolAction: - return ScriptedExecToolAction(vision_patch_action(request).command.replace('"as_needed"', json.dumps(advancement_policy))) - transport = ScriptedDoubaoExecTransport([ + command = (vision_patch_action if authoring == "apply_patch" else heredoc_action)(request).command + command = command.replace('"as_needed"', json.dumps(advancement_policy)) + if authoring == "heredoc_with_closeout": + command += "\n" + projected_refresh(request).command + " && " + projected_spend(request).command + return ScriptedExecToolAction(command) + actions = [ ScriptedExecToolAction(fixture.quota_guard_command), ScriptedExecToolAction("cat replan-frontier.json"), ScriptedExecToolAction("cat fixture/permission-config.json"), - authored_decision, projected_refresh, projected_spend, - ]) + authored_decision, + ] + if authoring == "apply_patch": + actions.extend([projected_refresh, projected_spend]) + transport = ScriptedDoubaoExecTransport(actions) receipt = DoubaoReplanSemanticActionBehaviorActor( api_key="test-only-placeholder", transport=transport, ).qualify(qualification_id="required-vision", fixture_root=tmp_path / "actor", required_vision=True) @@ -136,3 +150,14 @@ def test_authoring_cannot_escape_or_overwrite_fixture(tmp_path: Path) -> None: with pytest.raises(ValueError, match="path_outside_fixture"): _authored_file(command, tmp_path) assert (tmp_path / "decision.json").read_bytes() == original + + +def test_heredoc_is_inert_json_and_never_an_arbitrary_shell_program() -> None: + command = heredoc_action({}).command + name, content, suffix = _json_heredoc(command.replace("Explicit write grant", "$(touch outside); `echo untrusted`")) + assert name == "decision.json" and suffix == [] + assert "$(touch outside)" in json.loads(content)["path_delta"]["retained"][0] + for tail in ("\ntouch outside", "\nloopx quota spend-slot; touch outside", "\nloopx quota spend-slot | echo outside"): + with pytest.raises(ValueError, match="suffix_requires_loopx"): + _json_heredoc(command + tail) + assert _json_heredoc(command.replace("<<'JSON'", "< Date: Sat, 19 Sep 2026 17:30:29 +0800 Subject: [PATCH 05/22] fix(qualification): align bounded read host and recoverable tool errors Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../testing/model_tool_behavior.py | 11 +- .../replan_semantic_action_behavior.py | 67 ++++++++--- .../replan_vision_closeout_behavior.py | 14 +++ .../testing/selected_todo_tool_behavior.py | 50 ++++++-- .../test_required_vision_closeout_behavior.py | 109 +++++++++++++++++- 5 files changed, 219 insertions(+), 32 deletions(-) diff --git a/loopx/control_plane/testing/model_tool_behavior.py b/loopx/control_plane/testing/model_tool_behavior.py index 031d210145..c55fb4a5d8 100644 --- a/loopx/control_plane/testing/model_tool_behavior.py +++ b/loopx/control_plane/testing/model_tool_behavior.py @@ -225,17 +225,24 @@ def actor_ref(self) -> str: def next_tool_call( self, messages: list[dict[str, Any]], + *, + tool_description: str | None = None, ) -> ExecToolCall | None: - return self.next_step(messages).tool_call + return self.next_step(messages, tool_description=tool_description).tool_call def next_step( self, messages: list[dict[str, Any]], + *, + tool_description: str | None = None, ) -> ExecToolStep: body = { "model": self._model, "messages": messages, - "tools": [EXEC_COMMAND_TOOL], + "tools": [EXEC_COMMAND_TOOL if tool_description is None else { + **EXEC_COMMAND_TOOL, + "function": {**EXEC_COMMAND_TOOL["function"], "description": tool_description}, + }], "tool_choice": "auto", "thinking": {"type": "disabled"}, "temperature": 0, diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index f7177bf504..fbd5f0e563 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -32,13 +32,24 @@ loopx_command_tokens, ) from .selected_todo_tool_behavior import bounded_workspace_read_plan -from .replan_vision_closeout_behavior import dispatch_vision_closeout, VISION_HOST_INSTRUCTION +from .replan_vision_closeout_behavior import ( + dispatch_vision_closeout, VISION_HOST_INSTRUCTION, VISION_EXEC_TOOL_DESCRIPTION, +) REPLAN_SEMANTIC_ACTION_BEHAVIOR_RECEIPT_SCHEMA_VERSION = ( "replan_semantic_action_behavior_receipt_v0" ) REPLAN_SEMANTIC_ACTION_BEHAVIOR_MAX_CALLS = 7 + +class _WorkspaceReadFailed(RuntimeError): + """An admitted read's exit status is a tool result, not a semantic verdict.""" + + def __init__(self, output: str, exit_code: int) -> None: + super().__init__("workspace_read_nonzero") + self.output = output + self.exit_code = exit_code + _FIXTURE_GOAL_ID = "replan-semantic-action-fixture" _FIXTURE_AGENT_ID = "codex-replan-semantic-action" _FIXTURE_WORK_ITEM_ID = "todo-replan-semantic-action" @@ -690,7 +701,7 @@ def _bounded_workspace_read_plan( *, fixture: _ReplanSemanticActionFixture, ) -> list[Any] | None: - return bounded_workspace_read_plan(command, fixture=fixture) + return bounded_workspace_read_plan(command, fixture=fixture, allow_loopx_help=fixture.required_vision) def _bounded_refresh_state_help_limit(command: str) -> int | None: @@ -728,7 +739,7 @@ def _execute_workspace_read( command: str, *, fixture: _ReplanSemanticActionFixture, -) -> tuple[str, bool, bool]: +) -> tuple[str, bool, bool, int]: plan = _bounded_workspace_read_plan(command, fixture=fixture) if plan is None: raise ValueError("workspace read is outside the hermetic fixture") @@ -744,16 +755,21 @@ def _execute_workspace_read( ) if not should_run: continue - completed = subprocess.run( - list(step.argv), - cwd=fixture.project_root, - check=False, - capture_output=True, - text=True, encoding="utf-8", errors="replace", - timeout=10, - ) - status = completed.returncode - output = completed.stdout + if step.kind == "loopx_help": + output = execute_loopx_cli( + shlex.join(step.argv), source_root=fixture.source_root, + project_root=fixture.project_root, timeout_seconds=10, + ) + status = 0 + else: + completed = subprocess.run( + list(step.argv), cwd=fixture.project_root, check=False, + capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=10, + ) + status = completed.returncode + output = completed.stdout + if fixture.required_vision and not step.suppress_stderr: + output += completed.stderr if step.line_limit is not None: output = "".join(output.splitlines(keepends=True)[: step.line_limit]) outputs.append(output) @@ -767,9 +783,9 @@ def _execute_workspace_read( and step.target is not None and step.target.resolve() == fixture.work_source_target.resolve() ) - if status != 0: + if status != 0 and not fixture.required_vision: raise RuntimeError(f"workspace read failed with exit={status}") - return "".join(outputs), frontier_read, work_source_read + return "".join(outputs), frontier_read, work_source_read, status def _receipt( @@ -855,12 +871,14 @@ def _record_tool_step( *, kind: str, command: str, + exit_code: int | None = None, ) -> None: state.steps.append( { "ordinal": len(state.steps) + 1, "kind": kind, "command_digest": _digest(command), + **({"exit_code": exit_code, "error_code": "workspace_read_nonzero"} if exit_code else {}), } ) @@ -894,13 +912,17 @@ def _handle_clock_command(output: str, state: _QualificationState) -> str: def _handle_workspace_read(command: str, state: _QualificationState) -> str: - output, read_frontier, read_work_source = _execute_workspace_read( + if state.fixture.required_vision and not state.seen_quota: + raise ValueError("workspace_read_before_quota") + output, read_frontier, read_work_source, exit_code = _execute_workspace_read( command, fixture=state.fixture, ) state.frontier_context_read = state.frontier_context_read or read_frontier state.work_source_read = state.work_source_read or read_work_source state.read_only_host_commands_executed = True + if exit_code: + raise _WorkspaceReadFailed(output, exit_code) return output @@ -1141,6 +1163,7 @@ def _behavior_command_kind( _EXPECTED_BEHAVIOR_FAILURES = frozenset( { + "workspace_read_before_quota", "required_vision_fixture_trigger_mismatch", "vision_authoring_requires_single_add_file_patch", "vision_authoring_path_outside_fixture", @@ -1228,7 +1251,10 @@ def _run_qualification_loop( qualification_id: str, ) -> dict[str, Any]: for _ in range(REPLAN_SEMANTIC_ACTION_BEHAVIOR_MAX_CALLS): - tool_call = client.next_tool_call(state.messages) + tool_call = client.next_tool_call( + state.messages, + tool_description=VISION_EXEC_TOOL_DESCRIPTION if state.fixture.required_vision else None, + ) if tool_call is None: return _qualification_receipt( state, @@ -1244,6 +1270,13 @@ def _run_qualification_loop( state, ) kind = dispatched_kind + except _WorkspaceReadFailed as exc: + _record_tool_step(state, kind=kind, command=tool_call.command, exit_code=exc.exit_code) + _append_tool_response(state, tool_call=tool_call, output=json.dumps({ + "error_code": "workspace_read_nonzero", "exit_code": exc.exit_code, + "output": exc.output[:4096], "truncated": len(exc.output) > 4096, + })) + continue except (RuntimeError, ValueError, json.JSONDecodeError) as exc: _record_tool_step(state, kind=kind, command=tool_call.command) return _qualification_receipt( diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py index f8781702b1..e4588e35f7 100644 --- a/loopx/control_plane/testing/replan_vision_closeout_behavior.py +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -31,6 +31,20 @@ "from the evidence; file creation alone is not delivery." ) +VISION_EXEC_TOOL_DESCRIPTION = ( + "Execute bounded fixture commands, not an unrestricted shell. Workspace reads " + "support pwd, ls (-a/-l), find (maxdepth at most 4), rg --files, cat, head and " + "sed -n, optionally joined with &&, || or ; (at most 8 statements). Root " + "loopx --help and loopx refresh-state --help are read-only queries and may " + "be combined with reads or piped to head (at most 200 lines). Author a new " + "relative JSON file via apply_patch Add File or cat with a quoted heredoc; " + "a heredoc may be followed by up to two bound LoopX closeout commands using " + "newlines or &&. Other LoopX commands must be separate single invocations. " + "Returns stdout on success; nonzero workspace reads return exit_code and " + "output for correction within the same call budget. Unsupported programs " + "or syntax are never executed. No external network or outside-fixture writes." +) + def required_vision_scenario_contract( source_packet: Mapping[str, Any], contract: Mapping[str, Any], diff --git a/loopx/control_plane/testing/selected_todo_tool_behavior.py b/loopx/control_plane/testing/selected_todo_tool_behavior.py index 0545425f39..431fe0c957 100644 --- a/loopx/control_plane/testing/selected_todo_tool_behavior.py +++ b/loopx/control_plane/testing/selected_todo_tool_behavior.py @@ -5,7 +5,7 @@ import shlex import subprocess from collections.abc import Mapping -from dataclasses import dataclass +from dataclasses import dataclass, replace from hashlib import sha256 from pathlib import Path from typing import Any @@ -76,6 +76,7 @@ class BoundedWorkspaceReadStep: target: Path | None = None operator: str = ";" line_limit: int | None = None + suppress_stderr: bool = False def _build_fixture( @@ -409,11 +410,22 @@ def _is_fixture_metadata_target( return target.resolve() in allowed +def _loopx_help_argv(tokens: list[str]) -> tuple[str, ...] | None: + if tokens[-3:] == ["2", ">&", "1"]: + tokens = tokens[:-3] + if tokens and Path(tokens[0]).name == "loopx" and tokens[1:] in ( + ["--help"], ["refresh-state", "--help"], + ): + return ("loopx", *tokens[1:]) + return None + + def _metadata_pipeline_step( segment: list[str], *, fixture: _SelectedTodoToolFixture, operator: str, + allow_loopx_help: bool = False, ) -> BoundedWorkspaceReadStep | None: pipe_indices = [index for index, token in enumerate(segment) if token == "|"] if len(pipe_indices) not in {1, 2}: @@ -448,6 +460,8 @@ def _metadata_pipeline_step( return None if limit < 1 or limit > 200: return None + if allow_loopx_help and len(pipe_indices) == 1 and (help_argv := _loopx_help_argv(left)): + return BoundedWorkspaceReadStep("loopx_help", help_argv, operator=operator, line_limit=limit) discovery_argv = _discovery_tokens(left, fixture=fixture) if discovery_argv is not None: return BoundedWorkspaceReadStep( @@ -480,7 +494,8 @@ def _metadata_pipeline_step( ) -def _read_plan_tokens(command: str) -> list[str] | None: +def _read_plan_segments(command: str) -> list[tuple[str, list[str]]] | None: + """Decode the bounded read grammar before admitting any executable step.""" lexer = shlex.shlex(command, posix=True, punctuation_chars=";&|<>\n") lexer.whitespace = " \t\r" lexer.whitespace_split = True @@ -488,15 +503,6 @@ def _read_plan_tokens(command: str) -> list[str] | None: tokens = list(lexer) except ValueError: return None - return tokens or None - - -def bounded_workspace_read_plan( - command: str, - *, - fixture: _SelectedTodoToolFixture, -) -> list[BoundedWorkspaceReadStep] | None: - tokens = _read_plan_tokens(command) if not tokens: return None segments: list[list[str]] = [[]] @@ -515,15 +521,28 @@ def bounded_workspace_read_plan( segments[-1].append(token) if not segments[-1] or len(segments) > 8: return None + return list(zip(operators, segments, strict=True)) + + +def bounded_workspace_read_plan( + command: str, + *, + fixture: _SelectedTodoToolFixture, + allow_loopx_help: bool = False, +) -> list[BoundedWorkspaceReadStep] | None: + segments = _read_plan_segments(command) + if segments is None: + return None plan: list[BoundedWorkspaceReadStep] = [] - for operator, raw_segment in zip(operators, segments, strict=True): + for operator, raw_segment in segments: segment = list(raw_segment) if "|" in segment: pipeline_step = _metadata_pipeline_step( segment, fixture=fixture, operator=operator, + allow_loopx_help=allow_loopx_help, ) if pipeline_step is None: return None @@ -531,6 +550,9 @@ def bounded_workspace_read_plan( continue if len(segment) >= 3 and segment[-3:] == ["2", ">", "/dev/null"]: segment = segment[:-3] + if allow_loopx_help and (help_argv := _loopx_help_argv(segment)): + plan.append(BoundedWorkspaceReadStep("loopx_help", help_argv, operator=operator, line_limit=200)) + continue if not segment or any( token in {"&", "|", "<", ">", "<<", ">>", "<<<"} for token in segment @@ -626,6 +648,10 @@ def bounded_workspace_read_plan( operator, ) ) + for index, (_, segment) in enumerate(segments): + left = segment[:segment.index("|")] if "|" in segment else segment + if left[-3:] == ["2", ">", "/dev/null"]: + plan[index] = replace(plan[index], suppress_stderr=True) return plan diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py index 0a999677d3..1ed8bef525 100644 --- a/tests/control_plane/test_required_vision_closeout_behavior.py +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -9,10 +9,11 @@ import pytest from loopx.control_plane.testing.model_tool_behavior import ( - ScriptedAssistantAction, ScriptedDoubaoExecTransport, ScriptedExecToolAction, + EXEC_COMMAND_TOOL, ScriptedAssistantAction, ScriptedDoubaoExecTransport, ScriptedExecToolAction, ) from loopx.control_plane.testing.replan_semantic_action_behavior import ( DoubaoReplanSemanticActionBehaviorActor, _build_fixture, + _bounded_workspace_read_plan, _execute_workspace_read, ) from loopx.control_plane.testing.replan_vision_closeout_behavior import _authored_file, _json_heredoc @@ -161,3 +162,109 @@ def test_heredoc_is_inert_json_and_never_an_arbitrary_shell_program() -> None: with pytest.raises(ValueError, match="suffix_requires_loopx"): _json_heredoc(command + tail) assert _json_heredoc(command.replace("<<'JSON'", "< None: + fixture = _build_fixture(tmp_path, required_vision=True) + command = f'ls {shlex.quote(str(fixture.runtime_root))} 2>/dev/null && echo "---" && loopx --help 2>&1 | head -60' + plan = _bounded_workspace_read_plan(command, fixture=fixture) + assert plan is not None + assert plan[-1].kind == "loopx_help" + output, frontier_read, source_read, exit_code = _execute_workspace_read(command, fixture=fixture) + assert "usage:" in output.lower() and "loopx" in output + assert exit_code == 0 and not frontier_read and not source_read + # Adding an unadmitted effect rejects the entire plan before execution. + assert _bounded_workspace_read_plan(command + " && touch injected", fixture=fixture) is None + with pytest.raises(ValueError, match="outside the hermetic fixture"): + _execute_workspace_read(command + " && touch injected", fixture=fixture) + assert not (fixture.project_root / "injected").exists() + + +@pytest.mark.parametrize("command", ["loopx --help", "cat replan-frontier.json && loopx refresh-state --help"]) +def test_read_and_help_extension_preserves_quota_first(command: str, tmp_path: Path) -> None: + transport = ScriptedDoubaoExecTransport([ScriptedExecToolAction(command)]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="read-before-quota", fixture_root=tmp_path, required_vision=True, + ) + assert receipt["qualification_passed"] is False + assert receipt["failure_code"] == "workspace_read_before_quota" + assert receipt["semantic_action_accepted"] is False + + +def test_read_error_reaches_model_and_recovery_still_requires_full_closeout(tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: + error = json.loads(request["messages"][-1]["content"]) + assert error["exit_code"] != 0 + assert error["error_code"] == "workspace_read_nonzero" + assert "missing.json" in error["output"] + return ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json") + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat missing.json"), recover, + vision_patch_action, projected_refresh, projected_spend, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="read-error-recovery", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is True + assert receipt["vision_closeout"]["spend_count"] == 1 + assert receipt["tool_call_count"] == 6 + assert receipt["tool_call_receipts"][1]["error_code"] == "workspace_read_nonzero" + assert "bounded" in transport.requests[0]["tools"][0]["function"]["description"] + + +def test_read_failures_consume_the_unchanged_call_budget(tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + *[ScriptedExecToolAction("cat missing.json") for _ in range(6)], + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="read-errors-exhaust-budget", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is False + assert receipt["failure_code"] == "tool_call_budget_exhausted" + assert receipt["tool_call_count"] == 7 + assert receipt["semantic_action_accepted"] is False + + +def test_help_host_extension_does_not_change_narrow_actor_contract(tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path) + assert _bounded_workspace_read_plan("loopx --help", fixture=fixture) is None + with pytest.raises(RuntimeError, match="workspace read failed"): + _execute_workspace_read("cat missing.json", fixture=fixture) + + +def test_failed_read_does_not_supply_source_evidence(tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat replan-frontier.json"), + ScriptedExecToolAction("cat missing.json"), vision_patch_action, projected_refresh, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="failed-read-is-not-evidence", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is False + assert receipt["failure_code"] == "vision_closeout_requires_observed_source_and_authored_decision" + assert receipt["semantic_action_accepted"] is False + + +def test_read_short_circuit_and_stderr_suppression_are_preserved(tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path, required_vision=True) + output, _, _, code = _execute_workspace_read("cat missing.json 2>/dev/null && loopx --help", fixture=fixture) + assert code != 0 and output == "" + output, _, _, code = _execute_workspace_read("cat missing.json 2>/dev/null || loopx --help", fixture=fixture) + assert code == 0 and "usage:" in output.lower() + + +def test_tool_description_override_is_request_local() -> None: + transport = ScriptedDoubaoExecTransport([ScriptedAssistantAction("done"), ScriptedAssistantAction("done")]) + client = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport)._client + client.next_step([], tool_description="A bounded test host") + client.next_step([]) + custom = transport.requests[0]["tools"][0] + assert custom["function"]["description"] == "A bounded test host" + assert custom["function"]["parameters"] == EXEC_COMMAND_TOOL["function"]["parameters"] + assert transport.requests[1]["tools"] == [EXEC_COMMAND_TOOL] From cb9d800628bd32a6f3ee4b2ca84d5b4c6a2b2049 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 17:30:29 +0800 Subject: [PATCH 06/22] docs(qualification): distinguish host execution from semantic admission Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- docs/development/testing-and-quality.md | 20 +++++++++++++------ .../references/repair-patterns.md | 1 + 2 files changed, 15 insertions(+), 6 deletions(-) diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index 73ec1b907c..3dc307fe9d 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -800,12 +800,20 @@ and qualification scope; a successful ordinary `typed_progress_repeat` refresh cannot qualify this journey. The narrow semantic-action gate remains useful but does not prove full closeout. Run this focused journey with `uv run --extra test python scripts/qualify-doubao-replan-semantic-action-live.py --required-vision --qualification-id `. -Its seven-call bound is unchanged. The hermetic host supports bounded reads, -real LoopX CLI calls and confined JSON authoring (`apply_patch` or a quoted -heredoc, optionally followed by bound CLI closeout); it does not -execute arbitrary model shell programs or add missing semantic fields. -Receipts distinguish accepted writeback from settled closeout, retain bounded -failure-stage codes, and never persist raw model conversations. The other turn cases remain +Its seven-call bound is unchanged. The tool description declares the actual +bounded host grammar, not unrestricted shell access: workspace reads may compose +with root/refresh-state CLI help, and JSON authoring uses `apply_patch` or a quoted +heredoc, optionally followed by bound CLI closeout. The shared read parser admits +the entire plan before execution. Ordinary nonzero reads return an exit code and +bounded output to the model; recovery consumes the same call budget and a failed +read supplies no evidence. Unsupported programs, outside-fixture writes and +invalid semantic/binding claims remain rejected. The host does not add missing +semantic fields or execute arbitrary model shell programs. These host semantics +are explicit for required-vision qualification; other actors retain their current +contracts. Receipts distinguish tool execution errors, accepted writeback and +settled closeout without persisting raw conversations. A host grammar rejection +before writeback is not evidence of a core semantic-admission failure. +The other turn cases remain bounded packet-interpretation checks. Nine core-contract scenarios cover onboarding, agent identity and goal selection, selected todo, peer identity routing, same-agent continuation, final human gate, healthy continuation, and diff --git a/skills/loopx-self-repair/references/repair-patterns.md b/skills/loopx-self-repair/references/repair-patterns.md index bbeb80e44e..a4fb14ccf3 100644 --- a/skills/loopx-self-repair/references/repair-patterns.md +++ b/skills/loopx-self-repair/references/repair-patterns.md @@ -5,6 +5,7 @@ teaches a reusable control-plane lesson. | Pattern | Symptoms | Evidence To Read | Likely Root | Durable Repair | | --- | --- | --- | --- | --- | +| `qualification_host_contract_mismatch` | A model stops during ordinary file inspection or CLI help and the result is reported as a semantic-control failure. | Actual synthetic command shape, advertised tool contract, bounded parser decision, subprocess exit status, next model request and durable writeback attempts. | An unrestricted-shell description hid a narrow grammar; supported reads and help could not compose, or a nonzero read ended qualification before returning its error to the model. | Make host capabilities explicit, reuse the bounded read parser for admitted query composition, and return ordinary read errors within the unchanged call budget. Admit the complete plan before execution; preserve evidence, identity and boundary rejection. Separate host rejection, model budget exhaustion and core semantic admission, and preserve all earlier live failures. Do not add task-specific action instructions or expand arbitrary shell authority to obtain a passing sample. | | `replan_projection_discharge_mismatch` | A projected refresh cannot satisfy its own vision obligation, or a required-vision qualification executes an ordinary repeated-progress fixture. | Typed trigger kinds, required outcomes, projected input/authoring contract, actual actor fixture, bound durable writeback, checkpoint and post-settlement readback. | Projection and admission encoded separate outcome rules; a successful narrow action was mistaken for a complete journey. | Share obligation matching and executable input projection at the typed owner. Reuse vision authoring limits, preserve original-Turn recovery and legal successor/terminal exits. Match qualification scope and triggers to the executed fixture; distinguish accepted writeback, satisfied checkpoint and one-spend settlement. A legitimate new successor obligation is not a failed closeout. Never fix this by adding ACK rituals, weakening acceptance, raising call budgets or replaying live attempts until green. | | `budget_metric_overfitting` | A budget failure triggers automatic expansion, or mechanical compaction that removes useful semantics or breaks consumers. | Owning limit, matched base/head measurements, consumer/caller contract, original failure and revised validation. | A regression metric became the objective; historical ceilings or green tests replaced semantic judgment. | Follow the [budget decision guide](../../../docs/development/testing-and-quality.md#budget-failure-decisions), compare true redundancy, compatibility cost and justified headroom, and repair the existing contract/tests and review evidence. Preserve hard limits and frozen qualification results. | | `skill_import_recreation` | Duplicate LoopX skills return after successful cleanup; imported entries display a fallback brand casing. | Compare installed files and metadata with source-host skills; correlate file creation times with structured host import receipts. | A later external-host import recreates command facades in another discovered root, omits display metadata, and bypasses installer reconciliation. | Attribute the writer from import receipts without guessing the human initiator; exclude already-installed LoopX skills from later imports and rerun managed reconciliation. Ensure standalone workflow entry installation writes Codex metadata, previews missing metadata repair, and records the full installed tree. Preserve user metadata and exact-host invocation behavior. | From 0e999c7f166c461fb05629816d2854c422061b99 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 17:33:14 +0800 Subject: [PATCH 07/22] fix(qualification): return unexecuted host rejections to the model Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../replan_semantic_action_behavior.py | 30 +++++++++---- .../replan_vision_closeout_behavior.py | 8 ++-- .../test_required_vision_closeout_behavior.py | 42 +++++++++++++++++++ 3 files changed, 68 insertions(+), 12 deletions(-) diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index fbd5f0e563..fc65819558 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -42,11 +42,12 @@ REPLAN_SEMANTIC_ACTION_BEHAVIOR_MAX_CALLS = 7 -class _WorkspaceReadFailed(RuntimeError): - """An admitted read's exit status is a tool result, not a semantic verdict.""" +class _HostToolError(RuntimeError): + """A host tool result that supplies no semantic acceptance or extra budget.""" - def __init__(self, output: str, exit_code: int) -> None: - super().__init__("workspace_read_nonzero") + def __init__(self, code: str, output: str, exit_code: int) -> None: + super().__init__(code) + self.code = code self.output = output self.exit_code = exit_code @@ -872,13 +873,14 @@ def _record_tool_step( kind: str, command: str, exit_code: int | None = None, + error_code: str | None = None, ) -> None: state.steps.append( { "ordinal": len(state.steps) + 1, "kind": kind, "command_digest": _digest(command), - **({"exit_code": exit_code, "error_code": "workspace_read_nonzero"} if exit_code else {}), + **({"exit_code": exit_code, "error_code": error_code} if error_code else {}), } ) @@ -922,7 +924,7 @@ def _handle_workspace_read(command: str, state: _QualificationState) -> str: state.work_source_read = state.work_source_read or read_work_source state.read_only_host_commands_executed = True if exit_code: - raise _WorkspaceReadFailed(output, exit_code) + raise _HostToolError("workspace_read_nonzero", output, exit_code) return output @@ -1132,6 +1134,15 @@ def _dispatch_behavior_command( ) if "evidence-log" in command: raise ValueError("manual_evidence_read_is_not_replan") + if state.fixture.required_vision: + # No handler admitted this command, so no part of it was executed. + # Rejection is not semantic success, nor a reason to hide tool feedback. + raise _HostToolError( + "unsupported_host_command", + "Command not executed: outside the bounded tool grammar. " + "Use the operations and limits in the exec_command description.", + 2, + ) raise ValueError("unexpected_command") @@ -1270,10 +1281,11 @@ def _run_qualification_loop( state, ) kind = dispatched_kind - except _WorkspaceReadFailed as exc: - _record_tool_step(state, kind=kind, command=tool_call.command, exit_code=exc.exit_code) + except _HostToolError as exc: + _record_tool_step(state, kind=kind, command=tool_call.command, + exit_code=exc.exit_code, error_code=exc.code) _append_tool_response(state, tool_call=tool_call, output=json.dumps({ - "error_code": "workspace_read_nonzero", "exit_code": exc.exit_code, + "error_code": exc.code, "exit_code": exc.exit_code, "output": exc.output[:4096], "truncated": len(exc.output) > 4096, })) continue diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py index e4588e35f7..7c4332b987 100644 --- a/loopx/control_plane/testing/replan_vision_closeout_behavior.py +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -40,9 +40,11 @@ "relative JSON file via apply_patch Add File or cat with a quoted heredoc; " "a heredoc may be followed by up to two bound LoopX closeout commands using " "newlines or &&. Other LoopX commands must be separate single invocations. " - "Returns stdout on success; nonzero workspace reads return exit_code and " - "output for correction within the same call budget. Unsupported programs " - "or syntax are never executed. No external network or outside-fixture writes." + "Returns stdout on success; nonzero workspace reads and unsupported command " + "shapes return exit_code and output for correction within the same call " + "budget. Unsupported programs or syntax are never executed. Semantic, " + "identity and write-boundary errors fail qualification. No external network " + "or outside-fixture writes." ) diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py index 1ed8bef525..43d81168d7 100644 --- a/tests/control_plane/test_required_vision_closeout_behavior.py +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -229,6 +229,48 @@ def test_read_failures_consume_the_unchanged_call_budget(tmp_path: Path) -> None assert receipt["semantic_action_accepted"] is False +@pytest.mark.parametrize("command", [ + "ls -la .codex/ && find .codex -maxdepth 5 -type f | head -50", + "touch injected", +]) +def test_unadmitted_commands_return_feedback_without_execution_or_extra_budget(command: str, tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: + error = json.loads(request["messages"][-1]["content"]) + assert error["error_code"] == "unsupported_host_command" + assert error["exit_code"] != 0 + assert "not executed" in error["output"] + return vision_patch_action(request) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), + ScriptedExecToolAction(command), recover, projected_refresh, projected_spend, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="unadmitted-command-recovery", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is True + assert receipt["vision_closeout"]["spend_count"] == 1 + assert receipt["tool_call_count"] == 6 + assert receipt["tool_call_receipts"][2]["error_code"] == "unsupported_host_command" + assert not list(tmp_path.rglob("injected")) + + +def test_unadmitted_commands_never_supply_evidence_or_success(tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + *[ScriptedExecToolAction("find . -maxdepth 5 -type f") for _ in range(6)], + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="unadmitted-command-budget", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is False + assert receipt["failure_code"] == "tool_call_budget_exhausted" + assert receipt["semantic_action_accepted"] is False + assert receipt["tool_call_count"] == 7 + + def test_help_host_extension_does_not_change_narrow_actor_contract(tmp_path: Path) -> None: fixture = _build_fixture(tmp_path) assert _bounded_workspace_read_plan("loopx --help", fixture=fixture) is None From 39d001c63771fedd32999571d5985b2441507b9c Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 17:33:14 +0800 Subject: [PATCH 08/22] docs(qualification): preserve bounded recovery without extending grammar Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- docs/development/testing-and-quality.md | 6 ++++-- skills/loopx-self-repair/references/repair-patterns.md | 2 +- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index 3dc307fe9d..8089c1e97a 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -806,8 +806,10 @@ with root/refresh-state CLI help, and JSON authoring uses `apply_patch` or a quo heredoc, optionally followed by bound CLI closeout. The shared read parser admits the entire plan before execution. Ordinary nonzero reads return an exit code and bounded output to the model; recovery consumes the same call budget and a failed -read supplies no evidence. Unsupported programs, outside-fixture writes and -invalid semantic/binding claims remain rejected. The host does not add missing +read supplies no evidence. Unadmitted commands likewise return a no-execution +error instead of ending the conversation; their grammar limits are not raised. +Unsupported programs never execute, and detected outside-fixture authoring and +invalid semantic/binding claims still fail qualification. The host does not add missing semantic fields or execute arbitrary model shell programs. These host semantics are explicit for required-vision qualification; other actors retain their current contracts. Receipts distinguish tool execution errors, accepted writeback and diff --git a/skills/loopx-self-repair/references/repair-patterns.md b/skills/loopx-self-repair/references/repair-patterns.md index a4fb14ccf3..8132dc81bd 100644 --- a/skills/loopx-self-repair/references/repair-patterns.md +++ b/skills/loopx-self-repair/references/repair-patterns.md @@ -5,7 +5,7 @@ teaches a reusable control-plane lesson. | Pattern | Symptoms | Evidence To Read | Likely Root | Durable Repair | | --- | --- | --- | --- | --- | -| `qualification_host_contract_mismatch` | A model stops during ordinary file inspection or CLI help and the result is reported as a semantic-control failure. | Actual synthetic command shape, advertised tool contract, bounded parser decision, subprocess exit status, next model request and durable writeback attempts. | An unrestricted-shell description hid a narrow grammar; supported reads and help could not compose, or a nonzero read ended qualification before returning its error to the model. | Make host capabilities explicit, reuse the bounded read parser for admitted query composition, and return ordinary read errors within the unchanged call budget. Admit the complete plan before execution; preserve evidence, identity and boundary rejection. Separate host rejection, model budget exhaustion and core semantic admission, and preserve all earlier live failures. Do not add task-specific action instructions or expand arbitrary shell authority to obtain a passing sample. | +| `qualification_host_contract_mismatch` | A model stops during ordinary file inspection or CLI help and the result is reported as a semantic-control failure. | Actual synthetic command shape, advertised tool contract, bounded parser decision, subprocess exit status, next model request and durable writeback attempts. | An unrestricted-shell description hid a narrow grammar; supported reads and help could not compose, or a read failure / unadmitted command ended qualification without tool feedback. | Declare host capabilities, reuse bounded query composition, and return read errors or no-execution rejections within the unchanged call budget. Admit the complete plan before execution; do not raise grammar limits to match a model's command. Preserve evidence, identity and write-boundary rejection. Separate host rejection, model budget exhaustion and core semantic admission; retain every earlier live failure. Do not add task-specific action instructions or arbitrary shell authority to obtain a passing sample. | | `replan_projection_discharge_mismatch` | A projected refresh cannot satisfy its own vision obligation, or a required-vision qualification executes an ordinary repeated-progress fixture. | Typed trigger kinds, required outcomes, projected input/authoring contract, actual actor fixture, bound durable writeback, checkpoint and post-settlement readback. | Projection and admission encoded separate outcome rules; a successful narrow action was mistaken for a complete journey. | Share obligation matching and executable input projection at the typed owner. Reuse vision authoring limits, preserve original-Turn recovery and legal successor/terminal exits. Match qualification scope and triggers to the executed fixture; distinguish accepted writeback, satisfied checkpoint and one-spend settlement. A legitimate new successor obligation is not a failed closeout. Never fix this by adding ACK rituals, weakening acceptance, raising call budgets or replaying live attempts until green. | | `budget_metric_overfitting` | A budget failure triggers automatic expansion, or mechanical compaction that removes useful semantics or breaks consumers. | Owning limit, matched base/head measurements, consumer/caller contract, original failure and revised validation. | A regression metric became the objective; historical ceilings or green tests replaced semantic judgment. | Follow the [budget decision guide](../../../docs/development/testing-and-quality.md#budget-failure-decisions), compare true redundancy, compatibility cost and justified headroom, and repair the existing contract/tests and review evidence. Preserve hard limits and frozen qualification results. | | `skill_import_recreation` | Duplicate LoopX skills return after successful cleanup; imported entries display a fallback brand casing. | Compare installed files and metadata with source-host skills; correlate file creation times with structured host import receipts. | A later external-host import recreates command facades in another discovered root, omits display metadata, and bypasses installer reconciliation. | Attribute the writer from import receipts without guessing the human initiator; exclude already-installed LoopX skills from later imports and rerun managed reconciliation. Ensure standalone workflow entry installation writes Codex metadata, previews missing metadata repair, and records the full installed tree. Preserve user metadata and exact-host invocation behavior. | From e626beb323b57a721f578e8aabf243f449719e49 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 17:56:35 +0800 Subject: [PATCH 09/22] fix(qualification): budget complete vision closeout independently Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../replan_semantic_action_behavior.py | 9 +++- .../replan_vision_closeout_behavior.py | 4 ++ .../test_required_vision_closeout_behavior.py | 46 +++++++++++++++++-- 3 files changed, 53 insertions(+), 6 deletions(-) diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index fc65819558..7fb66787ad 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -34,6 +34,7 @@ from .selected_todo_tool_behavior import bounded_workspace_read_plan from .replan_vision_closeout_behavior import ( dispatch_vision_closeout, VISION_HOST_INSTRUCTION, VISION_EXEC_TOOL_DESCRIPTION, + REQUIRED_VISION_CLOSEOUT_MAX_CALLS, ) REPLAN_SEMANTIC_ACTION_BEHAVIOR_RECEIPT_SCHEMA_VERSION = ( @@ -98,6 +99,11 @@ class _QualificationState: semantic_reentry_observation: dict[str, Any] | None = None vision_closeout: dict[str, Any] | None = None + @property + def tool_call_limit(self) -> int: + return (REQUIRED_VISION_CLOSEOUT_MAX_CALLS if self.fixture.required_vision + else REPLAN_SEMANTIC_ACTION_BEHAVIOR_MAX_CALLS) + def _digest(value: str) -> str: return digest_text(value) @@ -861,6 +867,7 @@ def _qualification_receipt( receipt["qualification_scope"] = ( "required_vision_closeout" if state.fixture.required_vision else "semantic_action" ) + receipt["tool_call_limit"] = state.tool_call_limit if state.vision_closeout is not None: receipt["vision_closeout"] = state.vision_closeout receipt["semantic_action_accepted"] = bool(state.semantic_delta and state.semantic_delta.get("accepted")) @@ -1261,7 +1268,7 @@ def _run_qualification_loop( *, qualification_id: str, ) -> dict[str, Any]: - for _ in range(REPLAN_SEMANTIC_ACTION_BEHAVIOR_MAX_CALLS): + for _ in range(state.tool_call_limit): tool_call = client.next_tool_call( state.messages, tool_description=VISION_EXEC_TOOL_DESCRIPTION if state.fixture.required_vision else None, diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py index 7c4332b987..0c38b0d31c 100644 --- a/loopx/control_plane/testing/replan_vision_closeout_behavior.py +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -20,6 +20,10 @@ if TYPE_CHECKING: from .replan_semantic_action_behavior import _QualificationState +# Full closeout includes evidence discovery, authoring, refresh and settlement; +# its resource bound is independent of the narrower single-action qualifier. +REQUIRED_VISION_CLOSEOUT_MAX_CALLS = 16 + VISION_HOST_INSTRUCTION = ( "You are Codex operating a hermetic LoopX project. Follow the heartbeat and " "the live control-plane packet, including its semantic writeback and settlement. " diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py index 43d81168d7..5932c6ccfe 100644 --- a/tests/control_plane/test_required_vision_closeout_behavior.py +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -214,18 +214,18 @@ def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: assert "bounded" in transport.requests[0]["tools"][0]["function"]["description"] -def test_read_failures_consume_the_unchanged_call_budget(tmp_path: Path) -> None: +def test_read_failures_consume_the_full_closeout_call_budget(tmp_path: Path) -> None: fixture = _build_fixture(tmp_path / "oracle", required_vision=True) transport = ScriptedDoubaoExecTransport([ ScriptedExecToolAction(fixture.quota_guard_command), - *[ScriptedExecToolAction("cat missing.json") for _ in range(6)], + *[ScriptedExecToolAction("cat missing.json") for _ in range(15)], ]) receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( qualification_id="read-errors-exhaust-budget", fixture_root=tmp_path / "actor", required_vision=True, ) assert receipt["qualification_passed"] is False assert receipt["failure_code"] == "tool_call_budget_exhausted" - assert receipt["tool_call_count"] == 7 + assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 16 assert receipt["semantic_action_accepted"] is False @@ -260,7 +260,7 @@ def test_unadmitted_commands_never_supply_evidence_or_success(tmp_path: Path) -> fixture = _build_fixture(tmp_path / "oracle", required_vision=True) transport = ScriptedDoubaoExecTransport([ ScriptedExecToolAction(fixture.quota_guard_command), - *[ScriptedExecToolAction("find . -maxdepth 5 -type f") for _ in range(6)], + *[ScriptedExecToolAction("find . -maxdepth 5 -type f") for _ in range(15)], ]) receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( qualification_id="unadmitted-command-budget", fixture_root=tmp_path / "actor", required_vision=True, @@ -268,7 +268,43 @@ def test_unadmitted_commands_never_supply_evidence_or_success(tmp_path: Path) -> assert receipt["qualification_passed"] is False assert receipt["failure_code"] == "tool_call_budget_exhausted" assert receipt["semantic_action_accepted"] is False - assert receipt["tool_call_count"] == 7 + assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 16 + + +@pytest.mark.parametrize("extra_reads,passed", [(11, True), (12, False)]) +def test_full_closeout_budget_boundary_never_waives_settlement(extra_reads: int, passed: bool, tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), + *[ScriptedExecToolAction("pwd") for _ in range(extra_reads)], + vision_patch_action, projected_refresh, projected_spend, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="closeout-budget-boundary", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is passed + assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 16 + assert receipt["semantic_action_accepted"] is True + assert receipt["vision_closeout"]["settled"] is passed + if passed: + assert receipt["vision_closeout"]["spend_count"] == 1 + else: + assert receipt["failure_code"] == "tool_call_budget_exhausted" + + +def test_narrow_semantic_action_keeps_seven_call_budget(tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle") + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + *[ScriptedExecToolAction("pwd") for _ in range(7)], + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="narrow-budget-unchanged", fixture_root=tmp_path / "actor", + ) + assert receipt["qualification_passed"] is False + assert receipt["failure_code"] == "tool_call_budget_exhausted" + assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 7 def test_help_host_extension_does_not_change_narrow_actor_contract(tmp_path: Path) -> None: From cdde5180837535b8b1fe05a00a48d8b556b5afb8 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 17:56:35 +0800 Subject: [PATCH 10/22] docs(qualification): distinguish scenario budgets from semantic acceptance Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- docs/development/testing-and-quality.md | 9 ++++++++- skills/loopx-self-repair/references/repair-patterns.md | 4 ++-- 2 files changed, 10 insertions(+), 3 deletions(-) diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index 8089c1e97a..631f776d20 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -800,7 +800,14 @@ and qualification scope; a successful ordinary `typed_progress_repeat` refresh cannot qualify this journey. The narrow semantic-action gate remains useful but does not prove full closeout. Run this focused journey with `uv run --extra test python scripts/qualify-doubao-replan-semantic-action-live.py --required-vision --qualification-id `. -Its seven-call bound is unchanged. The tool description declares the actual +The complete required-vision journey has a 16-call bound; the narrow +single-semantic-action qualifier retains seven. The increased budget covers +evidence discovery, JSON authoring, refresh, settlement and bounded recovery; +it does not authorize more effects or weaken the checkpoint/readback oracle. +Each receipt records `tool_call_limit` beside actual usage. Earlier seven-call +failures remain failures; the revised budget defines a new qualification, not +a retrospective pass. Limit exhaustion after refresh but before spend still +fails. The tool description declares the actual bounded host grammar, not unrestricted shell access: workspace reads may compose with root/refresh-state CLI help, and JSON authoring uses `apply_patch` or a quoted heredoc, optionally followed by bound CLI closeout. The shared read parser admits diff --git a/skills/loopx-self-repair/references/repair-patterns.md b/skills/loopx-self-repair/references/repair-patterns.md index 8132dc81bd..baa3f60c21 100644 --- a/skills/loopx-self-repair/references/repair-patterns.md +++ b/skills/loopx-self-repair/references/repair-patterns.md @@ -5,8 +5,8 @@ teaches a reusable control-plane lesson. | Pattern | Symptoms | Evidence To Read | Likely Root | Durable Repair | | --- | --- | --- | --- | --- | -| `qualification_host_contract_mismatch` | A model stops during ordinary file inspection or CLI help and the result is reported as a semantic-control failure. | Actual synthetic command shape, advertised tool contract, bounded parser decision, subprocess exit status, next model request and durable writeback attempts. | An unrestricted-shell description hid a narrow grammar; supported reads and help could not compose, or a read failure / unadmitted command ended qualification without tool feedback. | Declare host capabilities, reuse bounded query composition, and return read errors or no-execution rejections within the unchanged call budget. Admit the complete plan before execution; do not raise grammar limits to match a model's command. Preserve evidence, identity and write-boundary rejection. Separate host rejection, model budget exhaustion and core semantic admission; retain every earlier live failure. Do not add task-specific action instructions or arbitrary shell authority to obtain a passing sample. | -| `replan_projection_discharge_mismatch` | A projected refresh cannot satisfy its own vision obligation, or a required-vision qualification executes an ordinary repeated-progress fixture. | Typed trigger kinds, required outcomes, projected input/authoring contract, actual actor fixture, bound durable writeback, checkpoint and post-settlement readback. | Projection and admission encoded separate outcome rules; a successful narrow action was mistaken for a complete journey. | Share obligation matching and executable input projection at the typed owner. Reuse vision authoring limits, preserve original-Turn recovery and legal successor/terminal exits. Match qualification scope and triggers to the executed fixture; distinguish accepted writeback, satisfied checkpoint and one-spend settlement. A legitimate new successor obligation is not a failed closeout. Never fix this by adding ACK rituals, weakening acceptance, raising call budgets or replaying live attempts until green. | +| `qualification_host_contract_mismatch` | A model stops during ordinary file inspection or CLI help and the result is reported as a semantic-control failure. | Actual synthetic command shape, advertised tool contract, bounded parser decision, subprocess exit status, next model request and durable writeback attempts. | An unrestricted-shell description hid a narrow grammar; supported reads and help could not compose, or a read failure / unadmitted command ended qualification without tool feedback. | Declare host capabilities, reuse bounded query composition, and return read errors or no-execution rejections within the declared scenario budget. Admit the complete plan before execution; do not raise grammar limits to match a model's command. Preserve evidence, identity and write-boundary rejection. Separate host rejection, model budget exhaustion and core semantic admission; retain every earlier live failure. Do not add task-specific action instructions or arbitrary shell authority to obtain a passing sample. | +| `replan_projection_discharge_mismatch` | A projected refresh cannot satisfy its own vision obligation, or a required-vision qualification executes an ordinary repeated-progress fixture. | Typed trigger kinds, required outcomes, projected input/authoring contract, actual actor fixture, bound durable writeback, checkpoint and post-settlement readback. | Projection and admission encoded separate outcome rules; a successful narrow action was mistaken for a complete journey. | Share outcome matching and executable input projection at the typed owner. Preserve original-Turn recovery, authoring limits and legal successor/terminal exits. Match qualification scope to the fixture; distinguish accepted writeback, checkpoint and one-spend settlement. A new successor duty is not a failed closeout. Do not add ACK rituals, weaken acceptance or replay until green. Full closeout and one action need not share a call cap: an authorized, cost-disclosed budget revision is a new qualification and must retain old failures, not relabel them. | | `budget_metric_overfitting` | A budget failure triggers automatic expansion, or mechanical compaction that removes useful semantics or breaks consumers. | Owning limit, matched base/head measurements, consumer/caller contract, original failure and revised validation. | A regression metric became the objective; historical ceilings or green tests replaced semantic judgment. | Follow the [budget decision guide](../../../docs/development/testing-and-quality.md#budget-failure-decisions), compare true redundancy, compatibility cost and justified headroom, and repair the existing contract/tests and review evidence. Preserve hard limits and frozen qualification results. | | `skill_import_recreation` | Duplicate LoopX skills return after successful cleanup; imported entries display a fallback brand casing. | Compare installed files and metadata with source-host skills; correlate file creation times with structured host import receipts. | A later external-host import recreates command facades in another discovered root, omits display metadata, and bypasses installer reconciliation. | Attribute the writer from import receipts without guessing the human initiator; exclude already-installed LoopX skills from later imports and rerun managed reconciliation. Ensure standalone workflow entry installation writes Codex metadata, previews missing metadata repair, and records the full installed tree. Preserve user metadata and exact-host invocation behavior. | | `skill_discovery_split_ownership` | Duplicate skill names, conflicting PR-review routes, or canonical and legacy aliases appear together. | Enumerate discovered roots, resolve directory symlinks, compare skill hashes, managed markers, install receipts, and generated metadata. | Workflow and command installers wrote independently to overlapping host roots; dedupe was optional, omitted the bare entry name, or retired copies without proving a replacement. | Repair the shared installer reconciliation and every active installation path; preserve user changes and rich workflows, retire managed aliases from the Codex picker, test repeated installs and custom profiles, then verify a fresh host catalog. Do not treat deleting one visible duplicate or changing invocation policy as a durable fix. | From 6a6ef5d59be9458ee66d3d0099a719b673a392c1 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:00:25 +0800 Subject: [PATCH 11/22] fix(qualification): return pre-execution authoring rejection as tool feedback Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../replan_semantic_action_behavior.py | 9 +++- .../replan_vision_closeout_behavior.py | 49 +++++++++++++------ .../test_required_vision_closeout_behavior.py | 36 ++++++++++++++ 3 files changed, 77 insertions(+), 17 deletions(-) diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index 7fb66787ad..4a6d6c14a3 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -35,6 +35,7 @@ from .replan_vision_closeout_behavior import ( dispatch_vision_closeout, VISION_HOST_INSTRUCTION, VISION_EXEC_TOOL_DESCRIPTION, REQUIRED_VISION_CLOSEOUT_MAX_CALLS, + VisionAuthoringRejected, ) REPLAN_SEMANTIC_ACTION_BEHAVIOR_RECEIPT_SCHEMA_VERSION = ( @@ -1288,7 +1289,13 @@ def _run_qualification_loop( state, ) kind = dispatched_kind - except _HostToolError as exc: + except (VisionAuthoringRejected, _HostToolError) as exc: + if isinstance(exc, VisionAuthoringRejected): + exc = _HostToolError( + str(exc), "File and suffix not executed: " + str(exc) + ". " + "Use a new relative JSON file and the declared authoring grammar; " + "other operations can be issued as separate tool calls.", 2, + ) _record_tool_step(state, kind=kind, command=tool_call.command, exit_code=exc.exit_code, error_code=exc.code) _append_tool_response(state, tool_call=tool_call, output=json.dumps({ diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py index 0c38b0d31c..79fffa47a7 100644 --- a/loopx/control_plane/testing/replan_vision_closeout_behavior.py +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -24,6 +24,11 @@ # its resource bound is independent of the narrower single-action qualifier. REQUIRED_VISION_CLOSEOUT_MAX_CALLS = 16 + +class VisionAuthoringRejected(ValueError): + """Pre-execution file admission failure; no file or CLI effect has run.""" + + VISION_HOST_INSTRUCTION = ( "You are Codex operating a hermetic LoopX project. Follow the heartbeat and " "the live control-plane packet, including its semantic writeback and settlement. " @@ -46,8 +51,9 @@ "newlines or &&. Other LoopX commands must be separate single invocations. " "Returns stdout on success; nonzero workspace reads and unsupported command " "shapes return exit_code and output for correction within the same call " - "budget. Unsupported programs or syntax are never executed. Semantic, " - "identity and write-boundary errors fail qualification. No external network " + "budget. Rejected file-authoring operations also return errors without " + "creating a file or executing any suffix. Unsupported programs never run; " + "semantic and identity errors fail qualification. No external network " "or outside-fixture writes." ) @@ -103,22 +109,26 @@ def _authored_file(command: str, project: Path) -> str: r"(.*?)\n\*\*\* End Patch\n\1", command.strip(), re.DOTALL, ) if not match: - raise ValueError("vision_authoring_requires_single_add_file_patch") + raise VisionAuthoringRejected("vision_authoring_requires_single_add_file_patch") lines = match[3].splitlines() if not lines or any(not line.startswith("+") for line in lines): - raise ValueError("vision_authoring_requires_added_lines") + raise VisionAuthoringRejected("vision_authoring_requires_added_lines") return _write_json_file(project, match[2], "\n".join(line[1:] for line in lines) + "\n") def _write_json_file(project: Path, name: str, content: str) -> str: relative = Path(name) if relative.is_absolute() or relative.suffix != ".json" or ".." in relative.parts: - raise ValueError("vision_authoring_path_outside_fixture") + raise VisionAuthoringRejected("vision_authoring_path_outside_fixture") target = (project / relative).resolve() if not target.is_relative_to(project.resolve()) or target.exists(): - raise ValueError("vision_authoring_path_outside_fixture") - if not isinstance(json.loads(content), dict): - raise ValueError("vision_authoring_requires_json_object") + raise VisionAuthoringRejected("vision_authoring_path_outside_fixture") + try: + value = json.loads(content) + except json.JSONDecodeError: + raise VisionAuthoringRejected("vision_authoring_requires_json_object") from None + if not isinstance(value, dict): + raise VisionAuthoringRejected("vision_authoring_requires_json_object") target.parent.mkdir(parents=True, exist_ok=True) target.write_text(content, encoding="utf-8") return json.dumps({"ok": True, "path": relative.as_posix()}) @@ -134,36 +144,43 @@ def _json_heredoc(command: str) -> tuple[str, str, list[str]] | None: return None lexer = shlex.shlex(header, posix=True, punctuation_chars="<>&;|") lexer.whitespace_split = True - tokens = list(lexer) + try: + tokens = list(lexer) + except ValueError: + raise VisionAuthoringRejected("vision_authoring_requires_literal_json_heredoc") from None if len(tokens) != 5 or tokens[0] != "cat" or set(tokens[1::2]) != {">", "<<"}: - raise ValueError("vision_authoring_requires_literal_json_heredoc") + raise VisionAuthoringRejected("vision_authoring_requires_literal_json_heredoc") name = tokens[tokens.index(">") + 1] delimiter = tokens[tokens.index("<<") + 1] if delimiter != quoted[2]: - raise ValueError("vision_authoring_requires_literal_json_heredoc") + raise VisionAuthoringRejected("vision_authoring_requires_literal_json_heredoc") lines = rest.splitlines(keepends=True) end = next((index for index, line in enumerate(lines) if line.rstrip("\r\n") == delimiter), None) if end is None: - raise ValueError("vision_authoring_requires_literal_json_heredoc") + raise VisionAuthoringRejected("vision_authoring_requires_literal_json_heredoc") suffix = "".join(lines[end + 1:]).strip() lexer = shlex.shlex(suffix, posix=True, punctuation_chars=";&|\n") lexer.whitespace = " \t\r" lexer.whitespace_split = True commands: list[str] = [] argv: list[str] = [] - for token in [*lexer, "\n"]: + try: + suffix_tokens = list(lexer) + except ValueError: + raise VisionAuthoringRejected("vision_authoring_suffix_requires_loopx") from None + for token in [*suffix_tokens, "\n"]: if token == "&&" or token.strip("\n") == "": if argv: if Path(argv[0]).name != "loopx": - raise ValueError("vision_authoring_suffix_requires_loopx") + raise VisionAuthoringRejected("vision_authoring_suffix_requires_loopx") commands.append(shlex.join(argv)) argv = [] elif token in {";", "|", "||", "&"}: - raise ValueError("vision_authoring_suffix_requires_loopx") + raise VisionAuthoringRejected("vision_authoring_suffix_requires_loopx") else: argv.append(token) if len(commands) > 2: - raise ValueError("vision_authoring_suffix_requires_loopx") + raise VisionAuthoringRejected("vision_authoring_suffix_requires_loopx") return name, "".join(lines[:end]), commands diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py index 5932c6ccfe..c895fc2af5 100644 --- a/tests/control_plane/test_required_vision_closeout_behavior.py +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -307,6 +307,42 @@ def test_narrow_semantic_action_keeps_seven_call_budget(tmp_path: Path) -> None: assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 7 +@pytest.mark.parametrize("rejection", ["suffix", "outside", "invalid_json", "overwrite"]) +def test_authoring_rejection_is_recoverable_but_has_no_file_or_suffix_effect(rejection: str, tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + root = tmp_path / "actor" / "project" + def rejected_authoring(request: Mapping[str, Any]) -> ScriptedExecToolAction: + command = heredoc_action(request).command + if rejection == "suffix": + command += "\ntouch injected" + elif rejection == "outside": + command = command.replace("decision.json", "../outside.json") + elif rejection == "invalid_json": + command = "cat > decision.json <<'JSON'\n{broken\nJSON" + else: + command = command.replace("decision.json", "fixture/permission-config.json") + return ScriptedExecToolAction(command) + def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: + error = json.loads(request["messages"][-1]["content"]) + assert error["exit_code"] != 0 and "not executed" in error["output"] + assert not (root / "decision.json").exists() + assert not list(tmp_path.rglob("injected")) + assert not list(tmp_path.rglob("outside.json")) + assert (root / "fixture/permission-config.json").read_bytes() == fixture.work_source_target.read_bytes() + return heredoc_action(request) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), + rejected_authoring, recover, projected_refresh, projected_spend, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="authoring-error-recovery", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is True + assert receipt["vision_closeout"]["spend_count"] == 1 + assert receipt["tool_call_count"] == 6 + + def test_help_host_extension_does_not_change_narrow_actor_contract(tmp_path: Path) -> None: fixture = _build_fixture(tmp_path) assert _bounded_workspace_read_plan("loopx --help", fixture=fixture) is None From 2918526070a4ded74540c8ce8dc15b92b37989f4 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:00:25 +0800 Subject: [PATCH 12/22] docs(qualification): clarify rejected authoring and post-effect failures Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- docs/development/testing-and-quality.md | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index 631f776d20..7bd8249624 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -815,8 +815,10 @@ the entire plan before execution. Ordinary nonzero reads return an exit code and bounded output to the model; recovery consumes the same call budget and a failed read supplies no evidence. Unadmitted commands likewise return a no-execution error instead of ending the conversation; their grammar limits are not raised. -Unsupported programs never execute, and detected outside-fixture authoring and -invalid semantic/binding claims still fail qualification. The host does not add missing +Unsupported programs and rejected file-authoring operations never execute. +Pre-execution path, overwrite, JSON or authoring-syntax rejection returns tool +feedback without creating the file or running a suffix. Invalid semantic/binding +claims and errors after an effect remain terminal failures. The host does not add missing semantic fields or execute arbitrary model shell programs. These host semantics are explicit for required-vision qualification; other actors retain their current contracts. Receipts distinguish tool execution errors, accepted writeback and From 10e3722096555e360cb416aa1e5ff70e6e58e411 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:06:14 +0800 Subject: [PATCH 13/22] fix(qualification): ground evidence references and admit literal CLI commands Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../replan_semantic_action_behavior.py | 10 ++-- .../replan_vision_closeout_behavior.py | 43 +++++++++------- .../test_required_vision_closeout_behavior.py | 50 +++++++++++++++++++ 3 files changed, 80 insertions(+), 23 deletions(-) diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index 4a6d6c14a3..60acce9872 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -35,7 +35,7 @@ from .replan_vision_closeout_behavior import ( dispatch_vision_closeout, VISION_HOST_INSTRUCTION, VISION_EXEC_TOOL_DESCRIPTION, REQUIRED_VISION_CLOSEOUT_MAX_CALLS, - VisionAuthoringRejected, + VisionHostAdmissionRejected, ) REPLAN_SEMANTIC_ACTION_BEHAVIOR_RECEIPT_SCHEMA_VERSION = ( @@ -1289,11 +1289,11 @@ def _run_qualification_loop( state, ) kind = dispatched_kind - except (VisionAuthoringRejected, _HostToolError) as exc: - if isinstance(exc, VisionAuthoringRejected): + except (VisionHostAdmissionRejected, _HostToolError) as exc: + if isinstance(exc, VisionHostAdmissionRejected): exc = _HostToolError( - str(exc), "File and suffix not executed: " + str(exc) + ". " - "Use a new relative JSON file and the declared authoring grammar; " + str(exc), "Command not executed: " + str(exc) + ". " + "Use literal LoopX arguments, new relative JSON files and the declared grammar; " "other operations can be issued as separate tool calls.", 2, ) _record_tool_step(state, kind=kind, command=tool_call.command, diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py index 79fffa47a7..dd4ad31c7b 100644 --- a/loopx/control_plane/testing/replan_vision_closeout_behavior.py +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -25,8 +25,8 @@ REQUIRED_VISION_CLOSEOUT_MAX_CALLS = 16 -class VisionAuthoringRejected(ValueError): - """Pre-execution file admission failure; no file or CLI effect has run.""" +class VisionHostAdmissionRejected(ValueError): + """Pre-execution host rejection; no file or CLI effect has run.""" VISION_HOST_INSTRUCTION = ( @@ -109,26 +109,26 @@ def _authored_file(command: str, project: Path) -> str: r"(.*?)\n\*\*\* End Patch\n\1", command.strip(), re.DOTALL, ) if not match: - raise VisionAuthoringRejected("vision_authoring_requires_single_add_file_patch") + raise VisionHostAdmissionRejected("vision_authoring_requires_single_add_file_patch") lines = match[3].splitlines() if not lines or any(not line.startswith("+") for line in lines): - raise VisionAuthoringRejected("vision_authoring_requires_added_lines") + raise VisionHostAdmissionRejected("vision_authoring_requires_added_lines") return _write_json_file(project, match[2], "\n".join(line[1:] for line in lines) + "\n") def _write_json_file(project: Path, name: str, content: str) -> str: relative = Path(name) if relative.is_absolute() or relative.suffix != ".json" or ".." in relative.parts: - raise VisionAuthoringRejected("vision_authoring_path_outside_fixture") + raise VisionHostAdmissionRejected("vision_authoring_path_outside_fixture") target = (project / relative).resolve() if not target.is_relative_to(project.resolve()) or target.exists(): - raise VisionAuthoringRejected("vision_authoring_path_outside_fixture") + raise VisionHostAdmissionRejected("vision_authoring_path_outside_fixture") try: value = json.loads(content) except json.JSONDecodeError: - raise VisionAuthoringRejected("vision_authoring_requires_json_object") from None + raise VisionHostAdmissionRejected("vision_authoring_requires_json_object") from None if not isinstance(value, dict): - raise VisionAuthoringRejected("vision_authoring_requires_json_object") + raise VisionHostAdmissionRejected("vision_authoring_requires_json_object") target.parent.mkdir(parents=True, exist_ok=True) target.write_text(content, encoding="utf-8") return json.dumps({"ok": True, "path": relative.as_posix()}) @@ -147,17 +147,17 @@ def _json_heredoc(command: str) -> tuple[str, str, list[str]] | None: try: tokens = list(lexer) except ValueError: - raise VisionAuthoringRejected("vision_authoring_requires_literal_json_heredoc") from None + raise VisionHostAdmissionRejected("vision_authoring_requires_literal_json_heredoc") from None if len(tokens) != 5 or tokens[0] != "cat" or set(tokens[1::2]) != {">", "<<"}: - raise VisionAuthoringRejected("vision_authoring_requires_literal_json_heredoc") + raise VisionHostAdmissionRejected("vision_authoring_requires_literal_json_heredoc") name = tokens[tokens.index(">") + 1] delimiter = tokens[tokens.index("<<") + 1] if delimiter != quoted[2]: - raise VisionAuthoringRejected("vision_authoring_requires_literal_json_heredoc") + raise VisionHostAdmissionRejected("vision_authoring_requires_literal_json_heredoc") lines = rest.splitlines(keepends=True) end = next((index for index, line in enumerate(lines) if line.rstrip("\r\n") == delimiter), None) if end is None: - raise VisionAuthoringRejected("vision_authoring_requires_literal_json_heredoc") + raise VisionHostAdmissionRejected("vision_authoring_requires_literal_json_heredoc") suffix = "".join(lines[end + 1:]).strip() lexer = shlex.shlex(suffix, posix=True, punctuation_chars=";&|\n") lexer.whitespace = " \t\r" @@ -167,20 +167,20 @@ def _json_heredoc(command: str) -> tuple[str, str, list[str]] | None: try: suffix_tokens = list(lexer) except ValueError: - raise VisionAuthoringRejected("vision_authoring_suffix_requires_loopx") from None + raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx") from None for token in [*suffix_tokens, "\n"]: if token == "&&" or token.strip("\n") == "": if argv: if Path(argv[0]).name != "loopx": - raise VisionAuthoringRejected("vision_authoring_suffix_requires_loopx") + raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx") commands.append(shlex.join(argv)) argv = [] elif token in {";", "|", "||", "&"}: - raise VisionAuthoringRejected("vision_authoring_suffix_requires_loopx") + raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx") else: argv.append(token) if len(commands) > 2: - raise VisionAuthoringRejected("vision_authoring_suffix_requires_loopx") + raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx") return name, "".join(lines[:end]), commands @@ -214,6 +214,11 @@ def dispatch_vision_closeout( tokens = loopx_command_tokens(command) or [] if "refresh-state" not in tokens and "spend-slot" not in tokens: return None + lexer = shlex.shlex(command.strip(), posix=True, punctuation_chars=";&|\n") + lexer.whitespace = " \t\r" + lexer.whitespace_split = True + if tokens != list(lexer): + raise VisionHostAdmissionRejected("vision_command_requires_literal_loopx_argv") packet = state.quota_packet or {} binding = dict(dict(dict(packet.get("interaction_contract") or {}).get("cli_channel") or {}).get("replan_settlement_contract") or {}).get("settlement_binding") or {} if not binding or argument_value(tokens, binding["cli_argument"]) != binding["id"]: @@ -229,13 +234,15 @@ def dispatch_vision_closeout( raise ValueError("vision_authoring_path_outside_fixture") vision = json.loads(target.read_text(encoding="utf-8")) source_evidence = json.loads(state.fixture.frontier_target.read_text(encoding="utf-8"))["uncovered"] - observed_ids = {item["evidence_id"] for item in source_evidence} + source_ref = state.fixture.work_source_target.relative_to(state.fixture.project_root).as_posix() + observed_refs = {ref for item in source_evidence if item.get("source_ref") == source_ref + for ref in (item["evidence_id"], source_ref)} path_delta = vision.get("path_delta") refs = path_delta.get("evidence_refs") if isinstance(path_delta, dict) else None if not isinstance(refs, list) or any(not isinstance(ref, str) for ref in refs): raise ValueError("vision_closeout_evidence_not_observed") refs = set(refs) - if not refs.intersection(observed_ids): + if not refs.intersection(observed_refs): raise ValueError("vision_closeout_evidence_not_observed") output = execute(command, fixture=state.fixture, turn_instance_id=state.turn_instance_id) row = _rows(state)[-1] diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py index c895fc2af5..cb3f508f54 100644 --- a/tests/control_plane/test_required_vision_closeout_behavior.py +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -343,6 +343,56 @@ def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: assert receipt["tool_call_count"] == 6 +@pytest.mark.parametrize("evidence_ref,passed", [ + ("evidence-permission-config", True), ("fixture/permission-config.json", True), + ("fixture/unread.json", False), ("replan-frontier.json", False), +]) +def test_closeout_matches_observed_evidence_not_one_identifier_spelling(evidence_ref: str, passed: bool, tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + def author(request: Mapping[str, Any]) -> ScriptedExecToolAction: + return ScriptedExecToolAction(heredoc_action(request).command.replace("evidence-permission-config", evidence_ref)) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), + author, projected_refresh, projected_spend, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="observed-evidence-reference", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is passed + if passed: + assert receipt["vision_closeout"]["spend_count"] == 1 + else: + assert receipt["failure_code"] == "vision_closeout_evidence_not_observed" + assert receipt["semantic_action_accepted"] is False + + +@pytest.mark.parametrize("shell_shape", ["assignment_prefix", "newline_sequence"]) +def test_shell_shape_is_not_silently_discarded_before_binding_check(shell_shape: str, tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + def prefixed(request: Mapping[str, Any]) -> ScriptedExecToolAction: + command = projected_refresh(request).command + command = ("TURN=unexpanded\n" + command if shell_shape == "assignment_prefix" + else command + "\n" + projected_spend(request).command) + return ScriptedExecToolAction(command) + def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: + error = json.loads(request["messages"][-1]["content"]) + assert error["error_code"] == "vision_command_requires_literal_loopx_argv" + assert "not executed" in error["output"] + return projected_refresh(request) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), + vision_patch_action, prefixed, recover, projected_spend, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="literal-cli-admission", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is True + assert receipt["vision_closeout"]["spend_count"] == 1 + assert receipt["tool_call_count"] == 6 + + def test_help_host_extension_does_not_change_narrow_actor_contract(tmp_path: Path) -> None: fixture = _build_fixture(tmp_path) assert _bounded_workspace_read_plan("loopx --help", fixture=fixture) is None From 76241000aa2b119ae4f525889df5d01df0a03c98 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:06:14 +0800 Subject: [PATCH 14/22] docs(qualification): define source reference and literal argv boundaries Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- docs/development/testing-and-quality.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index 7bd8249624..07e9558ec0 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -824,6 +824,9 @@ are explicit for required-vision qualification; other actors retain their curren contracts. Receipts distinguish tool execution errors, accepted writeback and settled closeout without persisting raw conversations. A host grammar rejection before writeback is not evidence of a core semantic-admission failure. +Shell assignment prefixes are rejected before CLI identity checks, not silently +discarded. An observed evidence item's id and declared source reference both +identify that same evidence; arbitrary or unread sources do not qualify it. The other turn cases remain bounded packet-interpretation checks. Nine core-contract scenarios cover onboarding, agent identity and goal selection, selected todo, peer identity From 41fe5240e25c632ddbebaeea0da82141c8114a85 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:16:49 +0800 Subject: [PATCH 15/22] fix(testing): let vision actors correct rejected drafts Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../testing/model_tool_behavior.py | 13 +++++-- .../replan_semantic_action_behavior.py | 34 ++++++++++------- .../replan_vision_closeout_behavior.py | 38 +++++++++++-------- .../test_required_vision_closeout_behavior.py | 38 ++++++++++++++++++- 4 files changed, 89 insertions(+), 34 deletions(-) diff --git a/loopx/control_plane/testing/model_tool_behavior.py b/loopx/control_plane/testing/model_tool_behavior.py index c55fb4a5d8..5bc68e5aa2 100644 --- a/loopx/control_plane/testing/model_tool_behavior.py +++ b/loopx/control_plane/testing/model_tool_behavior.py @@ -547,8 +547,13 @@ def execute_loopx_cli( bounded_detail = ( detail if len(detail) <= 1_000 else detail[:500] + "\n...\n" + detail[-500:] ) - raise RuntimeError( - "LoopX CLI command failed with " - f"exit={completed.returncode}: {bounded_detail}" - ) + raise LoopxCliExecutionError(completed.returncode, bounded_detail) return completed.stdout + + +class LoopxCliExecutionError(RuntimeError): + """An executed CLI returned nonzero; this does not imply state rollback.""" + + def __init__(self, returncode: int, detail: str) -> None: + super().__init__(f"LoopX CLI command failed with exit={returncode}: {detail}") + self.returncode = returncode diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index 60acce9872..d0603354ae 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -5,7 +5,7 @@ import shlex import subprocess from collections.abc import Mapping -from dataclasses import dataclass +from dataclasses import dataclass, field as dataclass_field from hashlib import sha256 from pathlib import Path from typing import Any @@ -25,6 +25,7 @@ _direct_ark_transport, ) from .model_tool_behavior import ( + LoopxCliExecutionError, DoubaoExecToolClient, argument_value, digest_text, @@ -99,6 +100,7 @@ class _QualificationState: successor_reentry_observation: dict[str, Any] | None = None semantic_reentry_observation: dict[str, Any] | None = None vision_closeout: dict[str, Any] | None = None + authored_paths: set[Path] = dataclass_field(default_factory=set) @property def tool_call_limit(self) -> int: @@ -692,16 +694,21 @@ def _execute_loopx( tokens[1:1] = ["--registry", str(fixture.global_registry_path)] if "refresh-state" in tokens: tokens.extend(["--no-global-sync", "--suppress-external-sinks"]) - return execute_loopx_cli( - shlex.join(tokens), - source_root=fixture.source_root, - project_root=fixture.project_root, - argument_overrides={ - "--registry": str(fixture.global_registry_path), - "--runtime-root": str(fixture.runtime_root), - "--turn-instance-id": turn_instance_id, - }, - ) + try: + return execute_loopx_cli( + shlex.join(tokens), + source_root=fixture.source_root, + project_root=fixture.project_root, + argument_overrides={ + "--registry": str(fixture.global_registry_path), + "--runtime-root": str(fixture.runtime_root), + "--turn-instance-id": turn_instance_id, + }, + ) + except LoopxCliExecutionError as exc: + if not fixture.required_vision: + raise + raise _HostToolError("loopx_cli_nonzero", str(exc), exc.returncode) from None def _bounded_workspace_read_plan( @@ -1292,8 +1299,9 @@ def _run_qualification_loop( except (VisionHostAdmissionRejected, _HostToolError) as exc: if isinstance(exc, VisionHostAdmissionRejected): exc = _HostToolError( - str(exc), "Command not executed: " + str(exc) + ". " - "Use literal LoopX arguments, new relative JSON files and the declared grammar; " + str(exc), "Rejected operation not executed: " + str(exc) + ". " + exc.detail + " " + "Earlier successful steps, if any, are not rolled back. " + "Use literal LoopX arguments, relative JSON drafts and the declared grammar; " "other operations can be issued as separate tool calls.", 2, ) _record_tool_step(state, kind=kind, command=tool_call.command, diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py index dd4ad31c7b..7ce81eb7e9 100644 --- a/loopx/control_plane/testing/replan_vision_closeout_behavior.py +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -26,7 +26,11 @@ class VisionHostAdmissionRejected(ValueError): - """Pre-execution host rejection; no file or CLI effect has run.""" + """Reject the current operation before execution, not prior compound steps.""" + + def __init__(self, code: str, detail: str = "") -> None: + super().__init__(code) + self.detail = detail VISION_HOST_INSTRUCTION = ( @@ -47,13 +51,15 @@ class VisionHostAdmissionRejected(ValueError): "loopx --help and loopx refresh-state --help are read-only queries and may " "be combined with reads or piped to head (at most 200 lines). Author a new " "relative JSON file via apply_patch Add File or cat with a quoted heredoc; " + "these operations may revise your own JSON drafts, never original fixture files. " "a heredoc may be followed by up to two bound LoopX closeout commands using " "newlines or &&. Other LoopX commands must be separate single invocations. " "Returns stdout on success; nonzero workspace reads and unsupported command " "shapes return exit_code and output for correction within the same call " "budget. Rejected file-authoring operations also return errors without " - "creating a file or executing any suffix. Unsupported programs never run; " - "semantic and identity errors fail qualification. No external network " + "creating a file or executing any suffix. CLI validation errors return " + "their diagnostics; identity and durable-verification errors fail " + "qualification. Unsupported programs never run. No external network " "or outside-fixture writes." ) @@ -100,7 +106,7 @@ def required_vision_scenario_contract( } -def _authored_file(command: str, project: Path) -> str: +def _authored_file(command: str, project: Path, *, authored_paths: set[Path] | None = None) -> str: # Accept data, never execute the model's shell program. JSON quotes and # shell metacharacters inside added lines remain literal file content. match = re.fullmatch( @@ -113,16 +119,16 @@ def _authored_file(command: str, project: Path) -> str: lines = match[3].splitlines() if not lines or any(not line.startswith("+") for line in lines): raise VisionHostAdmissionRejected("vision_authoring_requires_added_lines") - return _write_json_file(project, match[2], "\n".join(line[1:] for line in lines) + "\n") + return _write_json_file(project, match[2], "\n".join(line[1:] for line in lines) + "\n", authored_paths=authored_paths) -def _write_json_file(project: Path, name: str, content: str) -> str: +def _write_json_file(project: Path, name: str, content: str, *, authored_paths: set[Path] | None = None) -> str: relative = Path(name) if relative.is_absolute() or relative.suffix != ".json" or ".." in relative.parts: - raise VisionHostAdmissionRejected("vision_authoring_path_outside_fixture") + raise VisionHostAdmissionRejected("vision_authoring_path_outside_fixture", "The JSON path must be relative to the project, without parent traversal; absolute temporary paths are not writable.") target = (project / relative).resolve() - if not target.is_relative_to(project.resolve()) or target.exists(): - raise VisionHostAdmissionRejected("vision_authoring_path_outside_fixture") + if not target.is_relative_to(project.resolve()) or (target.exists() and target not in (authored_paths or set())): + raise VisionHostAdmissionRejected("vision_authoring_path_outside_fixture", "Only new JSON files or this actor's own drafts are writable; existing fixture inputs are immutable.") try: value = json.loads(content) except json.JSONDecodeError: @@ -131,6 +137,8 @@ def _write_json_file(project: Path, name: str, content: str) -> str: raise VisionHostAdmissionRejected("vision_authoring_requires_json_object") target.parent.mkdir(parents=True, exist_ok=True) target.write_text(content, encoding="utf-8") + if authored_paths is not None: + authored_paths.add(target) return json.dumps({"ok": True, "path": relative.as_posix()}) @@ -172,15 +180,15 @@ def _json_heredoc(command: str) -> tuple[str, str, list[str]] | None: if token == "&&" or token.strip("\n") == "": if argv: if Path(argv[0]).name != "loopx": - raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx") + raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx", f"The command after the heredoc delimiter starts with {argv[0]!r}; only literal LoopX closeout commands are admitted there. Submit the heredoc alone; other operations need separate tool calls.") commands.append(shlex.join(argv)) argv = [] elif token in {";", "|", "||", "&"}: - raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx") + raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx", f"The heredoc suffix contains unsupported operator {token!r}; use a standalone heredoc or literal LoopX commands separated by newlines or &&.") else: argv.append(token) if len(commands) > 2: - raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx") + raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx", "A heredoc permits at most two literal LoopX closeout commands; issue other operations in separate tool calls.") return name, "".join(lines[:end]), commands @@ -198,7 +206,7 @@ def dispatch_vision_closeout( if not state.seen_quota: raise ValueError("vision_authoring_before_quota") name, content, suffix = heredoc - result: tuple[str, str, bool] = (_write_json_file(state.fixture.project_root, name, content), "vision_file_authoring", False) + result: tuple[str, str, bool] = (_write_json_file(state.fixture.project_root, name, content, authored_paths=state.authored_paths), "vision_file_authoring", False) for following in suffix: if result[2]: raise ValueError("vision_authoring_suffix_requires_loopx") @@ -210,7 +218,7 @@ def dispatch_vision_closeout( if command.startswith("apply_patch"): if not state.seen_quota: raise ValueError("vision_authoring_before_quota") - return _authored_file(command, state.fixture.project_root), "vision_file_authoring", False + return _authored_file(command, state.fixture.project_root, authored_paths=state.authored_paths), "vision_file_authoring", False tokens = loopx_command_tokens(command) or [] if "refresh-state" not in tokens and "spend-slot" not in tokens: return None @@ -243,7 +251,7 @@ def dispatch_vision_closeout( raise ValueError("vision_closeout_evidence_not_observed") refs = set(refs) if not refs.intersection(observed_refs): - raise ValueError("vision_closeout_evidence_not_observed") + raise VisionHostAdmissionRejected("vision_closeout_evidence_not_observed", "No refresh was executed: path_delta.evidence_refs must identify the source evidence actually read. Accepted references: " + json.dumps(sorted(observed_refs))) output = execute(command, fixture=state.fixture, turn_instance_id=state.turn_instance_id) row = _rows(state)[-1] semantic = dict(row.get("autonomous_replan_ack") or {}).get("semantic_delta") or {} diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py index cb3f508f54..30d5862562 100644 --- a/tests/control_plane/test_required_vision_closeout_behavior.py +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -325,6 +325,8 @@ def rejected_authoring(request: Mapping[str, Any]) -> ScriptedExecToolAction: def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: error = json.loads(request["messages"][-1]["content"]) assert error["exit_code"] != 0 and "not executed" in error["output"] + if rejection == "suffix": + assert "touch" in error["output"] and "heredoc alone" in error["output"] assert not (root / "decision.json").exists() assert not list(tmp_path.rglob("injected")) assert not list(tmp_path.rglob("outside.json")) @@ -354,7 +356,8 @@ def author(request: Mapping[str, Any]) -> ScriptedExecToolAction: transport = ScriptedDoubaoExecTransport([ ScriptedExecToolAction(fixture.quota_guard_command), ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), - author, projected_refresh, projected_spend, + author, projected_refresh, + projected_spend if passed else ScriptedAssistantAction("No observed source reference supplied."), ]) receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( qualification_id="observed-evidence-reference", fixture_root=tmp_path / "actor", required_vision=True, @@ -363,7 +366,7 @@ def author(request: Mapping[str, Any]) -> ScriptedExecToolAction: if passed: assert receipt["vision_closeout"]["spend_count"] == 1 else: - assert receipt["failure_code"] == "vision_closeout_evidence_not_observed" + assert receipt["tool_call_receipts"][-1]["error_code"] == "vision_closeout_evidence_not_observed" assert receipt["semantic_action_accepted"] is False @@ -393,6 +396,37 @@ def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: assert receipt["tool_call_count"] == 6 +@pytest.mark.parametrize("rejection", ["unobserved_evidence", "oversized_vision"]) +def test_actor_can_correct_its_own_draft_after_rejection(rejection: str, tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + def ungrounded(request: Mapping[str, Any]) -> ScriptedExecToolAction: + command = heredoc_action(request).command + command = (command.replace("evidence-permission-config", "unobserved-evidence") + if rejection == "unobserved_evidence" else command.replace( + "Reader is default; writing requires an explicit grant.", "x" * 700)) + return ScriptedExecToolAction(command) + def corrected(request: Mapping[str, Any]) -> ScriptedExecToolAction: + error = json.loads(request["messages"][-1]["content"]) + if rejection == "unobserved_evidence": + assert error["error_code"] == "vision_closeout_evidence_not_observed" + assert "fixture/permission-config.json" in error["output"] + else: + assert error["error_code"] == "loopx_cli_nonzero" + assert error["exit_code"] != 0 + return heredoc_action(request) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction(fixture.quota_guard_command), + ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), + ungrounded, projected_refresh, corrected, projected_refresh, projected_spend, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="correct-owned-draft", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is True + assert receipt["vision_closeout"]["spend_count"] == 1 + assert receipt["tool_call_count"] == 7 + + def test_help_host_extension_does_not_change_narrow_actor_contract(tmp_path: Path) -> None: fixture = _build_fixture(tmp_path) assert _bounded_workspace_read_plan("loopx --help", fixture=fixture) is None From a42071737afe2b4a0a964b26ccdf2ee5397bbde9 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:16:49 +0800 Subject: [PATCH 16/22] docs(testing): define recoverable vision draft errors Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- docs/development/testing-and-quality.md | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index 07e9558ec0..9bacc5d881 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -817,8 +817,14 @@ read supplies no evidence. Unadmitted commands likewise return a no-execution error instead of ending the conversation; their grammar limits are not raised. Unsupported programs and rejected file-authoring operations never execute. Pre-execution path, overwrite, JSON or authoring-syntax rejection returns tool -feedback without creating the file or running a suffix. Invalid semantic/binding -claims and errors after an effect remain terminal failures. The host does not add missing +feedback without creating the file or running a suffix, with the rejected +operation and applicable restriction. An actor may revise only its own JSON +drafts; original fixture inputs stay immutable. A rejected evidence reference +returns the already-observed references before refresh executes, so the actor +can correct its draft. Real CLI nonzero results return their exit code and +bounded diagnostic; this does not imply rollback or waive original-Turn recovery +and settlement fences. Identity and durable-verification failures remain +terminal. The host does not add missing semantic fields or execute arbitrary model shell programs. These host semantics are explicit for required-vision qualification; other actors retain their current contracts. Receipts distinguish tool execution errors, accepted writeback and From 597b3ff16a56046408ac64b22c95124caacc72a5 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:18:34 +0800 Subject: [PATCH 17/22] fix(testing): return pre-admission workspace rejection to actor Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- docs/development/testing-and-quality.md | 2 ++ .../replan_semantic_action_behavior.py | 2 +- .../replan_vision_closeout_behavior.py | 7 +++--- .../test_required_vision_closeout_behavior.py | 22 +++++++++++++++++++ 4 files changed, 29 insertions(+), 4 deletions(-) diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index 9bacc5d881..7856a15ce8 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -815,6 +815,8 @@ the entire plan before execution. Ordinary nonzero reads return an exit code and bounded output to the model; recovery consumes the same call budget and a failed read supplies no evidence. Unadmitted commands likewise return a no-execution error instead of ending the conversation; their grammar limits are not raised. +Workspace access before quota admission is also rejected without execution and +returned as tool feedback; it supplies no evidence and consumes a call. Unsupported programs and rejected file-authoring operations never execute. Pre-execution path, overwrite, JSON or authoring-syntax rejection returns tool feedback without creating the file or running a suffix, with the rejected diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index d0603354ae..3cd673d925 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -930,7 +930,7 @@ def _handle_clock_command(output: str, state: _QualificationState) -> str: def _handle_workspace_read(command: str, state: _QualificationState) -> str: if state.fixture.required_vision and not state.seen_quota: - raise ValueError("workspace_read_before_quota") + raise VisionHostAdmissionRejected("workspace_read_before_quota", "Workspace access requires quota admission from the heartbeat's quota guard first. No workspace read ran or supplied evidence.") output, read_frontier, read_work_source, exit_code = _execute_workspace_read( command, fixture=state.fixture, diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py index 7ce81eb7e9..6d24f48c05 100644 --- a/loopx/control_plane/testing/replan_vision_closeout_behavior.py +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -45,7 +45,8 @@ def __init__(self, code: str, detail: str = "") -> None: ) VISION_EXEC_TOOL_DESCRIPTION = ( - "Execute bounded fixture commands, not an unrestricted shell. Workspace reads " + "Execute bounded fixture commands, not an unrestricted shell. Workspace access " + "requires quota admission first; premature access returns an error without execution. Workspace reads " "support pwd, ls (-a/-l), find (maxdepth at most 4), rg --files, cat, head and " "sed -n, optionally joined with &&, || or ; (at most 8 statements). Root " "loopx --help and loopx refresh-state --help are read-only queries and may " @@ -204,7 +205,7 @@ def dispatch_vision_closeout( heredoc = _json_heredoc(command) if heredoc is not None: if not state.seen_quota: - raise ValueError("vision_authoring_before_quota") + raise VisionHostAdmissionRejected("vision_authoring_before_quota", "Workspace authoring requires quota admission first.") name, content, suffix = heredoc result: tuple[str, str, bool] = (_write_json_file(state.fixture.project_root, name, content, authored_paths=state.authored_paths), "vision_file_authoring", False) for following in suffix: @@ -217,7 +218,7 @@ def dispatch_vision_closeout( return result if command.startswith("apply_patch"): if not state.seen_quota: - raise ValueError("vision_authoring_before_quota") + raise VisionHostAdmissionRejected("vision_authoring_before_quota", "Workspace authoring requires quota admission first.") return _authored_file(command, state.fixture.project_root, authored_paths=state.authored_paths), "vision_file_authoring", False tokens = loopx_command_tokens(command) or [] if "refresh-state" not in tokens and "spend-slot" not in tokens: diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py index 30d5862562..bf116eb76e 100644 --- a/tests/control_plane/test_required_vision_closeout_behavior.py +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -271,6 +271,28 @@ def test_unadmitted_commands_never_supply_evidence_or_success(tmp_path: Path) -> assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 16 +@pytest.mark.parametrize("premature", ["read", "author"]) +def test_pre_admission_workspace_access_is_rejected_with_recovery(premature: str, tmp_path: Path) -> None: + fixture = _build_fixture(tmp_path / "oracle", required_vision=True) + def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: + error = json.loads(request["messages"][-1]["content"]) + assert error["error_code"] == ("workspace_read_before_quota" if premature == "read" else "vision_authoring_before_quota") + assert "quota admission" in error["output"] + assert not (tmp_path / "actor/project/decision.json").exists() + return ScriptedExecToolAction(fixture.quota_guard_command) + transport = ScriptedDoubaoExecTransport([ + ScriptedExecToolAction("cat fixture/permission-config.json") if premature == "read" else heredoc_action, + recover, ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), + heredoc_action, projected_refresh, projected_spend, + ]) + receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="pre-admission-recovery", fixture_root=tmp_path / "actor", required_vision=True, + ) + assert receipt["qualification_passed"] is True + assert receipt["tool_call_count"] == 6 + assert receipt["vision_closeout"]["spend_count"] == 1 + + @pytest.mark.parametrize("extra_reads,passed", [(11, True), (12, False)]) def test_full_closeout_budget_boundary_never_waives_settlement(extra_reads: int, passed: bool, tmp_path: Path) -> None: fixture = _build_fixture(tmp_path / "oracle", required_vision=True) From 67a9e6f44319a959ec5b44c063eff93955207b33 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:23:36 +0800 Subject: [PATCH 18/22] fix(testing): budget complete closeout with validation recovery Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- docs/development/testing-and-quality.md | 7 ++++--- .../testing/replan_vision_closeout_behavior.py | 2 +- .../test_required_vision_closeout_behavior.py | 17 +++++++++-------- 3 files changed, 14 insertions(+), 12 deletions(-) diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index 7856a15ce8..7b893a41d4 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -800,11 +800,12 @@ and qualification scope; a successful ordinary `typed_progress_repeat` refresh cannot qualify this journey. The narrow semantic-action gate remains useful but does not prove full closeout. Run this focused journey with `uv run --extra test python scripts/qualify-doubao-replan-semantic-action-live.py --required-vision --qualification-id `. -The complete required-vision journey has a 16-call bound; the narrow +The complete required-vision journey has a 32-call bound; the narrow single-semantic-action qualifier retains seven. The increased budget covers -evidence discovery, JSON authoring, refresh, settlement and bounded recovery; +evidence discovery, JSON authoring, refresh, settlement and bounded recovery, +including multiple field-validation corrections before a final spend; it does not authorize more effects or weaken the checkpoint/readback oracle. -Each receipt records `tool_call_limit` beside actual usage. Earlier seven-call +Each receipt records `tool_call_limit` beside actual usage. Earlier seven- and sixteen-call failures remain failures; the revised budget defines a new qualification, not a retrospective pass. Limit exhaustion after refresh but before spend still fails. The tool description declares the actual diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py index 6d24f48c05..cb996bf7e7 100644 --- a/loopx/control_plane/testing/replan_vision_closeout_behavior.py +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -22,7 +22,7 @@ # Full closeout includes evidence discovery, authoring, refresh and settlement; # its resource bound is independent of the narrower single-action qualifier. -REQUIRED_VISION_CLOSEOUT_MAX_CALLS = 16 +REQUIRED_VISION_CLOSEOUT_MAX_CALLS = 32 class VisionHostAdmissionRejected(ValueError): diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py index bf116eb76e..928848f0a2 100644 --- a/tests/control_plane/test_required_vision_closeout_behavior.py +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -182,13 +182,14 @@ def test_host_composes_bounded_reads_and_real_cli_help(tmp_path: Path) -> None: @pytest.mark.parametrize("command", ["loopx --help", "cat replan-frontier.json && loopx refresh-state --help"]) def test_read_and_help_extension_preserves_quota_first(command: str, tmp_path: Path) -> None: - transport = ScriptedDoubaoExecTransport([ScriptedExecToolAction(command)]) + transport = ScriptedDoubaoExecTransport([ScriptedExecToolAction(command), ScriptedAssistantAction("Stopped before quota admission.")]) receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( qualification_id="read-before-quota", fixture_root=tmp_path, required_vision=True, ) assert receipt["qualification_passed"] is False - assert receipt["failure_code"] == "workspace_read_before_quota" + assert receipt["tool_call_receipts"][0]["error_code"] == "workspace_read_before_quota" assert receipt["semantic_action_accepted"] is False + assert receipt["boundary"]["read_only_host_commands_executed"] is False def test_read_error_reaches_model_and_recovery_still_requires_full_closeout(tmp_path: Path) -> None: @@ -218,14 +219,14 @@ def test_read_failures_consume_the_full_closeout_call_budget(tmp_path: Path) -> fixture = _build_fixture(tmp_path / "oracle", required_vision=True) transport = ScriptedDoubaoExecTransport([ ScriptedExecToolAction(fixture.quota_guard_command), - *[ScriptedExecToolAction("cat missing.json") for _ in range(15)], + *[ScriptedExecToolAction("cat missing.json") for _ in range(31)], ]) receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( qualification_id="read-errors-exhaust-budget", fixture_root=tmp_path / "actor", required_vision=True, ) assert receipt["qualification_passed"] is False assert receipt["failure_code"] == "tool_call_budget_exhausted" - assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 16 + assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 32 assert receipt["semantic_action_accepted"] is False @@ -260,7 +261,7 @@ def test_unadmitted_commands_never_supply_evidence_or_success(tmp_path: Path) -> fixture = _build_fixture(tmp_path / "oracle", required_vision=True) transport = ScriptedDoubaoExecTransport([ ScriptedExecToolAction(fixture.quota_guard_command), - *[ScriptedExecToolAction("find . -maxdepth 5 -type f") for _ in range(15)], + *[ScriptedExecToolAction("find . -maxdepth 5 -type f") for _ in range(31)], ]) receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( qualification_id="unadmitted-command-budget", fixture_root=tmp_path / "actor", required_vision=True, @@ -268,7 +269,7 @@ def test_unadmitted_commands_never_supply_evidence_or_success(tmp_path: Path) -> assert receipt["qualification_passed"] is False assert receipt["failure_code"] == "tool_call_budget_exhausted" assert receipt["semantic_action_accepted"] is False - assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 16 + assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 32 @pytest.mark.parametrize("premature", ["read", "author"]) @@ -293,7 +294,7 @@ def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: assert receipt["vision_closeout"]["spend_count"] == 1 -@pytest.mark.parametrize("extra_reads,passed", [(11, True), (12, False)]) +@pytest.mark.parametrize("extra_reads,passed", [(27, True), (28, False)]) def test_full_closeout_budget_boundary_never_waives_settlement(extra_reads: int, passed: bool, tmp_path: Path) -> None: fixture = _build_fixture(tmp_path / "oracle", required_vision=True) transport = ScriptedDoubaoExecTransport([ @@ -306,7 +307,7 @@ def test_full_closeout_budget_boundary_never_waives_settlement(extra_reads: int, qualification_id="closeout-budget-boundary", fixture_root=tmp_path / "actor", required_vision=True, ) assert receipt["qualification_passed"] is passed - assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 16 + assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 32 assert receipt["semantic_action_accepted"] is True assert receipt["vision_closeout"]["settled"] is passed if passed: From edf5566be6fdeb1a8740b35041298792da86e497 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:38:44 +0800 Subject: [PATCH 19/22] refactor(testing): qualify vision through an isolated native shell Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../testing/model_tool_behavior.py | 2 +- .../replan_semantic_action_behavior.py | 132 ++--- .../replan_vision_closeout_behavior.py | 158 +----- .../testing/selected_todo_tool_behavior.py | 50 +- .../testing/vision_shell_host.py | 145 +++++ .../test_required_vision_closeout_behavior.py | 511 ++++-------------- 6 files changed, 354 insertions(+), 644 deletions(-) create mode 100644 loopx/control_plane/testing/vision_shell_host.py diff --git a/loopx/control_plane/testing/model_tool_behavior.py b/loopx/control_plane/testing/model_tool_behavior.py index 5bc68e5aa2..77f8668177 100644 --- a/loopx/control_plane/testing/model_tool_behavior.py +++ b/loopx/control_plane/testing/model_tool_behavior.py @@ -534,7 +534,7 @@ def execute_loopx_cli( else os.pathsep.join((str(source_root), existing_pythonpath)) ) completed = subprocess.run( - [sys.executable, "-m", "loopx.cli", *argv], + [sys.executable, "-P", "-m", "loopx.cli", *argv], cwd=project_root, env=env, check=False, diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index 3cd673d925..fd41b9ca64 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -5,7 +5,7 @@ import shlex import subprocess from collections.abc import Mapping -from dataclasses import dataclass, field as dataclass_field +from dataclasses import dataclass from hashlib import sha256 from pathlib import Path from typing import Any @@ -36,8 +36,9 @@ from .replan_vision_closeout_behavior import ( dispatch_vision_closeout, VISION_HOST_INSTRUCTION, VISION_EXEC_TOOL_DESCRIPTION, REQUIRED_VISION_CLOSEOUT_MAX_CALLS, - VisionHostAdmissionRejected, + observe_shell_evidence, ) +from .vision_shell_host import VisionShellHost REPLAN_SEMANTIC_ACTION_BEHAVIOR_RECEIPT_SCHEMA_VERSION = ( "replan_semantic_action_behavior_receipt_v0" @@ -100,7 +101,6 @@ class _QualificationState: successor_reentry_observation: dict[str, Any] | None = None semantic_reentry_observation: dict[str, Any] | None = None vision_closeout: dict[str, Any] | None = None - authored_paths: set[Path] = dataclass_field(default_factory=set) @property def tool_call_limit(self) -> int: @@ -716,7 +716,7 @@ def _bounded_workspace_read_plan( *, fixture: _ReplanSemanticActionFixture, ) -> list[Any] | None: - return bounded_workspace_read_plan(command, fixture=fixture, allow_loopx_help=fixture.required_vision) + return bounded_workspace_read_plan(command, fixture=fixture) def _bounded_refresh_state_help_limit(command: str) -> int | None: @@ -754,7 +754,7 @@ def _execute_workspace_read( command: str, *, fixture: _ReplanSemanticActionFixture, -) -> tuple[str, bool, bool, int]: +) -> tuple[str, bool, bool]: plan = _bounded_workspace_read_plan(command, fixture=fixture) if plan is None: raise ValueError("workspace read is outside the hermetic fixture") @@ -770,21 +770,16 @@ def _execute_workspace_read( ) if not should_run: continue - if step.kind == "loopx_help": - output = execute_loopx_cli( - shlex.join(step.argv), source_root=fixture.source_root, - project_root=fixture.project_root, timeout_seconds=10, - ) - status = 0 - else: - completed = subprocess.run( - list(step.argv), cwd=fixture.project_root, check=False, - capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=10, - ) - status = completed.returncode - output = completed.stdout - if fixture.required_vision and not step.suppress_stderr: - output += completed.stderr + completed = subprocess.run( + list(step.argv), + cwd=fixture.project_root, + check=False, + capture_output=True, + text=True, encoding="utf-8", errors="replace", + timeout=10, + ) + status = completed.returncode + output = completed.stdout if step.line_limit is not None: output = "".join(output.splitlines(keepends=True)[: step.line_limit]) outputs.append(output) @@ -798,9 +793,9 @@ def _execute_workspace_read( and step.target is not None and step.target.resolve() == fixture.work_source_target.resolve() ) - if status != 0 and not fixture.required_vision: + if status != 0: raise RuntimeError(f"workspace read failed with exit={status}") - return "".join(outputs), frontier_read, work_source_read, status + return "".join(outputs), frontier_read, work_source_read def _receipt( @@ -876,6 +871,9 @@ def _qualification_receipt( "required_vision_closeout" if state.fixture.required_vision else "semantic_action" ) receipt["tool_call_limit"] = state.tool_call_limit + if state.fixture.required_vision: + receipt["execution_host"] = "os_isolated_shell" + receipt["boundary"]["shell_commands_executed"] = bool(state.steps) if state.vision_closeout is not None: receipt["vision_closeout"] = state.vision_closeout receipt["semantic_action_accepted"] = bool(state.semantic_delta and state.semantic_delta.get("accepted")) @@ -929,17 +927,13 @@ def _handle_clock_command(output: str, state: _QualificationState) -> str: def _handle_workspace_read(command: str, state: _QualificationState) -> str: - if state.fixture.required_vision and not state.seen_quota: - raise VisionHostAdmissionRejected("workspace_read_before_quota", "Workspace access requires quota admission from the heartbeat's quota guard first. No workspace read ran or supplied evidence.") - output, read_frontier, read_work_source, exit_code = _execute_workspace_read( + output, read_frontier, read_work_source = _execute_workspace_read( command, fixture=state.fixture, ) state.frontier_context_read = state.frontier_context_read or read_frontier state.work_source_read = state.work_source_read or read_work_source state.read_only_host_commands_executed = True - if exit_code: - raise _HostToolError("workspace_read_nonzero", output, exit_code) return output @@ -1135,10 +1129,6 @@ def _dispatch_behavior_command( "refresh_state_help", False, ) - if state.fixture.required_vision: - result = dispatch_vision_closeout(command, state, execute=_execute_loopx) - if result is not None: - return result if _is_replan_successor_create(command): return _handle_successor_command(command, state), "replan_successor_create", True if (observation := _progress_observation_from_command(command)) is not None: @@ -1149,15 +1139,6 @@ def _dispatch_behavior_command( ) if "evidence-log" in command: raise ValueError("manual_evidence_read_is_not_replan") - if state.fixture.required_vision: - # No handler admitted this command, so no part of it was executed. - # Rejection is not semantic success, nor a reason to hide tool feedback. - raise _HostToolError( - "unsupported_host_command", - "Command not executed: outside the bounded tool grammar. " - "Use the operations and limits in the exec_command description.", - 2, - ) raise ValueError("unexpected_command") @@ -1178,10 +1159,6 @@ def _behavior_command_kind( if _is_replan_successor_create(command): return "replan_successor_create" tokens = loopx_command_tokens(command) or [] - if command.startswith("apply_patch"): - return "vision_file_authoring" - if "spend-slot" in tokens: - return "quota_spend_slot" if "refresh-state" in tokens: return "semantic_replan_writeback" return "unexpected_command" @@ -1189,23 +1166,6 @@ def _behavior_command_kind( _EXPECTED_BEHAVIOR_FAILURES = frozenset( { - "workspace_read_before_quota", - "required_vision_fixture_trigger_mismatch", - "vision_authoring_requires_single_add_file_patch", - "vision_authoring_path_outside_fixture", - "vision_authoring_requires_added_lines", - "vision_authoring_requires_json_object", - "vision_authoring_requires_literal_json_heredoc", - "vision_authoring_suffix_requires_loopx", - "vision_authoring_before_quota", - "vision_closeout_binding_mismatch", - "vision_closeout_turn_mismatch", - "vision_closeout_requires_observed_source_and_authored_decision", - "vision_closeout_evidence_not_observed", - "vision_closeout_durable_writeback_incomplete", - "vision_closeout_spend_before_writeback", - "vision_closeout_settlement_readback_failed", - "vision_closeout_rearmed_after_settlement", "repeated_quota_should_run", "repeated_clock", "semantic_action_before_quota", @@ -1275,6 +1235,7 @@ def _run_qualification_loop( state: _QualificationState, *, qualification_id: str, + shell: VisionShellHost | None = None, ) -> dict[str, Any]: for _ in range(state.tool_call_limit): tool_call = client.next_tool_call( @@ -1289,21 +1250,18 @@ def _run_qualification_loop( passed=False, failure_code="model_returned_without_semantic_action", ) - kind = _behavior_command_kind(tool_call.command, state) + kind = "shell" if shell else _behavior_command_kind(tool_call.command, state) try: - output, dispatched_kind, completed = _dispatch_behavior_command( - tool_call.command, - state, - ) - kind = dispatched_kind - except (VisionHostAdmissionRejected, _HostToolError) as exc: - if isinstance(exc, VisionHostAdmissionRejected): - exc = _HostToolError( - str(exc), "Rejected operation not executed: " + str(exc) + ". " + exc.detail + " " - "Earlier successful steps, if any, are not rolled back. " - "Use literal LoopX arguments, relative JSON drafts and the declared grammar; " - "other operations can be issued as separate tool calls.", 2, - ) + if shell: + output, exit_code = shell.execute(tool_call.command) + observe_shell_evidence(output, state) + state.read_only_host_commands_executed = True + if exit_code: + raise _HostToolError("shell_nonzero", output, exit_code) + completed = bool((state.vision_closeout or {}).get("settled")) + else: + output, kind, completed = _dispatch_behavior_command(tool_call.command, state) + except _HostToolError as exc: _record_tool_step(state, kind=kind, command=tool_call.command, exit_code=exc.exit_code, error_code=exc.code) _append_tool_response(state, tool_call=tool_call, output=json.dumps({ @@ -1424,6 +1382,30 @@ def qualify( steps=[], turn_instance_id=f"qualification-{digest}", ) + if required_vision: + def invoke(argv: list[str], cwd: Path, output: str) -> str: + observe_shell_evidence(output, state) + tokens = ["loopx", *argv] + if "--agent-vision-json" in tokens: + index = tokens.index("--agent-vision-json") + 1 + if index < len(tokens): + tokens[index] = str((cwd / tokens[index]).resolve()) + command = shlex.join(tokens) + if "--help" in argv or "-h" in argv: + return _execute_loopx(command, fixture=fixture, turn_instance_id=state.turn_instance_id) + if _is_quota_guard(command): + return (_execute_loopx(command, fixture=fixture, turn_instance_id=state.turn_instance_id) + if state.seen_quota else _handle_quota_command(command, state)) + result = dispatch_vision_closeout(command, state, execute=_execute_loopx) + if result is None: + raise ValueError("This fixture grants CLI authority for quota, help, vision refresh and settlement only") + return result[0] + + shell = VisionShellHost(fixture.project_root, invoke, turn_instance_id=state.turn_instance_id) + try: + return _run_qualification_loop(self._client, state, qualification_id=qualification_id, shell=shell) + finally: + shell.close() return _run_qualification_loop( self._client, state, diff --git a/loopx/control_plane/testing/replan_vision_closeout_behavior.py b/loopx/control_plane/testing/replan_vision_closeout_behavior.py index cb996bf7e7..7dc0abe50f 100644 --- a/loopx/control_plane/testing/replan_vision_closeout_behavior.py +++ b/loopx/control_plane/testing/replan_vision_closeout_behavior.py @@ -7,10 +7,7 @@ from __future__ import annotations import json -import re -import shlex from collections.abc import Mapping -from pathlib import Path from typing import TYPE_CHECKING, Any, Callable from .model_tool_behavior import argument_value, loopx_command_tokens @@ -34,34 +31,18 @@ def __init__(self, code: str, detail: str = "") -> None: VISION_HOST_INSTRUCTION = ( - "You are Codex operating a hermetic LoopX project. Follow the heartbeat and " - "the live control-plane packet, including its semantic writeback and settlement. " - "The shell cwd is the connected project. Available operations are bounded " - "workspace reads, LoopX CLI commands, and new relative JSON file authoring " - "via apply_patch Add File or cat with a quoted heredoc. A cat write may be " - "followed by bound LoopX commands separated by newlines or &&. No external " - "network or writes outside the fixture are available. Choose the decision " - "from the evidence; file creation alone is not delivery." + "You are an agent working in an isolated LoopX project. Follow the heartbeat " + "and current control-plane packet. Use the shell normally to inspect evidence, " + "author your decision and execute the real CLI writeback and settlement." ) VISION_EXEC_TOOL_DESCRIPTION = ( - "Execute bounded fixture commands, not an unrestricted shell. Workspace access " - "requires quota admission first; premature access returns an error without execution. Workspace reads " - "support pwd, ls (-a/-l), find (maxdepth at most 4), rg --files, cat, head and " - "sed -n, optionally joined with &&, || or ; (at most 8 statements). Root " - "loopx --help and loopx refresh-state --help are read-only queries and may " - "be combined with reads or piped to head (at most 200 lines). Author a new " - "relative JSON file via apply_patch Add File or cat with a quoted heredoc; " - "these operations may revise your own JSON drafts, never original fixture files. " - "a heredoc may be followed by up to two bound LoopX closeout commands using " - "newlines or &&. Other LoopX commands must be separate single invocations. " - "Returns stdout on success; nonzero workspace reads and unsupported command " - "shapes return exit_code and output for correction within the same call " - "budget. Rejected file-authoring operations also return errors without " - "creating a file or executing any suffix. CLI validation errors return " - "their diagnostics; identity and durable-verification errors fail " - "qualification. Unsupported programs never run. No external network " - "or outside-fixture writes." + "Run a shell command in the project. Shell variables, pipelines, compound " + "commands, Python and JSON editing are available. Returns stdout/stderr and " + "exit_code, including errors so you can correct and retry. Project drafts and " + "$TMPDIR are writable; fixture inputs and authority stores are protected. " + "The loopx command uses the real checkout CLI for this isolated task. " + "External networking and access to private host data are unavailable." ) @@ -107,90 +88,20 @@ def required_vision_scenario_contract( } -def _authored_file(command: str, project: Path, *, authored_paths: set[Path] | None = None) -> str: - # Accept data, never execute the model's shell program. JSON quotes and - # shell metacharacters inside added lines remain literal file content. - match = re.fullmatch( - r"apply_patch\s+<<\s*'([A-Za-z_][A-Za-z0-9_]*)'\n" - r"\*\*\* Begin Patch\n\*\*\* Add File: ([^\n]+)\n" - r"(.*?)\n\*\*\* End Patch\n\1", command.strip(), re.DOTALL, - ) - if not match: - raise VisionHostAdmissionRejected("vision_authoring_requires_single_add_file_patch") - lines = match[3].splitlines() - if not lines or any(not line.startswith("+") for line in lines): - raise VisionHostAdmissionRejected("vision_authoring_requires_added_lines") - return _write_json_file(project, match[2], "\n".join(line[1:] for line in lines) + "\n", authored_paths=authored_paths) - - -def _write_json_file(project: Path, name: str, content: str, *, authored_paths: set[Path] | None = None) -> str: - relative = Path(name) - if relative.is_absolute() or relative.suffix != ".json" or ".." in relative.parts: - raise VisionHostAdmissionRejected("vision_authoring_path_outside_fixture", "The JSON path must be relative to the project, without parent traversal; absolute temporary paths are not writable.") - target = (project / relative).resolve() - if not target.is_relative_to(project.resolve()) or (target.exists() and target not in (authored_paths or set())): - raise VisionHostAdmissionRejected("vision_authoring_path_outside_fixture", "Only new JSON files or this actor's own drafts are writable; existing fixture inputs are immutable.") - try: - value = json.loads(content) - except json.JSONDecodeError: - raise VisionHostAdmissionRejected("vision_authoring_requires_json_object") from None - if not isinstance(value, dict): - raise VisionHostAdmissionRejected("vision_authoring_requires_json_object") - target.parent.mkdir(parents=True, exist_ok=True) - target.write_text(content, encoding="utf-8") - if authored_paths is not None: - authored_paths.add(target) - return json.dumps({"ok": True, "path": relative.as_posix()}) - - -def _json_heredoc(command: str) -> tuple[str, str, list[str]] | None: - """Decode inert JSON plus a bounded CLI suffix, never a shell program.""" - header, newline, rest = command.partition("\n") - if not newline or not header.lstrip().startswith("cat "): - return None - quoted = re.search(r"<<\s*(['\"])([A-Za-z_][A-Za-z0-9_]*)\1", header) - if quoted is None: - return None - lexer = shlex.shlex(header, posix=True, punctuation_chars="<>&;|") - lexer.whitespace_split = True - try: - tokens = list(lexer) - except ValueError: - raise VisionHostAdmissionRejected("vision_authoring_requires_literal_json_heredoc") from None - if len(tokens) != 5 or tokens[0] != "cat" or set(tokens[1::2]) != {">", "<<"}: - raise VisionHostAdmissionRejected("vision_authoring_requires_literal_json_heredoc") - name = tokens[tokens.index(">") + 1] - delimiter = tokens[tokens.index("<<") + 1] - if delimiter != quoted[2]: - raise VisionHostAdmissionRejected("vision_authoring_requires_literal_json_heredoc") - lines = rest.splitlines(keepends=True) - end = next((index for index, line in enumerate(lines) if line.rstrip("\r\n") == delimiter), None) - if end is None: - raise VisionHostAdmissionRejected("vision_authoring_requires_literal_json_heredoc") - suffix = "".join(lines[end + 1:]).strip() - lexer = shlex.shlex(suffix, posix=True, punctuation_chars=";&|\n") - lexer.whitespace = " \t\r" - lexer.whitespace_split = True - commands: list[str] = [] - argv: list[str] = [] - try: - suffix_tokens = list(lexer) - except ValueError: - raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx") from None - for token in [*suffix_tokens, "\n"]: - if token == "&&" or token.strip("\n") == "": - if argv: - if Path(argv[0]).name != "loopx": - raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx", f"The command after the heredoc delimiter starts with {argv[0]!r}; only literal LoopX closeout commands are admitted there. Submit the heredoc alone; other operations need separate tool calls.") - commands.append(shlex.join(argv)) - argv = [] - elif token in {";", "|", "||", "&"}: - raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx", f"The heredoc suffix contains unsupported operator {token!r}; use a standalone heredoc or literal LoopX commands separated by newlines or &&.") - else: - argv.append(token) - if len(commands) > 2: - raise VisionHostAdmissionRejected("vision_authoring_suffix_requires_loopx", "A heredoc permits at most two literal LoopX closeout commands; issue other operations in separate tool calls.") - return name, "".join(lines[:end]), commands +def observe_shell_evidence(output: str, state: _QualificationState) -> None: + """Observe returned source data, independent of the command used to read it.""" + expected = json.loads(state.fixture.work_source_target.read_text(encoding="utf-8")) + decoder = json.JSONDecoder() + for index, char in enumerate(output): + if char != "{": + continue + try: + value, _ = decoder.raw_decode(output[index:]) + except json.JSONDecodeError: + continue + if value == expected: + state.work_source_read = True + return def _rows(state: _QualificationState) -> list[dict[str, Any]]: @@ -202,32 +113,9 @@ def _rows(state: _QualificationState) -> list[dict[str, Any]]: def dispatch_vision_closeout( command: str, state: _QualificationState, *, execute: Callable[..., str], ) -> tuple[str, str, bool] | None: - heredoc = _json_heredoc(command) - if heredoc is not None: - if not state.seen_quota: - raise VisionHostAdmissionRejected("vision_authoring_before_quota", "Workspace authoring requires quota admission first.") - name, content, suffix = heredoc - result: tuple[str, str, bool] = (_write_json_file(state.fixture.project_root, name, content, authored_paths=state.authored_paths), "vision_file_authoring", False) - for following in suffix: - if result[2]: - raise ValueError("vision_authoring_suffix_requires_loopx") - next_result = dispatch_vision_closeout(following, state, execute=execute) - if next_result is None: - raise ValueError("vision_authoring_suffix_requires_loopx") - result = next_result - return result - if command.startswith("apply_patch"): - if not state.seen_quota: - raise VisionHostAdmissionRejected("vision_authoring_before_quota", "Workspace authoring requires quota admission first.") - return _authored_file(command, state.fixture.project_root, authored_paths=state.authored_paths), "vision_file_authoring", False tokens = loopx_command_tokens(command) or [] if "refresh-state" not in tokens and "spend-slot" not in tokens: return None - lexer = shlex.shlex(command.strip(), posix=True, punctuation_chars=";&|\n") - lexer.whitespace = " \t\r" - lexer.whitespace_split = True - if tokens != list(lexer): - raise VisionHostAdmissionRejected("vision_command_requires_literal_loopx_argv") packet = state.quota_packet or {} binding = dict(dict(dict(packet.get("interaction_contract") or {}).get("cli_channel") or {}).get("replan_settlement_contract") or {}).get("settlement_binding") or {} if not binding or argument_value(tokens, binding["cli_argument"]) != binding["id"]: diff --git a/loopx/control_plane/testing/selected_todo_tool_behavior.py b/loopx/control_plane/testing/selected_todo_tool_behavior.py index 431fe0c957..0545425f39 100644 --- a/loopx/control_plane/testing/selected_todo_tool_behavior.py +++ b/loopx/control_plane/testing/selected_todo_tool_behavior.py @@ -5,7 +5,7 @@ import shlex import subprocess from collections.abc import Mapping -from dataclasses import dataclass, replace +from dataclasses import dataclass from hashlib import sha256 from pathlib import Path from typing import Any @@ -76,7 +76,6 @@ class BoundedWorkspaceReadStep: target: Path | None = None operator: str = ";" line_limit: int | None = None - suppress_stderr: bool = False def _build_fixture( @@ -410,22 +409,11 @@ def _is_fixture_metadata_target( return target.resolve() in allowed -def _loopx_help_argv(tokens: list[str]) -> tuple[str, ...] | None: - if tokens[-3:] == ["2", ">&", "1"]: - tokens = tokens[:-3] - if tokens and Path(tokens[0]).name == "loopx" and tokens[1:] in ( - ["--help"], ["refresh-state", "--help"], - ): - return ("loopx", *tokens[1:]) - return None - - def _metadata_pipeline_step( segment: list[str], *, fixture: _SelectedTodoToolFixture, operator: str, - allow_loopx_help: bool = False, ) -> BoundedWorkspaceReadStep | None: pipe_indices = [index for index, token in enumerate(segment) if token == "|"] if len(pipe_indices) not in {1, 2}: @@ -460,8 +448,6 @@ def _metadata_pipeline_step( return None if limit < 1 or limit > 200: return None - if allow_loopx_help and len(pipe_indices) == 1 and (help_argv := _loopx_help_argv(left)): - return BoundedWorkspaceReadStep("loopx_help", help_argv, operator=operator, line_limit=limit) discovery_argv = _discovery_tokens(left, fixture=fixture) if discovery_argv is not None: return BoundedWorkspaceReadStep( @@ -494,8 +480,7 @@ def _metadata_pipeline_step( ) -def _read_plan_segments(command: str) -> list[tuple[str, list[str]]] | None: - """Decode the bounded read grammar before admitting any executable step.""" +def _read_plan_tokens(command: str) -> list[str] | None: lexer = shlex.shlex(command, posix=True, punctuation_chars=";&|<>\n") lexer.whitespace = " \t\r" lexer.whitespace_split = True @@ -503,6 +488,15 @@ def _read_plan_segments(command: str) -> list[tuple[str, list[str]]] | None: tokens = list(lexer) except ValueError: return None + return tokens or None + + +def bounded_workspace_read_plan( + command: str, + *, + fixture: _SelectedTodoToolFixture, +) -> list[BoundedWorkspaceReadStep] | None: + tokens = _read_plan_tokens(command) if not tokens: return None segments: list[list[str]] = [[]] @@ -521,28 +515,15 @@ def _read_plan_segments(command: str) -> list[tuple[str, list[str]]] | None: segments[-1].append(token) if not segments[-1] or len(segments) > 8: return None - return list(zip(operators, segments, strict=True)) - - -def bounded_workspace_read_plan( - command: str, - *, - fixture: _SelectedTodoToolFixture, - allow_loopx_help: bool = False, -) -> list[BoundedWorkspaceReadStep] | None: - segments = _read_plan_segments(command) - if segments is None: - return None plan: list[BoundedWorkspaceReadStep] = [] - for operator, raw_segment in segments: + for operator, raw_segment in zip(operators, segments, strict=True): segment = list(raw_segment) if "|" in segment: pipeline_step = _metadata_pipeline_step( segment, fixture=fixture, operator=operator, - allow_loopx_help=allow_loopx_help, ) if pipeline_step is None: return None @@ -550,9 +531,6 @@ def bounded_workspace_read_plan( continue if len(segment) >= 3 and segment[-3:] == ["2", ">", "/dev/null"]: segment = segment[:-3] - if allow_loopx_help and (help_argv := _loopx_help_argv(segment)): - plan.append(BoundedWorkspaceReadStep("loopx_help", help_argv, operator=operator, line_limit=200)) - continue if not segment or any( token in {"&", "|", "<", ">", "<<", ">>", "<<<"} for token in segment @@ -648,10 +626,6 @@ def bounded_workspace_read_plan( operator, ) ) - for index, (_, segment) in enumerate(segments): - left = segment[:segment.index("|")] if "|" in segment else segment - if left[-3:] == ["2", ">", "/dev/null"]: - plan[index] = replace(plan[index], suppress_stderr=True) return plan diff --git a/loopx/control_plane/testing/vision_shell_host.py b/loopx/control_plane/testing/vision_shell_host.py new file mode 100644 index 0000000000..316c4e3bbf --- /dev/null +++ b/loopx/control_plane/testing/vision_shell_host.py @@ -0,0 +1,145 @@ +"""OS-isolated shell for the required-vision actor; CLI effects stay supervised. + +The shell owns normal command syntax, not a qualification-specific parser. +Only the existing LoopX executor can write the fixture's authority stores. +""" +from __future__ import annotations + +import json +import os +import shutil +import signal +import socketserver +import subprocess +import sys +import tempfile +import threading +from collections.abc import Callable +from pathlib import Path +from typing import Any + + +_CLI_CLIENT = """import json, os, socket, sys +with socket.socket(socket.AF_UNIX) as connection: + connection.connect(os.environ['LOOPX_FIXTURE_SOCKET']) + connection.sendall(json.dumps({'argv': sys.argv[1:], 'cwd': os.getcwd()}).encode()) + connection.shutdown(socket.SHUT_WR) + with connection.makefile('rb') as stream: + result = json.load(stream) +sys.stdout.write(result['output']) +sys.exit(result['exit_code']) +""" + + +def shell_isolation_available() -> bool: + return (sys.platform == "darwin" and Path("/usr/bin/sandbox-exec").is_file()) or bool(shutil.which("bwrap")) + + +class VisionShellHost: + """One fixture, a normal shell, and an argv-only real-CLI bridge.""" + + def __init__(self, project: Path, invoke: Callable[[list[str], Path, str], str], *, turn_instance_id: str) -> None: + if not shell_isolation_available(): + raise RuntimeError("Native qualification needs sandbox-exec (macOS) or bubblewrap (Linux); unrestricted execution is not a fallback") + self.project = project.resolve() + self.fixture_root = self.project.parent + self.invoke = invoke + self.turn_instance_id = turn_instance_id + # Keep the socket short enough for AF_UNIX, and outside writable actor scope. + self.control = tempfile.TemporaryDirectory(prefix="lx-shell-", dir="/tmp") + self.control_path = Path(self.control.name).resolve() + self.socket_path = self.control_path / "cli.sock" + self.scratch = self.project / ".scratch" + self.originals = [path.resolve() for path in self.project.iterdir()] + self.scratch.mkdir() + client = self.control_path / "loopx" + client.write_text(f"#!{sys.executable}\n" + _CLI_CLIENT, encoding="utf-8") + client.chmod(0o700) + self._output: Any = None + self.cli_calls: list[dict[str, Any]] = [] + self._invocation_lock = threading.Lock() + self._active = False + host = self + + class Handler(socketserver.StreamRequestHandler): + def handle(self) -> None: + try: + request = json.load(self.rfile) + argv, cwd = request["argv"], Path(request["cwd"]).resolve() + if not isinstance(argv, list) or not all(isinstance(arg, str) for arg in argv) or not cwd.is_relative_to(host.project): + raise ValueError("CLI request must use literal argv from the isolated project") + with host._invocation_lock: + if not host._active: + raise ValueError("The shell command has ended; no later CLI effect is admitted") + output = host.invoke(argv, cwd, host.output()) + result = {"output": output, "exit_code": 0} + except (ValueError, RuntimeError, OSError) as exc: + detail = getattr(exc, "output", str(exc)) + if getattr(exc, "detail", ""): + detail += ": " + exc.detail + result = {"output": detail, "exit_code": getattr(exc, "returncode", getattr(exc, "exit_code", 1))} + host.cli_calls.append({"exit_code": result["exit_code"]}) + self.wfile.write(json.dumps(result).encode()) + + self.server = socketserver.UnixStreamServer(str(self.socket_path), Handler) + self.thread = threading.Thread(target=self.server.serve_forever, daemon=True) + self.thread.start() + + def _sandbox(self) -> list[str]: + readable = [Path(path).resolve() for path in ("/System", "/Library/Apple", "/usr", "/bin", "/sbin", "/lib", "/lib64", "/opt/homebrew/Cellar", sys.base_prefix, sys.prefix) if Path(path).exists()] + readable += [self.fixture_root, self.control_path] + if sys.platform == "darwin": + clauses = ["(version 1)", "(allow default)", "(deny file-read-data)", "(deny file-write*)", "(deny network*)", "(deny signal)", "(deny process-info*)", "(allow process-info* (target self))"] + clauses += [f"(allow file-read-data (subpath {json.dumps(str(path))}))" for path in readable] + clauses += ["(allow file-read-data (literal \"/\") (subpath \"/dev\"))", "(allow file-write-data (literal \"/dev/null\"))"] + clauses += [f"(allow file-write* (subpath {json.dumps(str(self.project))}))"] + clauses += [f"(deny file-write* (subpath {json.dumps(str(path))}))" for path in self.originals] + clauses += [f"(allow network* (remote unix-socket (literal {json.dumps(str(self.socket_path))})))"] + return ["/usr/bin/sandbox-exec", "-p", "\n".join(clauses)] + argv = [str(shutil.which("bwrap")), "--unshare-all", "--die-with-parent", "--new-session", "--proc", "/proc", "--dev", "/dev"] + for path in dict.fromkeys(readable): + argv += ["--ro-bind", str(path), str(path)] + argv += ["--bind", str(self.project), str(self.project)] + for path in self.originals: + argv += ["--ro-bind", str(path), str(path)] + return argv + + def output(self) -> str: + # pread does not move the shared offset used by the shell's stdout. + if self._output is None: + return "" + return os.pread(self._output.fileno(), 128_000, 0).decode("utf-8", errors="replace") + + def execute(self, command: str) -> tuple[str, int]: + self.cli_calls = [] + env = {"PATH": os.pathsep.join((str(self.control_path), str(Path(sys.executable).parent), "/opt/homebrew/bin", "/usr/bin", "/bin")), + "HOME": str(self.fixture_root / "home"), "TMPDIR": str(self.scratch), "LC_ALL": "en_US.UTF-8", "SHELL": "/bin/sh", + "PYTHONDONTWRITEBYTECODE": "1", "LOOPX_FIXTURE_SOCKET": str(self.socket_path), "LOOPX_TURN": self.turn_instance_id} + with tempfile.TemporaryFile() as output: + self._output = output + with self._invocation_lock: + self._active = True + process = subprocess.Popen([*self._sandbox(), "/bin/sh", "-c", command], cwd=self.project, env=env, + stdout=output, stderr=subprocess.STDOUT, start_new_session=True) + try: + code = process.wait(timeout=120) + except subprocess.TimeoutExpired: + os.killpg(process.pid, signal.SIGKILL) + process.wait() + code = 124 + finally: + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + with self._invocation_lock: + self._active = False + result = self.output() + self._output = None + return result, code + + def close(self) -> None: + self.server.shutdown() + self.server.server_close() + self.thread.join() + self.control.cleanup() diff --git a/tests/control_plane/test_required_vision_closeout_behavior.py b/tests/control_plane/test_required_vision_closeout_behavior.py index 928848f0a2..2cf7b692e1 100644 --- a/tests/control_plane/test_required_vision_closeout_behavior.py +++ b/tests/control_plane/test_required_vision_closeout_behavior.py @@ -13,15 +13,19 @@ ) from loopx.control_plane.testing.replan_semantic_action_behavior import ( DoubaoReplanSemanticActionBehaviorActor, _build_fixture, - _bounded_workspace_read_plan, _execute_workspace_read, ) -from loopx.control_plane.testing.replan_vision_closeout_behavior import _authored_file, _json_heredoc +from loopx.control_plane.testing.vision_shell_host import VisionShellHost, shell_isolation_available + +pytestmark = pytest.mark.skipif(not shell_isolation_available(), reason="Native shell needs sandbox-exec or bubblewrap") def _packet(request: Mapping[str, Any]) -> dict[str, Any]: for message in request["messages"]: if message["role"] == "tool": - value = json.loads(message["content"]) + try: + value = json.loads(message["content"]) + except json.JSONDecodeError: + continue if "replan_action_packet" in value: return value raise AssertionError("quota was not observed") @@ -46,11 +50,7 @@ def vision_patch_action(_: Mapping[str, Any]) -> ScriptedExecToolAction: "evidence_refs": ["evidence-permission-config"], }, } - lines = "\n".join("+" + line for line in json.dumps(decision, indent=2).splitlines()) - return ScriptedExecToolAction( - "apply_patch <<'PATCH'\n*** Begin Patch\n*** Add File: decision.json\n" - + lines + "\n*** End Patch\nPATCH" - ) + return ScriptedExecToolAction("cat > decision.json <<'JSON'\n" + json.dumps(decision, indent=2) + "\nJSON") def projected_refresh(request: Mapping[str, Any]) -> ScriptedExecToolAction: @@ -61,431 +61,152 @@ def projected_refresh(request: Mapping[str, Any]) -> ScriptedExecToolAction: def projected_spend(request: Mapping[str, Any]) -> ScriptedExecToolAction: - command = _packet(request)["interaction_contract"]["cli_channel"]["next_cli_actions"][1] - return ScriptedExecToolAction(command) + return ScriptedExecToolAction(_packet(request)["interaction_contract"]["cli_channel"]["next_cli_actions"][1]) -def heredoc_action(request: Mapping[str, Any]) -> ScriptedExecToolAction: - patch_lines = vision_patch_action(request).command.splitlines()[3:-2] - content = "\n".join(line[1:] for line in patch_lines) - return ScriptedExecToolAction("cat > decision.json <<'JSON'\n" + content + "\nJSON") +def _qualify(tmp_path: Path, actions: list[Any]) -> dict[str, Any]: + transport = ScriptedDoubaoExecTransport(actions) + return DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="native-vision-closeout", fixture_root=tmp_path / "actor", required_vision=True, + ) -@pytest.mark.parametrize("advancement_policy", ["as_needed", "repeat_until_closed"]) -@pytest.mark.parametrize("authoring", ["apply_patch", "heredoc_with_closeout"]) -def test_required_vision_uses_real_bound_refresh_spend_and_readback(tmp_path: Path, advancement_policy: str, authoring: str) -> None: +@pytest.mark.parametrize("policy", ["as_needed", "repeat_until_closed"]) +@pytest.mark.parametrize("compound", [False, True]) +def test_native_shell_closes_real_cli_turn_without_command_rituals(tmp_path: Path, policy: str, compound: bool) -> None: fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - def authored_decision(request: Mapping[str, Any]) -> ScriptedExecToolAction: - command = (vision_patch_action if authoring == "apply_patch" else heredoc_action)(request).command - command = command.replace('"as_needed"', json.dumps(advancement_policy)) - if authoring == "heredoc_with_closeout": - command += "\n" + projected_refresh(request).command + " && " + projected_spend(request).command + def author(request: Mapping[str, Any]) -> ScriptedExecToolAction: + command = vision_patch_action(request).command.replace("as_needed", policy) + command += "\necho authored\npython3 -m json.tool decision.json >/dev/null" + if compound: + refresh = projected_refresh(request).command.replace("'decision.json'", '"$VISION"') + command += "\nVISION=decision.json\n" + refresh + " && " + projected_spend(request).command return ScriptedExecToolAction(command) actions = [ + ScriptedExecToolAction("pwd && ls -la && loopx --help | head -100"), ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat replan-frontier.json"), - ScriptedExecToolAction("cat fixture/permission-config.json"), - authored_decision, + ScriptedExecToolAction("python3 -c \"import json; print(json.dumps(json.load(open('fixture/permission-config.json'))))\""), + author, ] - if authoring == "apply_patch": - actions.extend([projected_refresh, projected_spend]) - transport = ScriptedDoubaoExecTransport(actions) - receipt = DoubaoReplanSemanticActionBehaviorActor( - api_key="test-only-placeholder", transport=transport, - ).qualify(qualification_id="required-vision", fixture_root=tmp_path / "actor", required_vision=True) - assert receipt["qualification_passed"] is True, receipt - assert receipt["trigger_kinds"] == ["required_agent_vision_missing"] - assert receipt["selected_semantic_outcomes"] == ["fresh_vision_path_outcome"] - assert receipt["vision_closeout"] == { + if not compound: + actions += [projected_refresh, projected_spend] + result = _qualify(tmp_path, actions) + assert result["qualification_passed"] is True, result + assert result["execution_host"] == "os_isolated_shell" + assert result["boundary"]["shell_commands_executed"] is True + assert result["selected_semantic_outcomes"] == ["fresh_vision_path_outcome"] + assert result["vision_closeout"] == { "checkpoint_satisfied": True, "bound_writeback": True, "settled": True, "spend_count": 1, "original_obligation_closed": True, } - if advancement_policy == "as_needed": - assert receipt["semantic_reentry"]["effective_action"] == "monitor_quiet_skip" - assert "required_agent_vision_missing" not in receipt["semantic_reentry"]["trigger_kinds"] + assert "required_agent_vision_missing" not in result["semantic_reentry"]["trigger_kinds"] -def test_successful_refresh_without_settlement_is_not_qualified(tmp_path: Path) -> None: +@pytest.mark.parametrize("evidence", ["evidence-permission-config", "fixture/permission-config.json"]) +def test_cli_validation_errors_are_correctable_in_the_same_draft(tmp_path: Path, evidence: str) -> None: fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - transport = ScriptedDoubaoExecTransport([ + def oversized(request: Mapping[str, Any]) -> ScriptedExecToolAction: + return ScriptedExecToolAction(vision_patch_action(request).command.replace( + "Reader is default; writing requires an explicit grant.", "x" * 700)) + def correct(request: Mapping[str, Any]) -> ScriptedExecToolAction: + response = json.loads(request["messages"][-1]["content"]) + assert response["exit_code"] != 0 + assert "vision_budget_exceeded" in response["output"] + return ScriptedExecToolAction(vision_patch_action(request).command.replace("evidence-permission-config", evidence)) + result = _qualify(tmp_path, [ ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat replan-frontier.json"), ScriptedExecToolAction("cat fixture/permission-config.json"), - vision_patch_action, projected_refresh, ScriptedAssistantAction("Done"), + oversized, projected_refresh, correct, projected_refresh, projected_spend, ]) - receipt = DoubaoReplanSemanticActionBehaviorActor( - api_key="test-only-placeholder", transport=transport, - ).qualify(qualification_id="required-vision-incomplete", fixture_root=tmp_path / "actor", required_vision=True) - assert receipt["qualification_passed"] is False - assert receipt["vision_closeout"]["checkpoint_satisfied"] is True - assert receipt["vision_closeout"]["settled"] is False + assert result["qualification_passed"] is True, result + assert result["vision_closeout"]["spend_count"] == 1 -def test_wrong_turn_is_rejected_not_repaired_by_fixture_adapter(tmp_path: Path) -> None: +@pytest.mark.parametrize("invalid", ["no_source", "wrong_turn", "unread_reference", "no_spend"]) +def test_native_host_does_not_qualify_unproven_closeout(tmp_path: Path, invalid: str) -> None: fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - def wrong_turn(request: Mapping[str, Any]) -> ScriptedExecToolAction: + def author(request: Mapping[str, Any]) -> ScriptedExecToolAction: + command = vision_patch_action(request).command + if invalid == "unread_reference": + command = command.replace("evidence-permission-config", "unread-source") + return ScriptedExecToolAction(command) + def refresh(request: Mapping[str, Any]) -> ScriptedExecToolAction: tokens = shlex.split(projected_refresh(request).command) - tokens[tokens.index("--turn-instance-id") + 1] = "another-turn" + if invalid == "wrong_turn": + tokens[tokens.index("--turn-instance-id") + 1] = "wrong-turn" return ScriptedExecToolAction(shlex.join(tokens)) - transport = ScriptedDoubaoExecTransport([ - ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat replan-frontier.json"), - ScriptedExecToolAction("cat fixture/permission-config.json"), - vision_patch_action, wrong_turn, - ]) - receipt = DoubaoReplanSemanticActionBehaviorActor( - api_key="test-only-placeholder", transport=transport, - ).qualify(qualification_id="wrong-turn", fixture_root=tmp_path / "actor", required_vision=True) - assert receipt["qualification_passed"] is False - assert receipt["failure_code"] == "vision_closeout_turn_mismatch" - assert "vision_closeout" not in receipt - - -def test_authoring_cannot_escape_or_overwrite_fixture(tmp_path: Path) -> None: - command = vision_patch_action({}).command - for path in ("../outside.json", "/outside.json"): - with pytest.raises(ValueError, match="path_outside_fixture"): - _authored_file(command.replace("decision.json", path), tmp_path) - _authored_file(command, tmp_path) - original = (tmp_path / "decision.json").read_bytes() - with pytest.raises(ValueError, match="path_outside_fixture"): - _authored_file(command, tmp_path) - assert (tmp_path / "decision.json").read_bytes() == original - - -def test_heredoc_is_inert_json_and_never_an_arbitrary_shell_program() -> None: - command = heredoc_action({}).command - name, content, suffix = _json_heredoc(command.replace("Explicit write grant", "$(touch outside); `echo untrusted`")) - assert name == "decision.json" and suffix == [] - assert "$(touch outside)" in json.loads(content)["path_delta"]["retained"][0] - for tail in ("\ntouch outside", "\nloopx quota spend-slot; touch outside", "\nloopx quota spend-slot | echo outside"): - with pytest.raises(ValueError, match="suffix_requires_loopx"): - _json_heredoc(command + tail) - assert _json_heredoc(command.replace("<<'JSON'", "< None: - fixture = _build_fixture(tmp_path, required_vision=True) - command = f'ls {shlex.quote(str(fixture.runtime_root))} 2>/dev/null && echo "---" && loopx --help 2>&1 | head -60' - plan = _bounded_workspace_read_plan(command, fixture=fixture) - assert plan is not None - assert plan[-1].kind == "loopx_help" - output, frontier_read, source_read, exit_code = _execute_workspace_read(command, fixture=fixture) - assert "usage:" in output.lower() and "loopx" in output - assert exit_code == 0 and not frontier_read and not source_read - # Adding an unadmitted effect rejects the entire plan before execution. - assert _bounded_workspace_read_plan(command + " && touch injected", fixture=fixture) is None - with pytest.raises(ValueError, match="outside the hermetic fixture"): - _execute_workspace_read(command + " && touch injected", fixture=fixture) - assert not (fixture.project_root / "injected").exists() - - -@pytest.mark.parametrize("command", ["loopx --help", "cat replan-frontier.json && loopx refresh-state --help"]) -def test_read_and_help_extension_preserves_quota_first(command: str, tmp_path: Path) -> None: - transport = ScriptedDoubaoExecTransport([ScriptedExecToolAction(command), ScriptedAssistantAction("Stopped before quota admission.")]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="read-before-quota", fixture_root=tmp_path, required_vision=True, - ) - assert receipt["qualification_passed"] is False - assert receipt["tool_call_receipts"][0]["error_code"] == "workspace_read_before_quota" - assert receipt["semantic_action_accepted"] is False - assert receipt["boundary"]["read_only_host_commands_executed"] is False - - -def test_read_error_reaches_model_and_recovery_still_requires_full_closeout(tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: - error = json.loads(request["messages"][-1]["content"]) - assert error["exit_code"] != 0 - assert error["error_code"] == "workspace_read_nonzero" - assert "missing.json" in error["output"] - return ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json") - transport = ScriptedDoubaoExecTransport([ - ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat missing.json"), recover, - vision_patch_action, projected_refresh, projected_spend, - ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="read-error-recovery", fixture_root=tmp_path / "actor", required_vision=True, - ) - assert receipt["qualification_passed"] is True - assert receipt["vision_closeout"]["spend_count"] == 1 - assert receipt["tool_call_count"] == 6 - assert receipt["tool_call_receipts"][1]["error_code"] == "workspace_read_nonzero" - assert "bounded" in transport.requests[0]["tools"][0]["function"]["description"] - - -def test_read_failures_consume_the_full_closeout_call_budget(tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - transport = ScriptedDoubaoExecTransport([ - ScriptedExecToolAction(fixture.quota_guard_command), - *[ScriptedExecToolAction("cat missing.json") for _ in range(31)], - ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="read-errors-exhaust-budget", fixture_root=tmp_path / "actor", required_vision=True, - ) - assert receipt["qualification_passed"] is False - assert receipt["failure_code"] == "tool_call_budget_exhausted" - assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 32 - assert receipt["semantic_action_accepted"] is False - - -@pytest.mark.parametrize("command", [ - "ls -la .codex/ && find .codex -maxdepth 5 -type f | head -50", - "touch injected", -]) -def test_unadmitted_commands_return_feedback_without_execution_or_extra_budget(command: str, tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: - error = json.loads(request["messages"][-1]["content"]) - assert error["error_code"] == "unsupported_host_command" - assert error["exit_code"] != 0 - assert "not executed" in error["output"] - return vision_patch_action(request) - transport = ScriptedDoubaoExecTransport([ - ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), - ScriptedExecToolAction(command), recover, projected_refresh, projected_spend, - ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="unadmitted-command-recovery", fixture_root=tmp_path / "actor", required_vision=True, - ) - assert receipt["qualification_passed"] is True - assert receipt["vision_closeout"]["spend_count"] == 1 - assert receipt["tool_call_count"] == 6 - assert receipt["tool_call_receipts"][2]["error_code"] == "unsupported_host_command" - assert not list(tmp_path.rglob("injected")) - - -def test_unadmitted_commands_never_supply_evidence_or_success(tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - transport = ScriptedDoubaoExecTransport([ - ScriptedExecToolAction(fixture.quota_guard_command), - *[ScriptedExecToolAction("find . -maxdepth 5 -type f") for _ in range(31)], - ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="unadmitted-command-budget", fixture_root=tmp_path / "actor", required_vision=True, - ) - assert receipt["qualification_passed"] is False - assert receipt["failure_code"] == "tool_call_budget_exhausted" - assert receipt["semantic_action_accepted"] is False - assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 32 - - -@pytest.mark.parametrize("premature", ["read", "author"]) -def test_pre_admission_workspace_access_is_rejected_with_recovery(premature: str, tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: - error = json.loads(request["messages"][-1]["content"]) - assert error["error_code"] == ("workspace_read_before_quota" if premature == "read" else "vision_authoring_before_quota") - assert "quota admission" in error["output"] - assert not (tmp_path / "actor/project/decision.json").exists() - return ScriptedExecToolAction(fixture.quota_guard_command) - transport = ScriptedDoubaoExecTransport([ - ScriptedExecToolAction("cat fixture/permission-config.json") if premature == "read" else heredoc_action, - recover, ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), - heredoc_action, projected_refresh, projected_spend, - ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="pre-admission-recovery", fixture_root=tmp_path / "actor", required_vision=True, - ) - assert receipt["qualification_passed"] is True - assert receipt["tool_call_count"] == 6 - assert receipt["vision_closeout"]["spend_count"] == 1 + actions = [ScriptedExecToolAction(fixture.quota_guard_command)] + if invalid != "no_source": + actions.append(ScriptedExecToolAction("cat fixture/permission-config.json")) + actions += [author, refresh, ScriptedAssistantAction("Stopped before a verified settlement.")] + result = _qualify(tmp_path, actions) + assert result["qualification_passed"] is False + assert not (result.get("vision_closeout") or {}).get("settled") @pytest.mark.parametrize("extra_reads,passed", [(27, True), (28, False)]) -def test_full_closeout_budget_boundary_never_waives_settlement(extra_reads: int, passed: bool, tmp_path: Path) -> None: +def test_budget_boundary_still_requires_the_final_spend(tmp_path: Path, extra_reads: int, passed: bool) -> None: fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - transport = ScriptedDoubaoExecTransport([ + result = _qualify(tmp_path, [ ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), - *[ScriptedExecToolAction("pwd") for _ in range(extra_reads)], + ScriptedExecToolAction("cat fixture/permission-config.json"), + *[ScriptedExecToolAction("printf inspected") for _ in range(extra_reads)], vision_patch_action, projected_refresh, projected_spend, ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="closeout-budget-boundary", fixture_root=tmp_path / "actor", required_vision=True, - ) - assert receipt["qualification_passed"] is passed - assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 32 - assert receipt["semantic_action_accepted"] is True - assert receipt["vision_closeout"]["settled"] is passed - if passed: - assert receipt["vision_closeout"]["spend_count"] == 1 - else: - assert receipt["failure_code"] == "tool_call_budget_exhausted" + assert result["qualification_passed"] is passed + assert result["tool_call_count"] == result["tool_call_limit"] == 32 + assert result["vision_closeout"]["settled"] is passed -def test_narrow_semantic_action_keeps_seven_call_budget(tmp_path: Path) -> None: +def test_narrow_actor_retains_its_existing_budget_and_tool(tmp_path: Path) -> None: fixture = _build_fixture(tmp_path / "oracle") transport = ScriptedDoubaoExecTransport([ ScriptedExecToolAction(fixture.quota_guard_command), *[ScriptedExecToolAction("pwd") for _ in range(7)], ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="narrow-budget-unchanged", fixture_root=tmp_path / "actor", - ) - assert receipt["qualification_passed"] is False - assert receipt["failure_code"] == "tool_call_budget_exhausted" - assert receipt["tool_call_count"] == receipt["tool_call_limit"] == 7 - - -@pytest.mark.parametrize("rejection", ["suffix", "outside", "invalid_json", "overwrite"]) -def test_authoring_rejection_is_recoverable_but_has_no_file_or_suffix_effect(rejection: str, tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - root = tmp_path / "actor" / "project" - def rejected_authoring(request: Mapping[str, Any]) -> ScriptedExecToolAction: - command = heredoc_action(request).command - if rejection == "suffix": - command += "\ntouch injected" - elif rejection == "outside": - command = command.replace("decision.json", "../outside.json") - elif rejection == "invalid_json": - command = "cat > decision.json <<'JSON'\n{broken\nJSON" - else: - command = command.replace("decision.json", "fixture/permission-config.json") - return ScriptedExecToolAction(command) - def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: - error = json.loads(request["messages"][-1]["content"]) - assert error["exit_code"] != 0 and "not executed" in error["output"] - if rejection == "suffix": - assert "touch" in error["output"] and "heredoc alone" in error["output"] - assert not (root / "decision.json").exists() - assert not list(tmp_path.rglob("injected")) - assert not list(tmp_path.rglob("outside.json")) - assert (root / "fixture/permission-config.json").read_bytes() == fixture.work_source_target.read_bytes() - return heredoc_action(request) - transport = ScriptedDoubaoExecTransport([ - ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), - rejected_authoring, recover, projected_refresh, projected_spend, - ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="authoring-error-recovery", fixture_root=tmp_path / "actor", required_vision=True, + result = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( + qualification_id="narrow-unchanged", fixture_root=tmp_path / "actor", ) - assert receipt["qualification_passed"] is True - assert receipt["vision_closeout"]["spend_count"] == 1 - assert receipt["tool_call_count"] == 6 - - -@pytest.mark.parametrize("evidence_ref,passed", [ - ("evidence-permission-config", True), ("fixture/permission-config.json", True), - ("fixture/unread.json", False), ("replan-frontier.json", False), -]) -def test_closeout_matches_observed_evidence_not_one_identifier_spelling(evidence_ref: str, passed: bool, tmp_path: Path) -> None: + assert result["tool_call_limit"] == result["tool_call_count"] == 7 + assert transport.requests[0]["tools"] == [EXEC_COMMAND_TOOL] + + +def test_os_boundary_protects_inputs_authority_private_data_and_network(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + fixture = _build_fixture(tmp_path / "fixture", required_vision=True) + private = tmp_path / "private.txt" + private.write_text("synthetic-private-marker") + monkeypatch.setenv("ARK_API_KEY", "synthetic-do-not-inherit") + host = VisionShellHost(fixture.project_root, lambda *args: "ok", turn_instance_id="shell-isolation-test") + original = fixture.work_source_target.read_bytes() + try: + for command in [ + "echo forged > fixture/permission-config.json", + "mv fixture renamed-inputs", + f"echo forged > {shlex.quote(str(fixture.runtime_root / 'forged.json'))}", + f"cat {shlex.quote(str(private))}", + "python3 -c \"import socket; socket.create_connection(('127.0.0.1',9),timeout=1)\"", + ]: + output, code = host.execute(command) + assert code != 0 + assert "synthetic-private-marker" not in output + output, code = host.execute("printf '%s' \"$ARK_API_KEY\"; echo draft > draft.txt; python3 -c \"import json; print(json.dumps({'valid':True}))\"") + assert code == 0 and "synthetic-do-not-inherit" not in output + assert fixture.work_source_target.read_bytes() == original + assert (fixture.project_root / "draft.txt").read_text().strip() == "draft" + assert not (fixture.runtime_root / "forged.json").exists() + finally: + host.close() + + +def test_actor_cannot_shadow_the_trusted_cli_in_its_writable_project(tmp_path: Path) -> None: fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - def author(request: Mapping[str, Any]) -> ScriptedExecToolAction: - return ScriptedExecToolAction(heredoc_action(request).command.replace("evidence-permission-config", evidence_ref)) - transport = ScriptedDoubaoExecTransport([ + def check_real_cli(request: Mapping[str, Any]) -> ScriptedAssistantAction: + output = request["messages"][-1]["content"] + assert "loopx --help" in output and "SHADOWED_CLI" not in output + return ScriptedAssistantAction("Only checking executor source provenance.") + result = _qualify(tmp_path, [ ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), - author, projected_refresh, - projected_spend if passed else ScriptedAssistantAction("No observed source reference supplied."), + ScriptedExecToolAction("mkdir loopx; touch loopx/__init__.py; printf 'print(\"SHADOWED_CLI\")' > loopx/cli.py; loopx --help"), + check_real_cli, ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="observed-evidence-reference", fixture_root=tmp_path / "actor", required_vision=True, - ) - assert receipt["qualification_passed"] is passed - if passed: - assert receipt["vision_closeout"]["spend_count"] == 1 - else: - assert receipt["tool_call_receipts"][-1]["error_code"] == "vision_closeout_evidence_not_observed" - assert receipt["semantic_action_accepted"] is False - - -@pytest.mark.parametrize("shell_shape", ["assignment_prefix", "newline_sequence"]) -def test_shell_shape_is_not_silently_discarded_before_binding_check(shell_shape: str, tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - def prefixed(request: Mapping[str, Any]) -> ScriptedExecToolAction: - command = projected_refresh(request).command - command = ("TURN=unexpanded\n" + command if shell_shape == "assignment_prefix" - else command + "\n" + projected_spend(request).command) - return ScriptedExecToolAction(command) - def recover(request: Mapping[str, Any]) -> ScriptedExecToolAction: - error = json.loads(request["messages"][-1]["content"]) - assert error["error_code"] == "vision_command_requires_literal_loopx_argv" - assert "not executed" in error["output"] - return projected_refresh(request) - transport = ScriptedDoubaoExecTransport([ - ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), - vision_patch_action, prefixed, recover, projected_spend, - ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="literal-cli-admission", fixture_root=tmp_path / "actor", required_vision=True, - ) - assert receipt["qualification_passed"] is True - assert receipt["vision_closeout"]["spend_count"] == 1 - assert receipt["tool_call_count"] == 6 - - -@pytest.mark.parametrize("rejection", ["unobserved_evidence", "oversized_vision"]) -def test_actor_can_correct_its_own_draft_after_rejection(rejection: str, tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - def ungrounded(request: Mapping[str, Any]) -> ScriptedExecToolAction: - command = heredoc_action(request).command - command = (command.replace("evidence-permission-config", "unobserved-evidence") - if rejection == "unobserved_evidence" else command.replace( - "Reader is default; writing requires an explicit grant.", "x" * 700)) - return ScriptedExecToolAction(command) - def corrected(request: Mapping[str, Any]) -> ScriptedExecToolAction: - error = json.loads(request["messages"][-1]["content"]) - if rejection == "unobserved_evidence": - assert error["error_code"] == "vision_closeout_evidence_not_observed" - assert "fixture/permission-config.json" in error["output"] - else: - assert error["error_code"] == "loopx_cli_nonzero" - assert error["exit_code"] != 0 - return heredoc_action(request) - transport = ScriptedDoubaoExecTransport([ - ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat replan-frontier.json && cat fixture/permission-config.json"), - ungrounded, projected_refresh, corrected, projected_refresh, projected_spend, - ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="correct-owned-draft", fixture_root=tmp_path / "actor", required_vision=True, - ) - assert receipt["qualification_passed"] is True - assert receipt["vision_closeout"]["spend_count"] == 1 - assert receipt["tool_call_count"] == 7 - - -def test_help_host_extension_does_not_change_narrow_actor_contract(tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path) - assert _bounded_workspace_read_plan("loopx --help", fixture=fixture) is None - with pytest.raises(RuntimeError, match="workspace read failed"): - _execute_workspace_read("cat missing.json", fixture=fixture) - - -def test_failed_read_does_not_supply_source_evidence(tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path / "oracle", required_vision=True) - transport = ScriptedDoubaoExecTransport([ - ScriptedExecToolAction(fixture.quota_guard_command), - ScriptedExecToolAction("cat replan-frontier.json"), - ScriptedExecToolAction("cat missing.json"), vision_patch_action, projected_refresh, - ]) - receipt = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport).qualify( - qualification_id="failed-read-is-not-evidence", fixture_root=tmp_path / "actor", required_vision=True, - ) - assert receipt["qualification_passed"] is False - assert receipt["failure_code"] == "vision_closeout_requires_observed_source_and_authored_decision" - assert receipt["semantic_action_accepted"] is False - - -def test_read_short_circuit_and_stderr_suppression_are_preserved(tmp_path: Path) -> None: - fixture = _build_fixture(tmp_path, required_vision=True) - output, _, _, code = _execute_workspace_read("cat missing.json 2>/dev/null && loopx --help", fixture=fixture) - assert code != 0 and output == "" - output, _, _, code = _execute_workspace_read("cat missing.json 2>/dev/null || loopx --help", fixture=fixture) - assert code == 0 and "usage:" in output.lower() - - -def test_tool_description_override_is_request_local() -> None: - transport = ScriptedDoubaoExecTransport([ScriptedAssistantAction("done"), ScriptedAssistantAction("done")]) - client = DoubaoReplanSemanticActionBehaviorActor(api_key="test-only-placeholder", transport=transport)._client - client.next_step([], tool_description="A bounded test host") - client.next_step([]) - custom = transport.requests[0]["tools"][0] - assert custom["function"]["description"] == "A bounded test host" - assert custom["function"]["parameters"] == EXEC_COMMAND_TOOL["function"]["parameters"] - assert transport.requests[1]["tools"] == [EXEC_COMMAND_TOOL] + assert result["qualification_passed"] is False From 4e6f0037c347273c4084583f1774584d1aeb7f31 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:38:44 +0800 Subject: [PATCH 20/22] docs(testing): replace command grammar with native execution boundaries Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- docs/development/testing-and-quality.md | 56 +++++++++---------- .../references/repair-patterns.md | 2 +- 2 files changed, 29 insertions(+), 29 deletions(-) diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index 7b893a41d4..d6f50bb4c1 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -808,34 +808,34 @@ it does not authorize more effects or weaken the checkpoint/readback oracle. Each receipt records `tool_call_limit` beside actual usage. Earlier seven- and sixteen-call failures remain failures; the revised budget defines a new qualification, not a retrospective pass. Limit exhaustion after refresh but before spend still -fails. The tool description declares the actual -bounded host grammar, not unrestricted shell access: workspace reads may compose -with root/refresh-state CLI help, and JSON authoring uses `apply_patch` or a quoted -heredoc, optionally followed by bound CLI closeout. The shared read parser admits -the entire plan before execution. Ordinary nonzero reads return an exit code and -bounded output to the model; recovery consumes the same call budget and a failed -read supplies no evidence. Unadmitted commands likewise return a no-execution -error instead of ending the conversation; their grammar limits are not raised. -Workspace access before quota admission is also rejected without execution and -returned as tool feedback; it supplies no evidence and consumes a call. -Unsupported programs and rejected file-authoring operations never execute. -Pre-execution path, overwrite, JSON or authoring-syntax rejection returns tool -feedback without creating the file or running a suffix, with the rejected -operation and applicable restriction. An actor may revise only its own JSON -drafts; original fixture inputs stay immutable. A rejected evidence reference -returns the already-observed references before refresh executes, so the actor -can correct its draft. Real CLI nonzero results return their exit code and -bounded diagnostic; this does not imply rollback or waive original-Turn recovery -and settlement fences. Identity and durable-verification failures remain -terminal. The host does not add missing -semantic fields or execute arbitrary model shell programs. These host semantics -are explicit for required-vision qualification; other actors retain their current -contracts. Receipts distinguish tool execution errors, accepted writeback and -settled closeout without persisting raw conversations. A host grammar rejection -before writeback is not evidence of a core semantic-admission failure. -Shell assignment prefixes are rejected before CLI identity checks, not silently -discarded. An observed evidence item's id and declared source reference both -identify that same evidence; arbitrary or unread sources do not qualify it. +fails. + +Required-vision qualification runs a normal shell in an OS-isolated fixture, +using `sandbox-exec` on macOS or `bubblewrap` on Linux. These are execution +prerequisites: missing isolation fails explicitly rather than falling back to an +unrestricted host. Shell variables, pipelines, compound commands, Python/JSON +validation, and draft rewrites are ordinary operations, not an allowlisted +command language. Read-only inspection may precede quota; writeback still needs +the current quota-derived binding. The source must be observed in returned +tool data, but no particular read command or metadata-read sequence is required. + +The shell may write project drafts and its private `$TMPDIR`. Original inputs, +authority stores, host-private data and external network access are protected +by the execution boundary. A fixture-local `loopx` command forwards expanded +argv to the existing real CLI executor; only the task's quota, help, vision +refresh and settlement operations have authority. The shell cannot forge run +history or receipts by writing the store directly. Turn identity is supplied +as normal host context, never inferred from a model's incorrect binding. + +Shell and real CLI failures return output and exit status for correction within +the same call budget. Returning an error is not semantic acceptance or rollback. +The model authors its own decision; the host does not fill semantic fields or +change evidence. An observed source's evidence id and exact source reference +identify the same evidence. Checkpoint, one-spend and following-Turn verification +remain independent of shell success. Receipts disclose native-shell execution +without persisting raw conversations. Other actors retain their existing host +and seven-call contracts. Earlier bounded-grammar failures remain failed +qualifications and are not evidence of core semantic rejection. The other turn cases remain bounded packet-interpretation checks. Nine core-contract scenarios cover onboarding, agent identity and goal selection, selected todo, peer identity diff --git a/skills/loopx-self-repair/references/repair-patterns.md b/skills/loopx-self-repair/references/repair-patterns.md index baa3f60c21..54c32ed149 100644 --- a/skills/loopx-self-repair/references/repair-patterns.md +++ b/skills/loopx-self-repair/references/repair-patterns.md @@ -5,7 +5,7 @@ teaches a reusable control-plane lesson. | Pattern | Symptoms | Evidence To Read | Likely Root | Durable Repair | | --- | --- | --- | --- | --- | -| `qualification_host_contract_mismatch` | A model stops during ordinary file inspection or CLI help and the result is reported as a semantic-control failure. | Actual synthetic command shape, advertised tool contract, bounded parser decision, subprocess exit status, next model request and durable writeback attempts. | An unrestricted-shell description hid a narrow grammar; supported reads and help could not compose, or a read failure / unadmitted command ended qualification without tool feedback. | Declare host capabilities, reuse bounded query composition, and return read errors or no-execution rejections within the declared scenario budget. Admit the complete plan before execution; do not raise grammar limits to match a model's command. Preserve evidence, identity and write-boundary rejection. Separate host rejection, model budget exhaustion and core semantic admission; retain every earlier live failure. Do not add task-specific action instructions or arbitrary shell authority to obtain a passing sample. | +| `qualification_host_contract_mismatch` | A model stops on ordinary inspection, shell composition or draft correction and the result is reported as a semantic-control failure. | Actual synthetic operation, advertised tool contract, OS isolation, subprocess exit status, returned diagnostics and durable writeback attempts. | A shell-labelled host imposed a separate command language or ended execution without returning normal tool errors. | Use a normal shell inside an isolated execution environment, with real CLI effects supervised at their existing authority boundary. Observe source evidence and durable outcomes instead of requiring a command spelling or read ritual. Return errors within a disclosed scenario budget; keep original inputs and authority stores protected. Separate host rejection, budget exhaustion and core semantic admission, retaining earlier failures. Do not insert model answers, waive evidence or grow a command whitelist one failed trajectory at a time. | | `replan_projection_discharge_mismatch` | A projected refresh cannot satisfy its own vision obligation, or a required-vision qualification executes an ordinary repeated-progress fixture. | Typed trigger kinds, required outcomes, projected input/authoring contract, actual actor fixture, bound durable writeback, checkpoint and post-settlement readback. | Projection and admission encoded separate outcome rules; a successful narrow action was mistaken for a complete journey. | Share outcome matching and executable input projection at the typed owner. Preserve original-Turn recovery, authoring limits and legal successor/terminal exits. Match qualification scope to the fixture; distinguish accepted writeback, checkpoint and one-spend settlement. A new successor duty is not a failed closeout. Do not add ACK rituals, weaken acceptance or replay until green. Full closeout and one action need not share a call cap: an authorized, cost-disclosed budget revision is a new qualification and must retain old failures, not relabel them. | | `budget_metric_overfitting` | A budget failure triggers automatic expansion, or mechanical compaction that removes useful semantics or breaks consumers. | Owning limit, matched base/head measurements, consumer/caller contract, original failure and revised validation. | A regression metric became the objective; historical ceilings or green tests replaced semantic judgment. | Follow the [budget decision guide](../../../docs/development/testing-and-quality.md#budget-failure-decisions), compare true redundancy, compatibility cost and justified headroom, and repair the existing contract/tests and review evidence. Preserve hard limits and frozen qualification results. | | `skill_import_recreation` | Duplicate LoopX skills return after successful cleanup; imported entries display a fallback brand casing. | Compare installed files and metadata with source-host skills; correlate file creation times with structured host import receipts. | A later external-host import recreates command facades in another discovered root, omits display metadata, and bypasses installer reconciliation. | Attribute the writer from import receipts without guessing the human initiator; exclude already-installed LoopX skills from later imports and rerun managed reconciliation. Ensure standalone workflow entry installation writes Codex metadata, previews missing metadata repair, and records the full installed tree. Preserve user metadata and exact-host invocation behavior. | From 0404a1bf77e8f0e3cfd337318e9854e7b2c254ad Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:50:31 +0800 Subject: [PATCH 21/22] fix(testing): prepare native shell isolation on Linux Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .github/workflows/python-tests.yml | 17 +++++++++++++++++ .../control_plane/testing/vision_shell_host.py | 4 ++++ 2 files changed, 21 insertions(+) diff --git a/.github/workflows/python-tests.yml b/.github/workflows/python-tests.yml index d1323db2fb..2ed266160f 100644 --- a/.github/workflows/python-tests.yml +++ b/.github/workflows/python-tests.yml @@ -284,6 +284,23 @@ jobs: run: | python -m pip install --disable-pip-version-check -e ".[test]" npm ci --ignore-scripts + - name: Prepare native shell isolation + # Ubuntu 24.04 may need an executable-scoped userns profile for bwrap. + # Do not disable AppArmor or the system-wide unprivileged-userns guard. + run: | + sudo apt-get update + sudo apt-get install -y bubblewrap + if ! bwrap --unshare-all --ro-bind / / /bin/true; then + sudo tee /etc/apparmor.d/loopx-qualification-bwrap >/dev/null <<'PROFILE' + abi , + include + profile loopx-qualification-bwrap /usr/bin/bwrap flags=(unconfined) { + userns, + } + PROFILE + sudo apparmor_parser -r /etc/apparmor.d/loopx-qualification-bwrap + fi + bwrap --unshare-all --ro-bind / / /bin/true - name: Run test shard # Split the whole collection, not a hand-maintained list of directories. # Without timing history least_duration alternates equal-weight tests. diff --git a/loopx/control_plane/testing/vision_shell_host.py b/loopx/control_plane/testing/vision_shell_host.py index 316c4e3bbf..63d7213e37 100644 --- a/loopx/control_plane/testing/vision_shell_host.py +++ b/loopx/control_plane/testing/vision_shell_host.py @@ -99,6 +99,10 @@ def _sandbox(self) -> list[str]: argv = [str(shutil.which("bwrap")), "--unshare-all", "--die-with-parent", "--new-session", "--proc", "/proc", "--dev", "/dev"] for path in dict.fromkeys(readable): argv += ["--ro-bind", str(path), str(path)] + # usr-merged Linux needs the original loader and shell aliases too. + for name in ("/bin", "/sbin", "/lib", "/lib64"): + if Path(name).is_symlink(): + argv += ["--symlink", os.readlink(name), name] argv += ["--bind", str(self.project), str(self.project)] for path in self.originals: argv += ["--ro-bind", str(path), str(path)] From 2084d0a0a1c152f7334230a89a57be1ed38b7471 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Sat, 19 Sep 2026 19:30:30 +0800 Subject: [PATCH 22/22] test(replan): align blocked wait with vision writeback Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- tests/control_plane/test_goal_vision_blocked_successor.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/control_plane/test_goal_vision_blocked_successor.py b/tests/control_plane/test_goal_vision_blocked_successor.py index 707fd088c7..ec2eeb4b43 100644 --- a/tests/control_plane/test_goal_vision_blocked_successor.py +++ b/tests/control_plane/test_goal_vision_blocked_successor.py @@ -812,7 +812,8 @@ def test_two_identical_blocked_successor_waits_trigger_bounded_replan() -> None: ] == obligation["satisfying_semantic_outcomes"] cli_actions = guard["interaction_contract"]["cli_channel"]["next_cli_actions"] refresh_action = next(action for action in cli_actions if "refresh-state" in action) - assert "--progress-result-class" in refresh_action + assert "--agent-vision-json" in refresh_action + assert "--progress-result-class" not in refresh_action assert "--autonomous-replan-recorded" not in refresh_action assert "--repair-delta-kind" not in refresh_action