diff --git a/.github/workflows/eval-pr-run.yml b/.github/workflows/eval-pr-run.yml index 197a5ff4..ca3110c8 100644 --- a/.github/workflows/eval-pr-run.yml +++ b/.github/workflows/eval-pr-run.yml @@ -284,6 +284,7 @@ jobs: run-evals: name: Run PR Evals + timeout-minutes: 90 needs: [discover, gate] if: >- !cancelled() && @@ -359,6 +360,8 @@ jobs: ANTHROPIC_DEFAULT_OPUS_MODEL: "claude-opus-4-6" ANTHROPIC_MODEL: "claude-opus-4-6" CLAUDE_CODE_SUBPROCESS_ENV_SCRUB: "1" + # Wait for eval agents to finish; the job timeout bounds total runtime. + CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS: "0" run: | IFS=',' read -ra SKILLS <<< "$SKILLS_CSV" failed=0 diff --git a/plugins/sdlc-workflow/scripts/test_native_fullsend_eval_ci.py b/plugins/sdlc-workflow/scripts/test_native_fullsend_eval_ci.py index 1587b694..dd7e681a 100644 --- a/plugins/sdlc-workflow/scripts/test_native_fullsend_eval_ci.py +++ b/plugins/sdlc-workflow/scripts/test_native_fullsend_eval_ci.py @@ -515,6 +515,9 @@ def run_ordinary_eval_step(tmp_path, mode, skills="first"): workspace = Path(re.search(r"Workspace: (.+)", args[args.index("-p") + 1])[1]) with open("calls.jsonl", "a") as capture: capture.write(json.dumps(args) + "\n") +with open("wait-settings.jsonl", "a") as capture: + capture.write(json.dumps({name: os.environ.get(name) for name in ( + "CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS", "CLAUDE_CODE_SUBPROCESS_ENV_SCRUB")}) + "\n") mode = os.environ["EVAL_STUB_MODE"] if workspace.name == "first-eval-pr" else "correct" if mode == "exit": sys.exit(7) @@ -538,16 +541,37 @@ def run_ordinary_eval_step(tmp_path, mode, skills="first"): source = tmp_path / "pr-head/evals" / skill source.mkdir(parents=True) (source / "evals.json").write_text('{"evals": [{"id": 1}]}') - script = script_step("run-evals", "Run PR evals")["run"].replace( + step = script_step("run-evals", "Run PR evals") + script = step["run"].replace( 'workspace="/tmp/${skill}-eval-pr"', f'workspace="{tmp_path}/${{skill}}-eval-pr"') + env = dict(os.environ, PATH=f"{tools}:{os.environ['PATH']}", + SKILLS_CSV=skills, EVAL_STUB_MODE=mode) + env.pop("CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS", None) + for name in ("CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS", "CLAUDE_CODE_SUBPROCESS_ENV_SCRUB"): + if name in step["env"]: + env[name] = step["env"][name] result = subprocess.run(["bash", "-e", "-o", "pipefail", "-c", script], cwd=tmp_path, - env=dict(os.environ, PATH=f"{tools}:{os.environ['PATH']}", - SKILLS_CSV=skills, EVAL_STUB_MODE=mode), + env=env, capture_output=True, text=True) calls = [json.loads(line) for line in (tmp_path / "calls.jsonl").read_text().splitlines()] return result, calls +def test_ordinary_eval_passes_background_wait_setting_to_each_cli(tmp_path): + """Both real shell invocations inherit the workflow wait policy and scrubbing.""" + result, calls = run_ordinary_eval_step(tmp_path, "correct", "first,second") + assert result.returncode == 0, result.stdout + result.stderr + assert len(calls) == 2 + settings = [json.loads(line) for line in (tmp_path / "wait-settings.jsonl").read_text().splitlines()] + assert settings == [{"CLAUDE_CODE_PRINT_BG_WAIT_CEILING_MS": "0", + "CLAUDE_CODE_SUBPROCESS_ENV_SCRUB": "1"}] * 2 + + +def test_ordinary_eval_job_bounds_background_wait(): + """Disabling the CLI idle ceiling still leaves an explicit total CI limit.""" + assert workflow()["jobs"]["run-evals"].get("timeout-minutes") == 90 + + @pytest.mark.parametrize("mode", ["absent", "misplaced", "missing-benchmark.json", "missing-feedback.json", "missing-summary.md", "empty-benchmark.json", "empty-feedback.json", "empty-summary.md",