diff --git a/benchmarks/truthbench/generated_code.py b/benchmarks/truthbench/generated_code.py index 317a3dbd..58d7594f 100644 --- a/benchmarks/truthbench/generated_code.py +++ b/benchmarks/truthbench/generated_code.py @@ -12,9 +12,11 @@ from __future__ import annotations import ast +import os import subprocess import sys import tempfile +from collections.abc import Mapping from dataclasses import dataclass, field from pathlib import Path from typing import Any @@ -97,6 +99,11 @@ def _stderr_excerpt(stderr: str) -> str: #: and every crash is still recorded in ``GeneratedCodeResult.native_crashes``. _NATIVE_CRASH_RETRIES = 1 +#: The harness prints this line before anything else. A child that exits +#: without it never ran the harness (the interpreter failed to start), so none +#: of the post-run checks below observed the generated code. +_STARTED_MARKER = "truthbench-sandbox-child-started" + @dataclass(frozen=True) class GeneratedCodeResult: @@ -111,6 +118,48 @@ class GeneratedCodeResult: #: One redacted excerpt per signal exit, including crashes that a retry #: recovered from, so a flaky pass is never silent. native_crashes: tuple[str, ...] = () + #: Set when the sandbox child could not start the harness. The generated + #: code never ran, so this result says nothing about its behavior and the + #: caller must treat it as an infrastructure failure, not a verdict. + infrastructure_failure: str | None = None + + +def _sandbox_env( + workdir: Path, + *, + os_name: str | None = None, + environ: Mapping[str, str] | None = None, +) -> dict[str, str]: + """The deliberately minimal environment of the sandbox child.""" + + env = { + "PYTHONDONTWRITEBYTECODE": "1", + "PYTHONNOUSERSITE": "1", + "FRESHDATA_NO_NETWORK": "1", + "HOME": str(workdir), + "TMPDIR": str(workdir), + # The environment is otherwise empty, so native thread pools would + # size themselves from the host core count; pin them (and the + # locale) so runs are deterministic across machines. + "OMP_NUM_THREADS": "1", + "OPENBLAS_NUM_THREADS": "1", + "MKL_NUM_THREADS": "1", + "POLARS_MAX_THREADS": "1", + "RAYON_NUM_THREADS": "1", + "LANG": "C.UTF-8", + } + if (os_name if os_name is not None else os.name) == "nt": + # CPython <= 3.10 on Windows seeds hash randomization through + # CryptGenRandom, which cannot load its provider without SystemRoot: + # the child dies with "Fatal Python error: _Py_HashRandomization_Init: + # failed to get random numbers". 3.11+ uses BCryptGenRandom and starts + # without it. Windows variable names are case-insensitive, so one key + # covers SystemRoot and SYSTEMROOT. + source = os.environ if environ is None else environ + system_root = source.get("SystemRoot") or source.get("SYSTEMROOT") + if system_root: + env["SystemRoot"] = system_root + return env def _allowlist_failures(tree: ast.AST) -> list[str]: @@ -141,6 +190,10 @@ def _harness(code_path: str, input_name: str) -> str: return f""" import sys +# First output: proves the interpreter started and is running this harness. +sys.stdout.write({_STARTED_MARKER!r} + "\\n") +sys.stdout.flush() + # Pre-import the allowed libraries so their own internal use of stdlib # modules (pandas imports subprocess for locale probing) completes first... import pandas # noqa: F401 @@ -220,22 +273,7 @@ def leaks(label: str, payload: Any) -> None: harness_path.write_text( _harness(str(code_path), _INPUT_BASENAME), encoding="utf-8" ) - env = { - "PYTHONDONTWRITEBYTECODE": "1", - "PYTHONNOUSERSITE": "1", - "FRESHDATA_NO_NETWORK": "1", - "HOME": str(workdir), - "TMPDIR": str(workdir), - # The environment is otherwise empty, so native thread pools would - # size themselves from the host core count; pin them (and the - # locale) so runs are deterministic across machines. - "OMP_NUM_THREADS": "1", - "OPENBLAS_NUM_THREADS": "1", - "MKL_NUM_THREADS": "1", - "POLARS_MAX_THREADS": "1", - "RAYON_NUM_THREADS": "1", - "LANG": "C.UTF-8", - } + env = _sandbox_env(workdir) interpreter = python or sys.executable def redacted(text: str) -> str: @@ -270,8 +308,26 @@ def redacted(text: str) -> str: stages=(*stages, "execute"), native_crashes=tuple(native_crashes), ) + stdout, stderr = proc.stdout or "", proc.stderr or "" + started = f"{_STARTED_MARKER}\n" + if proc.returncode >= 0 and not stdout.startswith(started): + # Fail closed: the generated code never ran, so the overwrite and + # canary checks below would observe nothing; report no verdict. + reason = ( + "sandbox infrastructure failure: the child interpreter exited " + f"with code {proc.returncode} before starting the harness: " + f"{_stderr_excerpt(redacted(stderr))}" + ) + return GeneratedCodeResult( + False, + (*failures, reason), + stderr=redacted(stderr), + stages=tuple(stages), + native_crashes=tuple(native_crashes), + infrastructure_failure=reason, + ) + stdout = stdout[len(started) :] stages.append("execute") - stdout, stderr = proc.stdout, proc.stderr if proc.returncode != 0: failures.append( f"generated code exited {proc.returncode}: " diff --git a/benchmarks/truthbench/runner.py b/benchmarks/truthbench/runner.py index 0ad203d6..159a14b2 100644 --- a/benchmarks/truthbench/runner.py +++ b/benchmarks/truthbench/runner.py @@ -742,6 +742,14 @@ def run_release( file=sys.stderr, ) + if any(result.infrastructure_failure for result in generated_results): + # The sandbox child never ran the generated code, so the sandbox + # checks observed nothing; that is not a gate verdict. + raise TruthBenchRunError( + "generated-code sandbox could not run: " + + "; ".join(sandbox_failures) + ) + observations, failures, parity_ledger = _parity_observations(parity, backends) parity_failures.extend( failure for failure in failures if failure not in parity_failures diff --git a/tests/truthbench/test_generated_code.py b/tests/truthbench/test_generated_code.py index 17c4b578..c82be195 100644 --- a/tests/truthbench/test_generated_code.py +++ b/tests/truthbench/test_generated_code.py @@ -47,6 +47,9 @@ def _fixture(): print("rows:", len(df)) """ +#: What a child that started the harness prints first. +_STARTED = gc._STARTED_MARKER + "\n" + def test_wellformed_code_passes_all_stages(): result = verify_generated_code(GOOD, _fixture()) @@ -90,14 +93,16 @@ def test_runtime_poison_blocks_banned_module_even_if_statically_allowed(monkeypa gc, "ALLOWED_IMPORTS", frozenset({*gc.ALLOWED_IMPORTS, "socket"}) ) result = verify_generated_code("import socket\n", _fixture()) + assert result.infrastructure_failure is None, result.failures assert not result.passed - assert any("exited" in f for f in result.failures) + assert any("generated code exited" in f for f in result.failures) def test_timeout_is_enforced(): result = verify_generated_code( "while True:\n pass\n", _fixture(), timeout=3.0 ) + assert result.infrastructure_failure is None, result.failures assert not result.passed assert any("timeout" in f for f in result.failures) @@ -116,6 +121,7 @@ def test_pii_canary_in_stdout_is_reported(): 'print(df["memo"].tolist())\n' ) result = verify_generated_code(code, _fixture()) + assert result.infrastructure_failure is None, result.failures assert not result.passed assert any("stdout leaked canary" in f for f in result.failures) assert "example.invalid" not in result.stdout # evidence itself is redacted @@ -128,6 +134,7 @@ def test_input_file_overwrite_is_reported(): 'df.head(1).to_csv("your_data.csv", index=False)\n' ) result = verify_generated_code(code, _fixture()) + assert result.infrastructure_failure is None, result.failures assert not result.passed assert any("modified its input file" in f for f in result.failures) @@ -145,6 +152,7 @@ def test_sandbox_pins_native_threads_and_enables_faulthandler(tmp_path): python = _fake_interpreter( tmp_path, "import json, os, sys\n" + f"print({gc._STARTED_MARKER!r})\n" "print(json.dumps({'argv': sys.argv[1:], 'env': dict(os.environ)}))\n", ) result = verify_generated_code(GOOD, _fixture(), python=python) @@ -212,7 +220,7 @@ def test_ordinary_failure_keeps_traceback_tail(monkeypatch): stderr = "noise\n" * 300 + "ValueError: bad column\n" def failed_child(args, **kwargs): - return subprocess.CompletedProcess(args, 1, "", stderr) + return subprocess.CompletedProcess(args, 1, _STARTED, stderr) monkeypatch.setattr(gc.subprocess, "run", failed_child) result = verify_generated_code(GOOD, _fixture()) @@ -234,7 +242,7 @@ def _scripted_children(monkeypatch, outcomes): def child(args, **kwargs): code, stderr = outcomes[min(len(calls), len(outcomes) - 1)] calls.append(code) - return subprocess.CompletedProcess(args, code, "", stderr) + return subprocess.CompletedProcess(args, code, _STARTED, stderr) monkeypatch.setattr(gc.subprocess, "run", child) return calls @@ -265,3 +273,117 @@ def test_ordinary_failure_is_not_retried(monkeypatch): assert calls == [1] assert not result.passed assert result.native_crashes == () + + +_POSIX_ENV_KEYS = { + "PYTHONDONTWRITEBYTECODE", + "PYTHONNOUSERSITE", + "FRESHDATA_NO_NETWORK", + "HOME", + "TMPDIR", + "OMP_NUM_THREADS", + "OPENBLAS_NUM_THREADS", + "MKL_NUM_THREADS", + "POLARS_MAX_THREADS", + "RAYON_NUM_THREADS", + "LANG", +} +_HOST_ENV = {"SYSTEMROOT": r"C:\Windows", "PATH": r"C:\bin", "USERPROFILE": r"C:\u"} + + +@pytest.mark.parametrize("key", ["SYSTEMROOT", "SystemRoot"]) +def test_sandbox_env_keeps_system_root_on_windows(tmp_path, key): + # CPython <= 3.10 on Windows cannot seed hash randomization without it. + host = {key: r"C:\Windows", "PATH": r"C:\bin", "USERPROFILE": r"C:\u"} + env = gc._sandbox_env(tmp_path, os_name="nt", environ=host) + assert env["SystemRoot"] == r"C:\Windows" + assert set(env) == _POSIX_ENV_KEYS | {"SystemRoot"} + + +def test_sandbox_env_on_posix_is_unchanged(tmp_path): + env = gc._sandbox_env(tmp_path, os_name="posix", environ=_HOST_ENV) + assert env == { + "PYTHONDONTWRITEBYTECODE": "1", + "PYTHONNOUSERSITE": "1", + "FRESHDATA_NO_NETWORK": "1", + "HOME": str(tmp_path), + "TMPDIR": str(tmp_path), + "OMP_NUM_THREADS": "1", + "OPENBLAS_NUM_THREADS": "1", + "MKL_NUM_THREADS": "1", + "POLARS_MAX_THREADS": "1", + "RAYON_NUM_THREADS": "1", + "LANG": "C.UTF-8", + } + + +def test_windows_child_receives_system_root(monkeypatch): + class _WindowsOs: + name = "nt" + environ = _HOST_ENV + + seen = {} + + def child(args, **kwargs): + seen.update(kwargs["env"]) + return subprocess.CompletedProcess(args, 0, _STARTED + "rows: 2\n", "") + + monkeypatch.setattr(gc, "os", _WindowsOs) + monkeypatch.setattr(gc.subprocess, "run", child) + result = verify_generated_code(GOOD, _fixture()) + assert result.passed, result.failures + assert seen["SystemRoot"] == r"C:\Windows" + assert "PATH" not in seen + assert result.stdout == "rows: 2\n" + + +_STARTUP_FATAL = ( + "Fatal Python error: _Py_HashRandomization_Init: failed to get random " + "numbers to initialize Python\nPython runtime state: preinitialized\n\n" +) + + +@pytest.mark.parametrize( + "code", + [ + GOOD, + 'import pandas as pd\nprint(pd.read_csv("your_data.csv")["memo"].tolist())\n', + 'import pandas as pd\npd.DataFrame().to_csv("your_data.csv")\n', + ], +) +def test_child_that_cannot_start_is_an_infrastructure_failure(monkeypatch, code): + # Seen on Windows + CPython 3.9: the child died at startup, and each case + # reported an ordinary "generated code exited 1", so the canary and + # overwrite checks passed vacuously. + calls = [] + + def child_never_started(args, **kwargs): + calls.append(args) + return subprocess.CompletedProcess(args, 1, "", _STARTUP_FATAL) + + monkeypatch.setattr(gc.subprocess, "run", child_never_started) + result = verify_generated_code(code, _fixture()) + assert not result.passed + assert result.infrastructure_failure + assert "_Py_HashRandomization_Init" in result.infrastructure_failure + assert result.failures == (result.infrastructure_failure,) + assert "execute" not in result.stages + assert not any("generated code exited" in f for f in result.failures) + assert len(calls) == 1 # a startup failure is not retried as a native crash + + +@pytest.mark.skipif(sys.platform == "win32", reason="POSIX shebang interpreter") +def test_child_exiting_cleanly_without_running_the_harness_fails_closed(tmp_path): + python = _fake_interpreter(tmp_path, "import sys\nsys.exit(0)\n") + result = verify_generated_code(GOOD, _fixture(), python=python) + assert not result.passed + assert "exited with code 0 before starting the harness" in ( + result.infrastructure_failure or "" + ) + + +def test_normal_run_reports_generated_stdout_without_the_start_marker(): + result = verify_generated_code(GOOD, _fixture()) + assert result.passed, result.failures + assert result.infrastructure_failure is None + assert result.stdout == "rows: 2\n" diff --git a/tests/truthbench/test_runner_report_cli.py b/tests/truthbench/test_runner_report_cli.py index e60f0129..94d651f9 100644 --- a/tests/truthbench/test_runner_report_cli.py +++ b/tests/truthbench/test_runner_report_cli.py @@ -14,6 +14,7 @@ import pytest from benchmarks.truthbench import cli from benchmarks.truthbench import runner as runner_module +from benchmarks.truthbench.generated_code import GeneratedCodeResult from benchmarks.truthbench.minimize import minimize_failure from benchmarks.truthbench.models import GateResult, RunResult from benchmarks.truthbench.report import compare_to_baseline, write_artifacts @@ -109,6 +110,34 @@ def missing(name: str) -> str: run_release(domains=("finance",), write=False) +def test_sandbox_child_that_cannot_start_is_an_infrastructure_failure( + monkeypatch, tmp_path, capsys +): + # A sandbox child that dies at interpreter startup never ran the generated + # code; the run must fail as infrastructure (exit 2), not grade a gate the + # regression ratchet could wave through. + reason = "sandbox infrastructure failure: the child interpreter exited with code 1" + + def child_never_started(code, fixture, **_kwargs): + return GeneratedCodeResult( + False, (reason,), stages=("parse", "allowlist", "compile"), + infrastructure_failure=reason, + ) + + monkeypatch.setattr(runner_module, "verify_generated_code", child_never_started) + code = cli.main( + [ + "run", "--domains", "finance", "--backends", "pandas", + "--results-dir", str(tmp_path), "--check-regressions", + ] + ) + assert code == 2 + err = capsys.readouterr().err + assert "INFRASTRUCTURE FAILURE" in err + assert "sandbox could not run" in err + assert not (tmp_path / "latest.json").exists() + + def test_minimizer_never_removes_the_target_cell(): fixture = parity_fixture() target = next(c for c in fixture.cells if c.row_id == "par-01" and c.column == "name")