Skip to content

Commit 2d94d64

Browse files
fix(truthbench): keep SystemRoot for the sandbox child on Windows (#431)
On Windows with CPython 3.10 and earlier, the generated-code sandbox child died at startup with "Fatal Python error: _Py_HashRandomization_Init: failed to get random numbers": its scrubbed environment dropped SystemRoot, which CryptGenRandom needs to seed hash randomization (3.11+ uses BCryptGenRandom and is unaffected). Every sandbox case then reported an ordinary "generated code exited 1", so the timeout, canary and input-overwrite checks never observed the generated code while the negative tests still saw a failure and passed. Keep SystemRoot in the child environment on nt only; the POSIX environment is unchanged. Also fail closed when the sandbox child cannot start: the harness now prints a start marker first, and a child that exits without it yields a GeneratedCodeResult with infrastructure_failure set, no "execute" stage, and none of the post-run checks. run_release raises TruthBenchRunError on such a result, so the CLI exits 2 (infrastructure failure) instead of grading the generated_code_sandbox gate. Signal exits keep the existing native-crash retry path.
1 parent bf19ba7 commit 2d94d64

4 files changed

Lines changed: 235 additions & 20 deletions

File tree

‎benchmarks/truthbench/generated_code.py‎

Lines changed: 73 additions & 17 deletions
Original file line numberDiff line numberDiff line change
@@ -12,9 +12,11 @@
1212
from __future__ import annotations
1313

1414
import ast
15+
import os
1516
import subprocess
1617
import sys
1718
import tempfile
19+
from collections.abc import Mapping
1820
from dataclasses import dataclass, field
1921
from pathlib import Path
2022
from typing import Any
@@ -97,6 +99,11 @@ def _stderr_excerpt(stderr: str) -> str:
9799
#: and every crash is still recorded in ``GeneratedCodeResult.native_crashes``.
98100
_NATIVE_CRASH_RETRIES = 1
99101

102+
#: The harness prints this line before anything else. A child that exits
103+
#: without it never ran the harness (the interpreter failed to start), so none
104+
#: of the post-run checks below observed the generated code.
105+
_STARTED_MARKER = "truthbench-sandbox-child-started"
106+
100107

101108
@dataclass(frozen=True)
102109
class GeneratedCodeResult:
@@ -111,6 +118,48 @@ class GeneratedCodeResult:
111118
#: One redacted excerpt per signal exit, including crashes that a retry
112119
#: recovered from, so a flaky pass is never silent.
113120
native_crashes: tuple[str, ...] = ()
121+
#: Set when the sandbox child could not start the harness. The generated
122+
#: code never ran, so this result says nothing about its behavior and the
123+
#: caller must treat it as an infrastructure failure, not a verdict.
124+
infrastructure_failure: str | None = None
125+
126+
127+
def _sandbox_env(
128+
workdir: Path,
129+
*,
130+
os_name: str | None = None,
131+
environ: Mapping[str, str] | None = None,
132+
) -> dict[str, str]:
133+
"""The deliberately minimal environment of the sandbox child."""
134+
135+
env = {
136+
"PYTHONDONTWRITEBYTECODE": "1",
137+
"PYTHONNOUSERSITE": "1",
138+
"FRESHDATA_NO_NETWORK": "1",
139+
"HOME": str(workdir),
140+
"TMPDIR": str(workdir),
141+
# The environment is otherwise empty, so native thread pools would
142+
# size themselves from the host core count; pin them (and the
143+
# locale) so runs are deterministic across machines.
144+
"OMP_NUM_THREADS": "1",
145+
"OPENBLAS_NUM_THREADS": "1",
146+
"MKL_NUM_THREADS": "1",
147+
"POLARS_MAX_THREADS": "1",
148+
"RAYON_NUM_THREADS": "1",
149+
"LANG": "C.UTF-8",
150+
}
151+
if (os_name if os_name is not None else os.name) == "nt":
152+
# CPython <= 3.10 on Windows seeds hash randomization through
153+
# CryptGenRandom, which cannot load its provider without SystemRoot:
154+
# the child dies with "Fatal Python error: _Py_HashRandomization_Init:
155+
# failed to get random numbers". 3.11+ uses BCryptGenRandom and starts
156+
# without it. Windows variable names are case-insensitive, so one key
157+
# covers SystemRoot and SYSTEMROOT.
158+
source = os.environ if environ is None else environ
159+
system_root = source.get("SystemRoot") or source.get("SYSTEMROOT")
160+
if system_root:
161+
env["SystemRoot"] = system_root
162+
return env
114163

115164

116165
def _allowlist_failures(tree: ast.AST) -> list[str]:
@@ -141,6 +190,10 @@ def _harness(code_path: str, input_name: str) -> str:
141190
return f"""
142191
import sys
143192
193+
# First output: proves the interpreter started and is running this harness.
194+
sys.stdout.write({_STARTED_MARKER!r} + "\\n")
195+
sys.stdout.flush()
196+
144197
# Pre-import the allowed libraries so their own internal use of stdlib
145198
# modules (pandas imports subprocess for locale probing) completes first...
146199
import pandas # noqa: F401
@@ -220,22 +273,7 @@ def leaks(label: str, payload: Any) -> None:
220273
harness_path.write_text(
221274
_harness(str(code_path), _INPUT_BASENAME), encoding="utf-8"
222275
)
223-
env = {
224-
"PYTHONDONTWRITEBYTECODE": "1",
225-
"PYTHONNOUSERSITE": "1",
226-
"FRESHDATA_NO_NETWORK": "1",
227-
"HOME": str(workdir),
228-
"TMPDIR": str(workdir),
229-
# The environment is otherwise empty, so native thread pools would
230-
# size themselves from the host core count; pin them (and the
231-
# locale) so runs are deterministic across machines.
232-
"OMP_NUM_THREADS": "1",
233-
"OPENBLAS_NUM_THREADS": "1",
234-
"MKL_NUM_THREADS": "1",
235-
"POLARS_MAX_THREADS": "1",
236-
"RAYON_NUM_THREADS": "1",
237-
"LANG": "C.UTF-8",
238-
}
276+
env = _sandbox_env(workdir)
239277
interpreter = python or sys.executable
240278

241279
def redacted(text: str) -> str:
@@ -270,8 +308,26 @@ def redacted(text: str) -> str:
270308
stages=(*stages, "execute"),
271309
native_crashes=tuple(native_crashes),
272310
)
311+
stdout, stderr = proc.stdout or "", proc.stderr or ""
312+
started = f"{_STARTED_MARKER}\n"
313+
if proc.returncode >= 0 and not stdout.startswith(started):
314+
# Fail closed: the generated code never ran, so the overwrite and
315+
# canary checks below would observe nothing; report no verdict.
316+
reason = (
317+
"sandbox infrastructure failure: the child interpreter exited "
318+
f"with code {proc.returncode} before starting the harness: "
319+
f"{_stderr_excerpt(redacted(stderr))}"
320+
)
321+
return GeneratedCodeResult(
322+
False,
323+
(*failures, reason),
324+
stderr=redacted(stderr),
325+
stages=tuple(stages),
326+
native_crashes=tuple(native_crashes),
327+
infrastructure_failure=reason,
328+
)
329+
stdout = stdout[len(started) :]
273330
stages.append("execute")
274-
stdout, stderr = proc.stdout, proc.stderr
275331
if proc.returncode != 0:
276332
failures.append(
277333
f"generated code exited {proc.returncode}: "

‎benchmarks/truthbench/runner.py‎

Lines changed: 8 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -742,6 +742,14 @@ def run_release(
742742
file=sys.stderr,
743743
)
744744

745+
if any(result.infrastructure_failure for result in generated_results):
746+
# The sandbox child never ran the generated code, so the sandbox
747+
# checks observed nothing; that is not a gate verdict.
748+
raise TruthBenchRunError(
749+
"generated-code sandbox could not run: "
750+
+ "; ".join(sandbox_failures)
751+
)
752+
745753
observations, failures, parity_ledger = _parity_observations(parity, backends)
746754
parity_failures.extend(
747755
failure for failure in failures if failure not in parity_failures

‎tests/truthbench/test_generated_code.py‎

Lines changed: 125 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -47,6 +47,9 @@ def _fixture():
4747
print("rows:", len(df))
4848
"""
4949

50+
#: What a child that started the harness prints first.
51+
_STARTED = gc._STARTED_MARKER + "\n"
52+
5053

5154
def test_wellformed_code_passes_all_stages():
5255
result = verify_generated_code(GOOD, _fixture())
@@ -90,14 +93,16 @@ def test_runtime_poison_blocks_banned_module_even_if_statically_allowed(monkeypa
9093
gc, "ALLOWED_IMPORTS", frozenset({*gc.ALLOWED_IMPORTS, "socket"})
9194
)
9295
result = verify_generated_code("import socket\n", _fixture())
96+
assert result.infrastructure_failure is None, result.failures
9397
assert not result.passed
94-
assert any("exited" in f for f in result.failures)
98+
assert any("generated code exited" in f for f in result.failures)
9599

96100

97101
def test_timeout_is_enforced():
98102
result = verify_generated_code(
99103
"while True:\n pass\n", _fixture(), timeout=3.0
100104
)
105+
assert result.infrastructure_failure is None, result.failures
101106
assert not result.passed
102107
assert any("timeout" in f for f in result.failures)
103108

@@ -116,6 +121,7 @@ def test_pii_canary_in_stdout_is_reported():
116121
'print(df["memo"].tolist())\n'
117122
)
118123
result = verify_generated_code(code, _fixture())
124+
assert result.infrastructure_failure is None, result.failures
119125
assert not result.passed
120126
assert any("stdout leaked canary" in f for f in result.failures)
121127
assert "example.invalid" not in result.stdout # evidence itself is redacted
@@ -128,6 +134,7 @@ def test_input_file_overwrite_is_reported():
128134
'df.head(1).to_csv("your_data.csv", index=False)\n'
129135
)
130136
result = verify_generated_code(code, _fixture())
137+
assert result.infrastructure_failure is None, result.failures
131138
assert not result.passed
132139
assert any("modified its input file" in f for f in result.failures)
133140

@@ -145,6 +152,7 @@ def test_sandbox_pins_native_threads_and_enables_faulthandler(tmp_path):
145152
python = _fake_interpreter(
146153
tmp_path,
147154
"import json, os, sys\n"
155+
f"print({gc._STARTED_MARKER!r})\n"
148156
"print(json.dumps({'argv': sys.argv[1:], 'env': dict(os.environ)}))\n",
149157
)
150158
result = verify_generated_code(GOOD, _fixture(), python=python)
@@ -212,7 +220,7 @@ def test_ordinary_failure_keeps_traceback_tail(monkeypatch):
212220
stderr = "noise\n" * 300 + "ValueError: bad column\n"
213221

214222
def failed_child(args, **kwargs):
215-
return subprocess.CompletedProcess(args, 1, "", stderr)
223+
return subprocess.CompletedProcess(args, 1, _STARTED, stderr)
216224

217225
monkeypatch.setattr(gc.subprocess, "run", failed_child)
218226
result = verify_generated_code(GOOD, _fixture())
@@ -234,7 +242,7 @@ def _scripted_children(monkeypatch, outcomes):
234242
def child(args, **kwargs):
235243
code, stderr = outcomes[min(len(calls), len(outcomes) - 1)]
236244
calls.append(code)
237-
return subprocess.CompletedProcess(args, code, "", stderr)
245+
return subprocess.CompletedProcess(args, code, _STARTED, stderr)
238246

239247
monkeypatch.setattr(gc.subprocess, "run", child)
240248
return calls
@@ -265,3 +273,117 @@ def test_ordinary_failure_is_not_retried(monkeypatch):
265273
assert calls == [1]
266274
assert not result.passed
267275
assert result.native_crashes == ()
276+
277+
278+
_POSIX_ENV_KEYS = {
279+
"PYTHONDONTWRITEBYTECODE",
280+
"PYTHONNOUSERSITE",
281+
"FRESHDATA_NO_NETWORK",
282+
"HOME",
283+
"TMPDIR",
284+
"OMP_NUM_THREADS",
285+
"OPENBLAS_NUM_THREADS",
286+
"MKL_NUM_THREADS",
287+
"POLARS_MAX_THREADS",
288+
"RAYON_NUM_THREADS",
289+
"LANG",
290+
}
291+
_HOST_ENV = {"SYSTEMROOT": r"C:\Windows", "PATH": r"C:\bin", "USERPROFILE": r"C:\u"}
292+
293+
294+
@pytest.mark.parametrize("key", ["SYSTEMROOT", "SystemRoot"])
295+
def test_sandbox_env_keeps_system_root_on_windows(tmp_path, key):
296+
# CPython <= 3.10 on Windows cannot seed hash randomization without it.
297+
host = {key: r"C:\Windows", "PATH": r"C:\bin", "USERPROFILE": r"C:\u"}
298+
env = gc._sandbox_env(tmp_path, os_name="nt", environ=host)
299+
assert env["SystemRoot"] == r"C:\Windows"
300+
assert set(env) == _POSIX_ENV_KEYS | {"SystemRoot"}
301+
302+
303+
def test_sandbox_env_on_posix_is_unchanged(tmp_path):
304+
env = gc._sandbox_env(tmp_path, os_name="posix", environ=_HOST_ENV)
305+
assert env == {
306+
"PYTHONDONTWRITEBYTECODE": "1",
307+
"PYTHONNOUSERSITE": "1",
308+
"FRESHDATA_NO_NETWORK": "1",
309+
"HOME": str(tmp_path),
310+
"TMPDIR": str(tmp_path),
311+
"OMP_NUM_THREADS": "1",
312+
"OPENBLAS_NUM_THREADS": "1",
313+
"MKL_NUM_THREADS": "1",
314+
"POLARS_MAX_THREADS": "1",
315+
"RAYON_NUM_THREADS": "1",
316+
"LANG": "C.UTF-8",
317+
}
318+
319+
320+
def test_windows_child_receives_system_root(monkeypatch):
321+
class _WindowsOs:
322+
name = "nt"
323+
environ = _HOST_ENV
324+
325+
seen = {}
326+
327+
def child(args, **kwargs):
328+
seen.update(kwargs["env"])
329+
return subprocess.CompletedProcess(args, 0, _STARTED + "rows: 2\n", "")
330+
331+
monkeypatch.setattr(gc, "os", _WindowsOs)
332+
monkeypatch.setattr(gc.subprocess, "run", child)
333+
result = verify_generated_code(GOOD, _fixture())
334+
assert result.passed, result.failures
335+
assert seen["SystemRoot"] == r"C:\Windows"
336+
assert "PATH" not in seen
337+
assert result.stdout == "rows: 2\n"
338+
339+
340+
_STARTUP_FATAL = (
341+
"Fatal Python error: _Py_HashRandomization_Init: failed to get random "
342+
"numbers to initialize Python\nPython runtime state: preinitialized\n\n"
343+
)
344+
345+
346+
@pytest.mark.parametrize(
347+
"code",
348+
[
349+
GOOD,
350+
'import pandas as pd\nprint(pd.read_csv("your_data.csv")["memo"].tolist())\n',
351+
'import pandas as pd\npd.DataFrame().to_csv("your_data.csv")\n',
352+
],
353+
)
354+
def test_child_that_cannot_start_is_an_infrastructure_failure(monkeypatch, code):
355+
# Seen on Windows + CPython 3.9: the child died at startup, and each case
356+
# reported an ordinary "generated code exited 1", so the canary and
357+
# overwrite checks passed vacuously.
358+
calls = []
359+
360+
def child_never_started(args, **kwargs):
361+
calls.append(args)
362+
return subprocess.CompletedProcess(args, 1, "", _STARTUP_FATAL)
363+
364+
monkeypatch.setattr(gc.subprocess, "run", child_never_started)
365+
result = verify_generated_code(code, _fixture())
366+
assert not result.passed
367+
assert result.infrastructure_failure
368+
assert "_Py_HashRandomization_Init" in result.infrastructure_failure
369+
assert result.failures == (result.infrastructure_failure,)
370+
assert "execute" not in result.stages
371+
assert not any("generated code exited" in f for f in result.failures)
372+
assert len(calls) == 1 # a startup failure is not retried as a native crash
373+
374+
375+
@pytest.mark.skipif(sys.platform == "win32", reason="POSIX shebang interpreter")
376+
def test_child_exiting_cleanly_without_running_the_harness_fails_closed(tmp_path):
377+
python = _fake_interpreter(tmp_path, "import sys\nsys.exit(0)\n")
378+
result = verify_generated_code(GOOD, _fixture(), python=python)
379+
assert not result.passed
380+
assert "exited with code 0 before starting the harness" in (
381+
result.infrastructure_failure or ""
382+
)
383+
384+
385+
def test_normal_run_reports_generated_stdout_without_the_start_marker():
386+
result = verify_generated_code(GOOD, _fixture())
387+
assert result.passed, result.failures
388+
assert result.infrastructure_failure is None
389+
assert result.stdout == "rows: 2\n"

‎tests/truthbench/test_runner_report_cli.py‎

Lines changed: 29 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -14,6 +14,7 @@
1414
import pytest
1515
from benchmarks.truthbench import cli
1616
from benchmarks.truthbench import runner as runner_module
17+
from benchmarks.truthbench.generated_code import GeneratedCodeResult
1718
from benchmarks.truthbench.minimize import minimize_failure
1819
from benchmarks.truthbench.models import GateResult, RunResult
1920
from benchmarks.truthbench.report import compare_to_baseline, write_artifacts
@@ -109,6 +110,34 @@ def missing(name: str) -> str:
109110
run_release(domains=("finance",), write=False)
110111

111112

113+
def test_sandbox_child_that_cannot_start_is_an_infrastructure_failure(
114+
monkeypatch, tmp_path, capsys
115+
):
116+
# A sandbox child that dies at interpreter startup never ran the generated
117+
# code; the run must fail as infrastructure (exit 2), not grade a gate the
118+
# regression ratchet could wave through.
119+
reason = "sandbox infrastructure failure: the child interpreter exited with code 1"
120+
121+
def child_never_started(code, fixture, **_kwargs):
122+
return GeneratedCodeResult(
123+
False, (reason,), stages=("parse", "allowlist", "compile"),
124+
infrastructure_failure=reason,
125+
)
126+
127+
monkeypatch.setattr(runner_module, "verify_generated_code", child_never_started)
128+
code = cli.main(
129+
[
130+
"run", "--domains", "finance", "--backends", "pandas",
131+
"--results-dir", str(tmp_path), "--check-regressions",
132+
]
133+
)
134+
assert code == 2
135+
err = capsys.readouterr().err
136+
assert "INFRASTRUCTURE FAILURE" in err
137+
assert "sandbox could not run" in err
138+
assert not (tmp_path / "latest.json").exists()
139+
140+
112141
def test_minimizer_never_removes_the_target_cell():
113142
fixture = parity_fixture()
114143
target = next(c for c in fixture.cells if c.row_id == "par-01" and c.column == "name")

0 commit comments

Comments
 (0)