Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
90 changes: 73 additions & 17 deletions benchmarks/truthbench/generated_code.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,9 +12,11 @@
from __future__ import annotations

import ast
import os
import subprocess
import sys
import tempfile
from collections.abc import Mapping
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any
Expand Down Expand Up @@ -97,6 +99,11 @@ def _stderr_excerpt(stderr: str) -> str:
#: and every crash is still recorded in ``GeneratedCodeResult.native_crashes``.
_NATIVE_CRASH_RETRIES = 1

#: The harness prints this line before anything else. A child that exits
#: without it never ran the harness (the interpreter failed to start), so none
#: of the post-run checks below observed the generated code.
_STARTED_MARKER = "truthbench-sandbox-child-started"


@dataclass(frozen=True)
class GeneratedCodeResult:
Expand All @@ -111,6 +118,48 @@ class GeneratedCodeResult:
#: One redacted excerpt per signal exit, including crashes that a retry
#: recovered from, so a flaky pass is never silent.
native_crashes: tuple[str, ...] = ()
#: Set when the sandbox child could not start the harness. The generated
#: code never ran, so this result says nothing about its behavior and the
#: caller must treat it as an infrastructure failure, not a verdict.
infrastructure_failure: str | None = None


def _sandbox_env(
workdir: Path,
*,
os_name: str | None = None,
environ: Mapping[str, str] | None = None,
) -> dict[str, str]:
"""The deliberately minimal environment of the sandbox child."""

env = {
"PYTHONDONTWRITEBYTECODE": "1",
"PYTHONNOUSERSITE": "1",
"FRESHDATA_NO_NETWORK": "1",
"HOME": str(workdir),
"TMPDIR": str(workdir),
# The environment is otherwise empty, so native thread pools would
# size themselves from the host core count; pin them (and the
# locale) so runs are deterministic across machines.
"OMP_NUM_THREADS": "1",
"OPENBLAS_NUM_THREADS": "1",
"MKL_NUM_THREADS": "1",
"POLARS_MAX_THREADS": "1",
"RAYON_NUM_THREADS": "1",
"LANG": "C.UTF-8",
}
if (os_name if os_name is not None else os.name) == "nt":
# CPython <= 3.10 on Windows seeds hash randomization through
# CryptGenRandom, which cannot load its provider without SystemRoot:
# the child dies with "Fatal Python error: _Py_HashRandomization_Init:
# failed to get random numbers". 3.11+ uses BCryptGenRandom and starts
# without it. Windows variable names are case-insensitive, so one key
# covers SystemRoot and SYSTEMROOT.
source = os.environ if environ is None else environ
system_root = source.get("SystemRoot") or source.get("SYSTEMROOT")
if system_root:
env["SystemRoot"] = system_root
return env


def _allowlist_failures(tree: ast.AST) -> list[str]:
Expand Down Expand Up @@ -141,6 +190,10 @@ def _harness(code_path: str, input_name: str) -> str:
return f"""
import sys

# First output: proves the interpreter started and is running this harness.
sys.stdout.write({_STARTED_MARKER!r} + "\\n")
sys.stdout.flush()

# Pre-import the allowed libraries so their own internal use of stdlib
# modules (pandas imports subprocess for locale probing) completes first...
import pandas # noqa: F401
Expand Down Expand Up @@ -220,22 +273,7 @@ def leaks(label: str, payload: Any) -> None:
harness_path.write_text(
_harness(str(code_path), _INPUT_BASENAME), encoding="utf-8"
)
env = {
"PYTHONDONTWRITEBYTECODE": "1",
"PYTHONNOUSERSITE": "1",
"FRESHDATA_NO_NETWORK": "1",
"HOME": str(workdir),
"TMPDIR": str(workdir),
# The environment is otherwise empty, so native thread pools would
# size themselves from the host core count; pin them (and the
# locale) so runs are deterministic across machines.
"OMP_NUM_THREADS": "1",
"OPENBLAS_NUM_THREADS": "1",
"MKL_NUM_THREADS": "1",
"POLARS_MAX_THREADS": "1",
"RAYON_NUM_THREADS": "1",
"LANG": "C.UTF-8",
}
env = _sandbox_env(workdir)
interpreter = python or sys.executable

def redacted(text: str) -> str:
Expand Down Expand Up @@ -270,8 +308,26 @@ def redacted(text: str) -> str:
stages=(*stages, "execute"),
native_crashes=tuple(native_crashes),
)
stdout, stderr = proc.stdout or "", proc.stderr or ""
started = f"{_STARTED_MARKER}\n"
if proc.returncode >= 0 and not stdout.startswith(started):
# Fail closed: the generated code never ran, so the overwrite and
# canary checks below would observe nothing; report no verdict.
reason = (
"sandbox infrastructure failure: the child interpreter exited "
f"with code {proc.returncode} before starting the harness: "
f"{_stderr_excerpt(redacted(stderr))}"
)
return GeneratedCodeResult(
False,
(*failures, reason),
stderr=redacted(stderr),
stages=tuple(stages),
native_crashes=tuple(native_crashes),
infrastructure_failure=reason,
)
stdout = stdout[len(started) :]
stages.append("execute")
stdout, stderr = proc.stdout, proc.stderr
if proc.returncode != 0:
failures.append(
f"generated code exited {proc.returncode}: "
Expand Down
8 changes: 8 additions & 0 deletions benchmarks/truthbench/runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -742,6 +742,14 @@ def run_release(
file=sys.stderr,
)

if any(result.infrastructure_failure for result in generated_results):
# The sandbox child never ran the generated code, so the sandbox
# checks observed nothing; that is not a gate verdict.
raise TruthBenchRunError(
"generated-code sandbox could not run: "
+ "; ".join(sandbox_failures)
)

observations, failures, parity_ledger = _parity_observations(parity, backends)
parity_failures.extend(
failure for failure in failures if failure not in parity_failures
Expand Down
128 changes: 125 additions & 3 deletions tests/truthbench/test_generated_code.py
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,9 @@ def _fixture():
print("rows:", len(df))
"""

#: What a child that started the harness prints first.
_STARTED = gc._STARTED_MARKER + "\n"


def test_wellformed_code_passes_all_stages():
result = verify_generated_code(GOOD, _fixture())
Expand Down Expand Up @@ -90,14 +93,16 @@ def test_runtime_poison_blocks_banned_module_even_if_statically_allowed(monkeypa
gc, "ALLOWED_IMPORTS", frozenset({*gc.ALLOWED_IMPORTS, "socket"})
)
result = verify_generated_code("import socket\n", _fixture())
assert result.infrastructure_failure is None, result.failures
assert not result.passed
assert any("exited" in f for f in result.failures)
assert any("generated code exited" in f for f in result.failures)


def test_timeout_is_enforced():
result = verify_generated_code(
"while True:\n pass\n", _fixture(), timeout=3.0
)
assert result.infrastructure_failure is None, result.failures
assert not result.passed
assert any("timeout" in f for f in result.failures)

Expand All @@ -116,6 +121,7 @@ def test_pii_canary_in_stdout_is_reported():
'print(df["memo"].tolist())\n'
)
result = verify_generated_code(code, _fixture())
assert result.infrastructure_failure is None, result.failures
assert not result.passed
assert any("stdout leaked canary" in f for f in result.failures)
assert "example.invalid" not in result.stdout # evidence itself is redacted
Expand All @@ -128,6 +134,7 @@ def test_input_file_overwrite_is_reported():
'df.head(1).to_csv("your_data.csv", index=False)\n'
)
result = verify_generated_code(code, _fixture())
assert result.infrastructure_failure is None, result.failures
assert not result.passed
assert any("modified its input file" in f for f in result.failures)

Expand All @@ -145,6 +152,7 @@ def test_sandbox_pins_native_threads_and_enables_faulthandler(tmp_path):
python = _fake_interpreter(
tmp_path,
"import json, os, sys\n"
f"print({gc._STARTED_MARKER!r})\n"
"print(json.dumps({'argv': sys.argv[1:], 'env': dict(os.environ)}))\n",
)
result = verify_generated_code(GOOD, _fixture(), python=python)
Expand Down Expand Up @@ -212,7 +220,7 @@ def test_ordinary_failure_keeps_traceback_tail(monkeypatch):
stderr = "noise\n" * 300 + "ValueError: bad column\n"

def failed_child(args, **kwargs):
return subprocess.CompletedProcess(args, 1, "", stderr)
return subprocess.CompletedProcess(args, 1, _STARTED, stderr)

monkeypatch.setattr(gc.subprocess, "run", failed_child)
result = verify_generated_code(GOOD, _fixture())
Expand All @@ -234,7 +242,7 @@ def _scripted_children(monkeypatch, outcomes):
def child(args, **kwargs):
code, stderr = outcomes[min(len(calls), len(outcomes) - 1)]
calls.append(code)
return subprocess.CompletedProcess(args, code, "", stderr)
return subprocess.CompletedProcess(args, code, _STARTED, stderr)

monkeypatch.setattr(gc.subprocess, "run", child)
return calls
Expand Down Expand Up @@ -265,3 +273,117 @@ def test_ordinary_failure_is_not_retried(monkeypatch):
assert calls == [1]
assert not result.passed
assert result.native_crashes == ()


_POSIX_ENV_KEYS = {
"PYTHONDONTWRITEBYTECODE",
"PYTHONNOUSERSITE",
"FRESHDATA_NO_NETWORK",
"HOME",
"TMPDIR",
"OMP_NUM_THREADS",
"OPENBLAS_NUM_THREADS",
"MKL_NUM_THREADS",
"POLARS_MAX_THREADS",
"RAYON_NUM_THREADS",
"LANG",
}
_HOST_ENV = {"SYSTEMROOT": r"C:\Windows", "PATH": r"C:\bin", "USERPROFILE": r"C:\u"}


@pytest.mark.parametrize("key", ["SYSTEMROOT", "SystemRoot"])
def test_sandbox_env_keeps_system_root_on_windows(tmp_path, key):
# CPython <= 3.10 on Windows cannot seed hash randomization without it.
host = {key: r"C:\Windows", "PATH": r"C:\bin", "USERPROFILE": r"C:\u"}
env = gc._sandbox_env(tmp_path, os_name="nt", environ=host)
assert env["SystemRoot"] == r"C:\Windows"
assert set(env) == _POSIX_ENV_KEYS | {"SystemRoot"}


def test_sandbox_env_on_posix_is_unchanged(tmp_path):
env = gc._sandbox_env(tmp_path, os_name="posix", environ=_HOST_ENV)
assert env == {
"PYTHONDONTWRITEBYTECODE": "1",
"PYTHONNOUSERSITE": "1",
"FRESHDATA_NO_NETWORK": "1",
"HOME": str(tmp_path),
"TMPDIR": str(tmp_path),
"OMP_NUM_THREADS": "1",
"OPENBLAS_NUM_THREADS": "1",
"MKL_NUM_THREADS": "1",
"POLARS_MAX_THREADS": "1",
"RAYON_NUM_THREADS": "1",
"LANG": "C.UTF-8",
}


def test_windows_child_receives_system_root(monkeypatch):
class _WindowsOs:
name = "nt"
environ = _HOST_ENV

seen = {}

def child(args, **kwargs):
seen.update(kwargs["env"])
return subprocess.CompletedProcess(args, 0, _STARTED + "rows: 2\n", "")

monkeypatch.setattr(gc, "os", _WindowsOs)
monkeypatch.setattr(gc.subprocess, "run", child)
result = verify_generated_code(GOOD, _fixture())
assert result.passed, result.failures
assert seen["SystemRoot"] == r"C:\Windows"
assert "PATH" not in seen
assert result.stdout == "rows: 2\n"


_STARTUP_FATAL = (
"Fatal Python error: _Py_HashRandomization_Init: failed to get random "
"numbers to initialize Python\nPython runtime state: preinitialized\n\n"
)


@pytest.mark.parametrize(
"code",
[
GOOD,
'import pandas as pd\nprint(pd.read_csv("your_data.csv")["memo"].tolist())\n',
'import pandas as pd\npd.DataFrame().to_csv("your_data.csv")\n',
],
)
def test_child_that_cannot_start_is_an_infrastructure_failure(monkeypatch, code):
# Seen on Windows + CPython 3.9: the child died at startup, and each case
# reported an ordinary "generated code exited 1", so the canary and
# overwrite checks passed vacuously.
calls = []

def child_never_started(args, **kwargs):
calls.append(args)
return subprocess.CompletedProcess(args, 1, "", _STARTUP_FATAL)

monkeypatch.setattr(gc.subprocess, "run", child_never_started)
result = verify_generated_code(code, _fixture())
assert not result.passed
assert result.infrastructure_failure
assert "_Py_HashRandomization_Init" in result.infrastructure_failure
assert result.failures == (result.infrastructure_failure,)
assert "execute" not in result.stages
assert not any("generated code exited" in f for f in result.failures)
assert len(calls) == 1 # a startup failure is not retried as a native crash


@pytest.mark.skipif(sys.platform == "win32", reason="POSIX shebang interpreter")
def test_child_exiting_cleanly_without_running_the_harness_fails_closed(tmp_path):
python = _fake_interpreter(tmp_path, "import sys\nsys.exit(0)\n")
result = verify_generated_code(GOOD, _fixture(), python=python)
assert not result.passed
assert "exited with code 0 before starting the harness" in (
result.infrastructure_failure or ""
)


def test_normal_run_reports_generated_stdout_without_the_start_marker():
result = verify_generated_code(GOOD, _fixture())
assert result.passed, result.failures
assert result.infrastructure_failure is None
assert result.stdout == "rows: 2\n"
29 changes: 29 additions & 0 deletions tests/truthbench/test_runner_report_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
import pytest
from benchmarks.truthbench import cli
from benchmarks.truthbench import runner as runner_module
from benchmarks.truthbench.generated_code import GeneratedCodeResult
from benchmarks.truthbench.minimize import minimize_failure
from benchmarks.truthbench.models import GateResult, RunResult
from benchmarks.truthbench.report import compare_to_baseline, write_artifacts
Expand Down Expand Up @@ -109,6 +110,34 @@ def missing(name: str) -> str:
run_release(domains=("finance",), write=False)


def test_sandbox_child_that_cannot_start_is_an_infrastructure_failure(
monkeypatch, tmp_path, capsys
):
# A sandbox child that dies at interpreter startup never ran the generated
# code; the run must fail as infrastructure (exit 2), not grade a gate the
# regression ratchet could wave through.
reason = "sandbox infrastructure failure: the child interpreter exited with code 1"

def child_never_started(code, fixture, **_kwargs):
return GeneratedCodeResult(
False, (reason,), stages=("parse", "allowlist", "compile"),
infrastructure_failure=reason,
)

monkeypatch.setattr(runner_module, "verify_generated_code", child_never_started)
code = cli.main(
[
"run", "--domains", "finance", "--backends", "pandas",
"--results-dir", str(tmp_path), "--check-regressions",
]
)
assert code == 2
err = capsys.readouterr().err
assert "INFRASTRUCTURE FAILURE" in err
assert "sandbox could not run" in err
assert not (tmp_path / "latest.json").exists()


def test_minimizer_never_removes_the_target_cell():
fixture = parity_fixture()
target = next(c for c in fixture.cells if c.row_id == "par-01" and c.column == "name")
Expand Down
Loading