diff --git a/.github/workflows/perf-regression.yml b/.github/workflows/perf-regression.yml index b55a2a69..64ef21a9 100644 --- a/.github/workflows/perf-regression.yml +++ b/.github/workflows/perf-regression.yml @@ -1,58 +1,128 @@ name: Performance regression -# Nightly (and release-candidate) T5 perf gate: runtime slowdown <= 20% and -# memory overhead <= 15% versus the cached v1.0-equivalent baseline. +# Same-run A/B perf gate on the CleanBench T5 workload: HEAD is measured against +# a base commit on the same runner, interleaved, and fails only when it is >20% +# slower or uses >15% more peak memory than base AND a confirmation re-run +# reproduces the breach. No absolute numbers from other machines are compared. +# +# pull_request PR head vs the PR base commit (blocking) +# daily main vs main as of ~26 hours earlier (blocking, alert issue) +# weekly main vs the latest release tag (report only) +# manual any base_ref; accept_regression=true reports without failing on: schedule: - cron: "0 5 * * *" + - cron: "30 5 * * 1" + pull_request: + paths: + - "src/freshdata/**" + - "benchmarks/cleanbench/**" + - ".github/workflows/perf-regression.yml" workflow_dispatch: inputs: - update_baseline: - description: "Re-pin the perf baseline to this run" + base_ref: + description: "Base ref to compare against (default: main as of 26h ago)" + type: string + default: "" + accept_regression: + description: "Report only; do not fail on a regression" type: boolean default: false +concurrency: + group: perf-regression-${{ github.ref }} + cancel-in-progress: true + +permissions: + contents: read + jobs: perf: runs-on: ubuntu-latest - timeout-minutes: 20 + timeout-minutes: 45 permissions: - issues: write contents: read + issues: write steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + fetch-depth: 0 # the base commit and release tags must be resolvable - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: python-version: "3.12" + cache: pip - name: Install run: | python -m pip install -U pip pip install -e ".[bench]" - - name: Restore perf baseline - uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 - with: - path: benchmarks/cleanbench/results/baseline_v1.json - key: cleanbench-baseline-${{ runner.os }}-v1 - - name: T5 perf gate + - name: Resolve base commit + id: base + env: + EVENT: ${{ github.event_name }} + SCHEDULE: ${{ github.event.schedule }} + PR_BASE_SHA: ${{ github.event.pull_request.base.sha }} + INPUT_BASE_REF: ${{ inputs.base_ref }} + INPUT_ACCEPT: ${{ inputs.accept_regression }} + run: | + set -euo pipefail + mode=gate + if [ "$EVENT" = "pull_request" ]; then + ref="$PR_BASE_SHA" + elif [ "$EVENT" = "workflow_dispatch" ] && [ -n "$INPUT_BASE_REF" ]; then + ref="$INPUT_BASE_REF" + elif [ "$SCHEDULE" = "30 5 * * 1" ]; then + ref="$(git describe --tags --abbrev=0 --match 'v*' origin/main)" + mode=report + else + ref="$(git rev-list -1 --before='26 hours ago' origin/main)" + fi + if [ "$INPUT_ACCEPT" = "true" ]; then mode=report; fi + sha="$(git rev-parse --verify "${ref}^{commit}")" + { + echo "ref=$ref" + echo "sha=$sha" + echo "mode=$mode" + } >> "$GITHUB_OUTPUT" + if [ "$sha" = "$(git rev-parse HEAD)" ]; then + echo "skip=true" >> "$GITHUB_OUTPUT" + echo "Base \`$ref\` is HEAD; nothing to compare." >> "$GITHUB_STEP_SUMMARY" + else + echo "skip=false" >> "$GITHUB_OUTPUT" + git worktree add --detach "$RUNNER_TEMP/perf-base" "$sha" + fi + - name: Same-run A/B perf gate + if: steps.base.outputs.skip != 'true' + env: + BASE_REF: ${{ steps.base.outputs.ref }} + BASE_SHA: ${{ steps.base.outputs.sha }} + MODE: ${{ steps.base.outputs.mode }} run: | - EXTRA="" - if [ "${{ inputs.update_baseline }}" = "true" ]; then EXTRA="--update-baseline"; fi - python -m benchmarks.cleanbench --tracks T5 --check-gates $EXTRA + GATE="" + if [ "$MODE" = "gate" ]; then GATE="--check-gates"; fi + python -m benchmarks.cleanbench.ab \ + --base-src "$RUNNER_TEMP/perf-base/src" \ + --head-src src \ + --base-label "$BASE_REF (${BASE_SHA:0:7})" \ + --head-label "HEAD (${GITHUB_SHA:0:7})" \ + --output benchmarks/cleanbench/results/latest.ab.json \ + --summary "$GITHUB_STEP_SUMMARY" \ + $GATE - name: Upload results - if: always() + if: always() && steps.base.outputs.skip == 'false' uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: - name: perf-results - path: benchmarks/cleanbench/results/latest.* + name: perf-ab-results + path: benchmarks/cleanbench/results/latest.ab.json retention-days: 30 - name: Open/refresh alert issue on failure - if: failure() + if: failure() && github.event_name == 'schedule' uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 with: script: | const title = "nightly perf-regression gate failing"; - const body = `The scheduled T5 performance-regression gate failed ` + - `(runtime > 120% or memory > 115% of baseline).\n\n` + + const body = `The scheduled same-run A/B performance gate failed: main ran ` + + `>20% slower or used >15% more peak memory than main from ~26 hours ` + + `earlier, reproduced on a confirmation run (or the job errored).\n\n` + `Run: ${context.serverUrl}/${context.repo.owner}/${context.repo.repo}` + `/actions/runs/${context.runId}`; const open = await github.rest.issues.listForRepo({ diff --git a/benchmarks/cleanbench/ab.py b/benchmarks/cleanbench/ab.py new file mode 100644 index 00000000..cfa6b270 --- /dev/null +++ b/benchmarks/cleanbench/ab.py @@ -0,0 +1,350 @@ +"""Same-run A/B performance gate on the CleanBench T5 workload. + +The old T5 gate compared this run's wall-clock against a number recorded once +on a different machine, so it passed or failed with runner noise. This module +measures a *base* and a *head* checkout of freshdata in the same job, on the +same runner, and gates on their ratio instead: + +* every measurement runs in a fresh worker subprocess whose ``PYTHONPATH`` puts + that side's ``src`` first; the harness and the fixture always come from HEAD, + and the worker refuses to run if freshdata was imported from anywhere else; +* sides alternate (base/head, then head/base) across pairs, so slow drift on + the runner hits both equally; +* runtime compares the fastest run of each side (the least noisy estimator of + the true cost) and memory the median peak-RSS delta per worker; +* a breach only fails the gate when a full confirmation re-run breaches too. + +Usage:: + + python -m benchmarks.cleanbench.ab --base-src ../base/src --head-src src --check-gates +""" + +from __future__ import annotations + +import argparse +import gc +import json +import os +import statistics +import subprocess +import sys +import time +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Callable + +from .metrics import FULL_GATE_MEMORY_OVERHEAD, FULL_GATE_RUNTIME_SLOWDOWN + +REPO_ROOT = Path(__file__).resolve().parents[2] +RESULTS_DIR = REPO_ROOT / "benchmarks" / "cleanbench" / "results" + +#: Below this base peak-RSS delta the memory ratio is dominated by allocator +#: noise, so the memory gate is reported as not applicable. +MIN_RSS_DELTA_BYTES = 4 * 1024 * 1024 + +Worker = Callable[..., dict[str, Any]] + + +@dataclass +class SideSamples: + """Everything measured for one side (base or head) across all workers.""" + + seconds: list[float] = field(default_factory=list) + rss_deltas: list[int] = field(default_factory=list) + freshdata_path: str = "" + + +# --------------------------------------------------------------------------- # +# worker (runs inside a subprocess with one side's src first on sys.path) +# --------------------------------------------------------------------------- # + +def _worker(src: str, *, target_rows: int, warmups: int, runs: int) -> dict[str, Any]: + import freshdata as fd # noqa: PLC0415 - must resolve from this side's src + + loaded = Path(fd.__file__).resolve() + expected = Path(src).resolve() + if expected not in loaded.parents: + raise SystemExit(f"freshdata was imported from {loaded}, expected under {expected}") + + from .fixtures import make_t5_scale_fixture # noqa: PLC0415 + from .runner import _PeakRss # noqa: PLC0415 + + _truth, corrupted, kwargs = make_t5_scale_fixture(target_rows=target_rows) + + def clean_once() -> None: + fd.clean(corrupted, return_report=True, **kwargs) + + # Memory is measured on the first (cold) clean: once a clean has run, the + # allocator keeps its pages and later peak-RSS deltas read as ~0. That run + # also serves as the first warmup for the timings. + gc.collect() + with _PeakRss() as rss: + clean_once() + for _ in range(max(0, warmups - 1)): + clean_once() + seconds: list[float] = [] + for _ in range(runs): + gc.collect() + start = time.perf_counter() + clean_once() + seconds.append(time.perf_counter() - start) + return { + "freshdata_path": str(loaded), + "seconds": seconds, + "peak_rss_delta_bytes": rss.delta, + } + + +def _run_worker( + src: str | Path, + *, + target_rows: int, + warmups: int, + runs: int, + timeout: float = 900.0, +) -> dict[str, Any]: + """Run one measurement in a fresh interpreter importing freshdata from *src*.""" + env = dict(os.environ) + env["PYTHONPATH"] = os.pathsep.join([str(Path(src).resolve()), str(REPO_ROOT)]) + cmd = [ + sys.executable, "-m", "benchmarks.cleanbench.ab", "--worker", + "--src", str(src), + "--target-rows", str(target_rows), + "--warmups", str(warmups), + "--runs", str(runs), + ] + proc = subprocess.run( # noqa: S603 - fixed interpreter and arguments + cmd, cwd=REPO_ROOT, env=env, capture_output=True, text=True, + timeout=timeout, check=False, + ) + if proc.returncode != 0: + raise RuntimeError( + f"A/B worker for {src} exited {proc.returncode}:\n{proc.stderr[-2000:]}" + ) + return json.loads(proc.stdout.strip().splitlines()[-1]) + + +# --------------------------------------------------------------------------- # +# orchestration +# --------------------------------------------------------------------------- # + +def measure( + base_src: str | Path, + head_src: str | Path, + *, + pairs: int, + target_rows: int, + warmups: int, + runs: int, + worker: Worker = _run_worker, +) -> dict[str, SideSamples]: + """Interleave base/head workers, alternating which side goes first.""" + sources = {"base": base_src, "head": head_src} + sides = {"base": SideSamples(), "head": SideSamples()} + for pair in range(pairs): + order = ("base", "head") if pair % 2 == 0 else ("head", "base") + for label in order: + payload = worker( + sources[label], target_rows=target_rows, warmups=warmups, runs=runs + ) + side = sides[label] + side.seconds.extend(float(s) for s in payload["seconds"]) + side.rss_deltas.append(int(payload["peak_rss_delta_bytes"])) + side.freshdata_path = str(payload.get("freshdata_path", "")) + return sides + + +def summarize(sides: dict[str, SideSamples]) -> dict[str, Any]: + base, head = sides["base"], sides["head"] + base_fastest, head_fastest = min(base.seconds), min(head.seconds) + base_rss = statistics.median(base.rss_deltas) + head_rss = statistics.median(head.rss_deltas) + memory_overhead = ( + round(head_rss / base_rss - 1.0, 4) if base_rss >= MIN_RSS_DELTA_BYTES else None + ) + return { + "samples_per_side": len(base.seconds), + "base_fastest_seconds": round(base_fastest, 4), + "head_fastest_seconds": round(head_fastest, 4), + "base_median_seconds": round(statistics.median(base.seconds), 4), + "head_median_seconds": round(statistics.median(head.seconds), 4), + "runtime_slowdown": round(head_fastest / base_fastest - 1.0, 4), + "base_median_rss_delta_bytes": int(base_rss), + "head_median_rss_delta_bytes": int(head_rss), + "memory_overhead": memory_overhead, + "base_freshdata_path": base.freshdata_path, + "head_freshdata_path": head.freshdata_path, + } + + +def gate_failures( + summary: dict[str, Any], + *, + runtime_threshold: float = FULL_GATE_RUNTIME_SLOWDOWN, + memory_threshold: float = FULL_GATE_MEMORY_OVERHEAD, +) -> dict[str, str]: + """Breached gates for one measurement, keyed by metric name.""" + failures: dict[str, str] = {} + slowdown = summary["runtime_slowdown"] + if slowdown > runtime_threshold: + failures["runtime"] = ( + f"runtime slowdown {slowdown:+.1%} > {runtime_threshold:.0%} " + f"(fastest head {summary['head_fastest_seconds']}s vs " + f"base {summary['base_fastest_seconds']}s)" + ) + overhead = summary["memory_overhead"] + if overhead is not None and overhead > memory_threshold: + failures["memory"] = ( + f"memory overhead {overhead:+.1%} > {memory_threshold:.0%} " + f"(median peak RSS delta head {summary['head_median_rss_delta_bytes']} " + f"vs base {summary['base_median_rss_delta_bytes']} bytes)" + ) + return failures + + +def run_ab( + base_src: str | Path, + head_src: str | Path, + *, + pairs: int = 5, + target_rows: int = 200_000, + warmups: int = 2, + runs: int = 2, + confirm: bool = True, + worker: Worker = _run_worker, + runtime_threshold: float = FULL_GATE_RUNTIME_SLOWDOWN, + memory_threshold: float = FULL_GATE_MEMORY_OVERHEAD, +) -> dict[str, Any]: + """Measure, gate, and (on a breach) confirm with a full second measurement.""" + options = {"pairs": pairs, "target_rows": target_rows, "warmups": warmups, "runs": runs} + thresholds = {"runtime_threshold": runtime_threshold, "memory_threshold": memory_threshold} + first = summarize(measure(base_src, head_src, worker=worker, **options)) + failures = gate_failures(first, **thresholds) + confirmation = None + if failures and confirm: + confirmation = summarize(measure(base_src, head_src, worker=worker, **options)) + repeated = gate_failures(confirmation, **thresholds) + # A metric fails only when it breached on both measurements. + failures = {metric: repeated[metric] for metric in failures if metric in repeated} + return { + "base_src": str(base_src), + "head_src": str(head_src), + "options": options, + "thresholds": thresholds, + "measurement": first, + "confirmation": confirmation, + "failures": list(failures.values()), + "passed": not failures, + } + + +def render_markdown(result: dict[str, Any]) -> str: + m = result["measurement"] + base_label = result.get("base_label") or result["base_src"] + head_label = result.get("head_label") or result["head_src"] + mib = 1024 * 1024 + memory_change = ( + f"{m['memory_overhead']:+.1%}" if m["memory_overhead"] is not None else "n/a (tiny)" + ) + thresholds = result["thresholds"] + lines = [ + "## Performance A/B gate (CleanBench T5 workload)", + "", + f"Base **{base_label}** vs head **{head_label}**, " + f"{result['options']['target_rows']:,} rows, {result['options']['pairs']} interleaved " + f"worker pairs, {m['samples_per_side']} timed runs per side.", + "", + "| metric | base | head | change | limit |", + "|---|---|---|---|---|", + f"| runtime (fastest run) | {m['base_fastest_seconds']:.3f}s | " + f"{m['head_fastest_seconds']:.3f}s | {m['runtime_slowdown']:+.1%} | " + f"+{thresholds['runtime_threshold']:.0%} |", + f"| runtime (median run) | {m['base_median_seconds']:.3f}s | " + f"{m['head_median_seconds']:.3f}s | | |", + f"| peak RSS delta (median) | {m['base_median_rss_delta_bytes'] / mib:.1f} MiB | " + f"{m['head_median_rss_delta_bytes'] / mib:.1f} MiB | {memory_change} | " + f"+{thresholds['memory_threshold']:.0%} |", + "", + ] + if result["confirmation"] is not None: + c = result["confirmation"] + confirm_memory = ( + f"{c['memory_overhead']:+.1%}" if c["memory_overhead"] is not None else "n/a" + ) + lines.append( + f"A breach triggered a confirmation run: runtime {c['runtime_slowdown']:+.1%}, " + f"memory {confirm_memory}." + ) + lines.append("") + if result["passed"]: + lines.append("**PASS**") + else: + lines.append("**FAIL** (reproduced on the confirmation run):") + lines.extend(f"- {failure}" for failure in result["failures"]) + return "\n".join(lines) + "\n" + + +# --------------------------------------------------------------------------- # +# CLI +# --------------------------------------------------------------------------- # + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(prog="python -m benchmarks.cleanbench.ab") + parser.add_argument("--worker", action="store_true", help=argparse.SUPPRESS) + parser.add_argument("--src", help=argparse.SUPPRESS) + parser.add_argument("--base-src", help="src/ directory of the base checkout") + parser.add_argument("--head-src", default=str(REPO_ROOT / "src"), + help="src/ directory of the head checkout (default: this repo)") + parser.add_argument("--base-label", default=None, help="label for the report") + parser.add_argument("--head-label", default=None, help="label for the report") + parser.add_argument("--pairs", type=int, default=5, help="interleaved worker pairs") + parser.add_argument("--runs", type=int, default=2, help="timed runs per worker") + parser.add_argument("--warmups", type=int, default=2, help="untimed runs per worker") + parser.add_argument("--target-rows", type=int, default=200_000, help="T5 frame size") + parser.add_argument("--no-confirm", action="store_true", + help="fail on the first breach without a confirmation run") + parser.add_argument("--check-gates", action="store_true", + help="exit non-zero when a regression is confirmed") + parser.add_argument("--output", default=str(RESULTS_DIR / "latest.ab.json"), + help="where to write the JSON result") + parser.add_argument("--summary", default=None, + help="append the markdown summary to this file " + "(e.g. $GITHUB_STEP_SUMMARY)") + args = parser.parse_args(argv) + + if args.worker: + payload = _worker( + args.src, target_rows=args.target_rows, warmups=args.warmups, runs=args.runs + ) + print(json.dumps(payload)) + return 0 + if not args.base_src: + parser.error("--base-src is required") + + result = run_ab( + args.base_src, + args.head_src, + pairs=args.pairs, + target_rows=args.target_rows, + warmups=args.warmups, + runs=args.runs, + confirm=not args.no_confirm, + ) + result["base_label"] = args.base_label + result["head_label"] = args.head_label + + output = Path(args.output) + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text(json.dumps(result, indent=2) + "\n", encoding="utf-8") + markdown = render_markdown(result) + print(markdown) + if args.summary: + with open(args.summary, "a", encoding="utf-8") as handle: + handle.write(markdown) + for failure in result["failures"]: + print(f"GATE FAIL: {failure}", file=sys.stderr) + return 1 if args.check_gates and result["failures"] else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_cleanbench_ab.py b/tests/test_cleanbench_ab.py new file mode 100644 index 00000000..e0e8cdff --- /dev/null +++ b/tests/test_cleanbench_ab.py @@ -0,0 +1,133 @@ +"""Same-run A/B performance gate: ordering, statistics, confirmation, CLI.""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import pytest + +BENCH_DIR = Path(__file__).resolve().parents[1] / "benchmarks" +if str(BENCH_DIR) not in sys.path: # pragma: no cover - import plumbing + sys.path.insert(0, str(BENCH_DIR)) + +pytest.importorskip("cleanbench") + +from cleanbench import ab # noqa: E402 + +MIB = 1024 * 1024 + + +def _worker(timings): + """Fake worker: each call pops the next ``(seconds, rss)`` for that side.""" + calls: list[str] = [] + queues = {side: list(values) for side, values in timings.items()} + + def worker(src, *, target_rows, warmups, runs): + calls.append(src) + seconds, rss = queues[src].pop(0) + return {"seconds": seconds, "peak_rss_delta_bytes": rss, "freshdata_path": src} + + return worker, calls + + +def _steady(seconds, rss=64 * MIB, count=20): + return [([seconds, seconds * 1.05], rss)] * count + + +def test_sides_are_interleaved_with_alternating_order(): + worker, calls = _worker({"base": _steady(1.0), "head": _steady(1.0)}) + ab.measure("base", "head", pairs=3, target_rows=10, warmups=0, runs=2, worker=worker) + assert calls == ["base", "head", "head", "base", "base", "head"] + + +def test_summary_uses_fastest_run_and_median_rss(): + sides = { + "base": ab.SideSamples(seconds=[1.2, 1.0, 3.0], rss_deltas=[10 * MIB, 50 * MIB, 12 * MIB]), + "head": ab.SideSamples(seconds=[1.1, 1.3, 9.0], rss_deltas=[11 * MIB, 13 * MIB, 90 * MIB]), + } + summary = ab.summarize(sides) + assert summary["runtime_slowdown"] == pytest.approx(0.10, abs=1e-4) # 1.1 / 1.0 + assert summary["memory_overhead"] == pytest.approx(13 / 12 - 1, abs=1e-4) + + +def test_memory_gate_not_applicable_for_tiny_base_delta(): + sides = { + "base": ab.SideSamples(seconds=[1.0], rss_deltas=[MIB]), + "head": ab.SideSamples(seconds=[1.0], rss_deltas=[10 * MIB]), + } + summary = ab.summarize(sides) + assert summary["memory_overhead"] is None + assert ab.gate_failures(summary) == {} + + +def test_within_threshold_passes_without_confirmation(): + worker, calls = _worker({"base": _steady(1.0), "head": _steady(1.15)}) + result = ab.run_ab("base", "head", pairs=2, worker=worker) + assert result["passed"] + assert result["confirmation"] is None + assert len(calls) == 4 + + +def test_breach_that_does_not_reproduce_passes(): + head = _steady(1.5, count=2) + _steady(1.02, count=2) + worker, calls = _worker({"base": _steady(1.0), "head": head}) + result = ab.run_ab("base", "head", pairs=2, worker=worker) + assert result["passed"], result["failures"] + assert result["measurement"]["runtime_slowdown"] > 0.2 + assert result["confirmation"]["runtime_slowdown"] < 0.2 + assert len(calls) == 8 + + +def test_reproduced_runtime_breach_fails(): + worker, _ = _worker({"base": _steady(1.0), "head": _steady(1.4)}) + result = ab.run_ab("base", "head", pairs=2, worker=worker) + assert not result["passed"] + [failure] = result["failures"] + assert failure.startswith("runtime slowdown +40.0% > 20%") + + +def test_metric_must_breach_on_both_runs_to_fail(): + # First run breaches runtime only; confirmation breaches memory only. + head = _steady(1.5, rss=64 * MIB, count=2) + _steady(1.0, rss=128 * MIB, count=2) + worker, _ = _worker({"base": _steady(1.0, rss=64 * MIB), "head": head}) + result = ab.run_ab("base", "head", pairs=2, worker=worker) + assert result["passed"], result["failures"] + + +def test_reproduced_memory_breach_fails(): + worker, _ = _worker({"base": _steady(1.0, rss=64 * MIB), "head": _steady(1.0, rss=96 * MIB)}) + result = ab.run_ab("base", "head", pairs=2, worker=worker) + assert [f.split(" (")[0] for f in result["failures"]] == ["memory overhead +50.0% > 15%"] + + +@pytest.mark.parametrize(("check_gates", "expected_code"), [(True, 1), (False, 0)]) +def test_cli_writes_results_and_summary(monkeypatch, tmp_path, check_gates, expected_code): + worker, _ = _worker({"base": _steady(1.0), "head": _steady(1.4)}) + real_run_ab = ab.run_ab + monkeypatch.setattr(ab, "run_ab", lambda *a, **k: real_run_ab(*a, **{**k, "worker": worker})) + output = tmp_path / "ab.json" + summary = tmp_path / "summary.md" + argv = ["--base-src", "base", "--head-src", "head", "--pairs", "2", + "--base-label", "main@abc1234", "--output", str(output), "--summary", str(summary)] + if check_gates: + argv.append("--check-gates") + assert ab.main(argv) == expected_code + payload = json.loads(output.read_text()) + assert payload["passed"] is False + text = summary.read_text() + assert "main@abc1234" in text + assert "**FAIL**" in text + + +def test_real_worker_imports_freshdata_from_the_requested_src(): + src = Path(ab.REPO_ROOT) / "src" + payload = ab._run_worker(src, target_rows=300, warmups=0, runs=1, timeout=300) + assert Path(payload["freshdata_path"]).resolve().is_relative_to(src.resolve()) + assert len(payload["seconds"]) == 1 + + +def test_real_worker_refuses_a_src_without_freshdata(tmp_path): + with pytest.raises(RuntimeError, match="expected under"): + ab._run_worker(tmp_path, target_rows=300, warmups=0, runs=1, timeout=300)