Skip to content

Commit d92bc4f

Browse files
feat(benchmarks): Validation Gauntlet — gold-labelled disposition benchmark
Five deterministic fixtures (finance, healthcare, crm, ecommerce, adversarial text) label every injected defect with the disposition FreshData should choose — preserve / repair / flag / review — including the flagship 'apple' case: quarantined in a price column, preserved as a company name and ticker, routed to review as a lowercase ticker. The runner drives public surfaces only (fd.clean defaults, validate_fields, clean_text safe + explicit opt-in config, semantic auto mode, lint_text_encoding, detect_pii, domain packs) and the metrics score detection P/R/F1, repair accuracy per source, review routing, preservation, corruption, escapes, FPR, audit completeness, determinism, trust monotonicity, runtime and peak memory. python -m benchmarks.gauntlet run writes JSON + Markdown with a failure catalogue; --check gates against absolute thresholds and the stored baseline.json. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
1 parent 98d2eb4 commit d92bc4f

9 files changed

Lines changed: 1500 additions & 0 deletions

File tree

‎.gitignore‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -34,3 +34,4 @@ crates/freshcore/target/
3434
# dev artifact (regenerated by running teacher tasks), not committed content.
3535
training/cache/*
3636
!training/cache/.gitkeep
37+
benchmarks/gauntlet/results/

‎benchmarks/gauntlet/__init__.py‎

Lines changed: 28 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,28 @@
1+
"""FreshData Validation Gauntlet.
2+
3+
Gold-labelled adversarial fixtures plus a harness that measures how
4+
FreshData's validation surfaces (``fd.clean``, ``fd.validate_fields``,
5+
``fd.clean_text``, domain packs, the semantic layer, PII detection) treat
6+
each labelled cell: preserve, repair, flag, or route to review.
7+
8+
Unlike CleanBench (which scores whole-frame repair fidelity against a clean
9+
oracle), the gauntlet scores *dispositions*: every injected defect carries the
10+
disposition FreshData should choose, and every adversarial trap is a valid
11+
value that must survive cleaning untouched.
12+
13+
Run ``python -m benchmarks.gauntlet run`` from the repo root.
14+
"""
15+
16+
from .fixtures import FIXTURES, GauntletFixture, GoldCell, build_fixture
17+
from .metrics import compute_metrics
18+
from .runner import run_fixture, run_gauntlet
19+
20+
__all__ = [
21+
"FIXTURES",
22+
"GauntletFixture",
23+
"GoldCell",
24+
"build_fixture",
25+
"compute_metrics",
26+
"run_fixture",
27+
"run_gauntlet",
28+
]

‎benchmarks/gauntlet/__main__.py‎

Lines changed: 67 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,67 @@
1+
"""Validation Gauntlet CLI.
2+
3+
Run from the repo root::
4+
5+
python -m benchmarks.gauntlet run # run + write results/
6+
python -m benchmarks.gauntlet run --check # also gate (CI mode)
7+
python -m benchmarks.gauntlet run --update-baseline
8+
"""
9+
10+
from __future__ import annotations
11+
12+
import argparse
13+
import json
14+
import sys
15+
from pathlib import Path
16+
17+
from .fixtures import DEFAULT_ROWS, DEFAULT_SEED, FIXTURES
18+
from .metrics import compute_metrics
19+
from .report import check_gates, render_markdown, results_payload, write_json
20+
from .runner import run_gauntlet
21+
22+
RESULTS_DIR = Path(__file__).parent / "results"
23+
BASELINE_PATH = Path(__file__).parent / "baseline.json"
24+
25+
26+
def main(argv: list[str] | None = None) -> int:
27+
parser = argparse.ArgumentParser(prog="python -m benchmarks.gauntlet")
28+
sub = parser.add_subparsers(dest="command", required=True)
29+
run_p = sub.add_parser("run", help="run the gauntlet and write JSON + Markdown")
30+
run_p.add_argument("--rows", type=int, default=DEFAULT_ROWS)
31+
run_p.add_argument("--seed", type=int, default=DEFAULT_SEED)
32+
run_p.add_argument("--fixtures", nargs="*", choices=sorted(FIXTURES))
33+
run_p.add_argument("--check", action="store_true",
34+
help="exit 1 when a gate fails or the baseline regresses")
35+
run_p.add_argument("--update-baseline", action="store_true",
36+
help="write this run as the stored baseline")
37+
args = parser.parse_args(argv)
38+
39+
runs = run_gauntlet(n_rows=args.rows, seed=args.seed, fixtures=args.fixtures)
40+
metrics = {name: compute_metrics(r) for name, r in runs.items()}
41+
payload = results_payload(metrics, n_rows=args.rows, seed=args.seed)
42+
43+
write_json(payload, RESULTS_DIR / "gauntlet.json")
44+
markdown = render_markdown(payload)
45+
(RESULTS_DIR / "gauntlet.md").write_text(markdown)
46+
print(markdown)
47+
print(f"results: {RESULTS_DIR / 'gauntlet.json'}")
48+
49+
if args.update_baseline:
50+
write_json(payload, BASELINE_PATH)
51+
print(f"baseline updated: {BASELINE_PATH}")
52+
53+
if args.check:
54+
baseline = (json.loads(BASELINE_PATH.read_text())
55+
if BASELINE_PATH.exists() else None)
56+
problems = check_gates(payload, baseline)
57+
if problems:
58+
print("\nGATE FAILURES:", file=sys.stderr)
59+
for p in problems:
60+
print(f" - {p}", file=sys.stderr)
61+
return 1
62+
print("all gates passed")
63+
return 0
64+
65+
66+
if __name__ == "__main__":
67+
raise SystemExit(main())

‎benchmarks/gauntlet/baseline.json‎

Lines changed: 228 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,228 @@
1+
{
2+
"schema_version": 1,
3+
"generated_at": "2026-07-13T04:14:15+00:00",
4+
"n_rows": 300,
5+
"seed": 42,
6+
"fixtures": {
7+
"crm": {
8+
"fixture": "crm",
9+
"n_rows": 304,
10+
"labelled_cells": 19,
11+
"detection": {
12+
"precision": 1.0,
13+
"recall": 0.9167,
14+
"f1": 0.9565,
15+
"true_positives": 11,
16+
"false_negatives": 1,
17+
"false_positives": 0
18+
},
19+
"repair_accuracy": 1.0,
20+
"repair_sources": {
21+
"clean": 3,
22+
"clean_text_opt_in": 1
23+
},
24+
"review_routing": 1.0,
25+
"preservation_rate": 1.0,
26+
"corruption_count": 0,
27+
"escape_rate": 0.0833,
28+
"false_positive_rate": 0.0,
29+
"duplicates": {
30+
"expected": 4,
31+
"removed": 4,
32+
"ok": true
33+
},
34+
"audit_completeness": 1.0,
35+
"deterministic": true,
36+
"trust": {
37+
"pristine_frame": 100.0,
38+
"dirty_frame": 99.688,
39+
"monotonic": true
40+
},
41+
"performance": {
42+
"clean_seconds": 0.2098,
43+
"validate_seconds": 0.0133,
44+
"peak_memory_mb": 0.269
45+
},
46+
"failures": [
47+
{
48+
"row": 24,
49+
"column": "full_name",
50+
"kind": "injection_text",
51+
"expect": "flag",
52+
"detected": false,
53+
"dirty": "\"ROBERT'); DROP TABLE users;--\"",
54+
"verdict": "escaped"
55+
}
56+
],
57+
"detected_only": []
58+
},
59+
"ecommerce": {
60+
"fixture": "ecommerce",
61+
"n_rows": 303,
62+
"labelled_cells": 15,
63+
"detection": {
64+
"precision": 1.0,
65+
"recall": 1.0,
66+
"f1": 1.0,
67+
"true_positives": 11,
68+
"false_negatives": 0,
69+
"false_positives": 0
70+
},
71+
"repair_accuracy": 1.0,
72+
"repair_sources": {
73+
"clean": 2,
74+
"clean_text_opt_in": 1
75+
},
76+
"review_routing": 1.0,
77+
"preservation_rate": 1.0,
78+
"corruption_count": 0,
79+
"escape_rate": 0.0,
80+
"false_positive_rate": 0.0,
81+
"duplicates": {
82+
"expected": 3,
83+
"removed": 3,
84+
"ok": true
85+
},
86+
"audit_completeness": 1.0,
87+
"deterministic": true,
88+
"trust": {
89+
"pristine_frame": 100.0,
90+
"dirty_frame": 94.74,
91+
"monotonic": true
92+
},
93+
"performance": {
94+
"clean_seconds": 0.1883,
95+
"validate_seconds": 0.0189,
96+
"peak_memory_mb": 0.157
97+
},
98+
"failures": [],
99+
"detected_only": []
100+
},
101+
"finance": {
102+
"fixture": "finance",
103+
"n_rows": 303,
104+
"labelled_cells": 25,
105+
"detection": {
106+
"precision": 1.0,
107+
"recall": 1.0,
108+
"f1": 1.0,
109+
"true_positives": 21,
110+
"false_negatives": 0,
111+
"false_positives": 0
112+
},
113+
"repair_accuracy": 1.0,
114+
"repair_sources": {
115+
"clean": 10
116+
},
117+
"review_routing": 1.0,
118+
"preservation_rate": 1.0,
119+
"corruption_count": 0,
120+
"escape_rate": 0.0,
121+
"false_positive_rate": 0.0,
122+
"duplicates": {
123+
"expected": 3,
124+
"removed": 3,
125+
"ok": true
126+
},
127+
"audit_completeness": 1.0,
128+
"deterministic": true,
129+
"trust": {
130+
"pristine_frame": 100.0,
131+
"dirty_frame": 92.178,
132+
"monotonic": true
133+
},
134+
"performance": {
135+
"clean_seconds": 0.223,
136+
"validate_seconds": 0.021,
137+
"peak_memory_mb": 0.181
138+
},
139+
"failures": [],
140+
"detected_only": []
141+
},
142+
"healthcare": {
143+
"fixture": "healthcare",
144+
"n_rows": 303,
145+
"labelled_cells": 16,
146+
"detection": {
147+
"precision": 1.0,
148+
"recall": 1.0,
149+
"f1": 1.0,
150+
"true_positives": 15,
151+
"false_negatives": 0,
152+
"false_positives": 0
153+
},
154+
"repair_accuracy": 1.0,
155+
"repair_sources": {
156+
"clean": 1,
157+
"validate_fields": 1
158+
},
159+
"review_routing": 1.0,
160+
"preservation_rate": 1.0,
161+
"corruption_count": 0,
162+
"escape_rate": 0.0,
163+
"false_positive_rate": 0.0,
164+
"duplicates": {
165+
"expected": 3,
166+
"removed": 3,
167+
"ok": true
168+
},
169+
"audit_completeness": 1.0,
170+
"deterministic": true,
171+
"trust": {
172+
"pristine_frame": 100.0,
173+
"dirty_frame": 94.74,
174+
"monotonic": true
175+
},
176+
"performance": {
177+
"clean_seconds": 0.2001,
178+
"validate_seconds": 0.0217,
179+
"peak_memory_mb": 0.16
180+
},
181+
"failures": [],
182+
"detected_only": []
183+
},
184+
"text": {
185+
"fixture": "text",
186+
"n_rows": 302,
187+
"labelled_cells": 16,
188+
"detection": {
189+
"precision": 1.0,
190+
"recall": 1.0,
191+
"f1": 1.0,
192+
"true_positives": 10,
193+
"false_negatives": 0,
194+
"false_positives": 0
195+
},
196+
"repair_accuracy": 1.0,
197+
"repair_sources": {
198+
"clean": 1,
199+
"clean_text_opt_in": 2,
200+
"validate_fields": 4
201+
},
202+
"review_routing": null,
203+
"preservation_rate": 1.0,
204+
"corruption_count": 0,
205+
"escape_rate": 0.0,
206+
"false_positive_rate": 0.0,
207+
"duplicates": {
208+
"expected": 2,
209+
"removed": 2,
210+
"ok": true
211+
},
212+
"audit_completeness": 1.0,
213+
"deterministic": true,
214+
"trust": {
215+
"pristine_frame": 100.0,
216+
"dirty_frame": 99.801,
217+
"monotonic": true
218+
},
219+
"performance": {
220+
"clean_seconds": 0.0689,
221+
"validate_seconds": 0.0054,
222+
"peak_memory_mb": 0.103
223+
},
224+
"failures": [],
225+
"detected_only": []
226+
}
227+
}
228+
}

0 commit comments

Comments
 (0)