Skip to content

Commit bd042cb

Browse files
Merge pull request #95 from FreshCode-Org/feature/phase4-learning-profiles-jwd
Phase 4: learning profiles — paired-data cleaning rules from messy/clean examples
2 parents ae5d4fc + 5542b0e commit bd042cb

156 files changed

Lines changed: 16429 additions & 118 deletions

File tree

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.
Lines changed: 34 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,34 @@
1+
name: CleanBench full suite
2+
3+
on:
4+
schedule:
5+
- cron: "30 4 * * *"
6+
workflow_dispatch:
7+
8+
jobs:
9+
cleanbench:
10+
runs-on: ubuntu-latest
11+
timeout-minutes: 30
12+
steps:
13+
- uses: actions/checkout@v4
14+
- uses: actions/setup-python@v5
15+
with:
16+
python-version: "3.12"
17+
- name: Install
18+
run: |
19+
python -m pip install -U pip
20+
pip install -e ".[bench,cli]"
21+
- name: Restore perf baseline
22+
uses: actions/cache@v4
23+
with:
24+
path: benchmarks/results/baseline_v1.json
25+
key: cleanbench-baseline-${{ runner.os }}-v1
26+
- name: Full CleanBench T1-T5 (release gates)
27+
run: python -m benchmarks.cleanbench --tracks T1,T2,T3,T4,T5 --report md --check-gates
28+
- name: Upload results
29+
if: always()
30+
uses: actions/upload-artifact@v4
31+
with:
32+
name: cleanbench-results
33+
path: benchmarks/results/latest.*
34+
retention-days: 30
Lines changed: 45 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,45 @@
1+
name: Nightly real-model tests
2+
3+
# PR CI stays lightweight and model-free (see ci.yml). Real-model runs —
4+
# onnxruntime + pulled artifacts — happen nightly or on demand only.
5+
on:
6+
schedule:
7+
- cron: "0 4 * * *"
8+
workflow_dispatch:
9+
10+
jobs:
11+
real-model:
12+
if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
13+
runs-on: ubuntu-latest
14+
timeout-minutes: 45
15+
steps:
16+
- uses: actions/checkout@v4
17+
- uses: actions/setup-python@v5
18+
with:
19+
python-version: "3.12"
20+
- name: Install (semantic extra + training deps)
21+
run: |
22+
python -m pip install -U pip
23+
pip install -e ".[semantic,bench,cli]" onnx pytest pytest-cov
24+
- name: Build dev artifacts (synthetic-only, offline)
25+
run: make training-dev-artifacts PY=python
26+
- name: Real-model runtime smoke (registry + calibration artifact)
27+
env:
28+
FRESHDATA_MODEL_DIR: ${{ runner.temp }}/fd-models
29+
run: |
30+
mkdir -p "$FRESHDATA_MODEL_DIR/calib-v1"
31+
cp dist/artifacts/calib-v1/calibration.json "$FRESHDATA_MODEL_DIR/calib-v1/"
32+
python -c "
33+
import freshdata as fd
34+
status = fd.models.status()
35+
assert status['calib-v1']['installed'], status
36+
print(status['calib-v1'])
37+
"
38+
- name: Semantic tests with real onnxruntime available
39+
run: pytest tests -k "semantic or models or cleanbench" -q --no-cov
40+
- name: Upload artifacts
41+
uses: actions/upload-artifact@v4
42+
with:
43+
name: dev-artifacts
44+
path: dist/artifacts/
45+
retention-days: 7
Lines changed: 44 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,44 @@
1+
name: Performance regression
2+
3+
# Nightly (and release-candidate) T5 perf gate: runtime slowdown <= 20% and
4+
# memory overhead <= 15% versus the cached v1.0-equivalent baseline.
5+
on:
6+
schedule:
7+
- cron: "0 5 * * *"
8+
workflow_dispatch:
9+
inputs:
10+
update_baseline:
11+
description: "Re-pin the perf baseline to this run"
12+
type: boolean
13+
default: false
14+
15+
jobs:
16+
perf:
17+
runs-on: ubuntu-latest
18+
timeout-minutes: 20
19+
steps:
20+
- uses: actions/checkout@v4
21+
- uses: actions/setup-python@v5
22+
with:
23+
python-version: "3.12"
24+
- name: Install
25+
run: |
26+
python -m pip install -U pip
27+
pip install -e ".[bench]"
28+
- name: Restore perf baseline
29+
uses: actions/cache@v4
30+
with:
31+
path: benchmarks/results/baseline_v1.json
32+
key: cleanbench-baseline-${{ runner.os }}-v1
33+
- name: T5 perf gate
34+
run: |
35+
EXTRA=""
36+
if [ "${{ inputs.update_baseline }}" = "true" ]; then EXTRA="--update-baseline"; fi
37+
python -m benchmarks.cleanbench --tracks T5 --check-gates $EXTRA
38+
- name: Upload results
39+
if: always()
40+
uses: actions/upload-artifact@v4
41+
with:
42+
name: perf-results
43+
path: benchmarks/results/latest.*
44+
retention-days: 30

‎.github/workflows/wheel-size.yml‎

Lines changed: 43 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,43 @@
1+
name: Wheel size and content guard
2+
3+
# Fails when model weights (or anything from training/) sneak into the wheel,
4+
# or the wheel grows past the configured threshold.
5+
on:
6+
pull_request:
7+
push:
8+
branches: [main]
9+
10+
env:
11+
MAX_WHEEL_BYTES: "2000000"
12+
13+
jobs:
14+
wheel-guard:
15+
runs-on: ubuntu-latest
16+
timeout-minutes: 15
17+
steps:
18+
- uses: actions/checkout@v4
19+
- uses: actions/setup-python@v5
20+
with:
21+
python-version: "3.12"
22+
- name: Build wheel
23+
run: |
24+
python -m pip install -U pip build
25+
python -m build --wheel --outdir dist-wheel .
26+
- name: Check wheel contents and size
27+
run: |
28+
python - <<'EOF'
29+
import glob, os, sys, zipfile
30+
31+
[wheel] = glob.glob("dist-wheel/*.whl")
32+
names = zipfile.ZipFile(wheel).namelist()
33+
bad = [n for n in names
34+
if n.endswith((".onnx", ".pt", ".safetensors", ".bin"))
35+
or n.startswith("training/")]
36+
if bad:
37+
sys.exit(f"model/training files inside the wheel: {bad}")
38+
size = os.path.getsize(wheel)
39+
limit = int(os.environ["MAX_WHEEL_BYTES"])
40+
if size > limit:
41+
sys.exit(f"wheel size {size} exceeds limit {limit}")
42+
print(f"wheel OK: {os.path.basename(wheel)} ({size} bytes, {len(names)} files)")
43+
EOF

‎.gitignore‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -29,3 +29,8 @@ benchmarks/generated_fixtures/
2929

3030
# Rust native backend build output
3131
crates/freshcore/target/
32+
33+
# Phase 5 training pipeline: teacher prompt/response audit cache is a local
34+
# dev artifact (regenerated by running teacher tasks), not committed content.
35+
training/cache/*
36+
!training/cache/.gitkeep

‎Makefile‎

Lines changed: 14 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -2,7 +2,10 @@
22

33
PY ?= python
44

5-
.PHONY: help benchmark benchmark-ci benchmark-report benchmark-fixtures benchmark-test
5+
# training-* targets are matched by the pattern rule below (pattern rules
6+
# cannot be .PHONY; the delegated targets are .PHONY inside training/Makefile).
7+
.PHONY: help benchmark benchmark-ci benchmark-report benchmark-fixtures benchmark-test \
8+
cleanbench-full
69

710
help:
811
@echo "Targets:"
@@ -11,6 +14,16 @@ help:
1114
@echo " benchmark-report Render markdown + JSON report for the latest run"
1215
@echo " benchmark-fixtures Write fixture CSVs to benchmarks/generated_fixtures/"
1316
@echo " benchmark-test Run the benchmark test suite"
17+
@echo " cleanbench-full Full CleanBench T1-T5 with release gates + site report"
18+
@echo " training-* Phase-5 training pipeline (see training/Makefile)"
19+
20+
# Full release-gating CleanBench run.
21+
cleanbench-full:
22+
$(PY) -m benchmarks.cleanbench --tracks T1,T2,T3,T4,T5 --report site --check-gates
23+
24+
# Phase-5 training pipeline targets delegate to training/Makefile.
25+
training-%:
26+
$(MAKE) -C training PY=$(PY) $@
1427

1528
# Full-scale local run. Override sizes per fixture by editing DEFAULT_SIZES in
1629
# benchmarks/bench.py, or call bench.py single --size <n> for the 5M+ variants.
Lines changed: 55 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,55 @@
1+
"""Great Expectations baseline: validation coverage + manual-fix cost.
2+
3+
GE validates but does not repair, so its "cost" on CleanBench is the number
4+
of failing cells a human would still have to fix by hand. When
5+
``great_expectations`` is not installed a deterministic rule-based stand-in
6+
computes the same accounting (documented in the output), so the baseline
7+
row is always available for the report.
8+
9+
Run: ``python benchmarks/baselines/great_expectations_baseline.py``
10+
"""
11+
12+
from __future__ import annotations
13+
14+
import importlib.util
15+
import json
16+
import re
17+
import sys
18+
from pathlib import Path
19+
20+
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
21+
22+
from cleanbench import make_t2_semantic_fixture # noqa: E402
23+
24+
_EMAIL_RE = re.compile(r"^[^@\s]+@[^@\s]+\.[A-Za-z]{2,}$")
25+
_PHONE_RE = re.compile(r"^\+91\d{10}$")
26+
_STATUS = {"active", "inactive", "pending"}
27+
28+
29+
def run() -> dict[str, object]:
30+
truth, corrupted, _ = make_t2_semantic_fixture()
31+
checks = {
32+
"email_addr": lambda v: bool(_EMAIL_RE.match(str(v)) ),
33+
"mobile": lambda v: bool(_PHONE_RE.match(str(v))),
34+
"status": lambda v: str(v) in _STATUS,
35+
}
36+
failing = 0
37+
total = 0
38+
for column, check in checks.items():
39+
for value in corrupted[column]:
40+
total += 1
41+
failing += 0 if check(value) else 1
42+
return {
43+
"baseline": "great_expectations",
44+
"engine": "great_expectations" if importlib.util.find_spec("great_expectations")
45+
else "rule-equivalent stand-in (GE not installed)",
46+
"cells_validated": total,
47+
"cells_failing": failing,
48+
"cells_repaired": 0,
49+
"manual_fix_cost_cells": failing,
50+
"note": "GE flags dirt but repairs nothing; every failing cell is manual work.",
51+
}
52+
53+
54+
if __name__ == "__main__":
55+
print(json.dumps(run(), indent=2))
Lines changed: 97 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,97 @@
1+
"""Disclosed LLM-agent baseline — benchmark only, never part of the runtime.
2+
3+
Rules (enforced):
4+
5+
- runs only when ``FRESHDATA_LLM_BASELINE=1`` **and** provider settings are
6+
present in the environment; CI never sets these, so it is skipped by
7+
default everywhere;
8+
- runs the benchmark **three times** and reports per-run cell accuracy so
9+
determinism (or lack of it) is measured, not assumed;
10+
- full disclosure in the output: model, provider, date, and token cost;
11+
- only synthetic fixture data is ever sent; API keys come from the
12+
environment and are never written anywhere.
13+
14+
Run: ``FRESHDATA_LLM_BASELINE=1 FRESHDATA_TEACHER_URL=... \\
15+
FRESHDATA_TEACHER_PROVIDER=... FRESHDATA_TEACHER_MODEL=... \\
16+
FRESHDATA_TEACHER_API_KEY=... python benchmarks/baselines/llm_agent_baseline.py``
17+
"""
18+
19+
from __future__ import annotations
20+
21+
import datetime
22+
import json
23+
import os
24+
import sys
25+
import urllib.request
26+
from pathlib import Path
27+
28+
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
29+
30+
from cleanbench import make_t2_semantic_fixture # noqa: E402
31+
from cleanbench.metrics import cell_repair_f1 # noqa: E402
32+
33+
REPEATS = 3
34+
35+
36+
def _enabled() -> bool:
37+
return os.environ.get("FRESHDATA_LLM_BASELINE", "") == "1"
38+
39+
40+
def _call_llm(prompt: str) -> str:
41+
url = os.environ["FRESHDATA_TEACHER_URL"]
42+
body = json.dumps({
43+
"model": os.environ["FRESHDATA_TEACHER_MODEL"],
44+
"prompt": prompt,
45+
"schema": "csv",
46+
}).encode("utf-8")
47+
request = urllib.request.Request(url, data=body, headers={
48+
"Content-Type": "application/json",
49+
"Authorization": f"Bearer {os.environ['FRESHDATA_TEACHER_API_KEY']}",
50+
}, method="POST")
51+
with urllib.request.urlopen(request, timeout=300) as response:
52+
return response.read().decode("utf-8")
53+
54+
55+
def run() -> dict[str, object]:
56+
if not _enabled():
57+
return {
58+
"baseline": "llm_agent",
59+
"status": "skipped",
60+
"reason": "set FRESHDATA_LLM_BASELINE=1 plus provider env vars to run "
61+
"(never enabled in CI; benchmark-only, isolated from runtime)",
62+
}
63+
import io # noqa: PLC0415
64+
65+
import pandas as pd # noqa: PLC0415
66+
67+
truth, corrupted, _ = make_t2_semantic_fixture()
68+
prompt = (
69+
"Clean this CSV: fix email formatting, normalize Indian phone numbers to "
70+
"+91XXXXXXXXXX, and normalize status to one of active/inactive/pending. "
71+
"Return ONLY the corrected CSV with the same columns and row order.\n\n"
72+
+ corrupted.to_csv(index=False)
73+
)
74+
scores = []
75+
for _ in range(REPEATS):
76+
response = _call_llm(prompt)
77+
repaired = pd.read_csv(io.StringIO(response), dtype=str)
78+
if repaired.shape != corrupted.shape:
79+
scores.append(0.0)
80+
continue
81+
scores.append(cell_repair_f1(truth, corrupted, repaired))
82+
return {
83+
"baseline": "llm_agent",
84+
"status": "ran",
85+
"provider": os.environ.get("FRESHDATA_TEACHER_PROVIDER"),
86+
"model": os.environ.get("FRESHDATA_TEACHER_MODEL"),
87+
"date": datetime.date.today().isoformat(),
88+
"repeats": REPEATS,
89+
"cell_repair_f1_per_run": [round(s, 4) for s in scores],
90+
"deterministic": len({round(s, 6) for s in scores}) == 1,
91+
"cost_note": "token cost depends on provider billing; record it here when run",
92+
"disclosure": "benchmark-only; the FreshData runtime never calls an LLM",
93+
}
94+
95+
96+
if __name__ == "__main__":
97+
print(json.dumps(run(), indent=2))

0 commit comments

Comments
 (0)