Skip to content

Commit 613dc18

Browse files
Merge pull request #126 from FreshCode-Org/chore/repo-wide-ruff-jwd
chore: lint the whole repository — ruff check . in CI (#54)
2 parents 049df38 + 8f4e547 commit 613dc18

21 files changed

Lines changed: 487 additions & 239 deletions

‎.github/workflows/ci.yml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -28,7 +28,7 @@ jobs:
2828
# mypy targets Python 3.10; NumPy 2.5+ stubs require Python 3.12.
2929
pip install -e ".[dev,ml]" "numpy<2.5"
3030
- name: Lint
31-
run: ruff check src tests
31+
run: ruff check .
3232
- name: Typecheck
3333
run: mypy src/freshdata
3434
- name: Test (required fast lane)

‎.github/workflows/release.yml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -233,7 +233,7 @@ jobs:
233233
# mypy targets Python 3.10; NumPy 2.5+ stubs require Python 3.12.
234234
pip install -e ".[dev,ml,docs]" "numpy<2.5"
235235
- name: Lint
236-
run: ruff check src tests
236+
run: ruff check .
237237
- name: Typecheck
238238
run: mypy src/freshdata
239239
- name: Test (release fast lane)

‎CHANGELOG.md‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -61,6 +61,13 @@ adheres to [Semantic Versioning](https://semver.org/).
6161
(analyze → mask → clean under a compiled policy → merge category variants →
6262
re-score trust), and a new docs guide (`docs/ai-copilot.md`).
6363

64+
### Changed
65+
- Lint now covers the whole repository (`ruff check .` in CI, closing #54):
66+
benchmark and notebook lint debt fixed, dead code removed
67+
(`harness_metrics` unused gold-labels block), and the ASV-managed
68+
`freshdata-benchmarks/` sub-project excluded as tool-generated. No
69+
runtime behavior changes.
70+
6471
## [1.1.1] - 2026-07-06
6572

6673
### Fixed

‎benchmarks/baselines/pandas_baseline.py‎

Lines changed: 1 addition & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -15,7 +15,6 @@
1515

1616
import re
1717

18-
import numpy as np
1918
import pandas as pd
2019

2120
from . import count_authored_lines
@@ -50,8 +49,7 @@ def run(df: pd.DataFrame) -> pd.DataFrame:
5049
for col in out.select_dtypes(include="number").columns:
5150
if out[col].isna().any():
5251
out[col] = out[col].fillna(out[col].median())
53-
out = out.drop_duplicates().reset_index(drop=True)
54-
return out
52+
return out.drop_duplicates().reset_index(drop=True)
5553

5654

5755
AUTHORED_LINES: int = count_authored_lines(run)

‎benchmarks/baselines/pyjanitor_baseline.py‎

Lines changed: 1 addition & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -46,8 +46,7 @@ def run(df: pd.DataFrame) -> pd.DataFrame:
4646
out[col] = coerced
4747
for col in out.select_dtypes(include="number").columns:
4848
out[col] = out[col].fillna(out[col].median())
49-
out = out.drop_duplicates().reset_index(drop=True)
50-
return out
49+
return out.drop_duplicates().reset_index(drop=True)
5150

5251

5352
AUTHORED_LINES: int = count_authored_lines(run)

‎benchmarks/bench.py‎

Lines changed: 25 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -31,11 +31,11 @@
3131
if str(HERE) not in sys.path:
3232
sys.path.insert(0, str(HERE))
3333

34-
import freshdata as fd # noqa: E402
35-
3634
import harness_metrics as hm # noqa: E402
35+
from results_schema import SCHEMA_VERSION # noqa: E402
36+
37+
import freshdata as fd # noqa: E402
3738
from fixtures import REGISTRY # noqa: E402
38-
from results_schema import RESULTS_SCHEMA, SCHEMA_VERSION # noqa: E402
3939

4040
RESULTS_DIR = HERE / "results"
4141

@@ -143,13 +143,16 @@ def _write_result(run_id: str, result: dict) -> Path:
143143
def cmd_run(args: argparse.Namespace) -> int:
144144
run_id = _now()
145145
fixtures = args.fixtures or list(DEFAULT_SIZES)
146-
print(f"benchmark run {run_id} freshdata={fd.__version__} mode={'aggressive' if args.aggressive else 'balanced'}")
146+
mode = "aggressive" if args.aggressive else "balanced"
147+
print(f"benchmark run {run_id} freshdata={fd.__version__} mode={mode}")
147148
summary_rows = []
148149
for name in fixtures:
149150
size = args.size or DEFAULT_SIZES[name]
150-
result = run_single(name, size, seed=args.seed, aggressive=args.aggressive, repeat=args.repeat)
151+
result = run_single(
152+
name, size, seed=args.seed, aggressive=args.aggressive, repeat=args.repeat
153+
)
151154
result["run_id"] = run_id
152-
path = _write_result(run_id, result)
155+
_write_result(run_id, result)
153156
m = result["metrics"]
154157
print(f" {name:12s} n={result['n_rows']:>8,} cols={result['n_cols']:>4} "
155158
f"p50={m['wall_clock_p50_sec']:.3f}s peak={m['peak_memory_mb']:.1f}MB "
@@ -182,7 +185,10 @@ def cmd_compare(args: argparse.Namespace) -> int:
182185
config = hm.config_for(name, df)
183186

184187
print(f"compare on {name} n={len(df):,} cols={df.shape[1]}\n")
185-
header = f"{'tool':24s} {'n_rows':>8} {'n_cols':>6} {'p50_sec':>8} {'p95_sec':>8} {'peak_mb':>8} mode"
188+
header = (
189+
f"{'tool':24s} {'n_rows':>8} {'n_cols':>6} "
190+
f"{'p50_sec':>8} {'p95_sec':>8} {'peak_mb':>8} mode"
191+
)
186192
print(header)
187193
print("-" * len(header))
188194

@@ -258,16 +264,20 @@ def _render_markdown(run_id: str, results: list[dict]) -> str:
258264
f"- python: `{results[0]['python_version'] if results else '?'}`",
259265
f"- platform: `{results[0]['platform'] if results else '?'}`",
260266
"",
261-
"| fixture | n_rows | n_cols | p50 s | p95 s | peak MB | repair % | false-repair % | preserve % | trust | monotonic | export % |",
267+
"| fixture | n_rows | n_cols | p50 s | p95 s | peak MB | repair % "
268+
"| false-repair % | preserve % | trust | monotonic | export % |",
262269
"|---|--:|--:|--:|--:|--:|--:|--:|--:|--:|:--:|--:|",
263270
]
264271
for r in sorted(results, key=lambda x: x["fixture"]):
265272
m = r["metrics"]
266273
lines.append(
267274
f"| {r['fixture']} | {r['n_rows']:,} | {r['n_cols']} | "
268-
f"{m['wall_clock_p50_sec']:.3f} | {m['wall_clock_p95_sec']:.3f} | {m['peak_memory_mb']:.1f} | "
269-
f"{m['repair_fidelity_pct']} | {m['false_repair_rate_pct']} | {m['preservation_rate_pct']} | "
270-
f"{m['trust_score']} | {'✅' if m['trust_monotonic_valid'] else '❌'} | {m['export_completeness_pct']} |"
275+
f"{m['wall_clock_p50_sec']:.3f} | {m['wall_clock_p95_sec']:.3f} | "
276+
f"{m['peak_memory_mb']:.1f} | "
277+
f"{m['repair_fidelity_pct']} | {m['false_repair_rate_pct']} | "
278+
f"{m['preservation_rate_pct']} | "
279+
f"{m['trust_score']} | {'✅' if m['trust_monotonic_valid'] else '❌'} | "
280+
f"{m['export_completeness_pct']} |"
271281
)
272282
lines += ["", "## Authored-code reduction (Metric 6)", ""]
273283
if results:
@@ -291,7 +301,10 @@ def cmd_fixtures(args: argparse.Namespace) -> int:
291301
bundle = mod.generate(size, seed=args.seed)
292302
bundle.dirty_df.to_csv(out / "gold_dirty.csv", index=False)
293303
bundle.clean_df.to_csv(out / "gold_clean.csv", index=False)
294-
print(f" gold -> gold_dirty.csv ({bundle.dirty_df.shape}), gold_clean.csv ({bundle.clean_df.shape})")
304+
print(
305+
f" gold -> gold_dirty.csv ({bundle.dirty_df.shape}), "
306+
f"gold_clean.csv ({bundle.clean_df.shape})"
307+
)
295308
else:
296309
df = mod.generate(size, seed=args.seed)
297310
path = out / f"{name}.csv"

‎benchmarks/fixtures/__init__.py‎

Lines changed: 10 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -29,4 +29,13 @@
2929
#: Fixtures whose ``generate`` returns a plain DataFrame (everything but gold).
3030
FRAME_FIXTURES = ("crm", "finance", "event_log", "wide_schema", "provenance")
3131

32-
__all__ = ["REGISTRY", "FRAME_FIXTURES", *REGISTRY.keys()]
32+
__all__ = [
33+
"REGISTRY",
34+
"FRAME_FIXTURES",
35+
"crm",
36+
"finance",
37+
"event_log",
38+
"wide_schema",
39+
"provenance",
40+
"gold",
41+
]

‎benchmarks/fixtures/_common.py‎

Lines changed: 3 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -24,12 +24,11 @@
2424

2525
from __future__ import annotations
2626

27-
from dataclasses import asdict, dataclass, field
27+
from dataclasses import asdict, dataclass
2828
from datetime import date
2929
from typing import Any
3030

3131
import numpy as np
32-
import pandas as pd
3332

3433
# -- roles -----------------------------------------------------------------
3534
# These mirror the roles FreshData's decision engine infers
@@ -134,8 +133,8 @@ def uuid_series(rng: np.random.Generator, n: int, prefix: str = "") -> list[str]
134133
hi = rng.integers(0, 2**32, size=n, dtype=np.uint64)
135134
lo = rng.integers(0, 2**32, size=n, dtype=np.uint64)
136135
out = []
137-
for h, l in zip(hi.tolist(), lo.tolist()):
138-
out.append(f"{prefix}{h:08x}-{l:08x}")
136+
for high, low in zip(hi.tolist(), lo.tolist()):
137+
out.append(f"{prefix}{high:08x}-{low:08x}")
139138
return out
140139

141140

‎benchmarks/fixtures/crm.py‎

Lines changed: 5 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -21,15 +21,13 @@
2121
ACCOUNT_STATUS_REF,
2222
BAD_COUNTRY,
2323
COUNTRY_REF,
24-
Defect,
25-
GoldLabel,
26-
ROLE_BOOL,
2724
ROLE_CATEGORICAL,
2825
ROLE_DATETIME,
2926
ROLE_ID,
3027
ROLE_NUMERIC,
3128
ROLE_TEXT,
32-
SENTINELS,
29+
Defect,
30+
GoldLabel,
3331
defect_mask,
3432
format_iso_date,
3533
gold_to_records,
@@ -69,8 +67,8 @@
6967
def _names(rng: np.random.Generator, n: int) -> np.ndarray:
7068
f = pick(rng, _FIRST, n)
7169
m = pick(rng, _MIDDLE, n)
72-
l = pick(rng, _LAST, n)
73-
return np.array([f"{a} {b} {c}" for a, b, c in zip(f, m, l)], dtype=object)
70+
last = pick(rng, _LAST, n)
71+
return np.array([f"{a} {b} {c}" for a, b, c in zip(f, m, last)], dtype=object)
7472

7573

7674
def generate(n_rows: int, seed: int = 42, defect_rate: float | None = None) -> pd.DataFrame:
@@ -109,8 +107,7 @@ def generate(n_rows: int, seed: int = 42, defect_rate: float | None = None) -> p
109107
data[col] = np.array([str(x) for x in d], dtype=object)
110108

111109
df = pd.DataFrame(data)
112-
df = _inject(df, rng, defect_rate)
113-
return df
110+
return _inject(df, rng, defect_rate)
114111

115112

116113
def _inject(df: pd.DataFrame, rng: np.random.Generator, defect_rate: float | None) -> pd.DataFrame:

‎benchmarks/fixtures/event_log.py‎

Lines changed: 2 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -18,13 +18,12 @@
1818
from ._common import (
1919
BAD_OPERATION,
2020
OPERATION_REF,
21-
Defect,
22-
GoldLabel,
2321
ROLE_CATEGORICAL,
2422
ROLE_DATETIME,
2523
ROLE_ID,
2624
ROLE_NUMERIC,
27-
ROLE_TEXT,
25+
Defect,
26+
GoldLabel,
2827
defect_mask,
2928
gold_to_records,
3029
manifest_to_records,

0 commit comments

Comments
 (0)