Skip to content

Commit 598500a

Browse files
test(validation): one bad cell must trip each rule, and nothing may mutate (#476)
Three properties the suite did not assert directly. One-value violations. A rule tested only against a wholly-broken frame can pass while detecting something else, and a rule never tested against a clean frame can pass by always failing. Each of unique, allowed_values, range, nullable, max_missing_ratio, regex, compound_unique and min_rows is now asserted both ways: exactly one bad cell is detected, and the same rule stays silent on clean data. max_rows, a missing required column and strict_columns extras are covered too. All 16 already behave correctly. Non-mutation by digest. The existing checks use assert_frame_equal, which compares values and dtypes but ignores .attrs and the index name, so an in-place change to either would go unnoticed. These hash content, labels, dtypes, attrs and index name, and all seven of run_suite, validate_fields, profile, suggest_plan, infer_roles, explain_clean and clean leave the input byte-identical. Cross-field. The inventory recorded "no explicit min <= max test"; the capability exists through caller-supplied cross_rules callables, so this was a coverage gap rather than a missing feature. On a frame carrying one date inversion and one price inversion on the same row, both rules fire, only that row is reported, the action is manual_review rather than an automatic repair, and a row with a missing half of either pair is correctly not reported -- absence is not inversion. Remediation integrity. accepted + quarantined + rejected + needs_review equals the input row count, so no row is silently dropped; the original 'apple' stays recoverable from the quarantine frame; and the audit entry carries original, action, applied, classification and the reason "expected numeric in 'amount' but got text value 'apple'; not silently converted". The report's claim and the frames agree, which is the property worth guarding: row 1 is absent from accepted and present in quarantined. Every property already holds, so these are regression guards rather than bug reports. No library code changed. Full suite 6674 passed / 0 failed, coverage 93.91%; ruff clean repo-wide.
1 parent 815a9ec commit 598500a

1 file changed

Lines changed: 226 additions & 0 deletions

File tree

‎tests/test_validation_integrity.py‎

Lines changed: 226 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,226 @@
1+
"""Validation must detect one bad cell, never mutate, and never lose a row.
2+
3+
Three properties the suite did not assert directly:
4+
5+
**One-value violations.** A dataset containing exactly one bad cell must be
6+
enough to prove a rule fires. A rule tested only against a wholly-broken frame
7+
can pass while detecting something else entirely, and a rule never tested
8+
against a *clean* frame can pass by always failing. Each rule here is asserted
9+
both ways.
10+
11+
**Non-mutation, by digest.** The existing checks use
12+
``pd.testing.assert_frame_equal``, which compares values and dtypes but not
13+
``.attrs`` or the index name -- so an in-place change to either would go
14+
unnoticed. These hash content, labels, dtypes, ``attrs`` and the index name.
15+
16+
**Remediation integrity.** Every input row must end up in exactly one of
17+
accepted / quarantined / rejected / needs_review, the original value must stay
18+
recoverable, and the audit must explain the decision.
19+
"""
20+
21+
from __future__ import annotations
22+
23+
import hashlib
24+
25+
import pandas as pd
26+
import pytest
27+
28+
import freshdata as fd
29+
from freshdata import ColumnRule, ValidationSuite
30+
31+
32+
def _frame() -> pd.DataFrame:
33+
return pd.DataFrame(
34+
{"id": [1, 2, 3, 4], "cat": ["a", "b", "a", "b"], "n": [1.0, 2.0, 3.0, 4.0]}
35+
)
36+
37+
38+
def _suite(**kw) -> ValidationSuite:
39+
return ValidationSuite(name="t", **kw)
40+
41+
42+
def _run(df: pd.DataFrame, suite: ValidationSuite) -> bool:
43+
"""True when the suite reports a violation."""
44+
return not fd.run_suite(df, suite).passed
45+
46+
47+
# -- one bad cell must be enough --------------------------------------------
48+
49+
ONE_VALUE_CASES = [
50+
("unique", _suite(rules=(ColumnRule(name="id", unique=True),)),
51+
lambda d: d.__setitem__("id", [1, 2, 3, 3])),
52+
("allowed_values", _suite(rules=(ColumnRule(name="cat", allowed_values=("a", "b")),)),
53+
lambda d: d.__setitem__("cat", ["a", "b", "a", "z"])),
54+
("range", _suite(rules=(ColumnRule(name="n", min_value=0, max_value=10),)),
55+
lambda d: d.__setitem__("n", [1.0, 2.0, 3.0, 99.0])),
56+
("nullable", _suite(rules=(ColumnRule(name="n", nullable=False),)),
57+
lambda d: d.__setitem__("n", [1.0, 2.0, 3.0, None])),
58+
("max_missing_ratio", _suite(rules=(ColumnRule(name="n", max_missing_ratio=0.0),)),
59+
lambda d: d.__setitem__("n", [1.0, 2.0, 3.0, None])),
60+
("regex", _suite(rules=(ColumnRule(name="cat", regex=r"^[ab]$"),)),
61+
lambda d: d.__setitem__("cat", ["a", "b", "a", "zz"])),
62+
("compound_unique", _suite(compound_unique=(("id", "cat"),)),
63+
lambda d: (
64+
d.__setitem__("id", [1, 2, 3, 3]),
65+
d.__setitem__("cat", ["a", "b", "a", "a"]),
66+
)),
67+
("min_rows", _suite(min_rows=4), lambda d: d.drop(index=3, inplace=True)),
68+
]
69+
70+
71+
_IDS = [c[0] for c in ONE_VALUE_CASES]
72+
73+
74+
@pytest.mark.parametrize(("name", "suite", "break_it"), ONE_VALUE_CASES, ids=_IDS)
75+
def test_one_violation_is_detected(name, suite, break_it):
76+
df = _frame()
77+
break_it(df)
78+
assert _run(df, suite), f"{name} did not detect its single violation"
79+
80+
81+
@pytest.mark.parametrize(("name", "suite", "break_it"), ONE_VALUE_CASES, ids=_IDS)
82+
def test_the_same_rule_passes_a_clean_frame(name, suite, break_it):
83+
"""Guards the other direction: a rule that always fails detects nothing."""
84+
assert not _run(_frame(), suite), f"{name} reported a violation on clean data"
85+
86+
87+
def test_a_missing_required_column_is_detected():
88+
assert _run(_frame(), _suite(rules=(ColumnRule(name="absent", required=True),)))
89+
90+
91+
def test_extra_columns_are_detected_under_strict_columns():
92+
assert _run(_frame(), _suite(rules=(ColumnRule(name="id"),), strict_columns=True))
93+
94+
95+
def test_max_rows_is_detected():
96+
df = _frame()
97+
df.loc[4] = [5, "a", 5.0]
98+
assert _run(df, _suite(max_rows=4))
99+
100+
101+
# -- non-mutation, by digest ------------------------------------------------
102+
103+
104+
def _digest(df: pd.DataFrame) -> str:
105+
"""Hash content, labels, dtypes, attrs and index name.
106+
107+
Stricter than ``assert_frame_equal``, which ignores ``attrs`` and the index
108+
name -- an in-place change to either would otherwise pass unnoticed.
109+
"""
110+
h = hashlib.sha256()
111+
h.update(pd.util.hash_pandas_object(df, index=True).values.tobytes())
112+
h.update(repr([str(c) for c in df.columns]).encode())
113+
h.update(repr([str(t) for t in df.dtypes]).encode())
114+
h.update(repr(sorted(df.attrs.items())).encode())
115+
h.update(repr(df.index.name).encode())
116+
return h.hexdigest()
117+
118+
119+
def _annotated() -> pd.DataFrame:
120+
df = pd.DataFrame(
121+
{"id": [1, 2, 3, 4], "cat": ["a", "b", "a", "z"], "n": [1.0, 2.0, 3.0, None]}
122+
)
123+
df.attrs["provenance"] = "unit-test"
124+
df.index.name = "row"
125+
return df
126+
127+
128+
READ_ONLY_CALLS = {
129+
"run_suite": lambda df: fd.run_suite(
130+
df, ValidationSuite(name="t", rules=(ColumnRule(name="cat", allowed_values=("a", "b")),))
131+
),
132+
"validate_fields": lambda df: fd.validate_fields(df, {"n": "numeric"}),
133+
"profile": fd.profile,
134+
"suggest_plan": fd.suggest_plan,
135+
"infer_roles": fd.infer_roles,
136+
"explain_clean": fd.explain_clean,
137+
"clean": lambda df: fd.clean(df, verbose=False),
138+
}
139+
140+
141+
@pytest.mark.parametrize("name", sorted(READ_ONLY_CALLS))
142+
def test_the_input_frame_is_not_mutated(name):
143+
df = _annotated()
144+
before = _digest(df)
145+
READ_ONLY_CALLS[name](df)
146+
assert _digest(df) == before, f"{name} mutated its input"
147+
148+
149+
# -- cross-field ------------------------------------------------------------
150+
151+
152+
def _date_order(row):
153+
start, end = row.get("start_date"), row.get("end_date")
154+
if pd.isna(start) or pd.isna(end):
155+
return None
156+
return "start_date must be <= end_date" if start > end else None
157+
158+
159+
def _min_max(row):
160+
low, high = row.get("min_price"), row.get("max_price")
161+
if pd.isna(low) or pd.isna(high):
162+
return None
163+
return "min_price must be <= max_price" if low > high else None
164+
165+
166+
def _cross_frame() -> pd.DataFrame:
167+
return pd.DataFrame(
168+
{
169+
"start_date": pd.to_datetime(["2026-01-01", "2026-05-01", "2026-03-01", None]),
170+
"end_date": pd.to_datetime(["2026-02-01", "2026-04-01", "2026-04-01", "2026-04-01"]),
171+
"min_price": [10.0, 50.0, 5.0, 1.0],
172+
"max_price": [20.0, 40.0, 9.0, None],
173+
}
174+
)
175+
176+
177+
def test_both_cross_field_inversions_are_reported_on_the_offending_row():
178+
report = fd.validate_fields(_cross_frame(), cross_rules=(_date_order, _min_max))
179+
rows = {issue.row for issue in report.issues}
180+
assert rows == {1}, "only the inverted row should be reported"
181+
assert len([i for i in report.issues if i.row == 1]) == 2, "both rules should fire"
182+
assert all(i.classification == "cross_field_inconsistency" for i in report.issues)
183+
184+
185+
def test_a_missing_half_of_a_pair_is_not_a_violation():
186+
"""Row 3 has a null start_date and max_price; absence is not inversion."""
187+
report = fd.validate_fields(_cross_frame(), cross_rules=(_date_order, _min_max))
188+
assert all(issue.row != 3 for issue in report.issues)
189+
190+
191+
def test_a_cross_field_failure_routes_to_review_rather_than_repair():
192+
report = fd.validate_fields(_cross_frame(), cross_rules=(_date_order, _min_max))
193+
assert {i.action for i in report.issues} == {"manual_review"}
194+
195+
196+
# -- remediation integrity --------------------------------------------------
197+
198+
199+
def test_every_row_lands_in_exactly_one_bucket_and_stays_recoverable():
200+
df = pd.DataFrame({"id": ["a", "b", "c", "d"], "amount": [10.0, "apple", 30.0, 40.0]})
201+
report = fd.validate_fields(df, {"amount": "numeric"})
202+
result = fd.apply_field_policy(df, report)
203+
204+
total = (
205+
len(result.accepted)
206+
+ len(result.quarantined)
207+
+ len(result.rejected)
208+
+ len(result.needs_review)
209+
)
210+
assert total == len(df), "a row was silently dropped"
211+
assert "apple" in result.quarantined["amount"].astype(str).tolist()
212+
213+
214+
def test_the_audit_explains_the_decision_and_matches_the_outcome():
215+
df = pd.DataFrame({"id": ["a", "b", "c", "d"], "amount": [10.0, "apple", 30.0, 40.0]})
216+
report = fd.validate_fields(df, {"amount": "numeric"})
217+
result = fd.apply_field_policy(df, report)
218+
219+
(entry,) = [e for e in result.audit if e["row"] == 1]
220+
assert entry["original"] == "apple"
221+
assert entry["action"] == entry["applied"] == "quarantine"
222+
assert entry["classification"] == "semantic_mismatch"
223+
assert "not silently converted" in entry["reason"]
224+
# The report's claim and the frames must agree.
225+
assert 1 not in result.accepted.index
226+
assert 1 in result.quarantined.index

0 commit comments

Comments
 (0)