Skip to content

Commit 772457b

Browse files
test(idempotency): cleaning twice must be a fixed point in every mode (#479)
Idempotency was asserted for the balanced strategy only. tests/expectations.py drives {"balanced": {"idempotent": true}} for the fixtures and test_properties.py:17 covers defaults; nothing checked conservative or aggressive, any semantic_mode, a compiled context policy, lossy text options, imputation, outlier clipping, the native engines, or memory replay -- even though test_compare_matrix.py exercises all three strategies. It matters because the second pass sees its own output. A repair that is not a fixed point drifts on every rerun: case folding that re-triggers, a sentinel that re-matches what the first pass wrote, a dominant-variant vote that flips once the variants have changed, fences that keep pulling inward on each clip. A pipeline that cleans on ingest and again on export would keep moving the data. Fifteen cases: defaults; each of the three strategies; each of the three semantic modes; a context policy; string_case="lower"; impute="mean"; outliers="clip" against a frame carrying a real outlier; each of pandas, polars and duckdb; and memory replay. The engine cases use a uniformly-typed frame on purpose. A mixed-type object column makes polars and duckdb disclose a fallback to pandas, so the obvious frame would have quietly tested pandas three times. The test asserts report.backend == engine and an empty fallback_events, so if that ever changes the test fails instead of silently testing the wrong thing. All fifteen already hold, so these are regression guards. No library code changed. Full suite 6692 passed / 0 failed, coverage 93.92%; ruff clean repo-wide.
1 parent 567a30e commit 772457b

1 file changed

Lines changed: 122 additions & 0 deletions

File tree

‎tests/test_idempotency_modes.py‎

Lines changed: 122 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,122 @@
1+
"""Cleaning twice must equal cleaning once -- in every mode, not just one.
2+
3+
Idempotency was asserted for the ``balanced`` strategy only:
4+
``tests/expectations.py:174`` drives ``{"balanced": {"idempotent": true}}`` for
5+
the fixtures, and ``tests/test_properties.py:17`` covers defaults. Nothing
6+
checked ``conservative`` or ``aggressive``, any ``semantic_mode``, a compiled
7+
context policy, lossy text options, imputation, outlier clipping, the native
8+
engines, or memory replay -- even though ``test_compare_matrix.py`` exercises
9+
all three strategies.
10+
11+
That matters because a second pass sees its own output: a repair that is not a
12+
fixed point (case folding that re-triggers, a sentinel that re-matches, a
13+
dominant-variant vote that flips once the variants change) drifts on every
14+
rerun, and a pipeline that cleans on ingest and again on export would keep
15+
moving the data.
16+
"""
17+
18+
from __future__ import annotations
19+
20+
import pandas as pd
21+
import pytest
22+
23+
import freshdata as fd
24+
25+
26+
def _frame() -> pd.DataFrame:
27+
"""Dirty enough to engage several steps at once."""
28+
return pd.DataFrame(
29+
{
30+
"row_key": [f"r{i}" for i in range(10)],
31+
"customer_id": [f"{i:03d}" for i in range(1, 11)],
32+
"amount": [10.5, 20.0, " 30.5 ", 40.0, 50.0, "N/A", 70.0, "$1,200.50", 90.0, 100.0],
33+
"country": ["US", "GB", "FR", "DE", "JP", "CA", "AU", "BR", "IN", "US"],
34+
"gender": ["M", "F", "M", "F", "M", "F", "M", "F", "M", "F"],
35+
"note": ["a ", "b", " c", "d", "e", "f", "g", "h", "i", "j"],
36+
}
37+
)
38+
39+
40+
def _native_frame() -> pd.DataFrame:
41+
"""Uniformly typed, so polars and duckdb stay on their native path.
42+
43+
A mixed-type object column makes both engines disclose a fallback to
44+
pandas, which would make an "engine" test silently test pandas twice.
45+
"""
46+
return pd.DataFrame(
47+
{
48+
"row_key": [f"r{i}" for i in range(10)],
49+
"code": [f"{i:03d}" for i in range(1, 11)],
50+
"note": ["a ", "b", " c", "d", "N/A", "f", "g", "h", "i", "j"],
51+
"qty": [1, 2, 3, 4, 5, 6, 7, 8, 9, 10],
52+
}
53+
)
54+
55+
56+
def _assert_fixed_point(frame: pd.DataFrame, **kw) -> None:
57+
once = pd.DataFrame(fd.clean(frame, verbose=False, **kw))
58+
twice = pd.DataFrame(fd.clean(once.copy(), verbose=False, **kw))
59+
pd.testing.assert_frame_equal(twice, once)
60+
61+
62+
def test_defaults_are_a_fixed_point():
63+
_assert_fixed_point(_frame())
64+
65+
66+
@pytest.mark.parametrize("strategy", ["conservative", "balanced", "aggressive"])
67+
def test_every_strategy_is_a_fixed_point(strategy):
68+
"""Only `balanced` was covered before."""
69+
_assert_fixed_point(_frame(), strategy=strategy)
70+
71+
72+
@pytest.mark.parametrize("mode", ["assist", "review", "auto"])
73+
def test_every_semantic_mode_is_a_fixed_point(mode):
74+
"""`auto` matters most: it is the only mode that applies repairs itself."""
75+
_assert_fixed_point(_frame(), semantic_mode=mode)
76+
77+
78+
def test_a_compiled_context_policy_is_a_fixed_point():
79+
_assert_fixed_point(_frame(), context="Never modify customer_id values.")
80+
81+
82+
def test_lossy_text_options_are_a_fixed_point():
83+
"""Case folding must not re-trigger on its own output."""
84+
_assert_fixed_point(_frame(), string_case="lower")
85+
86+
87+
def test_imputation_is_a_fixed_point():
88+
"""The second pass sees no missing values, so it must do nothing."""
89+
_assert_fixed_point(_frame(), impute="mean")
90+
91+
92+
def test_outlier_clipping_is_a_fixed_point():
93+
"""Clipping must not keep pulling the fences inward."""
94+
df = _frame()
95+
df.loc[9, "amount"] = 9999.0
96+
_assert_fixed_point(df, outliers="clip")
97+
98+
99+
@pytest.mark.parametrize("engine", ["pandas", "polars", "duckdb"])
100+
def test_each_engine_is_a_fixed_point_on_its_native_path(engine):
101+
if engine != "pandas":
102+
pytest.importorskip(engine)
103+
kw = {"engine": engine, "strategy": "conservative", "fix_dtypes": False}
104+
once, report = fd.clean(_native_frame(), verbose=False, return_report=True, **kw)
105+
twice = fd.clean(pd.DataFrame(once).copy(), verbose=False, **kw)
106+
pd.testing.assert_frame_equal(pd.DataFrame(twice), pd.DataFrame(once))
107+
if engine != "pandas":
108+
# Guard the guard: if this fell back, the test would be testing pandas.
109+
assert report.backend == engine
110+
assert not report.fallback_events
111+
112+
113+
def test_memory_replay_is_a_fixed_point():
114+
"""Replaying learned decisions must not keep changing the frame."""
115+
memory = fd.CleaningMemory(dataset_id="idempotency")
116+
once = pd.DataFrame(
117+
fd.clean(_frame(), verbose=False, semantic_mode="auto", memory=memory)
118+
)
119+
twice = pd.DataFrame(
120+
fd.clean(once.copy(), verbose=False, semantic_mode="auto", memory=memory)
121+
)
122+
pd.testing.assert_frame_equal(twice, once)

0 commit comments

Comments
 (0)