Skip to content

Commit a2d4e73

Browse files
fix(labels): accept missing labels, keep label identity, reject duplicates alike
Four label defects found by fuzzing: * pandas coerces Index([0, None]) to float64 with NaN, and then cannot match that label against the frame's own columns, so DataFrame.duplicated() raised KeyError(Index([nan])) - breaking clean, profile, infer_roles and explain_clean on frames every other step handles. Duplicate detection now addresses columns by position when a label is missing. * infer_roles collected the labels into one Series, which coerces a mixed numeric/None or int/float label set, so frame[row["column"]] no longer round-tripped. The column is built with object dtype instead. * suggest_plan and plan raised TypeError("cannot convert the series to int") on duplicate column labels with the semantic layer active, where infer_roles and explain_clean raise a clear ValueError. * clean_text and lint_text_encoding raised AttributeError on the same input. The guard moves to _util.require_unique_labels so every entry point shares one message; api keeps its private alias. Closes #459 Closes #461 Closes #462 Closes #437
1 parent 3b61888 commit a2d4e73

9 files changed

Lines changed: 165 additions & 16 deletions

File tree

‎src/freshdata/_util.py‎

Lines changed: 46 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -104,6 +104,52 @@ def add_column(df: pd.DataFrame, name: object, values: object) -> None:
104104
PANDAS_MAJOR: int = int(pd.__version__.split(".")[0])
105105

106106

107+
def require_unique_labels(frame: pd.DataFrame, func: str) -> None:
108+
"""Reject duplicate column labels, which make ``frame[col]`` a DataFrame."""
109+
if not frame.columns.is_unique:
110+
duplicated = sorted({str(c) for c in frame.columns[frame.columns.duplicated()]})
111+
raise ValueError(f"{func} requires unique column labels; duplicated: {duplicated}")
112+
113+
114+
def _same_label(left: Any, right: Any) -> bool:
115+
"""Label equality that matches two missing labels."""
116+
if left is right:
117+
return True
118+
try:
119+
if pd.isna(left) and pd.isna(right):
120+
return True
121+
except (TypeError, ValueError):
122+
pass
123+
try:
124+
return bool(left == right)
125+
except Exception: # noqa: BLE001 - exotic labels compare however they like
126+
return False
127+
128+
129+
def duplicated_mask(
130+
df: pd.DataFrame, subset: Any = None, keep: Any = "first"
131+
) -> pd.Series[bool]:
132+
"""``df.duplicated`` for frames whose column labels may be missing (#461).
133+
134+
``Index([0, None])`` coerces to float64 with NaN, and pandas then fails to
135+
match that label against the frame's own columns, so duplicate detection
136+
raised ``KeyError`` on frames every other step handles. Addressing the
137+
columns by position sidesteps the lookup entirely.
138+
"""
139+
labels = df.columns
140+
if not labels.isna().any():
141+
return df.duplicated(subset=subset, keep=keep)
142+
work = df.set_axis(pd.RangeIndex(len(labels)), axis=1)
143+
if subset is None:
144+
return work.duplicated(keep=keep)
145+
wanted = [
146+
position
147+
for position, label in enumerate(labels)
148+
if any(_same_label(label, s) for s in subset)
149+
]
150+
return work.duplicated(subset=wanted, keep=keep)
151+
152+
107153
def json_scalar(value: Any) -> Any:
108154
"""One value in a JSON-representable form (``repr`` as a last resort).
109155

‎src/freshdata/api.py‎

Lines changed: 10 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -11,7 +11,7 @@
1111

1212
from ._csv_io import leading_zero_dtypes
1313
from ._reportframe import ReportFrame
14-
from ._util import sanitize_csv_formulas
14+
from ._util import require_unique_labels, sanitize_csv_formulas
1515
from .adapters.polars import from_pandas, to_pandas
1616
from .cleaner import Cleaner, run_pipeline
1717
from .config import CleanConfig, merge_options
@@ -1152,11 +1152,8 @@ def _engine_mode(cfg: CleanConfig) -> EngineMode:
11521152
return "balanced" if mode == "balanced" else "aggressive"
11531153

11541154

1155-
def _require_unique_labels(frame: pd.DataFrame, func: str) -> None:
1156-
"""Reject duplicate column labels, which make ``frame[col]`` a DataFrame."""
1157-
if not frame.columns.is_unique:
1158-
duplicated = sorted({str(c) for c in frame.columns[frame.columns.duplicated()]})
1159-
raise ValueError(f"{func} requires unique column labels; duplicated: {duplicated}")
1155+
#: Shared with the other entry points that index frames by label.
1156+
_require_unique_labels = require_unique_labels
11601157

11611158

11621159
def infer_roles(
@@ -1217,7 +1214,13 @@ def infer_roles(
12171214
),
12181215
}
12191216
)
1220-
return ReportFrame.wrap(pd.DataFrame(rows), "infer_roles")
1217+
out = pd.DataFrame(rows)
1218+
if rows:
1219+
# Collecting the labels into a Series coerces a mixed numeric/None set
1220+
# (0 and None become 0.0 and NaN), so frame[label] no longer round-trips
1221+
# (#462). Object dtype keeps every label exactly as it came in.
1222+
out["column"] = pd.Series([r["column"] for r in rows], dtype=object)
1223+
return ReportFrame.wrap(out, "infer_roles")
12211224

12221225

12231226
def profile(

‎src/freshdata/engine/context.py‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -22,7 +22,7 @@
2222
import pandas as pd
2323
from pandas.api.types import is_bool_dtype, is_datetime64_any_dtype, is_numeric_dtype
2424

25-
from .._util import _is_stringlike_dtype
25+
from .._util import _is_stringlike_dtype, duplicated_mask
2626
from ..config import CleanConfig
2727
from ..steps.outliers import safe_skew
2828

@@ -323,7 +323,7 @@ def build_contexts(
323323
duplicated_rows = None
324324
if stats is None and len(df) and columns:
325325
try:
326-
mask = df.duplicated()
326+
mask = duplicated_mask(df)
327327
except (TypeError, NotImplementedError): # unhashable cells / nested Arrow
328328
mask = None
329329
if mask is not None and mask.any():

‎src/freshdata/plan.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -10,6 +10,7 @@
1010
from pandas.api.types import is_bool_dtype, is_numeric_dtype
1111

1212
from ._reportframe import ReportFrame
13+
from ._util import require_unique_labels
1314
from .cleaner import run_pipeline
1415
from .config import CleanConfig, merge_options
1516
from .engine.context import build_contexts
@@ -341,6 +342,7 @@ def suggest_plan(
341342
:func:`freshdata.clean` does — user options and policy always win, and
342343
severe schema drift disables the fold entirely.
343344
"""
345+
require_unique_labels(df, "suggest_plan")
344346
if context is not None:
345347
options["context"] = context
346348
if policy is not None:

‎src/freshdata/profile.py‎

Lines changed: 8 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -14,7 +14,13 @@
1414
import pandas as pd
1515
from pandas.api.types import infer_dtype, is_bool_dtype, is_numeric_dtype
1616

17-
from ._util import _is_stringlike_dtype, format_bytes, json_scalar, memory_bytes
17+
from ._util import (
18+
_is_stringlike_dtype,
19+
duplicated_mask,
20+
format_bytes,
21+
json_scalar,
22+
memory_bytes,
23+
)
1824
from .config import CleanConfig
1925
from .render.mixins import HtmlReprMixin
2026
from .steps.dtypes import suggest_conversion
@@ -253,7 +259,7 @@ def build_profile(
253259
duplicate_rows: int | None = None
254260
else:
255261
try:
256-
duplicate_rows = int(work.duplicated().sum())
262+
duplicate_rows = int(duplicated_mask(work).sum())
257263
except (TypeError, NotImplementedError):
258264
# Unhashable cells: object lists/dicts raise TypeError, nested Arrow
259265
# dtypes (list/struct/map) raise ArrowNotImplementedError, a

‎src/freshdata/steps/duplicates.py‎

Lines changed: 4 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -25,6 +25,7 @@
2525
import pandas as pd
2626
from pandas.api.types import is_bool_dtype, is_numeric_dtype
2727

28+
from .._util import duplicated_mask
2829
from ..config import CleanConfig
2930
from ..report import CleanReport
3031

@@ -178,7 +179,7 @@ def drop_duplicate_rows(df: pd.DataFrame, config: CleanConfig,
178179
return df
179180
subset = _validated_subset(df, config)
180181
try:
181-
dup_any = df.duplicated(subset=subset, keep="first")
182+
dup_any = duplicated_mask(df, subset=subset, keep="first")
182183
except (TypeError, NotImplementedError): # nested Arrow: ArrowNotImplementedError
183184
report.add("drop_duplicates",
184185
"skipped: column(s) contain unhashable values (e.g. lists)")
@@ -223,9 +224,9 @@ def drop_duplicate_rows(df: pd.DataFrame, config: CleanConfig,
223224
df, subset, protected=hard_protected_columns(config, df.columns)
224225
)
225226
if keep in ("first", "last"):
226-
df = _filter_rows(df, ~df.duplicated(subset=subset, keep=keep))
227+
df = _filter_rows(df, ~duplicated_mask(df, subset=subset, keep=keep))
227228
elif keep == "drop":
228-
df = _filter_rows(df, ~df.duplicated(subset=subset, keep=False))
229+
df = _filter_rows(df, ~duplicated_mask(df, subset=subset, keep=False))
229230

230231
n_removed = n_before - len(df)
231232
verb = {"first": "dropped", "last": "dropped", "drop": "dropped",

‎src/freshdata/textclean.py‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -28,7 +28,7 @@
2828

2929
import pandas as pd
3030

31-
from ._util import _is_stringlike_dtype
31+
from ._util import _is_stringlike_dtype, require_unique_labels
3232

3333
__all__ = [
3434
"TextCleanConfig",
@@ -313,6 +313,7 @@ def clean_text(
313313
restricted via :func:`config_for_field` so e.g. punctuation stripping
314314
never runs on an amount or identifier column.
315315
"""
316+
require_unique_labels(df, "clean_text")
316317
if columns is None:
317318
cols = [c for c in df.columns if _is_stringlike_dtype(df[c].dtype)]
318319
else:

‎src/freshdata/textlint.py‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -20,7 +20,7 @@
2020

2121
import pandas as pd
2222

23-
from ._util import stringlike_columns
23+
from ._util import require_unique_labels, stringlike_columns
2424
from .render import html as H
2525
from .render.mixins import SimpleHtmlReport
2626

@@ -266,6 +266,7 @@ def lint_text_encoding(
266266
TextLintReport
267267
"""
268268
hints = list(locale_hints or [])
269+
require_unique_labels(df, "lint_text_encoding")
269270
if columns is None:
270271
cols = list(stringlike_columns(df))
271272
else:

‎tests/test_label_robustness.py‎

Lines changed: 89 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,89 @@
1+
"""Column labels that pandas itself handles awkwardly: missing, coerced, repeated.
2+
3+
Regressions for #459 (suggest_plan on duplicate labels), #461 (numeric-or-None
4+
labels), #462 (label identity in infer_roles) and #437 (the text entry points).
5+
"""
6+
7+
from __future__ import annotations
8+
9+
import warnings
10+
11+
import pandas as pd
12+
import pytest
13+
14+
import freshdata as fd
15+
16+
warnings.simplefilter("ignore")
17+
18+
19+
# ── #461: a NaN column label must not break duplicate detection ────────────────
20+
21+
22+
def _nan_label_frame() -> pd.DataFrame:
23+
# pandas coerces Index([0, None]) to float64, so the second label is NaN and
24+
# cannot be looked up by value — DataFrame.duplicated() raised KeyError.
25+
return pd.DataFrame({0: [1, 2, 1], None: [3, 4, 3]})
26+
27+
28+
@pytest.mark.parametrize("call", [
29+
lambda df: fd.clean(df, verbose=False),
30+
fd.profile,
31+
fd.infer_roles,
32+
fd.explain_clean,
33+
])
34+
def test_numeric_or_none_labels_are_accepted(call):
35+
assert call(_nan_label_frame()) is not None
36+
37+
38+
def test_duplicate_rows_are_still_detected_with_a_nan_label():
39+
_, report = fd.clean(
40+
_nan_label_frame(), drop_duplicates=True, return_report=True, verbose=False
41+
)
42+
assert any(a.step == "drop_duplicates" and a.count == 1 for a in report.actions)
43+
44+
45+
def test_duplicate_subset_still_applies_with_a_nan_label():
46+
df = pd.DataFrame({0: [1, 1, 2], None: [9, 8, 7]})
47+
out = fd.clean(
48+
df, drop_duplicates=True, duplicate_subset=[0.0], verbose=False
49+
)
50+
assert len(out) == 2 # deduplicated on the first column only
51+
52+
53+
# ── #462: infer_roles reports the labels the frame actually has ────────────────
54+
55+
56+
@pytest.mark.parametrize("labels", [[0, None], [-2, 0.78], ["a", 1]])
57+
def test_infer_roles_keeps_label_identity(labels):
58+
df = pd.DataFrame([[1, 2], [3, 4]])
59+
df.columns = pd.Index(labels, dtype=object)
60+
reported = fd.infer_roles(df)["column"].tolist()
61+
assert reported == sorted(labels, key=str) # rows are ordered by label text
62+
for label in reported:
63+
assert df[label].shape == (2,) # the documented round-trip
64+
65+
66+
# ── #459 / #437: duplicate labels raise the same error everywhere ──────────────
67+
68+
69+
@pytest.mark.parametrize("call", [
70+
fd.suggest_plan,
71+
fd.plan,
72+
fd.clean_text,
73+
fd.lint_text_encoding,
74+
fd.infer_roles,
75+
])
76+
def test_duplicate_labels_raise_value_error(call):
77+
df = pd.DataFrame([[1, 2], [3, 4]], columns=["a", "a"])
78+
with pytest.raises(ValueError, match="requires unique column labels"):
79+
call(df)
80+
assert df.columns.tolist() == ["a", "a"] # never modified
81+
82+
83+
def test_unique_labels_still_work_on_those_entry_points():
84+
df = pd.DataFrame({"a": [" x ", "y"], "n": [1, 2]})
85+
assert fd.suggest_plan(df) is not None
86+
assert fd.plan(df) is not None
87+
cleaned, _ = fd.clean_text(df)
88+
assert cleaned["a"].tolist() == ["x", "y"]
89+
assert fd.lint_text_encoding(df) is not None

0 commit comments

Comments
 (0)