Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,11 @@ adheres to [Semantic Versioning](https://semver.org/).
## [Unreleased]

### Fixed
- `CleanReport.revert()` no longer writes a restored value into other rows
that share a duplicate index label. The undo log now records positional
offsets and revert restores by position, so a frame with a non-unique index
reverts exactly to its input instead of overwriting rows that never held the
value (older reports without positions still revert best-effort by label).
- The native Polars backend no longer raises on polars versions that reject
`collect(engine="streaming")` with a `ValueError` (polars 1.1–1.24): the
streaming collect now falls back to a plain `collect()` on those versions,
Expand Down
12 changes: 8 additions & 4 deletions src/freshdata/repairplan.py
Original file line number Diff line number Diff line change
Expand Up @@ -702,13 +702,17 @@ def _capture_undo(
for action in executed:
column = str(action.column)
raw = action.params.get("raw_value")
indices = out.index[out[column] == raw].tolist()
if len(indices) <= budget:
budget -= len(indices)
mask = (out[column] == raw).to_numpy()
# Positional offsets, not just index labels: a non-unique index would
# make a label-based revert overwrite every row sharing the label.
positions = [i for i, hit in enumerate(mask) if hit]
indices = out.index[mask].tolist()
if len(positions) <= budget:
budget -= len(positions)
dtypes.setdefault(column, str(out[column].dtype))
entries.append(
{"action_id": action.id, "column": column, "index": indices,
"value": raw}
"positions": positions, "value": raw}
)
action.reversible = True
else:
Expand Down
14 changes: 11 additions & 3 deletions src/freshdata/report.py
Original file line number Diff line number Diff line change
Expand Up @@ -458,9 +458,17 @@ def revert(
if series.dtype != object:
series = series.astype(object)
for entry in column_entries:
labels = [i for i in entry["index"] if i in series.index]
if labels:
series.loc[labels] = entry["value"]
positions = entry.get("positions")
if positions is not None:
# Revert by position: a non-unique index makes label-based
# ``.loc`` write the value into every row sharing the label.
valid = [p for p in positions if 0 <= p < len(series)]
if valid:
series.iloc[valid] = entry["value"]
else: # older report without positions: best-effort by label
labels = [i for i in entry["index"] if i in series.index]
if labels:
series.loc[labels] = entry["value"]
original_dtype = (self.undo_log.get("column_dtypes") or {}).get(column)
if original_dtype is not None:
# Mixed values after a partial revert legitimately stay object.
Expand Down
16 changes: 16 additions & 0 deletions tests/test_report.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
import json
import unicodedata
from datetime import datetime, timezone
from pathlib import Path

Expand Down Expand Up @@ -286,3 +287,18 @@ def test_bool_reflects_whether_anything_changed(messy, already_clean):
def test_repr_is_compact(messy):
_, report = fd.clean(messy, return_report=True)
assert repr(report).startswith("<CleanReport:")


def test_revert_restores_only_original_rows_on_duplicate_index():
"""revert() must not write a restored value into other rows that share a
duplicate index label (FDC-L3-025: silent data fabrication)."""
df = pd.DataFrame(
{"name": [unicodedata.normalize("NFD", "caf\u00e9"), None, None]},
index=[0, 0, 0],
)
plan = fd.suggest_plan(df, semantic_mode="auto", verbose=False).repair_plan
plan.approve_all(max_risk="high")
cleaned, report = fd.apply_plan(df, plan, keep_undo=True)
restored = report.revert(cleaned)
assert list(restored["name"]) == list(df["name"])
assert int(restored["name"].isna().sum()) == 2
Loading