From f1ae7adc3e05e8279ac152a1e810cb88f683b685 Mon Sep 17 00:00:00 2001 From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:17:02 +0530 Subject: [PATCH] fix(freshcore): fall back when native casts create columns the kernels mishandle The adapter's input checks only see the input dtypes, but FreshCore's native fix_dtypes stage casts text columns before imputation and outlier detection run. Two casts reach known kernel gaps: - "yes"/"no" text with missing values becomes a boolean column, which the native imputer skips, so mode/auto imputation leaves it unfilled while pandas fills it. - numeric text holding "inf" becomes a float column holding inf, which the native fences do not exclude, so zscore/iqr flag nothing while pandas drops inf before fencing. After the native run, the adapter now checks the returned column dtypes for text columns cast to bool or float. When a cast column hits either gap, it records a fallback naming the column (step "impute" or "outliers") and reruns on pandas. Under fallback_policy="error" it raises FallbackError. The scan covers only cast columns and runs only when impute or outliers is set, so other frames stay native. --- docs/fallback-matrix.md | 2 + docs/freshcore.md | 12 ++ .../execution/backends/_freshcore.py | 57 ++++++++ tests/test_execution/test_freshcore_engine.py | 129 ++++++++++++++++++ .../test_freshcore_native_parity.py | 117 ++++++++++++++++ 5 files changed, 317 insertions(+) diff --git a/docs/fallback-matrix.md b/docs/fallback-matrix.md index 22f2a2b4..6a0b96f2 100644 --- a/docs/fallback-matrix.md +++ b/docs/fallback-matrix.md @@ -26,10 +26,12 @@ FreshCore also runs its own config and data checks in | detection-only dedup (`drop_duplicates=False`) with `duplicate_ratio_action="error"` | native | native | native | native (pandas with native modules that don't report `duplicates_detected`) | the escalation needs the duplicate-row count at the pandas dedup stage; FreshCore counts it natively, but modules built before that count existed would never raise | | global impute mean/median/mode | native | native | native | native | — | | impute `mode`/`auto` with missing values in a nullable `boolean` column | native | native | native | pandas | FreshCore v1 kernels do not impute boolean columns | +| impute `mode`/`auto` when `fix_dtypes` casts a text column (e.g. `"yes"`/`"no"`) to boolean and it still has missing values | native | native | native | pandas | same kernel gap as above, but the column only becomes boolean inside the native run, so the adapter checks the native result (`fallback_step="impute"`) and reruns on pandas | | per-column `impute_strategy` | pandas | pandas | pandas | pandas | unimplemented natively (no fundamental blocker) | | `impute="missforest"` | pandas | pandas | pandas | pandas | scikit-learn model | | outliers `iqr` / `zscore` | native | native | native | native | — | | outliers when a float column holds `±inf` | native | native | native | pandas | FreshCore v1 fences don't exclude non-finite values, so they flag or clip nothing | +| outliers when `fix_dtypes` casts a text column holding `"inf"`/`"-inf"` to float | native | native | native | pandas | same fence gap as above, but the infinity only appears inside the native run, so the adapter checks the native result (`fallback_step="outliers"`) and reruns on pandas | | `outlier_action="auto"` / model methods | pandas | pandas | pandas | pandas | data-dependent / model-based selection | | `fix_dtypes=True` (default) | pandas | pandas | pandas | partial | sampled heuristics on the pandas reference; FreshCore casts bool/numeric natively, defers datetimes | | `drop_constant_columns` | pandas | pandas | pandas | pandas | needs a data scan before planning (two-phase plan not built) | diff --git a/docs/freshcore.md b/docs/freshcore.md index 6b6694c0..91361ebc 100644 --- a/docs/freshcore.md +++ b/docs/freshcore.md @@ -57,6 +57,18 @@ back for `impute="missforest"`, per-column `impute_strategy`, outlier handling on float columns holding `±inf`, and mode/auto imputation of nullable boolean columns with missing values. +The same two gaps can open inside a native run. `fix_dtypes` casts text columns +before imputation and outlier handling, so a `"yes"`/`"no"` column with missing +values becomes a boolean column the kernels do not impute. A numeric text +column holding `"inf"` becomes a float column holding `±inf`, which leaves the +outlier fences undefined. The input frame shows neither, so the adapter checks +the returned column dtypes. When a cast column hits a gap, it reruns the frame +on pandas and records a fallback event naming the column +(`fallback_step="impute"` or `"outliers"`). Under `fallback_policy="error"` it +raises `FallbackError` instead. `fd.plan()` checks only the input, so it +cannot predict these fallbacks, and the discarded native run still costs time. +Frames without such casts stay on the native path. + With `drop_duplicates=False` (the default), the native module counts full-row duplicates at the same stage as the pandas step: after string cleaning, empty-row removal and casts, and before imputation and outliers. It returns diff --git a/src/freshdata/execution/backends/_freshcore.py b/src/freshdata/execution/backends/_freshcore.py index bdc7a893..dff35471 100644 --- a/src/freshdata/execution/backends/_freshcore.py +++ b/src/freshdata/execution/backends/_freshcore.py @@ -105,6 +105,11 @@ def execute( "which this FreshCore native module does not report", ) + cast_fallback = self._native_cast_reason(native, frame, config) + if cast_fallback is not None: + step, reason = cast_fallback + return self._fallback(source, config, engine_config, step, reason) + cleaned, dtype_changes = self._frame_from_native(native, frame, config) report = self._report_from_native(frame, cleaned, native, started, config) for column, detail in dtype_changes: @@ -306,6 +311,58 @@ def _boolean_columns_to_impute(frame: pd.DataFrame) -> list[str]: found.append(str(col)) return found + def _native_cast_reason( + self, native: dict[str, Any], frame: pd.DataFrame, config: CleanConfig + ) -> tuple[str, str] | None: + """Fallback ``(step, reason)`` when a native cast built a column the kernels mishandle. + + ``_unsupported_reason`` sees only the input dtypes, but the native + ``fix_dtypes`` stage casts text columns before imputation and outlier + detection run. A text column cast to ``bool`` or ``float`` has the same + gaps as a boolean or ±inf input column: missing booleans are never + imputed, and ±inf leaves the outlier fences undefined. The returned + column dtypes show which columns were cast, so only those are scanned, + and only when the configured step would reach the gap. + """ + check_bools = config.impute in ("mode", "auto") + check_inf = config.outliers is not None + if not config.fix_dtypes or not (check_bools or check_inf): + return None + sources = { + str(label): (label, frame.iloc[:, i]) + for i, label in enumerate(self._output_labels(frame, config)) + } + bools: list[str] = [] + infinite: list[str] = [] + for column in native.get("columns", []): + source = sources.get(column["name"]) + if source is None: # e.g. an outlier flag column added natively + continue + label, series = source + dtype = column.get("dtype") + values = column["values"] + if check_bools and dtype == "bool" and not is_bool_dtype(series): + n_missing = values.count(None) + if 0 < n_missing < len(values): + bools.append(str(label)) + elif check_inf and dtype == "float" and not is_numeric_dtype(series): + floats = np.asarray(values, dtype="float64") # None -> NaN + if np.isinf(floats).any(): + infinite.append(str(label)) + if bools: + return "impute", ( + f"text column(s) {self._shown(bools)} were cast to boolean by FreshCore v1 " + "and still hold missing values: FreshCore v1 does not impute boolean " + "columns, so imputation requires the pandas reference path" + ) + if infinite: + return "outliers", ( + f"text column(s) {self._shown(infinite)} were cast to float64 by FreshCore v1 " + "and hold ±inf: FreshCore v1 outlier fences do not exclude ±inf, so outlier " + "handling requires the pandas reference path" + ) + return None + @staticmethod def _has_unsupported_object_values(frame: pd.DataFrame) -> bool: for col in frame.columns: diff --git a/tests/test_execution/test_freshcore_engine.py b/tests/test_execution/test_freshcore_engine.py index 08d78726..92475ec9 100644 --- a/tests/test_execution/test_freshcore_engine.py +++ b/tests/test_execution/test_freshcore_engine.py @@ -7,6 +7,7 @@ from __future__ import annotations import pandas as pd +import pytest import freshdata as fd from freshdata.execution import EngineConfig, EngineSelector @@ -99,6 +100,134 @@ def execute_plan(payload): assert report.to_dict()["stage_timings"][0]["backend"] == "freshcore" +class _CastingNative: + """Echoes the input but reports *cast* columns as a native fix_dtypes would.""" + + calls = 0 + cast: dict = {} + + @classmethod + def execute_plan(cls, payload): + cls.calls += 1 + columns = [cls.cast.get(c["name"], c) for c in payload["columns"]] + return {"columns": columns, "actions": []} + + +def _casting(monkeypatch, **cast) -> type[_CastingNative]: + native = type("CastingNative", (_CastingNative,), {"calls": 0, "cast": cast}) + monkeypatch.setattr(FreshCoreEngine, "_load_native", staticmethod(lambda: native)) + return native + + +def _cast_cfg(**kwargs) -> fd.CleanConfig: + return fd.CleanConfig(strategy="conservative", verbose=False, **kwargs) + + +_YES_NO = pd.DataFrame({"flag": ["yes", "no", None, "yes"], "k": [1.0, 2, 3, 4]}) +_BOOL_CAST = {"name": "flag", "dtype": "bool", "values": [True, False, None, True]} +_INF_TEXT = pd.DataFrame({"x": ["1", "2", "3", "100", "inf"]}) +_INF_CAST = {"name": "x", "dtype": "float", "values": [1.0, 2.0, 3.0, 100.0, float("inf")]} + + +@pytest.mark.parametrize("impute", ["mode", "auto"]) +def test_native_bool_cast_with_missing_values_falls_back_for_imputation(monkeypatch, impute): + native = _casting(monkeypatch, flag=_BOOL_CAST) + out, report = fd.clean( + _YES_NO.copy(), config=_cast_cfg(impute=impute), engine="freshcore", return_report=True + ) + + assert native.calls == 1 + assert report.backend == "pandas" + [event] = report.fallback_events + assert event["fallback_step"] == "impute" + assert "'flag'" in event["fallback_reason"] + assert "cast to boolean" in event["fallback_reason"] + assert out["flag"].tolist() == [True, False, True, True] + + +@pytest.mark.parametrize("method", ["zscore", "iqr"]) +def test_native_float_cast_holding_inf_falls_back_for_outliers(monkeypatch, method): + native = _casting(monkeypatch, x=_INF_CAST) + _, report = fd.clean( + _INF_TEXT.copy(), + config=_cast_cfg(outliers="flag", outlier_method=method), + engine="freshcore", + return_report=True, + ) + + assert native.calls == 1 + assert report.backend == "pandas" + [event] = report.fallback_events + assert event["fallback_step"] == "outliers" + assert "'x'" in event["fallback_reason"] + assert "±inf" in event["fallback_reason"] + + +@pytest.mark.parametrize( + ("frame", "cast", "options"), + [ + pytest.param(_YES_NO, {"flag": _BOOL_CAST}, {"impute": "mode"}, id="bool"), + pytest.param(_INF_TEXT, {"x": _INF_CAST}, {"outliers": "flag"}, id="inf"), + ], +) +def test_native_cast_fallback_honours_error_policy(monkeypatch, frame, cast, options): + _casting(monkeypatch, **cast) + with pytest.raises(fd.FallbackError): + fd.clean( + frame.copy(), + config=_cast_cfg(**options), + engine="freshcore", + fallback_policy="error", + ) + + +@pytest.mark.parametrize( + ("frame", "cast", "options"), + [ + pytest.param( + _YES_NO, {"flag": _BOOL_CAST}, {"outliers": "flag"}, id="bool-cast-without-impute" + ), + pytest.param( + _YES_NO, {"flag": _BOOL_CAST}, {"impute": "median"}, id="bool-cast-median-skips" + ), + pytest.param( + _YES_NO, + {"flag": {**_BOOL_CAST, "values": [True, False, False, True]}}, + {"impute": "mode"}, + id="bool-cast-without-missing", + ), + pytest.param( + _INF_TEXT, {"x": _INF_CAST}, {"impute": "median"}, id="inf-cast-without-outliers" + ), + pytest.param( + _INF_TEXT, + {"x": {**_INF_CAST, "values": [1.0, 2.0, 3.0, 100.0, None]}}, + {"outliers": "flag"}, + id="finite-float-cast", + ), + pytest.param( + pd.DataFrame({"x": ["a", "b", None], "k": [1.0, 2, 3]}), + {}, + {"impute": "mode", "outliers": "flag"}, + id="no-cast", + ), + ], +) +def test_frames_without_mishandled_casts_stay_native(monkeypatch, frame, cast, options): + native = _casting(monkeypatch, **cast) + _, report = fd.clean( + frame.copy(), + config=_cast_cfg(**options), + engine="freshcore", + fallback_policy="error", + return_report=True, + ) + + assert native.calls == 1 + assert report.backend == "freshcore" + assert report.fallback_events == [] + + def test_string_case_available_on_reference_pipeline(): df = pd.DataFrame({"name": ["Alice", "BOB"]}) out, report = fd.clean(df, config=_cfg(string_case="lower"), return_report=True) diff --git a/tests/test_execution/test_freshcore_native_parity.py b/tests/test_execution/test_freshcore_native_parity.py index e16acfe5..ce7c13b0 100644 --- a/tests/test_execution/test_freshcore_native_parity.py +++ b/tests/test_execution/test_freshcore_native_parity.py @@ -256,3 +256,120 @@ def test_detection_counts_before_imputation_like_pandas(native): df = pd.DataFrame({"a": [1.0, 2.0, None, 2.0], "b": ["x", "y", "y", "z"]}) expected, report = _clean_both(native, df, impute="mean", **REPRO) assert _detections(report) == _detections(expected) == [] + + +# -- native casts that build columns the kernels mishandle ------------------- +# +# ``fix_dtypes`` runs natively, so a text column can turn into a boolean column +# with missing values (never imputed natively) or a float column holding ±inf +# (which leaves the native outlier fences undefined). The input frame shows +# neither, so the adapter checks the native result and falls back. + + +def _yes_no_frame(missing: bool = True) -> pd.DataFrame: + flags = ["yes", "no", None if missing else "no", "yes", "Yes"] + return pd.DataFrame({"flag": flags, "k": [1.0, 2, 3, 4, 5]}) + + +def _inf_text_frame(values: list[str] | None = None) -> pd.DataFrame: + if values is None: + values = ["1", "2", "3", "4", "5", "6", "7", "8", "9", "100", "inf"] + return pd.DataFrame({"x": values}) + + +INF_ZSCORE = {"outliers": "flag", "outlier_method": "zscore", "outlier_factor": 2.0} +CAST_REPROS = [ + pytest.param(_yes_no_frame(), {"impute": "mode"}, "impute", "'flag'", id="bool-mode"), + pytest.param(_yes_no_frame(), {"impute": "auto"}, "impute", "'flag'", id="bool-auto"), + pytest.param(_inf_text_frame(), INF_ZSCORE, "outliers", "'x'", id="inf-zscore"), + pytest.param( + _inf_text_frame(["1", "2", "3", "4", "inf", "inf", "inf", "inf"]), + {"outliers": "flag", "outlier_method": "iqr"}, + "outliers", + "'x'", + id="inf-iqr", + ), + pytest.param( + _inf_text_frame(["1", "2", "3", "4", "5", "6", "7", "8", "9", "100", "-inf"]), + INF_ZSCORE, + "outliers", + "'x'", + id="negative-inf-zscore", + ), +] + + +def test_native_casts_diverge_without_the_fallback(): + """Pins the kernel behaviour the adapter guards against.""" + booleans = _native_result(_yes_no_frame(), impute="mode") + [flag] = [c for c in booleans["columns"] if c["name"] == "flag"] + assert flag["dtype"] == "bool" + assert None in flag["values"] + + infinite = _native_result(_inf_text_frame(), **INF_ZSCORE) + assert [c["name"] for c in infinite["columns"]] == ["x"] # no flag column + assert infinite["outliers_handled"] == 0 + + +@pytest.mark.parametrize(("df", "options", "step", "column"), CAST_REPROS) +def test_native_cast_repros_fall_back_to_the_pandas_result(native, df, options, step, column): + expected, _ = fd.clean(df.copy(), engine="pandas", return_report=True, **KW, **options) + out, report = fd.clean(df.copy(), engine="freshcore", return_report=True, **KW, **options) + + assert native.calls == 1 + assert report.backend == "pandas" + [event] = report.fallback_events + assert event["fallback_step"] == step + assert column in event["fallback_reason"] + pd.testing.assert_frame_equal(pd.DataFrame(out), pd.DataFrame(expected)) + + +def test_native_cast_repro_values_match_pandas(native): + out = fd.clean(_yes_no_frame(), engine="freshcore", impute="mode", **KW) + assert out["flag"].tolist() == [True, False, True, True, True] + + flagged = fd.clean(_inf_text_frame(), engine="freshcore", **INF_ZSCORE, **KW) + assert flagged["x_outlier"].tolist() == [False] * 9 + [True, True] + + +@pytest.mark.parametrize(("df", "options", "step", "column"), CAST_REPROS) +def test_native_cast_repros_raise_under_error_policy(native, df, options, step, column): + with pytest.raises(fd.FallbackError, match=f"step {step!r}"): + fd.clean(df.copy(), engine="freshcore", fallback_policy="error", **KW, **options) + assert native.calls == 1 + + +@pytest.mark.parametrize( + ("df", "options"), + [ + pytest.param(_yes_no_frame(missing=False), {"impute": "mode"}, id="bool-cast-no-missing"), + pytest.param(_yes_no_frame(), {"outliers": "flag"}, id="bool-cast-without-impute"), + pytest.param(_inf_text_frame(), {"impute": "median"}, id="inf-cast-without-outliers"), + pytest.param( + _inf_text_frame(["1", "2", "3", "4", "5", "6", "7", "8", "9", "100", None]), + {**INF_ZSCORE, "impute": "median"}, + id="finite-float-cast", + ), + pytest.param( + pd.DataFrame({"s": ["a", "b", None, "a"], "v": [1.0, 2.0, 3.0, 40.0]}), + {"impute": "mode", "outliers": "flag"}, + id="no-cast", + ), + ], +) +def test_frames_without_mishandled_casts_stay_native(native, df, options): + expected, _ = fd.clean(df.copy(), engine="pandas", return_report=True, **KW, **options) + out, report = fd.clean( + df.copy(), + engine="freshcore", + return_report=True, + fallback_policy="error", + **KW, + **options, + ) + + assert native.calls == 1 + assert report.backend == "freshcore" + assert report.fallback_events == [] + # Values match; dtypes may not (e.g. native boolean vs pandas bool). + pd.testing.assert_frame_equal(pd.DataFrame(out), pd.DataFrame(expected), check_dtype=False)