Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions docs/fallback-matrix.md
Original file line number Diff line number Diff line change
Expand Up @@ -26,10 +26,12 @@ FreshCore also runs its own config and data checks in
| detection-only dedup (`drop_duplicates=False`) with `duplicate_ratio_action="error"` | native | native | native | native (pandas with native modules that don't report `duplicates_detected`) | the escalation needs the duplicate-row count at the pandas dedup stage; FreshCore counts it natively, but modules built before that count existed would never raise |
| global impute mean/median/mode | native | native | native | native | — |
| impute `mode`/`auto` with missing values in a nullable `boolean` column | native | native | native | pandas | FreshCore v1 kernels do not impute boolean columns |
| impute `mode`/`auto` when `fix_dtypes` casts a text column (e.g. `"yes"`/`"no"`) to boolean and it still has missing values | native | native | native | pandas | same kernel gap as above, but the column only becomes boolean inside the native run, so the adapter checks the native result (`fallback_step="impute"`) and reruns on pandas |
| per-column `impute_strategy` | pandas | pandas | pandas | pandas | unimplemented natively (no fundamental blocker) |
| `impute="missforest"` | pandas | pandas | pandas | pandas | scikit-learn model |
| outliers `iqr` / `zscore` | native | native | native | native | — |
| outliers when a float column holds `±inf` | native | native | native | pandas | FreshCore v1 fences don't exclude non-finite values, so they flag or clip nothing |
| outliers when `fix_dtypes` casts a text column holding `"inf"`/`"-inf"` to float | native | native | native | pandas | same fence gap as above, but the infinity only appears inside the native run, so the adapter checks the native result (`fallback_step="outliers"`) and reruns on pandas |
| `outlier_action="auto"` / model methods | pandas | pandas | pandas | pandas | data-dependent / model-based selection |
| `fix_dtypes=True` (default) | pandas | pandas | pandas | partial | sampled heuristics on the pandas reference; FreshCore casts bool/numeric natively, defers datetimes |
| `drop_constant_columns` | pandas | pandas | pandas | pandas | needs a data scan before planning (two-phase plan not built) |
Expand Down
12 changes: 12 additions & 0 deletions docs/freshcore.md
Original file line number Diff line number Diff line change
Expand Up @@ -57,6 +57,18 @@ back for `impute="missforest"`, per-column `impute_strategy`, outlier handling
on float columns holding `±inf`, and mode/auto imputation of nullable boolean
columns with missing values.

The same two gaps can open inside a native run. `fix_dtypes` casts text columns
before imputation and outlier handling, so a `"yes"`/`"no"` column with missing
values becomes a boolean column the kernels do not impute. A numeric text
column holding `"inf"` becomes a float column holding `±inf`, which leaves the
outlier fences undefined. The input frame shows neither, so the adapter checks
the returned column dtypes. When a cast column hits a gap, it reruns the frame
on pandas and records a fallback event naming the column
(`fallback_step="impute"` or `"outliers"`). Under `fallback_policy="error"` it
raises `FallbackError` instead. `fd.plan()` checks only the input, so it
cannot predict these fallbacks, and the discarded native run still costs time.
Frames without such casts stay on the native path.

With `drop_duplicates=False` (the default), the native module counts full-row
duplicates at the same stage as the pandas step: after string cleaning,
empty-row removal and casts, and before imputation and outliers. It returns
Expand Down
57 changes: 57 additions & 0 deletions src/freshdata/execution/backends/_freshcore.py
Original file line number Diff line number Diff line change
Expand Up @@ -105,6 +105,11 @@ def execute(
"which this FreshCore native module does not report",
)

cast_fallback = self._native_cast_reason(native, frame, config)
if cast_fallback is not None:
step, reason = cast_fallback
return self._fallback(source, config, engine_config, step, reason)

cleaned, dtype_changes = self._frame_from_native(native, frame, config)
report = self._report_from_native(frame, cleaned, native, started, config)
for column, detail in dtype_changes:
Expand Down Expand Up @@ -306,6 +311,58 @@ def _boolean_columns_to_impute(frame: pd.DataFrame) -> list[str]:
found.append(str(col))
return found

def _native_cast_reason(
self, native: dict[str, Any], frame: pd.DataFrame, config: CleanConfig
) -> tuple[str, str] | None:
"""Fallback ``(step, reason)`` when a native cast built a column the kernels mishandle.

``_unsupported_reason`` sees only the input dtypes, but the native
``fix_dtypes`` stage casts text columns before imputation and outlier
detection run. A text column cast to ``bool`` or ``float`` has the same
gaps as a boolean or ±inf input column: missing booleans are never
imputed, and ±inf leaves the outlier fences undefined. The returned
column dtypes show which columns were cast, so only those are scanned,
and only when the configured step would reach the gap.
"""
check_bools = config.impute in ("mode", "auto")
check_inf = config.outliers is not None
if not config.fix_dtypes or not (check_bools or check_inf):
return None
sources = {
str(label): (label, frame.iloc[:, i])
for i, label in enumerate(self._output_labels(frame, config))
}
bools: list[str] = []
infinite: list[str] = []
for column in native.get("columns", []):
source = sources.get(column["name"])
if source is None: # e.g. an outlier flag column added natively
continue
label, series = source
dtype = column.get("dtype")
values = column["values"]
if check_bools and dtype == "bool" and not is_bool_dtype(series):
n_missing = values.count(None)
if 0 < n_missing < len(values):
bools.append(str(label))
elif check_inf and dtype == "float" and not is_numeric_dtype(series):
floats = np.asarray(values, dtype="float64") # None -> NaN
if np.isinf(floats).any():
infinite.append(str(label))
if bools:
return "impute", (
f"text column(s) {self._shown(bools)} were cast to boolean by FreshCore v1 "
"and still hold missing values: FreshCore v1 does not impute boolean "
"columns, so imputation requires the pandas reference path"
)
if infinite:
return "outliers", (
f"text column(s) {self._shown(infinite)} were cast to float64 by FreshCore v1 "
"and hold ±inf: FreshCore v1 outlier fences do not exclude ±inf, so outlier "
"handling requires the pandas reference path"
)
return None

@staticmethod
def _has_unsupported_object_values(frame: pd.DataFrame) -> bool:
for col in frame.columns:
Expand Down
129 changes: 129 additions & 0 deletions tests/test_execution/test_freshcore_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
from __future__ import annotations

import pandas as pd
import pytest

import freshdata as fd
from freshdata.execution import EngineConfig, EngineSelector
Expand Down Expand Up @@ -99,6 +100,134 @@ def execute_plan(payload):
assert report.to_dict()["stage_timings"][0]["backend"] == "freshcore"


class _CastingNative:
"""Echoes the input but reports *cast* columns as a native fix_dtypes would."""

calls = 0
cast: dict = {}

@classmethod
def execute_plan(cls, payload):
cls.calls += 1
columns = [cls.cast.get(c["name"], c) for c in payload["columns"]]
return {"columns": columns, "actions": []}


def _casting(monkeypatch, **cast) -> type[_CastingNative]:
native = type("CastingNative", (_CastingNative,), {"calls": 0, "cast": cast})
monkeypatch.setattr(FreshCoreEngine, "_load_native", staticmethod(lambda: native))
return native


def _cast_cfg(**kwargs) -> fd.CleanConfig:
return fd.CleanConfig(strategy="conservative", verbose=False, **kwargs)


_YES_NO = pd.DataFrame({"flag": ["yes", "no", None, "yes"], "k": [1.0, 2, 3, 4]})
_BOOL_CAST = {"name": "flag", "dtype": "bool", "values": [True, False, None, True]}
_INF_TEXT = pd.DataFrame({"x": ["1", "2", "3", "100", "inf"]})
_INF_CAST = {"name": "x", "dtype": "float", "values": [1.0, 2.0, 3.0, 100.0, float("inf")]}


@pytest.mark.parametrize("impute", ["mode", "auto"])
def test_native_bool_cast_with_missing_values_falls_back_for_imputation(monkeypatch, impute):
native = _casting(monkeypatch, flag=_BOOL_CAST)
out, report = fd.clean(
_YES_NO.copy(), config=_cast_cfg(impute=impute), engine="freshcore", return_report=True
)

assert native.calls == 1
assert report.backend == "pandas"
[event] = report.fallback_events
assert event["fallback_step"] == "impute"
assert "'flag'" in event["fallback_reason"]
assert "cast to boolean" in event["fallback_reason"]
assert out["flag"].tolist() == [True, False, True, True]


@pytest.mark.parametrize("method", ["zscore", "iqr"])
def test_native_float_cast_holding_inf_falls_back_for_outliers(monkeypatch, method):
native = _casting(monkeypatch, x=_INF_CAST)
_, report = fd.clean(
_INF_TEXT.copy(),
config=_cast_cfg(outliers="flag", outlier_method=method),
engine="freshcore",
return_report=True,
)

assert native.calls == 1
assert report.backend == "pandas"
[event] = report.fallback_events
assert event["fallback_step"] == "outliers"
assert "'x'" in event["fallback_reason"]
assert "±inf" in event["fallback_reason"]


@pytest.mark.parametrize(
("frame", "cast", "options"),
[
pytest.param(_YES_NO, {"flag": _BOOL_CAST}, {"impute": "mode"}, id="bool"),
pytest.param(_INF_TEXT, {"x": _INF_CAST}, {"outliers": "flag"}, id="inf"),
],
)
def test_native_cast_fallback_honours_error_policy(monkeypatch, frame, cast, options):
_casting(monkeypatch, **cast)
with pytest.raises(fd.FallbackError):
fd.clean(
frame.copy(),
config=_cast_cfg(**options),
engine="freshcore",
fallback_policy="error",
)


@pytest.mark.parametrize(
("frame", "cast", "options"),
[
pytest.param(
_YES_NO, {"flag": _BOOL_CAST}, {"outliers": "flag"}, id="bool-cast-without-impute"
),
pytest.param(
_YES_NO, {"flag": _BOOL_CAST}, {"impute": "median"}, id="bool-cast-median-skips"
),
pytest.param(
_YES_NO,
{"flag": {**_BOOL_CAST, "values": [True, False, False, True]}},
{"impute": "mode"},
id="bool-cast-without-missing",
),
pytest.param(
_INF_TEXT, {"x": _INF_CAST}, {"impute": "median"}, id="inf-cast-without-outliers"
),
pytest.param(
_INF_TEXT,
{"x": {**_INF_CAST, "values": [1.0, 2.0, 3.0, 100.0, None]}},
{"outliers": "flag"},
id="finite-float-cast",
),
pytest.param(
pd.DataFrame({"x": ["a", "b", None], "k": [1.0, 2, 3]}),
{},
{"impute": "mode", "outliers": "flag"},
id="no-cast",
),
],
)
def test_frames_without_mishandled_casts_stay_native(monkeypatch, frame, cast, options):
native = _casting(monkeypatch, **cast)
_, report = fd.clean(
frame.copy(),
config=_cast_cfg(**options),
engine="freshcore",
fallback_policy="error",
return_report=True,
)

assert native.calls == 1
assert report.backend == "freshcore"
assert report.fallback_events == []


def test_string_case_available_on_reference_pipeline():
df = pd.DataFrame({"name": ["Alice", "BOB"]})
out, report = fd.clean(df, config=_cfg(string_case="lower"), return_report=True)
Expand Down
117 changes: 117 additions & 0 deletions tests/test_execution/test_freshcore_native_parity.py
Original file line number Diff line number Diff line change
Expand Up @@ -256,3 +256,120 @@ def test_detection_counts_before_imputation_like_pandas(native):
df = pd.DataFrame({"a": [1.0, 2.0, None, 2.0], "b": ["x", "y", "y", "z"]})
expected, report = _clean_both(native, df, impute="mean", **REPRO)
assert _detections(report) == _detections(expected) == []


# -- native casts that build columns the kernels mishandle -------------------
#
# ``fix_dtypes`` runs natively, so a text column can turn into a boolean column
# with missing values (never imputed natively) or a float column holding ±inf
# (which leaves the native outlier fences undefined). The input frame shows
# neither, so the adapter checks the native result and falls back.


def _yes_no_frame(missing: bool = True) -> pd.DataFrame:
flags = ["yes", "no", None if missing else "no", "yes", "Yes"]
return pd.DataFrame({"flag": flags, "k": [1.0, 2, 3, 4, 5]})


def _inf_text_frame(values: list[str] | None = None) -> pd.DataFrame:
if values is None:
values = ["1", "2", "3", "4", "5", "6", "7", "8", "9", "100", "inf"]
return pd.DataFrame({"x": values})


INF_ZSCORE = {"outliers": "flag", "outlier_method": "zscore", "outlier_factor": 2.0}
CAST_REPROS = [
pytest.param(_yes_no_frame(), {"impute": "mode"}, "impute", "'flag'", id="bool-mode"),
pytest.param(_yes_no_frame(), {"impute": "auto"}, "impute", "'flag'", id="bool-auto"),
pytest.param(_inf_text_frame(), INF_ZSCORE, "outliers", "'x'", id="inf-zscore"),
pytest.param(
_inf_text_frame(["1", "2", "3", "4", "inf", "inf", "inf", "inf"]),
{"outliers": "flag", "outlier_method": "iqr"},
"outliers",
"'x'",
id="inf-iqr",
),
pytest.param(
_inf_text_frame(["1", "2", "3", "4", "5", "6", "7", "8", "9", "100", "-inf"]),
INF_ZSCORE,
"outliers",
"'x'",
id="negative-inf-zscore",
),
]


def test_native_casts_diverge_without_the_fallback():
"""Pins the kernel behaviour the adapter guards against."""
booleans = _native_result(_yes_no_frame(), impute="mode")
[flag] = [c for c in booleans["columns"] if c["name"] == "flag"]
assert flag["dtype"] == "bool"
assert None in flag["values"]

infinite = _native_result(_inf_text_frame(), **INF_ZSCORE)
assert [c["name"] for c in infinite["columns"]] == ["x"] # no flag column
assert infinite["outliers_handled"] == 0


@pytest.mark.parametrize(("df", "options", "step", "column"), CAST_REPROS)
def test_native_cast_repros_fall_back_to_the_pandas_result(native, df, options, step, column):
expected, _ = fd.clean(df.copy(), engine="pandas", return_report=True, **KW, **options)
out, report = fd.clean(df.copy(), engine="freshcore", return_report=True, **KW, **options)

assert native.calls == 1
assert report.backend == "pandas"
[event] = report.fallback_events
assert event["fallback_step"] == step
assert column in event["fallback_reason"]
pd.testing.assert_frame_equal(pd.DataFrame(out), pd.DataFrame(expected))


def test_native_cast_repro_values_match_pandas(native):
out = fd.clean(_yes_no_frame(), engine="freshcore", impute="mode", **KW)
assert out["flag"].tolist() == [True, False, True, True, True]

flagged = fd.clean(_inf_text_frame(), engine="freshcore", **INF_ZSCORE, **KW)
assert flagged["x_outlier"].tolist() == [False] * 9 + [True, True]


@pytest.mark.parametrize(("df", "options", "step", "column"), CAST_REPROS)
def test_native_cast_repros_raise_under_error_policy(native, df, options, step, column):
with pytest.raises(fd.FallbackError, match=f"step {step!r}"):
fd.clean(df.copy(), engine="freshcore", fallback_policy="error", **KW, **options)
assert native.calls == 1


@pytest.mark.parametrize(
("df", "options"),
[
pytest.param(_yes_no_frame(missing=False), {"impute": "mode"}, id="bool-cast-no-missing"),
pytest.param(_yes_no_frame(), {"outliers": "flag"}, id="bool-cast-without-impute"),
pytest.param(_inf_text_frame(), {"impute": "median"}, id="inf-cast-without-outliers"),
pytest.param(
_inf_text_frame(["1", "2", "3", "4", "5", "6", "7", "8", "9", "100", None]),
{**INF_ZSCORE, "impute": "median"},
id="finite-float-cast",
),
pytest.param(
pd.DataFrame({"s": ["a", "b", None, "a"], "v": [1.0, 2.0, 3.0, 40.0]}),
{"impute": "mode", "outliers": "flag"},
id="no-cast",
),
],
)
def test_frames_without_mishandled_casts_stay_native(native, df, options):
expected, _ = fd.clean(df.copy(), engine="pandas", return_report=True, **KW, **options)
out, report = fd.clean(
df.copy(),
engine="freshcore",
return_report=True,
fallback_policy="error",
**KW,
**options,
)

assert native.calls == 1
assert report.backend == "freshcore"
assert report.fallback_events == []
# Values match; dtypes may not (e.g. native boolean vs pandas bool).
pd.testing.assert_frame_equal(pd.DataFrame(out), pd.DataFrame(expected), check_dtype=False)
Loading