diff --git a/crates/freshcore/src/kernels/outliers.rs b/crates/freshcore/src/kernels/outliers.rs index 27790ac2..8c0a09a2 100644 --- a/crates/freshcore/src/kernels/outliers.rs +++ b/crates/freshcore/src/kernels/outliers.rs @@ -91,14 +91,47 @@ pub fn bounds(values: &[Option], method: &str, factor: f64) -> Option<(f64, return Some((mean - factor * std, mean + factor * std)); } let q1 = quantile(vals.clone(), 0.25)?; - let q3 = quantile(vals, 0.75)?; + let q3 = quantile(vals.clone(), 0.75)?; let spread = q3 - q1; - if spread == 0.0 || spread.is_nan() { + if spread.is_nan() { return None; } + if spread == 0.0 { + return zero_iqr_bounds(&vals, q1, factor); + } Some((q1 - factor * spread, q3 + factor * spread)) } +/// IQR-equivalent spread per unit of mean absolute deviation for normal data: +/// sqrt(pi/2) (MeanAD -> sigma) x 2*Phi^-1(0.75) (sigma -> IQR). Mirrors +/// `freshdata.steps.outliers.MEANAD_TO_IQR`. +const MEANAD_TO_IQR: f64 = 1.6906950787902986; +/// Largest share of values the zero-IQR fallback may flag; mirrors +/// `freshdata.steps.outliers.ZERO_IQR_MAX_SHARE`. +const ZERO_IQR_MAX_SHARE: f64 = 0.05; + +/// Fences when both quartiles equal `center` (at least half the column sits on +/// one value). A zero IQR must not silently disable detection. Instead, the IQR +/// is replaced by the normal-consistent spread from the mean absolute deviation +/// around the median (the median equals the quartiles here). The result is +/// `None` for a constant column, or when the fences would flag more than +/// `ZERO_IQR_MAX_SHARE` of the values: that many is a second mode, not rare +/// outliers. +fn zero_iqr_bounds(vals: &[f64], center: f64, factor: f64) -> Option<(f64, f64)> { + let n = vals.len() as f64; + let mean_ad = vals.iter().map(|v| (v - center).abs()).sum::() / n; + if mean_ad == 0.0 || mean_ad.is_nan() { + return None; + } + let spread = MEANAD_TO_IQR * mean_ad; + let (lo, hi) = (center - factor * spread, center + factor * spread); + let flagged = vals.iter().filter(|v| **v < lo || **v > hi).count() as f64; + if flagged > ZERO_IQR_MAX_SHARE * n { + return None; + } + Some((lo, hi)) +} + fn quantile(mut vals: Vec, q: f64) -> Option { if vals.is_empty() { return None; @@ -136,4 +169,45 @@ mod tests { assert!(lo < 1.0); assert!(hi < 100.0); } + + fn zeros_with(extra: &[f64], zeros: usize) -> Vec> { + std::iter::repeat(0.0) + .take(zeros) + .chain(extra.iter().copied()) + .map(Some) + .collect() + } + + #[test] + fn zero_iqr_falls_back_to_mean_absolute_deviation() { + let values = zeros_with(&[1000.0, 5000.0, 2.0, 3.0, -800.0], 95); + let (lo, hi) = bounds(&values, "iqr", 1.5).unwrap(); + // MeanAD = 6805 / 100; fences = 0 +/- 1.5 x 1.6907 x MeanAD. + let expected = 1.5 * (MEANAD_TO_IQR * (6805.0 / 100.0)); + assert_eq!((lo, hi), (-expected, expected)); + let mut frame = Frame { + nrows: values.len(), + columns: vec![Column { + name: "x".into(), + data: ColumnData::Float(values), + }], + }; + let actions = handle_outliers(&mut frame, Some("clip"), "iqr", 1.5); + assert_eq!(actions.len(), 1); + assert_eq!(actions[0].count, 3); // the spikes, not 2.0 / 3.0 + } + + #[test] + fn zero_iqr_fallback_does_not_flag_a_second_mode() { + // 30% non-zero, split around zero: the IQR is still zero. + let extra: Vec = (1..=15).flat_map(|i| [-(i as f64), i as f64]).collect(); + assert_eq!(bounds(&zeros_with(&extra, 70), "iqr", 1.5), None); + } + + #[test] + fn constant_column_has_no_bounds() { + let values = vec![Some(5.0); 20]; + assert_eq!(bounds(&values, "iqr", 1.5), None); + assert_eq!(bounds(&values, "zscore", 3.0), None); + } } diff --git a/docs/backends.md b/docs/backends.md index 8a95422c..0db98e83 100644 --- a/docs/backends.md +++ b/docs/backends.md @@ -174,7 +174,11 @@ Notes: - **Outlier counts** match the pandas reference where the quantile statistics match (Polars/DuckDB linear interpolation); Spark may flag a different count. pandas, Polars, DuckDB and Spark compute the fences from finite values only; - `±inf` is still tested against them. + `±inf` is still tested against them. For a non-constant column whose IQR is + zero, pandas and FreshCore fall back to fences from the mean absolute + deviation (see [the cleaning engine](cleaning-engine.md#outliers)). Polars, + DuckDB and Spark still skip such columns, so they can flag fewer outliers + there. - **Float `NaN`** is read as missing by Polars, DuckDB and Spark, as in pandas (Spark keeps `NaN` as a value in float/double columns, so it is converted to null on ingestion). diff --git a/docs/cleaning-engine.md b/docs/cleaning-engine.md index 80a09761..948c2afc 100644 --- a/docs/cleaning-engine.md +++ b/docs/cleaning-engine.md @@ -52,6 +52,21 @@ Detection methods: IQR fences (default), z-score, `"auto"` (z-score for ~normal columns, IQR for skewed), or `"isolation_forest"` (scikit-learn, ≥ 100 rows, falls back to IQR). +**Zero IQR.** When at least half of a non-constant column sits on one value +(Q1 = Q3, for example a mostly-zero column with a few spikes), Tukey fences +would collapse to that value. IQR detection does not switch off in that case. +The IQR is replaced by its normal-consistent equivalent from the mean absolute +deviation around the median: `1.69 × MeanAD`, i.e. `sqrt(pi/2) × MeanAD` as the +σ estimate, the usual fallback when the MAD is zero, times 1.349. The fences +are `median ± factor × 1.69 × MeanAD`. Spikes inflate MeanAD, so these fences +lean wide. The fallback applies only if it flags at most 5% of the column's +non-missing values. Above that, the off-center values are a second mode (a +zero-inflated count, a two-level measurement), not rare outliers, and the +column is left untouched, like a constant column. When the fallback is used, +the outlier action's description says so: "IQR is zero, so fences use 1.69 x +mean absolute deviation from the median". FreshCore applies the same rule. The +Polars, DuckDB and Spark backends still skip zero-IQR columns. + The default `outlier_action="auto"` is context-aware: it **flags** (adds a boolean `_outlier` column) under **balanced** mode and **caps** (winsorizes to the fences) under **aggressive** mode, and flags heavy-tailed diff --git a/docs/freshcore.md b/docs/freshcore.md index 91361ebc..9a60077e 100644 --- a/docs/freshcore.md +++ b/docs/freshcore.md @@ -46,7 +46,8 @@ FreshCore v1 runs native kernels for conservative, deterministic cleaning: - full-row duplicate detection with `duplicate_keep="first"` or `"last"` - boolean and numeric string casts where safe - mean/median/mode imputation -- IQR/z-score outlier clipping or flagging +- IQR/z-score outlier clipping or flagging, including the pandas zero-IQR + fallback to mean-absolute-deviation fences - simple per-column profile metadata FreshCore falls back for semantic cleaning, context/policy protection, cleaning diff --git a/src/freshdata/engine/outliers.py b/src/freshdata/engine/outliers.py index 6678080d..044dd66d 100644 --- a/src/freshdata/engine/outliers.py +++ b/src/freshdata/engine/outliers.py @@ -43,7 +43,7 @@ from ..report import CleanReport from ..steps.outliers import ( capping_bounds, - detection_bounds, + detection_fences, factor_for, integer_safe_bounds, resolve_method, @@ -137,10 +137,12 @@ def _detect(s: pd.Series, config: CleanConfig): fallback_note = "" method = resolve_method(s, config) factor = factor_for(config, method) - bounds = detection_bounds(s, method, factor) - if bounds is None: + fences = detection_fences(s, method, factor) + if fences is None: return None - lo, hi = integer_safe_bounds(s, *bounds) + lo, hi = integer_safe_bounds(s, fences[0], fences[1]) + if fences[2]: + fallback_note += f"; {fences[2]}" mask = (s < lo) | (s > hi) return mask, lo, hi, f"method={method}, factor={factor:g}{fallback_note}" diff --git a/src/freshdata/steps/outliers.py b/src/freshdata/steps/outliers.py index e7dce455..8f41c18b 100644 --- a/src/freshdata/steps/outliers.py +++ b/src/freshdata/steps/outliers.py @@ -7,6 +7,17 @@ Methods: Tukey fences (``iqr``, factor 1.5), mean ± k standard deviations (``zscore``, factor 3.0), or ``auto`` — z-score for approximately normal columns (|skewness| < 0.5), IQR otherwise. + +Zero IQR: when at least half of a non-constant column sits on one value +(Q1 == Q3, e.g. a mostly-zero column with a few spikes), Tukey fences collapse +to that value. Instead of silently switching detection off, the IQR is replaced +by its normal-consistent equivalent from the mean absolute deviation around the +median, ``1.6907 × MeanAD`` (sqrt(pi/2) × MeanAD estimates σ, the fallback +standard for a zero MAD, and 1.349σ is a normal IQR). The median equals Q1 and +Q3 here, and MeanAD is non-zero for every non-constant column. The fallback is +used only when it flags at most :data:`ZERO_IQR_MAX_SHARE` of the values. +Anything more is a structural second mode, not rare outliers, so detection +stays off, as it does for a constant column. """ from __future__ import annotations @@ -69,8 +80,11 @@ def resolve_method(s: pd.Series, config: CleanConfig) -> str: nonnull = s.dropna() # Measure shape on the trimmed bulk: a single extreme spike must not make # an otherwise-normal column look "skewed" to the very detector hunting it. - inner = detection_bounds(nonnull, "iqr", 3.0) - if inner is not None: + inner = detection_fences(nonnull, "iqr", 3.0) + # Zero-IQR fallback fences trim a column down to its dominant value, whose + # skewness of ~0 would misread as "normal". Measure such columns untrimmed, + # as before the fallback existed. + if inner is not None and not inner[2]: trimmed = nonnull[(nonnull >= inner[0]) & (nonnull <= inner[1])] if len(trimmed) >= 3: nonnull = trimmed @@ -87,10 +101,28 @@ def factor_for(config: CleanConfig, method: str) -> float: return _DEFAULT_FACTOR[method] -def detection_bounds( +#: IQR-equivalent spread per unit of mean absolute deviation for normal data: +#: sqrt(pi/2) (MeanAD -> σ) × 2·Φ⁻¹(0.75) (σ -> IQR). The FreshCore kernel uses +#: the same literal. +MEANAD_TO_IQR = 1.6906950787902986 +#: Largest share of non-missing values the zero-IQR fallback may flag. If more +#: values fall outside the fallback fences, they form a second mode (a +#: zero-inflated count, a two-level measurement) rather than rare outliers, so +#: detection stays off. +ZERO_IQR_MAX_SHARE = 0.05 +#: Report note appended to the detection label when the fallback set the fences. +ZERO_IQR_NOTE = "IQR is zero, so fences use 1.69 x mean absolute deviation from the median" + + +def detection_fences( s: pd.Series, method: str, factor: float -) -> tuple[float, float] | None: - """(lower, upper) fences for *s*, or None when undefined (constant data).""" +) -> tuple[float, float, str] | None: + """``(lower, upper, note)`` fences for *s*, or None when undefined. + + *note* is empty for ordinary fences and :data:`ZERO_IQR_NOTE` when the + zero-IQR fallback produced them. None means constant data, or a zero IQR + whose fallback would flag more than :data:`ZERO_IQR_MAX_SHARE` of values. + """ # Fence on the finite bulk: ±inf would otherwise poison the fences and # silently disable detection for exactly the columns that need it most, # while the infinities themselves must land outside the fences. @@ -98,13 +130,45 @@ def detection_bounds( if method == "iqr": q1, q3 = s.quantile(0.25), s.quantile(0.75) spread = q3 - q1 - if pd.isna(spread) or spread == 0: + if pd.isna(spread): return None - return float(q1 - factor * spread), float(q3 + factor * spread) + if spread == 0: + return _zero_iqr_fences(s, float(q1), factor) + return float(q1 - factor * spread), float(q3 + factor * spread), "" mean, std = s.mean(), s.std() if pd.isna(std) or std == 0: return None - return float(mean - factor * std), float(mean + factor * std) + return float(mean - factor * std), float(mean + factor * std), "" + + +def _zero_iqr_fences( + s: pd.Series, center: float, factor: float +) -> tuple[float, float, str] | None: + """Fences for a finite series whose quartiles both equal *center*.""" + values = s.to_numpy(dtype="float64", na_value=np.nan) + values = values[~np.isnan(values)] + if len(values) == 0: + return None + mean_ad = float(np.abs(values - center).mean()) + if mean_ad == 0: + return None # constant column + spread = MEANAD_TO_IQR * mean_ad + lo, hi = center - factor * spread, center + factor * spread + flagged = int(np.count_nonzero((values < lo) | (values > hi))) + if flagged > ZERO_IQR_MAX_SHARE * len(values): + return None + return lo, hi, ZERO_IQR_NOTE + + +def detection_bounds( + s: pd.Series, method: str, factor: float +) -> tuple[float, float] | None: + """(lower, upper) fences for *s*, or None when undefined (constant data). + + See :func:`detection_fences` for the zero-IQR fallback. + """ + fences = detection_fences(s, method, factor) + return None if fences is None else (fences[0], fences[1]) def _bounds(s: pd.Series, config: CleanConfig) -> tuple[float, float] | None: @@ -181,16 +245,29 @@ def handle_outliers(df: pd.DataFrame, config: CleanConfig, """ if config.outliers is None or df.empty: return df + from ..guard import ( # noqa: PLC0415 — cycle-safe lazy import + _match_columns, + hard_protected_columns, + ) + + protected = hard_protected_columns(config, df.columns) + # Declared roles only (not name-inferred ones): clipping may never rewrite + # a declared identifier or target. Flagging leaves values untouched. + targets = (_match_columns([str(config.target_column)], df.columns) + if config.target_column is not None else ()) + identifiers = _match_columns([str(c) for c in config.id_columns], df.columns) numeric_cols = [c for c in df.columns if is_numeric_dtype(df[c]) and not is_bool_dtype(df[c])] for col in numeric_cols: + if config.outliers == "clip" and str(col) in protected: + continue # context-protected columns must stay byte-identical s = df[col] method = resolve_method(s, config) factor = factor_for(config, method) - bounds = detection_bounds(s, method, factor) - if bounds is None: + fences = detection_fences(s, method, factor) + if fences is None: continue - lo, hi = bounds + lo, hi, note = fences if config.outliers == "clip": # Rewriting values uses skew-aware fences so a legitimate heavy # tail (lognormal amounts) is not flattened to the raw fences. @@ -200,7 +277,11 @@ def handle_outliers(df: pd.DataFrame, config: CleanConfig, n = int(mask.sum()) if n == 0: continue - label = f"{method}, factor {factor:g}" + if config.outliers == "clip" and (str(col) in targets or str(col) in identifiers): + role = "target" if str(col) in targets else "identifier" + report.add("outliers", f"skipped: {role} column", column=str(col), count=0) + continue + label = f"{method}, factor {factor:g}" + (f"; {note}" if note else "") if config.outliers == "clip": df[col] = s.clip(lo, hi) report.add("outliers", f"clipped {n} outlier(s) to [{lo:g}, {hi:g}] ({label})", diff --git a/tests/fixtures/golden_diff_summary.jsonl b/tests/fixtures/golden_diff_summary.jsonl index 02856e6e..19bec2df 100644 --- a/tests/fixtures/golden_diff_summary.jsonl +++ b/tests/fixtures/golden_diff_summary.jsonl @@ -40,3 +40,4 @@ {"changed": false, "created": false, "fixture": "titanic", "new_action_count": 8, "online": true, "previous_action_count": 8, "strategy": "balanced"} {"changed": false, "created": false, "fixture": "weather_json", "new_action_count": 3, "online": true, "previous_action_count": 3, "strategy": "balanced"} {"changed": true, "created": false, "fixture": "wine_quality", "new_action_count": 13, "online": true, "previous_action_count": 13, "strategy": "balanced"} +{"changed": true, "created": false, "fixture": "adult_income", "new_action_count": 11, "online": true, "previous_action_count": 10, "strategy": "balanced"} diff --git a/tests/fixtures/online/golden/adult_income.balanced.report.json b/tests/fixtures/online/golden/adult_income.balanced.report.json index 41fa831d..64172cbc 100644 --- a/tests/fixtures/online/golden/adult_income.balanced.report.json +++ b/tests/fixtures/online/golden/adult_income.balanced.report.json @@ -126,6 +126,20 @@ "status": "automatic", "step": "outliers" }, + { + "column": "capital_loss", + "confidence": 0.7, + "count": 100, + "description": "flagged 100 outlier(s), 5.0% of values (method=iqr, factor=1.5; IQR is zero, so fences use 1.69 x mean absolute deviation from the median) in new column 'capital_loss_outlier'", + "human_review": false, + "memory_influenced": false, + "model_id": "flag", + "rationale": "flagging records the detection without altering any value", + "reversible": null, + "risk": "low", + "status": "automatic", + "step": "outliers" + }, { "column": "hours_per_week", "confidence": 0.5, @@ -141,7 +155,7 @@ "step": "outliers" } ], - "cols_after": 19, + "cols_after": 20, "cols_before": 15, "columns_dropped": [], "columns_imputed": [ @@ -153,7 +167,7 @@ "duplicates_removed": 0, "missing_after": 0, "missing_before": 0, - "outliers_handled": 594, + "outliers_handled": 694, "recommendations": [], "rows_after": 2000, "rows_before": 2000, diff --git a/tests/test_execution/test_freshcore_native_parity.py b/tests/test_execution/test_freshcore_native_parity.py index ce7c13b0..4c4f96ec 100644 --- a/tests/test_execution/test_freshcore_native_parity.py +++ b/tests/test_execution/test_freshcore_native_parity.py @@ -373,3 +373,58 @@ def test_frames_without_mishandled_casts_stay_native(native, df, options): assert report.fallback_events == [] # Values match; dtypes may not (e.g. native boolean vs pandas bool). pd.testing.assert_frame_equal(pd.DataFrame(out), pd.DataFrame(expected), check_dtype=False) + + +# -- zero-IQR outlier fallback ------------------------------------------------ + + +def _outlier_frame() -> pd.DataFrame: + """A zero-IQR column with spikes, a zero-IQR second mode, a constant and + an ordinary column.""" + n = 100 + return pd.DataFrame({ + "spiky": [0.0] * 95 + [1000.0, 5000.0, 2.0, 3.0, -800.0], + "second_mode": [0.0] * 70 + [float(s * i) for i in range(1, 16) for s in (-1, 1)], + "constant": [4.0] * n, + "normal": [(((i * 37) % 23) - 11) / 7.0 for i in range(n - 1)] + [90.0], + }) + + +def _outlier_counts(report) -> dict: + return {a.column: a.count for a in report.actions if a.step == "outliers"} + + +@pytest.mark.parametrize("method", ["iqr", "zscore"]) +def test_outlier_flags_match_pandas_with_zero_iqr(native, method): + df = _outlier_frame() + options = {"outliers": "flag", "outlier_method": method, "fix_dtypes": False} + expected_frame = pd.DataFrame(fd.clean(df.copy(), engine="pandas", **KW, **options)) + expected, report = _clean_both(native, df, **options) + native_frame = pd.DataFrame(fd.clean(df.copy(), engine="freshcore", **KW, **options)) + + assert _outlier_counts(report) == _outlier_counts(expected) + assert list(native_frame.columns) == list(expected_frame.columns) + for flag in [c for c in expected_frame.columns if str(c).endswith("_outlier")]: + assert native_frame[flag].astype(bool).tolist() == expected_frame[flag].tolist() + if method == "iqr": + assert _outlier_counts(expected) == {"spiky": 3, "normal": 1} + + +def test_outlier_clips_match_pandas_with_zero_iqr(): + # The adapter sends clip to pandas (skew-aware capping), so drive the + # kernel directly. These columns hold non-positive values, so pandas never + # widens its fences in log space and both engines clip to detection fences. + df = _outlier_frame() + options = {"outliers": "clip", "outlier_method": "iqr", "fix_dtypes": False} + expected, report = fd.clean(df.copy(), engine="pandas", return_report=True, **KW, **options) + result = _native_result(df, **options) + + assert result["outliers_handled"] == report.outliers_handled == 4 + columns = {c["name"]: c["values"] for c in result["columns"]} + for name in df.columns: + # MeanAD is summed in a different order, so allow round-off. + assert columns[name] == pytest.approx(expected[name].tolist(), rel=1e-12, abs=0.0) + assert columns["spiky"][95] < 1000.0 and columns["spiky"][96] == columns["spiky"][95] + assert columns["spiky"][97:99] == [2.0, 3.0] + assert columns["second_mode"] == df["second_mode"].tolist() + assert columns["constant"] == df["constant"].tolist() diff --git a/tests/test_outliers.py b/tests/test_outliers.py index 15d1a422..e90e0c0f 100644 --- a/tests/test_outliers.py +++ b/tests/test_outliers.py @@ -1,8 +1,18 @@ +import math import warnings +import numpy as np import pandas as pd +import pytest import freshdata as fd +from freshdata.steps.outliers import ( + MEANAD_TO_IQR, + ZERO_IQR_MAX_SHARE, + ZERO_IQR_NOTE, + detection_bounds, + detection_fences, +) BASE = [10.0, 11.0, 12.0, 11.0, 10.0, 12.0, 11.0, 10.0, 12.0, 11.0] @@ -118,3 +128,206 @@ def test_all_infinite_column_is_skipped_not_crashed(): out, report = fd.clean(df, outliers="clip", return_report=True, **ISOLATE) # no finite bulk -> no fences -> column left alone assert not [a for a in report if a.step == "outliers"] + + +# -- zero IQR on a non-constant column --------------------------------------- +# Q1 == Q3 used to return no fences, silently disabling IQR detection on +# mostly-zero columns with real spikes (while zscore still flagged one). + +SPIKES = [0.0] * 95 + [1000.0, 5000.0, -800.0] +QUIET = {**ISOLATE, "verbose": False} + + +def _zero_iqr_fence(values, factor=1.5): + mean_ad = float(np.abs(np.asarray(values)).mean()) # median is 0 + return factor * (MEANAD_TO_IQR * mean_ad) + + +def _outlier_actions(report): + return [a for a in report if a.step == "outliers"] + + +@pytest.mark.parametrize("method", ["iqr", "auto"]) +def test_zero_iqr_spikes_are_flagged(method): + df = pd.DataFrame({"x": SPIKES}) + out, report = fd.clean(df, outliers="flag", outlier_method=method, + return_report=True, **QUIET) + assert out["x"].tolist() == SPIKES # data untouched + assert out.loc[out["x_outlier"], "x"].tolist() == [1000.0, 5000.0, -800.0] + [action] = _outlier_actions(report) + assert action.count == 3 and report.outliers_handled == 3 + assert ZERO_IQR_NOTE in action.description + + +@pytest.mark.parametrize("method", ["iqr", "auto"]) +def test_zero_iqr_spikes_are_clipped_to_fallback_fences(method): + df = pd.DataFrame({"x": SPIKES}) + out, report = fd.clean(df, outliers="clip", outlier_method=method, + return_report=True, **QUIET) + fence = _zero_iqr_fence(SPIKES) + assert out["x"].tolist() == [0.0] * 95 + [fence, fence, -fence] + [action] = _outlier_actions(report) + assert action.count == 3 and ZERO_IQR_NOTE in action.description + + +def test_zero_iqr_small_deviations_are_not_outliers(): + # The spikes widen the fences, so 2 and 3 stay inliers. + values = [0.0] * 95 + [1000.0, 5000.0, 2.0, 3.0, -800.0] + out = fd.clean(pd.DataFrame({"x": values}), outliers="flag", **QUIET) + assert out.loc[out["x_outlier"], "x"].tolist() == [1000.0, 5000.0, -800.0] + + +def test_zero_iqr_engine_actions_record_the_fallback(): + df = pd.DataFrame({"x": SPIKES}) + flagged, report = fd.clean(df, return_report=True, **QUIET) # balanced default + assert int(flagged["x_outlier"].sum()) == 3 + assert ZERO_IQR_NOTE in _outlier_actions(report)[0].description + + capped = fd.clean(df, outlier_action="cap", **QUIET) + fence = _zero_iqr_fence(SPIKES) + assert capped["x"].max() == fence and capped["x"].min() == -fence + + removed = fd.clean(df, outlier_action="remove", **QUIET) + assert removed["x"].tolist() == [0.0] * 95 + + +def test_zero_iqr_integer_column_stays_integer_after_clip(): + df = pd.DataFrame({"x": [int(v) for v in SPIKES]}) + out = fd.clean(df, outliers="clip", **QUIET) + assert out["x"].dtype == "int64" + assert out["x"].max() == math.ceil(_zero_iqr_fence(SPIKES)) + + +def test_zero_iqr_nullable_column_with_missing_values(): + # The second column keeps the missing-x row from being an empty row. + df = pd.DataFrame({"x": pd.array([int(v) for v in SPIKES] + [None], dtype="Int64"), + "k": [f"r{i}" for i in range(len(SPIKES) + 1)]}) + out = fd.clean(df, outliers="flag", strategy="conservative", **QUIET) # no engine imputation + assert int(out["x_outlier"].sum()) == 3 + assert out["x"].isna().iloc[-1] and not bool(out["x_outlier"].iloc[-1]) + + +@pytest.mark.parametrize( + "values", + [ + # 30% non-zero, split around zero so the IQR is still zero. + [0.0] * 70 + [float(s * i) for i in range(1, 16) for s in (-1, 1)], + # 8% small non-zero values: a second mode, not rare spikes. + [0.0] * 92 + [float(i) for i in range(1, 9)], + ], + ids=["two_sided_30pct", "one_sided_8pct"], +) +def test_zero_iqr_second_mode_is_not_flagged_wholesale(values): + s = pd.Series(values) + assert s.quantile(0.25) == s.quantile(0.75) # the case under test + df = pd.DataFrame({"x": values}) + for options in ({"outliers": "flag"}, {"outliers": "clip"}, {}, + {"outlier_action": "cap"}): + out, report = fd.clean(df, return_report=True, **options, **QUIET) + assert not _outlier_actions(report), options + assert out["x"].tolist() == values + assert "x_outlier" not in out.columns + + +def test_zero_iqr_fallback_share_limit_is_inclusive(): + at_limit = [0.0] * 95 + [1000.0] * 5 # 5% flagged: allowed + over = [0.0] * 94 + [1000.0] * 6 # 6% flagged: a second mode + assert detection_fences(pd.Series(at_limit), "iqr", 1.5) is not None + assert detection_fences(pd.Series(over), "iqr", 1.5) is None + assert ZERO_IQR_MAX_SHARE == 0.05 + + +def test_constant_column_still_untouched_by_every_action(): + df = pd.DataFrame({"c": [0.0] * 98}) + for options in ({"outliers": "flag"}, {"outliers": "clip"}, {}, + {"outlier_action": "cap"}, {"outlier_action": "remove"}): + out, report = fd.clean(df, return_report=True, **options, **QUIET) + assert out["c"].tolist() == [0.0] * 98 + assert list(out.columns) == ["c"] + assert not _outlier_actions(report), options + assert detection_fences(pd.Series([7.0] * 10 + [np.nan]), "iqr", 1.5) is None + + +def test_non_zero_iqr_fences_are_unchanged(): + rng = np.random.default_rng(3) + s = pd.Series(np.r_[rng.lognormal(0, 1, 500), 400.0]) + q1, q3 = s.quantile(0.25), s.quantile(0.75) + expected = (float(q1 - 1.5 * (q3 - q1)), float(q3 + 1.5 * (q3 - q1))) + assert detection_fences(s, "iqr", 1.5) == (*expected, "") + assert detection_bounds(s, "iqr", 1.5) == expected + _, report = fd.clean(pd.DataFrame({"v": s}), outliers="flag", + return_report=True, **QUIET) + [action] = _outlier_actions(report) + assert action.description.endswith("(iqr, factor 1.5)") + + +def _customer_frame(): + """Declared id and target columns that both hold an outlier.""" + rng = np.random.default_rng(0) + n = 80 + customer_id = np.arange(1000, 1000 + n).astype(float) + customer_id[-1] = 10_000_000 # a legitimate large id + spend = rng.normal(100, 10, n) + spend[0] = 5_000.0 + x = rng.normal(size=n) + x[1] = 50.0 + return pd.DataFrame({"customer_id": customer_id, "spend_target": spend, "x": x}) + + +def test_clip_never_modifies_declared_id_or_target_columns(): + df = _customer_frame() + out, report = fd.clean(df.copy(), id_columns=("customer_id",), + target_column="spend_target", outliers="clip", + return_report=True, **QUIET) + pd.testing.assert_series_equal(out["customer_id"], df["customer_id"]) + pd.testing.assert_series_equal(out["spend_target"], df["spend_target"]) + assert out["spend_target"].max() == 5_000.0 + assert out["x"].max() < 50.0 # other numeric columns are still clipped + actions = {a.column: a for a in _outlier_actions(report)} + assert actions["customer_id"].description == "skipped: identifier column" + assert actions["spend_target"].description == "skipped: target column" + assert actions["customer_id"].count == actions["spend_target"].count == 0 + assert actions["x"].count == 1 and "clipped" in actions["x"].description + assert report.outliers_handled == 1 + + +def test_clip_resolves_declared_names_after_column_renaming(): + df = _customer_frame().rename(columns={"customer_id": "Customer ID", + "spend_target": "Spend Target"}) + out = fd.clean(df.copy(), id_columns=("Customer ID",), target_column="Spend Target", + outliers="clip", **QUIET) + assert out["customer_id"].tolist() == df["Customer ID"].tolist() + assert out["spend_target"].tolist() == df["Spend Target"].tolist() + + +def test_clip_skip_is_reported_only_when_values_would_change(): + df = _customer_frame() + df["customer_id"] = np.arange(1000, 1000 + len(df)).astype(float) # no outlier + _, report = fd.clean(df, id_columns=("customer_id",), outliers="clip", + return_report=True, **QUIET) + assert "customer_id" not in {a.column for a in _outlier_actions(report)} + + +def test_flag_still_reports_declared_id_and_target_without_changing_values(): + df = _customer_frame() + out, report = fd.clean(df.copy(), id_columns=("customer_id",), + target_column="spend_target", outliers="flag", + return_report=True, **QUIET) + pd.testing.assert_series_equal(out["customer_id"], df["customer_id"]) + pd.testing.assert_series_equal(out["spend_target"], df["spend_target"]) + assert bool(out["spend_target_outlier"].iloc[0]) + assert bool(out["customer_id_outlier"].iloc[-1]) + + +def test_clip_skips_context_protected_columns(): + df = _customer_frame() + context = {"columns": {"x": {"mutable": False}}} + out = fd.clean(df.copy(), outliers="clip", semantic_context=context, **QUIET) + pd.testing.assert_series_equal(out["x"], df["x"]) + assert out["spend_target"].max() < 5_000.0 # undeclared columns are clipped + + +def test_zero_iqr_profile_reports_the_spikes(): + prof = fd.profile(pd.DataFrame({"x": SPIKES})) + [col] = [c for c in prof.columns if c.name == "x"] + assert "3 potential outlier(s) (iqr)" in col.issues