Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
78 changes: 76 additions & 2 deletions crates/freshcore/src/kernels/outliers.rs
Original file line number Diff line number Diff line change
Expand Up @@ -91,14 +91,47 @@ pub fn bounds(values: &[Option<f64>], method: &str, factor: f64) -> Option<(f64,
return Some((mean - factor * std, mean + factor * std));
}
let q1 = quantile(vals.clone(), 0.25)?;
let q3 = quantile(vals, 0.75)?;
let q3 = quantile(vals.clone(), 0.75)?;
let spread = q3 - q1;
if spread == 0.0 || spread.is_nan() {
if spread.is_nan() {
return None;
}
if spread == 0.0 {
return zero_iqr_bounds(&vals, q1, factor);
}
Some((q1 - factor * spread, q3 + factor * spread))
}

/// IQR-equivalent spread per unit of mean absolute deviation for normal data:
/// sqrt(pi/2) (MeanAD -> sigma) x 2*Phi^-1(0.75) (sigma -> IQR). Mirrors
/// `freshdata.steps.outliers.MEANAD_TO_IQR`.
const MEANAD_TO_IQR: f64 = 1.6906950787902986;
/// Largest share of values the zero-IQR fallback may flag; mirrors
/// `freshdata.steps.outliers.ZERO_IQR_MAX_SHARE`.
const ZERO_IQR_MAX_SHARE: f64 = 0.05;

/// Fences when both quartiles equal `center` (at least half the column sits on
/// one value). A zero IQR must not silently disable detection. Instead, the IQR
/// is replaced by the normal-consistent spread from the mean absolute deviation
/// around the median (the median equals the quartiles here). The result is
/// `None` for a constant column, or when the fences would flag more than
/// `ZERO_IQR_MAX_SHARE` of the values: that many is a second mode, not rare
/// outliers.
fn zero_iqr_bounds(vals: &[f64], center: f64, factor: f64) -> Option<(f64, f64)> {
let n = vals.len() as f64;
let mean_ad = vals.iter().map(|v| (v - center).abs()).sum::<f64>() / n;
if mean_ad == 0.0 || mean_ad.is_nan() {
return None;
}
let spread = MEANAD_TO_IQR * mean_ad;
let (lo, hi) = (center - factor * spread, center + factor * spread);
let flagged = vals.iter().filter(|v| **v < lo || **v > hi).count() as f64;
if flagged > ZERO_IQR_MAX_SHARE * n {
return None;
}
Some((lo, hi))
}

fn quantile(mut vals: Vec<f64>, q: f64) -> Option<f64> {
if vals.is_empty() {
return None;
Expand Down Expand Up @@ -136,4 +169,45 @@ mod tests {
assert!(lo < 1.0);
assert!(hi < 100.0);
}

fn zeros_with(extra: &[f64], zeros: usize) -> Vec<Option<f64>> {
std::iter::repeat(0.0)
.take(zeros)
.chain(extra.iter().copied())
.map(Some)
.collect()
}

#[test]
fn zero_iqr_falls_back_to_mean_absolute_deviation() {
let values = zeros_with(&[1000.0, 5000.0, 2.0, 3.0, -800.0], 95);
let (lo, hi) = bounds(&values, "iqr", 1.5).unwrap();
// MeanAD = 6805 / 100; fences = 0 +/- 1.5 x 1.6907 x MeanAD.
let expected = 1.5 * (MEANAD_TO_IQR * (6805.0 / 100.0));
assert_eq!((lo, hi), (-expected, expected));
let mut frame = Frame {
nrows: values.len(),
columns: vec![Column {
name: "x".into(),
data: ColumnData::Float(values),
}],
};
let actions = handle_outliers(&mut frame, Some("clip"), "iqr", 1.5);
assert_eq!(actions.len(), 1);
assert_eq!(actions[0].count, 3); // the spikes, not 2.0 / 3.0
}

#[test]
fn zero_iqr_fallback_does_not_flag_a_second_mode() {
// 30% non-zero, split around zero: the IQR is still zero.
let extra: Vec<f64> = (1..=15).flat_map(|i| [-(i as f64), i as f64]).collect();
assert_eq!(bounds(&zeros_with(&extra, 70), "iqr", 1.5), None);
}

#[test]
fn constant_column_has_no_bounds() {
let values = vec![Some(5.0); 20];
assert_eq!(bounds(&values, "iqr", 1.5), None);
assert_eq!(bounds(&values, "zscore", 3.0), None);
}
}
6 changes: 5 additions & 1 deletion docs/backends.md
Original file line number Diff line number Diff line change
Expand Up @@ -174,7 +174,11 @@ Notes:
- **Outlier counts** match the pandas reference where the quantile statistics
match (Polars/DuckDB linear interpolation); Spark may flag a different count.
pandas, Polars, DuckDB and Spark compute the fences from finite values only;
`±inf` is still tested against them.
`±inf` is still tested against them. For a non-constant column whose IQR is
zero, pandas and FreshCore fall back to fences from the mean absolute
deviation (see [the cleaning engine](cleaning-engine.md#outliers)). Polars,
DuckDB and Spark still skip such columns, so they can flag fewer outliers
there.
- **Float `NaN`** is read as missing by Polars, DuckDB and Spark, as in pandas
(Spark keeps `NaN` as a value in float/double columns, so it is converted to
null on ingestion).
Expand Down
15 changes: 15 additions & 0 deletions docs/cleaning-engine.md
Original file line number Diff line number Diff line change
Expand Up @@ -52,6 +52,21 @@ Detection methods: IQR fences (default), z-score, `"auto"` (z-score for ~normal
columns, IQR for skewed), or `"isolation_forest"` (scikit-learn, ≥ 100 rows,
falls back to IQR).

**Zero IQR.** When at least half of a non-constant column sits on one value
(Q1 = Q3, for example a mostly-zero column with a few spikes), Tukey fences
would collapse to that value. IQR detection does not switch off in that case.
The IQR is replaced by its normal-consistent equivalent from the mean absolute
deviation around the median: `1.69 × MeanAD`, i.e. `sqrt(pi/2) × MeanAD` as the
σ estimate, the usual fallback when the MAD is zero, times 1.349. The fences
are `median ± factor × 1.69 × MeanAD`. Spikes inflate MeanAD, so these fences
lean wide. The fallback applies only if it flags at most 5% of the column's
non-missing values. Above that, the off-center values are a second mode (a
zero-inflated count, a two-level measurement), not rare outliers, and the
column is left untouched, like a constant column. When the fallback is used,
the outlier action's description says so: "IQR is zero, so fences use 1.69 x
mean absolute deviation from the median". FreshCore applies the same rule. The
Polars, DuckDB and Spark backends still skip zero-IQR columns.

The default `outlier_action="auto"` is context-aware: it **flags** (adds a
boolean `<col>_outlier` column) under **balanced** mode and **caps**
(winsorizes to the fences) under **aggressive** mode, and flags heavy-tailed
Expand Down
3 changes: 2 additions & 1 deletion docs/freshcore.md
Original file line number Diff line number Diff line change
Expand Up @@ -46,7 +46,8 @@ FreshCore v1 runs native kernels for conservative, deterministic cleaning:
- full-row duplicate detection with `duplicate_keep="first"` or `"last"`
- boolean and numeric string casts where safe
- mean/median/mode imputation
- IQR/z-score outlier clipping or flagging
- IQR/z-score outlier clipping or flagging, including the pandas zero-IQR
fallback to mean-absolute-deviation fences
- simple per-column profile metadata

FreshCore falls back for semantic cleaning, context/policy protection, cleaning
Expand Down
10 changes: 6 additions & 4 deletions src/freshdata/engine/outliers.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,7 @@
from ..report import CleanReport
from ..steps.outliers import (
capping_bounds,
detection_bounds,
detection_fences,
factor_for,
integer_safe_bounds,
resolve_method,
Expand Down Expand Up @@ -137,10 +137,12 @@ def _detect(s: pd.Series, config: CleanConfig):
fallback_note = ""
method = resolve_method(s, config)
factor = factor_for(config, method)
bounds = detection_bounds(s, method, factor)
if bounds is None:
fences = detection_fences(s, method, factor)
if fences is None:
return None
lo, hi = integer_safe_bounds(s, *bounds)
lo, hi = integer_safe_bounds(s, fences[0], fences[1])
if fences[2]:
fallback_note += f"; {fences[2]}"
mask = (s < lo) | (s > hi)
return mask, lo, hi, f"method={method}, factor={factor:g}{fallback_note}"

Expand Down
105 changes: 93 additions & 12 deletions src/freshdata/steps/outliers.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,17 @@
Methods: Tukey fences (``iqr``, factor 1.5), mean ± k standard deviations
(``zscore``, factor 3.0), or ``auto`` — z-score for approximately normal
columns (|skewness| < 0.5), IQR otherwise.

Zero IQR: when at least half of a non-constant column sits on one value
(Q1 == Q3, e.g. a mostly-zero column with a few spikes), Tukey fences collapse
to that value. Instead of silently switching detection off, the IQR is replaced
by its normal-consistent equivalent from the mean absolute deviation around the
median, ``1.6907 × MeanAD`` (sqrt(pi/2) × MeanAD estimates σ, the fallback
standard for a zero MAD, and 1.349σ is a normal IQR). The median equals Q1 and
Q3 here, and MeanAD is non-zero for every non-constant column. The fallback is
used only when it flags at most :data:`ZERO_IQR_MAX_SHARE` of the values.
Anything more is a structural second mode, not rare outliers, so detection
stays off, as it does for a constant column.
"""

from __future__ import annotations
Expand Down Expand Up @@ -69,8 +80,11 @@ def resolve_method(s: pd.Series, config: CleanConfig) -> str:
nonnull = s.dropna()
# Measure shape on the trimmed bulk: a single extreme spike must not make
# an otherwise-normal column look "skewed" to the very detector hunting it.
inner = detection_bounds(nonnull, "iqr", 3.0)
if inner is not None:
inner = detection_fences(nonnull, "iqr", 3.0)
# Zero-IQR fallback fences trim a column down to its dominant value, whose
# skewness of ~0 would misread as "normal". Measure such columns untrimmed,
# as before the fallback existed.
if inner is not None and not inner[2]:
trimmed = nonnull[(nonnull >= inner[0]) & (nonnull <= inner[1])]
if len(trimmed) >= 3:
nonnull = trimmed
Expand All @@ -87,24 +101,74 @@ def factor_for(config: CleanConfig, method: str) -> float:
return _DEFAULT_FACTOR[method]


def detection_bounds(
#: IQR-equivalent spread per unit of mean absolute deviation for normal data:
#: sqrt(pi/2) (MeanAD -> σ) × 2·Φ⁻¹(0.75) (σ -> IQR). The FreshCore kernel uses
#: the same literal.
MEANAD_TO_IQR = 1.6906950787902986
#: Largest share of non-missing values the zero-IQR fallback may flag. If more
#: values fall outside the fallback fences, they form a second mode (a
#: zero-inflated count, a two-level measurement) rather than rare outliers, so
#: detection stays off.
ZERO_IQR_MAX_SHARE = 0.05
#: Report note appended to the detection label when the fallback set the fences.
ZERO_IQR_NOTE = "IQR is zero, so fences use 1.69 x mean absolute deviation from the median"


def detection_fences(
s: pd.Series, method: str, factor: float
) -> tuple[float, float] | None:
"""(lower, upper) fences for *s*, or None when undefined (constant data)."""
) -> tuple[float, float, str] | None:
"""``(lower, upper, note)`` fences for *s*, or None when undefined.

*note* is empty for ordinary fences and :data:`ZERO_IQR_NOTE` when the
zero-IQR fallback produced them. None means constant data, or a zero IQR
whose fallback would flag more than :data:`ZERO_IQR_MAX_SHARE` of values.
"""
# Fence on the finite bulk: ±inf would otherwise poison the fences and
# silently disable detection for exactly the columns that need it most,
# while the infinities themselves must land outside the fences.
s = drop_infinite(s)
if method == "iqr":
q1, q3 = s.quantile(0.25), s.quantile(0.75)
spread = q3 - q1
if pd.isna(spread) or spread == 0:
if pd.isna(spread):
return None
return float(q1 - factor * spread), float(q3 + factor * spread)
if spread == 0:
return _zero_iqr_fences(s, float(q1), factor)
return float(q1 - factor * spread), float(q3 + factor * spread), ""
mean, std = s.mean(), s.std()
if pd.isna(std) or std == 0:
return None
return float(mean - factor * std), float(mean + factor * std)
return float(mean - factor * std), float(mean + factor * std), ""


def _zero_iqr_fences(
s: pd.Series, center: float, factor: float
) -> tuple[float, float, str] | None:
"""Fences for a finite series whose quartiles both equal *center*."""
values = s.to_numpy(dtype="float64", na_value=np.nan)
values = values[~np.isnan(values)]
if len(values) == 0:
return None
mean_ad = float(np.abs(values - center).mean())
if mean_ad == 0:
return None # constant column
spread = MEANAD_TO_IQR * mean_ad
lo, hi = center - factor * spread, center + factor * spread
flagged = int(np.count_nonzero((values < lo) | (values > hi)))
if flagged > ZERO_IQR_MAX_SHARE * len(values):
return None
return lo, hi, ZERO_IQR_NOTE


def detection_bounds(
s: pd.Series, method: str, factor: float
) -> tuple[float, float] | None:
"""(lower, upper) fences for *s*, or None when undefined (constant data).

See :func:`detection_fences` for the zero-IQR fallback.
"""
fences = detection_fences(s, method, factor)
return None if fences is None else (fences[0], fences[1])


def _bounds(s: pd.Series, config: CleanConfig) -> tuple[float, float] | None:
Expand Down Expand Up @@ -181,16 +245,29 @@ def handle_outliers(df: pd.DataFrame, config: CleanConfig,
"""
if config.outliers is None or df.empty:
return df
from ..guard import ( # noqa: PLC0415 — cycle-safe lazy import
_match_columns,
hard_protected_columns,
)

protected = hard_protected_columns(config, df.columns)
# Declared roles only (not name-inferred ones): clipping may never rewrite
# a declared identifier or target. Flagging leaves values untouched.
targets = (_match_columns([str(config.target_column)], df.columns)
if config.target_column is not None else ())
identifiers = _match_columns([str(c) for c in config.id_columns], df.columns)
numeric_cols = [c for c in df.columns
if is_numeric_dtype(df[c]) and not is_bool_dtype(df[c])]
for col in numeric_cols:
if config.outliers == "clip" and str(col) in protected:
continue # context-protected columns must stay byte-identical
s = df[col]
method = resolve_method(s, config)
factor = factor_for(config, method)
bounds = detection_bounds(s, method, factor)
if bounds is None:
fences = detection_fences(s, method, factor)
if fences is None:
continue
lo, hi = bounds
lo, hi, note = fences
if config.outliers == "clip":
# Rewriting values uses skew-aware fences so a legitimate heavy
# tail (lognormal amounts) is not flattened to the raw fences.
Expand All @@ -200,7 +277,11 @@ def handle_outliers(df: pd.DataFrame, config: CleanConfig,
n = int(mask.sum())
if n == 0:
continue
label = f"{method}, factor {factor:g}"
if config.outliers == "clip" and (str(col) in targets or str(col) in identifiers):
role = "target" if str(col) in targets else "identifier"
report.add("outliers", f"skipped: {role} column", column=str(col), count=0)
continue
label = f"{method}, factor {factor:g}" + (f"; {note}" if note else "")
if config.outliers == "clip":
df[col] = s.clip(lo, hi)
report.add("outliers", f"clipped {n} outlier(s) to [{lo:g}, {hi:g}] ({label})",
Expand Down
1 change: 1 addition & 0 deletions tests/fixtures/golden_diff_summary.jsonl
Original file line number Diff line number Diff line change
Expand Up @@ -40,3 +40,4 @@
{"changed": false, "created": false, "fixture": "titanic", "new_action_count": 8, "online": true, "previous_action_count": 8, "strategy": "balanced"}
{"changed": false, "created": false, "fixture": "weather_json", "new_action_count": 3, "online": true, "previous_action_count": 3, "strategy": "balanced"}
{"changed": true, "created": false, "fixture": "wine_quality", "new_action_count": 13, "online": true, "previous_action_count": 13, "strategy": "balanced"}
{"changed": true, "created": false, "fixture": "adult_income", "new_action_count": 11, "online": true, "previous_action_count": 10, "strategy": "balanced"}
18 changes: 16 additions & 2 deletions tests/fixtures/online/golden/adult_income.balanced.report.json
Original file line number Diff line number Diff line change
Expand Up @@ -126,6 +126,20 @@
"status": "automatic",
"step": "outliers"
},
{
"column": "capital_loss",
"confidence": 0.7,
"count": 100,
"description": "flagged 100 outlier(s), 5.0% of values (method=iqr, factor=1.5; IQR is zero, so fences use 1.69 x mean absolute deviation from the median) in new column 'capital_loss_outlier'",
"human_review": false,
"memory_influenced": false,
"model_id": "flag",
"rationale": "flagging records the detection without altering any value",
"reversible": null,
"risk": "low",
"status": "automatic",
"step": "outliers"
},
{
"column": "hours_per_week",
"confidence": 0.5,
Expand All @@ -141,7 +155,7 @@
"step": "outliers"
}
],
"cols_after": 19,
"cols_after": 20,
"cols_before": 15,
"columns_dropped": [],
"columns_imputed": [
Expand All @@ -153,7 +167,7 @@
"duplicates_removed": 0,
"missing_after": 0,
"missing_before": 0,
"outliers_handled": 594,
"outliers_handled": 694,
"recommendations": [],
"rows_after": 2000,
"rows_before": 2000,
Expand Down
Loading
Loading