Skip to content

Commit f35a166

Browse files
fix(impute): keep large nullable integers exact; fill_missing dtype handling
Imputing a nullable integer column cast it to float64 whenever the fill value was fractional, and computed mean/median in float64. For values beyond 2**53 that silently changed *present* values (2**53 + 1 became 2**53) with only a "column cast to float64" note. - New _util helpers: exceeds_float64_exact() detects such columns, exact_int_stat() computes mean/median in exact integer arithmetic (rounded half-to-even), and fill_na_exact() fills without a float64 cast for them. Columns within +-2**53 keep the existing behaviour (float64 cast note). - steps/missing.py (impute=...) and engine/missing.py (_fill, KNN _assign_filled) use them; the report says the dtype was kept. - fd.fill_missing crashed on nullable Int columns with a fractional mean/median (including the default method="auto"), on nullable boolean columns (numeric median into boolean), and on duplicate column labels. It now picks the fill value like fd.clean(impute=...) (booleans take the mode), fills through fill_na_exact, and rejects duplicate labels with a clear ValueError. Closes #210 Refs #208
1 parent 55a8044 commit f35a166

6 files changed

Lines changed: 195 additions & 46 deletions

File tree

‎src/freshdata/_util.py‎

Lines changed: 60 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -3,7 +3,9 @@
33
from __future__ import annotations
44

55
import hashlib
6+
import math
67
import warnings
8+
from fractions import Fraction
79
from typing import Any
810

911
import pandas as pd
@@ -23,6 +25,64 @@ def safe_median(s: pd.Series) -> Any:
2325
return s.median()
2426

2527

28+
#: float64 represents every integer up to this magnitude exactly.
29+
FLOAT64_EXACT_INT = 2**53
30+
31+
32+
def exceeds_float64_exact(s: pd.Series) -> bool:
33+
"""True for a nullable integer column holding a value float64 cannot represent.
34+
35+
Casting such a column to float64 silently changes *present* values
36+
(``2**53 + 1`` becomes ``2**53``), so fill paths must keep its integer dtype.
37+
"""
38+
if not isinstance(s.array, pd.arrays.IntegerArray):
39+
return False
40+
present = s.dropna()
41+
if present.empty:
42+
return False
43+
return int(present.max()) > FLOAT64_EXACT_INT or int(present.min()) < -FLOAT64_EXACT_INT
44+
45+
46+
def exact_int_stat(s: pd.Series, strategy: str) -> int:
47+
"""Mean or median of an integer column in exact integer arithmetic.
48+
49+
Rounded half-to-even to the nearest integer so the result fits the column's
50+
dtype. Only used for columns where :func:`exceeds_float64_exact` holds.
51+
"""
52+
values = sorted(int(v) for v in s.dropna())
53+
n = len(values)
54+
if strategy == "mean":
55+
return round(Fraction(sum(values), n))
56+
mid = n // 2
57+
if n % 2:
58+
return values[mid]
59+
return round(Fraction(values[mid - 1] + values[mid], 2))
60+
61+
62+
def fill_na_exact(s: pd.Series, value: Any) -> tuple[pd.Series, str]:
63+
"""``s.fillna(value)`` that never silently corrupts large nullable integers.
64+
65+
Returns the filled series and a note for the report. A column holding values
66+
beyond 2**53 keeps its integer dtype and receives an integer fill value.
67+
Any other numeric column that cannot hold a fractional *value* is cast to
68+
float64, which is exact for it. Raises ``TypeError``/``ValueError`` when
69+
*value* cannot be stored.
70+
"""
71+
if exceeds_float64_exact(s):
72+
if isinstance(value, float):
73+
if not math.isfinite(value):
74+
raise ValueError(f"cannot fill {s.dtype} with {value!r}")
75+
value = int(round(value))
76+
return s.fillna(value), f", kept {s.dtype} so values beyond 2**53 stay exact"
77+
try:
78+
return s.fillna(value), ""
79+
except (TypeError, ValueError):
80+
if pd.api.types.is_numeric_dtype(s) and isinstance(value, float):
81+
# e.g. a fractional median into an integer column
82+
return s.astype("float64").fillna(value), ", column cast to float64"
83+
raise
84+
85+
2686
def add_column(df: pd.DataFrame, name: object, values: object) -> None:
2787
"""Insert a new column in place, suppressing pandas' fragmentation notice.
2888

‎src/freshdata/engine/missing.py‎

Lines changed: 19 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -35,7 +35,13 @@
3535
import pandas as pd
3636
from pandas.api.types import is_bool_dtype, is_numeric_dtype
3737

38-
from .._util import add_column, safe_median
38+
from .._util import (
39+
add_column,
40+
exact_int_stat,
41+
exceeds_float64_exact,
42+
fill_na_exact,
43+
safe_median,
44+
)
3945
from ..config import CleanConfig
4046
from ..report import CleanReport
4147
from ..steps.missing import _mode_value
@@ -486,6 +492,8 @@ def _fill(df: pd.DataFrame, col: object, ctx: ColumnContext, report: CleanReport
486492
min_confidence=min_confidence, model_id=model_id, label=label,
487493
)
488494
s = df[col]
495+
if label in ("mean", "median") and exceeds_float64_exact(s):
496+
value = exact_int_stat(s, label) # a float statistic would lose digits
489497
if value is None or pd.isna(value):
490498
_preserve(df, col, ctx, report,
491499
rationale="no usable fill value could be derived "
@@ -494,18 +502,13 @@ def _fill(df: pd.DataFrame, col: object, ctx: ColumnContext, report: CleanReport
494502
return df
495503
if isinstance(s.dtype, pd.CategoricalDtype) and value not in s.cat.categories:
496504
s = s.cat.add_categories([value])
497-
cast_note = ""
498505
try:
499-
filled = s.fillna(value)
506+
filled, cast_note = fill_na_exact(s, value)
500507
except (TypeError, ValueError):
501-
if is_numeric_dtype(s) and isinstance(value, float):
502-
filled = s.astype("float64").fillna(value)
503-
cast_note = ", column cast to float64"
504-
else:
505-
_preserve(df, col, ctx, report,
506-
rationale=f"fill value not representable in dtype {s.dtype}",
507-
risk="medium", confidence=0.6, model_id="preserve")
508-
return df
508+
_preserve(df, col, ctx, report,
509+
rationale=f"fill value not representable in dtype {s.dtype}",
510+
risk="medium", confidence=0.6, model_id="preserve")
511+
return df
509512
df[col] = filled
510513
shown = f"{value:.6g}" if isinstance(value, float) else repr(value)
511514
report.add(_STEP,
@@ -533,7 +536,11 @@ def _assign_filled(df: pd.DataFrame, col: object, ctx: ColumnContext,
533536
try:
534537
combined = s.where(s.notna(), filled_values)
535538
except (TypeError, ValueError):
536-
combined = s.astype("float64").where(s.notna(), filled_values)
539+
if exceeds_float64_exact(s):
540+
# keep the integer dtype: a float64 cast would change present values
541+
combined = s.where(s.notna(), filled_values.round().astype(s.dtype))
542+
else:
543+
combined = s.astype("float64").where(s.notna(), filled_values)
537544
df[col] = combined
538545
report.add(_STEP, f"filled {ctx.n_missing} missing value(s) with {label}",
539546
column=str(col), count=ctx.n_missing, rationale=rationale,

‎src/freshdata/simple.py‎

Lines changed: 22 additions & 21 deletions
Original file line numberDiff line numberDiff line change
@@ -21,7 +21,8 @@
2121
import pandas as pd
2222
from pandas.api.types import is_bool_dtype, is_numeric_dtype
2323

24-
from ._util import safe_median
24+
from ._util import fill_na_exact
25+
from .steps.missing import _fill_value
2526

2627
__all__ = [
2728
"detect_outliers",
@@ -78,12 +79,6 @@ def _is_outlier_numeric(s: pd.Series) -> bool:
7879
return is_numeric_dtype(s) and not is_bool_dtype(s)
7980

8081

81-
def _mode_value(s: pd.Series) -> object | None:
82-
"""Most-frequent non-null value, or None when the column is all-null."""
83-
modes = s.dropna().mode()
84-
return modes.iloc[0] if not modes.empty else None
85-
86-
8782
def _outlier_mask(
8883
df: pd.DataFrame,
8984
cols: Sequence[str],
@@ -125,14 +120,22 @@ def fill_missing(
125120
"""Fill missing values in ``columns`` (all columns by default).
126121
127122
``method`` is one of ``"auto"`` (median for numeric columns, mode for the
128-
rest), ``"mean"``, ``"median"``, ``"mode"``, ``"constant"`` (fills with
129-
``value``), ``"ffill"`` or ``"bfill"``. Returns the filled frame -- a copy
130-
unless ``inplace=True``.
123+
rest, booleans included), ``"mean"``, ``"median"``, ``"mode"``,
124+
``"constant"`` (fills with ``value``), ``"ffill"`` or ``"bfill"``. A
125+
fractional mean/median is filled into an integer column by casting it to
126+
float64, except for nullable integers holding values beyond 2**53, which keep
127+
their dtype and get a rounded, exactly-computed fill. Returns the filled
128+
frame -- a copy unless ``inplace=True``.
131129
"""
132130
if method not in _FILL_METHODS:
133131
raise ValueError(f"method must be one of {_FILL_METHODS}, got {method!r}")
134132
if method == "constant" and value is None:
135133
raise ValueError("method='constant' requires a value")
134+
if not df.columns.is_unique:
135+
duplicated = sorted({str(c) for c in df.columns[df.columns.duplicated()]})
136+
raise ValueError(
137+
f"fill_missing requires unique column labels; duplicated: {duplicated}"
138+
)
136139
if not inplace:
137140
df = df.copy()
138141
cols = _resolve_columns(df, columns)
@@ -149,18 +152,16 @@ def fill_missing(
149152
elif method == "constant":
150153
df[col] = s.fillna(value)
151154
else:
152-
strategy = method
153-
if strategy == "auto":
154-
strategy = "median" if is_numeric_dtype(s) else "mode"
155-
if strategy == "mode":
156-
fill = _mode_value(s)
157-
elif is_numeric_dtype(s):
158-
fill = s.mean() if strategy == "mean" else safe_median(s)
159-
else:
160-
fill = None # mean/median are undefined for non-numeric columns
161-
if fill is None:
155+
# Same value choice as fd.clean(impute=...): booleans take the mode,
156+
# mean/median are undefined (None) for non-numeric columns.
157+
fill = _fill_value(s, method)
158+
if fill is None or pd.isna(fill):
162159
continue
163-
df[col] = s.fillna(fill)
160+
try:
161+
df[col], _ = fill_na_exact(s, fill)
162+
except (TypeError, ValueError):
163+
continue # the value cannot be stored in this dtype
164+
164165
filled += before - int(df[col].isna().sum())
165166
if verbose:
166167
print(

‎src/freshdata/steps/missing.py‎

Lines changed: 8 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -12,7 +12,7 @@
1212
import pandas as pd
1313
from pandas.api.types import is_bool_dtype, is_numeric_dtype
1414

15-
from .._util import safe_median
15+
from .._util import exact_int_stat, exceeds_float64_exact, fill_na_exact, safe_median
1616
from ..config import CleanConfig
1717
from ..report import CleanReport
1818

@@ -36,6 +36,8 @@ def _fill_value(s: pd.Series, strategy: str) -> Any | None:
3636
if strategy in ("mean", "median"):
3737
if not numeric:
3838
return None # not defined for this dtype; caller reports the skip
39+
if exceeds_float64_exact(s):
40+
return exact_int_stat(s, strategy) # a float statistic would lose digits
3941
return s.mean() if strategy == "mean" else safe_median(s)
4042
return _mode_value(s)
4143

@@ -87,19 +89,13 @@ def impute_missing(df: pd.DataFrame, config: CleanConfig,
8789
f"skipped ({strategy} is not defined for dtype {s.dtype})",
8890
column=str(col))
8991
continue
90-
cast_note = ""
9192
try:
92-
filled = s.fillna(value)
93+
filled, cast_note = fill_na_exact(s, value)
9394
except (TypeError, ValueError):
94-
if is_numeric_dtype(s) and isinstance(value, float):
95-
# e.g. fractional median into an integer column
96-
filled = s.astype("float64").fillna(value)
97-
cast_note = ", column cast to float64"
98-
else:
99-
# e.g. value not representable in this dtype
100-
report.add("impute", f"skipped (could not fill dtype {s.dtype})",
101-
column=str(col))
102-
continue
95+
# e.g. value not representable in this dtype
96+
report.add("impute", f"skipped (could not fill dtype {s.dtype})",
97+
column=str(col))
98+
continue
10399
df[col] = filled
104100
shown = f"{value:.6g}" if isinstance(value, float) else repr(value)
105101
report.add("impute",

‎tests/test_nullable_int_median.py‎

Lines changed: 58 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -6,12 +6,14 @@
66
imputation path must survive those columns and keep Int64/float behaviour.
77
"""
88

9+
from fractions import Fraction
10+
911
import numpy as np
1012
import pandas as pd
1113
import pytest
1214

1315
import freshdata as fd
14-
from freshdata._util import safe_median
16+
from freshdata._util import exact_int_stat, exceeds_float64_exact, fill_na_exact, safe_median
1517
from freshdata.engine import missing, model_select
1618
from freshdata.engine.utils import _has_outliers
1719

@@ -92,6 +94,61 @@ def test_seasonal_imputation_global_median_on_nullable_int(dtype):
9294
assert out["v"].isna().sum() == 0
9395

9496

97+
BIG = 2**53 + 1 # float64 rounds this to 2**53
98+
99+
100+
def test_exact_int_stat_and_detection():
101+
s = pd.Series(pd.array([BIG, 0, 1, None], dtype="Int64"))
102+
assert exceeds_float64_exact(s)
103+
assert not exceeds_float64_exact(pd.Series(pd.array([2**53, None], dtype="Int64")))
104+
assert not exceeds_float64_exact(pd.Series([float(BIG), np.nan]))
105+
assert exact_int_stat(s, "mean") == round(Fraction(BIG + 1, 3))
106+
assert exact_int_stat(s, "median") == 1
107+
even = pd.Series(pd.array([BIG, BIG + 2, None, BIG + 5, BIG + 7], dtype="Int64"))
108+
# (BIG+2 + BIG+5) / 2 == 2**53 + 4.5, which rounds half-to-even to 2**53 + 4
109+
assert exact_int_stat(even, "median") == BIG + 3
110+
111+
112+
def test_fill_na_exact_keeps_small_int_behaviour():
113+
s = pd.Series(pd.array([1, None, 2], dtype="Int64"))
114+
filled, note = fill_na_exact(s, 1.5)
115+
assert filled.tolist() == [1.0, 1.5, 2.0]
116+
assert note == ", column cast to float64"
117+
filled, note = fill_na_exact(s, 1)
118+
assert str(filled.dtype) == "Int64" and note == ""
119+
120+
121+
@pytest.mark.parametrize("impute", ["mean", "median", "auto"])
122+
def test_explicit_impute_keeps_int64_beyond_2_53_exact(impute):
123+
df = pd.DataFrame({"x": pd.array([BIG, 0, 1, None], dtype="Int64"), "y": [1.0, 2.0, 3.0, 4.0]})
124+
out, report = fd.clean(
125+
df, impute=impute, strategy="conservative", return_report=True, **KEEP_ROWS
126+
)
127+
assert str(out["x"].dtype) == "Int64"
128+
assert out["x"].iloc[:3].tolist() == [BIG, 0, 1] # present values untouched
129+
expected = round(Fraction(BIG + 1, 3)) if impute == "mean" else 1
130+
assert out["x"].iloc[3] == expected
131+
notes = [a.description for a in report if a.step == "impute" and a.column == "x"]
132+
assert notes and "2**53" in notes[0]
133+
134+
135+
def test_default_engine_keeps_int64_beyond_2_53_exact():
136+
base = 2**60
137+
values = pd.array([base + (i % 7) for i in range(60)], dtype="Int64")
138+
s = pd.Series(values)
139+
s.iloc[[3, 17]] = pd.NA
140+
df = pd.DataFrame({"v": s, "x": np.random.default_rng(0).normal(0, 1, 60)})
141+
out, report = fd.clean(df, return_report=True, **KEEP_ROWS)
142+
assert str(out["v"].dtype) == "Int64"
143+
present = s.notna()
144+
assert out["v"][present].tolist() == s[present].tolist()
145+
filled = [a for a in report if a.step == "missing" and a.column == "v"]
146+
if filled and "filled" in filled[0].description:
147+
assert out["v"].isna().sum() == 0
148+
assert base <= int(out["v"].iloc[3]) <= base + 6
149+
assert "2**53" in filled[0].description
150+
151+
95152
def test_has_outliers_single_definition():
96153
assert missing._has_outliers is _has_outliers
97154
assert model_select._has_outliers is _has_outliers

‎tests/test_simple.py‎

Lines changed: 28 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -79,6 +79,34 @@ def test_fill_missing_verbose(capsys):
7979
assert "fill_missing" in capsys.readouterr().out
8080

8181

82+
def test_fill_missing_nullable_int_fractional_median_casts_to_float():
83+
out = fd.fill_missing(pd.DataFrame({"x": pd.array([1, None, 2], dtype="Int64")}))
84+
assert out["x"].tolist() == [1.0, 1.5, 2.0]
85+
86+
87+
def test_fill_missing_nullable_boolean_uses_mode():
88+
df = pd.DataFrame({"b": pd.array([True, None, False, False], dtype="boolean")})
89+
out = fd.fill_missing(df)
90+
assert str(out["b"].dtype) == "boolean"
91+
assert out["b"].tolist() == [True, False, False, False]
92+
# mean/median are not defined for booleans: left as-is, no crash
93+
assert fd.fill_missing(df, method="median")["b"].isna().sum() == 1
94+
95+
96+
def test_fill_missing_duplicate_labels_raise_clear_error():
97+
df = pd.DataFrame([[1.0, None], [None, 2.0]], columns=["x", "x"])
98+
with pytest.raises(ValueError, match="unique column labels"):
99+
fd.fill_missing(df)
100+
101+
102+
def test_fill_missing_keeps_int64_beyond_2_53_exact():
103+
big = 2**53 + 1
104+
df = pd.DataFrame({"x": pd.array([big, None, big + 2], dtype="Int64")})
105+
out = fd.fill_missing(df, method="mean")
106+
assert str(out["x"].dtype) == "Int64"
107+
assert out["x"].tolist() == [big, big + 1, big + 2]
108+
109+
82110
def test_resolve_columns_missing_raises():
83111
with pytest.raises(KeyError, match="columns not found"):
84112
fd.fill_missing(pd.DataFrame({"a": [1]}), columns="ghost")

0 commit comments

Comments
 (0)