Skip to content

Commit 8d99332

Browse files
fix(polars): keep integers exact when a polars column holds nulls (#467)
pl.DataFrame.to_pandas() renders an integer column that has nulls as float64, because a NumPy integer array cannot hold one. That silently rounds every value a float64 cannot represent: 2**53 + 1 came back as 2**53, and a UInt64 beyond 2**63 lost far more. This is not limited to engine="polars". Every public entry point reads a polars source through adapters.polars.to_pandas(), so fd.clean(pl_df) on the DEFAULT engine already returned the rounded value. Integer columns that hold nulls are now rebuilt from the raw integers plus a null mask, giving the pandas nullable dtype of the same width (Int64, UInt64, Int32, ...) — lossless, and what the same data already looked like when passed in as pandas. Columns without nulls round-trip exactly today and are untouched. The native polars engine converts its result through the same adapter instead of calling frame.to_pandas() directly. Default-output change: a polars input whose integer column holds nulls now cleans as a nullable integer column instead of float64. Closes #444 Co-authored-by: Kevin Costner <kiran.gangalakunta@gmail.com>
1 parent 49d49b2 commit 8d99332

4 files changed

Lines changed: 98 additions & 2 deletions

File tree

‎CHANGELOG.md‎

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -20,6 +20,17 @@ adheres to [Semantic Versioning](https://semver.org/).
2020
Polars returned a period as its raw int64 ordinal (`2020-01` → `600`) and an
2121
interval as a `{left, right}` struct. Both engines now fall back to the
2222
pandas reference for those dtypes, with the reason recorded.
23+
- A Polars frame whose integer column holds nulls no longer loses large
24+
integers. `pl.DataFrame.to_pandas()` renders such a column as `float64`,
25+
which rounds every value a float64 cannot represent, so `fd.clean(pl_df)`
26+
returned `9007199254740992` for an input of `2**53 + 1` — on the **default**
27+
engine, because every public entry point reads a Polars source through this
28+
conversion. Those columns are now rebuilt from the raw integers plus a null
29+
mask, giving the pandas nullable dtype of the same width (`Int64`, `UInt64`,
30+
`Int32`, …). Integer columns without nulls are untouched. **Default-output
31+
change:** a Polars input whose integer column has nulls now cleans as a
32+
nullable integer column instead of `float64`, exactly as the same data
33+
already did when passed in as pandas.
2334
- `CleanReport.revert()` no longer writes a restored value into other rows
2435
that share a duplicate index label. The undo log now records positional
2536
offsets and revert restores by position, so a frame with a non-unique index

‎src/freshdata/adapters/polars.py‎

Lines changed: 32 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -4,10 +4,16 @@
44

55
from typing import Any
66

7+
import numpy as np
78
import pandas as pd
89

910
_POLARS: Any = None
1011

12+
#: Polars integer dtypes that map to a pandas nullable integer dtype of the same width.
13+
_NULLABLE_INT_DTYPES = frozenset(
14+
{"Int8", "Int16", "Int32", "Int64", "UInt8", "UInt16", "UInt32", "UInt64"}
15+
)
16+
1117

1218
def _polars_module():
1319
global _POLARS
@@ -49,16 +55,41 @@ def to_pandas(df: object) -> pd.DataFrame:
4955
if is_polars_frame(df):
5056
pl_df: Any = df
5157
try:
52-
return pl_df.to_pandas()
58+
converted = pl_df.to_pandas()
5359
except ModuleNotFoundError as exc:
5460
raise ModuleNotFoundError(
5561
"converting a Polars frame to pandas requires pyarrow; "
5662
'install it with `pip install "freshdata-cleaner[polars]"` '
5763
"(or `pip install pyarrow`)"
5864
) from exc
65+
return restore_nullable_integers(pl_df, converted)
5966
raise TypeError(f"expected pandas or polars DataFrame, got {type(df).__name__}")
6067

6168

69+
def restore_nullable_integers(pl_df: Any, converted: pd.DataFrame) -> pd.DataFrame:
70+
"""Rebuild integer columns that hold nulls as pandas nullable integers.
71+
72+
``pl.DataFrame.to_pandas()`` renders an integer column that has nulls as
73+
``float64``, because NumPy integers cannot hold one. That silently rounds
74+
every value a float64 cannot represent — ``2**53 + 1`` comes back as
75+
``2**53``, and a ``UInt64`` beyond ``2**63`` loses far more. The integers
76+
are exact in polars, so rebuild those columns from the raw values plus a
77+
null mask, which is both lossless and what the same data would be in pandas
78+
to begin with. Columns without nulls already round-trip exactly and are
79+
left alone.
80+
"""
81+
for name, dtype in zip(pl_df.columns, pl_df.dtypes):
82+
if str(dtype) not in _NULLABLE_INT_DTYPES:
83+
continue
84+
column = pl_df[name]
85+
if column.null_count() == 0:
86+
continue # to_pandas() already gave us the exact NumPy integers
87+
values = np.asarray(column.fill_null(0).to_numpy())
88+
mask = np.asarray(column.is_null().to_numpy(), dtype=bool)
89+
converted[name] = pd.arrays.IntegerArray(values, mask)
90+
return converted
91+
92+
6293
def from_pandas(df: pd.DataFrame, original: object | None = None) -> object:
6394
if original is None or isinstance(original, pd.DataFrame):
6495
return df

‎src/freshdata/execution/__init__.py‎

Lines changed: 5 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -124,7 +124,11 @@ def _convert_output(frame: Any, output_format: str, report: Any = None) -> Any:
124124
if output_format == "pandas":
125125
if is_pandas:
126126
return frame
127-
return frame.to_pandas()
127+
from ..adapters.polars import to_pandas as polars_to_pandas
128+
129+
# Not frame.to_pandas(): that renders an integer column holding nulls as
130+
# float64, which rounds values a float64 cannot represent.
131+
return polars_to_pandas(frame)
128132

129133
if output_format == "polars":
130134
from ._lazy import require_polars

‎tests/test_polars_adapter.py‎

Lines changed: 50 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -56,3 +56,53 @@ def test_lazy_frame_round_trips_through_default_clean():
5656
out = fd.clean(lf, verbose=False)
5757
assert isinstance(out, pl.LazyFrame) # LazyFrame in, LazyFrame out
5858
assert out.collect()["b"].to_list() == ["x", "y", "z"]
59+
60+
61+
# -- #444 integer columns holding nulls must not go through float64 ------------
62+
63+
64+
@pytest.mark.parametrize(
65+
("values", "dtype", "expected"),
66+
[
67+
([2**53 + 1, None, 7], pl.Int64, "Int64"),
68+
([2**63 + 1, None, 7], pl.UInt64, "UInt64"),
69+
([5, None, 7], pl.Int32, "Int32"),
70+
],
71+
)
72+
def test_to_pandas_keeps_integers_exact_when_the_column_has_nulls(values, dtype, expected):
73+
"""pl.to_pandas() renders these as float64, rounding past 2**53 (#444)."""
74+
out = to_pandas(pl.DataFrame({"v": pl.Series(values, dtype=dtype)}))
75+
assert str(out["v"].dtype) == expected
76+
assert out["v"].tolist()[0] == values[0]
77+
assert out["v"].isna().tolist() == [False, True, False]
78+
79+
80+
def test_to_pandas_leaves_integer_columns_without_nulls_alone():
81+
out = to_pandas(pl.DataFrame({"v": pl.Series([2**53 + 1, 3], dtype=pl.Int64)}))
82+
assert str(out["v"].dtype) == "int64"
83+
assert out["v"].tolist() == [2**53 + 1, 3]
84+
85+
86+
def test_clean_does_not_round_large_integers_from_a_polars_frame():
87+
"""The default engine reads a polars source through the same adapter (#444)."""
88+
df = pl.DataFrame({"v": pl.Series([2**53 + 1, None, 7], dtype=pl.Int64), "k": [1.0, 2.0, 3.0]})
89+
out = fd.clean(df, verbose=False) # polars in, polars out
90+
assert out.schema["v"] == pl.Int64
91+
assert out["v"].to_list() == [2**53 + 1, None, 7]
92+
93+
94+
def test_native_polars_engine_returns_exact_integers():
95+
"""The native path converts its result with the same adapter (#444)."""
96+
pytest.importorskip("polars")
97+
df = pd.DataFrame({"v": pd.array([2**53 + 1, None, 7], dtype="Int64"), "k": [1.0, 2.0, 3.0]})
98+
out, report = fd.clean(
99+
df,
100+
strategy="conservative",
101+
fix_dtypes=False,
102+
verbose=False,
103+
engine="polars",
104+
return_report=True,
105+
)
106+
assert report.backend == "polars"
107+
assert str(out["v"].dtype) == "Int64"
108+
assert out["v"].tolist()[0] == 2**53 + 1

0 commit comments

Comments
 (0)