|
6 | 6 | imputation path must survive those columns and keep Int64/float behaviour. |
7 | 7 | """ |
8 | 8 |
|
| 9 | +from fractions import Fraction |
| 10 | + |
9 | 11 | import numpy as np |
10 | 12 | import pandas as pd |
11 | 13 | import pytest |
12 | 14 |
|
13 | 15 | import freshdata as fd |
14 | | -from freshdata._util import safe_median |
| 16 | +from freshdata._util import exact_int_stat, exceeds_float64_exact, fill_na_exact, safe_median |
15 | 17 | from freshdata.engine import missing, model_select |
16 | 18 | from freshdata.engine.utils import _has_outliers |
17 | 19 |
|
@@ -92,6 +94,61 @@ def test_seasonal_imputation_global_median_on_nullable_int(dtype): |
92 | 94 | assert out["v"].isna().sum() == 0 |
93 | 95 |
|
94 | 96 |
|
| 97 | +BIG = 2**53 + 1 # float64 rounds this to 2**53 |
| 98 | + |
| 99 | + |
| 100 | +def test_exact_int_stat_and_detection(): |
| 101 | + s = pd.Series(pd.array([BIG, 0, 1, None], dtype="Int64")) |
| 102 | + assert exceeds_float64_exact(s) |
| 103 | + assert not exceeds_float64_exact(pd.Series(pd.array([2**53, None], dtype="Int64"))) |
| 104 | + assert not exceeds_float64_exact(pd.Series([float(BIG), np.nan])) |
| 105 | + assert exact_int_stat(s, "mean") == round(Fraction(BIG + 1, 3)) |
| 106 | + assert exact_int_stat(s, "median") == 1 |
| 107 | + even = pd.Series(pd.array([BIG, BIG + 2, None, BIG + 5, BIG + 7], dtype="Int64")) |
| 108 | + # (BIG+2 + BIG+5) / 2 == 2**53 + 4.5, which rounds half-to-even to 2**53 + 4 |
| 109 | + assert exact_int_stat(even, "median") == BIG + 3 |
| 110 | + |
| 111 | + |
| 112 | +def test_fill_na_exact_keeps_small_int_behaviour(): |
| 113 | + s = pd.Series(pd.array([1, None, 2], dtype="Int64")) |
| 114 | + filled, note = fill_na_exact(s, 1.5) |
| 115 | + assert filled.tolist() == [1.0, 1.5, 2.0] |
| 116 | + assert note == ", column cast to float64" |
| 117 | + filled, note = fill_na_exact(s, 1) |
| 118 | + assert str(filled.dtype) == "Int64" and note == "" |
| 119 | + |
| 120 | + |
| 121 | +@pytest.mark.parametrize("impute", ["mean", "median", "auto"]) |
| 122 | +def test_explicit_impute_keeps_int64_beyond_2_53_exact(impute): |
| 123 | + df = pd.DataFrame({"x": pd.array([BIG, 0, 1, None], dtype="Int64"), "y": [1.0, 2.0, 3.0, 4.0]}) |
| 124 | + out, report = fd.clean( |
| 125 | + df, impute=impute, strategy="conservative", return_report=True, **KEEP_ROWS |
| 126 | + ) |
| 127 | + assert str(out["x"].dtype) == "Int64" |
| 128 | + assert out["x"].iloc[:3].tolist() == [BIG, 0, 1] # present values untouched |
| 129 | + expected = round(Fraction(BIG + 1, 3)) if impute == "mean" else 1 |
| 130 | + assert out["x"].iloc[3] == expected |
| 131 | + notes = [a.description for a in report if a.step == "impute" and a.column == "x"] |
| 132 | + assert notes and "2**53" in notes[0] |
| 133 | + |
| 134 | + |
| 135 | +def test_default_engine_keeps_int64_beyond_2_53_exact(): |
| 136 | + base = 2**60 |
| 137 | + values = pd.array([base + (i % 7) for i in range(60)], dtype="Int64") |
| 138 | + s = pd.Series(values) |
| 139 | + s.iloc[[3, 17]] = pd.NA |
| 140 | + df = pd.DataFrame({"v": s, "x": np.random.default_rng(0).normal(0, 1, 60)}) |
| 141 | + out, report = fd.clean(df, return_report=True, **KEEP_ROWS) |
| 142 | + assert str(out["v"].dtype) == "Int64" |
| 143 | + present = s.notna() |
| 144 | + assert out["v"][present].tolist() == s[present].tolist() |
| 145 | + filled = [a for a in report if a.step == "missing" and a.column == "v"] |
| 146 | + if filled and "filled" in filled[0].description: |
| 147 | + assert out["v"].isna().sum() == 0 |
| 148 | + assert base <= int(out["v"].iloc[3]) <= base + 6 |
| 149 | + assert "2**53" in filled[0].description |
| 150 | + |
| 151 | + |
95 | 152 | def test_has_outliers_single_definition(): |
96 | 153 | assert missing._has_outliers is _has_outliers |
97 | 154 | assert model_select._has_outliers is _has_outliers |
|
0 commit comments