|
1 | 1 | import numpy as np |
2 | 2 | import pandas as pd |
| 3 | +import pytest |
3 | 4 |
|
4 | 5 | import freshdata as fd |
| 6 | +from freshdata._util import PANDAS_MAJOR |
5 | 7 |
|
6 | 8 |
|
7 | 9 | def test_whitespace_stripped_object_and_string_dtype(): |
@@ -88,3 +90,62 @@ def test_unhashable_values_pass_through(): |
88 | 90 | out = fd.clean(df) |
89 | 91 | assert out["v"].iloc[0] == [1, 2] |
90 | 92 | assert not np.any(out["w"].isna()) |
| 93 | + |
| 94 | + |
| 95 | +def _plain(values): |
| 96 | + return [None if pd.isna(v) else v for v in values] |
| 97 | + |
| 98 | + |
| 99 | +def _column_steps(report, column): |
| 100 | + return [(a.step, a.count) for a in report if a.column == column] |
| 101 | + |
| 102 | + |
| 103 | +@pytest.mark.skipif(PANDAS_MAJOR < 2, reason="pd.ArrowDtype strings need pandas >= 2") |
| 104 | +@pytest.mark.parametrize( |
| 105 | + "values", |
| 106 | + [ |
| 107 | + [" a ", "N/A", "3", "4"], # text: strip + sentinel |
| 108 | + ["1", " 2 ", "N/A", "4"], # numeric-looking: fix_dtypes converts it |
| 109 | + ["2024-01-01", " 2024-02-01", None, "2024-03-01"], # dates |
| 110 | + ], |
| 111 | +) |
| 112 | +def test_arrow_string_column_cleans_like_string_pyarrow(values): |
| 113 | + pa = pytest.importorskip("pyarrow") |
| 114 | + arrow = pd.DataFrame( |
| 115 | + {"s": pd.Series(values, dtype=pd.ArrowDtype(pa.string())), "k": [1.0, 2.0, 3.0, 4.0]} |
| 116 | + ) |
| 117 | + string = arrow.astype({"s": "string[pyarrow]"}) |
| 118 | + out_arrow, report_arrow = fd.clean(arrow, return_report=True, verbose=False) |
| 119 | + out_string, report_string = fd.clean(string, return_report=True, verbose=False) |
| 120 | + assert _plain(out_arrow["s"].astype(object)) == _plain(out_string["s"].astype(object)) |
| 121 | + assert _column_steps(report_arrow, "s") == _column_steps(report_string, "s") |
| 122 | + assert ("strip_whitespace", 1) in _column_steps(report_arrow, "s") |
| 123 | + |
| 124 | + |
| 125 | +def test_categorical_text_is_normalized_and_stays_categorical(): |
| 126 | + cat = pd.Categorical( |
| 127 | + [" a ", "N/A", "b", "null", "a"], |
| 128 | + categories=["a", " a ", "N/A", "b", "null"], |
| 129 | + ordered=True, |
| 130 | + ) |
| 131 | + df = pd.DataFrame({"c": cat, "k": [1.0, 2.0, 3.0, 4.0, 5.0]}) |
| 132 | + out, report = fd.clean( |
| 133 | + df, strategy="conservative", return_report=True, verbose=False, |
| 134 | + drop_empty_rows=False, drop_duplicates=False, |
| 135 | + ) |
| 136 | + assert isinstance(out["c"].dtype, pd.CategoricalDtype) |
| 137 | + assert out["c"].cat.ordered |
| 138 | + assert list(out["c"].cat.categories) == ["a", "b"] # " a " merged, sentinels gone |
| 139 | + assert _plain(out["c"]) == ["a", None, "b", None, "a"] |
| 140 | + counts = dict(_column_steps(report, "c")) |
| 141 | + assert counts["strip_whitespace"] == 1 |
| 142 | + assert counts["normalize_sentinels"] == 2 |
| 143 | + |
| 144 | + |
| 145 | +def test_categorical_values_match_object_column(): |
| 146 | + values = [" a ", "N/A", "b", "null"] |
| 147 | + cat = pd.DataFrame({"c": pd.Categorical(values), "k": [1.0, 2.0, 3.0, 4.0]}) |
| 148 | + out_cat = fd.clean(cat, verbose=False) |
| 149 | + out_obj = fd.clean(cat.astype({"c": object}), verbose=False) |
| 150 | + assert isinstance(out_cat["c"].dtype, pd.CategoricalDtype) |
| 151 | + assert _plain(out_cat["c"].astype(object)) == _plain(out_obj["c"]) |
0 commit comments