Skip to content

Commit 902d2e0

Browse files
fix(text): normalize ArrowDtype string and categorical columns
Whitespace stripping, sentinel normalization, fix_dtypes and the profile's text issues skipped two kinds of text column: - pd.ArrowDtype(pa.string()) (pandas >= 2): _is_stringlike_dtype only knew object and StringDtype. It now accepts Arrow string/large_string/string_view. Type inference parses through a string[pyarrow] view, because Arrow-backed parse results lack arithmetic the numeric check relies on (e.g. `%`), so the column cleans exactly like the same data as string[pyarrow]. - Categorical columns: categories with text are repaired too, and the column keeps its categorical dtype and `ordered` flag. Values are normalized like the equivalent object column (same counts); categories that become equal merge and sentinel categories disappear. fix_dtypes still leaves categoricals alone, and the profile reports their whitespace/sentinel issues without a dtype suggestion. fd.clean_text's default column selection uses the same text-dtype check. Closes #212
1 parent 55a8044 commit 902d2e0

8 files changed

Lines changed: 200 additions & 19 deletions

File tree

‎docs/cleaning-engine.md‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -18,7 +18,7 @@ level, and a confidence score.
1818
| order | step | what it does |
1919
|---|---|---|
2020
| 1 | `column_names` | snake_case names, deduplicate collisions (`"a", "a"` → `"a", "a_2"`) |
21-
| 2 | `strip_whitespace` | trim surrounding whitespace in text cells (internal spacing kept) |
21+
| 2 | `strip_whitespace` | trim surrounding whitespace in text cells (internal spacing kept) — object, `string`, Arrow `string` and categorical columns; a categorical keeps its dtype (its categories are repaired, and ones that become equal merge) |
2222
| 3 | `normalize_sentinels` | `"N/A"`, `"null"`, `"-"`, `""`, `"#REF!"`, … → missing |
2323
| 4 | `drop_empty_columns` / `drop_empty_rows` | remove all-missing columns and rows |
2424
| 5 | `fix_dtypes` | text → numeric (`"$1,234.56"` works) / datetime / boolean, validated |

‎src/freshdata/_util.py‎

Lines changed: 38 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -96,7 +96,44 @@ def stringlike_columns(df: pd.DataFrame) -> list:
9696

9797

9898
def _is_stringlike_dtype(dtype: object) -> bool:
99-
return pd.api.types.is_object_dtype(dtype) or isinstance(dtype, pd.StringDtype)
99+
return (
100+
pd.api.types.is_object_dtype(dtype)
101+
or isinstance(dtype, pd.StringDtype)
102+
or is_arrow_string_dtype(dtype)
103+
)
104+
105+
106+
def is_arrow_string_dtype(dtype: object) -> bool:
107+
"""True for a ``pd.ArrowDtype`` holding strings (pandas >= 2 only).
108+
109+
``pd.ArrowDtype(pa.string())`` carries the same text as ``string[pyarrow]``
110+
but is a different dtype class, so it needs its own check. pandas 1.5's
111+
experimental ``ArrowDtype`` is left alone.
112+
"""
113+
arrow_dtype_cls = getattr(pd, "ArrowDtype", None)
114+
if PANDAS_MAJOR < 2 or arrow_dtype_cls is None or not isinstance(dtype, arrow_dtype_cls):
115+
return False
116+
import pyarrow as pa # noqa: PLC0415 - an ArrowDtype implies pyarrow is installed
117+
118+
arrow_type = getattr(dtype, "pyarrow_dtype", None)
119+
is_string_view = getattr(pa.types, "is_string_view", None)
120+
return bool(
121+
pa.types.is_string(arrow_type)
122+
or pa.types.is_large_string(arrow_type)
123+
or (is_string_view is not None and is_string_view(arrow_type))
124+
)
125+
126+
127+
def as_string_view(s: pd.Series) -> pd.Series:
128+
"""``string[pyarrow]`` copy of an Arrow-string column; any other column as-is.
129+
130+
Type inference parses text into numbers/dates and then does arithmetic on the
131+
result; Arrow-backed results do not implement all of it (e.g. ``%``), while
132+
the ``string[pyarrow]`` path produces regular pandas dtypes.
133+
"""
134+
if is_arrow_string_dtype(s.dtype):
135+
return s.astype(pd.StringDtype("pyarrow"))
136+
return s
100137

101138

102139
#: Leading characters Excel/Sheets/LibreOffice interpret as a formula

‎src/freshdata/profile.py‎

Lines changed: 16 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -19,7 +19,12 @@
1919
from .render.mixins import HtmlReprMixin
2020
from .steps.dtypes import suggest_conversion
2121
from .steps.outliers import _bounds
22-
from .steps.strings import active_sentinels, normalize_text
22+
from .steps.strings import (
23+
active_sentinels,
24+
is_text_categorical_dtype,
25+
normalize_categorical,
26+
normalize_text,
27+
)
2328

2429

2530
@dataclass(frozen=True)
@@ -162,19 +167,22 @@ def _profile_column(name: str, s: pd.Series, config: CleanConfig,
162167
issues.append("mixed value types")
163168

164169
is_textual = _is_stringlike_dtype(s.dtype)
165-
if is_textual and non_null:
166-
normalized, n_stripped, n_sentinels, n_case = normalize_text(s, config, sentinels)
170+
text_categorical = is_text_categorical_dtype(s.dtype)
171+
if (is_textual or text_categorical) and non_null:
172+
normalize = normalize_categorical if text_categorical else normalize_text
173+
normalized, n_stripped, n_sentinels, n_case = normalize(s, config, sentinels)
167174
if n_stripped:
168175
issues.append(f"{n_stripped} value(s) with surrounding whitespace")
169176
if n_sentinels:
170177
issues.append(f"{n_sentinels} sentinel value(s) meaning missing")
171178
if n_case:
172179
issues.append(f"{n_case} value(s) would be converted to {config.string_case}case")
173-
target, converted, n_coerced = suggest_conversion(normalized, config)
174-
if converted is not None:
175-
suggested = str(converted.dtype)
176-
note = f", {n_coerced} unparseable" if n_coerced else ""
177-
issues.append(f"would convert to {suggested}{note}")
180+
if is_textual: # cleaning keeps categoricals categorical; no dtype suggestion
181+
target, converted, n_coerced = suggest_conversion(normalized, config)
182+
if converted is not None:
183+
suggested = str(converted.dtype)
184+
note = f", {n_coerced} unparseable" if n_coerced else ""
185+
issues.append(f"would convert to {suggested}{note}")
178186

179187
if is_numeric_dtype(s) and not is_bool_dtype(s) and non_null >= 20:
180188
bounds = _bounds(s, config)

‎src/freshdata/steps/dtypes.py‎

Lines changed: 12 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -23,7 +23,13 @@
2323
import pandas as pd
2424
from pandas.api.types import infer_dtype, is_datetime64_any_dtype
2525

26-
from .._util import PANDAS_MAJOR, mask_sensitive_value, sample_series, stringlike_columns
26+
from .._util import (
27+
PANDAS_MAJOR,
28+
as_string_view,
29+
mask_sensitive_value,
30+
sample_series,
31+
stringlike_columns,
32+
)
2733
from ..config import CleanConfig
2834
from ..report import CleanReport
2935

@@ -458,6 +464,7 @@ def suggest_conversion(
458464
the cleaning pipeline and :func:`freshdata.profile` so the preview always
459465
matches what cleaning would actually do.
460466
"""
467+
s = as_string_view(s) # Arrow strings parse through the string[pyarrow] path
461468
nonnull = s.dropna()
462469
if nonnull.empty:
463470
return "none", None, 0
@@ -665,14 +672,15 @@ def fix_dtypes(df: pd.DataFrame, config: CleanConfig, report: CleanReport) -> pd
665672
for col in stringlike_columns(df):
666673
if str(col) in protected:
667674
continue # context-protected columns must stay byte-identical
668-
target, converted, n_coerced = suggest_conversion(df[col], config)
675+
s = as_string_view(df[col])
676+
target, converted, n_coerced = suggest_conversion(s, config)
669677
if converted is None:
670-
_warn_type_contamination(str(col), df[col], config, report)
678+
_warn_type_contamination(str(col), s, config, report)
671679
continue
672680
description = f"converted to {converted.dtype}"
673681
if n_coerced:
674682
description += f" ({n_coerced} unparseable value(s) set to missing)"
675-
_record_coerced(str(col), df[col], converted, report, config)
683+
_record_coerced(str(col), s, converted, report, config)
676684
report.add("fix_dtypes", description, column=str(col),
677685
count=int(converted.notna().sum()) + n_coerced)
678686
df[col] = converted

‎src/freshdata/steps/strings.py‎

Lines changed: 51 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -15,7 +15,7 @@
1515
from pandas.api.types import infer_dtype
1616

1717
from .._sentinels import DEFAULT_SENTINELS
18-
from .._util import stringlike_columns
18+
from .._util import _is_stringlike_dtype
1919
from ..config import CleanConfig
2020
from ..report import CleanReport
2121

@@ -91,8 +91,51 @@ def normalize_text(
9191
return s, n_stripped, n_sentinels, n_case
9292

9393

94+
def is_text_categorical_dtype(dtype: object) -> bool:
95+
"""True for a categorical dtype whose categories hold text worth repairing."""
96+
return (
97+
isinstance(dtype, pd.CategoricalDtype)
98+
and infer_dtype(dtype.categories, skipna=True) in _TEXTUAL_KINDS
99+
)
100+
101+
102+
def normalize_categorical(
103+
s: pd.Series, config: CleanConfig, sentinels: frozenset[str]
104+
) -> tuple[pd.Series, int, int, int]:
105+
""":func:`normalize_text` for a text categorical, keeping the categorical dtype.
106+
107+
Values are repaired exactly as the equivalent object column would be (so
108+
counts match), then rebuilt as a categorical with the same ``ordered`` flag
109+
whose categories are the repaired originals: ``" a "`` and ``"a"`` merge,
110+
and sentinel categories disappear.
111+
"""
112+
normalized, n_stripped, n_sentinels, n_case = normalize_text(
113+
s.astype(object), config, sentinels
114+
)
115+
if not (n_stripped or n_sentinels or n_case):
116+
return s, 0, 0, 0
117+
categories, *_ = normalize_text(pd.Series(s.cat.categories, dtype=object), config, sentinels)
118+
rebuilt = pd.Categorical(
119+
normalized, categories=pd.unique(categories.dropna()), ordered=s.cat.ordered
120+
)
121+
return pd.Series(rebuilt, index=s.index, name=s.name), n_stripped, n_sentinels, n_case
122+
123+
124+
def _text_columns(df: pd.DataFrame) -> list:
125+
"""Object/string columns plus text categoricals, in frame order."""
126+
return [
127+
col
128+
for col, dtype in zip(df.columns, df.dtypes)
129+
if _is_stringlike_dtype(dtype) or is_text_categorical_dtype(dtype)
130+
]
131+
132+
94133
def clean_strings(df: pd.DataFrame, config: CleanConfig, report: CleanReport) -> pd.DataFrame:
95-
"""Apply whitespace stripping and sentinel→missing to text-capable columns."""
134+
"""Apply whitespace stripping and sentinel→missing to text-capable columns.
135+
136+
Categorical columns with text categories are repaired too; they keep their
137+
categorical dtype (see :func:`normalize_categorical`).
138+
"""
96139
if not (
97140
config.strip_whitespace
98141
or config.normalize_sentinels
@@ -103,10 +146,14 @@ def clean_strings(df: pd.DataFrame, config: CleanConfig, report: CleanReport) ->
103146
from ..guard import hard_protected_columns # noqa: PLC0415 — cycle-safe lazy import
104147

105148
protected = hard_protected_columns(config, df.columns)
106-
for col in stringlike_columns(df):
149+
for col in _text_columns(df):
107150
if str(col) in protected:
108151
continue # context-protected columns must stay byte-identical
109-
normalized, n_stripped, n_sentinels, n_case = normalize_text(df[col], config, sentinels)
152+
s = df[col]
153+
normalize = (
154+
normalize_categorical if isinstance(s.dtype, pd.CategoricalDtype) else normalize_text
155+
)
156+
normalized, n_stripped, n_sentinels, n_case = normalize(s, config, sentinels)
110157
if n_stripped:
111158
report.add("strip_whitespace", "trimmed surrounding whitespace",
112159
column=str(col), count=n_stripped)

‎src/freshdata/textclean.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -28,6 +28,8 @@
2828

2929
import pandas as pd
3030

31+
from ._util import _is_stringlike_dtype
32+
3133
__all__ = [
3234
"TextCleanConfig",
3335
"CleanedText",
@@ -301,7 +303,7 @@ def clean_text(
301303
never runs on an amount or identifier column.
302304
"""
303305
if columns is None:
304-
cols = [c for c in df.columns if df[c].dtype == object or str(df[c].dtype) == "string"]
306+
cols = [c for c in df.columns if _is_stringlike_dtype(df[c].dtype)]
305307
else:
306308
missing = [c for c in columns if c not in df.columns]
307309
if missing:

‎tests/test_profile.py‎

Lines changed: 18 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -96,3 +96,21 @@ def test_empty_frame_profile():
9696
p = fd.profile(pd.DataFrame())
9797
assert p.n_rows == 0 and p.n_cols == 0
9898
assert str(p) # renders without crashing
99+
100+
101+
def test_profile_flags_text_issues_in_categorical_columns():
102+
df = pd.DataFrame({"c": pd.Categorical([" a ", "N/A", "b", "b"])})
103+
issues = fd.profile(df).columns[0].issues
104+
assert "1 value(s) with surrounding whitespace" in issues
105+
assert "1 sentinel value(s) meaning missing" in issues
106+
assert not any("would convert" in issue for issue in issues) # stays categorical
107+
108+
109+
def test_profile_flags_text_issues_in_arrow_string_columns():
110+
if int(pd.__version__.split(".")[0]) < 2:
111+
pytest.skip("pd.ArrowDtype strings need pandas >= 2")
112+
pa = pytest.importorskip("pyarrow")
113+
df = pd.DataFrame({"s": pd.Series([" a ", "N/A", "b", "b"], dtype=pd.ArrowDtype(pa.string()))})
114+
issues = fd.profile(df).columns[0].issues
115+
assert "1 value(s) with surrounding whitespace" in issues
116+
assert "1 sentinel value(s) meaning missing" in issues

‎tests/test_strings.py‎

Lines changed: 61 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,7 +1,9 @@
11
import numpy as np
22
import pandas as pd
3+
import pytest
34

45
import freshdata as fd
6+
from freshdata._util import PANDAS_MAJOR
57

68

79
def test_whitespace_stripped_object_and_string_dtype():
@@ -88,3 +90,62 @@ def test_unhashable_values_pass_through():
8890
out = fd.clean(df)
8991
assert out["v"].iloc[0] == [1, 2]
9092
assert not np.any(out["w"].isna())
93+
94+
95+
def _plain(values):
96+
return [None if pd.isna(v) else v for v in values]
97+
98+
99+
def _column_steps(report, column):
100+
return [(a.step, a.count) for a in report if a.column == column]
101+
102+
103+
@pytest.mark.skipif(PANDAS_MAJOR < 2, reason="pd.ArrowDtype strings need pandas >= 2")
104+
@pytest.mark.parametrize(
105+
"values",
106+
[
107+
[" a ", "N/A", "3", "4"], # text: strip + sentinel
108+
["1", " 2 ", "N/A", "4"], # numeric-looking: fix_dtypes converts it
109+
["2024-01-01", " 2024-02-01", None, "2024-03-01"], # dates
110+
],
111+
)
112+
def test_arrow_string_column_cleans_like_string_pyarrow(values):
113+
pa = pytest.importorskip("pyarrow")
114+
arrow = pd.DataFrame(
115+
{"s": pd.Series(values, dtype=pd.ArrowDtype(pa.string())), "k": [1.0, 2.0, 3.0, 4.0]}
116+
)
117+
string = arrow.astype({"s": "string[pyarrow]"})
118+
out_arrow, report_arrow = fd.clean(arrow, return_report=True, verbose=False)
119+
out_string, report_string = fd.clean(string, return_report=True, verbose=False)
120+
assert _plain(out_arrow["s"].astype(object)) == _plain(out_string["s"].astype(object))
121+
assert _column_steps(report_arrow, "s") == _column_steps(report_string, "s")
122+
assert ("strip_whitespace", 1) in _column_steps(report_arrow, "s")
123+
124+
125+
def test_categorical_text_is_normalized_and_stays_categorical():
126+
cat = pd.Categorical(
127+
[" a ", "N/A", "b", "null", "a"],
128+
categories=["a", " a ", "N/A", "b", "null"],
129+
ordered=True,
130+
)
131+
df = pd.DataFrame({"c": cat, "k": [1.0, 2.0, 3.0, 4.0, 5.0]})
132+
out, report = fd.clean(
133+
df, strategy="conservative", return_report=True, verbose=False,
134+
drop_empty_rows=False, drop_duplicates=False,
135+
)
136+
assert isinstance(out["c"].dtype, pd.CategoricalDtype)
137+
assert out["c"].cat.ordered
138+
assert list(out["c"].cat.categories) == ["a", "b"] # " a " merged, sentinels gone
139+
assert _plain(out["c"]) == ["a", None, "b", None, "a"]
140+
counts = dict(_column_steps(report, "c"))
141+
assert counts["strip_whitespace"] == 1
142+
assert counts["normalize_sentinels"] == 2
143+
144+
145+
def test_categorical_values_match_object_column():
146+
values = [" a ", "N/A", "b", "null"]
147+
cat = pd.DataFrame({"c": pd.Categorical(values), "k": [1.0, 2.0, 3.0, 4.0]})
148+
out_cat = fd.clean(cat, verbose=False)
149+
out_obj = fd.clean(cat.astype({"c": object}), verbose=False)
150+
assert isinstance(out_cat["c"].dtype, pd.CategoricalDtype)
151+
assert _plain(out_cat["c"].astype(object)) == _plain(out_obj["c"])

0 commit comments

Comments
 (0)