Skip to content

Commit 7a651fd

Browse files
fix: route every to_numeric call through a shared crash guard
pandas < 3 reads the exponent digits of a numeric string into a C int with no overflow check, before it rejects trailing text (pandas-dev/pandas#62617, #63089, #63167; fixed in pandas 3.0 by pandas-dev/pandas#62741). #407 masked such cells in dtype inference only; about 30 other pd.to_numeric calls (domain validators, fieldcheck, CSV leading-zero detection, time-series scoring, MissForest, semantic checks, learning) still handed hash-like text such as "81e3104049863b72" straight to the parser and could crash the process. Add freshdata._numeric.safe_to_numeric, which keeps those cells away from pandas and otherwise forwards to pd.to_numeric unchanged (errors=, downcast=, dtype_backend=, index, name and dtype). A cell is guarded only when its leading exponent has ten or more significant digits, the smallest size that can overflow the C int accumulator; every shorter exponent, including subnormal and out-of-range values, is parsed exactly as pandas parses it. Numeric, boolean and datetime inputs skip the check; text is screened as one joined string, so clean columns stay close to free. The guard pieces move there and dtypes.py imports them, so dtype inference uses the same bound and keeps valid subnormal and underflow values. Calls on provably numeric dtypes stay direct, and a static test fails when a new unguarded call appears.
1 parent 57089a9 commit 7a651fd

21 files changed

Lines changed: 779 additions & 122 deletions

File tree

‎src/freshdata/_csv_io.py‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -15,6 +15,7 @@
1515

1616
import pandas as pd
1717

18+
from ._numeric import safe_to_numeric
1819
from .steps.dtypes import _has_leading_zero_ids
1920

2021
#: Rows read by the pre-scan. Zero padding that first appears after this many rows
@@ -63,6 +64,6 @@ def leading_zero_dtypes(
6364
values = sample.iloc[:, position].dropna()
6465
if values.empty or not _has_leading_zero_ids(values):
6566
continue
66-
if pd.to_numeric(values, errors="coerce").notna().all():
67+
if safe_to_numeric(values, errors="coerce").notna().all():
6768
padded[column] = str
6869
return padded

‎src/freshdata/_numeric.py‎

Lines changed: 220 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,220 @@
1+
"""Crash-safe :func:`pandas.to_numeric`.
2+
3+
pandas < 3 reads the exponent digits of a numeric string into a C int with no
4+
overflow check, and it does so *before* rejecting trailing text
5+
(pandas-dev/pandas#62617, #63089, #63167; fixed in pandas 3.0 by
6+
pandas-dev/pandas#62741). A cell that merely starts with such a token -- e.g.
7+
the hash-like value ``"81e3104049863b72"`` -- can segfault ``to_numeric`` and
8+
take the whole process down, whatever ``errors=`` says.
9+
10+
:func:`safe_to_numeric` keeps those cells away from the parser. Every
11+
``to_numeric`` call on data that may hold text must go through it
12+
(``tests/test_numeric.py`` enforces this for ``src/freshdata``).
13+
"""
14+
15+
from __future__ import annotations
16+
17+
import re
18+
from typing import Any
19+
20+
import numpy as np
21+
import pandas as pd
22+
from pandas.api.extensions import ExtensionArray
23+
24+
# The leading scientific-notation token of a cell, as pandas' C float parser
25+
# (precise_xstrtod) reads it. Match the prefix, not the whole cell: trailing
26+
# text is only rejected after the exponent has been accumulated.
27+
_SCIENTIFIC_PREFIX = re.compile(r"[+-]?(?:\d+(?:\.\d*)?|\.\d+)[eE]([+-]?\d+)")
28+
# precise_xstrtod accumulates the exponent digits in a C int (n = n * 10 + d)
29+
# and adds n to a mantissa adjustment, with no overflow check. Leading zeros
30+
# keep n at 0, so only significant digits matter: nine (below 10**9) cannot
31+
# overflow even after the adjustment, ten might. Every shorter exponent --
32+
# including subnormal and out-of-range values such as "4.9e-324" or "1e400" --
33+
# is left to pandas, which parses or rejects it safely.
34+
_MIN_UNSAFE_EXPONENT = 10**9
35+
# Column-level screen: an unsafe cell contains an exponent marker followed by
36+
# at least ten digits. A false positive (e.g. leading zeros) merely routes the
37+
# column to the exact per-cell check.
38+
_RISKY_EXPONENT = re.compile(r"[eE][+-]?\d{10,}")
39+
# dtype kinds pandas converts without its string parser: bool, integer,
40+
# unsigned, float, complex, datetime and timedelta.
41+
_PARSER_FREE_KINDS = frozenset("biufcmM")
42+
# Stand-in for masked cells under errors="ignore": text pandas cannot parse.
43+
_UNPARSEABLE = "x"
44+
45+
46+
def _has_unsafe_scientific_exponent(value: object) -> bool:
47+
"""True when *value* starts with scientific notation whose exponent can
48+
overflow the C int in pandas' parser (ten or more significant digits)."""
49+
if isinstance(value, bytes): # pandas parses bytes cells with the same C code
50+
value = value.decode("latin-1")
51+
if not isinstance(value, str):
52+
return False
53+
match = _SCIENTIFIC_PREFIX.match(value.lstrip())
54+
if match is None:
55+
return False
56+
try:
57+
return abs(int(match.group(1))) >= _MIN_UNSAFE_EXPONENT
58+
except ValueError: # e.g. more digits than int() accepts from a string
59+
return True
60+
61+
62+
def _text_blob(cells: np.ndarray) -> str:
63+
"""All text cells joined into one string, using C-level joins when possible."""
64+
try:
65+
return "\x1f".join(cells)
66+
except TypeError: # missing values or non-text cells
67+
pass
68+
try:
69+
return "\x1f".join(cells[pd.notna(cells)])
70+
except TypeError: # non-text objects (numbers, bytes, lists)
71+
pass
72+
return "\x1f".join(
73+
v.decode("latin-1") if isinstance(v, bytes) else v
74+
for v in cells
75+
if isinstance(v, (str, bytes))
76+
)
77+
78+
79+
def _unsafe_cells(cells: np.ndarray) -> np.ndarray | None:
80+
"""Positional mask of unsafe cells, or ``None`` when there are none."""
81+
blob = _text_blob(cells)
82+
if ("e" not in blob and "E" not in blob) or _RISKY_EXPONENT.search(blob) is None:
83+
return None
84+
unsafe = np.fromiter(
85+
map(_has_unsafe_scientific_exponent, cells), dtype=bool, count=len(cells)
86+
)
87+
return unsafe if unsafe.any() else None
88+
89+
90+
def _parsed_cells(values: Any) -> np.ndarray | None:
91+
"""The 1-D cells pandas would run through its string parser, else ``None``."""
92+
if isinstance(values, (pd.Series, pd.Index, np.ndarray, ExtensionArray)):
93+
if values.ndim != 1 or values.dtype.kind in _PARSER_FREE_KINDS:
94+
return None
95+
if isinstance(values.dtype, pd.CategoricalDtype) and _unsafe_cells(
96+
np.asarray(values.dtype.categories, dtype=object)
97+
) is None:
98+
return None
99+
return values if isinstance(values, np.ndarray) else values.to_numpy()
100+
if isinstance(values, (list, tuple)):
101+
cells = np.array(values, dtype=object)
102+
return cells if cells.ndim == 1 else None
103+
if isinstance(values, (str, bytes)):
104+
return np.array([values], dtype=object)
105+
return None
106+
107+
108+
def _categories(values: Any) -> Any:
109+
"""The ``.cat``-style accessor of a categorical Series or Categorical."""
110+
return values.cat if isinstance(values, pd.Series) else values
111+
112+
113+
def _is_categorical(values: Any) -> bool:
114+
return isinstance(getattr(values, "dtype", None), pd.CategoricalDtype)
115+
116+
117+
def _substitute(values: Any, unsafe: np.ndarray, fill: object) -> Any:
118+
"""A copy of *values* with the unsafe cells replaced by *fill*
119+
(``None`` means missing), keeping the container pandas dispatches on."""
120+
if isinstance(values, pd.Index):
121+
return values.where(~unsafe) if fill is None else values.where(~unsafe, fill)
122+
if isinstance(values, (pd.Series, ExtensionArray)) and fill is None:
123+
return (
124+
values.mask(unsafe)
125+
if isinstance(values, pd.Series)
126+
else pd.Series(values, copy=False).mask(unsafe).array
127+
)
128+
positions = np.flatnonzero(unsafe)
129+
if isinstance(values, (list, tuple)):
130+
out = np.array(values, dtype=object)
131+
elif isinstance(values, np.ndarray):
132+
# str/bytes arrays cannot hold None; pandas parses them as object anyway.
133+
out = values.astype(object) if fill is None else values.copy()
134+
elif isinstance(values, (pd.Series, ExtensionArray)):
135+
if _is_categorical(values) and fill not in values.dtype.categories:
136+
values = _categories(values).add_categories([fill])
137+
out = values.copy()
138+
else:
139+
return fill # scalar
140+
if isinstance(out, pd.Series):
141+
out.iloc[positions] = fill
142+
else:
143+
out[positions] = fill
144+
return out
145+
146+
147+
def _head(values: Any, stop: int) -> Any:
148+
"""The cells before position *stop*, in the same container."""
149+
if isinstance(values, pd.Series):
150+
return values.iloc[:stop]
151+
if isinstance(values, (str, bytes)):
152+
return np.array([], dtype=object)
153+
return values[:stop]
154+
155+
156+
def _restore(result: Any, values: Any, cells: np.ndarray, unsafe: np.ndarray) -> Any:
157+
"""Put the original cells back into an ``errors="ignore"`` result."""
158+
if isinstance(values, (str, bytes)):
159+
return values
160+
positions = np.flatnonzero(unsafe)
161+
if isinstance(result, pd.Index):
162+
arr = result.to_numpy(dtype=object, copy=True)
163+
arr[positions] = cells[positions]
164+
return pd.Index(arr, dtype=result.dtype, name=result.name)
165+
out = result.copy()
166+
if isinstance(out, pd.Series):
167+
out.iloc[positions] = cells[positions]
168+
else:
169+
out[positions] = cells[positions]
170+
if (
171+
_is_categorical(out)
172+
and _is_categorical(values)
173+
and _UNPARSEABLE not in values.dtype.categories
174+
):
175+
out = _categories(out).remove_categories([_UNPARSEABLE])
176+
return out
177+
178+
179+
def safe_to_numeric(values: Any, **kwargs: Any) -> Any:
180+
""":func:`pandas.to_numeric` that cannot crash on exponent-overflow text.
181+
182+
Accepts everything ``pd.to_numeric`` does (scalar, list, tuple, 1-D array,
183+
``Index`` or ``Series``) and forwards every keyword unchanged
184+
(``errors=``, ``downcast=``, ``dtype_backend=``). Input with no unsafe cell
185+
-- including every numeric, boolean or datetime dtype -- goes to pandas
186+
untouched, so the result is exactly pandas' result (index, name, dtype).
187+
188+
A cell that starts with scientific notation whose exponent has ten or more
189+
significant digits (enough to overflow pandas' C int) is treated as
190+
unparseable text: it becomes missing with ``errors="coerce"``, raises
191+
pandas' ``Unable to parse string`` ``ValueError`` with ``errors="raise"``,
192+
and is returned as-is with ``errors="ignore"``. pandas rejects almost all
193+
such cells too; the exception is a whole-cell negative exponent small
194+
enough not to overflow, which pandas would underflow to 0.0. Shorter
195+
exponents, including subnormal and out-of-range ones, are left to pandas.
196+
197+
The check is cheap: text columns are screened as one joined string, and
198+
only a column that contains an exponent marker followed by ten digits
199+
pays for the per-cell check.
200+
"""
201+
cells = _parsed_cells(values)
202+
unsafe = None if cells is None else _unsafe_cells(cells)
203+
if cells is None or unsafe is None:
204+
return pd.to_numeric(values, **kwargs)
205+
errors = kwargs.get("errors", "raise")
206+
if errors == "coerce":
207+
return pd.to_numeric(_substitute(values, unsafe, None), **kwargs)
208+
if errors == "raise":
209+
first = int(np.argmax(unsafe))
210+
# An earlier unparseable cell raises first, exactly as pandas would.
211+
pd.to_numeric(_head(values, first), **kwargs)
212+
cell = cells[first]
213+
text = cell.decode("latin-1") if isinstance(cell, bytes) else cell
214+
raise ValueError(f'Unable to parse string "{text}" at position {first}')
215+
# errors="ignore" (pandas itself rejects any other value).
216+
result = pd.to_numeric(_substitute(values, unsafe, _UNPARSEABLE), **kwargs)
217+
return _restore(result, values, cells, unsafe)
218+
219+
220+
__all__ = ["safe_to_numeric"]

‎src/freshdata/context/validate.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -167,7 +167,9 @@ def _check_allowed_values(series: pd.Series, c: ColumnConstraint) -> QualityFind
167167
def _check_range(series: pd.Series, c: ColumnConstraint) -> QualityFinding | None:
168168
import pandas as pd # noqa: PLC0415 - keep the context package import-light
169169

170-
numeric = pd.to_numeric(series, errors="coerce")
170+
from .._numeric import safe_to_numeric # noqa: PLC0415
171+
172+
numeric = safe_to_numeric(series, errors="coerce")
171173
lo, hi = c.params.get("lo"), c.params.get("hi")
172174
mask = pd.Series(False, index=series.index)
173175
if lo is not None:

‎src/freshdata/domains/_common.py‎

Lines changed: 6 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -18,6 +18,7 @@
1818

1919
import pandas as pd
2020

21+
from .._numeric import safe_to_numeric
2122
from .base import ColumnMapping, Rule
2223

2324
__all__ = [
@@ -197,14 +198,14 @@ def check_numeric(df: pd.DataFrame, mapping: ColumnMapping, rule: Rule) -> list[
197198
"""Flag present values that are not numeric."""
198199
col = mapping.actual(rule.fields[0])
199200
series = df[col]
200-
bad = series.notna() & pd.to_numeric(series, errors="coerce").isna()
201+
bad = series.notna() & safe_to_numeric(series, errors="coerce").isna()
201202
return df.index[bad].tolist()
202203

203204

204205
def check_nonneg(df: pd.DataFrame, mapping: ColumnMapping, rule: Rule) -> list[Any]:
205206
"""Flag present numeric values that are negative (non-numeric is a numeric rule's job)."""
206207
col = mapping.actual(rule.fields[0])
207-
numeric = pd.to_numeric(df[col], errors="coerce")
208+
numeric = safe_to_numeric(df[col], errors="coerce")
208209
return df.index[numeric.notna() & (numeric < 0)].tolist()
209210

210211

@@ -213,7 +214,7 @@ def check_nonneg_number(df: pd.DataFrame, mapping: ColumnMapping, rule: Rule) ->
213214
col = mapping.actual(rule.fields[0])
214215
series = df[col]
215216
present = series.notna()
216-
numeric = pd.to_numeric(series, errors="coerce")
217+
numeric = safe_to_numeric(series, errors="coerce")
217218
bad = (present & numeric.isna()) | (numeric.notna() & (numeric < 0))
218219
return df.index[bad].tolist()
219220

@@ -223,7 +224,7 @@ def check_positive(df: pd.DataFrame, mapping: ColumnMapping, rule: Rule) -> list
223224
col = mapping.actual(rule.fields[0])
224225
series = df[col]
225226
present = series.notna()
226-
numeric = pd.to_numeric(series, errors="coerce")
227+
numeric = safe_to_numeric(series, errors="coerce")
227228
bad = (present & numeric.isna()) | (numeric.notna() & (numeric <= 0))
228229
return df.index[bad].tolist()
229230

@@ -233,7 +234,7 @@ def check_positive_integer(df: pd.DataFrame, mapping: ColumnMapping, rule: Rule)
233234
col = mapping.actual(rule.fields[0])
234235
series = df[col]
235236
present = series.notna()
236-
numeric = pd.to_numeric(series, errors="coerce")
237+
numeric = safe_to_numeric(series, errors="coerce")
237238
is_pos_int = numeric.notna() & (numeric > 0) & (numeric == numeric.round())
238239
return df.index[present & ~is_pos_int].tolist()
239240

‎src/freshdata/domains/agriculture/validator.py‎

Lines changed: 3 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -17,6 +17,7 @@
1717

1818
import pandas as pd
1919

20+
from ..._numeric import safe_to_numeric
2021
from .._common import (
2122
check_both_present,
2223
check_iso_date,
@@ -113,7 +114,7 @@ def _check_season_year(
113114
) -> list[Any]:
114115
series = df[mapping.actual("season_year")]
115116
present = series.notna()
116-
numeric = pd.to_numeric(series, errors="coerce")
117+
numeric = safe_to_numeric(series, errors="coerce")
117118
max_year = pd.Timestamp.now().year + _MAX_YEAR_OFFSET
118119
valid = (
119120
numeric.notna()
@@ -127,7 +128,7 @@ def _check_date_year_matches_season(
127128
self, df: pd.DataFrame, mapping: ColumnMapping, rule: Rule
128129
) -> list[Any]:
129130
parsed = to_datetime_safe(df[mapping.actual("operation_date")])
130-
season = pd.to_numeric(df[mapping.actual("season_year")], errors="coerce")
131+
season = safe_to_numeric(df[mapping.actual("season_year")], errors="coerce")
131132
both = parsed.notna() & season.notna()
132133
bad = both & (parsed.dt.year != season)
133134
return df.index[bad].tolist()

‎src/freshdata/domains/base.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -22,6 +22,8 @@
2222

2323
import pandas as pd
2424

25+
from .._numeric import safe_to_numeric
26+
2527
#: Validation layers, executed in this strict order.
2628
LAYERS: tuple[str, ...] = ("schema", "format", "reference", "business", "semantic")
2729
#: Finding severities, in increasing order of seriousness.
@@ -578,7 +580,7 @@ def _check_enum(self, df: pd.DataFrame, mapping: ColumnMapping, rule: Rule) -> l
578580

579581
def _check_range(self, df: pd.DataFrame, mapping: ColumnMapping, rule: Rule) -> list[Any]:
580582
col = mapping.actual(rule.fields[0])
581-
numeric = pd.to_numeric(df[col], errors="coerce")
583+
numeric = safe_to_numeric(df[col], errors="coerce")
582584
present = df[col].notna()
583585
low = rule.params.get("min")
584586
high = rule.params.get("max")

‎src/freshdata/domains/education/validator.py‎

Lines changed: 3 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -16,6 +16,7 @@
1616

1717
import pandas as pd
1818

19+
from ..._numeric import safe_to_numeric
1920
from .._common import (
2021
check_both_present,
2122
check_ge_date,
@@ -121,7 +122,7 @@ def _check_school_year(
121122
) -> list[Any]:
122123
series = df[mapping.actual("school_year")]
123124
present = series.notna()
124-
numeric = pd.to_numeric(series, errors="coerce")
125+
numeric = safe_to_numeric(series, errors="coerce")
125126
max_year = pd.Timestamp.now().year + _MAX_YEAR_OFFSET
126127
valid = (
127128
numeric.notna()
@@ -135,7 +136,7 @@ def _check_enrollment_in_year(
135136
self, df: pd.DataFrame, mapping: ColumnMapping, rule: Rule
136137
) -> list[Any]:
137138
enroll = to_datetime_safe(df[mapping.actual("enrollment_date")])
138-
year = pd.to_numeric(df[mapping.actual("school_year")], errors="coerce")
139+
year = safe_to_numeric(df[mapping.actual("school_year")], errors="coerce")
139140
rows: list[Any] = []
140141
for idx in df.index[enroll.notna() & year.notna()]:
141142
school_year = int(year.at[idx])

‎src/freshdata/domains/energy/validator.py‎

Lines changed: 3 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -20,6 +20,7 @@
2020

2121
import pandas as pd
2222

23+
from ..._numeric import safe_to_numeric
2324
from .._common import check_iso_datetime, check_not_future, check_numeric
2425
from ..base import ColumnMapping, ConfigDrivenValidator, Rule, RuleResult
2526

@@ -96,7 +97,7 @@ def _check_function_code(self, df: pd.DataFrame, mapping: ColumnMapping,
9697
if col is None:
9798
return []
9899
allowed = set(_ref("modbus_function_codes")["codes"])
99-
codes = pd.to_numeric(df[col], errors="coerce")
100+
codes = safe_to_numeric(df[col], errors="coerce")
100101
present = df[col].notna()
101102
return list(df.index[present & ~codes.isin(allowed)])
102103

@@ -118,7 +119,7 @@ def _check_function_object(self, df: pd.DataFrame, mapping: ColumnMapping,
118119
if fc_col is None or obj_col is None:
119120
return []
120121
klass = {int(k): v for k, v in _ref("modbus_function_codes")["register_class"].items()}
121-
codes = pd.to_numeric(df[fc_col], errors="coerce")
122+
codes = safe_to_numeric(df[fc_col], errors="coerce")
122123
declared = df[obj_col].astype("string").str.strip().str.casefold()
123124
bad: list[Any] = []
124125
for idx in df.index:

0 commit comments

Comments
 (0)