|
| 1 | +"""Crash-safe :func:`pandas.to_numeric`. |
| 2 | +
|
| 3 | +pandas < 3 reads the exponent digits of a numeric string into a C int with no |
| 4 | +overflow check, and it does so *before* rejecting trailing text |
| 5 | +(pandas-dev/pandas#62617, #63089, #63167; fixed in pandas 3.0 by |
| 6 | +pandas-dev/pandas#62741). A cell that merely starts with such a token -- e.g. |
| 7 | +the hash-like value ``"81e3104049863b72"`` -- can segfault ``to_numeric`` and |
| 8 | +take the whole process down, whatever ``errors=`` says. |
| 9 | +
|
| 10 | +:func:`safe_to_numeric` keeps those cells away from the parser. Every |
| 11 | +``to_numeric`` call on data that may hold text must go through it |
| 12 | +(``tests/test_numeric.py`` enforces this for ``src/freshdata``). |
| 13 | +""" |
| 14 | + |
| 15 | +from __future__ import annotations |
| 16 | + |
| 17 | +import re |
| 18 | +from typing import Any |
| 19 | + |
| 20 | +import numpy as np |
| 21 | +import pandas as pd |
| 22 | +from pandas.api.extensions import ExtensionArray |
| 23 | + |
| 24 | +# The leading scientific-notation token of a cell, as pandas' C float parser |
| 25 | +# (precise_xstrtod) reads it. Match the prefix, not the whole cell: trailing |
| 26 | +# text is only rejected after the exponent has been accumulated. |
| 27 | +_SCIENTIFIC_PREFIX = re.compile(r"[+-]?(?:\d+(?:\.\d*)?|\.\d+)[eE]([+-]?\d+)") |
| 28 | +# precise_xstrtod accumulates the exponent digits in a C int (n = n * 10 + d) |
| 29 | +# and adds n to a mantissa adjustment, with no overflow check. Leading zeros |
| 30 | +# keep n at 0, so only significant digits matter: nine (below 10**9) cannot |
| 31 | +# overflow even after the adjustment, ten might. Every shorter exponent -- |
| 32 | +# including subnormal and out-of-range values such as "4.9e-324" or "1e400" -- |
| 33 | +# is left to pandas, which parses or rejects it safely. |
| 34 | +_MIN_UNSAFE_EXPONENT = 10**9 |
| 35 | +# Column-level screen: an unsafe cell contains an exponent marker followed by |
| 36 | +# at least ten digits. A false positive (e.g. leading zeros) merely routes the |
| 37 | +# column to the exact per-cell check. |
| 38 | +_RISKY_EXPONENT = re.compile(r"[eE][+-]?\d{10,}") |
| 39 | +# dtype kinds pandas converts without its string parser: bool, integer, |
| 40 | +# unsigned, float, complex, datetime and timedelta. |
| 41 | +_PARSER_FREE_KINDS = frozenset("biufcmM") |
| 42 | +# Stand-in for masked cells under errors="ignore": text pandas cannot parse. |
| 43 | +_UNPARSEABLE = "x" |
| 44 | + |
| 45 | + |
| 46 | +def _has_unsafe_scientific_exponent(value: object) -> bool: |
| 47 | + """True when *value* starts with scientific notation whose exponent can |
| 48 | + overflow the C int in pandas' parser (ten or more significant digits).""" |
| 49 | + if isinstance(value, bytes): # pandas parses bytes cells with the same C code |
| 50 | + value = value.decode("latin-1") |
| 51 | + if not isinstance(value, str): |
| 52 | + return False |
| 53 | + match = _SCIENTIFIC_PREFIX.match(value.lstrip()) |
| 54 | + if match is None: |
| 55 | + return False |
| 56 | + try: |
| 57 | + return abs(int(match.group(1))) >= _MIN_UNSAFE_EXPONENT |
| 58 | + except ValueError: # e.g. more digits than int() accepts from a string |
| 59 | + return True |
| 60 | + |
| 61 | + |
| 62 | +def _text_blob(cells: np.ndarray) -> str: |
| 63 | + """All text cells joined into one string, using C-level joins when possible.""" |
| 64 | + try: |
| 65 | + return "\x1f".join(cells) |
| 66 | + except TypeError: # missing values or non-text cells |
| 67 | + pass |
| 68 | + try: |
| 69 | + return "\x1f".join(cells[pd.notna(cells)]) |
| 70 | + except TypeError: # non-text objects (numbers, bytes, lists) |
| 71 | + pass |
| 72 | + return "\x1f".join( |
| 73 | + v.decode("latin-1") if isinstance(v, bytes) else v |
| 74 | + for v in cells |
| 75 | + if isinstance(v, (str, bytes)) |
| 76 | + ) |
| 77 | + |
| 78 | + |
| 79 | +def _unsafe_cells(cells: np.ndarray) -> np.ndarray | None: |
| 80 | + """Positional mask of unsafe cells, or ``None`` when there are none.""" |
| 81 | + blob = _text_blob(cells) |
| 82 | + if ("e" not in blob and "E" not in blob) or _RISKY_EXPONENT.search(blob) is None: |
| 83 | + return None |
| 84 | + unsafe = np.fromiter( |
| 85 | + map(_has_unsafe_scientific_exponent, cells), dtype=bool, count=len(cells) |
| 86 | + ) |
| 87 | + return unsafe if unsafe.any() else None |
| 88 | + |
| 89 | + |
| 90 | +def _parsed_cells(values: Any) -> np.ndarray | None: |
| 91 | + """The 1-D cells pandas would run through its string parser, else ``None``.""" |
| 92 | + if isinstance(values, (pd.Series, pd.Index, np.ndarray, ExtensionArray)): |
| 93 | + if values.ndim != 1 or values.dtype.kind in _PARSER_FREE_KINDS: |
| 94 | + return None |
| 95 | + if isinstance(values.dtype, pd.CategoricalDtype) and _unsafe_cells( |
| 96 | + np.asarray(values.dtype.categories, dtype=object) |
| 97 | + ) is None: |
| 98 | + return None |
| 99 | + return values if isinstance(values, np.ndarray) else values.to_numpy() |
| 100 | + if isinstance(values, (list, tuple)): |
| 101 | + cells = np.array(values, dtype=object) |
| 102 | + return cells if cells.ndim == 1 else None |
| 103 | + if isinstance(values, (str, bytes)): |
| 104 | + return np.array([values], dtype=object) |
| 105 | + return None |
| 106 | + |
| 107 | + |
| 108 | +def _categories(values: Any) -> Any: |
| 109 | + """The ``.cat``-style accessor of a categorical Series or Categorical.""" |
| 110 | + return values.cat if isinstance(values, pd.Series) else values |
| 111 | + |
| 112 | + |
| 113 | +def _is_categorical(values: Any) -> bool: |
| 114 | + return isinstance(getattr(values, "dtype", None), pd.CategoricalDtype) |
| 115 | + |
| 116 | + |
| 117 | +def _substitute(values: Any, unsafe: np.ndarray, fill: object) -> Any: |
| 118 | + """A copy of *values* with the unsafe cells replaced by *fill* |
| 119 | + (``None`` means missing), keeping the container pandas dispatches on.""" |
| 120 | + if isinstance(values, pd.Index): |
| 121 | + return values.where(~unsafe) if fill is None else values.where(~unsafe, fill) |
| 122 | + if isinstance(values, (pd.Series, ExtensionArray)) and fill is None: |
| 123 | + return ( |
| 124 | + values.mask(unsafe) |
| 125 | + if isinstance(values, pd.Series) |
| 126 | + else pd.Series(values, copy=False).mask(unsafe).array |
| 127 | + ) |
| 128 | + positions = np.flatnonzero(unsafe) |
| 129 | + if isinstance(values, (list, tuple)): |
| 130 | + out = np.array(values, dtype=object) |
| 131 | + elif isinstance(values, np.ndarray): |
| 132 | + # str/bytes arrays cannot hold None; pandas parses them as object anyway. |
| 133 | + out = values.astype(object) if fill is None else values.copy() |
| 134 | + elif isinstance(values, (pd.Series, ExtensionArray)): |
| 135 | + if _is_categorical(values) and fill not in values.dtype.categories: |
| 136 | + values = _categories(values).add_categories([fill]) |
| 137 | + out = values.copy() |
| 138 | + else: |
| 139 | + return fill # scalar |
| 140 | + if isinstance(out, pd.Series): |
| 141 | + out.iloc[positions] = fill |
| 142 | + else: |
| 143 | + out[positions] = fill |
| 144 | + return out |
| 145 | + |
| 146 | + |
| 147 | +def _head(values: Any, stop: int) -> Any: |
| 148 | + """The cells before position *stop*, in the same container.""" |
| 149 | + if isinstance(values, pd.Series): |
| 150 | + return values.iloc[:stop] |
| 151 | + if isinstance(values, (str, bytes)): |
| 152 | + return np.array([], dtype=object) |
| 153 | + return values[:stop] |
| 154 | + |
| 155 | + |
| 156 | +def _restore(result: Any, values: Any, cells: np.ndarray, unsafe: np.ndarray) -> Any: |
| 157 | + """Put the original cells back into an ``errors="ignore"`` result.""" |
| 158 | + if isinstance(values, (str, bytes)): |
| 159 | + return values |
| 160 | + positions = np.flatnonzero(unsafe) |
| 161 | + if isinstance(result, pd.Index): |
| 162 | + arr = result.to_numpy(dtype=object, copy=True) |
| 163 | + arr[positions] = cells[positions] |
| 164 | + return pd.Index(arr, dtype=result.dtype, name=result.name) |
| 165 | + out = result.copy() |
| 166 | + if isinstance(out, pd.Series): |
| 167 | + out.iloc[positions] = cells[positions] |
| 168 | + else: |
| 169 | + out[positions] = cells[positions] |
| 170 | + if ( |
| 171 | + _is_categorical(out) |
| 172 | + and _is_categorical(values) |
| 173 | + and _UNPARSEABLE not in values.dtype.categories |
| 174 | + ): |
| 175 | + out = _categories(out).remove_categories([_UNPARSEABLE]) |
| 176 | + return out |
| 177 | + |
| 178 | + |
| 179 | +def safe_to_numeric(values: Any, **kwargs: Any) -> Any: |
| 180 | + """:func:`pandas.to_numeric` that cannot crash on exponent-overflow text. |
| 181 | +
|
| 182 | + Accepts everything ``pd.to_numeric`` does (scalar, list, tuple, 1-D array, |
| 183 | + ``Index`` or ``Series``) and forwards every keyword unchanged |
| 184 | + (``errors=``, ``downcast=``, ``dtype_backend=``). Input with no unsafe cell |
| 185 | + -- including every numeric, boolean or datetime dtype -- goes to pandas |
| 186 | + untouched, so the result is exactly pandas' result (index, name, dtype). |
| 187 | +
|
| 188 | + A cell that starts with scientific notation whose exponent has ten or more |
| 189 | + significant digits (enough to overflow pandas' C int) is treated as |
| 190 | + unparseable text: it becomes missing with ``errors="coerce"``, raises |
| 191 | + pandas' ``Unable to parse string`` ``ValueError`` with ``errors="raise"``, |
| 192 | + and is returned as-is with ``errors="ignore"``. pandas rejects almost all |
| 193 | + such cells too; the exception is a whole-cell negative exponent small |
| 194 | + enough not to overflow, which pandas would underflow to 0.0. Shorter |
| 195 | + exponents, including subnormal and out-of-range ones, are left to pandas. |
| 196 | +
|
| 197 | + The check is cheap: text columns are screened as one joined string, and |
| 198 | + only a column that contains an exponent marker followed by ten digits |
| 199 | + pays for the per-cell check. |
| 200 | + """ |
| 201 | + cells = _parsed_cells(values) |
| 202 | + unsafe = None if cells is None else _unsafe_cells(cells) |
| 203 | + if cells is None or unsafe is None: |
| 204 | + return pd.to_numeric(values, **kwargs) |
| 205 | + errors = kwargs.get("errors", "raise") |
| 206 | + if errors == "coerce": |
| 207 | + return pd.to_numeric(_substitute(values, unsafe, None), **kwargs) |
| 208 | + if errors == "raise": |
| 209 | + first = int(np.argmax(unsafe)) |
| 210 | + # An earlier unparseable cell raises first, exactly as pandas would. |
| 211 | + pd.to_numeric(_head(values, first), **kwargs) |
| 212 | + cell = cells[first] |
| 213 | + text = cell.decode("latin-1") if isinstance(cell, bytes) else cell |
| 214 | + raise ValueError(f'Unable to parse string "{text}" at position {first}') |
| 215 | + # errors="ignore" (pandas itself rejects any other value). |
| 216 | + result = pd.to_numeric(_substitute(values, unsafe, _UNPARSEABLE), **kwargs) |
| 217 | + return _restore(result, values, cells, unsafe) |
| 218 | + |
| 219 | + |
| 220 | +__all__ = ["safe_to_numeric"] |
0 commit comments