|
42 | 42 | r"^[+-]?(?:\d+(?:\.\d*)?|\.\d+)[eE]([+-]?\d+)$" |
43 | 43 | ) |
44 | 44 | _MAX_SAFE_EXPONENT = 308 |
| 45 | +# Column-level screen for the guard in _to_numeric_or_none: any string that |
| 46 | +# pandas could parse as scientific notation with an exponent past the float |
| 47 | +# range must contain an exponent marker followed by at least three digits |
| 48 | +# (309 is the smallest unsafe magnitude). A false positive merely routes the |
| 49 | +# column to the exact per-cell check. |
| 50 | +_RISKY_EXPONENT = re.compile(r"[eE][+-]?\d{3,}") |
45 | 51 |
|
46 | 52 |
|
47 | 53 | def _number_format( |
@@ -153,22 +159,19 @@ def _to_numeric_or_none(values: pd.Series) -> pd.Series | None: |
153 | 159 | if pd.api.types.is_object_dtype(values.dtype) or pd.api.types.is_string_dtype( |
154 | 160 | values.dtype |
155 | 161 | ): |
156 | | - # Only strings containing an exponent marker can match the unsafe |
157 | | - # pattern, so find candidates with one vectorized pass and run the |
158 | | - # per-value regex on that (normally empty) subset only. ``.str`` |
159 | | - # refuses object columns that contain no strings at all — such a |
160 | | - # column has no unsafe tokens either, so treat it as candidate-free. |
| 162 | + # Screen the whole column as one joined blob first: a single C-level |
| 163 | + # join plus one regex scan, no per-cell Python work in the common |
| 164 | + # (safe) case. The join raises TypeError when non-string, non-missing |
| 165 | + # objects are present — treat such columns as risky and let the exact |
| 166 | + # per-cell predicate decide. |
161 | 167 | try: |
162 | | - candidates = values.str.contains("e", case=False, regex=False, na=False) |
163 | | - except (AttributeError, TypeError): |
164 | | - candidates = None |
165 | | - if candidates is not None: |
166 | | - if candidates.dtype != bool: |
167 | | - candidates = candidates.fillna(False).astype(bool) |
168 | | - if bool(candidates.any()): |
169 | | - unsafe = values[candidates].map(_has_unsafe_scientific_exponent) |
170 | | - if bool(unsafe.any()): |
171 | | - values = values.mask(unsafe.reindex(values.index, fill_value=False)) |
| 168 | + blob = "\x1f".join(values.dropna().to_numpy()) |
| 169 | + except TypeError: |
| 170 | + blob = None |
| 171 | + if blob is None or _RISKY_EXPONENT.search(blob) is not None: |
| 172 | + unsafe = values.map(_has_unsafe_scientific_exponent) |
| 173 | + if bool(unsafe.any()): |
| 174 | + values = values.mask(unsafe) |
172 | 175 | try: |
173 | 176 | return pd.to_numeric(values, errors="coerce") |
174 | 177 | except (TypeError, ValueError): |
|
0 commit comments