2020if TYPE_CHECKING :
2121 from ..context import ContextPolicy
2222
23+ from .._util import exceeds_float64_exact , fill_na_exact
2324from ..cleaner import run_pipeline
2425from ..config import CleanConfig , merge_options
2526from ..engine .context import infer_role
@@ -117,7 +118,7 @@ def clean_batch(self, batch: Any) -> tuple[pd.DataFrame, CleanReport]:
117118 # In time-series mode the TS processor owns missing-value policy for its
118119 # numeric columns (short-gap interpolation, long-gap preservation), so the
119120 # generic statistical imputer must not pre-fill them.
120- ts_skip = (set ( self ._ts .numeric_targets (cleaned , self ._roles ))
121+ ts_skip = ({ str ( c ) for c in self ._ts .numeric_targets (cleaned , self ._roles )}
121122 if self ._ts is not None else set ())
122123 # Context-protected columns are never touched by the statistical imputer
123124 # either (the representation pass already byte-guards them); leave their
@@ -413,11 +414,13 @@ def _impute(self, df: pd.DataFrame, report: CleanReport,
413414 miss = int (df [col ].isna ().sum ())
414415 if miss == 0 :
415416 continue
416- df = self ._impute_column (df , name , miss , report )
417+ df = self ._impute_column (df , col , name , miss , report )
417418 return df
418419
419- def _impute_column (self , df : pd .DataFrame , col : str , miss : int ,
420+ def _impute_column (self , df : pd .DataFrame , frame_col : object , col : str , miss : int ,
420421 report : CleanReport ) -> pd .DataFrame :
422+ """Fill one column. *frame_col* is the frame's own label (it may be an int);
423+ *col* is its str name, which keys the running state and the report."""
421424 cs = self .state .columns [col ]
422425 role , ratio = cs .role , cs .missing_ratio
423426 band = _band (ratio , self .config )
@@ -440,7 +443,7 @@ def _impute_column(self, df: pd.DataFrame, col: str, miss: int,
440443 return self ._preserve (df , col , miss , report ,
441444 rationale = "datetime without a usable order; fill would "
442445 "invent timestamps" , risk = "medium" )
443- df [col ] = df [col ].ffill ().bfill ()
446+ df [frame_col ] = df [frame_col ].ffill ().bfill ()
444447 return self ._record (df , report , col , miss , "forward/backward fill within batch" ,
445448 rationale = "datetime column with monotonic order" , confidence = 0.8 )
446449
@@ -457,24 +460,30 @@ def _impute_column(self, df: pd.DataFrame, col: str, miss: int,
457460 return self ._preserve (df , col , miss , report ,
458461 rationale = "no running statistic available yet" ,
459462 risk = "medium" , confidence = 0.5 )
460- df [col ] = (df [col ].astype ("float64" ) if df [col ].dtype .kind in "iu"
461- else df [col ]).fillna (value )
462- return self ._record (df , report , col , miss , f"{ label } ({ value :.6g} )" ,
463+ s = df [frame_col ]
464+ note = ""
465+ if exceeds_float64_exact (s ):
466+ # A float64 cast would change present values beyond 2**53: keep the
467+ # integer dtype and fill with the (rounded) running statistic.
468+ df [frame_col ], note = fill_na_exact (s , value )
469+ else :
470+ df [frame_col ] = (s .astype ("float64" ) if s .dtype .kind in "iu" else s ).fillna (value )
471+ return self ._record (df , report , col , miss , f"{ label } ({ value :.6g} { note } )" ,
463472 rationale = rationale , confidence = 0.8 if band == "low" else 0.7 )
464473
465474 # categorical / boolean
466475 mode , mode_ratio = cs .mode (), cs .mode_ratio ()
467476 threshold = 0.5 if band == "low" else 0.6
468477 if mode is not None and mode_ratio is not None and mode_ratio >= threshold :
469- df [col ] = df [col ].fillna (mode )
478+ df [frame_col ] = df [frame_col ].fillna (mode )
470479 return self ._record (df , report , col , miss , f"running mode ({ mode !r} )" ,
471480 rationale = f"dominant category ({ 100 * mode_ratio :.0f} % of seen)" ,
472481 confidence = 0.8 if band == "low" else 0.7 )
473482 sentinel = "Unknown" if band == "low" else "Missing"
474- s = df [col ]
483+ s = df [frame_col ]
475484 if isinstance (s .dtype , pd .CategoricalDtype ) and sentinel not in s .cat .categories :
476485 s = s .cat .add_categories ([sentinel ])
477- df [col ] = s .fillna (sentinel )
486+ df [frame_col ] = s .fillna (sentinel )
478487 return self ._record (df , report , col , miss , f'sentinel "{ sentinel } "' ,
479488 rationale = "no dominant category; sentinel keeps the gap visible" ,
480489 confidence = 0.7 , risk = "low" if band == "low" else "medium" )
0 commit comments