55import pandas as pd
66
77from .adapters .polars import from_pandas , to_pandas
8- from .cleaner import Cleaner
8+ from .cleaner import Cleaner , run_pipeline
99from .config import CleanConfig , merge_options
10+ from .domains import SEVERITY_TO_RISK , DomainOutcome , run_domain
1011from .engine .context import build_contexts
1112from .engine .model_select import EngineMode , rank_missing_models
1213from .plan import suggest_plan
@@ -19,6 +20,8 @@ def clean(
1920 * ,
2021 config : CleanConfig | None = None ,
2122 return_report : bool = False ,
23+ domain : str | None = None ,
24+ column_map : dict [str , str ] | None = None ,
2225 ** options : object ,
2326) -> pd .DataFrame | tuple [pd .DataFrame , CleanReport ]:
2427 """Clean a DataFrame and return a new, repaired one.
@@ -54,6 +57,16 @@ def clean(
5457 If True, return ``(cleaned_df, CleanReport)``. The report carries
5558 per-action rationale/risk/confidence, missing counts before/after,
5659 warnings, and recommendations for manual review.
60+ domain:
61+ Optional domain validator pack (e.g. ``"finance"``). When set, generic
62+ cleaning runs first (defaulting to ``strategy="conservative"`` so the
63+ statistical engine never silently alters ledgers/IDs unless you pass an
64+ explicit ``strategy``), then the pack validates in layers and repairs
65+ separately; findings and a ``domain_trust_score`` are folded into the
66+ report. Unknown names raise :class:`~freshdata.domains.UnknownDomainError`.
67+ column_map:
68+ Optional ``{actual_column: canonical_field}`` overrides for the domain
69+ pack's column detection. Requires ``domain`` to be set.
5770 **options:
5871 Any :class:`~freshdata.CleanConfig` field as a keyword override — e.g.
5972 ``strategy`` (``"balanced"`` default / ``"aggressive"`` / ``"conservative"``),
@@ -71,7 +84,15 @@ def clean(
7184
7285 >>> fd.clean(df, outlier_action="flag", target_column="churn",
7386 ... preserve_columns=("notes",), verbose=False)
87+
88+ >>> ledger = fd.clean(df, domain="finance") # validate + repair
89+ >>> ledger, rep = fd.clean(df, domain="finance", return_report=True)
90+ >>> rep.domain_trust_score # 0–1
7491 """
92+ if domain is not None :
93+ return _clean_with_domain (df , domain , column_map , config , return_report , options )
94+ if column_map is not None :
95+ raise TypeError ("column_map requires a domain= to be set" )
7596 cleaner = Cleaner (config = config , ** options )
7697 result = cleaner .clean (df , report = return_report )
7798 if return_report :
@@ -80,6 +101,65 @@ def clean(
80101 return from_pandas (result , df )
81102
82103
104+ def _clean_with_domain (
105+ df : pd .DataFrame ,
106+ domain : str ,
107+ column_map : dict [str , str ] | None ,
108+ config : CleanConfig | None ,
109+ return_report : bool ,
110+ options : dict [str , object ],
111+ ) -> pd .DataFrame | tuple [pd .DataFrame , CleanReport ]:
112+ """Generic clean (conservative by default) then domain validate + repair."""
113+ # With an explicit config the caller owns every setting. Otherwise default to
114+ # a conservative base that does *not* infer dtypes: the domain pack owns
115+ # format validation/coercion (per its audited rules), and generic dtype
116+ # inference would otherwise silently retype dates/amounts before validation.
117+ if config is None :
118+ options = {
119+ "strategy" : "conservative" ,
120+ "fix_dtypes" : False ,
121+ ** options , # explicit caller options win
122+ }
123+ cfg = merge_options (config , ** options )
124+ cleaned , rep = run_pipeline (df , cfg )
125+ repaired , outcome = run_domain (cleaned , domain , column_map = column_map )
126+ _fold_domain_outcome (rep , outcome )
127+ if cfg .verbose :
128+ print (rep .brief ())
129+ out = from_pandas (repaired , df )
130+ return (out , rep ) if return_report else out
131+
132+
133+ def _fold_domain_outcome (rep : CleanReport , outcome : DomainOutcome ) -> None :
134+ """Merge a domain run's findings/repairs into the existing CleanReport."""
135+ report = outcome .report
136+ rep .domain = outcome .domain
137+ rep .domain_trust_score = outcome .trust_score
138+ rep .domain_findings = [r .to_dict () for r in report .results ]
139+ rep .domain_repairs = [a .to_dict () for a in outcome .repairs .actions ]
140+ for result in report .results :
141+ if not result .violated :
142+ continue
143+ col = report .mapping .actual (result .fields [0 ]) if result .fields else None
144+ rep .add (
145+ step = f"domain:{ outcome .domain } :{ result .rule_id } " ,
146+ description = result .message or result .name ,
147+ column = col ,
148+ count = result .n_violations ,
149+ risk = SEVERITY_TO_RISK .get (result .severity , "low" ),
150+ rationale = result .name ,
151+ )
152+ if result .severity == "error" :
153+ rep .add_warning (
154+ f"[{ outcome .domain } ] { result .rule_id } : { result .message or result .name } "
155+ )
156+ applied = sum (1 for a in outcome .repairs .actions if a .status == "applied" )
157+ if applied :
158+ rep .add_recommendation (
159+ f"{ outcome .domain } : { applied } domain repair(s) applied — see domain_repairs"
160+ )
161+
162+
83163def _engine_mode (cfg : CleanConfig ) -> EngineMode :
84164 mode = cfg .engine_mode or "balanced"
85165 return "balanced" if mode == "balanced" else "aggressive"
0 commit comments