Skip to content

Commit df0839a

Browse files
Merge pull request #41 from FreshCode-Org/lean-core
Lean core
2 parents ab9e2c5 + 9682f0e commit df0839a

15 files changed

Lines changed: 1896 additions & 2 deletions

File tree

‎CONTRIBUTING_DOMAINS.md‎

Lines changed: 102 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,102 @@
1+
# Contributing a domain validator pack
2+
3+
A *domain pack* teaches `fd.clean(df, domain="…")` how to validate and repair a
4+
specific kind of tabular data against versioned, reviewable rules. Built-in packs
5+
live in `src/freshdata/domains/<name>/`; third-party packs ship as separate
6+
PyPI distributions and register through an entry point. Both implement the same
7+
interface.
8+
9+
## The interface
10+
11+
Every pack implements `freshdata.domains.DomainValidator`:
12+
13+
```python
14+
class DomainValidator(ABC):
15+
domain_name: str # "finance"
16+
version: str # "0.1.0"
17+
schema_version: str # "2024-01"
18+
19+
def detect_columns(self, df) -> ColumnMapping: ...
20+
def validate(self, df) -> ValidationReport: ...
21+
def repair(self, df, report) -> tuple[DataFrame, RepairLog]: ...
22+
def describe(self) -> dict: ...
23+
```
24+
25+
Most packs subclass **`ConfigDrivenValidator`**, which implements the layered
26+
engine for you — you supply a `rules.yaml`, declare canonical fields, and add any
27+
custom checks. The finance pack (`src/freshdata/domains/finance/`) is the
28+
reference example.
29+
30+
## Anatomy of a pack
31+
32+
```
33+
mypack/
34+
__init__.py # exports your Validator class
35+
validator.py # subclass ConfigDrivenValidator; register custom checks
36+
rules.yaml # the rules (see below)
37+
reference/*.json # bundled code sets, each with a _meta block
38+
```
39+
40+
### `validator.py`
41+
42+
```python
43+
from freshdata.domains import ConfigDrivenValidator
44+
45+
class MyValidator(ConfigDrivenValidator):
46+
domain_name = "mydomain"
47+
version = "0.1.0"
48+
schema_version = "2024-01"
49+
50+
canonical_fields = ("id", "amount", "code")
51+
required_fields = ("id", "amount")
52+
id_fields = ("id",) # never repaired
53+
aliases = {"id": (r"my_?id", r"identifier")} # regex, case-insensitive
54+
rules_path = str(Path(__file__).parent / "rules.yaml")
55+
56+
def register_extensions(self):
57+
self.register_check("my_check", self._my_check) # params.func
58+
self.register_repair("my_fix", self._my_fix) # repair_params.func
59+
60+
def load_reference_values(self, name): # for reference checks
61+
...
62+
```
63+
64+
### `rules.yaml`
65+
66+
Each rule carries `id`, `name`, `layer` (`schema|format|reference|business|
67+
semantic`), `severity` (`error|warning|info`), `field`/`fields`, `check`
68+
(`not_null|required|regex|enum|reference|range|custom`), optional `params`, and
69+
an optional `repair` (`fill_default|coerce|flag_only|reject|none`) with
70+
`repair_params`. Layers always run in order; a rule whose target column is
71+
absent is skipped (the schema layer reports it as `MISSING_REQUIRED_FIELD`).
72+
73+
### Reference data
74+
75+
Bundle code sets as static JSON with a `_meta` block so they stay auditable:
76+
77+
```json
78+
{ "_meta": {"source": "ISO 4217", "retrieved_date": "2024-01-15", "version": "2024-01"},
79+
"codes": ["USD", "EUR", "..."] }
80+
```
81+
82+
## Rules of the road
83+
84+
- **Validation never mutates data.** All changes happen in `repair`, and every
85+
attempt is logged to the `RepairLog` with from/to values, the rule id, and a
86+
status (`applied`/`flagged`/`unresolvable`).
87+
- **Identifier columns are never repaired** — list them in `id_fields`.
88+
- **Never guess column mappings silently** — detection logs how each match was made.
89+
- **No network calls or LLM calls** in a pack.
90+
91+
## Registering a third-party pack
92+
93+
Expose your validator through the `freshdata.domains` entry-point group in your
94+
package's `pyproject.toml`:
95+
96+
```toml
97+
[project.entry-points."freshdata.domains"]
98+
mydomain = "mypack.validator:MyValidator"
99+
```
100+
101+
Once installed, `fd.clean(df, domain="mydomain")` finds it automatically.
102+
Built-in names take precedence, so you cannot shadow `finance` (etc.).

‎pyproject.toml‎

Lines changed: 12 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -70,6 +70,10 @@ semantic = [
7070
cli = [
7171
"pyyaml>=5.1",
7272
]
73+
# Domain validator packs (finance/retail/transport) read their rules from YAML.
74+
domains = [
75+
"pyyaml>=5.1",
76+
]
7377
cleanlab = [
7478
"cleanlab>=2.0",
7579
"scikit-learn>=1.0",
@@ -113,6 +117,12 @@ docs = [
113117
[project.scripts]
114118
freshdata = "freshdata.enterprise.cli:main"
115119

120+
# Built-in domain packs are also exposed here so third-party packs can register
121+
# the same way (the registry resolves built-ins directly and does not require
122+
# this entry to be installed).
123+
[project.entry-points."freshdata.domains"]
124+
finance = "freshdata.domains.finance:FinanceValidator"
125+
116126
[project.urls]
117127
Homepage = "https://freshcode-org.github.io/freshdata/"
118128
Documentation = "https://freshcode-org.github.io/freshdata/"
@@ -193,6 +203,8 @@ ignore = [
193203
"src/freshdata/enterprise/interface.py" = ["PLC0415"]
194204
"src/freshdata/enterprise/cli.py" = ["PLC0415", "PLR0915"]
195205
"src/freshdata/__init__.py" = ["PLC0415"]
206+
# Domain packs defer the PyYAML import so the base infra stays dependency-free.
207+
"src/freshdata/domains/base.py" = ["PLC0415"]
196208

197209
[tool.mypy]
198210
python_version = "3.10"

‎src/freshdata/api.py‎

Lines changed: 81 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -5,8 +5,9 @@
55
import pandas as pd
66

77
from .adapters.polars import from_pandas, to_pandas
8-
from .cleaner import Cleaner
8+
from .cleaner import Cleaner, run_pipeline
99
from .config import CleanConfig, merge_options
10+
from .domains import SEVERITY_TO_RISK, DomainOutcome, run_domain
1011
from .engine.context import build_contexts
1112
from .engine.model_select import EngineMode, rank_missing_models
1213
from .plan import suggest_plan
@@ -19,6 +20,8 @@ def clean(
1920
*,
2021
config: CleanConfig | None = None,
2122
return_report: bool = False,
23+
domain: str | None = None,
24+
column_map: dict[str, str] | None = None,
2225
**options: object,
2326
) -> pd.DataFrame | tuple[pd.DataFrame, CleanReport]:
2427
"""Clean a DataFrame and return a new, repaired one.
@@ -54,6 +57,16 @@ def clean(
5457
If True, return ``(cleaned_df, CleanReport)``. The report carries
5558
per-action rationale/risk/confidence, missing counts before/after,
5659
warnings, and recommendations for manual review.
60+
domain:
61+
Optional domain validator pack (e.g. ``"finance"``). When set, generic
62+
cleaning runs first (defaulting to ``strategy="conservative"`` so the
63+
statistical engine never silently alters ledgers/IDs unless you pass an
64+
explicit ``strategy``), then the pack validates in layers and repairs
65+
separately; findings and a ``domain_trust_score`` are folded into the
66+
report. Unknown names raise :class:`~freshdata.domains.UnknownDomainError`.
67+
column_map:
68+
Optional ``{actual_column: canonical_field}`` overrides for the domain
69+
pack's column detection. Requires ``domain`` to be set.
5770
**options:
5871
Any :class:`~freshdata.CleanConfig` field as a keyword override — e.g.
5972
``strategy`` (``"balanced"`` default / ``"aggressive"`` / ``"conservative"``),
@@ -71,7 +84,15 @@ def clean(
7184
7285
>>> fd.clean(df, outlier_action="flag", target_column="churn",
7386
... preserve_columns=("notes",), verbose=False)
87+
88+
>>> ledger = fd.clean(df, domain="finance") # validate + repair
89+
>>> ledger, rep = fd.clean(df, domain="finance", return_report=True)
90+
>>> rep.domain_trust_score # 0–1
7491
"""
92+
if domain is not None:
93+
return _clean_with_domain(df, domain, column_map, config, return_report, options)
94+
if column_map is not None:
95+
raise TypeError("column_map requires a domain= to be set")
7596
cleaner = Cleaner(config=config, **options)
7697
result = cleaner.clean(df, report=return_report)
7798
if return_report:
@@ -80,6 +101,65 @@ def clean(
80101
return from_pandas(result, df)
81102

82103

104+
def _clean_with_domain(
105+
df: pd.DataFrame,
106+
domain: str,
107+
column_map: dict[str, str] | None,
108+
config: CleanConfig | None,
109+
return_report: bool,
110+
options: dict[str, object],
111+
) -> pd.DataFrame | tuple[pd.DataFrame, CleanReport]:
112+
"""Generic clean (conservative by default) then domain validate + repair."""
113+
# With an explicit config the caller owns every setting. Otherwise default to
114+
# a conservative base that does *not* infer dtypes: the domain pack owns
115+
# format validation/coercion (per its audited rules), and generic dtype
116+
# inference would otherwise silently retype dates/amounts before validation.
117+
if config is None:
118+
options = {
119+
"strategy": "conservative",
120+
"fix_dtypes": False,
121+
**options, # explicit caller options win
122+
}
123+
cfg = merge_options(config, **options)
124+
cleaned, rep = run_pipeline(df, cfg)
125+
repaired, outcome = run_domain(cleaned, domain, column_map=column_map)
126+
_fold_domain_outcome(rep, outcome)
127+
if cfg.verbose:
128+
print(rep.brief())
129+
out = from_pandas(repaired, df)
130+
return (out, rep) if return_report else out
131+
132+
133+
def _fold_domain_outcome(rep: CleanReport, outcome: DomainOutcome) -> None:
134+
"""Merge a domain run's findings/repairs into the existing CleanReport."""
135+
report = outcome.report
136+
rep.domain = outcome.domain
137+
rep.domain_trust_score = outcome.trust_score
138+
rep.domain_findings = [r.to_dict() for r in report.results]
139+
rep.domain_repairs = [a.to_dict() for a in outcome.repairs.actions]
140+
for result in report.results:
141+
if not result.violated:
142+
continue
143+
col = report.mapping.actual(result.fields[0]) if result.fields else None
144+
rep.add(
145+
step=f"domain:{outcome.domain}:{result.rule_id}",
146+
description=result.message or result.name,
147+
column=col,
148+
count=result.n_violations,
149+
risk=SEVERITY_TO_RISK.get(result.severity, "low"),
150+
rationale=result.name,
151+
)
152+
if result.severity == "error":
153+
rep.add_warning(
154+
f"[{outcome.domain}] {result.rule_id}: {result.message or result.name}"
155+
)
156+
applied = sum(1 for a in outcome.repairs.actions if a.status == "applied")
157+
if applied:
158+
rep.add_recommendation(
159+
f"{outcome.domain}: {applied} domain repair(s) applied — see domain_repairs"
160+
)
161+
162+
83163
def _engine_mode(cfg: CleanConfig) -> EngineMode:
84164
mode = cfg.engine_mode or "balanced"
85165
return "balanced" if mode == "balanced" else "aggressive"

‎src/freshdata/domains/__init__.py‎

Lines changed: 100 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,100 @@
1+
"""Domain-specific validator packs for :func:`freshdata.clean`.
2+
3+
A domain pack validates and (separately) repairs a specific kind of tabular data
4+
against versioned, config-driven rules, extending the existing clean audit trail
5+
with domain findings and a trust score. Use it through the normal entry point::
6+
7+
import freshdata as fd
8+
df_out = fd.clean(df, domain="finance")
9+
df_out, report = fd.clean(df, domain="finance", return_report=True)
10+
11+
Third-party packs register via the ``freshdata.domains`` entry-point group; see
12+
``CONTRIBUTING_DOMAINS.md``.
13+
"""
14+
15+
from __future__ import annotations
16+
17+
from collections.abc import Mapping
18+
from dataclasses import dataclass
19+
from typing import Any
20+
21+
import pandas as pd
22+
23+
from .base import (
24+
LAYERS,
25+
MISSING_REQUIRED_FIELD,
26+
SEVERITIES,
27+
SEVERITY_TO_RISK,
28+
ColumnMapping,
29+
ConfigDrivenValidator,
30+
DomainError,
31+
DomainValidator,
32+
RepairAction,
33+
RepairLog,
34+
Rule,
35+
RuleResult,
36+
ValidationReport,
37+
)
38+
from .registry import (
39+
UnknownDomainError,
40+
available,
41+
get_validator,
42+
register,
43+
)
44+
45+
__all__ = [
46+
"LAYERS",
47+
"MISSING_REQUIRED_FIELD",
48+
"SEVERITIES",
49+
"SEVERITY_TO_RISK",
50+
"ColumnMapping",
51+
"ConfigDrivenValidator",
52+
"DomainError",
53+
"DomainOutcome",
54+
"DomainValidator",
55+
"RepairAction",
56+
"RepairLog",
57+
"Rule",
58+
"RuleResult",
59+
"UnknownDomainError",
60+
"ValidationReport",
61+
"available",
62+
"get_validator",
63+
"register",
64+
"run_domain",
65+
]
66+
67+
68+
@dataclass
69+
class DomainOutcome:
70+
"""Everything a domain run produced, beyond the repaired frame."""
71+
72+
description: dict[str, Any]
73+
report: ValidationReport
74+
repairs: RepairLog
75+
76+
@property
77+
def domain(self) -> str:
78+
return self.report.domain
79+
80+
@property
81+
def trust_score(self) -> float:
82+
return self.report.domain_trust_score
83+
84+
85+
def run_domain(
86+
df: pd.DataFrame,
87+
domain: str,
88+
*,
89+
column_map: Mapping[str, str] | None = None,
90+
) -> tuple[pd.DataFrame, DomainOutcome]:
91+
"""Validate then (separately) repair *df* with the named domain pack.
92+
93+
Returns ``(repaired_df, outcome)``. Validation never mutates *df*; repair
94+
runs afterward and never touches identifier columns. Raises
95+
:class:`UnknownDomainError` if *domain* is not registered.
96+
"""
97+
validator = get_validator(domain, column_map=column_map)
98+
report = validator.validate(df)
99+
repaired, repairs = validator.repair(df, report)
100+
return repaired, DomainOutcome(validator.describe(), report, repairs)

0 commit comments

Comments
 (0)