From a5fcad1f0b383198d5f2ce617c6f213b1d3e834d Mon Sep 17 00:00:00 2001 From: Voyagerroc-Code <325343927+Voyagerroc-Code@users.noreply.github.com> Date: Mon, 14 Sep 2026 23:29:13 +0300 Subject: [PATCH 1/3] fix(masking): validate unmatched columns and support snake_case matching --- src/freshdata/enterprise/cleaner.py | 54 ++++++++++++++++++++++++++--- src/freshdata/enterprise/config.py | 4 +++ tests/test_enterprise_cleaner.py | 32 +++++++++++++++++ tests/test_enterprise_cli.py | 21 +++++++++++ 4 files changed, 106 insertions(+), 5 deletions(-) diff --git a/src/freshdata/enterprise/cleaner.py b/src/freshdata/enterprise/cleaner.py index f8dd9762..77588982 100644 --- a/src/freshdata/enterprise/cleaner.py +++ b/src/freshdata/enterprise/cleaner.py @@ -32,6 +32,7 @@ from .._util import _is_stringlike_dtype from ..adapters.polars import _polars_module, is_polars_frame, to_pandas +from ..steps.columns import snake_case from .config import ClusterConfig, MaskingRule, SemanticValidatorConfig @@ -342,11 +343,46 @@ def _scrub_patterns(rule: MaskingRule) -> list[str]: return [PII_PATTERNS[name] for name in rule.scrub_patterns] + list(rule.regexes) -def _resolve_columns(rule: MaskingRule, columns: Sequence) -> list: - selected = [c for c in columns if c in set(rule.columns)] +def _resolve_columns( + rule: MaskingRule, + columns: Sequence, + report: MaskReport | None = None, + strict: bool | None = None, +) -> list: + col_list = list(columns) + col_set = set(col_list) + col_by_snake = {snake_case(str(c)): c for c in col_list} + + selected: list = [] + unmatched: list = [] + + for c in rule.columns: + if c in col_set: + if c not in selected: + selected.append(c) + elif snake_case(str(c)) in col_by_snake: + matched = col_by_snake[snake_case(str(c))] + if matched not in selected: + selected.append(matched) + else: + unmatched.append(c) + if rule.pattern: regex = re.compile(rule.pattern) - selected += [c for c in columns if c not in selected and regex.search(str(c))] + selected += [c for c in col_list if c not in selected and regex.search(str(c))] + + if report is not None and unmatched: + for u in unmatched: + if str(u) not in report.unmatched_columns: + report.unmatched_columns.append(str(u)) + + is_strict = strict if strict is not None else getattr(rule, "strict", True) + if is_strict and unmatched: + missing = ", ".join(repr(c) for c in unmatched) + raise ValueError( + f"Masking rule {rule.name!r} specifies column(s) not found in dataframe: {missing}" + ) + return selected @@ -363,6 +399,8 @@ class MaskReport: #: Auditable provenance: one entry per masked column recording *which* rule #: masked it, with what strategy, under which policy id, and why. policy_provenance: list[dict[str, Any]] = field(default_factory=list) + #: Columns explicitly specified by rules that did not match any dataframe column. + unmatched_columns: list[str] = field(default_factory=list) def _record(self, column: str, rule: MaskingRule, n_cells: int) -> None: self.columns[column] = rule.strategy @@ -393,6 +431,7 @@ def to_dict(self) -> dict[str, Any]: "rules_applied": list(self.rules_applied), "retention": dict(self.retention), "policy_provenance": list(self.policy_provenance), + "unmatched_columns": list(self.unmatched_columns), } def __repr__(self) -> str: @@ -464,7 +503,11 @@ def _apply_mask_pandas( return out, changed -def mask_dataframe(df: Any, rules: Sequence[MaskingRule]) -> tuple[Any, MaskReport]: +def mask_dataframe( + df: Any, + rules: Sequence[MaskingRule], + strict: bool | None = None, +) -> tuple[Any, MaskReport]: """Apply PII masking *rules* to *df*; returns ``(df_same_type, MaskReport)``. Rules run in order; ``drop`` removes the column so later rules see the new schema. @@ -478,7 +521,8 @@ def mask_dataframe(df: Any, rules: Sequence[MaskingRule]) -> tuple[Any, MaskRepo f"strategy {rule.strategy!r} is a privacy strategy; apply it with " "freshdata.enterprise.privacy.anonymize (or fd.anonymize), not mask_dataframe" ) - for column in _resolve_columns(rule, _all_columns(out)): + resolved = _resolve_columns(rule, _all_columns(out), report=report, strict=strict) + for column in resolved: if column not in _all_columns(out): continue if is_polars_frame(out): diff --git a/src/freshdata/enterprise/config.py b/src/freshdata/enterprise/config.py index 4d9762c8..f932b832 100644 --- a/src/freshdata/enterprise/config.py +++ b/src/freshdata/enterprise/config.py @@ -107,6 +107,9 @@ class MaskingRule: policy_id: str | None = None #: Human-readable justification recorded alongside ``policy_id`` (the *why*). policy_reason: str | None = None + #: When True (default), raise ValueError if explicitly listed columns are + #: not present in the dataframe being masked. + strict: bool = True def __post_init__(self) -> None: # ``token`` is an accepted alias for the reversible ``tokenize`` strategy. @@ -138,6 +141,7 @@ def __post_init__(self) -> None: object.__setattr__(self, "entity_types", tuple(self.entity_types)) object.__setattr__(self, "hipaa_tags", tuple(self.hipaa_tags)) object.__setattr__(self, "gdpr_tags", tuple(self.gdpr_tags)) + object.__setattr__(self, "strict", bool(self.strict)) # Secure default: an empty salt on a hash rule would make low-entropy # PII trivially reversible, so generate a random per-rule salt instead. if self.strategy == "hash" and not self.salt: diff --git a/tests/test_enterprise_cleaner.py b/tests/test_enterprise_cleaner.py index 8f955b49..1eedb6f8 100644 --- a/tests/test_enterprise_cleaner.py +++ b/tests/test_enterprise_cleaner.py @@ -224,6 +224,38 @@ def test_pii_patterns_present(): assert {"email", "phone", "ssn", "credit_card", "ip", "iban"} <= set(PII_PATTERNS) +def test_masking_snake_case_column_matching(): + df = pd.DataFrame({"email": ["a@x.com"], "first_name": ["Alice"]}) + out, report = mask_dataframe( + df, + [ + MaskingRule(name="m1", columns=("Email",), strategy="hash"), + MaskingRule(name="m2", columns=("First Name",), strategy="redact", placeholder="[REDACTED]"), + ], + ) + assert out["email"].iloc[0] != "a@x.com" + assert out["first_name"].iloc[0] == "[REDACTED]" + assert report.columns["email"] == "hash" + assert report.columns["first_name"] == "redact" + assert report.unmatched_columns == [] + + +def test_masking_missing_column_strict_raises(): + df = pd.DataFrame({"email": ["a@x.com"]}) + with pytest.raises(ValueError, match="specifies column\\(s\\) not found in dataframe.*'ghost'"): + mask_dataframe(df, [MaskingRule(name="m", columns=("ghost",), strategy="hash")]) + + +def test_masking_missing_column_non_strict_records_unmatched(): + df = pd.DataFrame({"email": ["a@x.com"]}) + rule = MaskingRule(name="m", columns=("ghost",), strategy="hash", strict=False) + out, report = mask_dataframe(df, [rule]) + assert out["email"].iloc[0] == "a@x.com" + assert report.unmatched_columns == ["ghost"] + assert report.to_dict()["unmatched_columns"] == ["ghost"] + + + # ===================================================================== # Semantic validation # ===================================================================== diff --git a/tests/test_enterprise_cli.py b/tests/test_enterprise_cli.py index c8807cff..6b983de6 100644 --- a/tests/test_enterprise_cli.py +++ b/tests/test_enterprise_cli.py @@ -192,3 +192,24 @@ def test_unreadable_config_prints_one_line_error_not_traceback(tmp_path, capsys) code = cli.main(["clean", str(src), "-o", str(tmp_path / "o.csv"), "--config", str(cfg)]) assert code == 1 assert "Traceback" not in capsys.readouterr().err + + +def test_mask_missing_column_prints_one_line_error(tmp_path, capsys): + src = tmp_path / "in.csv" + pd.DataFrame({"email": ["a@x.com"]}).to_csv(src, index=False) + code = cli.main(["clean", str(src), "-o", str(tmp_path / "o.csv"), "--mask", "non_existent:hash"]) + err = capsys.readouterr().err + assert code == 1 + assert "not found in dataframe" in err + assert "non_existent" in err + assert "Traceback" not in err + + +def test_mask_case_normalization_in_cli(tmp_path): + src = tmp_path / "in.csv" + pd.DataFrame({"email": ["a@x.com"]}).to_csv(src, index=False) + out = tmp_path / "o.csv" + code = cli.main(["clean", str(src), "-o", str(out), "--mask", "Email:hash", "--quiet"]) + assert code == 0 + assert pd.read_csv(out)["email"].iloc[0] != "a@x.com" + From 30421d50e1cca8b557b3f4ffd2e54559b3f0402b Mon Sep 17 00:00:00 2001 From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com> Date: Tue, 15 Sep 2026 16:38:29 +0530 Subject: [PATCH 2/3] fix(masking): non-strict by default, mask every snake_case match, strict CLI --mask - MaskingRule.strict defaults to False: listed columns that match nothing are recorded in MaskReport.unmatched_columns, so shared rule sets keep working in fd.anonymize and clean_enterprise. The CLI --mask rules are strict. - A listed column masks every frame column with the same name or snake_case form, instead of only the last one that normalises to it. - The duplicate-label pre-check in anonymize never raises for absent columns. - Split the three E501 test lines; add tests, docstrings and a changelog entry. --- CHANGELOG.md | 6 +++ src/freshdata/enterprise/cleaner.py | 35 +++++++++++------- src/freshdata/enterprise/cli.py | 8 +++- src/freshdata/enterprise/config.py | 17 ++++++--- src/freshdata/enterprise/privacy.py | 4 +- tests/test_enterprise_cleaner.py | 57 ++++++++++++++++++++++++++--- tests/test_enterprise_cli.py | 3 +- 7 files changed, 102 insertions(+), 28 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 35e927f9..673c6321 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -235,6 +235,12 @@ adheres to [Semantic Versioning](https://semver.org/). `"1"` (#232, part 5). ### Fixed +- `freshdata clean --mask COLUMN:STRATEGY` exits 1 with a one-line error when + `COLUMN` matches no column, instead of silently masking nothing. Masking rule + columns also match by snake_case name (`"First Name"` matches `first_name`), + and every matching column is masked. Listed columns that match nothing are + recorded in `MaskReport.unmatched_columns`; set `MaskingRule(strict=True)` + (or `mask_dataframe(..., strict=True)`) to raise instead (#251). - Trust-gate integrations now validate `on_low_score` policies at configuration boundaries, rejecting typos instead of silently skipping failure handling (#345). - The minimum supported numpy is now 1.22. The numpy 1.21.6 wheel bundles an diff --git a/src/freshdata/enterprise/cleaner.py b/src/freshdata/enterprise/cleaner.py index 77588982..030dedae 100644 --- a/src/freshdata/enterprise/cleaner.py +++ b/src/freshdata/enterprise/cleaner.py @@ -349,34 +349,41 @@ def _resolve_columns( report: MaskReport | None = None, strict: bool | None = None, ) -> list: + """Return the columns *rule* selects from *columns*. + + A listed column matches a frame column with the same name or the same + snake_case form. Every matching column is selected, so a rule never masks + one of two columns that normalise to the same name and skips the other. + Listed columns that match nothing are recorded on *report*; they raise + ``ValueError`` only when *strict* (or ``rule.strict`` when *strict* is None). + """ col_list = list(columns) - col_set = set(col_list) - col_by_snake = {snake_case(str(c)): c for c in col_list} + col_by_snake: dict[str, list] = {} + for c in col_list: + col_by_snake.setdefault(snake_case(str(c)), []).append(c) selected: list = [] unmatched: list = [] - for c in rule.columns: - if c in col_set: - if c not in selected: - selected.append(c) - elif snake_case(str(c)) in col_by_snake: - matched = col_by_snake[snake_case(str(c))] - if matched not in selected: - selected.append(matched) - else: - unmatched.append(c) + for name in rule.columns: + matches = [c for c in col_list if c == name] + key = snake_case(str(name)) + if key: + matches += [c for c in col_by_snake.get(key, []) if c not in matches] + if not matches: + unmatched.append(name) + selected += [c for c in matches if c not in selected] if rule.pattern: regex = re.compile(rule.pattern) selected += [c for c in col_list if c not in selected and regex.search(str(c))] - if report is not None and unmatched: + if report is not None: for u in unmatched: if str(u) not in report.unmatched_columns: report.unmatched_columns.append(str(u)) - is_strict = strict if strict is not None else getattr(rule, "strict", True) + is_strict = rule.strict if strict is None else strict if is_strict and unmatched: missing = ", ".join(repr(c) for c in unmatched) raise ValueError( diff --git a/src/freshdata/enterprise/cli.py b/src/freshdata/enterprise/cli.py index fcf9cbd2..c7a4ae03 100644 --- a/src/freshdata/enterprise/cli.py +++ b/src/freshdata/enterprise/cli.py @@ -258,8 +258,14 @@ def cmd_clean(args: argparse.Namespace) -> int: extra_masks = [] for spec in args.mask or []: column, _, strategy = spec.partition(":") + # A column named on the command line must exist, so these rules are strict. extra_masks.append( - MaskingRule(name=f"cli_{column}", columns=(column,), strategy=strategy or "hash") + MaskingRule( + name=f"cli_{column}", + columns=(column,), + strategy=strategy or "hash", + strict=True, + ) ) masking = tuple(ec.masking) + tuple(extra_masks) diff --git a/src/freshdata/enterprise/config.py b/src/freshdata/enterprise/config.py index f932b832..84b111ba 100644 --- a/src/freshdata/enterprise/config.py +++ b/src/freshdata/enterprise/config.py @@ -42,9 +42,12 @@ class MaskingRule: """One PII masking rule applied to a set of columns. - Columns are selected by exact ``columns`` names (post-clean snake_case) and - by ``pattern`` (a regex matched against column *names*). At least one - selector must be given. + Columns are selected by ``columns`` names and by ``pattern`` (a regex + matched against column *names*). At least one selector must be given. A + listed name matches a column with the same name or the same snake_case + form (``"First Name"`` matches ``first_name``), and every column that + matches is masked. Listed names that match nothing are reported, and raise + only when ``strict=True``. Strategies ---------- @@ -107,9 +110,11 @@ class MaskingRule: policy_id: str | None = None #: Human-readable justification recorded alongside ``policy_id`` (the *why*). policy_reason: str | None = None - #: When True (default), raise ValueError if explicitly listed columns are - #: not present in the dataframe being masked. - strict: bool = True + #: When True, raise ``ValueError`` if a listed column matches no column of the + #: frame being masked. The default (False) records it in + #: ``MaskReport.unmatched_columns`` instead, so one rule set can be reused + #: across frames that don't all have every column. + strict: bool = False def __post_init__(self) -> None: # ``token`` is an accepted alias for the reversible ``tokenize`` strategy. diff --git a/src/freshdata/enterprise/privacy.py b/src/freshdata/enterprise/privacy.py index 3ebcd523..84dc740f 100644 --- a/src/freshdata/enterprise/privacy.py +++ b/src/freshdata/enterprise/privacy.py @@ -1133,7 +1133,9 @@ def anonymize( f"duplicated: {duplicated}" ) for rule in rules: - targeted = _resolve_columns(rule, duplicated) + # Only the duplicated labels are searched here, so a listed column + # that is absent from them is expected; never raise for it. + targeted = _resolve_columns(rule, duplicated, strict=False) if targeted: raise ValueError( f"anonymize requires unique column labels; rule {rule.name!r} " diff --git a/tests/test_enterprise_cleaner.py b/tests/test_enterprise_cleaner.py index 1eedb6f8..222ce9df 100644 --- a/tests/test_enterprise_cleaner.py +++ b/tests/test_enterprise_cleaner.py @@ -3,6 +3,7 @@ import pandas as pd import pytest +import freshdata as fd from freshdata.adapters.polars import is_polars_frame from freshdata.enterprise import ( PII_PATTERNS, @@ -230,7 +231,9 @@ def test_masking_snake_case_column_matching(): df, [ MaskingRule(name="m1", columns=("Email",), strategy="hash"), - MaskingRule(name="m2", columns=("First Name",), strategy="redact", placeholder="[REDACTED]"), + MaskingRule( + name="m2", columns=("First Name",), strategy="redact", placeholder="[REDACTED]" + ), ], ) assert out["email"].iloc[0] != "a@x.com" @@ -242,19 +245,63 @@ def test_masking_snake_case_column_matching(): def test_masking_missing_column_strict_raises(): df = pd.DataFrame({"email": ["a@x.com"]}) - with pytest.raises(ValueError, match="specifies column\\(s\\) not found in dataframe.*'ghost'"): - mask_dataframe(df, [MaskingRule(name="m", columns=("ghost",), strategy="hash")]) + rule = MaskingRule(name="m", columns=("ghost",), strategy="hash", strict=True) + with pytest.raises(ValueError, match=r"not found in dataframe.*'ghost'"): + mask_dataframe(df, [rule]) -def test_masking_missing_column_non_strict_records_unmatched(): +def test_masking_missing_column_is_recorded_by_default(): df = pd.DataFrame({"email": ["a@x.com"]}) - rule = MaskingRule(name="m", columns=("ghost",), strategy="hash", strict=False) + rule = MaskingRule(name="m", columns=("ghost",), strategy="hash") + assert rule.strict is False out, report = mask_dataframe(df, [rule]) assert out["email"].iloc[0] == "a@x.com" assert report.unmatched_columns == ["ghost"] assert report.to_dict()["unmatched_columns"] == ["ghost"] +def test_mask_dataframe_strict_argument_overrides_rule(): + df = pd.DataFrame({"email": ["a@x.com"]}) + with pytest.raises(ValueError, match="'ghost'"): + mask_dataframe(df, [MaskingRule(name="m", columns=("ghost",))], strict=True) + strict_rule = MaskingRule(name="m", columns=("ghost",), strict=True) + _, report = mask_dataframe(df, [strict_rule], strict=False) + assert report.unmatched_columns == ["ghost"] + + +def test_masking_masks_every_column_with_the_same_snake_case_name(): + df = pd.DataFrame({"E-mail": ["a@x.com"], "e_mail": ["b@y.com"]}) + rule = MaskingRule(name="r", columns=("E Mail",), strategy="redact", placeholder="X") + out, report = mask_dataframe(df, [rule]) + assert out["E-mail"].iloc[0] == "X" + assert out["e_mail"].iloc[0] == "X" + assert set(report.columns) == {"E-mail", "e_mail"} + + +def test_masking_exact_name_also_masks_snake_case_twin(): + df = pd.DataFrame({"email": ["a@x.com"], "Email": ["b@y.com"]}) + rule = MaskingRule(name="r", columns=("email",), strategy="redact", placeholder="X") + out, _ = mask_dataframe(df, [rule]) + assert out["email"].iloc[0] == "X" + assert out["Email"].iloc[0] == "X" + + +def test_anonymize_shared_rule_with_absent_column_does_not_raise(): + df = pd.DataFrame({"email": ["a@x.com"]}) + rule = MaskingRule(name="r", columns=("email", "ssn"), strategy="redact", placeholder="X") + out, _ = fd.anonymize(df, rules=(rule,)) + assert out["email"].iloc[0] == "X" + + +def test_anonymize_strict_rule_with_duplicated_labels_elsewhere(): + df = pd.DataFrame([["a@x.com", 1, 2]], columns=["email", "dup", "dup"]) + rule = MaskingRule( + name="r", columns=("email",), strategy="redact", placeholder="X", strict=True + ) + out, _ = fd.anonymize(df, rules=(rule,)) + assert out["email"].iloc[0] == "X" + + # ===================================================================== # Semantic validation diff --git a/tests/test_enterprise_cli.py b/tests/test_enterprise_cli.py index 6b983de6..d3d36be8 100644 --- a/tests/test_enterprise_cli.py +++ b/tests/test_enterprise_cli.py @@ -197,7 +197,8 @@ def test_unreadable_config_prints_one_line_error_not_traceback(tmp_path, capsys) def test_mask_missing_column_prints_one_line_error(tmp_path, capsys): src = tmp_path / "in.csv" pd.DataFrame({"email": ["a@x.com"]}).to_csv(src, index=False) - code = cli.main(["clean", str(src), "-o", str(tmp_path / "o.csv"), "--mask", "non_existent:hash"]) + out = tmp_path / "o.csv" + code = cli.main(["clean", str(src), "-o", str(out), "--mask", "non_existent:hash"]) err = capsys.readouterr().err assert code == 1 assert "not found in dataframe" in err From c26b60911d1905076a60f52aa0ec700d196de852 Mon Sep 17 00:00:00 2001 From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com> Date: Tue, 15 Sep 2026 17:10:08 +0530 Subject: [PATCH 3/3] test(masking): cover polars snake_case matching and strict rules in clean_enterprise Move the #251 changelog entry to Changed, since it changes matching, adds the unmatched_columns report key and makes CLI --mask exit 1. --- CHANGELOG.md | 14 ++++++++------ tests/test_enterprise_cleaner.py | 17 +++++++++++++++++ tests/test_enterprise_interface.py | 28 ++++++++++++++++++++++++++++ 3 files changed, 53 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 673c6321..2c02b7df 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -233,14 +233,16 @@ adheres to [Semantic Versioning](https://semver.org/). - `apply_privacy_policy` and `classify_columns` raise `ValueError` for duplicate column labels or labels that collide as strings, such as `1` and `"1"` (#232, part 5). +- A `MaskingRule` column name matches every column with the same name or the + same snake_case form, so `"First Name"` masks `first_name` and `"email"` + masks both `email` and `Email`. Listed names that match no column are + recorded under a new `MaskReport.unmatched_columns` key; they raise + `ValueError` only with `MaskingRule(strict=True)` or + `mask_dataframe(..., strict=True)`. `freshdata clean --mask COLUMN:STRATEGY` + rules are strict, so it exits 1 with a one-line error when `COLUMN` matches + no column, instead of masking nothing and exiting 0 (#251). ### Fixed -- `freshdata clean --mask COLUMN:STRATEGY` exits 1 with a one-line error when - `COLUMN` matches no column, instead of silently masking nothing. Masking rule - columns also match by snake_case name (`"First Name"` matches `first_name`), - and every matching column is masked. Listed columns that match nothing are - recorded in `MaskReport.unmatched_columns`; set `MaskingRule(strict=True)` - (or `mask_dataframe(..., strict=True)`) to raise instead (#251). - Trust-gate integrations now validate `on_low_score` policies at configuration boundaries, rejecting typos instead of silently skipping failure handling (#345). - The minimum supported numpy is now 1.22. The numpy 1.21.6 wheel bundles an diff --git a/tests/test_enterprise_cleaner.py b/tests/test_enterprise_cleaner.py index 222ce9df..50f5dfa2 100644 --- a/tests/test_enterprise_cleaner.py +++ b/tests/test_enterprise_cleaner.py @@ -302,6 +302,23 @@ def test_anonymize_strict_rule_with_duplicated_labels_elsewhere(): assert out["email"].iloc[0] == "X" +def test_masking_polars_snake_case_matching_masks_every_match(): + pl = pytest.importorskip("polars") + frame = pl.DataFrame( + {"first_name": ["Alice"], "email": ["a@x.com"], "Email": ["b@y.com"]} + ) + rules = [ + MaskingRule(name="n", columns=("First Name",), strategy="redact", placeholder="X"), + MaskingRule(name="e", columns=("email",), strategy="redact", placeholder="X"), + ] + out, report = mask_dataframe(frame, rules) + assert is_polars_frame(out) + assert out["first_name"][0] == "X" + assert out["email"][0] == "X" + assert out["Email"][0] == "X" + assert report.columns == {"first_name": "redact", "email": "redact", "Email": "redact"} + assert report.unmatched_columns == [] + # ===================================================================== # Semantic validation diff --git a/tests/test_enterprise_interface.py b/tests/test_enterprise_interface.py index 7e9251e9..63258a6e 100644 --- a/tests/test_enterprise_interface.py +++ b/tests/test_enterprise_interface.py @@ -157,3 +157,31 @@ def test_cli_clean_fails_closed_on_anonymization(raw, tmp_path, monkeypatch, cap assert code != 0 assert "anonymization is not supported" in capsys.readouterr().err assert not out.exists() + + +# -- Masking rules naming a missing column (#251) -------------------------- + + +def _missing_column_config(*, strict): + rule = MaskingRule( + name="pii", + columns=("email", "ghost"), + strategy="redact", + placeholder="X", + strict=strict, + ) + return EnterpriseConfig(masking=(rule,)) + + +def test_clean_enterprise_strict_masking_rule_raises_for_missing_column(raw): + with pytest.raises(ValueError, match=r"'pii'.*not found in dataframe.*'ghost'"): + clean_enterprise(raw, enterprise=_missing_column_config(strict=True), verbose=False) + + +def test_clean_enterprise_non_strict_masking_rule_reports_missing_column(raw): + result = clean_enterprise(raw, enterprise=_missing_column_config(strict=False), verbose=False) + assert (result.data["email"] == "X").all() + assert result.privacy_report is None + assert result.mask_report is not None + assert result.mask_report.unmatched_columns == ["ghost"] + assert result.mask_report.to_dict()["unmatched_columns"] == ["ghost"]