Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -233,6 +233,14 @@ adheres to [Semantic Versioning](https://semver.org/).
- `apply_privacy_policy` and `classify_columns` raise `ValueError` for
duplicate column labels or labels that collide as strings, such as `1` and
`"1"` (#232, part 5).
- A `MaskingRule` column name matches every column with the same name or the
same snake_case form, so `"First Name"` masks `first_name` and `"email"`
masks both `email` and `Email`. Listed names that match no column are
recorded under a new `MaskReport.unmatched_columns` key; they raise
`ValueError` only with `MaskingRule(strict=True)` or
`mask_dataframe(..., strict=True)`. `freshdata clean --mask COLUMN:STRATEGY`
rules are strict, so it exits 1 with a one-line error when `COLUMN` matches
no column, instead of masking nothing and exiting 0 (#251).

### Fixed
- Trust-gate integrations now validate `on_low_score` policies at configuration
Expand Down
61 changes: 56 additions & 5 deletions src/freshdata/enterprise/cleaner.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@

from .._util import _is_stringlike_dtype
from ..adapters.polars import _polars_module, is_polars_frame, to_pandas
from ..steps.columns import snake_case
from .config import ClusterConfig, MaskingRule, SemanticValidatorConfig


Expand Down Expand Up @@ -342,11 +343,53 @@ def _scrub_patterns(rule: MaskingRule) -> list[str]:
return [PII_PATTERNS[name] for name in rule.scrub_patterns] + list(rule.regexes)


def _resolve_columns(rule: MaskingRule, columns: Sequence) -> list:
selected = [c for c in columns if c in set(rule.columns)]
def _resolve_columns(
rule: MaskingRule,
columns: Sequence,
report: MaskReport | None = None,
strict: bool | None = None,
) -> list:
"""Return the columns *rule* selects from *columns*.

A listed column matches a frame column with the same name or the same
snake_case form. Every matching column is selected, so a rule never masks
one of two columns that normalise to the same name and skips the other.
Listed columns that match nothing are recorded on *report*; they raise
``ValueError`` only when *strict* (or ``rule.strict`` when *strict* is None).
"""
col_list = list(columns)
col_by_snake: dict[str, list] = {}
for c in col_list:
col_by_snake.setdefault(snake_case(str(c)), []).append(c)

selected: list = []
unmatched: list = []

for name in rule.columns:
matches = [c for c in col_list if c == name]
key = snake_case(str(name))
if key:
matches += [c for c in col_by_snake.get(key, []) if c not in matches]
if not matches:
unmatched.append(name)
selected += [c for c in matches if c not in selected]

if rule.pattern:
regex = re.compile(rule.pattern)
selected += [c for c in columns if c not in selected and regex.search(str(c))]
selected += [c for c in col_list if c not in selected and regex.search(str(c))]

if report is not None:
for u in unmatched:
if str(u) not in report.unmatched_columns:
report.unmatched_columns.append(str(u))

is_strict = rule.strict if strict is None else strict
if is_strict and unmatched:
missing = ", ".join(repr(c) for c in unmatched)
raise ValueError(
f"Masking rule {rule.name!r} specifies column(s) not found in dataframe: {missing}"
)

return selected


Expand All @@ -363,6 +406,8 @@ class MaskReport:
#: Auditable provenance: one entry per masked column recording *which* rule
#: masked it, with what strategy, under which policy id, and why.
policy_provenance: list[dict[str, Any]] = field(default_factory=list)
#: Columns explicitly specified by rules that did not match any dataframe column.
unmatched_columns: list[str] = field(default_factory=list)

def _record(self, column: str, rule: MaskingRule, n_cells: int) -> None:
self.columns[column] = rule.strategy
Expand Down Expand Up @@ -393,6 +438,7 @@ def to_dict(self) -> dict[str, Any]:
"rules_applied": list(self.rules_applied),
"retention": dict(self.retention),
"policy_provenance": list(self.policy_provenance),
"unmatched_columns": list(self.unmatched_columns),
}

def __repr__(self) -> str:
Expand Down Expand Up @@ -464,7 +510,11 @@ def _apply_mask_pandas(
return out, changed


def mask_dataframe(df: Any, rules: Sequence[MaskingRule]) -> tuple[Any, MaskReport]:
def mask_dataframe(
df: Any,
rules: Sequence[MaskingRule],
strict: bool | None = None,
) -> tuple[Any, MaskReport]:
"""Apply PII masking *rules* to *df*; returns ``(df_same_type, MaskReport)``.

Rules run in order; ``drop`` removes the column so later rules see the new schema.
Expand All @@ -478,7 +528,8 @@ def mask_dataframe(df: Any, rules: Sequence[MaskingRule]) -> tuple[Any, MaskRepo
f"strategy {rule.strategy!r} is a privacy strategy; apply it with "
"freshdata.enterprise.privacy.anonymize (or fd.anonymize), not mask_dataframe"
)
for column in _resolve_columns(rule, _all_columns(out)):
resolved = _resolve_columns(rule, _all_columns(out), report=report, strict=strict)
for column in resolved:
if column not in _all_columns(out):
continue
if is_polars_frame(out):
Expand Down
8 changes: 7 additions & 1 deletion src/freshdata/enterprise/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -258,8 +258,14 @@ def cmd_clean(args: argparse.Namespace) -> int:
extra_masks = []
for spec in args.mask or []:
column, _, strategy = spec.partition(":")
# A column named on the command line must exist, so these rules are strict.
extra_masks.append(
MaskingRule(name=f"cli_{column}", columns=(column,), strategy=strategy or "hash")
MaskingRule(
name=f"cli_{column}",
columns=(column,),
strategy=strategy or "hash",
strict=True,
)
)
masking = tuple(ec.masking) + tuple(extra_masks)

Expand Down
15 changes: 12 additions & 3 deletions src/freshdata/enterprise/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,9 +42,12 @@
class MaskingRule:
"""One PII masking rule applied to a set of columns.

Columns are selected by exact ``columns`` names (post-clean snake_case) and
by ``pattern`` (a regex matched against column *names*). At least one
selector must be given.
Columns are selected by ``columns`` names and by ``pattern`` (a regex
matched against column *names*). At least one selector must be given. A
listed name matches a column with the same name or the same snake_case
form (``"First Name"`` matches ``first_name``), and every column that
matches is masked. Listed names that match nothing are reported, and raise
only when ``strict=True``.

Strategies
----------
Expand Down Expand Up @@ -107,6 +110,11 @@ class MaskingRule:
policy_id: str | None = None
#: Human-readable justification recorded alongside ``policy_id`` (the *why*).
policy_reason: str | None = None
#: When True, raise ``ValueError`` if a listed column matches no column of the
#: frame being masked. The default (False) records it in
#: ``MaskReport.unmatched_columns`` instead, so one rule set can be reused
#: across frames that don't all have every column.
strict: bool = False

def __post_init__(self) -> None:
# ``token`` is an accepted alias for the reversible ``tokenize`` strategy.
Expand Down Expand Up @@ -138,6 +146,7 @@ def __post_init__(self) -> None:
object.__setattr__(self, "entity_types", tuple(self.entity_types))
object.__setattr__(self, "hipaa_tags", tuple(self.hipaa_tags))
object.__setattr__(self, "gdpr_tags", tuple(self.gdpr_tags))
object.__setattr__(self, "strict", bool(self.strict))
# Secure default: an empty salt on a hash rule would make low-entropy
# PII trivially reversible, so generate a random per-rule salt instead.
if self.strategy == "hash" and not self.salt:
Expand Down
4 changes: 3 additions & 1 deletion src/freshdata/enterprise/privacy.py
Original file line number Diff line number Diff line change
Expand Up @@ -1133,7 +1133,9 @@ def anonymize(
f"duplicated: {duplicated}"
)
for rule in rules:
targeted = _resolve_columns(rule, duplicated)
# Only the duplicated labels are searched here, so a listed column
# that is absent from them is expected; never raise for it.
targeted = _resolve_columns(rule, duplicated, strict=False)
if targeted:
raise ValueError(
f"anonymize requires unique column labels; rule {rule.name!r} "
Expand Down
96 changes: 96 additions & 0 deletions tests/test_enterprise_cleaner.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
import pandas as pd
import pytest

import freshdata as fd
from freshdata.adapters.polars import is_polars_frame
from freshdata.enterprise import (
PII_PATTERNS,
Expand Down Expand Up @@ -224,6 +225,101 @@ def test_pii_patterns_present():
assert {"email", "phone", "ssn", "credit_card", "ip", "iban"} <= set(PII_PATTERNS)


def test_masking_snake_case_column_matching():
df = pd.DataFrame({"email": ["a@x.com"], "first_name": ["Alice"]})
out, report = mask_dataframe(
df,
[
MaskingRule(name="m1", columns=("Email",), strategy="hash"),
MaskingRule(
name="m2", columns=("First Name",), strategy="redact", placeholder="[REDACTED]"
),
],
)
assert out["email"].iloc[0] != "a@x.com"
assert out["first_name"].iloc[0] == "[REDACTED]"
assert report.columns["email"] == "hash"
assert report.columns["first_name"] == "redact"
assert report.unmatched_columns == []


def test_masking_missing_column_strict_raises():
df = pd.DataFrame({"email": ["a@x.com"]})
rule = MaskingRule(name="m", columns=("ghost",), strategy="hash", strict=True)
with pytest.raises(ValueError, match=r"not found in dataframe.*'ghost'"):
mask_dataframe(df, [rule])


def test_masking_missing_column_is_recorded_by_default():
df = pd.DataFrame({"email": ["a@x.com"]})
rule = MaskingRule(name="m", columns=("ghost",), strategy="hash")
assert rule.strict is False
out, report = mask_dataframe(df, [rule])
assert out["email"].iloc[0] == "a@x.com"
assert report.unmatched_columns == ["ghost"]
assert report.to_dict()["unmatched_columns"] == ["ghost"]


def test_mask_dataframe_strict_argument_overrides_rule():
df = pd.DataFrame({"email": ["a@x.com"]})
with pytest.raises(ValueError, match="'ghost'"):
mask_dataframe(df, [MaskingRule(name="m", columns=("ghost",))], strict=True)
strict_rule = MaskingRule(name="m", columns=("ghost",), strict=True)
_, report = mask_dataframe(df, [strict_rule], strict=False)
assert report.unmatched_columns == ["ghost"]


def test_masking_masks_every_column_with_the_same_snake_case_name():
df = pd.DataFrame({"E-mail": ["a@x.com"], "e_mail": ["b@y.com"]})
rule = MaskingRule(name="r", columns=("E Mail",), strategy="redact", placeholder="X")
out, report = mask_dataframe(df, [rule])
assert out["E-mail"].iloc[0] == "X"
assert out["e_mail"].iloc[0] == "X"
assert set(report.columns) == {"E-mail", "e_mail"}


def test_masking_exact_name_also_masks_snake_case_twin():
df = pd.DataFrame({"email": ["a@x.com"], "Email": ["b@y.com"]})
rule = MaskingRule(name="r", columns=("email",), strategy="redact", placeholder="X")
out, _ = mask_dataframe(df, [rule])
assert out["email"].iloc[0] == "X"
assert out["Email"].iloc[0] == "X"


def test_anonymize_shared_rule_with_absent_column_does_not_raise():
df = pd.DataFrame({"email": ["a@x.com"]})
rule = MaskingRule(name="r", columns=("email", "ssn"), strategy="redact", placeholder="X")
out, _ = fd.anonymize(df, rules=(rule,))
assert out["email"].iloc[0] == "X"


def test_anonymize_strict_rule_with_duplicated_labels_elsewhere():
df = pd.DataFrame([["a@x.com", 1, 2]], columns=["email", "dup", "dup"])
rule = MaskingRule(
name="r", columns=("email",), strategy="redact", placeholder="X", strict=True
)
out, _ = fd.anonymize(df, rules=(rule,))
assert out["email"].iloc[0] == "X"


def test_masking_polars_snake_case_matching_masks_every_match():
pl = pytest.importorskip("polars")
frame = pl.DataFrame(
{"first_name": ["Alice"], "email": ["a@x.com"], "Email": ["b@y.com"]}
)
rules = [
MaskingRule(name="n", columns=("First Name",), strategy="redact", placeholder="X"),
MaskingRule(name="e", columns=("email",), strategy="redact", placeholder="X"),
]
out, report = mask_dataframe(frame, rules)
assert is_polars_frame(out)
assert out["first_name"][0] == "X"
assert out["email"][0] == "X"
assert out["Email"][0] == "X"
assert report.columns == {"first_name": "redact", "email": "redact", "Email": "redact"}
assert report.unmatched_columns == []


# =====================================================================
# Semantic validation
# =====================================================================
Expand Down
22 changes: 22 additions & 0 deletions tests/test_enterprise_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -192,3 +192,25 @@ def test_unreadable_config_prints_one_line_error_not_traceback(tmp_path, capsys)
code = cli.main(["clean", str(src), "-o", str(tmp_path / "o.csv"), "--config", str(cfg)])
assert code == 1
assert "Traceback" not in capsys.readouterr().err


def test_mask_missing_column_prints_one_line_error(tmp_path, capsys):
src = tmp_path / "in.csv"
pd.DataFrame({"email": ["a@x.com"]}).to_csv(src, index=False)
out = tmp_path / "o.csv"
code = cli.main(["clean", str(src), "-o", str(out), "--mask", "non_existent:hash"])
err = capsys.readouterr().err
assert code == 1
assert "not found in dataframe" in err
assert "non_existent" in err
assert "Traceback" not in err


def test_mask_case_normalization_in_cli(tmp_path):
src = tmp_path / "in.csv"
pd.DataFrame({"email": ["a@x.com"]}).to_csv(src, index=False)
out = tmp_path / "o.csv"
code = cli.main(["clean", str(src), "-o", str(out), "--mask", "Email:hash", "--quiet"])
assert code == 0
assert pd.read_csv(out)["email"].iloc[0] != "a@x.com"

28 changes: 28 additions & 0 deletions tests/test_enterprise_interface.py
Original file line number Diff line number Diff line change
Expand Up @@ -157,3 +157,31 @@ def test_cli_clean_fails_closed_on_anonymization(raw, tmp_path, monkeypatch, cap
assert code != 0
assert "anonymization is not supported" in capsys.readouterr().err
assert not out.exists()


# -- Masking rules naming a missing column (#251) --------------------------


def _missing_column_config(*, strict):
rule = MaskingRule(
name="pii",
columns=("email", "ghost"),
strategy="redact",
placeholder="X",
strict=strict,
)
return EnterpriseConfig(masking=(rule,))


def test_clean_enterprise_strict_masking_rule_raises_for_missing_column(raw):
with pytest.raises(ValueError, match=r"'pii'.*not found in dataframe.*'ghost'"):
clean_enterprise(raw, enterprise=_missing_column_config(strict=True), verbose=False)


def test_clean_enterprise_non_strict_masking_rule_reports_missing_column(raw):
result = clean_enterprise(raw, enterprise=_missing_column_config(strict=False), verbose=False)
assert (result.data["email"] == "X").all()
assert result.privacy_report is None
assert result.mask_report is not None
assert result.mask_report.unmatched_columns == ["ghost"]
assert result.mask_report.to_dict()["unmatched_columns"] == ["ghost"]
Loading