")``,
+ so equal values in different columns get different tokens. By default
+ (``None``) a random per-run key is used and tokens differ on every
+ run. Anyone holding the salt can confirm guesses of low-cardinality
+ values, so keep it secret; it is never written to the report, and
+ ``audit["mask_salt_source"]`` records only ``"caller"`` or
+ ``"per-run-random"``.
Column names in ``context_policy``, ``allow_unmasked_columns`` and
``sensitive_columns`` match a column's label or ``str(label)``, so
@@ -911,16 +948,15 @@ def analyze_dataset(
"""
if privacy not in _PRIVACY_MODES:
raise ValueError(f"privacy must be one of {_PRIVACY_MODES}, got {privacy!r}")
+ if mask_salt is not None and not isinstance(mask_salt, str):
+ raise TypeError(f"mask_salt must be a str or None, got {type(mask_salt).__name__}")
+ if mask_salt == "":
+ raise ValueError("mask_salt must be a non-empty str, or None for a per-run key")
frame = to_pandas(df)
labels = list(frame.columns)
_require_unique_label_strings(labels)
- unknown = [str(c) for c in allow_unmasked_columns if _label_position(labels, c) is None]
- if unknown:
- raise ValueError(f"allow_unmasked_columns contains unknown column(s): {unknown}")
- sensitive_positions = [_label_position(labels, c) for c in sensitive_columns]
- unknown = [str(c) for c, p in zip(sensitive_columns, sensitive_positions) if p is None]
- if unknown:
- raise ValueError(f"sensitive_columns contains unknown column(s): {unknown}")
+ _resolve_positions(labels, allow_unmasked_columns, "allow_unmasked_columns")
+ sensitive_positions = _resolve_positions(labels, sensitive_columns, "sensitive_columns")
intent = _parse_context_policy(context_policy)
prof = _profile_frame(frame)
@@ -945,7 +981,7 @@ def analyze_dataset(
# Declared-sensitive columns always join the mask set: pattern-based PII
# detection cannot recognise every sensitive token (an internal case ID,
# a synthetic SSN), so the caller's declaration is authoritative.
- declared_sensitive = [str(labels[p]) for p in sensitive_positions if p is not None]
+ declared_sensitive = [str(labels[p]) for p in sensitive_positions]
mask_for_code = sorted(
dict.fromkeys([*intent.mask_columns, *pii_columns, *declared_sensitive])
)
@@ -980,7 +1016,10 @@ def analyze_dataset(
if privacy == "mask_pii_before_reasoning" and sample_rows > 0:
sample_positions = _sample_mask_positions(frame, mask_for_code, allow_unmasked_columns)
sample_mask = sorted(str(labels[p]) for p in sample_positions)
- model_context["sample_rows_masked"] = _mask_sample(frame, sample_positions, sample_rows)
+ salt_key = mask_salt.encode("utf-8") if mask_salt is not None else None
+ model_context["sample_rows_masked"] = _mask_sample(
+ frame, sample_positions, sample_rows, salt_key
+ )
# --- optional provider hook (experimental) ----------------------------------
engine = "deterministic-local"
@@ -1020,6 +1059,7 @@ def analyze_dataset(
"pii_suppressed_date_like": found.pii_suppressed,
"masked_columns": mask_for_code,
"sample_masked_columns": sample_mask,
+ "mask_salt_source": "caller" if mask_salt is not None else "per-run-random",
"allow_unmasked_columns": sorted(str(c) for c in allow_unmasked_columns),
"policy_sentences": list(intent.sentences),
"compiled_policy": found.compiled_policy_summary,
diff --git a/tests/test_copilot_mask_salt.py b/tests/test_copilot_mask_salt.py
new file mode 100644
index 00000000..87752f2c
--- /dev/null
+++ b/tests/test_copilot_mask_salt.py
@@ -0,0 +1,101 @@
+"""``analyze_dataset(mask_salt=...)``: reproducible or per-run sample tokens (#288)."""
+
+from __future__ import annotations
+
+import hashlib
+import hmac
+import html
+import json
+
+import pandas as pd
+import pytest
+
+from freshdata.enterprise.cleaner import _hash_value
+from freshdata.experimental.ai_copilot import analyze_dataset
+
+SALT = "Zq7-copilot-mask-salt-4f1e9b"
+
+
+def _frame() -> pd.DataFrame:
+ return pd.DataFrame(
+ {
+ "name": ["Alice Johnson", "Bob Smith"],
+ "alias": ["Alice Johnson", "Carol White"],
+ "v": [1, 2],
+ }
+ )
+
+
+def _tokens(report) -> list[dict]:
+ return report.model_context["sample_rows_masked"]
+
+
+def test_default_runs_use_a_fresh_key_per_run() -> None:
+ a, b = analyze_dataset(_frame()), analyze_dataset(_frame())
+ assert _tokens(a) != _tokens(b)
+ assert a.audit["model_context_sha256"] != b.audit["model_context_sha256"]
+ assert a.audit["mask_salt_source"] == "per-run-random"
+
+
+def test_same_mask_salt_reproduces_model_context_and_fingerprint() -> None:
+ a = analyze_dataset(_frame(), mask_salt=SALT)
+ b = analyze_dataset(_frame(), mask_salt=SALT)
+ assert _tokens(a) == _tokens(b)
+ assert a.model_context == b.model_context
+ assert a.audit["model_context_sha256"] == b.audit["model_context_sha256"]
+ assert a.audit["mask_salt_source"] == "caller"
+ other = analyze_dataset(_frame(), mask_salt=SALT + "-other")
+ assert other.audit["model_context_sha256"] != a.audit["model_context_sha256"]
+
+
+def test_tokens_follow_the_documented_per_column_derivation() -> None:
+ report = analyze_dataset(_frame(), mask_salt=SALT)
+ for position, column in enumerate(["name", "alias"]):
+ column_salt = hmac.new(
+ SALT.encode("utf-8"), f"copilot-col:{position}".encode(), hashlib.sha256
+ ).hexdigest()
+ expected = [_hash_value(v, column_salt, 16) for v in _frame()[column]]
+ assert [row[column] for row in _tokens(report)] == expected
+
+
+def test_same_value_in_two_columns_gets_different_tokens() -> None:
+ row = _tokens(analyze_dataset(_frame(), mask_salt=SALT))[0]
+ assert row["name"] != row["alias"]
+
+
+def test_mask_salt_never_appears_in_any_sink() -> None:
+ prompts: list[str] = []
+
+ def provider(prompt: str) -> str:
+ prompts.append(prompt)
+ return "ok"
+
+ with pytest.warns(FutureWarning, match="experimental"):
+ report = analyze_dataset(_frame(), mask_salt=SALT, provider=provider)
+ rendered = report._repr_html_()
+ sinks = {
+ "prompt": prompts[0],
+ "model_context": json.dumps(report.model_context, default=str),
+ "audit": json.dumps(report.audit, default=str),
+ "to_json": report.to_json(),
+ "html": html.unescape(rendered),
+ "str": str(report),
+ "recommended_code": report.recommended_code,
+ }
+ for sink, text in sinks.items():
+ assert SALT not in text, sink
+ assert SALT.encode().hex() not in text, sink
+
+
+def test_mask_salt_source_is_recorded_without_sample_rows() -> None:
+ report = analyze_dataset(_frame(), privacy="schema_only", mask_salt=SALT)
+ assert "sample_rows_masked" not in report.model_context
+ assert report.audit["mask_salt_source"] == "caller"
+
+
+@pytest.mark.parametrize(
+ ("salt", "error"), [("", ValueError), (b"bytes", TypeError), (7, TypeError)]
+)
+def test_invalid_mask_salt_is_rejected(salt, error) -> None:
+ with pytest.raises(error, match="mask_salt"):
+ analyze_dataset(_frame(), mask_salt=salt)
From cb9080e3f8c161c211895e9f217e291a6ecbd9e5 Mon Sep 17 00:00:00 2001
From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com>
Date: Tue, 15 Sep 2026 19:03:14 +0530
Subject: [PATCH 13/17] docs(copilot): document positional masking and
mask_salt
- ai-copilot.md: non-str labels, sensitive_columns, fail-closed masking,
per-run sample tokens and mask_salt; reproducibility claims now say the
findings, plan and code are stable and model_context only with mask_salt.
- threat-model.md section 3: positional masking, label collisions, unknown
sensitive_columns, fail-closed check, per-run fingerprint; replace the
stale CLAIM_REGISTRY reference with the tests that enforce the claim.
- threat-model.md section 6: replace the copilot sentences, which described
a deterministic path that no longer exists, with the per-run key and
mask_salt behaviour.
- trust-claims.md: the copilot claim is no longer in the CLAIM_REGISTRY;
move it to the other claims table with its tests, and add mask_salt.
---
docs/ai-copilot.md | 30 +++++++++++++++++++++++++++---
docs/threat-model.md | 37 ++++++++++++++++++++++++++++---------
docs/trust-claims.md | 10 ++++++----
3 files changed, 61 insertions(+), 16 deletions(-)
diff --git a/docs/ai-copilot.md b/docs/ai-copilot.md
index 5a6d3982..6306c66b 100644
--- a/docs/ai-copilot.md
+++ b/docs/ai-copilot.md
@@ -41,7 +41,10 @@ Three properties make this different from "ask a chatbot about my data":
- **Deterministic and offline.** The analysis is rule-based, built from
freshdata's own primitives (profiling, PII detection, the context-policy
compiler, value clustering, trust scoring). The same input always produces
- the same report; it runs in CI with no API key and no network access.
+ the same findings, plan and code; it runs in CI with no API key and no
+ network access. Masked sample tokens use a per-run key unless you pass
+ `mask_salt`, so `model_context` and its fingerprint are reproducible only
+ with a pinned salt.
- **Privacy-first.** Raw string values never enter `report.model_context` —
the only payload an LLM provider would ever see. Every sample column that
is not numeric or boolean is hash-masked first (numeric and boolean values
@@ -109,7 +112,17 @@ The `privacy` parameter controls what goes into `report.model_context`:
age) are the residual risk — drop such columns first or use
`"schema_only"`.
`allow_unmasked_columns=[...]` is an explicit per-column opt-out; it never
- exempts a declared or detected PII column.
+ exempts a declared or detected PII column. `sensitive_columns=[...]`
+ declares columns that are always masked, whatever their dtype (an SSN
+ stored as an integer, an internal case ID).
+ Column names in `context_policy`, `sensitive_columns` and
+ `allow_unmasked_columns` match a column's label or its `str()` form, so
+ integer, float and tuple labels (e.g. from `read_csv(header=None)`) work;
+ unknown `sensitive_columns` / `allow_unmasked_columns` names raise, and so
+ do labels that collide once converted to `str` (`0` and `"0"`). Masking is
+ done by column position and fails closed: if a selected column does not
+ come back hash-masked, `analyze_dataset` raises `RuntimeError` instead of
+ building `model_context`.
- `"schema_only"` — no cell values at all; only column names, dtypes,
missing percentages, and aggregate statistics.
@@ -128,6 +141,16 @@ Two details worth knowing:
- `report.audit["model_context_sha256"]` fingerprints the exact payload a
provider would have seen, so you can prove after the fact what was (and
was not) shared.
+- Masked sample tokens are HMAC-SHA256 hashes with a separate salt per
+ column, so equal values in two columns get different tokens. By default
+ the salts come from a random per-run key: the same frame gives different
+ tokens and a different `model_context_sha256` on every run. Pass
+ `mask_salt="..."` to derive the salts from your value instead, which makes
+ `model_context` and its fingerprint reproducible (useful in CI). Treat
+ that value as a secret, since anyone holding it can confirm guesses of
+ low-cardinality values; it is never written to the report, and
+ `report.audit["mask_salt_source"]` records only `"caller"` or
+ `"per-run-random"`.
## Plugging in an LLM (optional, experimental)
@@ -183,7 +206,8 @@ every time. What the copilot (and freshdata underneath it) adds:
which spellings are the same category — with severity and evidence;
- an audit trail a reviewer can read (`CleanReport` actions with rationale,
masked-context SHA, privacy events with HIPAA/GDPR tags);
-- reproducibility: the same input produces the same report, plan, and code.
+- reproducibility: the same input produces the same findings, plan, and
+ code, and with `mask_salt` the same `model_context` and fingerprint.
## Limitations and responsible use
diff --git a/docs/threat-model.md b/docs/threat-model.md
index da09c452..b5f1c5c8 100644
--- a/docs/threat-model.md
+++ b/docs/threat-model.md
@@ -63,13 +63,28 @@ fact what was shared. Per privacy mode:
or dates of birth, so no non-numeric value is trusted to be safe.
`allow_unmasked_columns` is an explicit per-column opt-out that
never exempts a declared or detected PII column and rejects unknown
- names. Detected-problem details entering `model_context` are value-free
- in **every** mode (`category_noise` spelling previews stay local).
+ names. Masking works by column position, so integer, float and tuple
+ column labels are masked exactly like `str` labels: `must_mask`,
+ `sensitive_columns` and `allow_unmasked_columns` match a label or its
+ `str()` form, unknown `sensitive_columns` raise, and labels that collide
+ once converted to `str` (`0` and `"0"`) raise. Masking fails closed: if a
+ selected column does not come back as hash tokens, `analyze_dataset`
+ raises instead of building `model_context`. Detected-problem details
+ entering `model_context` are value-free in **every** mode
+ (`category_noise` spelling previews stay local).
- **`schema_only`** — no cell values at all.
-These guarantees are enforced by adversarial regression tests registered in
-the `CLAIM_REGISTRY` (`tests/test_experimental_ai_copilot.py`), which CI
-re-verifies against the README wording.
+Sample tokens are keyed per run by default, so `model_context` and
+`report.audit["model_context_sha256"]` differ between runs on the same
+frame; `analyze_dataset(mask_salt=...)` makes them reproducible (see
+boundary 6).
+
+These guarantees are enforced by adversarial regression tests in
+`tests/test_experimental_ai_copilot.py`, `tests/test_privacy_adversarial.py`,
+`tests/test_copilot_sample_dtype_allowlist.py` and
+`tests/test_copilot_positional_masking.py`, which CI runs on every change.
+The README no longer states this claim verbatim, so it is not part of the
+README `CLAIM_REGISTRY` audit.
**Residual risk (by design, documented):** numeric values pass through
unmasked. Numeric quasi-identifiers — an exact salary plus age plus a
@@ -116,10 +131,14 @@ these keyless paths used public constants, so keyless output from earlier
releases can be reversed by enumerating candidate values: re-pseudonymise
it with a secret key (see [Compliance](compliance.md#pseudonymisation-keys)).
-The copilot's internal masking uses the default
-deterministic path on purpose: a per-run random salt would break the
-documented reproducibility of `model_context` and its audit fingerprint.
-This trade-off is tracked as a roadmap item, not silently changed.
+The copilot's sample masking derives a separate salt for each column from a
+random per-run key, so masked sample tokens and
+`report.audit["model_context_sha256"]` differ between runs on the same
+frame. `analyze_dataset(mask_salt=...)` derives the column salts from your
+value instead, which makes `model_context` and its fingerprint reproducible.
+Treat that value as a secret: anyone holding it can confirm guesses of
+low-cardinality sample values. It is never written to the report;
+`report.audit["mask_salt_source"]` records only whether one was supplied.
Report stand-ins for declared `sensitive_columns` are a separate case. They
are the `[SENSITIVE:xxxxxxxx]` tokens in `CleanReport` warnings, coerced
diff --git a/docs/trust-claims.md b/docs/trust-claims.md
index ebaad302..28494569 100644
--- a/docs/trust-claims.md
+++ b/docs/trust-claims.md
@@ -1,9 +1,10 @@
# Trust claims — evidence map
Every trust-relevant claim FreshData makes, mapped to the thing that proves
-it. The three product-defining claims are additionally machine-enforced: the
-`CLAIM_REGISTRY` (`benchmarks/cleanbench/reproducibility.py`) pins their
-README wording verbatim to named tests, and CI fails if either side drifts.
+it. The product-defining claims the README still states verbatim are
+additionally machine-enforced: the `CLAIM_REGISTRY`
+(`benchmarks/cleanbench/reproducibility.py`) pins their README wording to
+named tests, and CI fails if either side drifts.
## Machine-enforced claims (CLAIM_REGISTRY)
@@ -11,12 +12,13 @@ README wording verbatim to named tests, and CI fails if either side drifts.
|---|---|
| protected columns are never modified | `tests/test_semantic_cleaning.py::test_id_columns_protected`, CleanBench `T2.protected_column_violation_rate` |
| nothing happens silently | `tests/test_semantic_cleaning.py::test_assist_records_without_mutating` |
-| raw PII never enters the copilot's model context | 4 tests in `tests/test_experimental_ai_copilot.py`, incl. adversarial cases: undeclared string-like columns masked, `category_noise` previews withheld; `tests/test_copilot_sample_dtype_allowlist.py` covers every non-numeric dtype (Arrow string / dictionary / list, categorical, bytes, datetime, timedelta, period) across the prompt, `model_context`, JSON, HTML and text sinks |
## Other README / docs claims
| Claim | Status | Evidence / boundary |
|---|---|---|
+| raw PII never enters the copilot's model context | **holds** | not in the `CLAIM_REGISTRY` (the README no longer states it verbatim). Tests: `tests/test_experimental_ai_copilot.py` and `tests/test_privacy_adversarial.py` (undeclared string-like columns masked, `category_noise` previews withheld); `tests/test_copilot_sample_dtype_allowlist.py` (every non-numeric dtype: Arrow string / dictionary / list, categorical, bytes, datetime, timedelta, period); `tests/test_copilot_positional_masking.py` (int, float and tuple labels, fail-closed masking); each across the prompt, `model_context`, JSON, HTML and text sinks. Boundary: numeric and boolean sample values pass through |
+| copilot `model_context` fingerprint is reproducible | **holds with `mask_salt`** | `tests/test_copilot_mask_salt.py`; without `mask_salt` sample tokens and `model_context_sha256` are per-run |
| 93% coverage gate enforced in CI | **holds** | `--cov-fail-under=93` in pyproject addopts; CI runs it on every PR (recent runs: 93.5%) |
| Safe defaults (never imputes identifiers, never touches targets, no blind outlier removal) | **holds** | protected-column guard + role inference tests; `strict=True` escalates ambiguity to errors |
| Fully offline; only network call is `fd.models.pull` | **holds** | default test gate is `-m "not online"`; no other network code paths |
From 640bad24babec57d360e4aa25ac0f5dc34146649 Mon Sep 17 00:00:00 2001
From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com>
Date: Tue, 15 Sep 2026 19:03:14 +0530
Subject: [PATCH 14/17] test(truthbench): pin copilot canary scanning and sink
determinism
- A canary whose digits sit in one run inside a hex token still flags;
the same digits split by letters inside a long leaf do not.
- A hash token carrying canary digits is flagged at $.rendered.html on
every run.
- With the pinned mask_salt the copilot adapter renders identical HTML,
prompt and model_context across seeded process randomness, with no leak
(3 seeds in the default lane, 25 seeds on every domain under -m large).
- An int/float/tuple label and object/string/Arrow/categorical dtype sweep
finds no canary in rendered.html, the prompt or model_context.
---
.../test_copilot_canary_determinism.py | 204 ++++++++++++++++++
1 file changed, 204 insertions(+)
create mode 100644 tests/truthbench/test_copilot_canary_determinism.py
diff --git a/tests/truthbench/test_copilot_canary_determinism.py b/tests/truthbench/test_copilot_canary_determinism.py
new file mode 100644
index 00000000..0ba3fc16
--- /dev/null
+++ b/tests/truthbench/test_copilot_canary_determinism.py
@@ -0,0 +1,204 @@
+"""Copilot TruthBench sinks are deterministic and the canary scanner is precise.
+
+A TruthBench release gate once reported a digit-only phone canary in the
+copilot ``rendered.html`` sink intermittently. Two things could produce that:
+the digit-only scanner joining digits from different tokens of one long HTML
+leaf, and masked sample tokens that differ on every run. These tests pin both.
+"""
+
+from __future__ import annotations
+
+import random
+import re
+import secrets
+from datetime import datetime, timezone
+
+import pandas as pd
+import pytest
+from benchmarks.truthbench.fixtures import DOMAINS, build_fixture
+from benchmarks.truthbench.privacy import SinkScanner
+from benchmarks.truthbench.surfaces.copilot import CopilotAdapter
+
+from freshdata.enterprise import privacy as privacy_module
+from freshdata.experimental import ai_copilot
+from freshdata.experimental.ai_copilot import analyze_dataset
+
+PHONE = "555-0110"
+PROSE = " ".join(["the quick brown fox jumps over the lazy dog"] * 40)
+
+
+def _education():
+ return build_fixture("education")
+
+
+def _phone_canary_id(fixture) -> str:
+ return next(k for k, v in sorted(fixture.pii_canaries.items()) if v == PHONE)
+
+
+def _digits(text: str) -> str:
+ return "".join(ch for ch in text if ch.isdigit())
+
+
+# (a) scanner precision on long leaves ----------------------------------------
+
+
+def test_contiguous_canary_digits_inside_a_hex_token_flag() -> None:
+ fixture = _education()
+ leaf = f"| {PROSE} token 3fa5550110c9 {PROSE} | "
+ # One numeric run carries every canary digit, so this is a genuine match
+ # (the scanner reports whichever normalised variant matches first).
+ assert any(_digits(PHONE) in run for run in re.findall(r"\d+", leaf))
+ leaks = CopilotAdapter().scanner_for(fixture).scan({"html": leaf})
+ assert _phone_canary_id(fixture) in {x.canary_id for x in leaks}
+
+
+def test_canary_digits_split_by_letters_in_a_long_leaf_do_not_flag() -> None:
+ fixture = _education()
+ leaf = f"{PROSE} 3fa555b01c10 {PROSE} | "
+ # The concatenated digit stream of the leaf does contain the canary;
+ # only digits within one numeric run may match it.
+ assert _digits(PHONE) in _digits(leaf)
+ assert CopilotAdapter().scanner_for(fixture).scan({"html": leaf}) == []
+
+
+# (b) a token carrying canary digits is caught every time ---------------------
+
+
+def _all_fixture_scanner() -> SinkScanner:
+ canaries: dict[str, object] = {}
+ for domain in DOMAINS:
+ canaries.update(build_fixture(domain).pii_canaries)
+ return SinkScanner.from_canaries(canaries, key=b"copilot-determinism-test")
+
+
+def test_hash_token_with_canary_digits_is_flagged_deterministically(monkeypatch) -> None:
+ monkeypatch.setattr(
+ privacy_module, "_hash_value", lambda value, salt, length: "ab5550110cdef012"
+ )
+ scanner = _all_fixture_scanner()
+ phone_id = _phone_canary_id(_education())
+ runs = []
+ for _ in range(2):
+ observation = CopilotAdapter().observe(build_fixture("crm"), {})
+ assert observation.unexpected_exception is None
+ leaks = scanner.scan(observation.audit_sinks)
+ runs.append(sorted((x.canary_id, x.variant, x.path) for x in leaks))
+ assert runs[0] == runs[1]
+ paths = {path for canary, _, path in runs[0] if canary == phone_id}
+ assert "$.rendered.html" in paths
+
+
+# (c) pinned mask_salt: rendered sinks do not depend on process randomness ----
+
+
+class _FrozenDatetime(datetime):
+ @classmethod
+ def now(cls, tz=None): # type: ignore[override]
+ return datetime(2026, 1, 2, 3, 4, 5, tzinfo=tz or timezone.utc)
+
+
+def _seed_randomness(monkeypatch, seed: int) -> None:
+ rng = random.Random(seed)
+
+ def token_bytes(nbytes: int | None = None) -> bytes:
+ n = 32 if nbytes is None else nbytes
+ return rng.getrandbits(8 * n).to_bytes(n, "big")
+
+ monkeypatch.setattr(secrets, "token_bytes", token_bytes)
+ monkeypatch.setattr(secrets, "token_hex", lambda nbytes=None: token_bytes(nbytes).hex())
+
+
+def _rendered_across_seeds(monkeypatch, domain: str, seeds) -> None:
+ monkeypatch.setattr(ai_copilot, "datetime", _FrozenDatetime)
+ fixture = build_fixture(domain)
+ scanner = CopilotAdapter().scanner_for(fixture)
+ outputs = []
+ for seed in seeds:
+ _seed_randomness(monkeypatch, seed)
+ observation = CopilotAdapter().observe(fixture, {})
+ assert observation.unexpected_exception is None
+ sinks = observation.audit_sinks
+ assert scanner.scan(sinks) == []
+ outputs.append((sinks["rendered"]["html"], sinks["prompt"], sinks["model_context"]))
+ assert all(output == outputs[0] for output in outputs[1:])
+
+
+def test_seeded_randomness_reaches_the_default_key(monkeypatch) -> None:
+ monkeypatch.setattr(ai_copilot, "datetime", _FrozenDatetime)
+ frame = _education().frame
+ htmls = []
+ for seed in (11, 12, 11):
+ _seed_randomness(monkeypatch, seed)
+ htmls.append(analyze_dataset(frame)._repr_html_())
+ assert htmls[0] == htmls[2]
+ assert htmls[0] != htmls[1]
+
+
+def test_pinned_mask_salt_renders_identically_across_seeds(monkeypatch) -> None:
+ _rendered_across_seeds(monkeypatch, "education", seeds=(0, 1, 2))
+
+
+@pytest.mark.large
+@pytest.mark.parametrize("domain", DOMAINS)
+def test_pinned_mask_salt_renders_identically_across_many_seeds(monkeypatch, domain) -> None:
+ _rendered_across_seeds(monkeypatch, domain, seeds=range(25))
+
+
+# (d) label and dtype sweep ---------------------------------------------------
+
+
+def _text_dtypes():
+ dtypes = {"object": object, "string": "string", "categorical": "category"}
+ try:
+ import pyarrow as pa # noqa: PLC0415
+
+ dtypes["arrow-string"] = pd.ArrowDtype(pa.string())
+ except ImportError:
+ pass
+ return dtypes
+
+
+LABELS = {
+ "int": (0, 1, 2, 3),
+ "float": (0.5, 1.5, 2.5, 3.5),
+ "tuple": (("g", "phone"), ("g", "email"), ("g", "notes"), ("g", "n")),
+}
+
+
+@pytest.mark.parametrize("labels", sorted(LABELS))
+@pytest.mark.parametrize("dtype", sorted(_text_dtypes()))
+def test_label_and_dtype_sweep_finds_no_leak(labels, dtype) -> None:
+ fixture = _education()
+ scanner = CopilotAdapter().scanner_for(fixture)
+ values = sorted(v for v in fixture.pii_canaries.values() if isinstance(v, str))
+ assert PHONE in values
+ others = [v for v in values if v != PHONE][:2]
+ names = LABELS[labels]
+ frame = pd.DataFrame(
+ {
+ 0: pd.Series([PHONE, "n/a", "unknown"], dtype=_text_dtypes()[dtype]),
+ 1: pd.Series([others[0], "none", "missing"], dtype=_text_dtypes()[dtype]),
+ 2: pd.Series([others[1], "blank", "empty"], dtype=_text_dtypes()[dtype]),
+ 3: [5550110, 1, 2],
+ }
+ )
+ frame.columns = pd.Index(list(names), tupleize_cols=False)
+ prompts: list[str] = []
+
+ def provider(prompt: str) -> str:
+ prompts.append(prompt)
+ return "ok"
+
+ with pytest.warns(FutureWarning, match="experimental"):
+ report = analyze_dataset(
+ frame,
+ provider=provider,
+ sensitive_columns=[names[3]],
+ mask_salt="sweep-salt",
+ )
+ sinks = {
+ "rendered": {"html": report._repr_html_()},
+ "prompt": prompts[0],
+ "model_context": report.model_context,
+ }
+ assert scanner.scan(sinks) == []
From 50efe91c95c5d583699da61f21be4994df4a89af Mon Sep 17 00:00:00 2001
From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com>
Date: Tue, 15 Sep 2026 20:35:42 +0530
Subject: [PATCH 15/17] docs: changelog for the security fixes
---
CHANGELOG.md | 52 ++++++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 52 insertions(+)
diff --git a/CHANGELOG.md b/CHANGELOG.md
index ea193dcd..8b0faa66 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -6,6 +6,52 @@ adheres to [Semantic Versioning](https://semver.org/).
## [Unreleased]
+### Security
+- GPX and SDMX parsing detects the document encoding (BOM, UTF-16/UTF-32
+ prefixes, XML declaration) and checks with expat before reading, so DTD and
+ entity declarations are rejected in every encoding. Previously a UTF-16
+ document bypassed the check and allowed entity expansion.
+- CSV formula sanitising now covers every level of multi-row headers, and index
+ labels and names, so crafted header cells from the input are no longer written
+ as live formulas.
+- The DuckDB engine no longer spills to the shared `/tmp/freshdata_spill`.
+ `EngineConfig.temp_directory` defaults to `None`, and each run spills into a
+ private (0700), per-run directory under the user's cache directory (or
+ `FRESHDATA_SPILL_DIR`), which is removed afterwards. An explicit
+ `temp_directory` is checked for ownership and permissions.
+- Baseline category labels are no longer unkeyed SHA-1. Without `label_key` a
+ baseline stores a label-free frequency profile; with `label_key` (or
+ `FRESHDATA_BASELINE_KEY`) labels are HMAC-SHA256. Baselines are written as
+ schema `freshdata-baseline-v2`; v1 baselines still load with a warning and
+ should be rebuilt.
+- `JsonTokenVault` and `SqliteTokenVault` create their files owner-only (0600)
+ at creation time. SQLite journal/WAL files inherit that mode. Existing
+ group/other-readable vault files trigger a warning.
+- `tokenize`, `surrogate` and keyless `fpe` masking rules, and the policy
+ `pseudonymize` action (the default in the GDPR, HIPAA and FERPA packs), no
+ longer fall back to public constants when no key is set. They use a random
+ per-call key and emit `EphemeralKeyWarning`; pass `key=`/`key_env=` for
+ stable, joinable output.
+- `fd.learn(privacy='mask')` treats every PII type `detect_pii` reports (payment
+ cards, IBANs, IP addresses, health and licence identifiers) as sensitive;
+ unknown types fail closed. It adds card and bank column-name hints.
+ `freshdata profile audit` flags raw card numbers and IBANs in existing
+ profiles.
+- `clean_enterprise` reports no longer contain raw values of masked columns:
+ cluster canonical, variant and key values, semantic-validation invalid
+ samples, and `clean_report.coerced_cells` originals (and the coercion warnings
+ quoting them) for masked columns are masked or redacted. The
+ `[SENSITIVE:xxxxxxxx]` tokens that stand in for declared `sensitive_columns`
+ values are now a truncated HMAC-SHA256 under a random per-process key instead
+ of an unkeyed SHA-256; they still match within a run but differ between runs.
+- Copilot sample masking uses an allow-list: only numeric and boolean sample
+ values pass through, so Arrow-backed string, dictionary and other non-numeric
+ columns (including datetimes) are hash-masked.
+- Copilot masks sample values by column position, so integer, float and tuple
+ column labels no longer bypass masking. `sensitive_columns` and `must_mask`
+ match non-string labels, unknown `sensitive_columns` raise, labels that
+ collide as strings raise, and masking fails closed.
+
### Added
- `fd.clean_excel()`, the Excel companion to `fd.clean_csv()`: reads one sheet,
cleans it, and optionally writes the result, with formula sanitization on by
@@ -509,6 +555,12 @@ adheres to [Semantic Versioning](https://semver.org/).
- The `quarantine` privacy-policy action works on nullable integer, boolean
and categorical columns instead of raising `TypeError`; those columns come
back as object dtype and missing cells stay missing.
+- `detect_pii` and detection-driven `anonymize` scan categorical text columns on
+ pandas 1.5, as on pandas 2 (#280).
+- Crypto FPE honours `visible` (#281).
+- `analyze_dataset(mask_salt=...)` makes `model_context` and its fingerprint
+ reproducible; by default they are per-run, and `audit["mask_salt_source"]`
+ records which was used (#288).
## [2.0.0] - 2026-07-20
From 4b3cd9a3655069c784ce2245fc0ca590228bfba6 Mon Sep 17 00:00:00 2001
From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com>
Date: Tue, 15 Sep 2026 20:42:40 +0530
Subject: [PATCH 16/17] fixup(enterprise): don't apply strict column matching
to report redaction
Integration fix. The report redaction added for masked columns resolves each
masking rule against the report's columns only (cluster, validation and
coercion reports) to find which report entries to mask. main's #406 made
_resolve_columns raise for listed columns that match nothing when the rule is
strict, and CLI --mask rules are strict. A masked column that appears in no
report (for example "email" without clustering hits) therefore raised
"specifies column(s) not found in dataframe" from clean_enterprise, failing
tests/test_enterprise_cli.py on the integrated branch.
Pass strict=False for this subset lookup, as privacy.py already does for its
duplicated-label lookup. The masking stage still enforces strict against the
real frame. Adds regression tests for both sides.
---
src/freshdata/enterprise/interface.py | 5 ++++-
tests/test_enterprise_report_masking.py | 20 ++++++++++++++++++++
2 files changed, 24 insertions(+), 1 deletion(-)
diff --git a/src/freshdata/enterprise/interface.py b/src/freshdata/enterprise/interface.py
index 15d8fdc8..9d426109 100644
--- a/src/freshdata/enterprise/interface.py
+++ b/src/freshdata/enterprise/interface.py
@@ -285,7 +285,10 @@ def _masked_report_columns(result: EnterpriseResult, ec: EnterpriseConfig) -> di
)
rules: dict[str, list[MaskingRule]] = {}
for rule in ec.masking:
- for column in _resolve_columns(rule, candidates):
+ # Only report columns are searched here, so a listed column that is
+ # absent from them is expected; the masking stage already enforced
+ # ``strict`` against the frame. Never raise for it.
+ for column in _resolve_columns(rule, candidates, strict=False):
rules.setdefault(str(column), []).append(rule)
# One rule reproduces its token; several stacked rules are just redacted.
maskers = {c: _report_masker(rs[0] if len(rs) == 1 else None) for c, rs in rules.items()}
diff --git a/tests/test_enterprise_report_masking.py b/tests/test_enterprise_report_masking.py
index aea75ffb..b868a8da 100644
--- a/tests/test_enterprise_report_masking.py
+++ b/tests/test_enterprise_report_masking.py
@@ -218,3 +218,23 @@ def test_coerced_cells_of_masked_date_like_column(strategy):
assert not [w for w in res.clean_report.warnings if "XYZZY" in w or "QWERTY" in w]
# The unmasked column keeps its reviewable original.
assert list(coerced["open"].values()) == ["7xyzOPEN"]
+
+
+def test_strict_rule_on_column_absent_from_reports_does_not_raise():
+ # "email" is masked but appears in no cluster, validation or coercion
+ # report; report redaction must not re-apply ``strict`` to that subset.
+ ec = EnterpriseConfig(
+ masking=(MaskingRule(name="m", columns=("Email",), strategy="hash", strict=True),),
+ )
+ df = pd.DataFrame({"email": ["a@x.com", "b@y.io"], "v": [1, 2]})
+ res = clean_enterprise(df, enterprise=ec)
+ assert "a@x.com" not in set(res.data["email"])
+ assert _leaks(res, ["a@x.com", "b@y.io"]) == []
+
+
+def test_strict_rule_on_column_missing_from_frame_still_raises():
+ ec = EnterpriseConfig(
+ masking=(MaskingRule(name="m", columns=("nope",), strategy="hash", strict=True),),
+ )
+ with pytest.raises(ValueError, match="not found in dataframe"):
+ clean_enterprise(pd.DataFrame({"email": ["a@x.com"]}), enterprise=ec)
From 25f5a5922fc209df57426c4bec692ea48af14e4d Mon Sep 17 00:00:00 2001
From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com>
Date: Tue, 15 Sep 2026 20:36:11 +0530
Subject: [PATCH 17/17] chore(release): 2.1.0
Bump version 2.0.0 -> 2.1.0 and finalize the changelog for release.
---
CHANGELOG.md | 2 +-
docs/production-readiness.md | 2 +-
pyproject.toml | 2 +-
src/freshdata/__init__.py | 2 +-
uv.lock | 2 +-
5 files changed, 5 insertions(+), 5 deletions(-)
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 8b0faa66..102c6dcb 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -4,7 +4,7 @@ All notable changes to this project are documented here. The format follows
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project
adheres to [Semantic Versioning](https://semver.org/).
-## [Unreleased]
+## [2.1.0] - 2026-09-15
### Security
- GPX and SDMX parsing detects the document encoding (BOM, UTF-16/UTF-32
diff --git a/docs/production-readiness.md b/docs/production-readiness.md
index 8bdb4dc5..1a101728 100644
--- a/docs/production-readiness.md
+++ b/docs/production-readiness.md
@@ -16,7 +16,7 @@ clean data nobody is watching. Each item links to the relevant guarantee on the
## Install & pin
-- [ ] Pin an exact version (`freshdata-cleaner==2.0.0`) and the extras you use
+- [ ] Pin an exact version (`freshdata-cleaner==2.1.0`) and the extras you use
(`freshdata-cleaner[polars,privacy]`). Cleaning defaults can tighten between
minor versions — pinning keeps decisions reproducible.
- [ ] Install only the extras you need. The base install has **no** heavy deps;
diff --git a/pyproject.toml b/pyproject.toml
index 50cad910..72e52101 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -7,7 +7,7 @@ build-backend = "hatchling.build"
# distribution MUST remain `freshdata-cleaner`: PyPI rejects `freshdata` as
# too similar to the existing `fresh-data` project (see commit 88aa495).
name = "freshdata-cleaner"
-version = "2.0.0"
+version = "2.1.0"
description = "Fast, safe, automatic data cleaning for real-world tabular data."
readme = "README.md"
requires-python = ">=3.9"
diff --git a/src/freshdata/__init__.py b/src/freshdata/__init__.py
index 12ad20d4..7c2dda65 100644
--- a/src/freshdata/__init__.py
+++ b/src/freshdata/__init__.py
@@ -112,7 +112,7 @@
)
from .textlint import TextIssue, TextLintReport, lint_text_encoding
-__version__ = "2.0.0"
+__version__ = "2.1.0"
__all__ = [
"Action",
diff --git a/uv.lock b/uv.lock
index 42044a3f..c125af60 100644
--- a/uv.lock
+++ b/uv.lock
@@ -2978,7 +2978,7 @@ wheels = [
[[package]]
name = "freshdata-cleaner"
-version = "2.0.0"
+version = "2.1.0"
source = { editable = "." }
dependencies = [
{ name = "numpy", version = "2.0.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.10'" },