From 84e95e914382d0d49d582359d9f99118b1c8ed7c Mon Sep 17 00:00:00 2001 From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:01:01 +0530 Subject: [PATCH 1/2] fix(quality): report unmeasured quality-debt dimensions as not assessed evaluate_quality_debt scored a dimension a clean 0.0 when it was never measured: type_instability when profiling failed, pii_risk when the PII scan was unavailable or failed, schema_drift and category_churn when no baseline was passed, and any dimension missing from the scores. Each item counted as assessed, so the gate could report pass on evidence it never collected, and the ledger stored a 0.0 that reset escalation history. These dimensions now use the DebtItem.assessed mechanism from #414: score and over_threshold serialise as None, the detail says why, the item is listed in gate.unassessed and shown as "? ... not assessed", and it adds nothing to the total, the gate status or the ledger. Runs with a baseline and a working profile and PII scan are byte-identical. --- CHANGELOG.md | 11 ++ docs/decision-workflow.md | 19 ++- src/freshdata/quality.py | 27 ++-- tests/test_quality_debt_dimensions.py | 200 +++++++++++++++++++++++++- 4 files changed, 236 insertions(+), 21 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c3b772c8..44f97797 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,17 @@ adheres to [Semantic Versioning](https://semver.org/). ## [Unreleased] +### Fixed +- `fd.evaluate_quality_debt` no longer scores a dimension a clean 0.0 when it + was never measured. `type_instability` when profiling fails, `pii_risk` when + the PII scan is unavailable or fails, and `schema_drift` and `category_churn` + when no `baseline=` is passed are now not assessed, using the same mechanism + as undetectable duplicates (#414): `score` and `over_threshold` serialise as + `None`, the detail says why, the item is listed in + `QualityDebtGate.unassessed`, and it counts toward neither the total, the gate + status nor the ledger history. Runs with a baseline and a working profile and + PII scan score exactly as before. + ## [2.1.0] - 2026-09-15 ### Security diff --git a/docs/decision-workflow.md b/docs/decision-workflow.md index 0440caa0..d2f02b31 100644 --- a/docs/decision-workflow.md +++ b/docs/decision-workflow.md @@ -76,12 +76,19 @@ review backlog), persists the history to SQLite, and **escalates warn→fail whe an issue repeats or worsens** across runs. The duplicates dimension counts duplicate rows left in the cleaned output (or -the rows removed, if more). When duplicates cannot be checked because a column -holds unhashable values (lists, dicts, or nested Arrow list/struct/map columns), -the dimension is **not assessed** rather than scored clean: `to_dict()` gives -`score` and `over_threshold` as `None`, the detail names the columns, -`gate.unassessed` lists it, and `summary()` prints a `? duplicates: not -assessed` line. An unassessed dimension adds nothing to the total, never +the rows removed, if more). A dimension that could not be measured is **not +assessed** rather than scored clean. That happens when: + +- duplicates cannot be checked because a column holds unhashable values (lists, + dicts, or nested Arrow list/struct/map columns); the detail names the columns, +- profiling fails, for `type_instability`, +- the PII scan is unavailable or fails, for `pii_risk`, +- no `baseline=` is passed, for `schema_drift` and `category_churn`, which only + have something to compare against when a baseline is supplied. + +For an unassessed dimension `to_dict()` gives `score` and `over_threshold` as +`None`, the detail says why, `gate.unassessed` lists it, and `summary()` prints +a `? : not assessed` line. It adds nothing to the total, never changes the gate status on its own, and is not written to the ledger, so escalation compares against the last run that measured it. diff --git a/src/freshdata/quality.py b/src/freshdata/quality.py index 3a702e25..89db3e1f 100644 --- a/src/freshdata/quality.py +++ b/src/freshdata/quality.py @@ -49,8 +49,9 @@ class DebtItem: detail: str previous: float | None = None #: False when the dimension could not be measured on this run (for example - #: duplicate rows in a frame with unhashable cells). An unassessed item is - #: neither over threshold nor evidence of a clean result: it serialises + #: duplicate rows in a frame with unhashable cells, a failed profile or PII + #: scan, or schema drift and category churn with no baseline). An unassessed + #: item is neither over threshold nor evidence of a clean result: it serialises #: with ``score`` and ``over_threshold`` as ``None``, adds nothing to the #: total, never drives the gate status and is not written to the ledger. assessed: bool = True @@ -235,8 +236,10 @@ def _score_debt( if c.suggested_dtype and c.suggested_dtype != c.dtype) out["type_instability"] = (retype / max(1, report.cols_after), f"{retype} column(s) with unstable types") - except Exception: # pragma: no cover - profiling is best-effort - out["type_instability"] = (0.0, "not assessed") + except Exception as exc: # profiling is best-effort + # A failed profile measured nothing, so it is not a clean result. + out["type_instability"] = ( + None, f"column types could not be checked: profiling failed ({type(exc).__name__})") # PII risk (best-effort, lazy enterprise import). try: @@ -248,10 +251,13 @@ def _score_debt( n_pii = len([col for col in scan.by_column() if col]) out["pii_risk"] = (min(1.0, n_pii / max(1, report.cols_after)), f"{n_pii} potential PII column(s)") - except Exception: - out["pii_risk"] = (0.0, "PII scan unavailable") + except Exception as exc: + # No scan ran (detector missing or failing): unknown, not "no PII". + out["pii_risk"] = ( + None, f"PII could not be checked: scan unavailable ({type(exc).__name__})") - # Schema drift + category churn need a baseline. + # Schema drift + category churn need a baseline. Without one there is + # nothing to compare against, so both are unassessed rather than clean. if baseline is not None: added = set(map(str, df.columns)) - set(map(str, baseline.columns)) removed = set(map(str, baseline.columns)) - set(map(str, df.columns)) @@ -261,8 +267,8 @@ def _score_debt( churn = _category_churn(baseline, df) out["category_churn"] = (churn, "category distribution churn vs baseline") else: - out["schema_drift"] = (0.0, "no baseline supplied") - out["category_churn"] = (0.0, "no baseline supplied") + out["schema_drift"] = (None, "no baseline supplied") + out["category_churn"] = (None, "no baseline supplied") return out @@ -355,6 +361,7 @@ def evaluate_quality_debt( ``None`` keeps the run in memory only (no escalation history). baseline: Optional prior frame enabling schema-drift and category-churn scoring. + Without it both dimensions are reported as not assessed. thresholds: Per-dimension overrides of the default "in debt" thresholds. **clean_options: @@ -381,7 +388,7 @@ def evaluate_quality_debt( items: list[DebtItem] = [] for dim in DEBT_DIMENSIONS: - score, detail = scores.get(dim, (0.0, "not assessed")) + score, detail = scores.get(dim, (None, "dimension was not scored")) items.append(DebtItem(dim, 0.0 if score is None else score, thr[dim], detail, previous.get(dim), assessed=score is not None)) diff --git a/tests/test_quality_debt_dimensions.py b/tests/test_quality_debt_dimensions.py index 05f15cc1..51d5a9c0 100644 --- a/tests/test_quality_debt_dimensions.py +++ b/tests/test_quality_debt_dimensions.py @@ -2,6 +2,7 @@ from __future__ import annotations +import importlib import json import sqlite3 @@ -60,8 +61,10 @@ def test_no_duplicates_scores_zero() -> None: def test_plain_frame_duplicates_serialisation_unchanged() -> None: """Detectable duplicates keep the exact prior scores and output.""" + # A baseline keeps schema_drift and category_churn measured (#16). _, gate = fd.evaluate_quality_debt( - _half_duplicated(), debt_policy="fail", ledger=None, verbose=False + _half_duplicated(), baseline=_half_duplicated(), debt_policy="fail", ledger=None, + verbose=False, ) dup = _item(gate, "duplicates") assert dup.assessed @@ -75,8 +78,9 @@ def test_plain_frame_duplicates_serialisation_unchanged() -> None: assert "not assessed" not in gate.summary() assert gate.total_score == pytest.approx(sum(i.score for i in gate.items)) + clean = pd.DataFrame({"a": range(20), "b": list("xy" * 10)}) _, clean_gate = fd.evaluate_quality_debt( - pd.DataFrame({"a": range(20), "b": list("xy" * 10)}), ledger=None, verbose=False + clean, baseline=clean.copy(), ledger=None, verbose=False ) assert _item(clean_gate, "duplicates").to_dict() == { "dimension": "duplicates", "score": 0.0, "threshold": 0.1, "over_threshold": False, @@ -137,7 +141,7 @@ def _assert_duplicates_unassessed(gate: QualityDebtGate) -> None: @pytest.mark.parametrize("policy", ["warn", "fail", "warn_then_fail"]) def test_object_list_duplicates_unassessed(policy: str) -> None: _, gate = fd.evaluate_quality_debt( - _list_frame(), debt_policy=policy, ledger=None, verbose=False + _list_frame(), baseline=_list_frame(), debt_policy=policy, ledger=None, verbose=False ) _assert_duplicates_unassessed(gate) assert gate.status == "pass" # undetectable duplicates never gate on their own @@ -145,8 +149,9 @@ def test_object_list_duplicates_unassessed(policy: str) -> None: @pytest.mark.parametrize("kind", ["list", "struct"]) def test_nested_arrow_duplicates_unassessed(kind: str) -> None: + frame = _arrow_frame(kind) _, gate = fd.evaluate_quality_debt( - _arrow_frame(kind), debt_policy="fail", ledger=None, verbose=False + frame, baseline=frame.copy(), debt_policy="fail", ledger=None, verbose=False ) _assert_duplicates_unassessed(gate) assert gate.status == "pass" @@ -204,7 +209,8 @@ def test_unassessed_duplicates_skip_the_ledger(tmp_path) -> None: finally: conn.close() assert [r[0] for r in rows] == [1] # the unassessed run recorded no score - assert n_items == 9 + 8 + # No baseline: schema_drift and category_churn are unassessed on both runs. + assert n_items == 7 + 6 # The next measured run escalates from, and compares against, the last run # that measured duplicates. _, g3 = fd.evaluate_quality_debt( @@ -268,3 +274,187 @@ def test_pii_risk_ignores_findings_without_column(monkeypatch: pytest.MonkeyPatc pii = _item(gate, "pii_risk") assert pii.detail == "1 potential PII column(s)" assert pii.score == pytest.approx(0.25) + + +# -- Dimensions that were never measured are not assessed (#16) --------------- + + +def _raise(exc: type[Exception]): + def boom(*_args, **_kwargs): + raise exc("simulated failure") + + return boom + + +def _assert_unassessed(gate: QualityDebtGate, dimension: str, detail: str) -> None: + item = _item(gate, dimension) + assert not item.assessed + assert not item.over + assert not item.worsening + assert item.detail == detail + payload = item.to_dict() + assert list(payload) == _DUP_KEYS + assert payload["score"] is None + assert payload["over_threshold"] is None + assert item in gate.unassessed + assert item not in gate.warned + assert gate.total_score == pytest.approx( + round(sum(i.score for i in gate.items if i.assessed), 4)) + assert f"? {dimension}: not assessed — {detail}" in gate.summary() + frame = gate.to_frame() + assert pd.isna(frame.loc[frame["dimension"] == dimension, "score"].iloc[0]) + assert json.loads(json.dumps(gate.to_dict())) == gate.to_dict() + + +def _assert_other_items_unchanged(gate: QualityDebtGate, reference: QualityDebtGate, + dimension: str) -> None: + assert [i.to_dict() for i in gate.items if i.dimension != dimension] == [ + i.to_dict() for i in reference.items if i.dimension != dimension] + + +def _baseline_for(df: pd.DataFrame) -> pd.DataFrame: + return df.copy() + + +@pytest.mark.parametrize("exc", [RuntimeError, ValueError, MemoryError]) +def test_profiling_failure_is_not_assessed(monkeypatch: pytest.MonkeyPatch, exc) -> None: + df = pd.DataFrame({"a": ["1", "2", "x"]}) + kwargs = {"baseline": _baseline_for(df), "debt_policy": "fail", "ledger": None, + "verbose": False, "include_trust_score": False} + _, reference = fd.evaluate_quality_debt(df, **kwargs) + assert _item(reference, "type_instability").assessed + + # freshdata.profile is shadowed by the fd.profile function; patch the module. + monkeypatch.setattr(importlib.import_module("freshdata.profile"), "build_profile", + _raise(exc)) + _, gate = fd.evaluate_quality_debt(df, **kwargs) + _assert_unassessed( + gate, "type_instability", + f"column types could not be checked: profiling failed ({exc.__name__})") + assert [i.dimension for i in gate.unassessed] == ["type_instability"] + _assert_other_items_unchanged(gate, reference, "type_instability") + assert "n/a" in gate.to_html() + + +@pytest.mark.parametrize("exc", [RuntimeError, ImportError]) +def test_pii_scan_failure_is_not_assessed(monkeypatch: pytest.MonkeyPatch, exc) -> None: + df = pd.DataFrame({"email": ["a@b.com", "c@d.com"]}) + kwargs = {"baseline": _baseline_for(df), "debt_policy": "fail", "ledger": None, + "verbose": False, "include_trust_score": False} + _, reference = fd.evaluate_quality_debt(df, **kwargs) + measured = _item(reference, "pii_risk") + assert measured.assessed and measured.over # a real scan finds the email column + + monkeypatch.setattr(privacy, "detect_pii", _raise(exc)) + _, gate = fd.evaluate_quality_debt(df, **kwargs) + _assert_unassessed(gate, "pii_risk", + f"PII could not be checked: scan unavailable ({exc.__name__})") + assert [i.dimension for i in gate.unassessed] == ["pii_risk"] + _assert_other_items_unchanged(gate, reference, "pii_risk") + # The scan that never ran is not reported as a clean 0.0. + assert "0 potential PII column(s)" not in gate.summary() + + +def test_no_baseline_drift_and_churn_are_not_assessed() -> None: + df = pd.DataFrame({"id": [1, 2, 3, 4], "cat": ["a", "b", "a", "c"]}) + _, gate = fd.evaluate_quality_debt(df, debt_policy="fail", ledger=None, verbose=False) + for dimension in ("schema_drift", "category_churn"): + _assert_unassessed(gate, dimension, "no baseline supplied") + assert [i.dimension for i in gate.unassessed] == ["schema_drift", "category_churn"] + + _, with_baseline = fd.evaluate_quality_debt( + df, baseline=df.copy(), debt_policy="fail", ledger=None, verbose=False) + assert with_baseline.unassessed == [] + assert _item(with_baseline, "schema_drift").to_dict()["score"] == 0.0 + assert _item(with_baseline, "category_churn").to_dict()["score"] == 0.0 + + +def test_missing_dimension_is_not_assessed(monkeypatch: pytest.MonkeyPatch) -> None: + quality = importlib.import_module("freshdata.quality") + real = quality._score_debt + + def without_outliers(*args, **kwargs): + scores = real(*args, **kwargs) + del scores["outlier_spikes"] + return scores + + monkeypatch.setattr(quality, "_score_debt", without_outliers) + df = pd.DataFrame({"a": range(20), "b": list("xy" * 10)}) + _, gate = fd.evaluate_quality_debt(df, baseline=df.copy(), ledger=None, verbose=False) + _assert_unassessed(gate, "outlier_spikes", "dimension was not scored") + + +def test_unmeasured_dimensions_skip_the_ledger(tmp_path) -> None: + """A run without a baseline neither records nor resets drift history.""" + ledger = str(tmp_path / "debt.sqlite") + df = pd.DataFrame({"a": range(20), "b": list("xy" * 10)}) + baseline = df.drop(columns=["b"]) # "b" is added: drift 1.0 + + _, g1 = fd.evaluate_quality_debt(df, baseline=baseline, debt_policy="warn_then_fail", + ledger=ledger, verbose=False) + assert _item(g1, "schema_drift").over + _, g2 = fd.evaluate_quality_debt(df, debt_policy="warn_then_fail", ledger=ledger, + verbose=False) + drift2 = _item(g2, "schema_drift") + assert not drift2.assessed and drift2.previous == pytest.approx(1.0) + assert drift2 not in g2.warned + _, g3 = fd.evaluate_quality_debt(df, baseline=baseline, debt_policy="warn_then_fail", + ledger=ledger, verbose=False) + drift3 = _item(g3, "schema_drift") + # Compared against run 1, not a fabricated 0.0 from the baseline-less run 2. + assert drift3.previous == pytest.approx(1.0) + assert not drift3.worsening + assert g3.status == "fail" # drift over threshold again: repeated + conn = sqlite3.connect(ledger) + try: + rows = conn.execute( + "SELECT run_id, dimension, over_threshold FROM debt_items " + "WHERE dimension IN ('schema_drift', 'category_churn') ORDER BY run_id, dimension" + ).fetchall() + finally: + conn.close() + assert rows == [(1, "category_churn", 0), (1, "schema_drift", 1), + (3, "category_churn", 0), (3, "schema_drift", 1)] + + +def _parity_frame() -> pd.DataFrame: + return pd.DataFrame({ + "amount": [1.0, 2.0, None, 4.0, 100000.0, 2.0], + "name": ["x", "x", "y", None, "z", "x"], + "id": [1, 2, 3, 4, 5, 2], + "email": ["a@b.com", "c@d.com", "e@f.com", "g@h.com", "i@j.com", "a@b.com"], + "num_text": ["1", "2", "3", "x", "5", "6"], + }) + + +def test_fully_measured_run_output_unchanged() -> None: + """Profiling and PII scan succeed and a baseline is given: output as before #16.""" + baseline = _parity_frame().drop(columns=["id"]).assign(name=["x", "q", "y", None, "w", "x"]) + _, gate = fd.evaluate_quality_debt(_parity_frame(), baseline=baseline, debt_policy="warn", + ledger=None, verbose=False, include_trust_score=False) + + def row(dimension, score, threshold, over, detail): + return {"dimension": dimension, "score": score, "threshold": threshold, + "over_threshold": over, "previous": None, "worsening": False, "detail": detail} + + assert [i.to_dict() for i in gate.items] == [ + row("missingness", 0.0556, 0.1, False, "2 missing cell(s) remain"), + row("duplicates", 0.0, 0.1, False, "0 duplicate row(s) detected (0 removed)"), + row("schema_drift", 0.25, 0.0, True, "1 added, 0 removed column(s)"), + row("type_instability", 0.0, 0.1, False, "0 column(s) with unstable types"), + row("outlier_spikes", 0.1667, 0.1, True, "1 outlier(s) flagged"), + row("pii_risk", 0.1667, 0.0, True, "1 potential PII column(s)"), + row("category_churn", 0.0625, 0.1, False, "category distribution churn vs baseline"), + row("failed_repairs", 0.0, 0.0, False, "0 high-risk action(s)"), + row("human_review_backlog", 0.2, 0.0, True, "2 item(s) awaiting review"), + ] + assert gate.status == "warn" + assert gate.total_score == 0.9014 + assert gate.unassessed == [] + assert gate.summary() == "\n".join([ + "freshdata quality-debt gate: WARN (policy=warn, total debt 0.90)", + " ! schema_drift: 0.25 (threshold 0.00) — 1 added, 0 removed column(s)", + " ! human_review_backlog: 0.20 (threshold 0.00) — 2 item(s) awaiting review", + " ! outlier_spikes: 0.17 (threshold 0.10) — 1 outlier(s) flagged", + " ! pii_risk: 0.17 (threshold 0.00) — 1 potential PII column(s)", + ]) From d42a9ccd7af445387f6407b3d8d75994bf165bcb Mon Sep 17 00:00:00 2001 From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:01:19 +0530 Subject: [PATCH 2/2] fix(enterprise): do not flag uniform list or dict columns as mixed types compute_trust_score tagged an object column holding Python lists or dicts as "mixed types" and lowered consistency (66.7 for one such column out of three), because infer_dtype returns "mixed" for any object column of containers. The equivalent nested Arrow list column infers as "unknown-array" and scored 100. A column whose non-null values are all lists, all dicts or all tuples is now uniform and scores like the Arrow equivalent. Columns that mix kinds (strings with numbers, lists with scalars, lists with dicts or tuples) are still flagged. The container check only runs when infer_dtype already reports mixed, so scores for plain frames are unchanged. --- CHANGELOG.md | 5 ++ src/freshdata/enterprise/metrics.py | 30 +++++++++-- tests/test_enterprise_metrics_nested.py | 69 +++++++++++++++++++++++-- 3 files changed, 98 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 44f97797..a159b6cf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,11 @@ adheres to [Semantic Versioning](https://semver.org/). `QualityDebtGate.unassessed`, and it counts toward neither the total, the gate status nor the ledger history. Runs with a baseline and a working profile and PII scan score exactly as before. +- `compute_trust_score` no longer flags an object column as "mixed types", or + lowers consistency for it, when every non-null value is a list (or every one + a dict, or every one a tuple). Such a column now scores like the equivalent + nested Arrow column. Columns that mix kinds, such as strings with numbers or + lists with scalars, are still flagged. ## [2.1.0] - 2026-09-15 diff --git a/src/freshdata/enterprise/metrics.py b/src/freshdata/enterprise/metrics.py index a069ab7c..fe573b5b 100644 --- a/src/freshdata/enterprise/metrics.py +++ b/src/freshdata/enterprise/metrics.py @@ -12,7 +12,9 @@ :meth:`TrustScore.to_dict`) and the overall score is blended from the other three dimensions with their weights renormalised, - **consistency** — share of columns free of structural defects that - corruption can introduce (mixed types, duplicate labels). Constant columns + corruption can introduce (mixed types, duplicate labels). A column whose + non-null values are all lists, all dicts or all tuples is uniform, not + mixed, matching the nested Arrow equivalent. Constant columns are surfaced as per-column issues instead of lowering this dimension: counting them here made the score *rise* when a constant column was corrupted into varying, breaking trust monotonicity. @@ -206,18 +208,40 @@ def _column_validity( invalid += n_out issues.append(f"{n_out} outlier") - if infer_dtype(s, skipna=True) in ("mixed", "mixed-integer"): + if _has_mixed_types(s): issues.append("mixed types") return min(invalid, non_null), issues +#: Container types that make up a uniform nested column when every non-null +#: value is one of them. +_CONTAINER_TYPES = (list, dict, tuple) + + +def _has_mixed_types(s: pd.Series) -> bool: + """True when the column's non-null values are of genuinely different kinds. + + ``infer_dtype`` reports ``"mixed"`` for any object column of lists, dicts or + tuples, even when every value is a list, while the nested Arrow equivalent + infers as ``"unknown-array"``. A column whose non-null values are all lists + (or all dicts, or all tuples) is uniform, so it is not mixed; lists next to + scalars, strings or dicts still are. + """ + if infer_dtype(s, skipna=True) not in ("mixed", "mixed-integer"): + return False + values = s.dropna() + return not any( + all(isinstance(v, kind) for v in values) for kind in _CONTAINER_TYPES + ) + + def _is_structurally_inconsistent(s: pd.Series, n_rows: int) -> bool: # Only defects that corruption can *introduce* may lower consistency. # A constant column is suspicious but corruption clears it (the column # starts varying), so counting it here made the trust score rise after # corruption; it is surfaced as a per-column issue instead. del n_rows - return infer_dtype(s, skipna=True) in ("mixed", "mixed-integer") + return _has_mixed_types(s) def _is_constant(s: pd.Series, n_rows: int) -> bool: diff --git a/tests/test_enterprise_metrics_nested.py b/tests/test_enterprise_metrics_nested.py index 5006d7d6..1205c791 100644 --- a/tests/test_enterprise_metrics_nested.py +++ b/tests/test_enterprise_metrics_nested.py @@ -127,10 +127,10 @@ def test_object_list_column_matches_nested_fallback(): score = compute_trust_score(df) assert score.completeness == 100.0 assert score.validity == 100.0 - assert score.consistency == 50.0 # "tags" infers as mixed + assert score.consistency == 100.0 # every "tags" value is a list: uniform _assert_uniqueness_unknown(score) - # (0.3 * 100 + 0.3 * 100 + 0.2 * 50) / 0.8; previously 90.0 with uniqueness 100. - assert score.overall == pytest.approx(87.5) + # (0.3 * 100 + 0.3 * 100 + 0.2 * 100) / 0.8, the same as the nested Arrow column. + assert score.overall == pytest.approx(100.0) by_name = {c.name: c for c in score.columns} assert "constant column" not in by_name["tags"].issues same = pd.Series([["x"], ["x"], ["x"]]) @@ -196,3 +196,66 @@ def test_clean_enterprise_arrow_nested_column(kind): payload = json.loads(result.to_json()) assert payload["trust_after"]["n_rows"] == 3 assert payload["trust_before"]["dimensions"]["uniqueness"] is None + + +# -- Uniform list / dict / tuple columns are not "mixed types" (#3) ----------- + +_UNIFORM_CONTAINERS = { + "lists": [[1, 2], [3], None, [4, 5]], + "empty_lists": [[], [1], None, []], + "dicts": [{"a": 1}, {"b": 2}, None, {"c": 3}], + "tuples": [(1, 2), (3,), None, (4, 5)], +} + + +def _with_tags(values: list) -> pd.DataFrame: + return pd.DataFrame({"a": [1, 2, 3, 4], "b": ["x", "y", "z", "w"], + "tags": pd.Series(values, dtype=object)}) + + +@pytest.mark.parametrize("kind", sorted(_UNIFORM_CONTAINERS)) +def test_uniform_container_column_is_consistent(kind): + score = compute_trust_score(_with_tags(_UNIFORM_CONTAINERS[kind])) + assert score.consistency == 100.0 + by_name = {c.name: c for c in score.columns} + assert "mixed types" not in by_name["tags"].issues + + +def test_object_list_column_scores_like_arrow_list(): + pa = pytest.importorskip("pyarrow") + if not hasattr(pd, "ArrowDtype"): + pytest.skip("pd.ArrowDtype is not available in this pandas version") + values = [[1, 2], [3], [4, 5], [6]] + obj = _with_tags(values) + try: + arrow_tags = pd.Series(values, dtype=pd.ArrowDtype(pa.list_(pa.int64()))) + except (TypeError, ValueError, NotImplementedError) as exc: # pragma: no cover + pytest.skip(f"nested ArrowDtype unsupported here: {exc}") + arrow = obj.assign(tags=arrow_tags) + obj_score, arrow_score = compute_trust_score(obj), compute_trust_score(arrow) + assert obj_score.consistency == arrow_score.consistency == 100.0 + assert obj_score.overall == pytest.approx(arrow_score.overall) + assert obj_score.overall == pytest.approx(100.0) # was 91.7 with "mixed types" + for score in (obj_score, arrow_score): + by_name = {c.name: c for c in score.columns} + assert by_name["tags"].issues == (_UNHASHABLE,) + + +@pytest.mark.parametrize( + "values", + [ + [[1], 2, [3], [4]], + [[1], "a", [3], [4]], + [{"a": 1}, [1], {"b": 2}, {"c": 3}], + [[1], (2,), [3], [4]], + [1, "a", 2.0, "b"], + ], + ids=["lists-and-ints", "lists-and-strings", "dicts-and-lists", "lists-and-tuples", + "strings-and-numbers"], +) +def test_mixed_kinds_still_flagged(values): + score = compute_trust_score(pd.DataFrame({"a": [1, 2, 3, 4], + "m": pd.Series(values, dtype=object)})) + assert score.consistency == 50.0 + by_name = {c.name: c for c in score.columns} + assert "mixed types" in by_name["m"].issues