From 34b4915f9603f6ef3308ace925c560698bae3101 Mon Sep 17 00:00:00 2001 From: Kevin Costner <120246174+kevincostner17@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:27:49 +0530 Subject: [PATCH] fix(enterprise): handle nested Arrow columns in trust scoring compute_trust_score, and therefore clean_enterprise, crashed with ArrowNotImplementedError on a frame with a nested Arrow column (list, large_list, struct or map). _is_constant calls Series.nunique and the uniqueness dimension calls DataFrame.duplicated; for nested Arrow dtypes pyarrow has no unique/dictionary_encode kernel and raises ArrowNotImplementedError, a NotImplementedError subclass. Both guards only caught TypeError, which is what object columns holding lists raise. Catch NotImplementedError alongside TypeError so nested Arrow columns follow the existing unhashable-cell path: the column is not reported as constant and uniqueness falls back to 100. Scores for frames without nested columns are unchanged. --- src/freshdata/enterprise/metrics.py | 7 +- tests/test_enterprise_metrics_nested.py | 129 ++++++++++++++++++++++++ 2 files changed, 134 insertions(+), 2 deletions(-) create mode 100644 tests/test_enterprise_metrics_nested.py diff --git a/src/freshdata/enterprise/metrics.py b/src/freshdata/enterprise/metrics.py index f4ea419c..63f1aab9 100644 --- a/src/freshdata/enterprise/metrics.py +++ b/src/freshdata/enterprise/metrics.py @@ -204,7 +204,10 @@ def _is_structurally_inconsistent(s: pd.Series, n_rows: int) -> bool: def _is_constant(s: pd.Series, n_rows: int) -> bool: try: return n_rows > 1 and int(s.nunique(dropna=True)) <= 1 - except TypeError: + except (TypeError, NotImplementedError): + # Unhashable cells cannot be counted: object lists/dicts raise + # TypeError, nested Arrow dtypes (list/struct/map) raise + # ArrowNotImplementedError, a NotImplementedError subclass. return False @@ -270,7 +273,7 @@ def compute_trust_score( try: dup_rows = int(frame.duplicated().sum()) uniqueness = 100.0 * (1.0 - dup_rows / n_rows) if n_rows else 100.0 - except TypeError: # pragma: no cover - unhashable cells (rare; mirrors profile.py) + except (TypeError, NotImplementedError): # lists/dicts, nested Arrow: undetectable uniqueness = 100.0 dup_labels = int(frame.columns.duplicated().sum()) diff --git a/tests/test_enterprise_metrics_nested.py b/tests/test_enterprise_metrics_nested.py new file mode 100644 index 00000000..aaea69ab --- /dev/null +++ b/tests/test_enterprise_metrics_nested.py @@ -0,0 +1,129 @@ +"""Trust scoring on frames with nested / unhashable cell values. + +Object columns holding Python lists make ``Series.nunique`` and +``DataFrame.duplicated`` raise ``TypeError``; nested Arrow dtypes (list, +large_list, struct, map) raise ``pyarrow.lib.ArrowNotImplementedError``, a +``NotImplementedError`` subclass, because they cannot be hashed or +dictionary-encoded. Both mean "cannot compute": the column is not reported as +constant and uniqueness falls back to 100 instead of crashing. +""" + +from __future__ import annotations + +import json + +import pandas as pd +import pytest + +from freshdata.enterprise import clean_enterprise, compute_trust_score +from freshdata.enterprise.metrics import _is_constant + + +def _arrow_nested_series(kind: str, values: list) -> pd.Series: + pa = pytest.importorskip("pyarrow") + if not hasattr(pd, "ArrowDtype"): + pytest.skip("pd.ArrowDtype is not available in this pandas version") + types = { + "list": lambda: pa.list_(pa.string()), + "large_list": lambda: pa.large_list(pa.string()), + "struct": lambda: pa.struct([("k", pa.int64()), ("v", pa.string())]), + "map": lambda: pa.map_(pa.string(), pa.int64()), + } + try: + return pd.Series(pd.array(values, dtype=pd.ArrowDtype(types[kind]()))) + except (TypeError, ValueError, NotImplementedError) as exc: # pragma: no cover + pytest.skip(f"nested ArrowDtype {kind!r} unsupported here: {exc}") + + +# Two distinct nested payloads per kind; rows 2 and 3 of each frame repeat. +_PAYLOADS = { + "list": (["x"], ["y", "z"]), + "large_list": (["x"], ["y", "z"]), + "struct": ({"k": 1, "v": "x"}, {"k": 2, "v": "y"}), + "map": ([("k", 1)], [("j", 2)]), +} +_KINDS = sorted(_PAYLOADS) + + +def _nested_frame(kind: str) -> pd.DataFrame: + first, second = _PAYLOADS[kind] + return pd.DataFrame({ + "a": [1, 2, 2], + "tags": _arrow_nested_series(kind, [first, second, second]), + }) + + +def test_trust_score_arrow_list_column_repro(): + pa = pytest.importorskip("pyarrow") + if not hasattr(pd, "ArrowDtype"): + pytest.skip("pd.ArrowDtype is not available in this pandas version") + tags = pd.array([["x"], ["y", "z"], ["y", "z"]], + dtype=pd.ArrowDtype(pa.list_(pa.string()))) + df = pd.DataFrame({"a": [1, 2, 2], "tags": tags}) + score = compute_trust_score(df) + assert score.n_rows == 3 + assert score.completeness == 100.0 + assert score.uniqueness == 100.0 # duplicate detection impossible + assert json.dumps(score.to_dict()) + + +@pytest.mark.parametrize("kind", _KINDS) +def test_trust_score_arrow_nested_column(kind): + df = _nested_frame(kind) + snapshot = df.copy(deep=True) + score = compute_trust_score(df) + assert (score.n_rows, score.n_cols) == (3, 2) + assert score.completeness == 100.0 + assert score.validity == 100.0 + assert score.uniqueness == 100.0 + assert 0.0 <= score.overall <= 100.0 + by_name = {c.name: c for c in score.columns} + assert "constant column" not in by_name["tags"].issues + assert str(score) + assert json.dumps(score.to_dict()) + pd.testing.assert_frame_equal(df, snapshot) # scoring never mutates + + +@pytest.mark.parametrize("kind", _KINDS) +def test_identical_nested_values_are_not_reported_constant(kind): + first, _ = _PAYLOADS[kind] + s = _arrow_nested_series(kind, [first, first, first]) + assert _is_constant(s, len(s)) is False + + +def test_object_list_column_matches_nested_fallback(): + df = pd.DataFrame({"a": [1, 2, 2], "tags": [["x"], ["y", "z"], ["y", "z"]]}) + score = compute_trust_score(df) + assert score.completeness == 100.0 + assert score.validity == 100.0 + assert score.uniqueness == 100.0 + by_name = {c.name: c for c in score.columns} + assert "constant column" not in by_name["tags"].issues + same = pd.Series([["x"], ["x"], ["x"]]) + assert _is_constant(same, len(same)) is False + + +def test_plain_frame_scores_unchanged(): + df = pd.DataFrame({ + "a": [1, 2, 2, None], + "b": ["x", "x", "x", "x"], + "m": [1, "a", "a", 2.5], + }) + score = compute_trust_score(df) + assert score.completeness == pytest.approx(100.0 * 11 / 12) + assert score.validity == 100.0 + assert score.uniqueness == 75.0 # row 3 repeats row 2 + assert score.consistency == pytest.approx(100.0 * 2 / 3) # "m" is mixed + by_name = {c.name: c for c in score.columns} + assert by_name["b"].issues == ("constant column",) + assert by_name["m"].issues == ("mixed types",) + assert by_name["a"].issues == () + + +@pytest.mark.parametrize("kind", _KINDS) +def test_clean_enterprise_arrow_nested_column(kind): + df = _nested_frame(kind) + result = clean_enterprise(df, verbose=False) + assert result.trust_before.uniqueness == 100.0 + assert result.trust_after.n_rows == 3 + assert json.loads(result.to_json())["trust_after"]["n_rows"] == 3