Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,22 @@ adheres to [Semantic Versioning](https://semver.org/).

## [Unreleased]

### Fixed
- `fd.evaluate_quality_debt` no longer scores a dimension a clean 0.0 when it
was never measured. `type_instability` when profiling fails, `pii_risk` when
the PII scan is unavailable or fails, and `schema_drift` and `category_churn`
when no `baseline=` is passed are now not assessed, using the same mechanism
as undetectable duplicates (#414): `score` and `over_threshold` serialise as
`None`, the detail says why, the item is listed in
`QualityDebtGate.unassessed`, and it counts toward neither the total, the gate
status nor the ledger history. Runs with a baseline and a working profile and
PII scan score exactly as before.
- `compute_trust_score` no longer flags an object column as "mixed types", or
lowers consistency for it, when every non-null value is a list (or every one
a dict, or every one a tuple). Such a column now scores like the equivalent
nested Arrow column. Columns that mix kinds, such as strings with numbers or
lists with scalars, are still flagged.

## [2.1.0] - 2026-09-15

### Security
Expand Down
19 changes: 13 additions & 6 deletions docs/decision-workflow.md
Original file line number Diff line number Diff line change
Expand Up @@ -76,12 +76,19 @@ review backlog), persists the history to SQLite, and **escalates warn→fail whe
an issue repeats or worsens** across runs.

The duplicates dimension counts duplicate rows left in the cleaned output (or
the rows removed, if more). When duplicates cannot be checked because a column
holds unhashable values (lists, dicts, or nested Arrow list/struct/map columns),
the dimension is **not assessed** rather than scored clean: `to_dict()` gives
`score` and `over_threshold` as `None`, the detail names the columns,
`gate.unassessed` lists it, and `summary()` prints a `? duplicates: not
assessed` line. An unassessed dimension adds nothing to the total, never
the rows removed, if more). A dimension that could not be measured is **not
assessed** rather than scored clean. That happens when:

- duplicates cannot be checked because a column holds unhashable values (lists,
dicts, or nested Arrow list/struct/map columns); the detail names the columns,
- profiling fails, for `type_instability`,
- the PII scan is unavailable or fails, for `pii_risk`,
- no `baseline=` is passed, for `schema_drift` and `category_churn`, which only
have something to compare against when a baseline is supplied.

For an unassessed dimension `to_dict()` gives `score` and `over_threshold` as
`None`, the detail says why, `gate.unassessed` lists it, and `summary()` prints
a `? <dimension>: not assessed` line. It adds nothing to the total, never
changes the gate status on its own, and is not written to the ledger, so
escalation compares against the last run that measured it.

Expand Down
30 changes: 27 additions & 3 deletions src/freshdata/enterprise/metrics.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,9 @@
:meth:`TrustScore.to_dict`) and the overall score is blended from the other
three dimensions with their weights renormalised,
- **consistency** — share of columns free of structural defects that
corruption can introduce (mixed types, duplicate labels). Constant columns
corruption can introduce (mixed types, duplicate labels). A column whose
non-null values are all lists, all dicts or all tuples is uniform, not
mixed, matching the nested Arrow equivalent. Constant columns
are surfaced as per-column issues instead of lowering this dimension:
counting them here made the score *rise* when a constant column was
corrupted into varying, breaking trust monotonicity.
Expand Down Expand Up @@ -206,18 +208,40 @@ def _column_validity(
invalid += n_out
issues.append(f"{n_out} outlier")

if infer_dtype(s, skipna=True) in ("mixed", "mixed-integer"):
if _has_mixed_types(s):
issues.append("mixed types")
return min(invalid, non_null), issues


#: Container types that make up a uniform nested column when every non-null
#: value is one of them.
_CONTAINER_TYPES = (list, dict, tuple)


def _has_mixed_types(s: pd.Series) -> bool:
"""True when the column's non-null values are of genuinely different kinds.

``infer_dtype`` reports ``"mixed"`` for any object column of lists, dicts or
tuples, even when every value is a list, while the nested Arrow equivalent
infers as ``"unknown-array"``. A column whose non-null values are all lists
(or all dicts, or all tuples) is uniform, so it is not mixed; lists next to
scalars, strings or dicts still are.
"""
if infer_dtype(s, skipna=True) not in ("mixed", "mixed-integer"):
return False
values = s.dropna()
return not any(
all(isinstance(v, kind) for v in values) for kind in _CONTAINER_TYPES
)


def _is_structurally_inconsistent(s: pd.Series, n_rows: int) -> bool:
# Only defects that corruption can *introduce* may lower consistency.
# A constant column is suspicious but corruption clears it (the column
# starts varying), so counting it here made the trust score rise after
# corruption; it is surfaced as a per-column issue instead.
del n_rows
return infer_dtype(s, skipna=True) in ("mixed", "mixed-integer")
return _has_mixed_types(s)


def _is_constant(s: pd.Series, n_rows: int) -> bool:
Expand Down
27 changes: 17 additions & 10 deletions src/freshdata/quality.py
Original file line number Diff line number Diff line change
Expand Up @@ -49,8 +49,9 @@ class DebtItem:
detail: str
previous: float | None = None
#: False when the dimension could not be measured on this run (for example
#: duplicate rows in a frame with unhashable cells). An unassessed item is
#: neither over threshold nor evidence of a clean result: it serialises
#: duplicate rows in a frame with unhashable cells, a failed profile or PII
#: scan, or schema drift and category churn with no baseline). An unassessed
#: item is neither over threshold nor evidence of a clean result: it serialises
#: with ``score`` and ``over_threshold`` as ``None``, adds nothing to the
#: total, never drives the gate status and is not written to the ledger.
assessed: bool = True
Expand Down Expand Up @@ -235,8 +236,10 @@ def _score_debt(
if c.suggested_dtype and c.suggested_dtype != c.dtype)
out["type_instability"] = (retype / max(1, report.cols_after),
f"{retype} column(s) with unstable types")
except Exception: # pragma: no cover - profiling is best-effort
out["type_instability"] = (0.0, "not assessed")
except Exception as exc: # profiling is best-effort
# A failed profile measured nothing, so it is not a clean result.
out["type_instability"] = (
None, f"column types could not be checked: profiling failed ({type(exc).__name__})")

# PII risk (best-effort, lazy enterprise import).
try:
Expand All @@ -248,10 +251,13 @@ def _score_debt(
n_pii = len([col for col in scan.by_column() if col])
out["pii_risk"] = (min(1.0, n_pii / max(1, report.cols_after)),
f"{n_pii} potential PII column(s)")
except Exception:
out["pii_risk"] = (0.0, "PII scan unavailable")
except Exception as exc:
# No scan ran (detector missing or failing): unknown, not "no PII".
out["pii_risk"] = (
None, f"PII could not be checked: scan unavailable ({type(exc).__name__})")

# Schema drift + category churn need a baseline.
# Schema drift + category churn need a baseline. Without one there is
# nothing to compare against, so both are unassessed rather than clean.
if baseline is not None:
added = set(map(str, df.columns)) - set(map(str, baseline.columns))
removed = set(map(str, baseline.columns)) - set(map(str, df.columns))
Expand All @@ -261,8 +267,8 @@ def _score_debt(
churn = _category_churn(baseline, df)
out["category_churn"] = (churn, "category distribution churn vs baseline")
else:
out["schema_drift"] = (0.0, "no baseline supplied")
out["category_churn"] = (0.0, "no baseline supplied")
out["schema_drift"] = (None, "no baseline supplied")
out["category_churn"] = (None, "no baseline supplied")

return out

Expand Down Expand Up @@ -355,6 +361,7 @@ def evaluate_quality_debt(
``None`` keeps the run in memory only (no escalation history).
baseline:
Optional prior frame enabling schema-drift and category-churn scoring.
Without it both dimensions are reported as not assessed.
thresholds:
Per-dimension overrides of the default "in debt" thresholds.
**clean_options:
Expand All @@ -381,7 +388,7 @@ def evaluate_quality_debt(

items: list[DebtItem] = []
for dim in DEBT_DIMENSIONS:
score, detail = scores.get(dim, (0.0, "not assessed"))
score, detail = scores.get(dim, (None, "dimension was not scored"))
items.append(DebtItem(dim, 0.0 if score is None else score, thr[dim], detail,
previous.get(dim), assessed=score is not None))

Expand Down
69 changes: 66 additions & 3 deletions tests/test_enterprise_metrics_nested.py
Original file line number Diff line number Diff line change
Expand Up @@ -127,10 +127,10 @@ def test_object_list_column_matches_nested_fallback():
score = compute_trust_score(df)
assert score.completeness == 100.0
assert score.validity == 100.0
assert score.consistency == 50.0 # "tags" infers as mixed
assert score.consistency == 100.0 # every "tags" value is a list: uniform
_assert_uniqueness_unknown(score)
# (0.3 * 100 + 0.3 * 100 + 0.2 * 50) / 0.8; previously 90.0 with uniqueness 100.
assert score.overall == pytest.approx(87.5)
# (0.3 * 100 + 0.3 * 100 + 0.2 * 100) / 0.8, the same as the nested Arrow column.
assert score.overall == pytest.approx(100.0)
by_name = {c.name: c for c in score.columns}
assert "constant column" not in by_name["tags"].issues
same = pd.Series([["x"], ["x"], ["x"]])
Expand Down Expand Up @@ -196,3 +196,66 @@ def test_clean_enterprise_arrow_nested_column(kind):
payload = json.loads(result.to_json())
assert payload["trust_after"]["n_rows"] == 3
assert payload["trust_before"]["dimensions"]["uniqueness"] is None


# -- Uniform list / dict / tuple columns are not "mixed types" (#3) -----------

_UNIFORM_CONTAINERS = {
"lists": [[1, 2], [3], None, [4, 5]],
"empty_lists": [[], [1], None, []],
"dicts": [{"a": 1}, {"b": 2}, None, {"c": 3}],
"tuples": [(1, 2), (3,), None, (4, 5)],
}


def _with_tags(values: list) -> pd.DataFrame:
return pd.DataFrame({"a": [1, 2, 3, 4], "b": ["x", "y", "z", "w"],
"tags": pd.Series(values, dtype=object)})


@pytest.mark.parametrize("kind", sorted(_UNIFORM_CONTAINERS))
def test_uniform_container_column_is_consistent(kind):
score = compute_trust_score(_with_tags(_UNIFORM_CONTAINERS[kind]))
assert score.consistency == 100.0
by_name = {c.name: c for c in score.columns}
assert "mixed types" not in by_name["tags"].issues


def test_object_list_column_scores_like_arrow_list():
pa = pytest.importorskip("pyarrow")
if not hasattr(pd, "ArrowDtype"):
pytest.skip("pd.ArrowDtype is not available in this pandas version")
values = [[1, 2], [3], [4, 5], [6]]
obj = _with_tags(values)
try:
arrow_tags = pd.Series(values, dtype=pd.ArrowDtype(pa.list_(pa.int64())))
except (TypeError, ValueError, NotImplementedError) as exc: # pragma: no cover
pytest.skip(f"nested ArrowDtype unsupported here: {exc}")
arrow = obj.assign(tags=arrow_tags)
obj_score, arrow_score = compute_trust_score(obj), compute_trust_score(arrow)
assert obj_score.consistency == arrow_score.consistency == 100.0
assert obj_score.overall == pytest.approx(arrow_score.overall)
assert obj_score.overall == pytest.approx(100.0) # was 91.7 with "mixed types"
for score in (obj_score, arrow_score):
by_name = {c.name: c for c in score.columns}
assert by_name["tags"].issues == (_UNHASHABLE,)


@pytest.mark.parametrize(
"values",
[
[[1], 2, [3], [4]],
[[1], "a", [3], [4]],
[{"a": 1}, [1], {"b": 2}, {"c": 3}],
[[1], (2,), [3], [4]],
[1, "a", 2.0, "b"],
],
ids=["lists-and-ints", "lists-and-strings", "dicts-and-lists", "lists-and-tuples",
"strings-and-numbers"],
)
def test_mixed_kinds_still_flagged(values):
score = compute_trust_score(pd.DataFrame({"a": [1, 2, 3, 4],
"m": pd.Series(values, dtype=object)}))
assert score.consistency == 50.0
by_name = {c.name: c for c in score.columns}
assert "mixed types" in by_name["m"].issues
Loading
Loading