|
| 1 | +"""#305: ``24:00`` in a date-less time column is suggested, never auto-applied. |
| 2 | +
|
| 3 | +Without a date, ``24:00`` (end of day) and ``00:00`` (start of day) are |
| 4 | +different clock times, so the rewrite is held for review in every mode. |
| 5 | +""" |
| 6 | + |
| 7 | +from __future__ import annotations |
| 8 | + |
| 9 | +import pandas as pd |
| 10 | +import pytest |
| 11 | + |
| 12 | +import freshdata as fd |
| 13 | +from freshdata.config import CleanConfig |
| 14 | +from freshdata.semantic.canonical import TimeCanonicalExpert |
| 15 | +from freshdata.semantic.context import build_semantic_context |
| 16 | + |
| 17 | +_RATIONALE = ( |
| 18 | + "24:00 marks end of day; in a time-only column 00:00 would read as start " |
| 19 | + "of day, so this is held for review" |
| 20 | +) |
| 21 | + |
| 22 | + |
| 23 | +def _shift_frame() -> pd.DataFrame: |
| 24 | + return pd.DataFrame( |
| 25 | + { |
| 26 | + "shift_start": ["14:00", "15:00", "13:30", "16:00", "12:00"], |
| 27 | + "shift_end": ["22:00", "23:00", "21:30", "24:00", "20:00"], |
| 28 | + } |
| 29 | + ) |
| 30 | + |
| 31 | + |
| 32 | +def _time_actions(report: fd.CleanReport, column: str) -> list[fd.Action]: |
| 33 | + return [ |
| 34 | + a |
| 35 | + for a in report.actions |
| 36 | + if a.step == "semantic" |
| 37 | + and a.column == column |
| 38 | + and a.metadata.get("expert") == "time_canonical" |
| 39 | + ] |
| 40 | + |
| 41 | + |
| 42 | +@pytest.mark.parametrize("mode", ["auto", "review", "assist"]) |
| 43 | +def test_issue_305_repro_keeps_24_00_and_suggests(mode: str) -> None: |
| 44 | + df = _shift_frame() |
| 45 | + out, report = fd.clean(df, semantic_mode=mode, return_report=True, verbose=False) |
| 46 | + |
| 47 | + assert out["shift_end"].iloc[3] == "24:00" |
| 48 | + assert out["shift_end"].tolist() == df["shift_end"].tolist() |
| 49 | + actions = _time_actions(report, "shift_end") |
| 50 | + assert [a.status for a in actions] == ["suggested"] |
| 51 | + assert actions[0].rationale == _RATIONALE |
| 52 | + assert actions[0].metadata.get("raw_value") == "24:00" |
| 53 | + |
| 54 | + |
| 55 | +@pytest.mark.parametrize("raw, proposed", [("24:00", "00:00"), ("24:00:00", "00:00:00")]) |
| 56 | +def test_proposal_scores_between_review_and_auto_thresholds(raw: str, proposed: str) -> None: |
| 57 | + df = pd.DataFrame({"end_time": ["08:00", "09:30", "10:15", "11:00", raw]}) |
| 58 | + config = CleanConfig(semantic_mode="auto") |
| 59 | + ctx = build_semantic_context(df, config) |
| 60 | + info = ctx.columns["end_time"] |
| 61 | + expert = TimeCanonicalExpert() |
| 62 | + assert expert.applies(info) |
| 63 | + |
| 64 | + proposals = expert.propose(df["end_time"], info) |
| 65 | + assert len(proposals) == 1 |
| 66 | + proposal = proposals[0] |
| 67 | + assert proposal.raw_value == raw |
| 68 | + assert proposal.proposed_value == proposed |
| 69 | + assert config.semantic_review_threshold <= proposal.confidence |
| 70 | + assert proposal.confidence < config.semantic_auto_threshold |
| 71 | + assert proposal.rationale == _RATIONALE |
| 72 | + assert "instant is unchanged" not in proposal.rationale |
| 73 | + |
| 74 | + |
| 75 | +def test_seconds_form_is_kept_in_auto_mode() -> None: |
| 76 | + df = pd.DataFrame({"end_time": ["08:00:00", "09:30:00", "10:15:00", "11:00:00", "24:00:00"]}) |
| 77 | + out, report = fd.clean(df, semantic_mode="auto", return_report=True, verbose=False) |
| 78 | + assert out["end_time"].iloc[4] == "24:00:00" |
| 79 | + assert [a.status for a in _time_actions(report, "end_time")] == ["suggested"] |
0 commit comments