Skip to content

Commit b2901ff

Browse files
test(ingestion): assert the fixed Excel behaviour, not the defect (#485)
main went red at c0fbb08: 7064 passed, 2 failed. #478 found that preserve_leading_zeros was a no-op for clean_excel and, following the repo's convention of not using xfail, pinned the defect as it behaved: assert list(fd.clean_excel(xlsx)["zip"]) == [2134, 501, 10001] # DEFECT #480 then fixed that defect. Each PR was green on its own branch, and each was rebased on a main that did not contain the other, so nothing caught the pair until both had landed. The two assertions were tripwires for the fix, and the fix duly tripped them. Both now assert the repaired behaviour, and each docstring keeps the history so the tests still explain why they exist. The round-trip case gains nothing artificial -- it simply asserts that the Excel hop no longer undoes a correct CSV clean, which is the property that was missing in the first place. The opt-out and explicit-dtype paths are asserted alongside, so the fix cannot drift into applying unconditionally. Full suite 7066 passed / 0 failed, coverage 94.21%; ruff clean repo-wide.
1 parent c0fbb08 commit b2901ff

1 file changed

Lines changed: 19 additions & 16 deletions

File tree

‎tests/test_ingestion_roundtrip.py‎

Lines changed: 19 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -320,14 +320,15 @@ def test_sanitize_formulas_false_round_trips_the_values_byte_exactly(tmp_path):
320320
# --------------------------------------------------------------------------
321321

322322

323-
def test_excel_ingestion_drops_leading_zeros_that_the_csv_path_preserves(tmp_path):
324-
"""DEFECT: ``preserve_leading_zeros`` is a no-op for ``clean_excel``.
325-
326-
``clean_csv`` pre-scans the file (``_csv_io.leading_zero_dtypes``) so a
327-
zero-padded numeric column is read as text. ``clean_excel`` calls
328-
``pd.read_excel`` directly, and pandas' type inference turns the *text* cell
329-
``"02134"`` into the integer ``2134`` — silently, with no report entry, even
330-
though ``clean_excel`` documents "the same options" as ``clean_csv``.
323+
def test_excel_ingestion_preserves_the_leading_zeros_the_csv_path_keeps(tmp_path):
324+
"""``clean_excel`` honours ``preserve_leading_zeros``, like ``clean_csv``.
325+
326+
This test was originally written to pin the defect: ``clean_excel`` called
327+
``pd.read_excel`` directly, so pandas' type inference turned the *text* cell
328+
``"02134"`` into the integer ``2134`` -- silently, with no report entry, even
329+
though ``clean_excel`` documents "the same options" as ``clean_csv``. The
330+
fix gave ``clean_excel`` the ``read_excel`` counterpart of the CSV pre-scan,
331+
so the assertion is now the fixed behaviour rather than the defect.
331332
"""
332333
rows = [["zip", "city"], ["02134", "Boston"], ["00501", "Holtsville"], ["10001", "NY"]]
333334
xlsx = _write_xlsx(tmp_path / "zips.xlsx", rows)
@@ -339,17 +340,19 @@ def test_excel_ingestion_drops_leading_zeros_that_the_csv_path_preserves(tmp_pat
339340
stored = [c.value for c in openpyxl.load_workbook(xlsx).active["A"]]
340341
assert stored == ["zip", "02134", "00501", "10001"]
341342

342-
assert list(fd.clean_csv(csv_path)["zip"]) == ["02134", "00501", "10001"]
343-
assert list(fd.clean_excel(xlsx)["zip"]) == [2134, 501, 10001] # DEFECT
344-
assert list(fd.clean_excel(xlsx, preserve_leading_zeros=True)["zip"]) == [2134, 501, 10001]
343+
padded = ["02134", "00501", "10001"]
344+
assert list(fd.clean_csv(csv_path)["zip"]) == padded
345+
assert list(fd.clean_excel(xlsx)["zip"]) == padded
346+
assert list(fd.clean_excel(xlsx, preserve_leading_zeros=True)["zip"]) == padded
345347

346-
# The supported workaround, which callers must know to reach for.
348+
# An explicit dtype still wins, and opting out still opts out.
347349
forced = fd.clean_excel(xlsx, read_excel_kwargs={"dtype": {"zip": str}})
348-
assert list(forced["zip"]) == ["02134", "00501", "10001"]
350+
assert list(forced["zip"]) == padded
351+
assert list(fd.clean_excel(xlsx, preserve_leading_zeros=False)["zip"]) == [2134, 501, 10001]
349352

350353

351-
def test_csv_to_excel_round_trip_loses_the_leading_zeros_csv_had_kept(tmp_path):
352-
"""The same defect seen end to end: a correct CSV clean is undone by the Excel hop."""
354+
def test_csv_to_excel_round_trip_keeps_the_leading_zeros_csv_had_kept(tmp_path):
355+
"""The same path end to end: the Excel hop no longer undoes a correct CSV clean."""
353356
csv_path = _write_csv(tmp_path / "zips.csv", "zip,v\n02134,1\n00501,2\n10001,3\n")
354357
xlsx = tmp_path / "zips.xlsx"
355358

@@ -358,7 +361,7 @@ def test_csv_to_excel_round_trip_loses_the_leading_zeros_csv_had_kept(tmp_path):
358361
second = fd.clean_excel(xlsx)
359362

360363
assert list(first["zip"]) == ["02134", "00501", "10001"]
361-
assert list(second["zip"]) == [2134, 501, 10001] # DEFECT: silent, no report entry
364+
assert list(second["zip"]) == ["02134", "00501", "10001"]
362365

363366

364367
def test_leading_zero_prescan_stops_at_the_documented_row_limit(tmp_path):

0 commit comments

Comments
 (0)