Skip to content

Commit 8a96f97

Browse files
Merge pull request #123 from FreshCode-Org/fix/csv-formula-injection-jwd
fix: neutralize CSV formula injection in spreadsheet-bound exports (OWASP)
2 parents 613dc18 + 7c1c34a commit 8a96f97

6 files changed

Lines changed: 251 additions & 9 deletions

File tree

‎CHANGELOG.md‎

Lines changed: 8 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -39,6 +39,14 @@ adheres to [Semantic Versioning](https://semver.org/).
3939
materialization — pipeline stages currently collect intermediates
4040
eagerly, so peak memory during cleaning matches eager output. The DuckDB
4141
handle path is the measured lower-peak-memory route (#52, #53).
42+
- **CSV formula-injection protection** (OWASP): `export_review_queue` now
43+
neutralizes spreadsheet formula payloads in CSV exports **by default**
44+
(string cells and column labels starting with `= + - @ <tab> <cr>` get a
45+
leading `'`; opt out with `sanitize_formulas=False`) — review queues are
46+
built to be opened by humans in spreadsheets. `fd.clean_csv` and the
47+
streaming CLI keep byte-exact output by default and gain an explicit
48+
opt-in (`sanitize_formulas=True` / `--sanitize-formulas`) covering the
49+
cleaned output and the quarantine export. JSONL/Parquet are never altered.
4250

4351
### Added
4452
- `benchmarks/bench_outofcore.py`: subprocess-isolated peak-RSS evidence for

‎src/freshdata/_util.py‎

Lines changed: 28 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -82,3 +82,31 @@ def stringlike_columns(df: pd.DataFrame) -> list:
8282

8383
def _is_stringlike_dtype(dtype: object) -> bool:
8484
return pd.api.types.is_object_dtype(dtype) or isinstance(dtype, pd.StringDtype)
85+
86+
87+
#: Leading characters Excel/Sheets/LibreOffice interpret as a formula
88+
#: (OWASP CSV-injection guidance).
89+
_FORMULA_PREFIXES = ("=", "+", "-", "@", "\t", "\r")
90+
91+
92+
def _formula_guard(value: object) -> object:
93+
if isinstance(value, str) and value.startswith(_FORMULA_PREFIXES):
94+
return "'" + value
95+
return value
96+
97+
98+
def sanitize_csv_formulas(df: pd.DataFrame) -> pd.DataFrame:
99+
"""Copy of *df* safe to open in a spreadsheet: string cells (and column
100+
labels) starting with ``= + - @ <tab> <cr>`` are prefixed with ``'`` so
101+
they render as text instead of executing as formulas. Non-string cells
102+
(including negative numbers) are untouched.
103+
"""
104+
out = df.copy()
105+
for i, dtype in enumerate(out.dtypes):
106+
if _is_stringlike_dtype(dtype) or isinstance(dtype, pd.CategoricalDtype):
107+
column = out.iloc[:, i]
108+
guarded = column.astype(object).map(_formula_guard)
109+
if not guarded.equals(column.astype(object)):
110+
out.isetitem(i, guarded)
111+
out.columns = pd.Index([_formula_guard(c) for c in out.columns])
112+
return out

‎src/freshdata/api.py‎

Lines changed: 10 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -10,6 +10,7 @@
1010
import pandas as pd
1111

1212
from ._reportframe import ReportFrame
13+
from ._util import sanitize_csv_formulas
1314
from .adapters.polars import from_pandas, to_pandas
1415
from .cleaner import Cleaner, run_pipeline
1516
from .config import CleanConfig, merge_options
@@ -423,6 +424,7 @@ def clean_csv(
423424
policy: object | None = None,
424425
strict: bool = False,
425426
profile: object | None = None,
427+
sanitize_formulas: bool = False,
426428
**options: object,
427429
) -> pd.DataFrame | tuple[pd.DataFrame, CleanReport]:
428430
"""Read a CSV file, clean it, and optionally write the result to disk.
@@ -433,6 +435,12 @@ def clean_csv(
433435
Path to the input CSV file.
434436
output_path:
435437
Optional path to write the cleaned CSV.
438+
sanitize_formulas:
439+
If ``True``, string cells (and column labels) in the **written**
440+
file that start with ``= + - @ <tab> <cr>`` are prefixed with ``'``
441+
so spreadsheets render them as text instead of executing them
442+
(OWASP CSV-injection guidance). Off by default to keep the output
443+
byte-exact; the returned DataFrame is never altered.
436444
return_report:
437445
If True, return ``(cleaned_df, CleanReport)``.
438446
read_csv_kwargs:
@@ -473,7 +481,8 @@ def clean_csv(
473481
)
474482
cleaned_df = cast(pd.DataFrame, result[0] if return_report else result)
475483
if output_path is not None:
476-
cleaned_df.to_csv(output_path, **{"index": False, **(to_csv_kwargs or {})})
484+
to_write = sanitize_csv_formulas(cleaned_df) if sanitize_formulas else cleaned_df
485+
to_write.to_csv(output_path, **{"index": False, **(to_csv_kwargs or {})})
477486
return result
478487

479488

‎src/freshdata/enterprise/entity_resolution.py‎

Lines changed: 12 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -36,6 +36,7 @@
3636

3737
import pandas as pd
3838

39+
from .._util import sanitize_csv_formulas
3940
from ..adapters.polars import from_pandas, to_pandas
4041
from .config import ( # noqa: F401 (configs re-exported for discoverability)
4142
BlockingRule,
@@ -1424,11 +1425,18 @@ def export_review_queue(
14241425
*,
14251426
format: str | None = None,
14261427
config: ReviewQueueConfig | None = None,
1428+
sanitize_formulas: bool = True,
14271429
) -> Path:
14281430
"""Write a review queue to *path* as ``csv``, ``jsonl``, or ``parquet``.
14291431
14301432
*report* may be a freshly-built :class:`ReviewQueueReport` or a raw
14311433
:class:`EntityResolutionReport` (in which case a queue is built first).
1434+
1435+
Review queues exist to be opened by humans in spreadsheets, so the
1436+
``csv`` format neutralizes formula-injection payloads by default: string
1437+
cells starting with ``= + - @ <tab> <cr>`` are prefixed with ``'``
1438+
(OWASP CSV-injection guidance). Pass ``sanitize_formulas=False`` for a
1439+
byte-exact export. Other formats are never altered.
14321440
"""
14331441
queue = (
14341442
report
@@ -1443,7 +1451,10 @@ def export_review_queue(
14431451
for it in queue.items:
14441452
fh.write(json.dumps(it.to_dict(), default=str) + "\n")
14451453
elif fmt == "csv":
1446-
queue.to_frame().to_csv(out, index=False)
1454+
frame = queue.to_frame()
1455+
if sanitize_formulas:
1456+
frame = sanitize_csv_formulas(frame)
1457+
frame.to_csv(out, index=False)
14471458
else: # parquet
14481459
queue.to_frame().to_parquet(out, index=False)
14491460
return out

‎src/freshdata/streaming/_cli.py‎

Lines changed: 25 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -17,6 +17,7 @@
1717

1818
import pandas as pd
1919

20+
from .._util import sanitize_csv_formulas
2021
from ._cleaner import StreamingCleaner
2122

2223

@@ -34,10 +35,11 @@ def _read_chunks(path: str, batch_size: int) -> Iterator[pd.DataFrame]:
3435
class _BatchWriter:
3536
"""Append cleaned batches to one CSV or Parquet file without buffering them all."""
3637

37-
def __init__(self, path: str | None) -> None:
38+
def __init__(self, path: str | None, sanitize_formulas: bool = False) -> None:
3839
self.path = path
3940
self.fmt = None if path is None else ("parquet"
4041
if path.lower().endswith((".parquet", ".pq")) else "csv")
42+
self.sanitize_formulas = sanitize_formulas
4143
self._pq_writer: Any = None
4244
self._csv_header = True
4345

@@ -53,6 +55,8 @@ def write(self, df: pd.DataFrame) -> None:
5355
self._pq_writer = pq.ParquetWriter(self.path, table.schema)
5456
self._pq_writer.write_table(table)
5557
else:
58+
if self.sanitize_formulas:
59+
df = sanitize_csv_formulas(df)
5660
df.to_csv(self.path, mode="w" if self._csv_header else "a",
5761
header=self._csv_header, index=False)
5862
self._csv_header = False
@@ -109,7 +113,8 @@ def _timeseries_config(args: argparse.Namespace) -> Any:
109113
return TimeSeriesCleanConfig(**kwargs)
110114

111115

112-
def _write_exceptions(cleaner: StreamingCleaner, path: str | None) -> None:
116+
def _write_exceptions(cleaner: StreamingCleaner, path: str | None,
117+
sanitize_formulas: bool = False) -> None:
113118
"""Persist any quarantined (late/anomalous) rows to *path* (CSV or Parquet)."""
114119
if not path:
115120
return
@@ -119,12 +124,15 @@ def _write_exceptions(cleaner: StreamingCleaner, path: str | None) -> None:
119124
if path.lower().endswith((".parquet", ".pq")):
120125
exc.to_parquet(path, index=False)
121126
else:
127+
if sanitize_formulas:
128+
exc = sanitize_csv_formulas(exc)
122129
exc.to_csv(path, index=False)
123130

124131

125132
def _run_stream(cleaner: StreamingCleaner, batches: Iterator[pd.DataFrame],
126133
writer: _BatchWriter, report_dir: str | None, quiet: bool,
127-
quarantine_path: str | None = None) -> int:
134+
quarantine_path: str | None = None,
135+
sanitize_formulas: bool = False) -> int:
128136
if report_dir:
129137
os.makedirs(report_dir, exist_ok=True)
130138
for cleaned, report in cleaner.clean_batches(batches):
@@ -137,7 +145,7 @@ def _run_stream(cleaner: StreamingCleaner, batches: Iterator[pd.DataFrame],
137145
print(json.dumps(report.streaming))
138146
writer.close()
139147
final = cleaner.finalize()
140-
_write_exceptions(cleaner, quarantine_path)
148+
_write_exceptions(cleaner, quarantine_path, sanitize_formulas=sanitize_formulas)
141149
if report_dir:
142150
with open(os.path.join(report_dir, "summary.json"), "w") as fh:
143151
json.dump(final.to_dict(), fh, default=str)
@@ -148,17 +156,21 @@ def _run_stream(cleaner: StreamingCleaner, batches: Iterator[pd.DataFrame],
148156

149157
def cmd_stream(args: argparse.Namespace) -> int:
150158
cleaner = StreamingCleaner(**_stream_options(args))
159+
sanitize = getattr(args, "sanitize_formulas", False)
151160
return _run_stream(cleaner, _read_chunks(args.input, args.batch_size),
152-
_BatchWriter(args.output), args.report, args.quiet,
153-
getattr(args, "quarantine", None))
161+
_BatchWriter(args.output, sanitize_formulas=sanitize),
162+
args.report, args.quiet,
163+
getattr(args, "quarantine", None),
164+
sanitize_formulas=sanitize)
154165

155166

156167
def cmd_stream_kafka(args: argparse.Namespace) -> int:
157168
cleaner = StreamingCleaner(**_stream_options(args))
158169
batches = cleaner.clean_kafka( # validates the kafka dependency
159170
topic=args.topic, bootstrap_servers=args.bootstrap_servers,
160171
batch_size=args.batch_size, max_batches=args.max_batches)
161-
writer = _BatchWriter(args.output)
172+
writer = _BatchWriter(args.output,
173+
sanitize_formulas=getattr(args, "sanitize_formulas", False))
162174
if args.report:
163175
os.makedirs(args.report, exist_ok=True)
164176
for cleaned, report in batches:
@@ -234,6 +246,9 @@ def add_stream_subparsers(subparsers: argparse._SubParsersAction) -> None:
234246
s = subparsers.add_parser("stream", help="clean a CSV/Parquet file in micro-batches")
235247
s.add_argument("input")
236248
s.add_argument("-o", "--output")
249+
s.add_argument("--sanitize-formulas", action="store_true",
250+
help="prefix ' to string cells starting with = + - @ tab/CR in "
251+
"CSV outputs so spreadsheets render them as text (OWASP)")
237252
s.add_argument("--batch-size", "--chunksize", type=int, default=100_000, dest="batch_size")
238253
s.add_argument("--report", metavar="DIR", help="directory for per-batch + summary JSON")
239254
s.add_argument("--target-column")
@@ -252,6 +267,9 @@ def add_stream_subparsers(subparsers: argparse._SubParsersAction) -> None:
252267
k.add_argument("--batch-size", type=int, default=10_000)
253268
k.add_argument("--max-batches", type=int)
254269
k.add_argument("-o", "--output")
270+
k.add_argument("--sanitize-formulas", action="store_true",
271+
help="prefix ' to string cells starting with = + - @ tab/CR in "
272+
"CSV outputs so spreadsheets render them as text (OWASP)")
255273
k.add_argument("--report", metavar="DIR")
256274
k.add_argument("--target-column")
257275
k.add_argument("--id-columns", nargs="*", default=())

0 commit comments

Comments
 (0)