data: add sanitized test fixtures and their generator

Real samples under data/test_input/ stay local and gitignored; these
committed fixtures are format-identical copies with every identifying
value replaced one-way via salted hashes: epsilon contract cards and
customer numbers into synthetic disjoint ranges (batches renumbered
9405+, dates +2y; cumulative slice batches 5001+, dates -6y), tsdrms
R/A / DBR / location ids into ZZ-form synthetic ids (dates +2y,
filenames shifted to match cutoffs), subfranchise statements to
extracted text with sender, partner, invoice numbers and amounts
replaced (EUR x rate = SEK arithmetic deliberately NOT preserved).
Amounts, GL codes, descriptions, and station/terminal/pump/receipt
numbers carry no personal data and are kept verbatim per readme
'Data formats'.

scripts/sanitize_samples.py regenerates fixtures deterministically
from local raw samples; scripts/check_fixture_leaks.py verifies no
real value appears in fixtures (content or filenames), exit 0 =
clean. Verified passing for all sources.
This commit is contained in:
hermes
2026-10-08 11:30:13 +02:00
parent 676e55eeb6
commit 3f0974a1d1
21 changed files with 8388 additions and 0 deletions
+121
View File
@@ -0,0 +1,121 @@
#!/usr/bin/env python3
"""One-way leak check: no real sensitive value may appear in data/fixtures.
Compares raw (gitignored) samples in data/test_input against generated
fixtures in data/fixtures. Exit 0 = clean, 1 = leaks found.
Design (scripts/sanitize_samples.py): amounts, GL codes, descriptions,
station/terminal/pump/receipt/control numbers are KEPT verbatim — they are
not PII. What must not leak: customer numbers, card numbers, tsdrms R/A /
DBR / location ids, and subfranchise sender/partner/invoice identity.
Checks are scoped per fixture kind (a naive blob scan false-positives
because kept amounts contain digit runs that coincide with 4-digit
customer numbers). Synthetic values are also shape-checked: epsilon
customers must all be 5xxxx, contract cards 98-prefixed.
"""
import glob
import os
import re
import sys
import openpyxl
root = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "data")
problems = []
def report(name, hits, total):
status = "OK" if not hits else f"LEAK {len(hits)}"
print(f"{name}: {status} (checked {total})", hits[:3] if hits else "")
problems.extend(hits)
# ---------------------------------------------------------------- real side #
cust, cards = set(), set()
for f in glob.glob(os.path.join(root, "test_input/export_from_epsilon/*.txt")):
for line in open(f, "rb").read().decode("ascii").split("\r\n")[1:]:
if not line.strip():
continue
fl = line.split("\t")
if fl[9].strip('"'):
cust.add(fl[9].strip('"'))
if fl[7].strip('"'):
cards.add(fl[7].strip('"'))
ra_ids = set()
for f in glob.glob(os.path.join(root, "test_input/export_from_tsdrms/*.xlsx")):
ws = openpyxl.load_workbook(f, read_only=True)["Sheet"]
idx = None
for r in ws.iter_rows(values_only=True):
if idx is None:
idx = {n: i for i, n in enumerate(r)}
continue
for k in ("R/A #", "DBR", "Location"):
if r[idx[k]] not in (None, ""):
ra_ids.add(str(r[idx[k]]))
subfr_strings = {"516411-7920", "Recamp", "Luleå", "Vinvägen",
"Stockholm-Arlanda", "Shared Mobility"}
subfr_invoices = {"821100844", "821100853", "821100865", "821100876",
"821100882", "821100891"}
# ---------------------------------------------------------- fixture side #
eps_fields = {"customer": set(), "card": set()}
for p in glob.glob(os.path.join(root, "fixtures/epsilon/*.txt")):
for line in open(p, "rb").read().decode("ascii").split("\r\n")[1:]:
if not line.strip():
continue
fl = line.split("\t")
if fl[9].strip('"'):
eps_fields["customer"].add(fl[9].strip('"'))
if fl[7].strip('"'):
eps_fields["card"].add(fl[7].strip('"'))
tsdrms_ids = set()
for p in glob.glob(os.path.join(root, "fixtures/tsdrms/*.xlsx")):
ws = openpyxl.load_workbook(p, read_only=True)["Sheet"]
idx = None
for r in ws.iter_rows(values_only=True):
if idx is None:
idx = {n: i for i, n in enumerate(r)}
continue
for k in ("R/A #", "DBR", "Location"):
if r[idx[k]] not in (None, ""):
tsdrms_ids.add(str(r[idx[k]]))
subfr_blob = ""
for p in glob.glob(os.path.join(root, "fixtures/subfranchise/*.txt")):
subfr_blob += open(p, encoding="utf-8").read() + "\n"
subfr_blob += "\n".join(glob.glob(os.path.join(root, "fixtures/*/*")))
# epsilon: value sets must be disjoint AND synthetic in shape
report("epsilon customer numbers", eps_fields["customer"] & cust, len(cust))
report("epsilon customer shape (5xxxx)",
[c for c in eps_fields["customer"] if not re.fullmatch(r"5\d{4}", c)],
len(eps_fields["customer"]))
report("epsilon card numbers", eps_fields["card"] & cards, len(cards))
report("epsilon contract-card shape (98..)",
[c for c in eps_fields["card"] if "*" not in c and not c.startswith("98")],
len(eps_fields["card"]))
report("epsilon masked-card shape (dddddd******dddd)",
[c for c in eps_fields["card"] if "*" in c
and not re.fullmatch(r"\d{6}\*{6}\d{4}", c)], len(eps_fields["card"]))
# tsdrms: identifier sets disjoint; fixture ids all ZZ-prefixed synthetic
report("tsdrms identifiers", tsdrms_ids & ra_ids, len(ra_ids))
report("tsdrms id shape (ZZ.. / 9xxxxxxC)",
[i for i in tsdrms_ids
if not (i.startswith("ZZ") or re.fullmatch(r"9\d{6}C?", i))],
len(tsdrms_ids))
# subfranchise: identity substrings and invoice numbers must be absent
report("subfranchise identity strings",
[s for s in subfr_strings if s in subfr_blob], len(subfr_strings))
report("subfranchise invoice numbers",
[s for s in subfr_invoices if s in subfr_blob], len(subfr_invoices))
# filenames must not carry real identifiers either
report("fixture filenames",
[p for p in glob.glob(os.path.join(root, "fixtures/*/*"))
if any(s in p for s in subfr_strings | subfr_invoices)
or any(c in p for c in cust)], "all")
sys.exit(1 if problems else 0)