data: add sanitized test fixtures and their generator

Real samples under data/test_input/ stay local and gitignored; these
committed fixtures are format-identical copies with every identifying
value replaced one-way via salted hashes: epsilon contract cards and
customer numbers into synthetic disjoint ranges (batches renumbered
9405+, dates +2y; cumulative slice batches 5001+, dates -6y), tsdrms
R/A / DBR / location ids into ZZ-form synthetic ids (dates +2y,
filenames shifted to match cutoffs), subfranchise statements to
extracted text with sender, partner, invoice numbers and amounts
replaced (EUR x rate = SEK arithmetic deliberately NOT preserved).
Amounts, GL codes, descriptions, and station/terminal/pump/receipt
numbers carry no personal data and are kept verbatim per readme
'Data formats'.

scripts/sanitize_samples.py regenerates fixtures deterministically
from local raw samples; scripts/check_fixture_leaks.py verifies no
real value appears in fixtures (content or filenames), exit 0 =
clean. Verified passing for all sources.
This commit is contained in:
hermes
2026-10-08 11:30:13 +02:00
parent 676e55eeb6
commit 3f0974a1d1
21 changed files with 8388 additions and 0 deletions
+121
View File
@@ -0,0 +1,121 @@
#!/usr/bin/env python3
"""One-way leak check: no real sensitive value may appear in data/fixtures.
Compares raw (gitignored) samples in data/test_input against generated
fixtures in data/fixtures. Exit 0 = clean, 1 = leaks found.
Design (scripts/sanitize_samples.py): amounts, GL codes, descriptions,
station/terminal/pump/receipt/control numbers are KEPT verbatim — they are
not PII. What must not leak: customer numbers, card numbers, tsdrms R/A /
DBR / location ids, and subfranchise sender/partner/invoice identity.
Checks are scoped per fixture kind (a naive blob scan false-positives
because kept amounts contain digit runs that coincide with 4-digit
customer numbers). Synthetic values are also shape-checked: epsilon
customers must all be 5xxxx, contract cards 98-prefixed.
"""
import glob
import os
import re
import sys
import openpyxl
root = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "data")
problems = []
def report(name, hits, total):
status = "OK" if not hits else f"LEAK {len(hits)}"
print(f"{name}: {status} (checked {total})", hits[:3] if hits else "")
problems.extend(hits)
# ---------------------------------------------------------------- real side #
cust, cards = set(), set()
for f in glob.glob(os.path.join(root, "test_input/export_from_epsilon/*.txt")):
for line in open(f, "rb").read().decode("ascii").split("\r\n")[1:]:
if not line.strip():
continue
fl = line.split("\t")
if fl[9].strip('"'):
cust.add(fl[9].strip('"'))
if fl[7].strip('"'):
cards.add(fl[7].strip('"'))
ra_ids = set()
for f in glob.glob(os.path.join(root, "test_input/export_from_tsdrms/*.xlsx")):
ws = openpyxl.load_workbook(f, read_only=True)["Sheet"]
idx = None
for r in ws.iter_rows(values_only=True):
if idx is None:
idx = {n: i for i, n in enumerate(r)}
continue
for k in ("R/A #", "DBR", "Location"):
if r[idx[k]] not in (None, ""):
ra_ids.add(str(r[idx[k]]))
subfr_strings = {"516411-7920", "Recamp", "Luleå", "Vinvägen",
"Stockholm-Arlanda", "Shared Mobility"}
subfr_invoices = {"821100844", "821100853", "821100865", "821100876",
"821100882", "821100891"}
# ---------------------------------------------------------- fixture side #
eps_fields = {"customer": set(), "card": set()}
for p in glob.glob(os.path.join(root, "fixtures/epsilon/*.txt")):
for line in open(p, "rb").read().decode("ascii").split("\r\n")[1:]:
if not line.strip():
continue
fl = line.split("\t")
if fl[9].strip('"'):
eps_fields["customer"].add(fl[9].strip('"'))
if fl[7].strip('"'):
eps_fields["card"].add(fl[7].strip('"'))
tsdrms_ids = set()
for p in glob.glob(os.path.join(root, "fixtures/tsdrms/*.xlsx")):
ws = openpyxl.load_workbook(p, read_only=True)["Sheet"]
idx = None
for r in ws.iter_rows(values_only=True):
if idx is None:
idx = {n: i for i, n in enumerate(r)}
continue
for k in ("R/A #", "DBR", "Location"):
if r[idx[k]] not in (None, ""):
tsdrms_ids.add(str(r[idx[k]]))
subfr_blob = ""
for p in glob.glob(os.path.join(root, "fixtures/subfranchise/*.txt")):
subfr_blob += open(p, encoding="utf-8").read() + "\n"
subfr_blob += "\n".join(glob.glob(os.path.join(root, "fixtures/*/*")))
# epsilon: value sets must be disjoint AND synthetic in shape
report("epsilon customer numbers", eps_fields["customer"] & cust, len(cust))
report("epsilon customer shape (5xxxx)",
[c for c in eps_fields["customer"] if not re.fullmatch(r"5\d{4}", c)],
len(eps_fields["customer"]))
report("epsilon card numbers", eps_fields["card"] & cards, len(cards))
report("epsilon contract-card shape (98..)",
[c for c in eps_fields["card"] if "*" not in c and not c.startswith("98")],
len(eps_fields["card"]))
report("epsilon masked-card shape (dddddd******dddd)",
[c for c in eps_fields["card"] if "*" in c
and not re.fullmatch(r"\d{6}\*{6}\d{4}", c)], len(eps_fields["card"]))
# tsdrms: identifier sets disjoint; fixture ids all ZZ-prefixed synthetic
report("tsdrms identifiers", tsdrms_ids & ra_ids, len(ra_ids))
report("tsdrms id shape (ZZ.. / 9xxxxxxC)",
[i for i in tsdrms_ids
if not (i.startswith("ZZ") or re.fullmatch(r"9\d{6}C?", i))],
len(tsdrms_ids))
# subfranchise: identity substrings and invoice numbers must be absent
report("subfranchise identity strings",
[s for s in subfr_strings if s in subfr_blob], len(subfr_strings))
report("subfranchise invoice numbers",
[s for s in subfr_invoices if s in subfr_blob], len(subfr_invoices))
# filenames must not carry real identifiers either
report("fixture filenames",
[p for p in glob.glob(os.path.join(root, "fixtures/*/*"))
if any(s in p for s in subfr_strings | subfr_invoices)
or any(c in p for c in cust)], "all")
sys.exit(1 if problems else 0)
+310
View File
@@ -0,0 +1,310 @@
#!/usr/bin/env python3
"""Generate sanitized, committed test fixtures from the raw samples.
The raw exports under Application/data/test_input/ contain real customer,
card and financial data and are gitignored (readme.md §Data formats).
This script reads them locally and writes format-identical fixtures under
Application/data/fixtures/.
Outputs (all safe to commit):
fixtures/epsilon/raw-export-from-epsilon-<n>.txt
Batch files for 405-412, all rows, sanitized. Batches are renumbered
9405.. and dates shifted +2 years so fixtures can never collide with
real data. One extra file, raw-export-from-epsilon-cumulative.txt,
is a head+tail slice of the 2019-2025 cumulative export (few rows
from the earliest and latest batches, batches 5001+, dates shifted
-6 years). Customer numbers map into 5xxxx, contract cards into
98-prefixed 19-digit numbers — disjoint from any real value.
fixtures/tsdrms/<same filenames>.xlsx
Full, sanitized: GL account numbers, descriptions, CODEs, product
types and amounts kept verbatim; R/A #, DBR and location ids replaced
with synthetic ids (ZZ.. prefix); dates shifted +2 years.
fixtures/subfranchise/<source filename>.txt
Extracted text of each statement PDF with sender, partner, address,
org.nr., location, dates, invoice numbers and all money amounts
replaced by deterministic synthetic values. Layout (European
separators, () negatives, column alignment) is preserved; EUR x rate
= SEK arithmetic is NOT preserved — parsers must not cross-check it.
Determinism: every substitution derives from sha256(SALT, original value),
so re-running reproduces byte-identical files. Mapping is one-way; nothing
is written out. Requires openpyxl and pypdf. Never modifies data/test_input.
After running, verify no real values leaked before committing, e.g.:
grep -R -F -f <(cut real card/customer numbers) data/fixtures/
"""
import glob
import hashlib
import re
from pathlib import Path
SALT = "rustyrpn-fixtures-v1/"
DATE_SHIFT_YEARS = 2
TSDRMS_DATE_SHIFT_YEARS = 2
HERE = Path(__file__).resolve().parent
DATA = HERE.parent / "data"
RAW = DATA / "test_input"
OUT = DATA / "fixtures"
def h(*key) -> str:
return hashlib.sha256((SALT + "\x1f".join(str(k) for k in key)).encode()).hexdigest()
def hint(*key, lo, hi) -> int:
return lo + int(h(*key), 16) % (hi - lo + 1)
# --------------------------------------------------------------------------- #
# epsilon
# --------------------------------------------------------------------------- #
EPS_DATE_RE = re.compile(r"^(\d{1,2})/(\d{1,2})/(\d{4}) (\d{1,2}):(\d{2}):(\d{2}) (AM|PM)$")
def shift_us_date(value: str, years: int) -> str:
m = EPS_DATE_RE.match(value)
if not m:
raise ValueError(f"unparsed epsilon date: {value!r}")
mo, d, y, hh, mi, s, ap = m.groups()
h24 = int(hh) % 12 + (12 if ap == "PM" else 0)
h12 = (h24 % 12) or 12
ampm = "PM" if h24 >= 12 else "AM"
return f"{int(mo)}/{int(d)}/{int(y) + years} {h12}:{mi}:{s} {ampm}"
def synth_card(card: str) -> str:
"""Masked consumer cards keep the masked shape; contract cards keep the
19-digit shape with a synthetic 98 prefix. Values never derive from the
original digits beyond the hash seed."""
s = h("card", card)
if "*" in card:
# keep the exact masked shape: 6 digits ****** 4 digits (hex would
# break decimal-only card parsers)
return str(int(s[:8], 16) % 10**6).zfill(6) + "******" + \
str(int(s[8:12], 16) % 10**4).zfill(4)
digits = "98" + "".join(str(int(c, 16) % 10) for c in s)
return digits[:len(card)].ljust(len(card), "0")
def synth_customer(cust: str) -> str:
# range 50000-59999: clearly synthetic and disjoint from real customer
# numbers (4 digits), so a leak grep can never false-positive
return str(int(h("cust", cust), 16) % 10000 + 50000)
def sanitize_epsilon_file(src: Path, dst: Path, year_shift: int, batch_base: int = 9000) -> int:
lines = src.read_bytes().decode("ascii").split("\r\n")
out = [lines[0]] # header verbatim
n = 0
for line in lines[1:]:
if not line.strip():
continue
f = line.split("\t")
assert len(f) == 16, f"{src.name}: row with {len(f)} fields"
# fields: 0 Date 1 Batch 2 Amount 3 Volume 4 Price 5 Quality
# 6 QualityName 7 CardNo 8 CardType 9 CustomerNo 10 Station
# 11 Terminal 12 Pump 13 Receipt 14 Group 15 Control
f[0] = '"' + shift_us_date(f[0].strip('"'), year_shift) + '"'
f[1] = '"%d"' % (batch_base + int(f[1].strip('"')))
card = synth_card(f[7].strip('"'))
f[7] = f'"{card}"'
f[8] = f'"{card}"' # card type equals card number in all samples
if f[9].strip('"'):
f[9] = f'"{synth_customer(f[9].strip(chr(34)))}"'
out.append("\t".join(f))
n += 1
dst.write_bytes(("\r\n".join(out) + "\r\n").encode("ascii"))
return n
# --------------------------------------------------------------------------- #
# tsdrms
# --------------------------------------------------------------------------- #
def shift_eu_date(v, years: int):
m = re.fullmatch(r"(\d{2})/(\d{2})/(\d{4})", str(v))
return v if not m else f"{m.group(1)}/{m.group(2)}/{int(m.group(3)) + years}"
def synth_id(kind: str, v):
if v in (None, ""):
return v
s = h(kind, v)
# tail 5xxxx: disjoint from real customer numbers (4 digits) so the
# leak check never has to disambiguate
return "ZZ%s-%s" % (str(int(s[:4], 16) % 100).zfill(2), str(int(s[4:10], 16) % 10000 + 50000))
def synth_ra(v):
if v in (None, ""):
return v
s = h("ra", v)
if re.fullmatch(r"\d{6,7}[A-Za-z]?", str(v)): # numeric style, e.g. 345670C
# '9' prefix keeps length/shape but is disjoint from real ids
return "9" + str(int(s[:6], 16) % 10**6).zfill(6) + ("C" if str(v)[-1].isalpha() else "")
return synth_id("ra", v)
def sanitize_tsdrms(src: Path, dst: Path) -> int:
import openpyxl
wb = openpyxl.load_workbook(src, read_only=True)
assert wb.sheetnames == ["Sheet"], f"{src.name}: {wb.sheetnames}"
rows = list(wb["Sheet"].iter_rows(values_only=True))
header = list(rows[0])
ix = {name: i for i, name in enumerate(header)}
out_wb = openpyxl.Workbook()
ws = out_wb.active
ws.title = "Sheet"
ws.append(header)
for r in rows[1:]:
r = list(r)
r[ix["R/A #"]] = synth_ra(r[ix["R/A #"]])
r[ix["DBR"]] = synth_id("dbr", r[ix["DBR"]])
r[ix["Location"]] = synth_id("loc", r[ix["Location"]])
for col in ("Transaction Date", "Cutoff Date"):
r[ix[col]] = shift_eu_date(r[ix[col]], 2)
ws.append(r)
out_wb.save(dst)
return len(rows) - 1
# --------------------------------------------------------------------------- #
# subfranchise statements (PDF -> sanitized text fixture)
# --------------------------------------------------------------------------- #
IDENTITY = [
("SHARED MOBILITY SVERIGE FILIAL", "EXEMPEL MOBILITY SVERIGE FILIAL"),
("Shared Mobility Sverige Filial", "Exempel Mobility Sverige Filial"),
("Shared Mobility", "Exempel Mobility"),
("Vinvägen 4", "Exempelvägen 1"),
("190 60 Stockholm-Arlanda", "111 22 EXEMPELSTAD"),
("Org.nr. 516411-7920", "Org.nr. 556000-0001"),
("Recamp Nordic AB", "Demo Uthyrning AB"),
("Luleå", "Demostad"),
("Sverige", "Exempelland"),
("LLAT", "ZZTT"),
]
# a money number: grouped-thousands w/ comma decimals, or a bare group,
# optionally parenthesised for negatives; never the 10,66-style rate line
# Keyed on the currency marker: EUR amounts are tight to '€'
# ("130.121,94€"), SEK amounts precede ' kr'. The '@daily rate'
# column ("10,66") is followed by neither marker and is kept verbatim
# (exchange rate is not business data). Zero cells ("€ -") match nothing.
MONEY_RE = re.compile(r"(\(?)(\d{1,3}(?:\.\d{3})*(?:,\d{2})?)(\)?)(\s*€|\s+kr)")
# invoice numbers are sequential in the real world; keep sequence visible in
# the fixture with a disjoint synthetic base (98xxxxxx) mapped by rank
_INVOICE_BASE = 98000001
# real invoice numbers are sequential from ~821100844; keep the step visible
# (x3, still monotonic over the small sample range) with a disjoint base
_INVOICE_MIN = 821100844
def synth_invoice(n: str) -> str:
return str(_INVOICE_BASE + (int(n) - _INVOICE_MIN) * 3)
def sanitize_subfranchise_text(t: str) -> str:
for a, b in IDENTITY:
t = t.replace(a, b)
# settlement period year lands on the address line after text extraction
# ("111 22 EXEMPELSTAD 2026"); shift it like any other date
t = re.sub(r"(EXEMPELSTAD )20(\d{2})\b",
lambda m: m.group(1) + "20%02d" % (int(m.group(2)) + 2), t)
t = re.sub(
r"(January|February|March|April|May|June|July|August|September|October|"
r"November|December) (\d{1,2}), (\d{4})",
lambda m: f"{m.group(1)} {int(m.group(2))}, {int(m.group(3)) + 2}",
t,
)
# month+year without day (settlement period, line-item names); run after
# the full-date form so "January 5, 2026" isn't hit twice
t = re.sub(
r"(January|February|March|April|May|June|July|August|September|October|"
r"November|December) (\d{4})",
lambda m: f"{m.group(1)} {int(m.group(2)) + 2}",
t,
)
t = re.sub(
r"Invoice #:\s*(\d+)",
lambda m: "Invoice #: " + synth_invoice(m.group(1)),
t,
)
t = re.sub(
r"Invoice (\d{8,})",
lambda m: "Invoice " + synth_invoice(m.group(1)),
t,
)
def repl(m):
whole = m.group(2)
if "." not in whole and "," not in whole and len(whole) <= 3:
return m.group(0) # a bare small int next to kr is not money
digits = int(whole.replace(".", "").split(",")[0])
frac = whole.split(",")[1] if "," in whole else "00"
scaled = digits * hint("scale", m.string, m.start(), lo=60, hi=150) // 100
return f"{m.group(1)}{group_eu(scaled)},{frac}{m.group(3)}{m.group(4)}"
return MONEY_RE.sub(repl, t)
def group_eu(n: int) -> str:
s, parts = str(abs(int(n))), []
while s:
parts.insert(0, s[-3:])
s = s[:-3]
return ".".join(parts)
# --------------------------------------------------------------------------- #
def main():
for sub in ("epsilon", "tsdrms", "subfranchise"):
(OUT / sub).mkdir(parents=True, exist_ok=True)
for src in sorted((RAW / "export_from_epsilon").glob("raw-export-from-epsilon-4*.txt")):
n = sanitize_epsilon_file(src, OUT / "epsilon" / src.name, 2)
print(f"epsilon {src.name}: {n} rows")
huge = RAW / "export_from_epsilon" / "raw-export-from-epsilon-huge.txt"
if huge.exists():
rows = [ln for ln in huge.read_bytes().decode("ascii").split("\r\n")[1:] if ln.strip()]
slice_path = OUT / "epsilon" / ".slice.tmp"
slice_path.write_bytes(
("\r\n".join([huge.read_bytes().decode("ascii").split("\r\n")[0]] + rows[:4] + rows[-4:]) + "\r\n").encode("ascii"))
n = sanitize_epsilon_file(slice_path, OUT / "epsilon" / "raw-export-from-epsilon-cumulative.txt", -6, batch_base=5000)
slice_path.unlink()
print(f"epsilon cumulative slice: {n} rows (batches 5001+)")
for src in sorted((RAW / "export_from_tsdrms").glob("*.xlsx")):
# filename months must match the shifted cutoff dates inside
name = re.sub(r"\b(20\d{2})-",
lambda m: str(int(m.group(1)) + TSDRMS_DATE_SHIFT_YEARS) + "-",
src.name)
n = sanitize_tsdrms(src, OUT / "tsdrms" / name)
print(f"tsdrms {src.name} -> {name}: {n} rows")
from pypdf import PdfReader
for src in sorted((RAW / "subfranchise statement").glob("*.pdf")):
text = "\n".join(p.extract_text() for p in PdfReader(str(src)).pages)
# fixture filename carries the same metadata shapes, sanitized:
# "<received date +2y> -- <synthetic invoice> - Subfranchise
# Statement - <settlement month +2y>.txt"
m = re.match(r"(\d{4})-(\d{2})-(\d{2}) -- (\d+) - Subfranchise Statement - (\d{4})-(\d{2})", src.stem)
assert m, src.name
name = "%s-%s-%s -- %s - Subfranchise Statement - %s-%s" % (
int(m.group(1)) + 2, m.group(2), m.group(3),
synth_invoice(m.group(4)),
int(m.group(5)) + 2, m.group(6))
dst = OUT / "subfranchise" / (name + ".txt")
dst.write_text(sanitize_subfranchise_text(text), "utf-8")
print(f"subfranchise {src.name} -> {dst.name}")
if __name__ == "__main__":
main()