Real samples under data/test_input/ stay local and gitignored; these committed fixtures are format-identical copies with every identifying value replaced one-way via salted hashes: epsilon contract cards and customer numbers into synthetic disjoint ranges (batches renumbered 9405+, dates +2y; cumulative slice batches 5001+, dates -6y), tsdrms R/A / DBR / location ids into ZZ-form synthetic ids (dates +2y, filenames shifted to match cutoffs), subfranchise statements to extracted text with sender, partner, invoice numbers and amounts replaced (EUR x rate = SEK arithmetic deliberately NOT preserved). Amounts, GL codes, descriptions, and station/terminal/pump/receipt numbers carry no personal data and are kept verbatim per readme 'Data formats'. scripts/sanitize_samples.py regenerates fixtures deterministically from local raw samples; scripts/check_fixture_leaks.py verifies no real value appears in fixtures (content or filenames), exit 0 = clean. Verified passing for all sources.
311 lines
12 KiB
Python
311 lines
12 KiB
Python
#!/usr/bin/env python3
|
|
"""Generate sanitized, committed test fixtures from the raw samples.
|
|
|
|
The raw exports under Application/data/test_input/ contain real customer,
|
|
card and financial data and are gitignored (readme.md §Data formats).
|
|
This script reads them locally and writes format-identical fixtures under
|
|
Application/data/fixtures/.
|
|
|
|
Outputs (all safe to commit):
|
|
fixtures/epsilon/raw-export-from-epsilon-<n>.txt
|
|
Batch files for 405-412, all rows, sanitized. Batches are renumbered
|
|
9405.. and dates shifted +2 years so fixtures can never collide with
|
|
real data. One extra file, raw-export-from-epsilon-cumulative.txt,
|
|
is a head+tail slice of the 2019-2025 cumulative export (few rows
|
|
from the earliest and latest batches, batches 5001+, dates shifted
|
|
-6 years). Customer numbers map into 5xxxx, contract cards into
|
|
98-prefixed 19-digit numbers — disjoint from any real value.
|
|
fixtures/tsdrms/<same filenames>.xlsx
|
|
Full, sanitized: GL account numbers, descriptions, CODEs, product
|
|
types and amounts kept verbatim; R/A #, DBR and location ids replaced
|
|
with synthetic ids (ZZ.. prefix); dates shifted +2 years.
|
|
fixtures/subfranchise/<source filename>.txt
|
|
Extracted text of each statement PDF with sender, partner, address,
|
|
org.nr., location, dates, invoice numbers and all money amounts
|
|
replaced by deterministic synthetic values. Layout (European
|
|
separators, () negatives, column alignment) is preserved; EUR x rate
|
|
= SEK arithmetic is NOT preserved — parsers must not cross-check it.
|
|
|
|
Determinism: every substitution derives from sha256(SALT, original value),
|
|
so re-running reproduces byte-identical files. Mapping is one-way; nothing
|
|
is written out. Requires openpyxl and pypdf. Never modifies data/test_input.
|
|
|
|
After running, verify no real values leaked before committing, e.g.:
|
|
grep -R -F -f <(cut real card/customer numbers) data/fixtures/
|
|
"""
|
|
|
|
import glob
|
|
import hashlib
|
|
import re
|
|
from pathlib import Path
|
|
|
|
SALT = "rustyrpn-fixtures-v1/"
|
|
DATE_SHIFT_YEARS = 2
|
|
TSDRMS_DATE_SHIFT_YEARS = 2
|
|
HERE = Path(__file__).resolve().parent
|
|
DATA = HERE.parent / "data"
|
|
RAW = DATA / "test_input"
|
|
OUT = DATA / "fixtures"
|
|
|
|
|
|
def h(*key) -> str:
|
|
return hashlib.sha256((SALT + "\x1f".join(str(k) for k in key)).encode()).hexdigest()
|
|
|
|
|
|
def hint(*key, lo, hi) -> int:
|
|
return lo + int(h(*key), 16) % (hi - lo + 1)
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# epsilon
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
EPS_DATE_RE = re.compile(r"^(\d{1,2})/(\d{1,2})/(\d{4}) (\d{1,2}):(\d{2}):(\d{2}) (AM|PM)$")
|
|
|
|
|
|
def shift_us_date(value: str, years: int) -> str:
|
|
m = EPS_DATE_RE.match(value)
|
|
if not m:
|
|
raise ValueError(f"unparsed epsilon date: {value!r}")
|
|
mo, d, y, hh, mi, s, ap = m.groups()
|
|
h24 = int(hh) % 12 + (12 if ap == "PM" else 0)
|
|
h12 = (h24 % 12) or 12
|
|
ampm = "PM" if h24 >= 12 else "AM"
|
|
return f"{int(mo)}/{int(d)}/{int(y) + years} {h12}:{mi}:{s} {ampm}"
|
|
|
|
|
|
def synth_card(card: str) -> str:
|
|
"""Masked consumer cards keep the masked shape; contract cards keep the
|
|
19-digit shape with a synthetic 98 prefix. Values never derive from the
|
|
original digits beyond the hash seed."""
|
|
s = h("card", card)
|
|
if "*" in card:
|
|
# keep the exact masked shape: 6 digits ****** 4 digits (hex would
|
|
# break decimal-only card parsers)
|
|
return str(int(s[:8], 16) % 10**6).zfill(6) + "******" + \
|
|
str(int(s[8:12], 16) % 10**4).zfill(4)
|
|
digits = "98" + "".join(str(int(c, 16) % 10) for c in s)
|
|
return digits[:len(card)].ljust(len(card), "0")
|
|
|
|
|
|
def synth_customer(cust: str) -> str:
|
|
# range 50000-59999: clearly synthetic and disjoint from real customer
|
|
# numbers (4 digits), so a leak grep can never false-positive
|
|
return str(int(h("cust", cust), 16) % 10000 + 50000)
|
|
|
|
|
|
def sanitize_epsilon_file(src: Path, dst: Path, year_shift: int, batch_base: int = 9000) -> int:
|
|
lines = src.read_bytes().decode("ascii").split("\r\n")
|
|
out = [lines[0]] # header verbatim
|
|
n = 0
|
|
for line in lines[1:]:
|
|
if not line.strip():
|
|
continue
|
|
f = line.split("\t")
|
|
assert len(f) == 16, f"{src.name}: row with {len(f)} fields"
|
|
# fields: 0 Date 1 Batch 2 Amount 3 Volume 4 Price 5 Quality
|
|
# 6 QualityName 7 CardNo 8 CardType 9 CustomerNo 10 Station
|
|
# 11 Terminal 12 Pump 13 Receipt 14 Group 15 Control
|
|
f[0] = '"' + shift_us_date(f[0].strip('"'), year_shift) + '"'
|
|
f[1] = '"%d"' % (batch_base + int(f[1].strip('"')))
|
|
card = synth_card(f[7].strip('"'))
|
|
f[7] = f'"{card}"'
|
|
f[8] = f'"{card}"' # card type equals card number in all samples
|
|
if f[9].strip('"'):
|
|
f[9] = f'"{synth_customer(f[9].strip(chr(34)))}"'
|
|
out.append("\t".join(f))
|
|
n += 1
|
|
dst.write_bytes(("\r\n".join(out) + "\r\n").encode("ascii"))
|
|
return n
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# tsdrms
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
def shift_eu_date(v, years: int):
|
|
m = re.fullmatch(r"(\d{2})/(\d{2})/(\d{4})", str(v))
|
|
return v if not m else f"{m.group(1)}/{m.group(2)}/{int(m.group(3)) + years}"
|
|
|
|
|
|
def synth_id(kind: str, v):
|
|
if v in (None, ""):
|
|
return v
|
|
s = h(kind, v)
|
|
# tail 5xxxx: disjoint from real customer numbers (4 digits) so the
|
|
# leak check never has to disambiguate
|
|
return "ZZ%s-%s" % (str(int(s[:4], 16) % 100).zfill(2), str(int(s[4:10], 16) % 10000 + 50000))
|
|
|
|
|
|
def synth_ra(v):
|
|
if v in (None, ""):
|
|
return v
|
|
s = h("ra", v)
|
|
if re.fullmatch(r"\d{6,7}[A-Za-z]?", str(v)): # numeric style, e.g. 345670C
|
|
# '9' prefix keeps length/shape but is disjoint from real ids
|
|
return "9" + str(int(s[:6], 16) % 10**6).zfill(6) + ("C" if str(v)[-1].isalpha() else "")
|
|
return synth_id("ra", v)
|
|
|
|
|
|
def sanitize_tsdrms(src: Path, dst: Path) -> int:
|
|
import openpyxl
|
|
|
|
wb = openpyxl.load_workbook(src, read_only=True)
|
|
assert wb.sheetnames == ["Sheet"], f"{src.name}: {wb.sheetnames}"
|
|
rows = list(wb["Sheet"].iter_rows(values_only=True))
|
|
header = list(rows[0])
|
|
ix = {name: i for i, name in enumerate(header)}
|
|
out_wb = openpyxl.Workbook()
|
|
ws = out_wb.active
|
|
ws.title = "Sheet"
|
|
ws.append(header)
|
|
for r in rows[1:]:
|
|
r = list(r)
|
|
r[ix["R/A #"]] = synth_ra(r[ix["R/A #"]])
|
|
r[ix["DBR"]] = synth_id("dbr", r[ix["DBR"]])
|
|
r[ix["Location"]] = synth_id("loc", r[ix["Location"]])
|
|
for col in ("Transaction Date", "Cutoff Date"):
|
|
r[ix[col]] = shift_eu_date(r[ix[col]], 2)
|
|
ws.append(r)
|
|
out_wb.save(dst)
|
|
return len(rows) - 1
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# subfranchise statements (PDF -> sanitized text fixture)
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
IDENTITY = [
|
|
("SHARED MOBILITY SVERIGE FILIAL", "EXEMPEL MOBILITY SVERIGE FILIAL"),
|
|
("Shared Mobility Sverige Filial", "Exempel Mobility Sverige Filial"),
|
|
("Shared Mobility", "Exempel Mobility"),
|
|
("Vinvägen 4", "Exempelvägen 1"),
|
|
("190 60 Stockholm-Arlanda", "111 22 EXEMPELSTAD"),
|
|
("Org.nr. 516411-7920", "Org.nr. 556000-0001"),
|
|
("Recamp Nordic AB", "Demo Uthyrning AB"),
|
|
("Luleå", "Demostad"),
|
|
("Sverige", "Exempelland"),
|
|
("LLAT", "ZZTT"),
|
|
]
|
|
|
|
# a money number: grouped-thousands w/ comma decimals, or a bare group,
|
|
# optionally parenthesised for negatives; never the 10,66-style rate line
|
|
# Keyed on the currency marker: EUR amounts are tight to '€'
|
|
# ("130.121,94€"), SEK amounts precede ' kr'. The '@daily rate'
|
|
# column ("10,66") is followed by neither marker and is kept verbatim
|
|
# (exchange rate is not business data). Zero cells ("€ -") match nothing.
|
|
MONEY_RE = re.compile(r"(\(?)(\d{1,3}(?:\.\d{3})*(?:,\d{2})?)(\)?)(\s*€|\s+kr)")
|
|
|
|
# invoice numbers are sequential in the real world; keep sequence visible in
|
|
# the fixture with a disjoint synthetic base (98xxxxxx) mapped by rank
|
|
_INVOICE_BASE = 98000001
|
|
# real invoice numbers are sequential from ~821100844; keep the step visible
|
|
# (x3, still monotonic over the small sample range) with a disjoint base
|
|
_INVOICE_MIN = 821100844
|
|
|
|
|
|
def synth_invoice(n: str) -> str:
|
|
return str(_INVOICE_BASE + (int(n) - _INVOICE_MIN) * 3)
|
|
|
|
|
|
def sanitize_subfranchise_text(t: str) -> str:
|
|
for a, b in IDENTITY:
|
|
t = t.replace(a, b)
|
|
# settlement period year lands on the address line after text extraction
|
|
# ("111 22 EXEMPELSTAD 2026"); shift it like any other date
|
|
t = re.sub(r"(EXEMPELSTAD )20(\d{2})\b",
|
|
lambda m: m.group(1) + "20%02d" % (int(m.group(2)) + 2), t)
|
|
|
|
t = re.sub(
|
|
r"(January|February|March|April|May|June|July|August|September|October|"
|
|
r"November|December) (\d{1,2}), (\d{4})",
|
|
lambda m: f"{m.group(1)} {int(m.group(2))}, {int(m.group(3)) + 2}",
|
|
t,
|
|
)
|
|
# month+year without day (settlement period, line-item names); run after
|
|
# the full-date form so "January 5, 2026" isn't hit twice
|
|
t = re.sub(
|
|
r"(January|February|March|April|May|June|July|August|September|October|"
|
|
r"November|December) (\d{4})",
|
|
lambda m: f"{m.group(1)} {int(m.group(2)) + 2}",
|
|
t,
|
|
)
|
|
t = re.sub(
|
|
r"Invoice #:\s*(\d+)",
|
|
lambda m: "Invoice #: " + synth_invoice(m.group(1)),
|
|
t,
|
|
)
|
|
t = re.sub(
|
|
r"Invoice (\d{8,})",
|
|
lambda m: "Invoice " + synth_invoice(m.group(1)),
|
|
t,
|
|
)
|
|
|
|
def repl(m):
|
|
whole = m.group(2)
|
|
if "." not in whole and "," not in whole and len(whole) <= 3:
|
|
return m.group(0) # a bare small int next to kr is not money
|
|
digits = int(whole.replace(".", "").split(",")[0])
|
|
frac = whole.split(",")[1] if "," in whole else "00"
|
|
scaled = digits * hint("scale", m.string, m.start(), lo=60, hi=150) // 100
|
|
return f"{m.group(1)}{group_eu(scaled)},{frac}{m.group(3)}{m.group(4)}"
|
|
|
|
return MONEY_RE.sub(repl, t)
|
|
|
|
|
|
def group_eu(n: int) -> str:
|
|
s, parts = str(abs(int(n))), []
|
|
while s:
|
|
parts.insert(0, s[-3:])
|
|
s = s[:-3]
|
|
return ".".join(parts)
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
def main():
|
|
for sub in ("epsilon", "tsdrms", "subfranchise"):
|
|
(OUT / sub).mkdir(parents=True, exist_ok=True)
|
|
|
|
for src in sorted((RAW / "export_from_epsilon").glob("raw-export-from-epsilon-4*.txt")):
|
|
n = sanitize_epsilon_file(src, OUT / "epsilon" / src.name, 2)
|
|
print(f"epsilon {src.name}: {n} rows")
|
|
|
|
huge = RAW / "export_from_epsilon" / "raw-export-from-epsilon-huge.txt"
|
|
if huge.exists():
|
|
rows = [ln for ln in huge.read_bytes().decode("ascii").split("\r\n")[1:] if ln.strip()]
|
|
slice_path = OUT / "epsilon" / ".slice.tmp"
|
|
slice_path.write_bytes(
|
|
("\r\n".join([huge.read_bytes().decode("ascii").split("\r\n")[0]] + rows[:4] + rows[-4:]) + "\r\n").encode("ascii"))
|
|
n = sanitize_epsilon_file(slice_path, OUT / "epsilon" / "raw-export-from-epsilon-cumulative.txt", -6, batch_base=5000)
|
|
slice_path.unlink()
|
|
print(f"epsilon cumulative slice: {n} rows (batches 5001+)")
|
|
|
|
for src in sorted((RAW / "export_from_tsdrms").glob("*.xlsx")):
|
|
# filename months must match the shifted cutoff dates inside
|
|
name = re.sub(r"\b(20\d{2})-",
|
|
lambda m: str(int(m.group(1)) + TSDRMS_DATE_SHIFT_YEARS) + "-",
|
|
src.name)
|
|
n = sanitize_tsdrms(src, OUT / "tsdrms" / name)
|
|
print(f"tsdrms {src.name} -> {name}: {n} rows")
|
|
|
|
from pypdf import PdfReader
|
|
for src in sorted((RAW / "subfranchise statement").glob("*.pdf")):
|
|
text = "\n".join(p.extract_text() for p in PdfReader(str(src)).pages)
|
|
# fixture filename carries the same metadata shapes, sanitized:
|
|
# "<received date +2y> -- <synthetic invoice> - Subfranchise
|
|
# Statement - <settlement month +2y>.txt"
|
|
m = re.match(r"(\d{4})-(\d{2})-(\d{2}) -- (\d+) - Subfranchise Statement - (\d{4})-(\d{2})", src.stem)
|
|
assert m, src.name
|
|
name = "%s-%s-%s -- %s - Subfranchise Statement - %s-%s" % (
|
|
int(m.group(1)) + 2, m.group(2), m.group(3),
|
|
synth_invoice(m.group(4)),
|
|
int(m.group(5)) + 2, m.group(6))
|
|
dst = OUT / "subfranchise" / (name + ".txt")
|
|
dst.write_text(sanitize_subfranchise_text(text), "utf-8")
|
|
print(f"subfranchise {src.name} -> {dst.name}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|