#!/usr/bin/env python3 """Generate sanitized, committed test fixtures from the raw samples. The raw exports under Application/data/test_input/ contain real customer, card and financial data and are gitignored (readme.md §Data formats). This script reads them locally and writes format-identical fixtures under Application/data/fixtures/. Outputs (all safe to commit): fixtures/epsilon/raw-export-from-epsilon-.txt Batch files for 405-412, all rows, sanitized. Batches are renumbered 9405.. and dates shifted +2 years so fixtures can never collide with real data. One extra file, raw-export-from-epsilon-cumulative.txt, is a head+tail slice of the 2019-2025 cumulative export (few rows from the earliest and latest batches, batches 5001+, dates shifted -6 years). Customer numbers map into 5xxxx, contract cards into 98-prefixed 19-digit numbers — disjoint from any real value. fixtures/tsdrms/.xlsx Full, sanitized: GL account numbers, descriptions, CODEs, product types and amounts kept verbatim; R/A #, DBR and location ids replaced with synthetic ids (ZZ.. prefix); dates shifted +2 years. fixtures/subfranchise/.txt Extracted text of each statement PDF with sender, partner, address, org.nr., location, dates, invoice numbers and all money amounts replaced by deterministic synthetic values. Layout (European separators, () negatives, column alignment) is preserved; EUR x rate = SEK arithmetic is NOT preserved — parsers must not cross-check it. Determinism: every substitution derives from sha256(SALT, original value), so re-running reproduces byte-identical files. Mapping is one-way; nothing is written out. Requires openpyxl and pypdf. Never modifies data/test_input. After running, verify no real values leaked before committing, e.g.: grep -R -F -f <(cut real card/customer numbers) data/fixtures/ """ import glob import hashlib import re from pathlib import Path SALT = "rustyrpn-fixtures-v1/" DATE_SHIFT_YEARS = 2 TSDRMS_DATE_SHIFT_YEARS = 2 HERE = Path(__file__).resolve().parent DATA = HERE.parent / "data" RAW = DATA / "test_input" OUT = DATA / "fixtures" def h(*key) -> str: return hashlib.sha256((SALT + "\x1f".join(str(k) for k in key)).encode()).hexdigest() def hint(*key, lo, hi) -> int: return lo + int(h(*key), 16) % (hi - lo + 1) # --------------------------------------------------------------------------- # # epsilon # --------------------------------------------------------------------------- # EPS_DATE_RE = re.compile(r"^(\d{1,2})/(\d{1,2})/(\d{4}) (\d{1,2}):(\d{2}):(\d{2}) (AM|PM)$") def shift_us_date(value: str, years: int) -> str: m = EPS_DATE_RE.match(value) if not m: raise ValueError(f"unparsed epsilon date: {value!r}") mo, d, y, hh, mi, s, ap = m.groups() h24 = int(hh) % 12 + (12 if ap == "PM" else 0) h12 = (h24 % 12) or 12 ampm = "PM" if h24 >= 12 else "AM" return f"{int(mo)}/{int(d)}/{int(y) + years} {h12}:{mi}:{s} {ampm}" def synth_card(card: str) -> str: """Masked consumer cards keep the masked shape; contract cards keep the 19-digit shape with a synthetic 98 prefix. Values never derive from the original digits beyond the hash seed.""" s = h("card", card) if "*" in card: # keep the exact masked shape: 6 digits ****** 4 digits (hex would # break decimal-only card parsers) return str(int(s[:8], 16) % 10**6).zfill(6) + "******" + \ str(int(s[8:12], 16) % 10**4).zfill(4) digits = "98" + "".join(str(int(c, 16) % 10) for c in s) return digits[:len(card)].ljust(len(card), "0") def synth_customer(cust: str) -> str: # range 50000-59999: clearly synthetic and disjoint from real customer # numbers (4 digits), so a leak grep can never false-positive return str(int(h("cust", cust), 16) % 10000 + 50000) def sanitize_epsilon_file(src: Path, dst: Path, year_shift: int, batch_base: int = 9000) -> int: lines = src.read_bytes().decode("ascii").split("\r\n") out = [lines[0]] # header verbatim n = 0 for line in lines[1:]: if not line.strip(): continue f = line.split("\t") assert len(f) == 16, f"{src.name}: row with {len(f)} fields" # fields: 0 Date 1 Batch 2 Amount 3 Volume 4 Price 5 Quality # 6 QualityName 7 CardNo 8 CardType 9 CustomerNo 10 Station # 11 Terminal 12 Pump 13 Receipt 14 Group 15 Control f[0] = '"' + shift_us_date(f[0].strip('"'), year_shift) + '"' f[1] = '"%d"' % (batch_base + int(f[1].strip('"'))) card = synth_card(f[7].strip('"')) f[7] = f'"{card}"' f[8] = f'"{card}"' # card type equals card number in all samples if f[9].strip('"'): f[9] = f'"{synth_customer(f[9].strip(chr(34)))}"' out.append("\t".join(f)) n += 1 dst.write_bytes(("\r\n".join(out) + "\r\n").encode("ascii")) return n # --------------------------------------------------------------------------- # # tsdrms # --------------------------------------------------------------------------- # def shift_eu_date(v, years: int): m = re.fullmatch(r"(\d{2})/(\d{2})/(\d{4})", str(v)) return v if not m else f"{m.group(1)}/{m.group(2)}/{int(m.group(3)) + years}" def synth_id(kind: str, v): if v in (None, ""): return v s = h(kind, v) # tail 5xxxx: disjoint from real customer numbers (4 digits) so the # leak check never has to disambiguate return "ZZ%s-%s" % (str(int(s[:4], 16) % 100).zfill(2), str(int(s[4:10], 16) % 10000 + 50000)) def synth_ra(v): if v in (None, ""): return v s = h("ra", v) if re.fullmatch(r"\d{6,7}[A-Za-z]?", str(v)): # numeric style, e.g. 345670C # '9' prefix keeps length/shape but is disjoint from real ids return "9" + str(int(s[:6], 16) % 10**6).zfill(6) + ("C" if str(v)[-1].isalpha() else "") return synth_id("ra", v) def sanitize_tsdrms(src: Path, dst: Path) -> int: import openpyxl wb = openpyxl.load_workbook(src, read_only=True) assert wb.sheetnames == ["Sheet"], f"{src.name}: {wb.sheetnames}" rows = list(wb["Sheet"].iter_rows(values_only=True)) header = list(rows[0]) ix = {name: i for i, name in enumerate(header)} out_wb = openpyxl.Workbook() ws = out_wb.active ws.title = "Sheet" ws.append(header) for r in rows[1:]: r = list(r) r[ix["R/A #"]] = synth_ra(r[ix["R/A #"]]) r[ix["DBR"]] = synth_id("dbr", r[ix["DBR"]]) r[ix["Location"]] = synth_id("loc", r[ix["Location"]]) for col in ("Transaction Date", "Cutoff Date"): r[ix[col]] = shift_eu_date(r[ix[col]], 2) ws.append(r) out_wb.save(dst) return len(rows) - 1 # --------------------------------------------------------------------------- # # subfranchise statements (PDF -> sanitized text fixture) # --------------------------------------------------------------------------- # IDENTITY = [ ("SHARED MOBILITY SVERIGE FILIAL", "EXEMPEL MOBILITY SVERIGE FILIAL"), ("Shared Mobility Sverige Filial", "Exempel Mobility Sverige Filial"), ("Shared Mobility", "Exempel Mobility"), ("Vinvägen 4", "Exempelvägen 1"), ("190 60 Stockholm-Arlanda", "111 22 EXEMPELSTAD"), ("Org.nr. 516411-7920", "Org.nr. 556000-0001"), ("Recamp Nordic AB", "Demo Uthyrning AB"), ("Luleå", "Demostad"), ("Sverige", "Exempelland"), ("LLAT", "ZZTT"), ] # a money number: grouped-thousands w/ comma decimals, or a bare group, # optionally parenthesised for negatives; never the 10,66-style rate line # Keyed on the currency marker: EUR amounts are tight to '€' # ("130.121,94€"), SEK amounts precede ' kr'. The '@daily rate' # column ("10,66") is followed by neither marker and is kept verbatim # (exchange rate is not business data). Zero cells ("€ -") match nothing. MONEY_RE = re.compile(r"(\(?)(\d{1,3}(?:\.\d{3})*(?:,\d{2})?)(\)?)(\s*€|\s+kr)") # invoice numbers are sequential in the real world; keep sequence visible in # the fixture with a disjoint synthetic base (98xxxxxx) mapped by rank _INVOICE_BASE = 98000001 # real invoice numbers are sequential from ~821100844; keep the step visible # (x3, still monotonic over the small sample range) with a disjoint base _INVOICE_MIN = 821100844 def synth_invoice(n: str) -> str: return str(_INVOICE_BASE + (int(n) - _INVOICE_MIN) * 3) def sanitize_subfranchise_text(t: str) -> str: for a, b in IDENTITY: t = t.replace(a, b) # settlement period year lands on the address line after text extraction # ("111 22 EXEMPELSTAD 2026"); shift it like any other date t = re.sub(r"(EXEMPELSTAD )20(\d{2})\b", lambda m: m.group(1) + "20%02d" % (int(m.group(2)) + 2), t) t = re.sub( r"(January|February|March|April|May|June|July|August|September|October|" r"November|December) (\d{1,2}), (\d{4})", lambda m: f"{m.group(1)} {int(m.group(2))}, {int(m.group(3)) + 2}", t, ) # month+year without day (settlement period, line-item names); run after # the full-date form so "January 5, 2026" isn't hit twice t = re.sub( r"(January|February|March|April|May|June|July|August|September|October|" r"November|December) (\d{4})", lambda m: f"{m.group(1)} {int(m.group(2)) + 2}", t, ) t = re.sub( r"Invoice #:\s*(\d+)", lambda m: "Invoice #: " + synth_invoice(m.group(1)), t, ) t = re.sub( r"Invoice (\d{8,})", lambda m: "Invoice " + synth_invoice(m.group(1)), t, ) def repl(m): whole = m.group(2) if "." not in whole and "," not in whole and len(whole) <= 3: return m.group(0) # a bare small int next to kr is not money digits = int(whole.replace(".", "").split(",")[0]) frac = whole.split(",")[1] if "," in whole else "00" scaled = digits * hint("scale", m.string, m.start(), lo=60, hi=150) // 100 return f"{m.group(1)}{group_eu(scaled)},{frac}{m.group(3)}{m.group(4)}" return MONEY_RE.sub(repl, t) def group_eu(n: int) -> str: s, parts = str(abs(int(n))), [] while s: parts.insert(0, s[-3:]) s = s[:-3] return ".".join(parts) # --------------------------------------------------------------------------- # def main(): for sub in ("epsilon", "tsdrms", "subfranchise"): (OUT / sub).mkdir(parents=True, exist_ok=True) for src in sorted((RAW / "export_from_epsilon").glob("raw-export-from-epsilon-4*.txt")): n = sanitize_epsilon_file(src, OUT / "epsilon" / src.name, 2) print(f"epsilon {src.name}: {n} rows") huge = RAW / "export_from_epsilon" / "raw-export-from-epsilon-huge.txt" if huge.exists(): rows = [ln for ln in huge.read_bytes().decode("ascii").split("\r\n")[1:] if ln.strip()] slice_path = OUT / "epsilon" / ".slice.tmp" slice_path.write_bytes( ("\r\n".join([huge.read_bytes().decode("ascii").split("\r\n")[0]] + rows[:4] + rows[-4:]) + "\r\n").encode("ascii")) n = sanitize_epsilon_file(slice_path, OUT / "epsilon" / "raw-export-from-epsilon-cumulative.txt", -6, batch_base=5000) slice_path.unlink() print(f"epsilon cumulative slice: {n} rows (batches 5001+)") for src in sorted((RAW / "export_from_tsdrms").glob("*.xlsx")): # filename months must match the shifted cutoff dates inside name = re.sub(r"\b(20\d{2})-", lambda m: str(int(m.group(1)) + TSDRMS_DATE_SHIFT_YEARS) + "-", src.name) n = sanitize_tsdrms(src, OUT / "tsdrms" / name) print(f"tsdrms {src.name} -> {name}: {n} rows") from pypdf import PdfReader for src in sorted((RAW / "subfranchise statement").glob("*.pdf")): text = "\n".join(p.extract_text() for p in PdfReader(str(src)).pages) # fixture filename carries the same metadata shapes, sanitized: # " -- - Subfranchise # Statement - .txt" m = re.match(r"(\d{4})-(\d{2})-(\d{2}) -- (\d+) - Subfranchise Statement - (\d{4})-(\d{2})", src.stem) assert m, src.name name = "%s-%s-%s -- %s - Subfranchise Statement - %s-%s" % ( int(m.group(1)) + 2, m.group(2), m.group(3), synth_invoice(m.group(4)), int(m.group(5)) + 2, m.group(6)) dst = OUT / "subfranchise" / (name + ".txt") dst.write_text(sanitize_subfranchise_text(text), "utf-8") print(f"subfranchise {src.name} -> {dst.name}") if __name__ == "__main__": main()