data: add sanitized test fixtures and their generator
Real samples under data/test_input/ stay local and gitignored; these committed fixtures are format-identical copies with every identifying value replaced one-way via salted hashes: epsilon contract cards and customer numbers into synthetic disjoint ranges (batches renumbered 9405+, dates +2y; cumulative slice batches 5001+, dates -6y), tsdrms R/A / DBR / location ids into ZZ-form synthetic ids (dates +2y, filenames shifted to match cutoffs), subfranchise statements to extracted text with sender, partner, invoice numbers and amounts replaced (EUR x rate = SEK arithmetic deliberately NOT preserved). Amounts, GL codes, descriptions, and station/terminal/pump/receipt numbers carry no personal data and are kept verbatim per readme 'Data formats'. scripts/sanitize_samples.py regenerates fixtures deterministically from local raw samples; scripts/check_fixture_leaks.py verifies no real value appears in fixtures (content or filenames), exit 0 = clean. Verified passing for all sources.
This commit is contained in:
@@ -0,0 +1,310 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate sanitized, committed test fixtures from the raw samples.
|
||||
|
||||
The raw exports under Application/data/test_input/ contain real customer,
|
||||
card and financial data and are gitignored (readme.md §Data formats).
|
||||
This script reads them locally and writes format-identical fixtures under
|
||||
Application/data/fixtures/.
|
||||
|
||||
Outputs (all safe to commit):
|
||||
fixtures/epsilon/raw-export-from-epsilon-<n>.txt
|
||||
Batch files for 405-412, all rows, sanitized. Batches are renumbered
|
||||
9405.. and dates shifted +2 years so fixtures can never collide with
|
||||
real data. One extra file, raw-export-from-epsilon-cumulative.txt,
|
||||
is a head+tail slice of the 2019-2025 cumulative export (few rows
|
||||
from the earliest and latest batches, batches 5001+, dates shifted
|
||||
-6 years). Customer numbers map into 5xxxx, contract cards into
|
||||
98-prefixed 19-digit numbers — disjoint from any real value.
|
||||
fixtures/tsdrms/<same filenames>.xlsx
|
||||
Full, sanitized: GL account numbers, descriptions, CODEs, product
|
||||
types and amounts kept verbatim; R/A #, DBR and location ids replaced
|
||||
with synthetic ids (ZZ.. prefix); dates shifted +2 years.
|
||||
fixtures/subfranchise/<source filename>.txt
|
||||
Extracted text of each statement PDF with sender, partner, address,
|
||||
org.nr., location, dates, invoice numbers and all money amounts
|
||||
replaced by deterministic synthetic values. Layout (European
|
||||
separators, () negatives, column alignment) is preserved; EUR x rate
|
||||
= SEK arithmetic is NOT preserved — parsers must not cross-check it.
|
||||
|
||||
Determinism: every substitution derives from sha256(SALT, original value),
|
||||
so re-running reproduces byte-identical files. Mapping is one-way; nothing
|
||||
is written out. Requires openpyxl and pypdf. Never modifies data/test_input.
|
||||
|
||||
After running, verify no real values leaked before committing, e.g.:
|
||||
grep -R -F -f <(cut real card/customer numbers) data/fixtures/
|
||||
"""
|
||||
|
||||
import glob
|
||||
import hashlib
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
SALT = "rustyrpn-fixtures-v1/"
|
||||
DATE_SHIFT_YEARS = 2
|
||||
TSDRMS_DATE_SHIFT_YEARS = 2
|
||||
HERE = Path(__file__).resolve().parent
|
||||
DATA = HERE.parent / "data"
|
||||
RAW = DATA / "test_input"
|
||||
OUT = DATA / "fixtures"
|
||||
|
||||
|
||||
def h(*key) -> str:
|
||||
return hashlib.sha256((SALT + "\x1f".join(str(k) for k in key)).encode()).hexdigest()
|
||||
|
||||
|
||||
def hint(*key, lo, hi) -> int:
|
||||
return lo + int(h(*key), 16) % (hi - lo + 1)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# epsilon
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
EPS_DATE_RE = re.compile(r"^(\d{1,2})/(\d{1,2})/(\d{4}) (\d{1,2}):(\d{2}):(\d{2}) (AM|PM)$")
|
||||
|
||||
|
||||
def shift_us_date(value: str, years: int) -> str:
|
||||
m = EPS_DATE_RE.match(value)
|
||||
if not m:
|
||||
raise ValueError(f"unparsed epsilon date: {value!r}")
|
||||
mo, d, y, hh, mi, s, ap = m.groups()
|
||||
h24 = int(hh) % 12 + (12 if ap == "PM" else 0)
|
||||
h12 = (h24 % 12) or 12
|
||||
ampm = "PM" if h24 >= 12 else "AM"
|
||||
return f"{int(mo)}/{int(d)}/{int(y) + years} {h12}:{mi}:{s} {ampm}"
|
||||
|
||||
|
||||
def synth_card(card: str) -> str:
|
||||
"""Masked consumer cards keep the masked shape; contract cards keep the
|
||||
19-digit shape with a synthetic 98 prefix. Values never derive from the
|
||||
original digits beyond the hash seed."""
|
||||
s = h("card", card)
|
||||
if "*" in card:
|
||||
# keep the exact masked shape: 6 digits ****** 4 digits (hex would
|
||||
# break decimal-only card parsers)
|
||||
return str(int(s[:8], 16) % 10**6).zfill(6) + "******" + \
|
||||
str(int(s[8:12], 16) % 10**4).zfill(4)
|
||||
digits = "98" + "".join(str(int(c, 16) % 10) for c in s)
|
||||
return digits[:len(card)].ljust(len(card), "0")
|
||||
|
||||
|
||||
def synth_customer(cust: str) -> str:
|
||||
# range 50000-59999: clearly synthetic and disjoint from real customer
|
||||
# numbers (4 digits), so a leak grep can never false-positive
|
||||
return str(int(h("cust", cust), 16) % 10000 + 50000)
|
||||
|
||||
|
||||
def sanitize_epsilon_file(src: Path, dst: Path, year_shift: int, batch_base: int = 9000) -> int:
|
||||
lines = src.read_bytes().decode("ascii").split("\r\n")
|
||||
out = [lines[0]] # header verbatim
|
||||
n = 0
|
||||
for line in lines[1:]:
|
||||
if not line.strip():
|
||||
continue
|
||||
f = line.split("\t")
|
||||
assert len(f) == 16, f"{src.name}: row with {len(f)} fields"
|
||||
# fields: 0 Date 1 Batch 2 Amount 3 Volume 4 Price 5 Quality
|
||||
# 6 QualityName 7 CardNo 8 CardType 9 CustomerNo 10 Station
|
||||
# 11 Terminal 12 Pump 13 Receipt 14 Group 15 Control
|
||||
f[0] = '"' + shift_us_date(f[0].strip('"'), year_shift) + '"'
|
||||
f[1] = '"%d"' % (batch_base + int(f[1].strip('"')))
|
||||
card = synth_card(f[7].strip('"'))
|
||||
f[7] = f'"{card}"'
|
||||
f[8] = f'"{card}"' # card type equals card number in all samples
|
||||
if f[9].strip('"'):
|
||||
f[9] = f'"{synth_customer(f[9].strip(chr(34)))}"'
|
||||
out.append("\t".join(f))
|
||||
n += 1
|
||||
dst.write_bytes(("\r\n".join(out) + "\r\n").encode("ascii"))
|
||||
return n
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# tsdrms
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def shift_eu_date(v, years: int):
|
||||
m = re.fullmatch(r"(\d{2})/(\d{2})/(\d{4})", str(v))
|
||||
return v if not m else f"{m.group(1)}/{m.group(2)}/{int(m.group(3)) + years}"
|
||||
|
||||
|
||||
def synth_id(kind: str, v):
|
||||
if v in (None, ""):
|
||||
return v
|
||||
s = h(kind, v)
|
||||
# tail 5xxxx: disjoint from real customer numbers (4 digits) so the
|
||||
# leak check never has to disambiguate
|
||||
return "ZZ%s-%s" % (str(int(s[:4], 16) % 100).zfill(2), str(int(s[4:10], 16) % 10000 + 50000))
|
||||
|
||||
|
||||
def synth_ra(v):
|
||||
if v in (None, ""):
|
||||
return v
|
||||
s = h("ra", v)
|
||||
if re.fullmatch(r"\d{6,7}[A-Za-z]?", str(v)): # numeric style, e.g. 345670C
|
||||
# '9' prefix keeps length/shape but is disjoint from real ids
|
||||
return "9" + str(int(s[:6], 16) % 10**6).zfill(6) + ("C" if str(v)[-1].isalpha() else "")
|
||||
return synth_id("ra", v)
|
||||
|
||||
|
||||
def sanitize_tsdrms(src: Path, dst: Path) -> int:
|
||||
import openpyxl
|
||||
|
||||
wb = openpyxl.load_workbook(src, read_only=True)
|
||||
assert wb.sheetnames == ["Sheet"], f"{src.name}: {wb.sheetnames}"
|
||||
rows = list(wb["Sheet"].iter_rows(values_only=True))
|
||||
header = list(rows[0])
|
||||
ix = {name: i for i, name in enumerate(header)}
|
||||
out_wb = openpyxl.Workbook()
|
||||
ws = out_wb.active
|
||||
ws.title = "Sheet"
|
||||
ws.append(header)
|
||||
for r in rows[1:]:
|
||||
r = list(r)
|
||||
r[ix["R/A #"]] = synth_ra(r[ix["R/A #"]])
|
||||
r[ix["DBR"]] = synth_id("dbr", r[ix["DBR"]])
|
||||
r[ix["Location"]] = synth_id("loc", r[ix["Location"]])
|
||||
for col in ("Transaction Date", "Cutoff Date"):
|
||||
r[ix[col]] = shift_eu_date(r[ix[col]], 2)
|
||||
ws.append(r)
|
||||
out_wb.save(dst)
|
||||
return len(rows) - 1
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# subfranchise statements (PDF -> sanitized text fixture)
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
IDENTITY = [
|
||||
("SHARED MOBILITY SVERIGE FILIAL", "EXEMPEL MOBILITY SVERIGE FILIAL"),
|
||||
("Shared Mobility Sverige Filial", "Exempel Mobility Sverige Filial"),
|
||||
("Shared Mobility", "Exempel Mobility"),
|
||||
("Vinvägen 4", "Exempelvägen 1"),
|
||||
("190 60 Stockholm-Arlanda", "111 22 EXEMPELSTAD"),
|
||||
("Org.nr. 516411-7920", "Org.nr. 556000-0001"),
|
||||
("Recamp Nordic AB", "Demo Uthyrning AB"),
|
||||
("Luleå", "Demostad"),
|
||||
("Sverige", "Exempelland"),
|
||||
("LLAT", "ZZTT"),
|
||||
]
|
||||
|
||||
# a money number: grouped-thousands w/ comma decimals, or a bare group,
|
||||
# optionally parenthesised for negatives; never the 10,66-style rate line
|
||||
# Keyed on the currency marker: EUR amounts are tight to '€'
|
||||
# ("130.121,94€"), SEK amounts precede ' kr'. The '@daily rate'
|
||||
# column ("10,66") is followed by neither marker and is kept verbatim
|
||||
# (exchange rate is not business data). Zero cells ("€ -") match nothing.
|
||||
MONEY_RE = re.compile(r"(\(?)(\d{1,3}(?:\.\d{3})*(?:,\d{2})?)(\)?)(\s*€|\s+kr)")
|
||||
|
||||
# invoice numbers are sequential in the real world; keep sequence visible in
|
||||
# the fixture with a disjoint synthetic base (98xxxxxx) mapped by rank
|
||||
_INVOICE_BASE = 98000001
|
||||
# real invoice numbers are sequential from ~821100844; keep the step visible
|
||||
# (x3, still monotonic over the small sample range) with a disjoint base
|
||||
_INVOICE_MIN = 821100844
|
||||
|
||||
|
||||
def synth_invoice(n: str) -> str:
|
||||
return str(_INVOICE_BASE + (int(n) - _INVOICE_MIN) * 3)
|
||||
|
||||
|
||||
def sanitize_subfranchise_text(t: str) -> str:
|
||||
for a, b in IDENTITY:
|
||||
t = t.replace(a, b)
|
||||
# settlement period year lands on the address line after text extraction
|
||||
# ("111 22 EXEMPELSTAD 2026"); shift it like any other date
|
||||
t = re.sub(r"(EXEMPELSTAD )20(\d{2})\b",
|
||||
lambda m: m.group(1) + "20%02d" % (int(m.group(2)) + 2), t)
|
||||
|
||||
t = re.sub(
|
||||
r"(January|February|March|April|May|June|July|August|September|October|"
|
||||
r"November|December) (\d{1,2}), (\d{4})",
|
||||
lambda m: f"{m.group(1)} {int(m.group(2))}, {int(m.group(3)) + 2}",
|
||||
t,
|
||||
)
|
||||
# month+year without day (settlement period, line-item names); run after
|
||||
# the full-date form so "January 5, 2026" isn't hit twice
|
||||
t = re.sub(
|
||||
r"(January|February|March|April|May|June|July|August|September|October|"
|
||||
r"November|December) (\d{4})",
|
||||
lambda m: f"{m.group(1)} {int(m.group(2)) + 2}",
|
||||
t,
|
||||
)
|
||||
t = re.sub(
|
||||
r"Invoice #:\s*(\d+)",
|
||||
lambda m: "Invoice #: " + synth_invoice(m.group(1)),
|
||||
t,
|
||||
)
|
||||
t = re.sub(
|
||||
r"Invoice (\d{8,})",
|
||||
lambda m: "Invoice " + synth_invoice(m.group(1)),
|
||||
t,
|
||||
)
|
||||
|
||||
def repl(m):
|
||||
whole = m.group(2)
|
||||
if "." not in whole and "," not in whole and len(whole) <= 3:
|
||||
return m.group(0) # a bare small int next to kr is not money
|
||||
digits = int(whole.replace(".", "").split(",")[0])
|
||||
frac = whole.split(",")[1] if "," in whole else "00"
|
||||
scaled = digits * hint("scale", m.string, m.start(), lo=60, hi=150) // 100
|
||||
return f"{m.group(1)}{group_eu(scaled)},{frac}{m.group(3)}{m.group(4)}"
|
||||
|
||||
return MONEY_RE.sub(repl, t)
|
||||
|
||||
|
||||
def group_eu(n: int) -> str:
|
||||
s, parts = str(abs(int(n))), []
|
||||
while s:
|
||||
parts.insert(0, s[-3:])
|
||||
s = s[:-3]
|
||||
return ".".join(parts)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def main():
|
||||
for sub in ("epsilon", "tsdrms", "subfranchise"):
|
||||
(OUT / sub).mkdir(parents=True, exist_ok=True)
|
||||
|
||||
for src in sorted((RAW / "export_from_epsilon").glob("raw-export-from-epsilon-4*.txt")):
|
||||
n = sanitize_epsilon_file(src, OUT / "epsilon" / src.name, 2)
|
||||
print(f"epsilon {src.name}: {n} rows")
|
||||
|
||||
huge = RAW / "export_from_epsilon" / "raw-export-from-epsilon-huge.txt"
|
||||
if huge.exists():
|
||||
rows = [ln for ln in huge.read_bytes().decode("ascii").split("\r\n")[1:] if ln.strip()]
|
||||
slice_path = OUT / "epsilon" / ".slice.tmp"
|
||||
slice_path.write_bytes(
|
||||
("\r\n".join([huge.read_bytes().decode("ascii").split("\r\n")[0]] + rows[:4] + rows[-4:]) + "\r\n").encode("ascii"))
|
||||
n = sanitize_epsilon_file(slice_path, OUT / "epsilon" / "raw-export-from-epsilon-cumulative.txt", -6, batch_base=5000)
|
||||
slice_path.unlink()
|
||||
print(f"epsilon cumulative slice: {n} rows (batches 5001+)")
|
||||
|
||||
for src in sorted((RAW / "export_from_tsdrms").glob("*.xlsx")):
|
||||
# filename months must match the shifted cutoff dates inside
|
||||
name = re.sub(r"\b(20\d{2})-",
|
||||
lambda m: str(int(m.group(1)) + TSDRMS_DATE_SHIFT_YEARS) + "-",
|
||||
src.name)
|
||||
n = sanitize_tsdrms(src, OUT / "tsdrms" / name)
|
||||
print(f"tsdrms {src.name} -> {name}: {n} rows")
|
||||
|
||||
from pypdf import PdfReader
|
||||
for src in sorted((RAW / "subfranchise statement").glob("*.pdf")):
|
||||
text = "\n".join(p.extract_text() for p in PdfReader(str(src)).pages)
|
||||
# fixture filename carries the same metadata shapes, sanitized:
|
||||
# "<received date +2y> -- <synthetic invoice> - Subfranchise
|
||||
# Statement - <settlement month +2y>.txt"
|
||||
m = re.match(r"(\d{4})-(\d{2})-(\d{2}) -- (\d+) - Subfranchise Statement - (\d{4})-(\d{2})", src.stem)
|
||||
assert m, src.name
|
||||
name = "%s-%s-%s -- %s - Subfranchise Statement - %s-%s" % (
|
||||
int(m.group(1)) + 2, m.group(2), m.group(3),
|
||||
synth_invoice(m.group(4)),
|
||||
int(m.group(5)) + 2, m.group(6))
|
||||
dst = OUT / "subfranchise" / (name + ".txt")
|
||||
dst.write_text(sanitize_subfranchise_text(text), "utf-8")
|
||||
print(f"subfranchise {src.name} -> {dst.name}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user