Files
python_scripts/nlb-export/trade_republic.py
T
2026-08-08 15:28:24 +02:00

156 lines
4.8 KiB
Python

import csv
import re
import subprocess
import sys
from datetime import datetime
from decimal import Decimal
from pathlib import Path
# extract transactions from trade republic account statement pdf
HEADER = re.compile(r"^\s*DATE\b.*\bMONEY IN\b.*\bMONEY OUT\b.*\bBALANCE\b")
TEXT_COLUMNS = ("DATE", "TYPE", "DESCRIPTION")
MONEY_COLUMNS = ("MONEY IN", "MONEY OUT", "BALANCE")
AMOUNT = re.compile(r"-?€\s?-?[\d,]+\.\d{2}")
TOKEN = re.compile(r"\S+")
FULL_DATE = re.compile(r"^\d{2} [A-Z][a-z]{2} \d{4}$")
IBAN = re.compile(r"\b[A-Z]{2}\d{2}[A-Z0-9]{11,30}\b")
AMOUNT_SLACK = 5
def pdf_to_text(pdf_path):
result = subprocess.run(
["pdftotext", "-layout", str(pdf_path), "-"],
capture_output=True, text=True, check=True,
)
return result.stdout
def column_starts(header_line):
return {c: header_line.index(c) for c in TEXT_COLUMNS + MONEY_COLUMNS}
def nearest(pos, cols):
return min(cols, key=lambda c: abs(pos - cols[c]))
def to_number(text):
return Decimal(text.replace("€", "").replace(",", "").replace(" ", ""))
def join_wrapped(chunks):
"""Glue description fragments split across lines, honouring hyphen wraps."""
out = ""
for chunk in chunks:
if not out:
out = chunk
elif len(out) > 1 and out.endswith("-") and not out.endswith(" -"):
out += chunk
else:
out += " " + chunk
return out
def parse_block(lines, cols):
"""A block is the consecutive lines making up one transaction.
Layout varies between statements: everything on one line, or the date,
type and description stacked across three lines. Both are handled by
assigning every token to the column its start position is closest to.
"""
text_cols = {c: cols[c] for c in TEXT_COLUMNS}
money_cols = {c: cols[c] for c in MONEY_COLUMNS}
words = {c: [] for c in TEXT_COLUMNS}
desc_chunks = []
amounts = {c: None for c in MONEY_COLUMNS}
for line in lines:
cut = len(line)
for m in AMOUNT.finditer(line):
if m.start() < cols["MONEY IN"] - AMOUNT_SLACK:
continue
cut = min(cut, m.start())
amounts[nearest(m.start(), money_cols)] = to_number(m.group())
words["DESCRIPTION"] = []
for m in TOKEN.finditer(line[:cut]):
words[nearest(m.start(), text_cols)].append(m.group())
if words["DESCRIPTION"]:
desc_chunks.append(" ".join(words["DESCRIPTION"]))
date = " ".join(words["DATE"])
if not FULL_DATE.match(date) or amounts["BALANCE"] is None:
return None
desc = join_wrapped(desc_chunks)
account = IBAN.search(desc)
amount = (amounts["MONEY IN"] or 0) - (amounts["MONEY OUT"] or 0)
return {
"date": datetime.strptime(date, "%d %b %Y").strftime("%d.%m.%y"),
"desc": desc,
"account": account.group() if account else "",
"amount": f"{amount:+.2f}",
"balance": f"{amounts['BALANCE']:.2f}",
"currency": "EUR",
"type": " ".join(words["TYPE"]),
}
def blocks(lines):
block = []
for line in lines:
if line.strip():
block.append(line)
elif block:
yield block
block = []
if block:
yield block
def extract_transactions(path):
print(path)
rows = []
for page_text in pdf_to_text(path).split("\f"):
page_lines = page_text.split("\n")
header = next((i for i, l in enumerate(page_lines) if HEADER.match(l)), None)
if header is None:
continue
cols = column_starts(page_lines[header])
for block in blocks(page_lines[header + 1:]):
row = parse_block(block, cols)
if row:
rows.append(row)
return rows
def check_balances(rows, path):
"""Every balance must be the previous one plus the transaction amount."""
for prev, row in zip(rows, rows[1:]):
expected = Decimal(prev["balance"]) + Decimal(row["amount"])
if expected != Decimal(row["balance"]):
print(f" warning: {path.name} {row['date']} {row['desc'][:40]!r} "
f"balance {row['balance']} != {expected}")
def main():
target = Path(sys.argv[1])
paths = sorted(target.glob("*.pdf")) if target.is_dir() else [target]
out_path = target.with_suffix(".csv")
all_rows = []
for path in paths:
rows = extract_transactions(path)
check_balances(rows, path)
all_rows.extend(rows)
with open(out_path, "w", newline="") as f:
writer = csv.DictWriter(
f, fieldnames=["date", "desc", "account", "amount", "balance", "currency", "type"]
)
writer.writeheader()
writer.writerows(all_rows)
print(f"Wrote {len(all_rows)} transactions to {out_path}")
if __name__ == "__main__":
main()