Files
money/internal/parser/traderepublic.go
T
nikolaandClaude Opus 5 16c2585637 Drop counterparty, the generic parsers and .money/
counterparty was a structured field only nlb could fill honestly. revolut
and traderepublic invented one by running an IBAN-shaped regex over the
description they had just built, and the two spellings disagreed --
SI56 1234 5678 9012 345 against SI56123456789012345 -- so a literal rule
pattern that worked on one account silently matched nothing on another. It
is gone from the model, the index, the rule keys, ls --wide and the rules
screen. nlb now appends its IBAN column to the end of the description,
where the other two already keep theirs, so match = "*SI56*" works
everywhere. That changes those descriptions and with them their
fingerprints, so a statement overlapping an already-imported period will
re-add rather than dedupe those rows until the index is rebuilt. An index
built by an older binary drops the column when it is opened.

The index itself moves from .money/index.db up to index.db beside
rules.toml. Nothing looks in the old location, so an existing one has to be
moved by hand -- otherwise the tool quietly starts a fresh index and the
manual tags in the old file, the only thing statements cannot reproduce,
stay behind in it.

The csv and cmd parsers are gone along with the [csv] and [cmd] config they
carried. cmd shelled out to the Python extractors, which were ported to Go
and deleted, so it bridged to nothing; csv was a generic column-mapped
fallback that no account used, and between them they were the largest
configuration surface in the tool. A bank is now described in Go, where it
can be tested. The importer tests register their own three-column parser
rather than borrow a bank's, so they stay about the directory walk, dedupe
and per-file error reporting.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-11 22:51:20 +02:00

232 lines
6.3 KiB
Go

package parser
import (
"fmt"
"regexp"
"strings"
"time"
"git.petrovv.com/nikola/money/internal/config"
)
func init() {
Register("traderepublic", func(acc *config.Account) (Parser, error) {
return &tradeRepublicParser{digits: acc.Digits()}, nil
})
}
// tradeRepublicParser reads a Trade Republic account statement PDF.
//
// The layout varies between statements: a transaction may sit entirely on one
// line, or have its date, type and description stacked across several. Rather
// than guess, every token is assigned to whichever column heading its start
// position is closest to, which handles both shapes.
type tradeRepublicParser struct {
digits int
}
var (
trHeader = regexp.MustCompile(`^\s*DATE\b.*\bMONEY IN\b.*\bMONEY OUT\b.*\bBALANCE\b`)
// The sign may be an ASCII hyphen or a typographic minus, depending on the
// font the PDF was produced with.
trAmount = regexp.MustCompile(`[-\x{2212}]?€\s?[-\x{2212}]?[\d,]+\.\d{2}`)
trToken = regexp.MustCompile(`\S+`)
trFullDate = regexp.MustCompile(`^\d{2} [A-Z][a-z]{2} \d{4}$`)
)
var (
trTextColumns = []string{"DATE", "TYPE", "DESCRIPTION"}
trMoneyColumns = []string{"MONEY IN", "MONEY OUT", "BALANCE"}
)
// trAmountSlack lets an amount sit slightly left of the MONEY IN column
// without being mistaken for description text.
const trAmountSlack = 5
// trDateLayout is the date Trade Republic prints, e.g. "05 Jan 2026".
const trDateLayout = "02 Jan 2006"
func (p *tradeRepublicParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
text, err := pdfToText(path)
if err != nil {
return nil, err
}
return parseTradeRepublicText(text, p.digits)
}
// parseTradeRepublicText holds the whole parser, separated from PDF extraction
// so it can be tested against captured pdftotext output.
func parseTradeRepublicText(text string, digits int) ([]RawTxn, error) {
var txns []RawTxn
for _, page := range pages(text) {
lines := strings.Split(page, "\n")
header := -1
for i, line := range lines {
if trHeader.MatchString(line) {
header = i
break
}
}
if header < 0 {
continue // a cover page or disclaimer, with no transaction table
}
cols, err := trColumnStarts(lines[header])
if err != nil {
return nil, err
}
for _, block := range trBlocks(lines[header+1:]) {
txn, ok, err := parseTradeRepublicBlock(block, cols, digits)
if err != nil {
return nil, err
}
if ok {
txns = append(txns, txn)
}
}
}
return txns, nil
}
// trColumnStarts records where each heading begins on the header line; those
// positions are what every token is measured against.
func trColumnStarts(header string) (map[string]int, error) {
cols := map[string]int{}
for _, name := range append(append([]string{}, trTextColumns...), trMoneyColumns...) {
i := strings.Index(header, name)
if i < 0 {
return nil, fmt.Errorf("statement header has no %q column: %q", name, strings.TrimSpace(header))
}
cols[name] = i
}
return cols, nil
}
// trBlocks groups the consecutive non-blank lines that make up one transaction.
func trBlocks(lines []string) [][]string {
var (
out [][]string
block []string
)
for _, line := range lines {
if strings.TrimSpace(line) != "" {
block = append(block, line)
continue
}
if len(block) > 0 {
out = append(out, block)
block = nil
}
}
if len(block) > 0 {
out = append(out, block)
}
return out
}
// nearest returns the column whose start is closest to pos.
func nearest(pos int, cols map[string]int, names []string) string {
best, bestDist := "", -1
for _, name := range names {
d := pos - cols[name]
if d < 0 {
d = -d
}
if bestDist < 0 || d < bestDist {
best, bestDist = name, d
}
}
return best
}
func parseTradeRepublicBlock(lines []string, cols map[string]int, digits int) (RawTxn, bool, error) {
words := map[string][]string{}
amounts := map[string]*int64{}
var descChunks []string
for _, line := range lines {
// Amounts are found first: everything to their left is text, and the
// cut keeps them from being read as description tokens.
cut := len(line)
for _, loc := range trAmount.FindAllStringIndex(line, -1) {
if loc[0] < cols["MONEY IN"]-trAmountSlack {
continue // a figure inside the description, not a money column
}
if loc[0] < cut {
cut = loc[0]
}
v, err := parseTradeRepublicNumber(line[loc[0]:loc[1]], digits)
if err != nil {
return RawTxn{}, false, fmt.Errorf("amount %q: %w", line[loc[0]:loc[1]], err)
}
amounts[nearest(loc[0], cols, trMoneyColumns)] = &v
}
// A description fragment per line, so wrapped text can be rejoined.
var lineDesc []string
for _, loc := range trToken.FindAllStringIndex(line[:cut], -1) {
column := nearest(loc[0], cols, trTextColumns)
token := line[loc[0]:loc[1]]
if column == "DESCRIPTION" {
lineDesc = append(lineDesc, token)
continue
}
words[column] = append(words[column], token)
}
if len(lineDesc) > 0 {
descChunks = append(descChunks, strings.Join(lineDesc, " "))
}
}
// A block without a full date and a balance is a heading or a footer.
date := strings.Join(words["DATE"], " ")
if !trFullDate.MatchString(date) || amounts["BALANCE"] == nil {
return RawTxn{}, false, nil
}
d, err := time.Parse(trDateLayout, date)
if err != nil {
return RawTxn{}, false, fmt.Errorf("date %q: %w", date, err)
}
desc := trJoinWrapped(descChunks)
amount := deref(amounts["MONEY IN"]) - deref(amounts["MONEY OUT"])
return RawTxn{
Date: d.Format("2006-01-02"),
Description: desc,
AmountMinor: amount,
Type: strings.Join(words["TYPE"], " "),
BalanceMinor: amounts["BALANCE"],
}, true, nil
}
// trJoinWrapped glues description fragments split across lines. A fragment
// ending in a hyphen was broken mid-word, so it joins without a space.
func trJoinWrapped(chunks []string) string {
var out string
for _, chunk := range chunks {
switch {
case out == "":
out = chunk
case len(out) > 1 && strings.HasSuffix(out, "-") && !strings.HasSuffix(out, " -"):
out += chunk
default:
out += " " + chunk
}
}
return out
}
// parseTradeRepublicNumber reads the €1,234.56 format.
func parseTradeRepublicNumber(s string, digits int) (int64, error) {
return ParseAmount(strings.ReplaceAll(s, "€", ""), ".", ",", digits)
}
func deref(v *int64) int64 {
if v == nil {
return 0
}
return *v
}