Trade Republic statements all download as document-N.pdf, which says
nothing about what they hold. traderepublic now implements parser.Namer:
it reads the period from the address block at the top of the first page
("DATE 01 May 2025 - 31 Jul 2026") and names the statement
traderepublic_2025_05_01_2026_07_31, lowercase like the NLB names, so the
folder sorts by start date and two downloads of one statement meet under
one name. The transaction table's own DATE heading never matches, since
no date range follows it.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
265 lines
7.4 KiB
Go
265 lines
7.4 KiB
Go
package parser
|
|
|
|
import (
|
|
"fmt"
|
|
"regexp"
|
|
"strings"
|
|
"time"
|
|
|
|
"git.petrovv.com/nikola/money/internal/config"
|
|
)
|
|
|
|
func init() {
|
|
Register("traderepublic", func(acc *config.Account) (Parser, error) {
|
|
return &tradeRepublicParser{digits: acc.Digits()}, nil
|
|
})
|
|
}
|
|
|
|
// tradeRepublicParser reads a Trade Republic account statement PDF.
|
|
//
|
|
// The layout varies between statements: a transaction may sit entirely on one
|
|
// line, or have its date, type and description stacked across several. Rather
|
|
// than guess, every token is assigned to whichever column heading its start
|
|
// position is closest to, which handles both shapes.
|
|
type tradeRepublicParser struct {
|
|
digits int
|
|
}
|
|
|
|
var (
|
|
trHeader = regexp.MustCompile(`^\s*DATE\b.*\bMONEY IN\b.*\bMONEY OUT\b.*\bBALANCE\b`)
|
|
// The sign may be an ASCII hyphen or a typographic minus, depending on the
|
|
// font the PDF was produced with.
|
|
trAmount = regexp.MustCompile(`[-\x{2212}]?€\s?[-\x{2212}]?[\d,]+\.\d{2}`)
|
|
trToken = regexp.MustCompile(`\S+`)
|
|
trFullDate = regexp.MustCompile(`^\d{2} [A-Z][a-z]{2} \d{4}$`)
|
|
)
|
|
|
|
var (
|
|
trTextColumns = []string{"DATE", "TYPE", "DESCRIPTION"}
|
|
trMoneyColumns = []string{"MONEY IN", "MONEY OUT", "BALANCE"}
|
|
)
|
|
|
|
// trAmountSlack lets an amount sit slightly left of the MONEY IN column
|
|
// without being mistaken for description text.
|
|
const trAmountSlack = 5
|
|
|
|
// trDateLayout is the date Trade Republic prints, e.g. "05 Jan 2026".
|
|
const trDateLayout = "02 Jan 2006"
|
|
|
|
// trPeriod is the statement period in the address block at the top of the
|
|
// first page, e.g. "DATE 01 May 2025 - 31 Jul 2026". It names the
|
|
// statement: Trade Republic's downloads are all called document-N.pdf. The
|
|
// transaction table's own DATE heading never matches, since no dates follow it.
|
|
var trPeriod = regexp.MustCompile(`\bDATE\s+(\d{2} [A-Z][a-z]{2} \d{4})\s*[-\x{2013}\x{2212}]\s*(\d{2} [A-Z][a-z]{2} \d{4})`)
|
|
|
|
func (p *tradeRepublicParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
|
|
text, err := pdfToText(path)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
return parseTradeRepublicText(text, p.digits)
|
|
}
|
|
|
|
// StatementName names a statement after its period,
|
|
// traderepublic_YYYY_MM_DD_YYYY_MM_DD, so the folder sorts by when it starts
|
|
// and two downloads of one statement collide.
|
|
func (p *tradeRepublicParser) StatementName(path string) (string, error) {
|
|
text, err := pdfToText(path)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
return tradeRepublicStatementName(text), nil
|
|
}
|
|
|
|
func tradeRepublicStatementName(text string) string {
|
|
m := trPeriod.FindStringSubmatch(text)
|
|
if m == nil {
|
|
return ""
|
|
}
|
|
from, err := time.Parse(trDateLayout, m[1])
|
|
if err != nil {
|
|
return ""
|
|
}
|
|
to, err := time.Parse(trDateLayout, m[2])
|
|
if err != nil {
|
|
return ""
|
|
}
|
|
return "traderepublic_" + from.Format("2006_01_02") + "_" + to.Format("2006_01_02")
|
|
}
|
|
|
|
// parseTradeRepublicText holds the whole parser, separated from PDF extraction
|
|
// so it can be tested against captured pdftotext output.
|
|
func parseTradeRepublicText(text string, digits int) ([]RawTxn, error) {
|
|
var txns []RawTxn
|
|
|
|
for _, page := range pages(text) {
|
|
lines := strings.Split(page, "\n")
|
|
header := -1
|
|
for i, line := range lines {
|
|
if trHeader.MatchString(line) {
|
|
header = i
|
|
break
|
|
}
|
|
}
|
|
if header < 0 {
|
|
continue // a cover page or disclaimer, with no transaction table
|
|
}
|
|
|
|
cols, err := trColumnStarts(lines[header])
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
for _, block := range trBlocks(lines[header+1:]) {
|
|
txn, ok, err := parseTradeRepublicBlock(block, cols, digits)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if ok {
|
|
txns = append(txns, txn)
|
|
}
|
|
}
|
|
}
|
|
return txns, nil
|
|
}
|
|
|
|
// trColumnStarts records where each heading begins on the header line; those
|
|
// positions are what every token is measured against.
|
|
func trColumnStarts(header string) (map[string]int, error) {
|
|
cols := map[string]int{}
|
|
for _, name := range append(append([]string{}, trTextColumns...), trMoneyColumns...) {
|
|
i := strings.Index(header, name)
|
|
if i < 0 {
|
|
return nil, fmt.Errorf("statement header has no %q column: %q", name, strings.TrimSpace(header))
|
|
}
|
|
cols[name] = i
|
|
}
|
|
return cols, nil
|
|
}
|
|
|
|
// trBlocks groups the consecutive non-blank lines that make up one transaction.
|
|
func trBlocks(lines []string) [][]string {
|
|
var (
|
|
out [][]string
|
|
block []string
|
|
)
|
|
for _, line := range lines {
|
|
if strings.TrimSpace(line) != "" {
|
|
block = append(block, line)
|
|
continue
|
|
}
|
|
if len(block) > 0 {
|
|
out = append(out, block)
|
|
block = nil
|
|
}
|
|
}
|
|
if len(block) > 0 {
|
|
out = append(out, block)
|
|
}
|
|
return out
|
|
}
|
|
|
|
// nearest returns the column whose start is closest to pos.
|
|
func nearest(pos int, cols map[string]int, names []string) string {
|
|
best, bestDist := "", -1
|
|
for _, name := range names {
|
|
d := pos - cols[name]
|
|
if d < 0 {
|
|
d = -d
|
|
}
|
|
if bestDist < 0 || d < bestDist {
|
|
best, bestDist = name, d
|
|
}
|
|
}
|
|
return best
|
|
}
|
|
|
|
func parseTradeRepublicBlock(lines []string, cols map[string]int, digits int) (RawTxn, bool, error) {
|
|
words := map[string][]string{}
|
|
amounts := map[string]*int64{}
|
|
var descChunks []string
|
|
|
|
for _, line := range lines {
|
|
// Amounts are found first: everything to their left is text, and the
|
|
// cut keeps them from being read as description tokens.
|
|
cut := len(line)
|
|
for _, loc := range trAmount.FindAllStringIndex(line, -1) {
|
|
if loc[0] < cols["MONEY IN"]-trAmountSlack {
|
|
continue // a figure inside the description, not a money column
|
|
}
|
|
if loc[0] < cut {
|
|
cut = loc[0]
|
|
}
|
|
v, err := parseTradeRepublicNumber(line[loc[0]:loc[1]], digits)
|
|
if err != nil {
|
|
return RawTxn{}, false, fmt.Errorf("amount %q: %w", line[loc[0]:loc[1]], err)
|
|
}
|
|
amounts[nearest(loc[0], cols, trMoneyColumns)] = &v
|
|
}
|
|
|
|
// A description fragment per line, so wrapped text can be rejoined.
|
|
var lineDesc []string
|
|
for _, loc := range trToken.FindAllStringIndex(line[:cut], -1) {
|
|
column := nearest(loc[0], cols, trTextColumns)
|
|
token := line[loc[0]:loc[1]]
|
|
if column == "DESCRIPTION" {
|
|
lineDesc = append(lineDesc, token)
|
|
continue
|
|
}
|
|
words[column] = append(words[column], token)
|
|
}
|
|
if len(lineDesc) > 0 {
|
|
descChunks = append(descChunks, strings.Join(lineDesc, " "))
|
|
}
|
|
}
|
|
|
|
// A block without a full date and a balance is a heading or a footer.
|
|
date := strings.Join(words["DATE"], " ")
|
|
if !trFullDate.MatchString(date) || amounts["BALANCE"] == nil {
|
|
return RawTxn{}, false, nil
|
|
}
|
|
d, err := time.Parse(trDateLayout, date)
|
|
if err != nil {
|
|
return RawTxn{}, false, fmt.Errorf("date %q: %w", date, err)
|
|
}
|
|
|
|
desc := trJoinWrapped(descChunks)
|
|
amount := deref(amounts["MONEY IN"]) - deref(amounts["MONEY OUT"])
|
|
|
|
return RawTxn{
|
|
Date: d.Format("2006-01-02"),
|
|
Description: desc,
|
|
AmountMinor: amount,
|
|
Type: strings.Join(words["TYPE"], " "),
|
|
BalanceMinor: amounts["BALANCE"],
|
|
}, true, nil
|
|
}
|
|
|
|
// trJoinWrapped glues description fragments split across lines. A fragment
|
|
// ending in a hyphen was broken mid-word, so it joins without a space.
|
|
func trJoinWrapped(chunks []string) string {
|
|
var out string
|
|
for _, chunk := range chunks {
|
|
switch {
|
|
case out == "":
|
|
out = chunk
|
|
case len(out) > 1 && strings.HasSuffix(out, "-") && !strings.HasSuffix(out, " -"):
|
|
out += chunk
|
|
default:
|
|
out += " " + chunk
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
// parseTradeRepublicNumber reads the €1,234.56 format.
|
|
func parseTradeRepublicNumber(s string, digits int) (int64, error) {
|
|
return ParseAmount(strings.ReplaceAll(s, "€", ""), ".", ",", digits)
|
|
}
|
|
|
|
func deref(v *int64) int64 {
|
|
if v == nil {
|
|
return 0
|
|
}
|
|
return *v
|
|
}
|