Add money: statement-driven personal finance tracker

A data directory holds one folder per account. Statements dropped into
those folders are parsed into a rebuildable SQLite index, categorised by
ordered glob rules in rules.toml, and browsed or hand-tagged in a Bubble
Tea TUI. Movements between the user's own accounts are marked as
transfers by the same rules and excluded from spending totals.

Manual tags and transfer marks are stored separately from the rule-derived
ones and always win, so editing rules.toml and re-running retag never
destroys hand edits.

Parsers are pluggable. Three are ported from the Python extractors they
replace -- nlb and traderepublic read PDFs via pdftotext -layout, revolut
reads the CSV export -- alongside a configurable-column CSV parser and a
cmd parser that shells out to an external script.

Both ports fix two latent bugs in the originals: the sign character class
rejected the typographic minus U+2212 that some PDF fonts emit, and NLB's
hardcoded continuation indent broke when pdftotext compressed runs of
spaces, so the threshold is now measured from the description column.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-09 00:37:20 +02:00
co-authored by Claude Opus 5
commit b0026c5a79
30 changed files with 5048 additions and 0 deletions
+233
View File
@@ -0,0 +1,233 @@
package parser
import (
"fmt"
"regexp"
"strings"
"time"
"git.petrovv.com/nikola/money/internal/config"
)
func init() {
Register("traderepublic", func(acc *config.Account) (Parser, error) {
return &tradeRepublicParser{digits: acc.Digits()}, nil
})
}
// tradeRepublicParser reads a Trade Republic account statement PDF.
//
// The layout varies between statements: a transaction may sit entirely on one
// line, or have its date, type and description stacked across several. Rather
// than guess, every token is assigned to whichever column heading its start
// position is closest to, which handles both shapes.
type tradeRepublicParser struct {
digits int
}
var (
trHeader = regexp.MustCompile(`^\s*DATE\b.*\bMONEY IN\b.*\bMONEY OUT\b.*\bBALANCE\b`)
// The sign may be an ASCII hyphen or a typographic minus, depending on the
// font the PDF was produced with.
trAmount = regexp.MustCompile(`[-\x{2212}]?€\s?[-\x{2212}]?[\d,]+\.\d{2}`)
trToken = regexp.MustCompile(`\S+`)
trFullDate = regexp.MustCompile(`^\d{2} [A-Z][a-z]{2} \d{4}$`)
trIBAN = regexp.MustCompile(`\b[A-Z]{2}\d{2}[A-Z0-9]{11,30}\b`)
)
var (
trTextColumns = []string{"DATE", "TYPE", "DESCRIPTION"}
trMoneyColumns = []string{"MONEY IN", "MONEY OUT", "BALANCE"}
)
// trAmountSlack lets an amount sit slightly left of the MONEY IN column
// without being mistaken for description text.
const trAmountSlack = 5
// trDateLayout is the date Trade Republic prints, e.g. "05 Jan 2026".
const trDateLayout = "02 Jan 2006"
func (p *tradeRepublicParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
text, err := pdfToText(path)
if err != nil {
return nil, err
}
return parseTradeRepublicText(text, p.digits)
}
// parseTradeRepublicText holds the whole parser, separated from PDF extraction
// so it can be tested against captured pdftotext output.
func parseTradeRepublicText(text string, digits int) ([]RawTxn, error) {
var txns []RawTxn
for _, page := range pages(text) {
lines := strings.Split(page, "\n")
header := -1
for i, line := range lines {
if trHeader.MatchString(line) {
header = i
break
}
}
if header < 0 {
continue // a cover page or disclaimer, with no transaction table
}
cols, err := trColumnStarts(lines[header])
if err != nil {
return nil, err
}
for _, block := range trBlocks(lines[header+1:]) {
txn, ok, err := parseTradeRepublicBlock(block, cols, digits)
if err != nil {
return nil, err
}
if ok {
txns = append(txns, txn)
}
}
}
return txns, nil
}
// trColumnStarts records where each heading begins on the header line; those
// positions are what every token is measured against.
func trColumnStarts(header string) (map[string]int, error) {
cols := map[string]int{}
for _, name := range append(append([]string{}, trTextColumns...), trMoneyColumns...) {
i := strings.Index(header, name)
if i < 0 {
return nil, fmt.Errorf("statement header has no %q column: %q", name, strings.TrimSpace(header))
}
cols[name] = i
}
return cols, nil
}
// trBlocks groups the consecutive non-blank lines that make up one transaction.
func trBlocks(lines []string) [][]string {
var (
out [][]string
block []string
)
for _, line := range lines {
if strings.TrimSpace(line) != "" {
block = append(block, line)
continue
}
if len(block) > 0 {
out = append(out, block)
block = nil
}
}
if len(block) > 0 {
out = append(out, block)
}
return out
}
// nearest returns the column whose start is closest to pos.
func nearest(pos int, cols map[string]int, names []string) string {
best, bestDist := "", -1
for _, name := range names {
d := pos - cols[name]
if d < 0 {
d = -d
}
if bestDist < 0 || d < bestDist {
best, bestDist = name, d
}
}
return best
}
func parseTradeRepublicBlock(lines []string, cols map[string]int, digits int) (RawTxn, bool, error) {
words := map[string][]string{}
amounts := map[string]*int64{}
var descChunks []string
for _, line := range lines {
// Amounts are found first: everything to their left is text, and the
// cut keeps them from being read as description tokens.
cut := len(line)
for _, loc := range trAmount.FindAllStringIndex(line, -1) {
if loc[0] < cols["MONEY IN"]-trAmountSlack {
continue // a figure inside the description, not a money column
}
if loc[0] < cut {
cut = loc[0]
}
v, err := parseTradeRepublicNumber(line[loc[0]:loc[1]], digits)
if err != nil {
return RawTxn{}, false, fmt.Errorf("amount %q: %w", line[loc[0]:loc[1]], err)
}
amounts[nearest(loc[0], cols, trMoneyColumns)] = &v
}
// A description fragment per line, so wrapped text can be rejoined.
var lineDesc []string
for _, loc := range trToken.FindAllStringIndex(line[:cut], -1) {
column := nearest(loc[0], cols, trTextColumns)
token := line[loc[0]:loc[1]]
if column == "DESCRIPTION" {
lineDesc = append(lineDesc, token)
continue
}
words[column] = append(words[column], token)
}
if len(lineDesc) > 0 {
descChunks = append(descChunks, strings.Join(lineDesc, " "))
}
}
// A block without a full date and a balance is a heading or a footer.
date := strings.Join(words["DATE"], " ")
if !trFullDate.MatchString(date) || amounts["BALANCE"] == nil {
return RawTxn{}, false, nil
}
d, err := time.Parse(trDateLayout, date)
if err != nil {
return RawTxn{}, false, fmt.Errorf("date %q: %w", date, err)
}
desc := trJoinWrapped(descChunks)
amount := deref(amounts["MONEY IN"]) - deref(amounts["MONEY OUT"])
return RawTxn{
Date: d.Format("2006-01-02"),
Description: desc,
AmountMinor: amount,
Counterparty: trIBAN.FindString(desc),
Type: strings.Join(words["TYPE"], " "),
BalanceMinor: amounts["BALANCE"],
}, true, nil
}
// trJoinWrapped glues description fragments split across lines. A fragment
// ending in a hyphen was broken mid-word, so it joins without a space.
func trJoinWrapped(chunks []string) string {
var out string
for _, chunk := range chunks {
switch {
case out == "":
out = chunk
case len(out) > 1 && strings.HasSuffix(out, "-") && !strings.HasSuffix(out, " -"):
out += chunk
default:
out += " " + chunk
}
}
return out
}
// parseTradeRepublicNumber reads the €1,234.56 format.
func parseTradeRepublicNumber(s string, digits int) (int64, error) {
return ParseAmount(strings.ReplaceAll(s, "€", ""), ".", ",", digits)
}
func deref(v *int64) int64 {
if v == nil {
return 0
}
return *v
}