Files
money/internal/parser/revolut.go
T
nikolaandClaude Opus 5 b0026c5a79 Add money: statement-driven personal finance tracker
A data directory holds one folder per account. Statements dropped into
those folders are parsed into a rebuildable SQLite index, categorised by
ordered glob rules in rules.toml, and browsed or hand-tagged in a Bubble
Tea TUI. Movements between the user's own accounts are marked as
transfers by the same rules and excluded from spending totals.

Manual tags and transfer marks are stored separately from the rule-derived
ones and always win, so editing rules.toml and re-running retag never
destroys hand edits.

Parsers are pluggable. Three are ported from the Python extractors they
replace -- nlb and traderepublic read PDFs via pdftotext -layout, revolut
reads the CSV export -- alongside a configurable-column CSV parser and a
cmd parser that shells out to an external script.

Both ports fix two latent bugs in the originals: the sign character class
rejected the typographic minus U+2212 that some PDF fonts emit, and NLB's
hardcoded continuation indent broke when pdftotext compressed runs of
spaces, so the threshold is now measured from the description column.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-09 00:37:20 +02:00

223 lines
5.8 KiB
Go

package parser
import (
"encoding/csv"
"fmt"
"io"
"os"
"regexp"
"sort"
"strings"
"time"
"git.petrovv.com/nikola/money/internal/config"
"git.petrovv.com/nikola/money/internal/model"
)
func init() {
Register("revolut", func(acc *config.Account) (Parser, error) {
return &revolutParser{digits: acc.Digits(), currency: strings.ToUpper(acc.Currency)}, nil
})
}
// revolutParser reads a Revolut account-statement CSV export.
//
// Revolut books the fee alongside the transaction rather than as its own line,
// and reports pending transactions that have no balance yet. Both are handled
// the way the original extraction script did: the fee is folded into the
// amount, and anything not COMPLETED is skipped.
//
// A single export can mix currencies. Since an account here has one currency,
// rows in others are skipped and reported; importing them is a matter of
// giving that currency its own account folder.
type revolutParser struct {
digits int
currency string
warnings []string
}
// revolutIBAN finds a counterparty account inside a description.
var revolutIBAN = regexp.MustCompile(`\b[A-Z]{2}\d{2}[A-Z0-9]{11,30}\b`)
// Columns the parser needs; a missing one is a hard error rather than a
// silently empty field.
var revolutColumns = []string{
"Type", "Completed Date", "Description", "Amount", "Fee", "Currency", "State", "Balance",
}
func (p *revolutParser) Warnings() []string { return p.warnings }
func (p *revolutParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
p.warnings = nil
f, err := os.Open(path)
if err != nil {
return nil, err
}
defer f.Close()
r := csv.NewReader(f)
r.FieldsPerRecord = -1
r.LazyQuotes = true
header, err := r.Read()
if err == io.EOF {
return nil, fmt.Errorf("%s is empty", path)
}
if err != nil {
return nil, fmt.Errorf("%s: %w", path, err)
}
index, err := revolutHeaderIndex(header)
if err != nil {
return nil, fmt.Errorf("%s: %w", path, err)
}
type record struct {
fields []string
line int
}
var records []record
for line := 2; ; line++ {
rec, err := r.Read()
if err == io.EOF {
break
}
if err != nil {
return nil, fmt.Errorf("%s: row %d: %w", path, line, err)
}
if isBlank(rec) {
continue
}
records = append(records, record{fields: rec, line: line})
}
// The export is not in date order, and the running balance only makes
// sense along one. A stable sort also keeps identical rows in file order,
// which the importer's deduplication relies on.
sort.SliceStable(records, func(i, j int) bool {
return index.get(records[i].fields, "Completed Date") < index.get(records[j].fields, "Completed Date")
})
var (
txns []RawTxn
skipped = map[string]int{}
currencies = map[string]int{}
)
for _, rec := range records {
get := func(name string) string { return index.get(rec.fields, name) }
if state := get("State"); state != "COMPLETED" || get("Balance") == "" {
if state == "" {
state = "unfinished"
}
skipped[state]++
continue
}
if currency := strings.ToUpper(get("Currency")); currency != p.currency {
currencies[currency]++
continue
}
txn, err := p.row(get, p.digits)
if err != nil {
return nil, fmt.Errorf("%s: row %d: %w", path, rec.line, err)
}
txns = append(txns, txn)
}
for _, state := range sortedKeys(skipped) {
p.warnings = append(p.warnings, fmt.Sprintf("skipped %d %s transactions", skipped[state], state))
}
for _, currency := range sortedKeys(currencies) {
p.warnings = append(p.warnings, fmt.Sprintf(
"skipped %d rows in %s; give that currency its own account folder to import them",
currencies[currency], currency))
}
return txns, nil
}
func (p *revolutParser) row(get func(string) string, digits int) (RawTxn, error) {
d, err := time.Parse("2006-01-02", firstN(get("Completed Date"), 10))
if err != nil {
return RawTxn{}, fmt.Errorf("completed date %q: %w", get("Completed Date"), err)
}
amount, err := ParseAmount(get("Amount"), ".", "", digits)
if err != nil {
return RawTxn{}, fmt.Errorf("amount: %w", err)
}
var fee int64
if raw := strings.TrimSpace(get("Fee")); raw != "" {
if fee, err = ParseAmount(raw, ".", "", digits); err != nil {
return RawTxn{}, fmt.Errorf("fee: %w", err)
}
}
balance, err := ParseAmount(get("Balance"), ".", "", digits)
if err != nil {
return RawTxn{}, fmt.Errorf("balance: %w", err)
}
// The fee is deducted along with the amount rather than booked separately,
// so it has to be folded in for the balance chain to hold.
desc := strings.TrimSpace(get("Description"))
if fee != 0 {
desc += fmt.Sprintf(" (fee %s)", model.FormatMinor(fee, digits))
}
counterparty := revolutIBAN.FindString(desc)
return RawTxn{
Date: d.Format("2006-01-02"),
Description: desc,
AmountMinor: amount - fee,
Counterparty: counterparty,
Type: strings.TrimSpace(get("Type")),
BalanceMinor: &balance,
}, nil
}
// headerIndex maps a Revolut column name to its position.
type headerIndex map[string]int
func (h headerIndex) get(fields []string, name string) string {
i, ok := h[name]
if !ok || i >= len(fields) {
return ""
}
return strings.TrimSpace(fields[i])
}
func revolutHeaderIndex(header []string) (headerIndex, error) {
index := headerIndex{}
for i, name := range header {
index[strings.TrimSpace(name)] = i
}
var missing []string
for _, name := range revolutColumns {
if _, ok := index[name]; !ok {
missing = append(missing, name)
}
}
if len(missing) > 0 {
return nil, fmt.Errorf("statement is missing the %s column(s); header was %v",
strings.Join(missing, ", "), header)
}
return index, nil
}
func sortedKeys(m map[string]int) []string {
out := make([]string, 0, len(m))
for k := range m {
out = append(out, k)
}
sort.Strings(out)
return out
}
func firstN(s string, n int) string {
if len(s) < n {
return s
}
return s[:n]
}