Add money: statement-driven personal finance tracker
A data directory holds one folder per account. Statements dropped into those folders are parsed into a rebuildable SQLite index, categorised by ordered glob rules in rules.toml, and browsed or hand-tagged in a Bubble Tea TUI. Movements between the user's own accounts are marked as transfers by the same rules and excluded from spending totals. Manual tags and transfer marks are stored separately from the rule-derived ones and always win, so editing rules.toml and re-running retag never destroys hand edits. Parsers are pluggable. Three are ported from the Python extractors they replace -- nlb and traderepublic read PDFs via pdftotext -layout, revolut reads the CSV export -- alongside a configurable-column CSV parser and a cmd parser that shells out to an external script. Both ports fix two latent bugs in the originals: the sign character class rejected the typographic minus U+2212 that some PDF fonts emit, and NLB's hardcoded continuation indent broke when pdftotext compressed runs of spaces, so the threshold is now measured from the description column. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,109 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ParseAmount converts a decimal string from a statement into signed minor
|
||||
// units. It is deliberately forgiving, because statements are messy:
|
||||
//
|
||||
// "1.234,56" (decimal ",", thousands ".") -> 123456
|
||||
// "-45.20" -> -4520
|
||||
// "45,20-" (trailing minus) -> -4520
|
||||
// "(45.20)" (accounting negative) -> -4520
|
||||
// "1 234.56 EUR" -> 123456
|
||||
//
|
||||
// decimal defaults to "."; every other separator is discarded, so thousands is
|
||||
// only a readability hint in account.toml and needs no special handling here.
|
||||
// digits is the account's minor-unit scale; extra fraction digits are rounded
|
||||
// half away from zero.
|
||||
func ParseAmount(s, decimal, thousands string, digits int) (int64, error) {
|
||||
_ = thousands // dropped along with all other non-decimal separators
|
||||
orig := s
|
||||
if decimal == "" {
|
||||
decimal = "."
|
||||
}
|
||||
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return 0, fmt.Errorf("empty amount")
|
||||
}
|
||||
|
||||
neg := false
|
||||
if strings.HasPrefix(s, "(") && strings.HasSuffix(s, ")") {
|
||||
neg = true
|
||||
s = strings.TrimSuffix(strings.TrimPrefix(s, "("), ")")
|
||||
}
|
||||
if strings.HasSuffix(s, "-") {
|
||||
neg = true
|
||||
s = strings.TrimSuffix(s, "-")
|
||||
}
|
||||
|
||||
// Keep only digits and the decimal separator. Everything else -- currency
|
||||
// codes, symbols, thin and non-breaking spaces, thousands separators --
|
||||
// is noise. A minus anywhere flips the sign; a plus is ignored.
|
||||
var b strings.Builder
|
||||
for _, r := range s {
|
||||
switch {
|
||||
case r >= '0' && r <= '9':
|
||||
b.WriteRune(r)
|
||||
case r == '-' || r == '−':
|
||||
neg = !neg
|
||||
case string(r) == decimal:
|
||||
b.WriteRune('.')
|
||||
}
|
||||
}
|
||||
clean := b.String()
|
||||
if clean == "" || clean == "." {
|
||||
return 0, fmt.Errorf("cannot parse amount %q", orig)
|
||||
}
|
||||
|
||||
intPart, fracPart, _ := strings.Cut(clean, ".")
|
||||
if strings.Contains(fracPart, ".") {
|
||||
return 0, fmt.Errorf("cannot parse amount %q: multiple decimal separators", orig)
|
||||
}
|
||||
if intPart == "" {
|
||||
intPart = "0"
|
||||
}
|
||||
|
||||
whole, err := strconv.ParseInt(intPart, 10, 64)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("cannot parse amount %q: %w", orig, err)
|
||||
}
|
||||
|
||||
scale := int64(1)
|
||||
for range digits {
|
||||
scale *= 10
|
||||
}
|
||||
|
||||
var frac int64
|
||||
if digits > 0 {
|
||||
switch {
|
||||
case len(fracPart) > digits:
|
||||
// Round half away from zero on the first dropped digit.
|
||||
kept, err := strconv.ParseInt(fracPart[:digits], 10, 64)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("cannot parse amount %q: %w", orig, err)
|
||||
}
|
||||
frac = kept
|
||||
if fracPart[digits] >= '5' {
|
||||
frac++
|
||||
}
|
||||
case fracPart != "":
|
||||
padded := fracPart + strings.Repeat("0", digits-len(fracPart))
|
||||
if frac, err = strconv.ParseInt(padded, 10, 64); err != nil {
|
||||
return 0, fmt.Errorf("cannot parse amount %q: %w", orig, err)
|
||||
}
|
||||
}
|
||||
} else if fracPart != "" && fracPart[0] >= '5' {
|
||||
whole++
|
||||
}
|
||||
|
||||
v := whole*scale + frac
|
||||
if neg {
|
||||
v = -v
|
||||
}
|
||||
return v, nil
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
package parser
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestParseAmount(t *testing.T) {
|
||||
cases := []struct {
|
||||
in string
|
||||
decimal string
|
||||
thousands string
|
||||
digits int
|
||||
want int64
|
||||
}{
|
||||
{"45.20", ".", "", 2, 4520},
|
||||
{"-45.20", ".", "", 2, -4520},
|
||||
{"45,20", ",", ".", 2, 4520},
|
||||
{"1.234,56", ",", ".", 2, 123456},
|
||||
{"1,234.56", ".", ",", 2, 123456},
|
||||
{"1 234.56", ".", " ", 2, 123456},
|
||||
{"45.20 EUR", ".", "", 2, 4520},
|
||||
{"BGN 45.20", ".", "", 2, 4520},
|
||||
{"45,20-", ",", "", 2, -4520},
|
||||
{"(45.20)", ".", "", 2, -4520},
|
||||
{"+45.20", ".", "", 2, 4520},
|
||||
{"0.00", ".", "", 2, 0},
|
||||
{".50", ".", "", 2, 50},
|
||||
{"45", ".", "", 2, 4500},
|
||||
{"45.2", ".", "", 2, 4520},
|
||||
{"45.205", ".", "", 2, 4521}, // round half away from zero
|
||||
{"45.204", ".", "", 2, 4520}, // round down
|
||||
{"-45.205", ".", "", 2, -4521},
|
||||
{"1234", ".", "", 0, 1234}, // zero-digit currency
|
||||
{"1234.6", ".", "", 0, 1235},
|
||||
{"1.234567", ".", "", 3, 1235},
|
||||
{"−45.20", ".", "", 2, -4520}, // U+2212 minus
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, err := ParseAmount(c.in, c.decimal, c.thousands, c.digits)
|
||||
if err != nil {
|
||||
t.Errorf("ParseAmount(%q) error: %v", c.in, err)
|
||||
continue
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("ParseAmount(%q, decimal=%q, digits=%d) = %d, want %d",
|
||||
c.in, c.decimal, c.digits, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseAmountErrors(t *testing.T) {
|
||||
for _, in := range []string{"", " ", "abc", "-", "."} {
|
||||
if v, err := ParseAmount(in, ".", "", 2); err == nil {
|
||||
t.Errorf("ParseAmount(%q) = %d, want error", in, v)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/csv"
|
||||
"fmt"
|
||||
"io"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"git.petrovv.com/nikola/money/internal/config"
|
||||
)
|
||||
|
||||
func init() {
|
||||
Register("cmd", newCmdParser)
|
||||
}
|
||||
|
||||
// FileToken is replaced with the statement's absolute path in a cmd parser's argv.
|
||||
const FileToken = "{{file}}"
|
||||
|
||||
// cmdRunTimeout bounds an extractor run so a hung script cannot wedge an import.
|
||||
const cmdRunTimeout = 2 * time.Minute
|
||||
|
||||
// cmdParser runs an external extractor and reads normalised CSV from its
|
||||
// stdout: date,description,amount with any further columns ignored. This is
|
||||
// the bridge that lets the existing Python extractors be used unchanged.
|
||||
type cmdParser struct {
|
||||
cfg config.CmdConfig
|
||||
digits int
|
||||
}
|
||||
|
||||
func newCmdParser(acc *config.Account) (Parser, error) {
|
||||
if acc.Cmd == nil || len(acc.Cmd.Argv) == 0 {
|
||||
return nil, fmt.Errorf("account %s: parser \"cmd\" requires [cmd] with a non-empty argv", acc.Slug)
|
||||
}
|
||||
cfg := *acc.Cmd
|
||||
if cfg.Layout == "" {
|
||||
cfg.Layout = "2006-01-02"
|
||||
}
|
||||
if !hasFileToken(cfg.Argv) {
|
||||
return nil, fmt.Errorf("account %s: [cmd] argv must contain %s so the script knows which file to read",
|
||||
acc.Slug, FileToken)
|
||||
}
|
||||
return &cmdParser{cfg: cfg, digits: acc.Digits()}, nil
|
||||
}
|
||||
|
||||
func hasFileToken(argv []string) bool {
|
||||
for _, a := range argv {
|
||||
if strings.Contains(a, FileToken) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func (p *cmdParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
|
||||
argv := make([]string, len(p.cfg.Argv))
|
||||
for i, a := range p.cfg.Argv {
|
||||
argv[i] = strings.ReplaceAll(a, FileToken, path)
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), cmdRunTimeout)
|
||||
defer cancel()
|
||||
|
||||
cmd := exec.CommandContext(ctx, argv[0], argv[1:]...)
|
||||
cmd.Dir = acc.Dir
|
||||
var stdout, stderr bytes.Buffer
|
||||
cmd.Stdout = &stdout
|
||||
cmd.Stderr = &stderr
|
||||
|
||||
if err := cmd.Run(); err != nil {
|
||||
msg := strings.TrimSpace(stderr.String())
|
||||
if ctx.Err() == context.DeadlineExceeded {
|
||||
return nil, fmt.Errorf("extractor %v timed out after %s", argv, cmdRunTimeout)
|
||||
}
|
||||
if msg != "" {
|
||||
return nil, fmt.Errorf("extractor %v failed: %w: %s", argv, err, msg)
|
||||
}
|
||||
return nil, fmt.Errorf("extractor %v failed: %w", argv, err)
|
||||
}
|
||||
|
||||
return p.parseOutput(&stdout, argv)
|
||||
}
|
||||
|
||||
func (p *cmdParser) parseOutput(out io.Reader, argv []string) ([]RawTxn, error) {
|
||||
r := csv.NewReader(out)
|
||||
r.FieldsPerRecord = -1
|
||||
r.LazyQuotes = true
|
||||
|
||||
var txns []RawTxn
|
||||
for row := 0; ; row++ {
|
||||
rec, err := r.Read()
|
||||
if err == io.EOF {
|
||||
break
|
||||
}
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("extractor %v: output row %d: %w", argv, row+1, err)
|
||||
}
|
||||
if row < p.cfg.SkipRows || isBlank(rec) {
|
||||
continue
|
||||
}
|
||||
if len(rec) < 3 {
|
||||
return nil, fmt.Errorf("extractor %v: output row %d has %d columns, want date,description,amount",
|
||||
argv, row+1, len(rec))
|
||||
}
|
||||
d, err := time.Parse(p.cfg.Layout, strings.TrimSpace(rec[0]))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("extractor %v: output row %d: date %q does not match layout %q",
|
||||
argv, row+1, rec[0], p.cfg.Layout)
|
||||
}
|
||||
amount, err := ParseAmount(rec[2], ".", "", p.digits)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("extractor %v: output row %d: %w", argv, row+1, err)
|
||||
}
|
||||
txns = append(txns, RawTxn{
|
||||
Date: d.Format("2006-01-02"),
|
||||
Description: strings.TrimSpace(rec[1]),
|
||||
AmountMinor: amount,
|
||||
})
|
||||
}
|
||||
return txns, nil
|
||||
}
|
||||
@@ -0,0 +1,175 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"encoding/csv"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"git.petrovv.com/nikola/money/internal/config"
|
||||
)
|
||||
|
||||
func init() {
|
||||
Register("csv", newCSVParser)
|
||||
}
|
||||
|
||||
// csvParser reads a delimited statement using column positions from
|
||||
// account.toml. Amounts come either from one signed column, or from a
|
||||
// debit/credit pair.
|
||||
type csvParser struct {
|
||||
cfg config.CSVConfig
|
||||
digits int
|
||||
}
|
||||
|
||||
func newCSVParser(acc *config.Account) (Parser, error) {
|
||||
if acc.CSV == nil {
|
||||
return nil, fmt.Errorf("account %s: parser \"csv\" requires a [csv] section", acc.Slug)
|
||||
}
|
||||
cfg := *acc.CSV
|
||||
if cfg.Amount == nil && cfg.Debit == nil && cfg.Credit == nil {
|
||||
return nil, fmt.Errorf("account %s: [csv] needs either amount or debit/credit columns", acc.Slug)
|
||||
}
|
||||
if cfg.Amount != nil && (cfg.Debit != nil || cfg.Credit != nil) {
|
||||
return nil, fmt.Errorf("account %s: [csv] sets both amount and debit/credit; pick one", acc.Slug)
|
||||
}
|
||||
if cfg.Date.Layout == "" {
|
||||
return nil, fmt.Errorf("account %s: [csv] date needs a layout, e.g. layout = \"02.01.2006\"", acc.Slug)
|
||||
}
|
||||
return &csvParser{cfg: cfg, digits: acc.Digits()}, nil
|
||||
}
|
||||
|
||||
func (p *csvParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
r := csv.NewReader(f)
|
||||
r.FieldsPerRecord = -1 // statements are ragged more often than not
|
||||
r.LazyQuotes = true
|
||||
if d := p.cfg.Delimiter; d != "" {
|
||||
runes := []rune(d)
|
||||
if len(runes) != 1 {
|
||||
return nil, fmt.Errorf("delimiter %q must be a single character", d)
|
||||
}
|
||||
r.Comma = runes[0]
|
||||
}
|
||||
|
||||
var out []RawTxn
|
||||
for row := 0; ; row++ {
|
||||
rec, err := r.Read()
|
||||
if err == io.EOF {
|
||||
break
|
||||
}
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: row %d: %w", path, row+1, err)
|
||||
}
|
||||
if row < p.cfg.SkipRows {
|
||||
continue
|
||||
}
|
||||
if isBlank(rec) {
|
||||
continue
|
||||
}
|
||||
txn, err := p.row(rec)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: row %d: %w", path, row+1, err)
|
||||
}
|
||||
out = append(out, txn)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (p *csvParser) row(rec []string) (RawTxn, error) {
|
||||
var t RawTxn
|
||||
|
||||
raw, err := field(rec, p.cfg.Date.Col)
|
||||
if err != nil {
|
||||
return t, fmt.Errorf("date column: %w", err)
|
||||
}
|
||||
d, err := time.Parse(p.cfg.Date.Layout, strings.TrimSpace(raw))
|
||||
if err != nil {
|
||||
return t, fmt.Errorf("date %q does not match layout %q", raw, p.cfg.Date.Layout)
|
||||
}
|
||||
t.Date = d.Format("2006-01-02")
|
||||
|
||||
desc, err := field(rec, p.cfg.Description.Col)
|
||||
if err != nil {
|
||||
return t, fmt.Errorf("description column: %w", err)
|
||||
}
|
||||
t.Description = strings.TrimSpace(desc)
|
||||
|
||||
switch {
|
||||
case p.cfg.Amount != nil:
|
||||
raw, err := field(rec, p.cfg.Amount.Col)
|
||||
if err != nil {
|
||||
return t, fmt.Errorf("amount column: %w", err)
|
||||
}
|
||||
v, err := ParseAmount(raw, p.cfg.Amount.Decimal, p.cfg.Amount.Thousands, p.digits)
|
||||
if err != nil {
|
||||
return t, err
|
||||
}
|
||||
t.AmountMinor = v
|
||||
default:
|
||||
// Debit/credit pair: exactly one of the two carries a value, and both
|
||||
// are written as positive numbers.
|
||||
debit, err := p.optional(rec, p.cfg.Debit)
|
||||
if err != nil {
|
||||
return t, fmt.Errorf("debit column: %w", err)
|
||||
}
|
||||
credit, err := p.optional(rec, p.cfg.Credit)
|
||||
if err != nil {
|
||||
return t, fmt.Errorf("credit column: %w", err)
|
||||
}
|
||||
if debit != 0 && credit != 0 {
|
||||
return t, fmt.Errorf("both debit (%d) and credit (%d) are set", debit, credit)
|
||||
}
|
||||
t.AmountMinor = credit - abs(debit)
|
||||
}
|
||||
|
||||
if p.cfg.Invert {
|
||||
t.AmountMinor = -t.AmountMinor
|
||||
}
|
||||
return t, nil
|
||||
}
|
||||
|
||||
// optional parses a column that may legitimately be blank, as debit/credit
|
||||
// columns always are for half the rows.
|
||||
func (p *csvParser) optional(rec []string, c *config.Column) (int64, error) {
|
||||
if c == nil {
|
||||
return 0, nil
|
||||
}
|
||||
raw, err := field(rec, c.Col)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if strings.TrimSpace(raw) == "" {
|
||||
return 0, nil
|
||||
}
|
||||
return ParseAmount(raw, c.Decimal, c.Thousands, p.digits)
|
||||
}
|
||||
|
||||
func field(rec []string, i int) (string, error) {
|
||||
if i < 0 || i >= len(rec) {
|
||||
return "", fmt.Errorf("index %d out of range, row has %d columns", i, len(rec))
|
||||
}
|
||||
return rec[i], nil
|
||||
}
|
||||
|
||||
func isBlank(rec []string) bool {
|
||||
for _, f := range rec {
|
||||
if strings.TrimSpace(f) != "" {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func abs(v int64) int64 {
|
||||
if v < 0 {
|
||||
return -v
|
||||
}
|
||||
return v
|
||||
}
|
||||
@@ -0,0 +1,112 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"git.petrovv.com/nikola/money/internal/config"
|
||||
)
|
||||
|
||||
func writeFile(t *testing.T, name, content string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), name)
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
func TestCSVSignedAmountColumn(t *testing.T) {
|
||||
path := writeFile(t, "st.csv", `Date,Ref,Description,Amount
|
||||
02.01.2026,X1,LIDL SOFIA 1234,"-45,20"
|
||||
03.01.2026,X2,ACME PAYROLL,"1.500,00"
|
||||
|
||||
04.01.2026,X3,COFFEE,"-3,50"
|
||||
`)
|
||||
acc := &config.Account{Slug: "checking", Currency: "EUR", Parser: "csv", CSV: &config.CSVConfig{
|
||||
SkipRows: 1,
|
||||
Date: config.Column{Col: 0, Layout: "02.01.2006"},
|
||||
Description: config.Column{Col: 2},
|
||||
Amount: &config.Column{Col: 3, Decimal: ",", Thousands: "."},
|
||||
}}
|
||||
p, err := For(acc)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := p.Parse(path, acc)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := []RawTxn{
|
||||
{Date: "2026-01-02", Description: "LIDL SOFIA 1234", AmountMinor: -4520},
|
||||
{Date: "2026-01-03", Description: "ACME PAYROLL", AmountMinor: 150000},
|
||||
{Date: "2026-01-04", Description: "COFFEE", AmountMinor: -350},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("got %d txns, want %d: %+v", len(got), len(want), got)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("txn %d = %+v, want %+v", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestCSVDebitCreditPair(t *testing.T) {
|
||||
path := writeFile(t, "st.csv", `2026-02-01;RENT;800.00;
|
||||
2026-02-05;SALARY;;2500.00
|
||||
`)
|
||||
acc := &config.Account{Slug: "checking", Currency: "EUR", Parser: "csv", CSV: &config.CSVConfig{
|
||||
Delimiter: ";",
|
||||
Date: config.Column{Col: 0, Layout: "2006-01-02"},
|
||||
Description: config.Column{Col: 1},
|
||||
Debit: &config.Column{Col: 2},
|
||||
Credit: &config.Column{Col: 3},
|
||||
}}
|
||||
p, err := For(acc)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := p.Parse(path, acc)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("got %d txns, want 2: %+v", len(got), got)
|
||||
}
|
||||
if got[0].AmountMinor != -80000 {
|
||||
t.Errorf("debit row = %d, want -80000", got[0].AmountMinor)
|
||||
}
|
||||
if got[1].AmountMinor != 250000 {
|
||||
t.Errorf("credit row = %d, want 250000", got[1].AmountMinor)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCSVConfigErrors(t *testing.T) {
|
||||
cases := map[string]*config.Account{
|
||||
"no csv section": {Slug: "a", Parser: "csv"},
|
||||
"no amount columns": {Slug: "a", Parser: "csv", CSV: &config.CSVConfig{
|
||||
Date: config.Column{Layout: "2006-01-02"},
|
||||
}},
|
||||
"amount and debit together": {Slug: "a", Parser: "csv", CSV: &config.CSVConfig{
|
||||
Date: config.Column{Layout: "2006-01-02"},
|
||||
Amount: &config.Column{Col: 1},
|
||||
Debit: &config.Column{Col: 2},
|
||||
}},
|
||||
"no date layout": {Slug: "a", Parser: "csv", CSV: &config.CSVConfig{
|
||||
Amount: &config.Column{Col: 1},
|
||||
}},
|
||||
}
|
||||
for name, acc := range cases {
|
||||
if _, err := For(acc); err == nil {
|
||||
t.Errorf("%s: expected an error", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnknownParser(t *testing.T) {
|
||||
if _, err := For(&config.Account{Slug: "a", Parser: "nope"}); err == nil {
|
||||
t.Error("expected an error for an unregistered parser")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"regexp"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"git.petrovv.com/nikola/money/internal/config"
|
||||
)
|
||||
|
||||
func init() {
|
||||
Register("nlb", func(acc *config.Account) (Parser, error) {
|
||||
return &nlbParser{digits: acc.Digits()}, nil
|
||||
})
|
||||
}
|
||||
|
||||
// nlbParser reads an NLB izpisek PDF.
|
||||
//
|
||||
// A transaction starts on a line beginning with a dd.mm.yy date and ends with
|
||||
// the signed amount and the running balance. Long descriptions and the
|
||||
// counterparty account wrap onto indented continuation lines below.
|
||||
type nlbParser struct {
|
||||
digits int
|
||||
}
|
||||
|
||||
var (
|
||||
// A transaction line starts with a two-digit date.
|
||||
nlbStart = regexp.MustCompile(`^\d{2}\.\d{2}\.\d{2}\s`)
|
||||
// The whole line: date, free-form middle, signed amount, running balance.
|
||||
// Both the ASCII hyphen and the typographic minus U+2212 count as a sign;
|
||||
// which one a PDF carries depends on the font it was produced with.
|
||||
nlbLine = regexp.MustCompile(
|
||||
`^(?P<date>\d{2}\.\d{2}\.\d{2})\s+` +
|
||||
`(?P<middle>.*?)\s+` +
|
||||
`(?P<amount>[-+\x{2212}][\d.,]+)\s+` +
|
||||
`(?P<balance>[-\x{2212}]?[\d.,]+[-\x{2212}]?)\s*$`)
|
||||
// Columns inside a line are separated by four or more spaces.
|
||||
nlbFieldSplit = regexp.MustCompile(`\s{4,}`)
|
||||
// A Slovenian IBAN, which is the counterparty account when present.
|
||||
nlbAccount = regexp.MustCompile(`^SI\d{2}(?:\s?\d{4}){3}\s?\d{3}$`)
|
||||
)
|
||||
|
||||
// nlbMinContinuationIndent is the shallowest indent a wrapped description line
|
||||
// may have. The real threshold is the description column of the transaction
|
||||
// the line belongs to, measured per line rather than hardcoded: pdftotext
|
||||
// squeezes runs of spaces, so absolute columns shift with the font and page
|
||||
// size of the PDF. This floor only rejects flush-left page furniture.
|
||||
const nlbMinContinuationIndent = 2
|
||||
|
||||
// nlbDateLayout is the two-digit-year date NLB prints.
|
||||
const nlbDateLayout = "02.01.06"
|
||||
|
||||
func (p *nlbParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
|
||||
text, err := pdfToText(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return parseNLBText(text, p.digits)
|
||||
}
|
||||
|
||||
// parseNLBText holds the whole parser, separated from PDF extraction so it can
|
||||
// be tested against captured pdftotext output.
|
||||
func parseNLBText(text string, digits int) ([]RawTxn, error) {
|
||||
var (
|
||||
txns []RawTxn
|
||||
accts []string // counterparty per transaction, built up alongside
|
||||
descAt int // description column of the last transaction line
|
||||
)
|
||||
|
||||
for _, page := range pages(text) {
|
||||
// Continuation lines may only attach to a transaction started on the
|
||||
// same page, so a wrapped line at the top of a page is page furniture.
|
||||
pageStart := len(txns)
|
||||
|
||||
for i, line := range strings.Split(page, "\n") {
|
||||
if strings.TrimSpace(line) == "" {
|
||||
continue
|
||||
}
|
||||
|
||||
if nlbStart.MatchString(line) {
|
||||
txn, account, col, err := parseNLBLine(line, digits)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("line %d: %w", i+1, err)
|
||||
}
|
||||
txns = append(txns, txn)
|
||||
accts = append(accts, account)
|
||||
descAt = col
|
||||
continue
|
||||
}
|
||||
|
||||
if len(txns) <= pageStart {
|
||||
continue // header, before any transaction on this page
|
||||
}
|
||||
// A wrapped description sits under the description column of the
|
||||
// transaction it continues; anything to the left of that is a
|
||||
// footer or a column heading.
|
||||
threshold := max(descAt, nlbMinContinuationIndent)
|
||||
if indent(line) < threshold {
|
||||
continue
|
||||
}
|
||||
|
||||
parts := nlbFieldSplit.Split(strings.TrimSpace(line), -1)
|
||||
last := len(txns) - 1
|
||||
txns[last].Description += " " + parts[0]
|
||||
if len(parts) > 1 {
|
||||
accts[last] = strings.TrimSpace(accts[last] + " " + parts[1])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for i := range txns {
|
||||
txns[i].Counterparty = accts[i]
|
||||
txns[i].Description = strings.Join(strings.Fields(txns[i].Description), " ")
|
||||
}
|
||||
return txns, nil
|
||||
}
|
||||
|
||||
// parseNLBLine parses one transaction line, returning the transaction, the
|
||||
// counterparty account found among its columns, and the column at which the
|
||||
// description starts, which is where its wrapped lines will sit.
|
||||
func parseNLBLine(line string, digits int) (RawTxn, string, int, error) {
|
||||
trimmed := strings.TrimRight(line, " \t\r")
|
||||
loc := nlbLine.FindStringSubmatchIndex(trimmed)
|
||||
if loc == nil {
|
||||
return RawTxn{}, "", 0, fmt.Errorf("line starts with a date but has no amount and balance: %q", strings.TrimSpace(line))
|
||||
}
|
||||
group := func(name string) string {
|
||||
i := nlbLine.SubexpIndex(name) * 2
|
||||
if loc[i] < 0 {
|
||||
return ""
|
||||
}
|
||||
return trimmed[loc[i]:loc[i+1]]
|
||||
}
|
||||
var (
|
||||
date = group("date")
|
||||
middle = group("middle")
|
||||
amount = group("amount")
|
||||
balance = group("balance")
|
||||
descCol = loc[nlbLine.SubexpIndex("middle")*2]
|
||||
)
|
||||
|
||||
d, err := time.Parse(nlbDateLayout, date)
|
||||
if err != nil {
|
||||
return RawTxn{}, "", 0, fmt.Errorf("date %q: %w", date, err)
|
||||
}
|
||||
|
||||
amountMinor, err := parseNLBNumber(amount, digits)
|
||||
if err != nil {
|
||||
return RawTxn{}, "", 0, fmt.Errorf("amount %q: %w", amount, err)
|
||||
}
|
||||
balanceMinor, err := parseNLBNumber(balance, digits)
|
||||
if err != nil {
|
||||
return RawTxn{}, "", 0, fmt.Errorf("balance %q: %w", balance, err)
|
||||
}
|
||||
|
||||
// The middle holds the description and, sometimes, the counterparty IBAN.
|
||||
var (
|
||||
account string
|
||||
desc []string
|
||||
)
|
||||
for _, f := range nlbFieldSplit.Split(strings.TrimSpace(middle), -1) {
|
||||
if f == "" {
|
||||
continue
|
||||
}
|
||||
if nlbAccount.MatchString(f) {
|
||||
account = f
|
||||
continue
|
||||
}
|
||||
desc = append(desc, f)
|
||||
}
|
||||
|
||||
return RawTxn{
|
||||
Date: d.Format("2006-01-02"),
|
||||
Description: strings.Join(desc, " "),
|
||||
AmountMinor: amountMinor,
|
||||
BalanceMinor: &balanceMinor,
|
||||
}, account, descCol, nil
|
||||
}
|
||||
|
||||
// parseNLBNumber reads the 1.234,56 format, where a trailing minus marks a
|
||||
// negative balance.
|
||||
func parseNLBNumber(s string, digits int) (int64, error) {
|
||||
return ParseAmount(s, ",", ".", digits)
|
||||
}
|
||||
@@ -0,0 +1,116 @@
|
||||
package parser
|
||||
|
||||
import "testing"
|
||||
|
||||
// A page as pdftotext -layout renders it: a header, transaction lines ending
|
||||
// in amount and balance, and indented continuation lines carrying wrapped
|
||||
// descriptions and the counterparty IBAN.
|
||||
// The description column sits at 15, which is what the script's
|
||||
// CONTINUATION_INDENT tells us about the real layout.
|
||||
const nlbPage = ` NLB d.d.
|
||||
IZPISEK 001/2026
|
||||
|
||||
DATUM OPIS ZNESEK STANJE
|
||||
|
||||
02.01.26 PLACILO S KARTICO -45,20 1.234,56
|
||||
LIDL SOFIA 4412 SI56 1234 5678 9012 345
|
||||
05.01.26 PRILIV PLACE +2.500,00 3.734,56
|
||||
ACME DOO
|
||||
10.01.26 PRENOS NA VARCEVALNI -500,00 3.234,56
|
||||
15.01.26 NEGATIVNO STANJE -3.300,00 65,44-
|
||||
|
||||
Stran 1
|
||||
`
|
||||
|
||||
func TestParseNLBText(t *testing.T) {
|
||||
txns, err := parseNLBText(nlbPage, 2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(txns) != 4 {
|
||||
t.Fatalf("got %d transactions, want 4: %+v", len(txns), txns)
|
||||
}
|
||||
|
||||
first := txns[0]
|
||||
if first.Date != "2026-01-02" {
|
||||
t.Errorf("date = %q, want 2026-01-02", first.Date)
|
||||
}
|
||||
// The continuation line is appended to the description.
|
||||
if first.Description != "PLACILO S KARTICO LIDL SOFIA 4412" {
|
||||
t.Errorf("description = %q", first.Description)
|
||||
}
|
||||
// ...and the IBAN in its second column becomes the counterparty.
|
||||
if first.Counterparty != "SI56 1234 5678 9012 345" {
|
||||
t.Errorf("counterparty = %q", first.Counterparty)
|
||||
}
|
||||
if first.AmountMinor != -4520 {
|
||||
t.Errorf("amount = %d, want -4520", first.AmountMinor)
|
||||
}
|
||||
if first.BalanceMinor == nil || *first.BalanceMinor != 123456 {
|
||||
t.Errorf("balance = %v, want 123456", first.BalanceMinor)
|
||||
}
|
||||
|
||||
if txns[1].AmountMinor != 250000 {
|
||||
t.Errorf("credit amount = %d, want 250000", txns[1].AmountMinor)
|
||||
}
|
||||
if txns[1].Description != "PRILIV PLACE ACME DOO" {
|
||||
t.Errorf("description = %q", txns[1].Description)
|
||||
}
|
||||
|
||||
// A transaction with no continuation line keeps an empty counterparty.
|
||||
if txns[2].Counterparty != "" {
|
||||
t.Errorf("counterparty = %q, want empty", txns[2].Counterparty)
|
||||
}
|
||||
|
||||
// A trailing minus marks a negative balance.
|
||||
if txns[3].BalanceMinor == nil || *txns[3].BalanceMinor != -6544 {
|
||||
t.Errorf("balance = %v, want -6544", txns[3].BalanceMinor)
|
||||
}
|
||||
}
|
||||
|
||||
// A wrapped line at the top of a page must not attach to the last transaction
|
||||
// of the previous page.
|
||||
func TestParseNLBPageBoundary(t *testing.T) {
|
||||
text := "02.01.26 FIRST -10,00 100,00\n" +
|
||||
"\f" +
|
||||
" STRAY CONTINUATION AT TOP OF PAGE\n" +
|
||||
"03.01.26 SECOND -20,00 80,00\n" +
|
||||
" REAL CONTINUATION\n"
|
||||
|
||||
txns, err := parseNLBText(text, 2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(txns) != 2 {
|
||||
t.Fatalf("got %d transactions, want 2", len(txns))
|
||||
}
|
||||
if txns[0].Description != "FIRST" {
|
||||
t.Errorf("first description = %q, want FIRST (page furniture must not leak in)", txns[0].Description)
|
||||
}
|
||||
if txns[1].Description != "SECOND REAL CONTINUATION" {
|
||||
t.Errorf("second description = %q", txns[1].Description)
|
||||
}
|
||||
}
|
||||
|
||||
// Shallowly indented lines are footers, not wrapped descriptions.
|
||||
func TestParseNLBIgnoresShallowIndent(t *testing.T) {
|
||||
text := "02.01.26 FIRST -10,00 100,00\n" +
|
||||
" Stran 1 od 2\n"
|
||||
|
||||
txns, err := parseNLBText(text, 2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(txns) != 1 || txns[0].Description != "FIRST" {
|
||||
t.Errorf("got %+v, want a single FIRST transaction", txns)
|
||||
}
|
||||
}
|
||||
|
||||
// A date-led line without an amount and balance is a real problem, not
|
||||
// something to silently drop.
|
||||
func TestParseNLBRejectsMalformedLine(t *testing.T) {
|
||||
text := "02.01.26 NO AMOUNT OR BALANCE HERE\n"
|
||||
if _, err := parseNLBText(text, 2); err == nil {
|
||||
t.Error("expected an error for a transaction line with no amount")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,91 @@
|
||||
// Package parser turns a statement file into raw transactions.
|
||||
//
|
||||
// Every bank needs its own extraction logic, so parsers are looked up by name
|
||||
// from a registry. Two are built in: "csv" for delimited exports with
|
||||
// configurable columns, and "cmd" for shelling out to an external extractor
|
||||
// (which is how the existing Python scripts are used until they are ported).
|
||||
// Ported extractors register themselves here and become usable by putting
|
||||
// their name in an account.toml.
|
||||
package parser
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"sync"
|
||||
|
||||
"git.petrovv.com/nikola/money/internal/config"
|
||||
)
|
||||
|
||||
// RawTxn is one line as it came out of a statement, before fingerprinting,
|
||||
// deduplication or tagging.
|
||||
type RawTxn struct {
|
||||
Date string // YYYY-MM-DD
|
||||
Description string
|
||||
AmountMinor int64 // signed; negative is an outflow
|
||||
|
||||
// Counterparty is the other side's account number, where the statement
|
||||
// gives one. Optional.
|
||||
Counterparty string
|
||||
// Type is the bank's own classification of the transaction. Optional.
|
||||
Type string
|
||||
// BalanceMinor is the running balance after this transaction, when the
|
||||
// statement reports one. Optional; enables the balance-chain check.
|
||||
BalanceMinor *int64
|
||||
}
|
||||
|
||||
// Parser extracts transactions from a single statement file.
|
||||
type Parser interface {
|
||||
// Parse reads the statement at path. digits is the account's minor-unit
|
||||
// scale, so parsers can convert decimal strings without guessing.
|
||||
Parse(path string, acc *config.Account) ([]RawTxn, error)
|
||||
}
|
||||
|
||||
// Warner is an optional interface for parsers that legitimately drop rows --
|
||||
// pending transactions, other currencies -- and want to say so. The importer
|
||||
// collects the warnings from the most recent Parse call and reports them.
|
||||
type Warner interface {
|
||||
Warnings() []string
|
||||
}
|
||||
|
||||
// Factory builds a parser from an account's config, validating it up front so
|
||||
// a bad account.toml fails before any file is read.
|
||||
type Factory func(acc *config.Account) (Parser, error)
|
||||
|
||||
var (
|
||||
mu sync.RWMutex
|
||||
registry = map[string]Factory{}
|
||||
)
|
||||
|
||||
// Register adds a named parser. It panics on a duplicate name, since that can
|
||||
// only be a programming error at init time.
|
||||
func Register(name string, f Factory) {
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
if _, dup := registry[name]; dup {
|
||||
panic("parser: duplicate registration of " + name)
|
||||
}
|
||||
registry[name] = f
|
||||
}
|
||||
|
||||
// For builds the parser named by acc.Parser.
|
||||
func For(acc *config.Account) (Parser, error) {
|
||||
mu.RLock()
|
||||
f, ok := registry[acc.Parser]
|
||||
mu.RUnlock()
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("account %s: unknown parser %q (available: %v)", acc.Slug, acc.Parser, Names())
|
||||
}
|
||||
return f(acc)
|
||||
}
|
||||
|
||||
// Names lists the registered parsers, for error messages and `money parsers`.
|
||||
func Names() []string {
|
||||
mu.RLock()
|
||||
defer mu.RUnlock()
|
||||
out := make([]string, 0, len(registry))
|
||||
for name := range registry {
|
||||
out = append(out, name)
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// pdfToTextTimeout bounds extraction so a malformed PDF cannot wedge an import.
|
||||
const pdfToTextTimeout = 2 * time.Minute
|
||||
|
||||
// pdfToText renders a PDF as text with the original column positions
|
||||
// preserved, which is what the statement parsers key off.
|
||||
//
|
||||
// This shells out to poppler's pdftotext rather than decoding the PDF in Go:
|
||||
// the layout reconstruction it does is the whole reason the column-based
|
||||
// parsers work, and no Go library matches it.
|
||||
func pdfToText(path string) (string, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), pdfToTextTimeout)
|
||||
defer cancel()
|
||||
|
||||
cmd := exec.CommandContext(ctx, "pdftotext", "-layout", path, "-")
|
||||
var stdout, stderr bytes.Buffer
|
||||
cmd.Stdout = &stdout
|
||||
cmd.Stderr = &stderr
|
||||
|
||||
if err := cmd.Run(); err != nil {
|
||||
if ctx.Err() == context.DeadlineExceeded {
|
||||
return "", fmt.Errorf("pdftotext timed out after %s on %s", pdfToTextTimeout, path)
|
||||
}
|
||||
if errors := strings.TrimSpace(stderr.String()); errors != "" {
|
||||
return "", fmt.Errorf("pdftotext %s: %w: %s", path, err, errors)
|
||||
}
|
||||
if _, lookErr := exec.LookPath("pdftotext"); lookErr != nil {
|
||||
return "", fmt.Errorf("pdftotext is not installed (it ships with poppler-utils): %w", lookErr)
|
||||
}
|
||||
return "", fmt.Errorf("pdftotext %s: %w", path, err)
|
||||
}
|
||||
return stdout.String(), nil
|
||||
}
|
||||
|
||||
// pages splits pdftotext output on form feeds.
|
||||
func pages(text string) []string {
|
||||
return strings.Split(text, "\f")
|
||||
}
|
||||
|
||||
// indent counts the leading spaces of a line.
|
||||
func indent(line string) int {
|
||||
return len(line) - len(strings.TrimLeft(line, " "))
|
||||
}
|
||||
@@ -0,0 +1,222 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"encoding/csv"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"git.petrovv.com/nikola/money/internal/config"
|
||||
"git.petrovv.com/nikola/money/internal/model"
|
||||
)
|
||||
|
||||
func init() {
|
||||
Register("revolut", func(acc *config.Account) (Parser, error) {
|
||||
return &revolutParser{digits: acc.Digits(), currency: strings.ToUpper(acc.Currency)}, nil
|
||||
})
|
||||
}
|
||||
|
||||
// revolutParser reads a Revolut account-statement CSV export.
|
||||
//
|
||||
// Revolut books the fee alongside the transaction rather than as its own line,
|
||||
// and reports pending transactions that have no balance yet. Both are handled
|
||||
// the way the original extraction script did: the fee is folded into the
|
||||
// amount, and anything not COMPLETED is skipped.
|
||||
//
|
||||
// A single export can mix currencies. Since an account here has one currency,
|
||||
// rows in others are skipped and reported; importing them is a matter of
|
||||
// giving that currency its own account folder.
|
||||
type revolutParser struct {
|
||||
digits int
|
||||
currency string
|
||||
warnings []string
|
||||
}
|
||||
|
||||
// revolutIBAN finds a counterparty account inside a description.
|
||||
var revolutIBAN = regexp.MustCompile(`\b[A-Z]{2}\d{2}[A-Z0-9]{11,30}\b`)
|
||||
|
||||
// Columns the parser needs; a missing one is a hard error rather than a
|
||||
// silently empty field.
|
||||
var revolutColumns = []string{
|
||||
"Type", "Completed Date", "Description", "Amount", "Fee", "Currency", "State", "Balance",
|
||||
}
|
||||
|
||||
func (p *revolutParser) Warnings() []string { return p.warnings }
|
||||
|
||||
func (p *revolutParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
|
||||
p.warnings = nil
|
||||
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
r := csv.NewReader(f)
|
||||
r.FieldsPerRecord = -1
|
||||
r.LazyQuotes = true
|
||||
|
||||
header, err := r.Read()
|
||||
if err == io.EOF {
|
||||
return nil, fmt.Errorf("%s is empty", path)
|
||||
}
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", path, err)
|
||||
}
|
||||
index, err := revolutHeaderIndex(header)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", path, err)
|
||||
}
|
||||
|
||||
type record struct {
|
||||
fields []string
|
||||
line int
|
||||
}
|
||||
var records []record
|
||||
for line := 2; ; line++ {
|
||||
rec, err := r.Read()
|
||||
if err == io.EOF {
|
||||
break
|
||||
}
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: row %d: %w", path, line, err)
|
||||
}
|
||||
if isBlank(rec) {
|
||||
continue
|
||||
}
|
||||
records = append(records, record{fields: rec, line: line})
|
||||
}
|
||||
|
||||
// The export is not in date order, and the running balance only makes
|
||||
// sense along one. A stable sort also keeps identical rows in file order,
|
||||
// which the importer's deduplication relies on.
|
||||
sort.SliceStable(records, func(i, j int) bool {
|
||||
return index.get(records[i].fields, "Completed Date") < index.get(records[j].fields, "Completed Date")
|
||||
})
|
||||
|
||||
var (
|
||||
txns []RawTxn
|
||||
skipped = map[string]int{}
|
||||
currencies = map[string]int{}
|
||||
)
|
||||
for _, rec := range records {
|
||||
get := func(name string) string { return index.get(rec.fields, name) }
|
||||
|
||||
if state := get("State"); state != "COMPLETED" || get("Balance") == "" {
|
||||
if state == "" {
|
||||
state = "unfinished"
|
||||
}
|
||||
skipped[state]++
|
||||
continue
|
||||
}
|
||||
if currency := strings.ToUpper(get("Currency")); currency != p.currency {
|
||||
currencies[currency]++
|
||||
continue
|
||||
}
|
||||
|
||||
txn, err := p.row(get, p.digits)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: row %d: %w", path, rec.line, err)
|
||||
}
|
||||
txns = append(txns, txn)
|
||||
}
|
||||
|
||||
for _, state := range sortedKeys(skipped) {
|
||||
p.warnings = append(p.warnings, fmt.Sprintf("skipped %d %s transactions", skipped[state], state))
|
||||
}
|
||||
for _, currency := range sortedKeys(currencies) {
|
||||
p.warnings = append(p.warnings, fmt.Sprintf(
|
||||
"skipped %d rows in %s; give that currency its own account folder to import them",
|
||||
currencies[currency], currency))
|
||||
}
|
||||
return txns, nil
|
||||
}
|
||||
|
||||
func (p *revolutParser) row(get func(string) string, digits int) (RawTxn, error) {
|
||||
d, err := time.Parse("2006-01-02", firstN(get("Completed Date"), 10))
|
||||
if err != nil {
|
||||
return RawTxn{}, fmt.Errorf("completed date %q: %w", get("Completed Date"), err)
|
||||
}
|
||||
|
||||
amount, err := ParseAmount(get("Amount"), ".", "", digits)
|
||||
if err != nil {
|
||||
return RawTxn{}, fmt.Errorf("amount: %w", err)
|
||||
}
|
||||
var fee int64
|
||||
if raw := strings.TrimSpace(get("Fee")); raw != "" {
|
||||
if fee, err = ParseAmount(raw, ".", "", digits); err != nil {
|
||||
return RawTxn{}, fmt.Errorf("fee: %w", err)
|
||||
}
|
||||
}
|
||||
balance, err := ParseAmount(get("Balance"), ".", "", digits)
|
||||
if err != nil {
|
||||
return RawTxn{}, fmt.Errorf("balance: %w", err)
|
||||
}
|
||||
|
||||
// The fee is deducted along with the amount rather than booked separately,
|
||||
// so it has to be folded in for the balance chain to hold.
|
||||
desc := strings.TrimSpace(get("Description"))
|
||||
if fee != 0 {
|
||||
desc += fmt.Sprintf(" (fee %s)", model.FormatMinor(fee, digits))
|
||||
}
|
||||
|
||||
counterparty := revolutIBAN.FindString(desc)
|
||||
|
||||
return RawTxn{
|
||||
Date: d.Format("2006-01-02"),
|
||||
Description: desc,
|
||||
AmountMinor: amount - fee,
|
||||
Counterparty: counterparty,
|
||||
Type: strings.TrimSpace(get("Type")),
|
||||
BalanceMinor: &balance,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// headerIndex maps a Revolut column name to its position.
|
||||
type headerIndex map[string]int
|
||||
|
||||
func (h headerIndex) get(fields []string, name string) string {
|
||||
i, ok := h[name]
|
||||
if !ok || i >= len(fields) {
|
||||
return ""
|
||||
}
|
||||
return strings.TrimSpace(fields[i])
|
||||
}
|
||||
|
||||
func revolutHeaderIndex(header []string) (headerIndex, error) {
|
||||
index := headerIndex{}
|
||||
for i, name := range header {
|
||||
index[strings.TrimSpace(name)] = i
|
||||
}
|
||||
var missing []string
|
||||
for _, name := range revolutColumns {
|
||||
if _, ok := index[name]; !ok {
|
||||
missing = append(missing, name)
|
||||
}
|
||||
}
|
||||
if len(missing) > 0 {
|
||||
return nil, fmt.Errorf("statement is missing the %s column(s); header was %v",
|
||||
strings.Join(missing, ", "), header)
|
||||
}
|
||||
return index, nil
|
||||
}
|
||||
|
||||
func sortedKeys(m map[string]int) []string {
|
||||
out := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
out = append(out, k)
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
|
||||
func firstN(s string, n int) string {
|
||||
if len(s) < n {
|
||||
return s
|
||||
}
|
||||
return s[:n]
|
||||
}
|
||||
@@ -0,0 +1,119 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"git.petrovv.com/nikola/money/internal/config"
|
||||
)
|
||||
|
||||
// A Revolut export: out of date order, mixed currencies, a pending row, a row
|
||||
// with a fee, and a transfer carrying an IBAN.
|
||||
const revolutCSV = `Type,Product,Started Date,Completed Date,Description,Amount,Fee,Currency,State,Balance
|
||||
CARD_PAYMENT,Current,2026-01-06 09:12:00,2026-01-06 09:12:00,LIDL SOFIA 4412,-45.20,0.00,EUR,COMPLETED,1234.56
|
||||
TOPUP,Current,2026-01-05 08:00:00,2026-01-05 08:00:00,Payment from ACME,2500.00,0.00,EUR,COMPLETED,1279.76
|
||||
TRANSFER,Current,2026-01-10 10:00:00,2026-01-10 10:00:00,To savings SI56123456789012345,-500.00,0.35,EUR,COMPLETED,734.41
|
||||
CARD_PAYMENT,Current,2026-01-11 10:00:00,,Pending coffee,-3.50,0.00,EUR,PENDING,
|
||||
EXCHANGE,Current,2026-01-12 10:00:00,2026-01-12 10:00:00,Tokyo hotel,-15000,0,JPY,COMPLETED,50000
|
||||
`
|
||||
|
||||
func revolutAccount(currency string, digits int) *config.Account {
|
||||
return &config.Account{Slug: "revolut", Currency: currency, MinorDigits: &digits, Parser: "revolut"}
|
||||
}
|
||||
|
||||
func TestRevolutParse(t *testing.T) {
|
||||
path := writeFile(t, "account-statement.csv", revolutCSV)
|
||||
acc := revolutAccount("EUR", 2)
|
||||
|
||||
p, err := For(acc)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
txns, err := p.Parse(path, acc)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(txns) != 3 {
|
||||
t.Fatalf("got %d transactions, want 3: %+v", len(txns), txns)
|
||||
}
|
||||
|
||||
// Rows come back in completed-date order, not file order.
|
||||
wantDates := []string{"2026-01-05", "2026-01-06", "2026-01-10"}
|
||||
for i, want := range wantDates {
|
||||
if txns[i].Date != want {
|
||||
t.Errorf("txn %d date = %q, want %q", i, txns[i].Date, want)
|
||||
}
|
||||
}
|
||||
|
||||
if txns[0].Type != "TOPUP" {
|
||||
t.Errorf("type = %q, want TOPUP", txns[0].Type)
|
||||
}
|
||||
if txns[1].AmountMinor != -4520 {
|
||||
t.Errorf("card payment = %d, want -4520", txns[1].AmountMinor)
|
||||
}
|
||||
|
||||
// The fee is folded into the amount and noted in the description.
|
||||
transfer := txns[2]
|
||||
if transfer.AmountMinor != -50035 {
|
||||
t.Errorf("transfer with fee = %d, want -50035 (amount minus fee)", transfer.AmountMinor)
|
||||
}
|
||||
if !strings.Contains(transfer.Description, "(fee 0.35)") {
|
||||
t.Errorf("description = %q, want a fee note", transfer.Description)
|
||||
}
|
||||
if transfer.Counterparty != "SI56123456789012345" {
|
||||
t.Errorf("counterparty = %q, want the IBAN from the description", transfer.Counterparty)
|
||||
}
|
||||
if transfer.BalanceMinor == nil || *transfer.BalanceMinor != 73441 {
|
||||
t.Errorf("balance = %v, want 73441", transfer.BalanceMinor)
|
||||
}
|
||||
|
||||
// Pending and foreign-currency rows are skipped, and said so.
|
||||
warnings := strings.Join(p.(Warner).Warnings(), "\n")
|
||||
if !strings.Contains(warnings, "1 PENDING") {
|
||||
t.Errorf("warnings = %q, want a note about the pending row", warnings)
|
||||
}
|
||||
if !strings.Contains(warnings, "JPY") {
|
||||
t.Errorf("warnings = %q, want a note about the JPY row", warnings)
|
||||
}
|
||||
}
|
||||
|
||||
// The same export read as a JPY account picks up the row the EUR account skipped.
|
||||
func TestRevolutOtherCurrencyViaSecondAccount(t *testing.T) {
|
||||
path := writeFile(t, "account-statement.csv", revolutCSV)
|
||||
acc := revolutAccount("JPY", 0)
|
||||
|
||||
p, err := For(acc)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
txns, err := p.Parse(path, acc)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(txns) != 1 {
|
||||
t.Fatalf("got %d transactions, want 1: %+v", len(txns), txns)
|
||||
}
|
||||
// JPY has no minor unit, so the amount is whole yen.
|
||||
if txns[0].AmountMinor != -15000 {
|
||||
t.Errorf("amount = %d, want -15000", txns[0].AmountMinor)
|
||||
}
|
||||
if txns[0].BalanceMinor == nil || *txns[0].BalanceMinor != 50000 {
|
||||
t.Errorf("balance = %v, want 50000", txns[0].BalanceMinor)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRevolutMissingColumn(t *testing.T) {
|
||||
path := writeFile(t, "bad.csv", "Type,Description,Amount\nCARD_PAYMENT,X,-1.00\n")
|
||||
acc := revolutAccount("EUR", 2)
|
||||
p, err := For(acc)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, err = p.Parse(path, acc)
|
||||
if err == nil {
|
||||
t.Fatal("expected an error for a statement missing required columns")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "Balance") {
|
||||
t.Errorf("error = %v, want it to name the missing columns", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,233 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"regexp"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"git.petrovv.com/nikola/money/internal/config"
|
||||
)
|
||||
|
||||
func init() {
|
||||
Register("traderepublic", func(acc *config.Account) (Parser, error) {
|
||||
return &tradeRepublicParser{digits: acc.Digits()}, nil
|
||||
})
|
||||
}
|
||||
|
||||
// tradeRepublicParser reads a Trade Republic account statement PDF.
|
||||
//
|
||||
// The layout varies between statements: a transaction may sit entirely on one
|
||||
// line, or have its date, type and description stacked across several. Rather
|
||||
// than guess, every token is assigned to whichever column heading its start
|
||||
// position is closest to, which handles both shapes.
|
||||
type tradeRepublicParser struct {
|
||||
digits int
|
||||
}
|
||||
|
||||
var (
|
||||
trHeader = regexp.MustCompile(`^\s*DATE\b.*\bMONEY IN\b.*\bMONEY OUT\b.*\bBALANCE\b`)
|
||||
// The sign may be an ASCII hyphen or a typographic minus, depending on the
|
||||
// font the PDF was produced with.
|
||||
trAmount = regexp.MustCompile(`[-\x{2212}]?€\s?[-\x{2212}]?[\d,]+\.\d{2}`)
|
||||
trToken = regexp.MustCompile(`\S+`)
|
||||
trFullDate = regexp.MustCompile(`^\d{2} [A-Z][a-z]{2} \d{4}$`)
|
||||
trIBAN = regexp.MustCompile(`\b[A-Z]{2}\d{2}[A-Z0-9]{11,30}\b`)
|
||||
)
|
||||
|
||||
var (
|
||||
trTextColumns = []string{"DATE", "TYPE", "DESCRIPTION"}
|
||||
trMoneyColumns = []string{"MONEY IN", "MONEY OUT", "BALANCE"}
|
||||
)
|
||||
|
||||
// trAmountSlack lets an amount sit slightly left of the MONEY IN column
|
||||
// without being mistaken for description text.
|
||||
const trAmountSlack = 5
|
||||
|
||||
// trDateLayout is the date Trade Republic prints, e.g. "05 Jan 2026".
|
||||
const trDateLayout = "02 Jan 2006"
|
||||
|
||||
func (p *tradeRepublicParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
|
||||
text, err := pdfToText(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return parseTradeRepublicText(text, p.digits)
|
||||
}
|
||||
|
||||
// parseTradeRepublicText holds the whole parser, separated from PDF extraction
|
||||
// so it can be tested against captured pdftotext output.
|
||||
func parseTradeRepublicText(text string, digits int) ([]RawTxn, error) {
|
||||
var txns []RawTxn
|
||||
|
||||
for _, page := range pages(text) {
|
||||
lines := strings.Split(page, "\n")
|
||||
header := -1
|
||||
for i, line := range lines {
|
||||
if trHeader.MatchString(line) {
|
||||
header = i
|
||||
break
|
||||
}
|
||||
}
|
||||
if header < 0 {
|
||||
continue // a cover page or disclaimer, with no transaction table
|
||||
}
|
||||
|
||||
cols, err := trColumnStarts(lines[header])
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
for _, block := range trBlocks(lines[header+1:]) {
|
||||
txn, ok, err := parseTradeRepublicBlock(block, cols, digits)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if ok {
|
||||
txns = append(txns, txn)
|
||||
}
|
||||
}
|
||||
}
|
||||
return txns, nil
|
||||
}
|
||||
|
||||
// trColumnStarts records where each heading begins on the header line; those
|
||||
// positions are what every token is measured against.
|
||||
func trColumnStarts(header string) (map[string]int, error) {
|
||||
cols := map[string]int{}
|
||||
for _, name := range append(append([]string{}, trTextColumns...), trMoneyColumns...) {
|
||||
i := strings.Index(header, name)
|
||||
if i < 0 {
|
||||
return nil, fmt.Errorf("statement header has no %q column: %q", name, strings.TrimSpace(header))
|
||||
}
|
||||
cols[name] = i
|
||||
}
|
||||
return cols, nil
|
||||
}
|
||||
|
||||
// trBlocks groups the consecutive non-blank lines that make up one transaction.
|
||||
func trBlocks(lines []string) [][]string {
|
||||
var (
|
||||
out [][]string
|
||||
block []string
|
||||
)
|
||||
for _, line := range lines {
|
||||
if strings.TrimSpace(line) != "" {
|
||||
block = append(block, line)
|
||||
continue
|
||||
}
|
||||
if len(block) > 0 {
|
||||
out = append(out, block)
|
||||
block = nil
|
||||
}
|
||||
}
|
||||
if len(block) > 0 {
|
||||
out = append(out, block)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// nearest returns the column whose start is closest to pos.
|
||||
func nearest(pos int, cols map[string]int, names []string) string {
|
||||
best, bestDist := "", -1
|
||||
for _, name := range names {
|
||||
d := pos - cols[name]
|
||||
if d < 0 {
|
||||
d = -d
|
||||
}
|
||||
if bestDist < 0 || d < bestDist {
|
||||
best, bestDist = name, d
|
||||
}
|
||||
}
|
||||
return best
|
||||
}
|
||||
|
||||
func parseTradeRepublicBlock(lines []string, cols map[string]int, digits int) (RawTxn, bool, error) {
|
||||
words := map[string][]string{}
|
||||
amounts := map[string]*int64{}
|
||||
var descChunks []string
|
||||
|
||||
for _, line := range lines {
|
||||
// Amounts are found first: everything to their left is text, and the
|
||||
// cut keeps them from being read as description tokens.
|
||||
cut := len(line)
|
||||
for _, loc := range trAmount.FindAllStringIndex(line, -1) {
|
||||
if loc[0] < cols["MONEY IN"]-trAmountSlack {
|
||||
continue // a figure inside the description, not a money column
|
||||
}
|
||||
if loc[0] < cut {
|
||||
cut = loc[0]
|
||||
}
|
||||
v, err := parseTradeRepublicNumber(line[loc[0]:loc[1]], digits)
|
||||
if err != nil {
|
||||
return RawTxn{}, false, fmt.Errorf("amount %q: %w", line[loc[0]:loc[1]], err)
|
||||
}
|
||||
amounts[nearest(loc[0], cols, trMoneyColumns)] = &v
|
||||
}
|
||||
|
||||
// A description fragment per line, so wrapped text can be rejoined.
|
||||
var lineDesc []string
|
||||
for _, loc := range trToken.FindAllStringIndex(line[:cut], -1) {
|
||||
column := nearest(loc[0], cols, trTextColumns)
|
||||
token := line[loc[0]:loc[1]]
|
||||
if column == "DESCRIPTION" {
|
||||
lineDesc = append(lineDesc, token)
|
||||
continue
|
||||
}
|
||||
words[column] = append(words[column], token)
|
||||
}
|
||||
if len(lineDesc) > 0 {
|
||||
descChunks = append(descChunks, strings.Join(lineDesc, " "))
|
||||
}
|
||||
}
|
||||
|
||||
// A block without a full date and a balance is a heading or a footer.
|
||||
date := strings.Join(words["DATE"], " ")
|
||||
if !trFullDate.MatchString(date) || amounts["BALANCE"] == nil {
|
||||
return RawTxn{}, false, nil
|
||||
}
|
||||
d, err := time.Parse(trDateLayout, date)
|
||||
if err != nil {
|
||||
return RawTxn{}, false, fmt.Errorf("date %q: %w", date, err)
|
||||
}
|
||||
|
||||
desc := trJoinWrapped(descChunks)
|
||||
amount := deref(amounts["MONEY IN"]) - deref(amounts["MONEY OUT"])
|
||||
|
||||
return RawTxn{
|
||||
Date: d.Format("2006-01-02"),
|
||||
Description: desc,
|
||||
AmountMinor: amount,
|
||||
Counterparty: trIBAN.FindString(desc),
|
||||
Type: strings.Join(words["TYPE"], " "),
|
||||
BalanceMinor: amounts["BALANCE"],
|
||||
}, true, nil
|
||||
}
|
||||
|
||||
// trJoinWrapped glues description fragments split across lines. A fragment
|
||||
// ending in a hyphen was broken mid-word, so it joins without a space.
|
||||
func trJoinWrapped(chunks []string) string {
|
||||
var out string
|
||||
for _, chunk := range chunks {
|
||||
switch {
|
||||
case out == "":
|
||||
out = chunk
|
||||
case len(out) > 1 && strings.HasSuffix(out, "-") && !strings.HasSuffix(out, " -"):
|
||||
out += chunk
|
||||
default:
|
||||
out += " " + chunk
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// parseTradeRepublicNumber reads the €1,234.56 format.
|
||||
func parseTradeRepublicNumber(s string, digits int) (int64, error) {
|
||||
return ParseAmount(strings.ReplaceAll(s, "€", ""), ".", ",", digits)
|
||||
}
|
||||
|
||||
func deref(v *int64) int64 {
|
||||
if v == nil {
|
||||
return 0
|
||||
}
|
||||
return *v
|
||||
}
|
||||
@@ -0,0 +1,167 @@
|
||||
package parser
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// place builds a fixed-width line by putting each string at a given column,
|
||||
// which keeps these fixtures honest about the layout the parser measures.
|
||||
func place(cells map[int]string) string {
|
||||
width := 0
|
||||
for col, text := range cells {
|
||||
if end := col + len(text); end > width {
|
||||
width = end
|
||||
}
|
||||
}
|
||||
line := []byte(strings.Repeat(" ", width))
|
||||
for col, text := range cells {
|
||||
copy(line[col:], text)
|
||||
}
|
||||
return string(line)
|
||||
}
|
||||
|
||||
// Column starts, spaced as a real statement is: wide enough that every token
|
||||
// of "05 Jan 2026" stays closest to DATE rather than drifting into TYPE.
|
||||
const (
|
||||
colDate = 2
|
||||
colType = 20
|
||||
colDesc = 40
|
||||
colMoneyIn = 75
|
||||
colMoneyOut = 90
|
||||
colBalance = 105
|
||||
)
|
||||
|
||||
func trHeaderLine() string {
|
||||
return place(map[int]string{
|
||||
colDate: "DATE", colType: "TYPE", colDesc: "DESCRIPTION",
|
||||
colMoneyIn: "MONEY IN", colMoneyOut: "MONEY OUT", colBalance: "BALANCE",
|
||||
})
|
||||
}
|
||||
|
||||
// A statement mixing both layouts: one transaction on a single line, one with
|
||||
// its fields stacked across three, and one whose description wraps.
|
||||
func trStatement() string {
|
||||
lines := []string{
|
||||
" Trade Republic Bank GmbH",
|
||||
"",
|
||||
trHeaderLine(),
|
||||
"",
|
||||
// Everything on one line.
|
||||
place(map[int]string{
|
||||
colDate: "05 Jan 2026", colType: "Deposit", colDesc: "Payment from ACME",
|
||||
colMoneyIn: "€2,500.00", colBalance: "€2,500.00",
|
||||
}),
|
||||
"",
|
||||
// Stacked across three lines.
|
||||
place(map[int]string{
|
||||
colDate: "06 Jan 2026", colMoneyOut: "€45.20", colBalance: "€2,454.80",
|
||||
}),
|
||||
place(map[int]string{colType: "Card"}),
|
||||
place(map[int]string{colDesc: "LIDL SOFIA 4412"}),
|
||||
"",
|
||||
// Wrapped description, hyphen-broken on the first line, plus an IBAN.
|
||||
place(map[int]string{
|
||||
colDate: "10 Jan 2026", colType: "Transfer", colDesc: "Standing order to sav-",
|
||||
colMoneyOut: "€500.00", colBalance: "€1,954.80",
|
||||
}),
|
||||
place(map[int]string{colDesc: "ings SI56123456789012345"}),
|
||||
"",
|
||||
" Page 1 of 2",
|
||||
}
|
||||
return strings.Join(lines, "\n")
|
||||
}
|
||||
|
||||
func TestParseTradeRepublicText(t *testing.T) {
|
||||
txns, err := parseTradeRepublicText(trStatement(), 2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(txns) != 3 {
|
||||
t.Fatalf("got %d transactions, want 3: %+v", len(txns), txns)
|
||||
}
|
||||
|
||||
deposit := txns[0]
|
||||
if deposit.Date != "2026-01-05" {
|
||||
t.Errorf("date = %q, want 2026-01-05", deposit.Date)
|
||||
}
|
||||
if deposit.AmountMinor != 250000 {
|
||||
t.Errorf("money in = %d, want 250000", deposit.AmountMinor)
|
||||
}
|
||||
if deposit.Type != "Deposit" {
|
||||
t.Errorf("type = %q, want Deposit", deposit.Type)
|
||||
}
|
||||
if deposit.Description != "Payment from ACME" {
|
||||
t.Errorf("description = %q", deposit.Description)
|
||||
}
|
||||
|
||||
// The stacked layout must produce the same shape as the single-line one.
|
||||
card := txns[1]
|
||||
if card.Date != "2026-01-06" {
|
||||
t.Errorf("date = %q, want 2026-01-06", card.Date)
|
||||
}
|
||||
if card.AmountMinor != -4520 {
|
||||
t.Errorf("money out = %d, want -4520", card.AmountMinor)
|
||||
}
|
||||
if card.Type != "Card" {
|
||||
t.Errorf("type = %q, want Card", card.Type)
|
||||
}
|
||||
if card.Description != "LIDL SOFIA 4412" {
|
||||
t.Errorf("description = %q", card.Description)
|
||||
}
|
||||
if card.BalanceMinor == nil || *card.BalanceMinor != 245480 {
|
||||
t.Errorf("balance = %v, want 245480", card.BalanceMinor)
|
||||
}
|
||||
|
||||
// A hyphen-broken word rejoins without a space. The hyphen itself is kept,
|
||||
// as the original script does, because descriptions contain real hyphens
|
||||
// that must not be swallowed.
|
||||
transfer := txns[2]
|
||||
if transfer.Description != "Standing order to sav-ings SI56123456789012345" {
|
||||
t.Errorf("description = %q, want the wrapped fragment joined without a space", transfer.Description)
|
||||
}
|
||||
if transfer.Counterparty != "SI56123456789012345" {
|
||||
t.Errorf("counterparty = %q", transfer.Counterparty)
|
||||
}
|
||||
if transfer.AmountMinor != -50000 {
|
||||
t.Errorf("amount = %d, want -50000", transfer.AmountMinor)
|
||||
}
|
||||
}
|
||||
|
||||
// Pages with no transaction table (cover pages, disclaimers) are skipped.
|
||||
func TestParseTradeRepublicSkipsPagesWithoutTable(t *testing.T) {
|
||||
text := " Some cover page with no table at all\n" + "\f" + trStatement()
|
||||
txns, err := parseTradeRepublicText(text, 2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(txns) != 3 {
|
||||
t.Errorf("got %d transactions, want 3", len(txns))
|
||||
}
|
||||
}
|
||||
|
||||
// A figure inside the description must not be mistaken for a money column.
|
||||
func TestParseTradeRepublicIgnoresAmountsInDescription(t *testing.T) {
|
||||
text := strings.Join([]string{
|
||||
trHeaderLine(),
|
||||
"",
|
||||
place(map[int]string{
|
||||
colDate: "05 Jan 2026", colType: "Card", colDesc: "Refund of €12.00 order",
|
||||
colMoneyIn: "€12.00", colBalance: "€100.00",
|
||||
}),
|
||||
}, "\n")
|
||||
|
||||
txns, err := parseTradeRepublicText(text, 2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(txns) != 1 {
|
||||
t.Fatalf("got %d transactions, want 1", len(txns))
|
||||
}
|
||||
if txns[0].AmountMinor != 1200 {
|
||||
t.Errorf("amount = %d, want 1200", txns[0].AmountMinor)
|
||||
}
|
||||
if !strings.Contains(txns[0].Description, "€12.00") {
|
||||
t.Errorf("description = %q, want the inline figure kept", txns[0].Description)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user