Files
money/internal/importer/importer_test.go
T
nikolaandClaude Opus 5 16c2585637 Drop counterparty, the generic parsers and .money/
counterparty was a structured field only nlb could fill honestly. revolut
and traderepublic invented one by running an IBAN-shaped regex over the
description they had just built, and the two spellings disagreed --
SI56 1234 5678 9012 345 against SI56123456789012345 -- so a literal rule
pattern that worked on one account silently matched nothing on another. It
is gone from the model, the index, the rule keys, ls --wide and the rules
screen. nlb now appends its IBAN column to the end of the description,
where the other two already keep theirs, so match = "*SI56*" works
everywhere. That changes those descriptions and with them their
fingerprints, so a statement overlapping an already-imported period will
re-add rather than dedupe those rows until the index is rebuilt. An index
built by an older binary drops the column when it is opened.

The index itself moves from .money/index.db up to index.db beside
rules.toml. Nothing looks in the old location, so an existing one has to be
moved by hand -- otherwise the tool quietly starts a fresh index and the
manual tags in the old file, the only thing statements cannot reproduce,
stay behind in it.

The csv and cmd parsers are gone along with the [csv] and [cmd] config they
carried. cmd shelled out to the Python extractors, which were ported to Go
and deleted, so it bridged to nothing; csv was a generic column-mapped
fallback that no account used, and between them they were the largest
configuration surface in the tool. A bank is now described in Go, where it
can be tested. The importer tests register their own three-column parser
rather than borrow a bank's, so they stay about the directory walk, dedupe
and per-file error reporting.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-11 22:51:20 +02:00

259 lines
8.1 KiB
Go

package importer
import (
"fmt"
"os"
"path/filepath"
"strings"
"testing"
"time"
"git.petrovv.com/nikola/money/internal/config"
"git.petrovv.com/nikola/money/internal/parser"
"git.petrovv.com/nikola/money/internal/rules"
"git.petrovv.com/nikola/money/internal/store"
)
const accountTOML = `
name = "Checking"
currency = "EUR"
parser = "test"
`
// These tests are about the directory walk, the checksum skip, dedupe and
// per-file error reporting -- not about any bank's layout. Driving them with a
// real bank parser would drag that bank's quirks (Revolut's fee folding and
// COMPLETED filter, the PDF parsers' dependency on pdftotext) into every
// fixture, so they register the smallest parser that will do instead.
func init() {
parser.Register("test", func(acc *config.Account) (parser.Parser, error) {
return testParser{digits: acc.Digits()}, nil
})
}
// testParser reads "date,description,amount" with one header row.
type testParser struct{ digits int }
func (p testParser) Parse(path string, acc *config.Account) ([]parser.RawTxn, error) {
body, err := os.ReadFile(path)
if err != nil {
return nil, err
}
var txns []parser.RawTxn
for i, line := range strings.Split(strings.TrimSpace(string(body)), "\n") {
if i == 0 || strings.TrimSpace(line) == "" {
continue // header
}
fields := strings.Split(line, ",")
if len(fields) != 3 {
return nil, fmt.Errorf("row %d: got %d fields, want 3", i+1, len(fields))
}
if _, err := time.Parse("2006-01-02", fields[0]); err != nil {
return nil, fmt.Errorf("row %d: %w", i+1, err)
}
amount, err := parser.ParseAmount(fields[2], ".", "", p.digits)
if err != nil {
return nil, fmt.Errorf("row %d: %w", i+1, err)
}
txns = append(txns, parser.RawTxn{Date: fields[0], Description: fields[1], AmountMinor: amount})
}
return txns, nil
}
// newRoot builds a data root with one account and the given statement files.
func newRoot(t *testing.T, statements map[string]string) (string, *store.DB, []*config.Account, *rules.Engine) {
t.Helper()
root := t.TempDir()
dir := filepath.Join(root, "checking")
if err := os.MkdirAll(dir, 0o755); err != nil {
t.Fatal(err)
}
write(t, filepath.Join(dir, config.AccountFile), accountTOML)
for name, body := range statements {
write(t, filepath.Join(dir, name), body)
}
db, err := store.Open(config.IndexPath(root))
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() { db.Close() })
accounts, err := config.LoadAccounts(root)
if err != nil {
t.Fatal(err)
}
engine := rules.New(&config.Rules{Rule: []config.Rule{{Match: "*LIDL*", Tag: "groceries"}}})
return root, db, accounts, engine
}
func write(t *testing.T, path, body string) {
t.Helper()
if err := os.WriteFile(path, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
}
func mustRun(t *testing.T, root string, db *store.DB, accounts []*config.Account, e *rules.Engine, opts Options) Result {
t.Helper()
res, err := Run(root, db, accounts, e, opts)
if err != nil {
t.Fatal(err)
}
for _, f := range res.Errs() {
t.Fatalf("import of %s failed: %v", f.Path, f.Err)
}
return res
}
// Identical lines inside one statement are distinct transactions; the same
// line seen again in an overlapping statement is not.
func TestDedupe(t *testing.T) {
root, db, accounts, engine := newRoot(t, map[string]string{
"2026-01.csv": `date,description,amount
2026-01-06,LIDL SOFIA,-45.20
2026-01-06,LIDL SOFIA,-45.20
2026-01-10,RENT,-800.00
`,
})
res := mustRun(t, root, db, accounts, engine, Options{})
if _, added, _ := res.Total(); added != 3 {
t.Fatalf("first import added %d rows, want 3 (identical same-day lines must both survive)", added)
}
// Unchanged file: skipped without even parsing.
res = mustRun(t, root, db, accounts, engine, Options{})
if parsed, added, _ := res.Total(); parsed != 0 || added != 0 {
t.Errorf("re-import parsed %d and added %d, want 0 and 0", parsed, added)
}
// Same file, parsed again: every row is recognised as a duplicate.
res = mustRun(t, root, db, accounts, engine, Options{Force: true})
if _, added, skipped := res.Total(); added != 0 || skipped != 3 {
t.Errorf("forced re-import added %d, skipped %d; want 0 and 3", added, skipped)
}
// An overlapping statement contributes only its genuinely new rows.
write(t, filepath.Join(root, "checking", "2026-02.csv"), `date,description,amount
2026-01-10,RENT,-800.00
2026-02-10,RENT,-800.00
`)
accounts, err := config.LoadAccounts(root)
if err != nil {
t.Fatal(err)
}
res = mustRun(t, root, db, accounts, engine, Options{})
if _, added, skipped := res.Total(); added != 1 || skipped != 1 {
t.Errorf("overlapping import added %d, skipped %d; want 1 and 1", added, skipped)
}
txns, err := db.Transactions(store.Filter{})
if err != nil {
t.Fatal(err)
}
if len(txns) != 4 {
t.Errorf("index holds %d transactions, want 4", len(txns))
}
}
// Rules are applied as rows are inserted, so a fresh import is already tagged.
func TestImportAppliesRules(t *testing.T) {
root, db, accounts, engine := newRoot(t, map[string]string{
"2026-01.csv": `date,description,amount
2026-01-06,LIDL SOFIA,-45.20
2026-01-10,RENT,-800.00
`,
})
res := mustRun(t, root, db, accounts, engine, Options{})
if res.Retagged != 0 {
t.Errorf("retag changed %d rows after import, want 0: rules should already be applied", res.Retagged)
}
untagged, err := db.Transactions(store.Filter{Untagged: true})
if err != nil {
t.Fatal(err)
}
if len(untagged) != 1 || untagged[0].Description != "RENT" {
t.Errorf("untagged = %+v, want only RENT", untagged)
}
}
// A broken statement must be reported without aborting the rest of the run.
func TestBadFileIsReportedNotFatal(t *testing.T) {
root, db, accounts, engine := newRoot(t, map[string]string{
"good.csv": `date,description,amount
2026-01-06,LIDL SOFIA,-45.20
`,
"bad.csv": `date,description,amount
not-a-date,BROKEN,-1.00
`,
})
res, err := Run(root, db, accounts, engine, Options{})
if err != nil {
t.Fatalf("Run returned a fatal error, want a per-file report: %v", err)
}
failures := res.Errs()
if len(failures) != 1 || filepath.Base(failures[0].Path) != "bad.csv" {
t.Fatalf("failures = %+v, want exactly bad.csv", failures)
}
if _, added, _ := res.Total(); added != 1 {
t.Errorf("added %d rows, want 1 from good.csv", added)
}
}
func TestCheckBalances(t *testing.T) {
balance := func(v int64) *int64 { return &v }
good := []parser.RawTxn{
{Date: "2026-01-01", Description: "A", AmountMinor: -1000, BalanceMinor: balance(9000)},
{Date: "2026-01-02", Description: "B", AmountMinor: -500, BalanceMinor: balance(8500)},
{Date: "2026-01-03", Description: "C", AmountMinor: 2000, BalanceMinor: balance(10500)},
}
if w := checkBalances(good, 2); len(w) != 0 {
t.Errorf("a consistent chain produced warnings: %v", w)
}
// A missed row shows up as a break at the row after it.
broken := []parser.RawTxn{
{Date: "2026-01-01", Description: "A", AmountMinor: -1000, BalanceMinor: balance(9000)},
{Date: "2026-01-02", Description: "B", AmountMinor: -500, BalanceMinor: balance(7000)},
}
w := checkBalances(broken, 2)
if len(w) != 1 {
t.Fatalf("got %d warnings, want 1: %v", len(w), w)
}
if !strings.Contains(w[0], "70.00") || !strings.Contains(w[0], "85.00") {
t.Errorf("warning = %q, want both the reported and the expected balance", w[0])
}
// Statements that report no balances are not checked.
none := []parser.RawTxn{
{Date: "2026-01-01", Description: "A", AmountMinor: -1000},
{Date: "2026-01-02", Description: "B", AmountMinor: -500},
}
if w := checkBalances(none, 2); len(w) != 0 {
t.Errorf("statements without balances produced warnings: %v", w)
}
}
// Only files matching include globs are treated as statements.
func TestIncludeGlobs(t *testing.T) {
root, db, accounts, engine := newRoot(t, map[string]string{
"2026-01.csv": `date,description,amount
2026-01-06,LIDL SOFIA,-45.20
`,
"notes.txt": "not a statement",
})
accounts[0].Include = []string{"*.csv"}
res := mustRun(t, root, db, accounts, engine, Options{})
if len(res.Files) != 1 {
t.Fatalf("processed %d files, want 1: %+v", len(res.Files), res.Files)
}
if _, added, _ := res.Total(); added != 1 {
t.Errorf("added %d rows, want 1", added)
}
}