counterparty was a structured field only nlb could fill honestly. revolut and traderepublic invented one by running an IBAN-shaped regex over the description they had just built, and the two spellings disagreed -- SI56 1234 5678 9012 345 against SI56123456789012345 -- so a literal rule pattern that worked on one account silently matched nothing on another. It is gone from the model, the index, the rule keys, ls --wide and the rules screen. nlb now appends its IBAN column to the end of the description, where the other two already keep theirs, so match = "*SI56*" works everywhere. That changes those descriptions and with them their fingerprints, so a statement overlapping an already-imported period will re-add rather than dedupe those rows until the index is rebuilt. An index built by an older binary drops the column when it is opened. The index itself moves from .money/index.db up to index.db beside rules.toml. Nothing looks in the old location, so an existing one has to be moved by hand -- otherwise the tool quietly starts a fresh index and the manual tags in the old file, the only thing statements cannot reproduce, stay behind in it. The csv and cmd parsers are gone along with the [csv] and [cmd] config they carried. cmd shelled out to the Python extractors, which were ported to Go and deleted, so it bridged to nothing; csv was a generic column-mapped fallback that no account used, and between them they were the largest configuration surface in the tool. A bank is now described in Go, where it can be tested. The importer tests register their own three-column parser rather than borrow a bank's, so they stay about the directory walk, dedupe and per-file error reporting. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
283 lines
6.9 KiB
Go
283 lines
6.9 KiB
Go
// Package importer walks the data root, extracts transactions from every
|
|
// statement file, and inserts the ones not already in the index.
|
|
package importer
|
|
|
|
import (
|
|
"crypto/sha256"
|
|
"encoding/hex"
|
|
"fmt"
|
|
"io"
|
|
"os"
|
|
"path/filepath"
|
|
"sort"
|
|
"strings"
|
|
"time"
|
|
|
|
"git.petrovv.com/nikola/money/internal/config"
|
|
"git.petrovv.com/nikola/money/internal/glob"
|
|
"git.petrovv.com/nikola/money/internal/model"
|
|
"git.petrovv.com/nikola/money/internal/parser"
|
|
"git.petrovv.com/nikola/money/internal/rules"
|
|
"git.petrovv.com/nikola/money/internal/store"
|
|
)
|
|
|
|
// FileResult is what happened to one statement file.
|
|
type FileResult struct {
|
|
Account string
|
|
Path string // relative to the data root
|
|
Parsed int
|
|
New int
|
|
Skipped int // already present, i.e. deduplicated
|
|
Err error
|
|
// Warnings are non-fatal notes: rows the parser deliberately dropped, and
|
|
// breaks in the statement's balance chain.
|
|
Warnings []string
|
|
}
|
|
|
|
// Result summarises a whole import run.
|
|
type Result struct {
|
|
Files []FileResult
|
|
Retagged int
|
|
}
|
|
|
|
// Total counts new rows across every file.
|
|
func (r Result) Total() (parsed, added, skipped int) {
|
|
for _, f := range r.Files {
|
|
parsed += f.Parsed
|
|
added += f.New
|
|
skipped += f.Skipped
|
|
}
|
|
return
|
|
}
|
|
|
|
// Errs collects per-file failures. One unreadable statement must not abort the
|
|
// whole run, so errors are reported rather than returned.
|
|
func (r Result) Errs() []FileResult {
|
|
var out []FileResult
|
|
for _, f := range r.Files {
|
|
if f.Err != nil {
|
|
out = append(out, f)
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
// Options tunes an import run.
|
|
type Options struct {
|
|
// Force reimports statement files whose contents are unchanged. Without
|
|
// it, files with a matching checksum are skipped entirely.
|
|
Force bool
|
|
}
|
|
|
|
// Run imports every account under root into db and then applies the rules.
|
|
func Run(root string, db *store.DB, accounts []*config.Account, engine *rules.Engine, opts Options) (Result, error) {
|
|
var res Result
|
|
for _, acc := range accounts {
|
|
accountID, err := db.UpsertAccount(model.Account{
|
|
Slug: acc.Slug,
|
|
Name: acc.Name,
|
|
Currency: acc.Currency,
|
|
MinorDigits: acc.Digits(),
|
|
})
|
|
if err != nil {
|
|
return res, err
|
|
}
|
|
|
|
p, err := parser.For(acc)
|
|
if err != nil {
|
|
// A misconfigured account is worth reporting, but the other
|
|
// accounts should still import.
|
|
res.Files = append(res.Files, FileResult{Account: acc.Slug, Path: acc.Slug, Err: err})
|
|
continue
|
|
}
|
|
|
|
files, err := statementFiles(acc)
|
|
if err != nil {
|
|
return res, err
|
|
}
|
|
for _, path := range files {
|
|
fr := importFile(root, db, acc, accountID, p, engine, path, opts)
|
|
res.Files = append(res.Files, fr)
|
|
}
|
|
}
|
|
|
|
n, err := engine.Retag(db)
|
|
if err != nil {
|
|
return res, err
|
|
}
|
|
res.Retagged = n
|
|
return res, nil
|
|
}
|
|
|
|
func importFile(root string, db *store.DB, acc *config.Account, accountID int64,
|
|
p parser.Parser, engine *rules.Engine, path string, opts Options) FileResult {
|
|
|
|
rel, err := filepath.Rel(root, path)
|
|
if err != nil {
|
|
rel = path
|
|
}
|
|
fr := FileResult{Account: acc.Slug, Path: rel}
|
|
|
|
sum, err := checksum(path)
|
|
if err != nil {
|
|
fr.Err = err
|
|
return fr
|
|
}
|
|
if !opts.Force {
|
|
prev, seen, err := db.SourceFileSHA(accountID, rel)
|
|
if err != nil {
|
|
fr.Err = err
|
|
return fr
|
|
}
|
|
if seen && prev == sum {
|
|
return fr // unchanged since last import; nothing to do
|
|
}
|
|
}
|
|
|
|
txns, err := p.Parse(path, acc)
|
|
if err != nil {
|
|
fr.Err = err
|
|
return fr
|
|
}
|
|
fr.Parsed = len(txns)
|
|
|
|
if w, ok := p.(parser.Warner); ok {
|
|
fr.Warnings = append(fr.Warnings, w.Warnings()...)
|
|
}
|
|
fr.Warnings = append(fr.Warnings, checkBalances(txns, acc.Digits())...)
|
|
|
|
sourceID, err := db.SourceFile(accountID, rel, sum, time.Now().UTC().Format(time.RFC3339))
|
|
if err != nil {
|
|
fr.Err = err
|
|
return fr
|
|
}
|
|
|
|
// Identical lines within one statement (two coffees on the same day) are
|
|
// distinguished by their ordinal, so both survive; the same line seen
|
|
// again in an overlapping statement gets the same fingerprint and is
|
|
// deduplicated.
|
|
seen := map[string]int{}
|
|
for _, t := range txns {
|
|
key := fingerprintKey(t)
|
|
ordinal := seen[key]
|
|
seen[key]++
|
|
|
|
txn := model.Transaction{
|
|
AccountID: accountID,
|
|
SourceFileID: sourceID,
|
|
Fingerprint: fingerprint(key, ordinal),
|
|
Date: t.Date,
|
|
Description: t.Description,
|
|
AmountMinor: t.AmountMinor,
|
|
Type: t.Type,
|
|
BalanceMinor: t.BalanceMinor,
|
|
}
|
|
txn.RuleTag, txn.RuleTransfer = engine.ApplyTxn(acc.Slug, txn)
|
|
added, err := db.InsertTransaction(txn)
|
|
if err != nil {
|
|
fr.Err = err
|
|
return fr
|
|
}
|
|
if added {
|
|
fr.New++
|
|
} else {
|
|
fr.Skipped++
|
|
}
|
|
}
|
|
return fr
|
|
}
|
|
|
|
// checkBalances verifies that each reported balance is the previous one plus
|
|
// the transaction amount, which is what the original extraction scripts did.
|
|
// A break means a row was missed or misparsed, so it is worth saying out loud
|
|
// even though the import still proceeds.
|
|
func checkBalances(txns []parser.RawTxn, digits int) []string {
|
|
var warnings []string
|
|
var prev *parser.RawTxn
|
|
|
|
for i := range txns {
|
|
t := &txns[i]
|
|
if t.BalanceMinor == nil {
|
|
continue // this statement does not report running balances
|
|
}
|
|
if prev != nil {
|
|
expected := *prev.BalanceMinor + t.AmountMinor
|
|
if expected != *t.BalanceMinor {
|
|
warnings = append(warnings, fmt.Sprintf(
|
|
"balance chain breaks at %s %q: statement says %s, previous balance plus amount is %s",
|
|
t.Date, truncate(t.Description, 40),
|
|
model.FormatMinor(*t.BalanceMinor, digits),
|
|
model.FormatMinor(expected, digits)))
|
|
}
|
|
}
|
|
prev = t
|
|
}
|
|
return warnings
|
|
}
|
|
|
|
func truncate(s string, n int) string {
|
|
r := []rune(s)
|
|
if len(r) <= n {
|
|
return s
|
|
}
|
|
return string(r[:n]) + "…"
|
|
}
|
|
|
|
func fingerprintKey(t parser.RawTxn) string {
|
|
return strings.Join([]string{
|
|
t.Date,
|
|
fmt.Sprintf("%d", t.AmountMinor),
|
|
model.NormalizeDescription(t.Description),
|
|
}, "\x00")
|
|
}
|
|
|
|
func fingerprint(key string, ordinal int) string {
|
|
sum := sha256.Sum256([]byte(fmt.Sprintf("%s\x00%d", key, ordinal)))
|
|
return hex.EncodeToString(sum[:])
|
|
}
|
|
|
|
// statementFiles lists the files in an account folder that should be parsed:
|
|
// every regular file except account.toml and dotfiles, narrowed by the
|
|
// account's optional include globs.
|
|
func statementFiles(acc *config.Account) ([]string, error) {
|
|
entries, err := os.ReadDir(acc.Dir)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("read account dir %s: %w", acc.Dir, err)
|
|
}
|
|
var out []string
|
|
for _, e := range entries {
|
|
name := e.Name()
|
|
if e.IsDir() || strings.HasPrefix(name, ".") || name == config.AccountFile {
|
|
continue
|
|
}
|
|
if len(acc.Include) > 0 && !matchAny(acc.Include, name) {
|
|
continue
|
|
}
|
|
out = append(out, filepath.Join(acc.Dir, name))
|
|
}
|
|
sort.Strings(out)
|
|
return out, nil
|
|
}
|
|
|
|
func matchAny(patterns []string, name string) bool {
|
|
for _, p := range patterns {
|
|
if glob.Match(p, name) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
func checksum(path string) (string, error) {
|
|
f, err := os.Open(path)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
defer f.Close()
|
|
h := sha256.New()
|
|
if _, err := io.Copy(h, f); err != nil {
|
|
return "", err
|
|
}
|
|
return hex.EncodeToString(h.Sum(nil)), nil
|
|
}
|