Files
money/internal/importer/importer.go
T
nikolaandClaude Opus 5.5 a5f7541980 Name NLB uploads, delete statements, and stop importing on upload
Four changes to statement handling in the web app, made together and
touching the same upload and statements-list code.

Name NLB uploads by statement date. parser.Namer is an optional
interface, like Warner, through which a parser names its statements;
nlb reads the "Datum izpiska" from the izpisek header and names it
izpisek_YYYY_MM_DD, lowercase, extension included -- ported from the
rename_izpiski.py it replaces. Uploads are staged as dotfiles, invisible
to import, so the parser can read them; two downloads of one statement
then meet under one name and the second is recognised as already there,
while a different statement of the same date is numbered _2 as the
script did. Only uploads are named: source_files records statements by
path, so renaming a file already in a folder would orphan its rows.

Delete a statement from the statements list. The file is removed from
disk for good -- the page says so before it asks -- and
store.ForgetSourceFile drops its transactions and their transfer rows.
A row two overlapping statements share is stored once, under the file
imported first, so it goes too; the account's other statements forget
their checksums and show as changed until the next Import re-reads them
and restores it. A file already gone from disk can be forgotten.

Upload and delete no longer import. Importing stays the user's call,
made with the Import button, so a batch can be put together and looked
over first. Delete still re-pairs transfers, which reads no statement.

Show rows and new rows per statement. The list read "0" for a file
whose rows an earlier, overlapping statement already held, which looked
like a file that failed to parse. source_files now records how many
transactions each statement holds, and the list reads "3 rows · 0 new".

This adds a column the code reads, so an index built by an earlier
version fails with "no such column: s.rows": delete index.db and import
again.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-02 21:48:29 +02:00

302 lines
7.8 KiB
Go

// Package importer walks the data root, extracts transactions from every
// statement file, and inserts the ones not already in the index.
package importer
import (
"crypto/sha256"
"encoding/hex"
"fmt"
"io"
"os"
"path/filepath"
"sort"
"strings"
"time"
"git.petrovv.com/nikola/money/internal/config"
"git.petrovv.com/nikola/money/internal/glob"
"git.petrovv.com/nikola/money/internal/model"
"git.petrovv.com/nikola/money/internal/parser"
"git.petrovv.com/nikola/money/internal/rules"
"git.petrovv.com/nikola/money/internal/store"
"git.petrovv.com/nikola/money/internal/transfers"
)
// FileResult is what happened to one statement file.
type FileResult struct {
Account string
Path string // relative to the data root
Parsed int
New int
Skipped int // already present, i.e. deduplicated
Err error
// Warnings are non-fatal notes: rows the parser deliberately dropped, and
// breaks in the statement's balance chain.
Warnings []string
}
// Result summarises a whole import run.
type Result struct {
Files []FileResult
Retagged int
// Paired and Unpaired are the state of the transfer pairing after the
// import: how many movements were matched, and how many legs a definition
// caught without finding the other side.
Paired int
Unpaired int
}
// Total counts new rows across every file.
func (r Result) Total() (parsed, added, skipped int) {
for _, f := range r.Files {
parsed += f.Parsed
added += f.New
skipped += f.Skipped
}
return
}
// Errs collects per-file failures. One unreadable statement must not abort the
// whole run, so errors are reported rather than returned.
func (r Result) Errs() []FileResult {
var out []FileResult
for _, f := range r.Files {
if f.Err != nil {
out = append(out, f)
}
}
return out
}
// Options tunes an import run.
type Options struct {
// Force reimports statement files whose contents are unchanged. Without
// it, files with a matching checksum are skipped entirely.
Force bool
}
// Run imports every account under root into db and then applies the rules and
// the transfer definitions. Both are derived from rules.toml and both are
// recomputed over the whole index, since a statement imported now can complete
// a transfer whose other leg arrived months ago.
func Run(root string, db *store.DB, accounts []*config.Account, engine *rules.Engine,
links *transfers.Engine, opts Options) (Result, error) {
var res Result
for _, acc := range accounts {
accountID, err := db.UpsertAccount(model.Account{
Slug: acc.Slug,
Name: acc.Name,
Currency: acc.Currency,
MinorDigits: acc.Digits(),
})
if err != nil {
return res, err
}
p, err := parser.For(acc)
if err != nil {
// A misconfigured account is worth reporting, but the other
// accounts should still import.
res.Files = append(res.Files, FileResult{Account: acc.Slug, Path: acc.Slug, Err: err})
continue
}
files, err := StatementFiles(acc)
if err != nil {
return res, err
}
for _, path := range files {
fr := importFile(root, db, acc, accountID, p, engine, path, opts)
res.Files = append(res.Files, fr)
}
}
n, err := engine.Retag(db)
if err != nil {
return res, err
}
res.Retagged = n
paired, unpaired, err := links.Link(db)
if err != nil {
return res, err
}
res.Paired, res.Unpaired = paired, unpaired
return res, nil
}
func importFile(root string, db *store.DB, acc *config.Account, accountID int64,
p parser.Parser, engine *rules.Engine, path string, opts Options) FileResult {
rel, err := filepath.Rel(root, path)
if err != nil {
rel = path
}
fr := FileResult{Account: acc.Slug, Path: rel}
sum, err := Checksum(path)
if err != nil {
fr.Err = err
return fr
}
if !opts.Force {
prev, seen, err := db.SourceFileSHA(accountID, rel)
if err != nil {
fr.Err = err
return fr
}
if seen && prev == sum {
return fr // unchanged since last import; nothing to do
}
}
txns, err := p.Parse(path, acc)
if err != nil {
fr.Err = err
return fr
}
fr.Parsed = len(txns)
if w, ok := p.(parser.Warner); ok {
fr.Warnings = append(fr.Warnings, w.Warnings()...)
}
fr.Warnings = append(fr.Warnings, checkBalances(txns, acc.Digits())...)
sourceID, err := db.SourceFile(accountID, rel, sum, time.Now().UTC().Format(time.RFC3339), len(txns))
if err != nil {
fr.Err = err
return fr
}
// Identical lines within one statement (two coffees on the same day) are
// distinguished by their ordinal, so both survive; the same line seen
// again in an overlapping statement gets the same fingerprint and is
// deduplicated.
seen := map[string]int{}
for _, t := range txns {
key := fingerprintKey(t)
ordinal := seen[key]
seen[key]++
txn := model.Transaction{
AccountID: accountID,
SourceFileID: sourceID,
Fingerprint: fingerprint(key, ordinal),
Date: t.Date,
Description: t.Description,
AmountMinor: t.AmountMinor,
Type: t.Type,
BalanceMinor: t.BalanceMinor,
}
txn.RuleTag = engine.ApplyTxn(acc.Slug, txn)
added, err := db.InsertTransaction(txn)
if err != nil {
fr.Err = err
return fr
}
if added {
fr.New++
} else {
fr.Skipped++
}
}
return fr
}
// checkBalances verifies that each reported balance is the previous one plus
// the transaction amount, which is what the original extraction scripts did.
// A break means a row was missed or misparsed, so it is worth saying out loud
// even though the import still proceeds.
func checkBalances(txns []parser.RawTxn, digits int) []string {
var warnings []string
var prev *parser.RawTxn
for i := range txns {
t := &txns[i]
if t.BalanceMinor == nil {
continue // this statement does not report running balances
}
if prev != nil {
expected := *prev.BalanceMinor + t.AmountMinor
if expected != *t.BalanceMinor {
warnings = append(warnings, fmt.Sprintf(
"balance chain breaks at %s %q: statement says %s, previous balance plus amount is %s",
t.Date, truncate(t.Description, 40),
model.FormatMinor(*t.BalanceMinor, digits),
model.FormatMinor(expected, digits)))
}
}
prev = t
}
return warnings
}
func truncate(s string, n int) string {
r := []rune(s)
if len(r) <= n {
return s
}
return string(r[:n]) + "…"
}
func fingerprintKey(t parser.RawTxn) string {
return strings.Join([]string{
t.Date,
fmt.Sprintf("%d", t.AmountMinor),
model.NormalizeDescription(t.Description),
}, "\x00")
}
func fingerprint(key string, ordinal int) string {
sum := sha256.Sum256([]byte(fmt.Sprintf("%s\x00%d", key, ordinal)))
return hex.EncodeToString(sum[:])
}
// StatementFiles lists the files in an account folder that should be parsed:
// every regular file except account.toml and dotfiles, narrowed by the
// account's optional include globs. It is also what the web app lists and
// serves, so a file is shown exactly when import would read it.
func StatementFiles(acc *config.Account) ([]string, error) {
entries, err := os.ReadDir(acc.Dir)
if err != nil {
return nil, fmt.Errorf("read account dir %s: %w", acc.Dir, err)
}
var out []string
for _, e := range entries {
name := e.Name()
if e.IsDir() || strings.HasPrefix(name, ".") || name == config.AccountFile {
continue
}
if len(acc.Include) > 0 && !matchAny(acc.Include, name) {
continue
}
out = append(out, filepath.Join(acc.Dir, name))
}
sort.Strings(out)
return out, nil
}
func matchAny(patterns []string, name string) bool {
for _, p := range patterns {
if glob.Match(p, name) {
return true
}
}
return false
}
// Checksum is the sha256 a statement is recorded under; an import skips a file
// whose checksum has not changed.
func Checksum(path string) (string, error) {
f, err := os.Open(path)
if err != nil {
return "", err
}
defer f.Close()
h := sha256.New()
if _, err := io.Copy(h, f); err != nil {
return "", err
}
return hex.EncodeToString(h.Sum(nil)), nil
}