Files
money/internal/store/store.go
T
nikolaandClaude Opus 5.5 a5f7541980 Name NLB uploads, delete statements, and stop importing on upload
Four changes to statement handling in the web app, made together and
touching the same upload and statements-list code.

Name NLB uploads by statement date. parser.Namer is an optional
interface, like Warner, through which a parser names its statements;
nlb reads the "Datum izpiska" from the izpisek header and names it
izpisek_YYYY_MM_DD, lowercase, extension included -- ported from the
rename_izpiski.py it replaces. Uploads are staged as dotfiles, invisible
to import, so the parser can read them; two downloads of one statement
then meet under one name and the second is recognised as already there,
while a different statement of the same date is numbered _2 as the
script did. Only uploads are named: source_files records statements by
path, so renaming a file already in a folder would orphan its rows.

Delete a statement from the statements list. The file is removed from
disk for good -- the page says so before it asks -- and
store.ForgetSourceFile drops its transactions and their transfer rows.
A row two overlapping statements share is stored once, under the file
imported first, so it goes too; the account's other statements forget
their checksums and show as changed until the next Import re-reads them
and restores it. A file already gone from disk can be forgotten.

Upload and delete no longer import. Importing stays the user's call,
made with the Import button, so a batch can be put together and looked
over first. Delete still re-pairs transfers, which reads no statement.

Show rows and new rows per statement. The list read "0" for a file
whose rows an earlier, overlapping statement already held, which looked
like a file that failed to parse. source_files now records how many
transactions each statement holds, and the list reads "3 rows · 0 new".

This adds a column the code reads, so an index built by an earlier
version fails with "no such column: s.rows": delete index.db and import
again.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-02 21:48:29 +02:00

507 lines
16 KiB
Go

// Package store is the SQLite index over the statements. It is entirely
// rebuildable: delete index.db and re-import to get it back. Nothing lives
// only here.
package store
import (
"database/sql"
"fmt"
"os"
"path/filepath"
"strings"
_ "modernc.org/sqlite"
"git.petrovv.com/nikola/money/internal/model"
)
// DB wraps the SQLite handle.
type DB struct {
sql *sql.DB
}
const schema = `
PRAGMA foreign_keys = ON;
CREATE TABLE IF NOT EXISTS accounts (
id INTEGER PRIMARY KEY AUTOINCREMENT,
slug TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
currency TEXT NOT NULL,
minor_digits INTEGER NOT NULL DEFAULT 2
);
CREATE TABLE IF NOT EXISTS source_files (
id INTEGER PRIMARY KEY AUTOINCREMENT,
account_id INTEGER NOT NULL REFERENCES accounts(id),
path TEXT NOT NULL,
sha256 TEXT NOT NULL,
imported_at TEXT NOT NULL,
-- How many transactions the statement holds, as parsed. Not how many it
-- brought in: rows an overlapping statement already had are deduplicated.
rows INTEGER NOT NULL DEFAULT 0,
UNIQUE(account_id, path)
);
CREATE TABLE IF NOT EXISTS transactions (
id INTEGER PRIMARY KEY AUTOINCREMENT,
account_id INTEGER NOT NULL REFERENCES accounts(id),
source_file_id INTEGER NOT NULL REFERENCES source_files(id),
fingerprint TEXT NOT NULL,
date TEXT NOT NULL,
description TEXT NOT NULL,
amount_minor INTEGER NOT NULL,
type TEXT NOT NULL DEFAULT '',
balance_minor INTEGER,
rule_tag TEXT,
UNIQUE(account_id, fingerprint)
);
CREATE INDEX IF NOT EXISTS idx_txn_date ON transactions(date);
CREATE INDEX IF NOT EXISTS idx_txn_account ON transactions(account_id);
-- One row per matched movement between the user's own accounts, derived from
-- the [[transfer]] blocks in rules.toml. A leg belongs to at most one transfer,
-- which UNIQUE enforces rather than trusting the pairing to be well behaved.
CREATE TABLE IF NOT EXISTS transfers (
id INTEGER PRIMARY KEY AUTOINCREMENT,
def_index INTEGER NOT NULL,
out_txn_id INTEGER NOT NULL UNIQUE REFERENCES transactions(id),
in_txn_id INTEGER NOT NULL UNIQUE REFERENCES transactions(id)
);
`
// Open opens (creating if needed) the index at path.
//
// There is no schema migration: the index caches what the statements and
// rules.toml already say, so a build that changes the schema is answered by
// deleting index.db and importing again, not by patching the old one in place.
// Until it is deleted, queries against it fail on the columns it lacks.
func Open(path string) (*DB, error) {
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
return nil, fmt.Errorf("create index dir: %w", err)
}
sqlDB, err := sql.Open("sqlite", path)
if err != nil {
return nil, fmt.Errorf("open index %s: %w", path, err)
}
if _, err := sqlDB.Exec(schema); err != nil {
sqlDB.Close()
return nil, fmt.Errorf("apply schema: %w", err)
}
return &DB{sql: sqlDB}, nil
}
// Close releases the underlying handle.
func (d *DB) Close() error { return d.sql.Close() }
// UpsertAccount inserts or updates an account by slug and returns its id.
func (d *DB) UpsertAccount(a model.Account) (int64, error) {
_, err := d.sql.Exec(`
INSERT INTO accounts (slug, name, currency, minor_digits)
VALUES (?, ?, ?, ?)
ON CONFLICT(slug) DO UPDATE SET
name = excluded.name,
currency = excluded.currency,
minor_digits = excluded.minor_digits`,
a.Slug, a.Name, a.Currency, a.MinorDigits)
if err != nil {
return 0, fmt.Errorf("upsert account %s: %w", a.Slug, err)
}
var id int64
if err := d.sql.QueryRow(`SELECT id FROM accounts WHERE slug = ?`, a.Slug).Scan(&id); err != nil {
return 0, fmt.Errorf("read account id %s: %w", a.Slug, err)
}
return id, nil
}
// Accounts lists every known account, ordered by slug.
func (d *DB) Accounts() ([]model.Account, error) {
rows, err := d.sql.Query(`SELECT id, slug, name, currency, minor_digits FROM accounts ORDER BY slug`)
if err != nil {
return nil, fmt.Errorf("list accounts: %w", err)
}
defer rows.Close()
var out []model.Account
for rows.Next() {
var a model.Account
if err := rows.Scan(&a.ID, &a.Slug, &a.Name, &a.Currency, &a.MinorDigits); err != nil {
return nil, err
}
out = append(out, a)
}
return out, rows.Err()
}
// SourceFile records that a statement file was imported, holding rows
// transactions, and returns its id.
func (d *DB) SourceFile(accountID int64, path, sha, importedAt string, rows int) (int64, error) {
_, err := d.sql.Exec(`
INSERT INTO source_files (account_id, path, sha256, imported_at, rows)
VALUES (?, ?, ?, ?, ?)
ON CONFLICT(account_id, path) DO UPDATE SET
sha256 = excluded.sha256,
imported_at = excluded.imported_at,
rows = excluded.rows`,
accountID, path, sha, importedAt, rows)
if err != nil {
return 0, fmt.Errorf("record source file %s: %w", path, err)
}
var id int64
if err := d.sql.QueryRow(
`SELECT id FROM source_files WHERE account_id = ? AND path = ?`, accountID, path).Scan(&id); err != nil {
return 0, fmt.Errorf("read source file id %s: %w", path, err)
}
return id, nil
}
// SourceFileSHA returns the recorded checksum for a statement file, and whether
// it has been imported before.
func (d *DB) SourceFileSHA(accountID int64, path string) (string, bool, error) {
var sha string
err := d.sql.QueryRow(
`SELECT sha256 FROM source_files WHERE account_id = ? AND path = ?`, accountID, path).Scan(&sha)
if err == sql.ErrNoRows {
return "", false, nil
}
if err != nil {
return "", false, err
}
return sha, true, nil
}
// InsertTransaction adds a transaction unless its fingerprint already exists
// for that account. It reports whether a new row was created.
//
// Existing rows are deliberately left untouched, so re-importing a statement
// that overlaps one already imported adds nothing rather than duplicating it.
func (d *DB) InsertTransaction(t model.Transaction) (bool, error) {
var balance any
if t.BalanceMinor != nil {
balance = *t.BalanceMinor
}
res, err := d.sql.Exec(`
INSERT INTO transactions
(account_id, source_file_id, fingerprint, date, description, amount_minor,
type, balance_minor, rule_tag)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, NULLIF(?, ''))
ON CONFLICT(account_id, fingerprint) DO NOTHING`,
t.AccountID, t.SourceFileID, t.Fingerprint, t.Date, t.Description,
t.AmountMinor, t.Type, balance, t.RuleTag)
if err != nil {
return false, fmt.Errorf("insert transaction: %w", err)
}
n, err := res.RowsAffected()
if err != nil {
return false, err
}
return n > 0, nil
}
// Filter narrows a transaction query.
type Filter struct {
AccountSlug string
// Untagged selects the rows still waiting for a verdict: no tag, and not a
// leg of a matched transfer. A paired leg has been accounted for by the
// transfer that claimed it, so listing it as untagged would ask the user to
// write a rule for something that is already spoken for and that the report
// leaves out anyway.
Untagged bool
Month string // YYYY-MM
// From and To bound the dates inclusively, as YYYY-MM-DD, either side empty
// for unbounded. Dates are stored ISO-8601, so a string comparison is a date
// comparison; nothing has to be parsed to filter on a range.
From string
To string
Search string // case-insensitive substring of the description
Limit int
}
// Transactions returns rows matching f, newest first.
func (d *DB) Transactions(f Filter) ([]model.Transaction, error) {
q := `
SELECT t.id, t.account_id, a.slug, a.currency, a.minor_digits,
t.fingerprint, t.date, t.description, t.amount_minor,
COALESCE(s.path, ''), t.type, t.balance_minor,
COALESCE(t.rule_tag, ''), x.id
FROM transactions t
JOIN accounts a ON a.id = t.account_id
LEFT JOIN source_files s ON s.id = t.source_file_id
LEFT JOIN transfers x ON x.out_txn_id = t.id OR x.in_txn_id = t.id
WHERE 1 = 1`
var args []any
if f.AccountSlug != "" {
q += ` AND a.slug = ?`
args = append(args, f.AccountSlug)
}
if f.Untagged {
q += ` AND NULLIF(t.rule_tag, '') IS NULL AND x.id IS NULL`
}
if f.Month != "" {
q += ` AND substr(t.date, 1, 7) = ?`
args = append(args, f.Month)
}
if f.From != "" {
q += ` AND t.date >= ?`
args = append(args, f.From)
}
if f.To != "" {
q += ` AND t.date <= ?`
args = append(args, f.To)
}
q += ` ORDER BY t.date DESC, t.id DESC`
// Search and Limit are applied in Go: SQLite's upper()/LIKE fold ASCII
// only, which would silently fail on Cyrillic statement descriptions.
rows, err := d.sql.Query(q, args...)
if err != nil {
return nil, fmt.Errorf("query transactions: %w", err)
}
defer rows.Close()
needle := model.NormalizeDescription(f.Search)
var out []model.Transaction
for rows.Next() {
var (
t model.Transaction
balance sql.NullInt64
transfer sql.NullInt64
)
if err := rows.Scan(&t.ID, &t.AccountID, &t.AccountSlug, &t.Currency, &t.MinorDigits,
&t.Fingerprint, &t.Date, &t.Description, &t.AmountMinor, &t.SourcePath,
&t.Type, &balance, &t.RuleTag, &transfer); err != nil {
return nil, err
}
if balance.Valid {
v := balance.Int64
t.BalanceMinor = &v
}
if transfer.Valid {
v := transfer.Int64
t.TransferID = &v
}
if needle != "" && !strings.Contains(model.NormalizeDescription(t.Description), needle) {
continue
}
out = append(out, t)
if f.Limit > 0 && len(out) >= f.Limit {
break
}
}
return out, rows.Err()
}
// RuleAssignment is one row's recomputed rule verdict.
type RuleAssignment struct {
ID int64
Tag string
}
// ApplyRuleResults rewrites rule_tag for every listed row in a single
// transaction. The manual column is never touched.
func (d *DB) ApplyRuleResults(rs []RuleAssignment) error {
tx, err := d.sql.Begin()
if err != nil {
return err
}
defer tx.Rollback()
stmt, err := tx.Prepare(
`UPDATE transactions SET rule_tag = NULLIF(?, '') WHERE id = ?`)
if err != nil {
return err
}
defer stmt.Close()
for _, r := range rs {
if _, err := stmt.Exec(r.Tag, r.ID); err != nil {
return fmt.Errorf("apply rules to txn %d: %w", r.ID, err)
}
}
return tx.Commit()
}
// TransferLink is one matched pair, as decided by the transfer definitions.
type TransferLink struct {
DefIndex int
OutID int64
InID int64
}
// ReplaceTransfers rewrites the whole pairing in one transaction. Like the
// tags, it is derived from a file the user edits, so it is replaced wholesale
// rather than patched: a definition removed from rules.toml must take its pairs
// with it.
func (d *DB) ReplaceTransfers(links []TransferLink) error {
tx, err := d.sql.Begin()
if err != nil {
return err
}
defer tx.Rollback()
if _, err := tx.Exec(`DELETE FROM transfers`); err != nil {
return fmt.Errorf("clear transfers: %w", err)
}
stmt, err := tx.Prepare(
`INSERT INTO transfers (def_index, out_txn_id, in_txn_id) VALUES (?, ?, ?)`)
if err != nil {
return err
}
defer stmt.Close()
for _, l := range links {
if _, err := stmt.Exec(l.DefIndex, l.OutID, l.InID); err != nil {
return fmt.Errorf("link transfer %d→%d: %w", l.OutID, l.InID, err)
}
}
return tx.Commit()
}
// Balance sums every transaction in an account.
func (d *DB) Balance(accountID int64) (int64, error) {
var v sql.NullInt64
err := d.sql.QueryRow(
`SELECT SUM(amount_minor) FROM transactions WHERE account_id = ?`, accountID).Scan(&v)
if err != nil {
return 0, err
}
return v.Int64, nil
}
// Count returns the number of transactions in an account.
func (d *DB) Count(accountID int64) (int, error) {
var n int
err := d.sql.QueryRow(
`SELECT COUNT(*) FROM transactions WHERE account_id = ?`, accountID).Scan(&n)
return n, err
}
// Months lists the distinct YYYY-MM the index holds, most recent first. It is
// asked of the whole index rather than of a filtered view on purpose: it is the
// time axis the report screen steps along, and an axis that grew and shrank as
// the account or search filter changed would move under the cursor.
func (d *DB) Months() ([]string, error) {
rows, err := d.sql.Query(
`SELECT DISTINCT substr(date, 1, 7) FROM transactions ORDER BY 1 DESC`)
if err != nil {
return nil, fmt.Errorf("list months: %w", err)
}
defer rows.Close()
var out []string
for rows.Next() {
var s string
if err := rows.Scan(&s); err != nil {
return nil, err
}
out = append(out, s)
}
return out, rows.Err()
}
// SourceFileInfo is what the index recorded about one imported statement.
type SourceFileInfo struct {
AccountSlug string
Path string // relative to the data root, as the importer records it
SHA256 string
ImportedAt string
// Rows is how many transactions the statement holds, as parsed.
Rows int
// Added counts the transactions this file introduced. A row seen again in
// an overlapping statement is deduplicated and stays with the file that
// brought it first, so this is not how many rows the file holds.
Added int
}
// SourceFiles lists every statement the index has imported.
func (d *DB) SourceFiles() ([]SourceFileInfo, error) {
rows, err := d.sql.Query(`
SELECT a.slug, s.path, s.sha256, s.imported_at, s.rows, COUNT(t.id)
FROM source_files s
JOIN accounts a ON a.id = s.account_id
LEFT JOIN transactions t ON t.source_file_id = s.id
GROUP BY s.id
ORDER BY a.slug, s.path`)
if err != nil {
return nil, fmt.Errorf("list source files: %w", err)
}
defer rows.Close()
var out []SourceFileInfo
for rows.Next() {
var f SourceFileInfo
if err := rows.Scan(&f.AccountSlug, &f.Path, &f.SHA256, &f.ImportedAt, &f.Rows, &f.Added); err != nil {
return nil, err
}
out = append(out, f)
}
return out, rows.Err()
}
// ForgetSourceFile removes a statement from the index: its transactions, any
// transfer pairing that used them, and the record of it, returning how many
// transactions went. Pairing is derived state that the next Link rewrites, so
// dropping a pair here loses nothing.
//
// A transaction two overlapping statements share is stored once, under the
// file that brought it first, so removing that file can take rows another
// statement still holds. The account's other statements therefore forget
// their checksums, and the next import re-reads them and restores those rows.
func (d *DB) ForgetSourceFile(accountSlug, path string) (int, error) {
tx, err := d.sql.Begin()
if err != nil {
return 0, err
}
defer tx.Rollback()
var id, accountID int64
err = tx.QueryRow(`
SELECT s.id, s.account_id FROM source_files s JOIN accounts a ON a.id = s.account_id
WHERE a.slug = ? AND s.path = ?`, accountSlug, path).Scan(&id, &accountID)
if err == sql.ErrNoRows {
return 0, nil
}
if err != nil {
return 0, fmt.Errorf("find source file %s: %w", path, err)
}
if _, err := tx.Exec(`
DELETE FROM transfers WHERE
out_txn_id IN (SELECT id FROM transactions WHERE source_file_id = ?) OR
in_txn_id IN (SELECT id FROM transactions WHERE source_file_id = ?)`, id, id); err != nil {
return 0, fmt.Errorf("forget transfers from %s: %w", path, err)
}
res, err := tx.Exec(`DELETE FROM transactions WHERE source_file_id = ?`, id)
if err != nil {
return 0, fmt.Errorf("forget transactions from %s: %w", path, err)
}
n, err := res.RowsAffected()
if err != nil {
return 0, err
}
if _, err := tx.Exec(`DELETE FROM source_files WHERE id = ?`, id); err != nil {
return 0, fmt.Errorf("forget %s: %w", path, err)
}
if _, err := tx.Exec(`UPDATE source_files SET sha256 = '' WHERE account_id = ?`, accountID); err != nil {
return 0, fmt.Errorf("mark %s's statements for re-reading: %w", accountSlug, err)
}
return int(n), tx.Commit()
}
// Tags lists every tag in use, for completion in the rule builder.
func (d *DB) Tags() ([]string, error) {
rows, err := d.sql.Query(`
SELECT DISTINCT rule_tag FROM transactions
WHERE rule_tag IS NOT NULL AND rule_tag != '' ORDER BY rule_tag`)
if err != nil {
return nil, err
}
defer rows.Close()
var out []string
for rows.Next() {
var s string
if err := rows.Scan(&s); err != nil {
return nil, err
}
out = append(out, s)
}
return out, rows.Err()
}