Four changes to statement handling in the web app, made together and touching the same upload and statements-list code. Name NLB uploads by statement date. parser.Namer is an optional interface, like Warner, through which a parser names its statements; nlb reads the "Datum izpiska" from the izpisek header and names it izpisek_YYYY_MM_DD, lowercase, extension included -- ported from the rename_izpiski.py it replaces. Uploads are staged as dotfiles, invisible to import, so the parser can read them; two downloads of one statement then meet under one name and the second is recognised as already there, while a different statement of the same date is numbered _2 as the script did. Only uploads are named: source_files records statements by path, so renaming a file already in a folder would orphan its rows. Delete a statement from the statements list. The file is removed from disk for good -- the page says so before it asks -- and store.ForgetSourceFile drops its transactions and their transfer rows. A row two overlapping statements share is stored once, under the file imported first, so it goes too; the account's other statements forget their checksums and show as changed until the next Import re-reads them and restores it. A file already gone from disk can be forgotten. Upload and delete no longer import. Importing stays the user's call, made with the Import button, so a batch can be put together and looked over first. Delete still re-pairs transfers, which reads no statement. Show rows and new rows per statement. The list read "0" for a file whose rows an earlier, overlapping statement already held, which looked like a file that failed to parse. source_files now records how many transactions each statement holds, and the list reads "3 rows · 0 new". This adds a column the code reads, so an index built by an earlier version fails with "no such column: s.rows": delete index.db and import again. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
507 lines
16 KiB
Go
507 lines
16 KiB
Go
// Package store is the SQLite index over the statements. It is entirely
|
|
// rebuildable: delete index.db and re-import to get it back. Nothing lives
|
|
// only here.
|
|
package store
|
|
|
|
import (
|
|
"database/sql"
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
|
|
_ "modernc.org/sqlite"
|
|
|
|
"git.petrovv.com/nikola/money/internal/model"
|
|
)
|
|
|
|
// DB wraps the SQLite handle.
|
|
type DB struct {
|
|
sql *sql.DB
|
|
}
|
|
|
|
const schema = `
|
|
PRAGMA foreign_keys = ON;
|
|
|
|
CREATE TABLE IF NOT EXISTS accounts (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
slug TEXT NOT NULL UNIQUE,
|
|
name TEXT NOT NULL,
|
|
currency TEXT NOT NULL,
|
|
minor_digits INTEGER NOT NULL DEFAULT 2
|
|
);
|
|
|
|
CREATE TABLE IF NOT EXISTS source_files (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
account_id INTEGER NOT NULL REFERENCES accounts(id),
|
|
path TEXT NOT NULL,
|
|
sha256 TEXT NOT NULL,
|
|
imported_at TEXT NOT NULL,
|
|
-- How many transactions the statement holds, as parsed. Not how many it
|
|
-- brought in: rows an overlapping statement already had are deduplicated.
|
|
rows INTEGER NOT NULL DEFAULT 0,
|
|
UNIQUE(account_id, path)
|
|
);
|
|
|
|
CREATE TABLE IF NOT EXISTS transactions (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
account_id INTEGER NOT NULL REFERENCES accounts(id),
|
|
source_file_id INTEGER NOT NULL REFERENCES source_files(id),
|
|
fingerprint TEXT NOT NULL,
|
|
date TEXT NOT NULL,
|
|
description TEXT NOT NULL,
|
|
amount_minor INTEGER NOT NULL,
|
|
type TEXT NOT NULL DEFAULT '',
|
|
balance_minor INTEGER,
|
|
rule_tag TEXT,
|
|
UNIQUE(account_id, fingerprint)
|
|
);
|
|
|
|
CREATE INDEX IF NOT EXISTS idx_txn_date ON transactions(date);
|
|
CREATE INDEX IF NOT EXISTS idx_txn_account ON transactions(account_id);
|
|
|
|
-- One row per matched movement between the user's own accounts, derived from
|
|
-- the [[transfer]] blocks in rules.toml. A leg belongs to at most one transfer,
|
|
-- which UNIQUE enforces rather than trusting the pairing to be well behaved.
|
|
CREATE TABLE IF NOT EXISTS transfers (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
def_index INTEGER NOT NULL,
|
|
out_txn_id INTEGER NOT NULL UNIQUE REFERENCES transactions(id),
|
|
in_txn_id INTEGER NOT NULL UNIQUE REFERENCES transactions(id)
|
|
);
|
|
`
|
|
|
|
// Open opens (creating if needed) the index at path.
|
|
//
|
|
// There is no schema migration: the index caches what the statements and
|
|
// rules.toml already say, so a build that changes the schema is answered by
|
|
// deleting index.db and importing again, not by patching the old one in place.
|
|
// Until it is deleted, queries against it fail on the columns it lacks.
|
|
func Open(path string) (*DB, error) {
|
|
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
|
return nil, fmt.Errorf("create index dir: %w", err)
|
|
}
|
|
sqlDB, err := sql.Open("sqlite", path)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("open index %s: %w", path, err)
|
|
}
|
|
if _, err := sqlDB.Exec(schema); err != nil {
|
|
sqlDB.Close()
|
|
return nil, fmt.Errorf("apply schema: %w", err)
|
|
}
|
|
return &DB{sql: sqlDB}, nil
|
|
}
|
|
|
|
// Close releases the underlying handle.
|
|
func (d *DB) Close() error { return d.sql.Close() }
|
|
|
|
// UpsertAccount inserts or updates an account by slug and returns its id.
|
|
func (d *DB) UpsertAccount(a model.Account) (int64, error) {
|
|
_, err := d.sql.Exec(`
|
|
INSERT INTO accounts (slug, name, currency, minor_digits)
|
|
VALUES (?, ?, ?, ?)
|
|
ON CONFLICT(slug) DO UPDATE SET
|
|
name = excluded.name,
|
|
currency = excluded.currency,
|
|
minor_digits = excluded.minor_digits`,
|
|
a.Slug, a.Name, a.Currency, a.MinorDigits)
|
|
if err != nil {
|
|
return 0, fmt.Errorf("upsert account %s: %w", a.Slug, err)
|
|
}
|
|
var id int64
|
|
if err := d.sql.QueryRow(`SELECT id FROM accounts WHERE slug = ?`, a.Slug).Scan(&id); err != nil {
|
|
return 0, fmt.Errorf("read account id %s: %w", a.Slug, err)
|
|
}
|
|
return id, nil
|
|
}
|
|
|
|
// Accounts lists every known account, ordered by slug.
|
|
func (d *DB) Accounts() ([]model.Account, error) {
|
|
rows, err := d.sql.Query(`SELECT id, slug, name, currency, minor_digits FROM accounts ORDER BY slug`)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("list accounts: %w", err)
|
|
}
|
|
defer rows.Close()
|
|
var out []model.Account
|
|
for rows.Next() {
|
|
var a model.Account
|
|
if err := rows.Scan(&a.ID, &a.Slug, &a.Name, &a.Currency, &a.MinorDigits); err != nil {
|
|
return nil, err
|
|
}
|
|
out = append(out, a)
|
|
}
|
|
return out, rows.Err()
|
|
}
|
|
|
|
// SourceFile records that a statement file was imported, holding rows
|
|
// transactions, and returns its id.
|
|
func (d *DB) SourceFile(accountID int64, path, sha, importedAt string, rows int) (int64, error) {
|
|
_, err := d.sql.Exec(`
|
|
INSERT INTO source_files (account_id, path, sha256, imported_at, rows)
|
|
VALUES (?, ?, ?, ?, ?)
|
|
ON CONFLICT(account_id, path) DO UPDATE SET
|
|
sha256 = excluded.sha256,
|
|
imported_at = excluded.imported_at,
|
|
rows = excluded.rows`,
|
|
accountID, path, sha, importedAt, rows)
|
|
if err != nil {
|
|
return 0, fmt.Errorf("record source file %s: %w", path, err)
|
|
}
|
|
var id int64
|
|
if err := d.sql.QueryRow(
|
|
`SELECT id FROM source_files WHERE account_id = ? AND path = ?`, accountID, path).Scan(&id); err != nil {
|
|
return 0, fmt.Errorf("read source file id %s: %w", path, err)
|
|
}
|
|
return id, nil
|
|
}
|
|
|
|
// SourceFileSHA returns the recorded checksum for a statement file, and whether
|
|
// it has been imported before.
|
|
func (d *DB) SourceFileSHA(accountID int64, path string) (string, bool, error) {
|
|
var sha string
|
|
err := d.sql.QueryRow(
|
|
`SELECT sha256 FROM source_files WHERE account_id = ? AND path = ?`, accountID, path).Scan(&sha)
|
|
if err == sql.ErrNoRows {
|
|
return "", false, nil
|
|
}
|
|
if err != nil {
|
|
return "", false, err
|
|
}
|
|
return sha, true, nil
|
|
}
|
|
|
|
// InsertTransaction adds a transaction unless its fingerprint already exists
|
|
// for that account. It reports whether a new row was created.
|
|
//
|
|
// Existing rows are deliberately left untouched, so re-importing a statement
|
|
// that overlaps one already imported adds nothing rather than duplicating it.
|
|
func (d *DB) InsertTransaction(t model.Transaction) (bool, error) {
|
|
var balance any
|
|
if t.BalanceMinor != nil {
|
|
balance = *t.BalanceMinor
|
|
}
|
|
res, err := d.sql.Exec(`
|
|
INSERT INTO transactions
|
|
(account_id, source_file_id, fingerprint, date, description, amount_minor,
|
|
type, balance_minor, rule_tag)
|
|
VALUES (?, ?, ?, ?, ?, ?, ?, ?, NULLIF(?, ''))
|
|
ON CONFLICT(account_id, fingerprint) DO NOTHING`,
|
|
t.AccountID, t.SourceFileID, t.Fingerprint, t.Date, t.Description,
|
|
t.AmountMinor, t.Type, balance, t.RuleTag)
|
|
if err != nil {
|
|
return false, fmt.Errorf("insert transaction: %w", err)
|
|
}
|
|
n, err := res.RowsAffected()
|
|
if err != nil {
|
|
return false, err
|
|
}
|
|
return n > 0, nil
|
|
}
|
|
|
|
// Filter narrows a transaction query.
|
|
type Filter struct {
|
|
AccountSlug string
|
|
// Untagged selects the rows still waiting for a verdict: no tag, and not a
|
|
// leg of a matched transfer. A paired leg has been accounted for by the
|
|
// transfer that claimed it, so listing it as untagged would ask the user to
|
|
// write a rule for something that is already spoken for and that the report
|
|
// leaves out anyway.
|
|
Untagged bool
|
|
Month string // YYYY-MM
|
|
// From and To bound the dates inclusively, as YYYY-MM-DD, either side empty
|
|
// for unbounded. Dates are stored ISO-8601, so a string comparison is a date
|
|
// comparison; nothing has to be parsed to filter on a range.
|
|
From string
|
|
To string
|
|
Search string // case-insensitive substring of the description
|
|
Limit int
|
|
}
|
|
|
|
// Transactions returns rows matching f, newest first.
|
|
func (d *DB) Transactions(f Filter) ([]model.Transaction, error) {
|
|
q := `
|
|
SELECT t.id, t.account_id, a.slug, a.currency, a.minor_digits,
|
|
t.fingerprint, t.date, t.description, t.amount_minor,
|
|
COALESCE(s.path, ''), t.type, t.balance_minor,
|
|
COALESCE(t.rule_tag, ''), x.id
|
|
FROM transactions t
|
|
JOIN accounts a ON a.id = t.account_id
|
|
LEFT JOIN source_files s ON s.id = t.source_file_id
|
|
LEFT JOIN transfers x ON x.out_txn_id = t.id OR x.in_txn_id = t.id
|
|
WHERE 1 = 1`
|
|
var args []any
|
|
if f.AccountSlug != "" {
|
|
q += ` AND a.slug = ?`
|
|
args = append(args, f.AccountSlug)
|
|
}
|
|
if f.Untagged {
|
|
q += ` AND NULLIF(t.rule_tag, '') IS NULL AND x.id IS NULL`
|
|
}
|
|
if f.Month != "" {
|
|
q += ` AND substr(t.date, 1, 7) = ?`
|
|
args = append(args, f.Month)
|
|
}
|
|
if f.From != "" {
|
|
q += ` AND t.date >= ?`
|
|
args = append(args, f.From)
|
|
}
|
|
if f.To != "" {
|
|
q += ` AND t.date <= ?`
|
|
args = append(args, f.To)
|
|
}
|
|
q += ` ORDER BY t.date DESC, t.id DESC`
|
|
// Search and Limit are applied in Go: SQLite's upper()/LIKE fold ASCII
|
|
// only, which would silently fail on Cyrillic statement descriptions.
|
|
|
|
rows, err := d.sql.Query(q, args...)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("query transactions: %w", err)
|
|
}
|
|
defer rows.Close()
|
|
|
|
needle := model.NormalizeDescription(f.Search)
|
|
var out []model.Transaction
|
|
for rows.Next() {
|
|
var (
|
|
t model.Transaction
|
|
balance sql.NullInt64
|
|
transfer sql.NullInt64
|
|
)
|
|
if err := rows.Scan(&t.ID, &t.AccountID, &t.AccountSlug, &t.Currency, &t.MinorDigits,
|
|
&t.Fingerprint, &t.Date, &t.Description, &t.AmountMinor, &t.SourcePath,
|
|
&t.Type, &balance, &t.RuleTag, &transfer); err != nil {
|
|
return nil, err
|
|
}
|
|
if balance.Valid {
|
|
v := balance.Int64
|
|
t.BalanceMinor = &v
|
|
}
|
|
if transfer.Valid {
|
|
v := transfer.Int64
|
|
t.TransferID = &v
|
|
}
|
|
if needle != "" && !strings.Contains(model.NormalizeDescription(t.Description), needle) {
|
|
continue
|
|
}
|
|
out = append(out, t)
|
|
if f.Limit > 0 && len(out) >= f.Limit {
|
|
break
|
|
}
|
|
}
|
|
return out, rows.Err()
|
|
}
|
|
|
|
// RuleAssignment is one row's recomputed rule verdict.
|
|
type RuleAssignment struct {
|
|
ID int64
|
|
Tag string
|
|
}
|
|
|
|
// ApplyRuleResults rewrites rule_tag for every listed row in a single
|
|
// transaction. The manual column is never touched.
|
|
func (d *DB) ApplyRuleResults(rs []RuleAssignment) error {
|
|
tx, err := d.sql.Begin()
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer tx.Rollback()
|
|
|
|
stmt, err := tx.Prepare(
|
|
`UPDATE transactions SET rule_tag = NULLIF(?, '') WHERE id = ?`)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer stmt.Close()
|
|
|
|
for _, r := range rs {
|
|
if _, err := stmt.Exec(r.Tag, r.ID); err != nil {
|
|
return fmt.Errorf("apply rules to txn %d: %w", r.ID, err)
|
|
}
|
|
}
|
|
return tx.Commit()
|
|
}
|
|
|
|
// TransferLink is one matched pair, as decided by the transfer definitions.
|
|
type TransferLink struct {
|
|
DefIndex int
|
|
OutID int64
|
|
InID int64
|
|
}
|
|
|
|
// ReplaceTransfers rewrites the whole pairing in one transaction. Like the
|
|
// tags, it is derived from a file the user edits, so it is replaced wholesale
|
|
// rather than patched: a definition removed from rules.toml must take its pairs
|
|
// with it.
|
|
func (d *DB) ReplaceTransfers(links []TransferLink) error {
|
|
tx, err := d.sql.Begin()
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer tx.Rollback()
|
|
|
|
if _, err := tx.Exec(`DELETE FROM transfers`); err != nil {
|
|
return fmt.Errorf("clear transfers: %w", err)
|
|
}
|
|
stmt, err := tx.Prepare(
|
|
`INSERT INTO transfers (def_index, out_txn_id, in_txn_id) VALUES (?, ?, ?)`)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer stmt.Close()
|
|
|
|
for _, l := range links {
|
|
if _, err := stmt.Exec(l.DefIndex, l.OutID, l.InID); err != nil {
|
|
return fmt.Errorf("link transfer %d→%d: %w", l.OutID, l.InID, err)
|
|
}
|
|
}
|
|
return tx.Commit()
|
|
}
|
|
|
|
// Balance sums every transaction in an account.
|
|
func (d *DB) Balance(accountID int64) (int64, error) {
|
|
var v sql.NullInt64
|
|
err := d.sql.QueryRow(
|
|
`SELECT SUM(amount_minor) FROM transactions WHERE account_id = ?`, accountID).Scan(&v)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
return v.Int64, nil
|
|
}
|
|
|
|
// Count returns the number of transactions in an account.
|
|
func (d *DB) Count(accountID int64) (int, error) {
|
|
var n int
|
|
err := d.sql.QueryRow(
|
|
`SELECT COUNT(*) FROM transactions WHERE account_id = ?`, accountID).Scan(&n)
|
|
return n, err
|
|
}
|
|
|
|
// Months lists the distinct YYYY-MM the index holds, most recent first. It is
|
|
// asked of the whole index rather than of a filtered view on purpose: it is the
|
|
// time axis the report screen steps along, and an axis that grew and shrank as
|
|
// the account or search filter changed would move under the cursor.
|
|
func (d *DB) Months() ([]string, error) {
|
|
rows, err := d.sql.Query(
|
|
`SELECT DISTINCT substr(date, 1, 7) FROM transactions ORDER BY 1 DESC`)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("list months: %w", err)
|
|
}
|
|
defer rows.Close()
|
|
var out []string
|
|
for rows.Next() {
|
|
var s string
|
|
if err := rows.Scan(&s); err != nil {
|
|
return nil, err
|
|
}
|
|
out = append(out, s)
|
|
}
|
|
return out, rows.Err()
|
|
}
|
|
|
|
// SourceFileInfo is what the index recorded about one imported statement.
|
|
type SourceFileInfo struct {
|
|
AccountSlug string
|
|
Path string // relative to the data root, as the importer records it
|
|
SHA256 string
|
|
ImportedAt string
|
|
// Rows is how many transactions the statement holds, as parsed.
|
|
Rows int
|
|
// Added counts the transactions this file introduced. A row seen again in
|
|
// an overlapping statement is deduplicated and stays with the file that
|
|
// brought it first, so this is not how many rows the file holds.
|
|
Added int
|
|
}
|
|
|
|
// SourceFiles lists every statement the index has imported.
|
|
func (d *DB) SourceFiles() ([]SourceFileInfo, error) {
|
|
rows, err := d.sql.Query(`
|
|
SELECT a.slug, s.path, s.sha256, s.imported_at, s.rows, COUNT(t.id)
|
|
FROM source_files s
|
|
JOIN accounts a ON a.id = s.account_id
|
|
LEFT JOIN transactions t ON t.source_file_id = s.id
|
|
GROUP BY s.id
|
|
ORDER BY a.slug, s.path`)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("list source files: %w", err)
|
|
}
|
|
defer rows.Close()
|
|
var out []SourceFileInfo
|
|
for rows.Next() {
|
|
var f SourceFileInfo
|
|
if err := rows.Scan(&f.AccountSlug, &f.Path, &f.SHA256, &f.ImportedAt, &f.Rows, &f.Added); err != nil {
|
|
return nil, err
|
|
}
|
|
out = append(out, f)
|
|
}
|
|
return out, rows.Err()
|
|
}
|
|
|
|
// ForgetSourceFile removes a statement from the index: its transactions, any
|
|
// transfer pairing that used them, and the record of it, returning how many
|
|
// transactions went. Pairing is derived state that the next Link rewrites, so
|
|
// dropping a pair here loses nothing.
|
|
//
|
|
// A transaction two overlapping statements share is stored once, under the
|
|
// file that brought it first, so removing that file can take rows another
|
|
// statement still holds. The account's other statements therefore forget
|
|
// their checksums, and the next import re-reads them and restores those rows.
|
|
func (d *DB) ForgetSourceFile(accountSlug, path string) (int, error) {
|
|
tx, err := d.sql.Begin()
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
defer tx.Rollback()
|
|
|
|
var id, accountID int64
|
|
err = tx.QueryRow(`
|
|
SELECT s.id, s.account_id FROM source_files s JOIN accounts a ON a.id = s.account_id
|
|
WHERE a.slug = ? AND s.path = ?`, accountSlug, path).Scan(&id, &accountID)
|
|
if err == sql.ErrNoRows {
|
|
return 0, nil
|
|
}
|
|
if err != nil {
|
|
return 0, fmt.Errorf("find source file %s: %w", path, err)
|
|
}
|
|
if _, err := tx.Exec(`
|
|
DELETE FROM transfers WHERE
|
|
out_txn_id IN (SELECT id FROM transactions WHERE source_file_id = ?) OR
|
|
in_txn_id IN (SELECT id FROM transactions WHERE source_file_id = ?)`, id, id); err != nil {
|
|
return 0, fmt.Errorf("forget transfers from %s: %w", path, err)
|
|
}
|
|
res, err := tx.Exec(`DELETE FROM transactions WHERE source_file_id = ?`, id)
|
|
if err != nil {
|
|
return 0, fmt.Errorf("forget transactions from %s: %w", path, err)
|
|
}
|
|
n, err := res.RowsAffected()
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
if _, err := tx.Exec(`DELETE FROM source_files WHERE id = ?`, id); err != nil {
|
|
return 0, fmt.Errorf("forget %s: %w", path, err)
|
|
}
|
|
if _, err := tx.Exec(`UPDATE source_files SET sha256 = '' WHERE account_id = ?`, accountID); err != nil {
|
|
return 0, fmt.Errorf("mark %s's statements for re-reading: %w", accountSlug, err)
|
|
}
|
|
return int(n), tx.Commit()
|
|
}
|
|
|
|
// Tags lists every tag in use, for completion in the rule builder.
|
|
func (d *DB) Tags() ([]string, error) {
|
|
rows, err := d.sql.Query(`
|
|
SELECT DISTINCT rule_tag FROM transactions
|
|
WHERE rule_tag IS NOT NULL AND rule_tag != '' ORDER BY rule_tag`)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
defer rows.Close()
|
|
var out []string
|
|
for rows.Next() {
|
|
var s string
|
|
if err := rows.Scan(&s); err != nil {
|
|
return nil, err
|
|
}
|
|
out = append(out, s)
|
|
}
|
|
return out, rows.Err()
|
|
}
|