A data directory holds one folder per account. Statements dropped into those folders are parsed into a rebuildable SQLite index, categorised by ordered glob rules in rules.toml, and browsed or hand-tagged in a Bubble Tea TUI. Movements between the user's own accounts are marked as transfers by the same rules and excluded from spending totals. Manual tags and transfer marks are stored separately from the rule-derived ones and always win, so editing rules.toml and re-running retag never destroys hand edits. Parsers are pluggable. Three are ported from the Python extractors they replace -- nlb and traderepublic read PDFs via pdftotext -layout, revolut reads the CSV export -- alongside a configurable-column CSV parser and a cmd parser that shells out to an external script. Both ports fix two latent bugs in the originals: the sign character class rejected the typographic minus U+2212 that some PDF fonts emit, and NLB's hardcoded continuation indent broke when pdftotext compressed runs of spaces, so the threshold is now measured from the description column. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
125 lines
3.3 KiB
Go
125 lines
3.3 KiB
Go
package parser
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"encoding/csv"
|
|
"fmt"
|
|
"io"
|
|
"os/exec"
|
|
"strings"
|
|
"time"
|
|
|
|
"git.petrovv.com/nikola/money/internal/config"
|
|
)
|
|
|
|
func init() {
|
|
Register("cmd", newCmdParser)
|
|
}
|
|
|
|
// FileToken is replaced with the statement's absolute path in a cmd parser's argv.
|
|
const FileToken = "{{file}}"
|
|
|
|
// cmdRunTimeout bounds an extractor run so a hung script cannot wedge an import.
|
|
const cmdRunTimeout = 2 * time.Minute
|
|
|
|
// cmdParser runs an external extractor and reads normalised CSV from its
|
|
// stdout: date,description,amount with any further columns ignored. This is
|
|
// the bridge that lets the existing Python extractors be used unchanged.
|
|
type cmdParser struct {
|
|
cfg config.CmdConfig
|
|
digits int
|
|
}
|
|
|
|
func newCmdParser(acc *config.Account) (Parser, error) {
|
|
if acc.Cmd == nil || len(acc.Cmd.Argv) == 0 {
|
|
return nil, fmt.Errorf("account %s: parser \"cmd\" requires [cmd] with a non-empty argv", acc.Slug)
|
|
}
|
|
cfg := *acc.Cmd
|
|
if cfg.Layout == "" {
|
|
cfg.Layout = "2006-01-02"
|
|
}
|
|
if !hasFileToken(cfg.Argv) {
|
|
return nil, fmt.Errorf("account %s: [cmd] argv must contain %s so the script knows which file to read",
|
|
acc.Slug, FileToken)
|
|
}
|
|
return &cmdParser{cfg: cfg, digits: acc.Digits()}, nil
|
|
}
|
|
|
|
func hasFileToken(argv []string) bool {
|
|
for _, a := range argv {
|
|
if strings.Contains(a, FileToken) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
func (p *cmdParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
|
|
argv := make([]string, len(p.cfg.Argv))
|
|
for i, a := range p.cfg.Argv {
|
|
argv[i] = strings.ReplaceAll(a, FileToken, path)
|
|
}
|
|
|
|
ctx, cancel := context.WithTimeout(context.Background(), cmdRunTimeout)
|
|
defer cancel()
|
|
|
|
cmd := exec.CommandContext(ctx, argv[0], argv[1:]...)
|
|
cmd.Dir = acc.Dir
|
|
var stdout, stderr bytes.Buffer
|
|
cmd.Stdout = &stdout
|
|
cmd.Stderr = &stderr
|
|
|
|
if err := cmd.Run(); err != nil {
|
|
msg := strings.TrimSpace(stderr.String())
|
|
if ctx.Err() == context.DeadlineExceeded {
|
|
return nil, fmt.Errorf("extractor %v timed out after %s", argv, cmdRunTimeout)
|
|
}
|
|
if msg != "" {
|
|
return nil, fmt.Errorf("extractor %v failed: %w: %s", argv, err, msg)
|
|
}
|
|
return nil, fmt.Errorf("extractor %v failed: %w", argv, err)
|
|
}
|
|
|
|
return p.parseOutput(&stdout, argv)
|
|
}
|
|
|
|
func (p *cmdParser) parseOutput(out io.Reader, argv []string) ([]RawTxn, error) {
|
|
r := csv.NewReader(out)
|
|
r.FieldsPerRecord = -1
|
|
r.LazyQuotes = true
|
|
|
|
var txns []RawTxn
|
|
for row := 0; ; row++ {
|
|
rec, err := r.Read()
|
|
if err == io.EOF {
|
|
break
|
|
}
|
|
if err != nil {
|
|
return nil, fmt.Errorf("extractor %v: output row %d: %w", argv, row+1, err)
|
|
}
|
|
if row < p.cfg.SkipRows || isBlank(rec) {
|
|
continue
|
|
}
|
|
if len(rec) < 3 {
|
|
return nil, fmt.Errorf("extractor %v: output row %d has %d columns, want date,description,amount",
|
|
argv, row+1, len(rec))
|
|
}
|
|
d, err := time.Parse(p.cfg.Layout, strings.TrimSpace(rec[0]))
|
|
if err != nil {
|
|
return nil, fmt.Errorf("extractor %v: output row %d: date %q does not match layout %q",
|
|
argv, row+1, rec[0], p.cfg.Layout)
|
|
}
|
|
amount, err := ParseAmount(rec[2], ".", "", p.digits)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("extractor %v: output row %d: %w", argv, row+1, err)
|
|
}
|
|
txns = append(txns, RawTxn{
|
|
Date: d.Format("2006-01-02"),
|
|
Description: strings.TrimSpace(rec[1]),
|
|
AmountMinor: amount,
|
|
})
|
|
}
|
|
return txns, nil
|
|
}
|