Four changes to statement handling in the web app, made together and touching the same upload and statements-list code. Name NLB uploads by statement date. parser.Namer is an optional interface, like Warner, through which a parser names its statements; nlb reads the "Datum izpiska" from the izpisek header and names it izpisek_YYYY_MM_DD, lowercase, extension included -- ported from the rename_izpiski.py it replaces. Uploads are staged as dotfiles, invisible to import, so the parser can read them; two downloads of one statement then meet under one name and the second is recognised as already there, while a different statement of the same date is numbered _2 as the script did. Only uploads are named: source_files records statements by path, so renaming a file already in a folder would orphan its rows. Delete a statement from the statements list. The file is removed from disk for good -- the page says so before it asks -- and store.ForgetSourceFile drops its transactions and their transfer rows. A row two overlapping statements share is stored once, under the file imported first, so it goes too; the account's other statements forget their checksums and show as changed until the next Import re-reads them and restores it. A file already gone from disk can be forgotten. Upload and delete no longer import. Importing stays the user's call, made with the Import button, so a batch can be put together and looked over first. Delete still re-pairs transfers, which reads no statement. Show rows and new rows per statement. The list read "0" for a file whose rows an earlier, overlapping statement already held, which looked like a file that failed to parse. source_files now records how many transactions each statement holds, and the list reads "3 rows · 0 new". This adds a column the code reads, so an index built by an earlier version fails with "no such column: s.rows": delete index.db and import again. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
215 lines
6.7 KiB
Go
215 lines
6.7 KiB
Go
package parser
|
|
|
|
import (
|
|
"fmt"
|
|
"regexp"
|
|
"strings"
|
|
"time"
|
|
|
|
"git.petrovv.com/nikola/money/internal/config"
|
|
)
|
|
|
|
func init() {
|
|
Register("nlb", func(acc *config.Account) (Parser, error) {
|
|
return &nlbParser{digits: acc.Digits()}, nil
|
|
})
|
|
}
|
|
|
|
// nlbParser reads an NLB izpisek PDF.
|
|
//
|
|
// A transaction starts on a line beginning with a dd.mm.yy date and ends with
|
|
// the signed amount and the running balance. Long descriptions and the other
|
|
// side's account number wrap onto indented continuation lines below.
|
|
type nlbParser struct {
|
|
digits int
|
|
}
|
|
|
|
var (
|
|
// A transaction line starts with a two-digit date.
|
|
nlbStart = regexp.MustCompile(`^\d{2}\.\d{2}\.\d{2}\s`)
|
|
// The whole line: date, free-form middle, signed amount, running balance.
|
|
// Both the ASCII hyphen and the typographic minus U+2212 count as a sign;
|
|
// which one a PDF carries depends on the font it was produced with.
|
|
nlbLine = regexp.MustCompile(
|
|
`^(?P<date>\d{2}\.\d{2}\.\d{2})\s+` +
|
|
`(?P<middle>.*?)\s+` +
|
|
`(?P<amount>[-+\x{2212}][\d.,]+)\s+` +
|
|
`(?P<balance>[-\x{2212}]?[\d.,]+[-\x{2212}]?)\s*$`)
|
|
// Columns inside a line are separated by four or more spaces.
|
|
nlbFieldSplit = regexp.MustCompile(`\s{4,}`)
|
|
// A Slovenian IBAN, which NLB prints in a column of its own. It is split
|
|
// out only so that its wrapped fragments can be rejoined; the result goes
|
|
// onto the end of the description, where every other parser keeps it.
|
|
nlbAccount = regexp.MustCompile(`^SI\d{2}(?:\s?\d{4}){3}\s?\d{3}$`)
|
|
)
|
|
|
|
// nlbStatementDate is the issue date in an izpisek's header. It names the
|
|
// statement: NLB's downloads are not named for what they hold.
|
|
var nlbStatementDate = regexp.MustCompile(`Datum izpiska\s+(\d{2})\.(\d{2})\.(\d{4})`)
|
|
|
|
// nlbMinContinuationIndent is the shallowest indent a wrapped description line
|
|
// may have. The real threshold is the description column of the transaction
|
|
// the line belongs to, measured per line rather than hardcoded: pdftotext
|
|
// squeezes runs of spaces, so absolute columns shift with the font and page
|
|
// size of the PDF. This floor only rejects flush-left page furniture.
|
|
const nlbMinContinuationIndent = 2
|
|
|
|
// nlbDateLayout is the two-digit-year date NLB prints.
|
|
const nlbDateLayout = "02.01.06"
|
|
|
|
func (p *nlbParser) Parse(path string, acc *config.Account) ([]RawTxn, error) {
|
|
text, err := pdfToText(path)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
return parseNLBText(text, p.digits)
|
|
}
|
|
|
|
// StatementName names an izpisek after its statement date, izpisek_YYYY_MM_DD,
|
|
// so the folder sorts by date and two downloads of one statement collide.
|
|
func (p *nlbParser) StatementName(path string) (string, error) {
|
|
text, err := pdfToText(path)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
return nlbStatementName(text), nil
|
|
}
|
|
|
|
func nlbStatementName(text string) string {
|
|
m := nlbStatementDate.FindStringSubmatch(text)
|
|
if m == nil {
|
|
return ""
|
|
}
|
|
return fmt.Sprintf("izpisek_%s_%s_%s", m[3], m[2], m[1])
|
|
}
|
|
|
|
// parseNLBText holds the whole parser, separated from PDF extraction so it can
|
|
// be tested against captured pdftotext output.
|
|
func parseNLBText(text string, digits int) ([]RawTxn, error) {
|
|
var (
|
|
txns []RawTxn
|
|
// The account column per transaction, built up alongside because it
|
|
// wraps in fragments of its own and has to be rejoined before it can
|
|
// be appended to the description.
|
|
trailing []string
|
|
descAt int // description column of the last transaction line
|
|
)
|
|
|
|
for _, page := range pages(text) {
|
|
// Continuation lines may only attach to a transaction started on the
|
|
// same page, so a wrapped line at the top of a page is page furniture.
|
|
pageStart := len(txns)
|
|
|
|
for i, line := range strings.Split(page, "\n") {
|
|
if strings.TrimSpace(line) == "" {
|
|
continue
|
|
}
|
|
|
|
if nlbStart.MatchString(line) {
|
|
txn, account, col, err := parseNLBLine(line, digits)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("line %d: %w", i+1, err)
|
|
}
|
|
txns = append(txns, txn)
|
|
trailing = append(trailing, account)
|
|
descAt = col
|
|
continue
|
|
}
|
|
|
|
if len(txns) <= pageStart {
|
|
continue // header, before any transaction on this page
|
|
}
|
|
// A wrapped description sits under the description column of the
|
|
// transaction it continues; anything to the left of that is a
|
|
// footer or a column heading.
|
|
threshold := max(descAt, nlbMinContinuationIndent)
|
|
if indent(line) < threshold {
|
|
continue
|
|
}
|
|
|
|
parts := nlbFieldSplit.Split(strings.TrimSpace(line), -1)
|
|
last := len(txns) - 1
|
|
txns[last].Description += " " + parts[0]
|
|
if len(parts) > 1 {
|
|
trailing[last] = strings.TrimSpace(trailing[last] + " " + parts[1])
|
|
}
|
|
}
|
|
}
|
|
|
|
// The account column goes last rather than in the position it occupied on
|
|
// the page, so that an IBAN wrapped over several lines stays contiguous
|
|
// and a glob can match it.
|
|
for i := range txns {
|
|
txns[i].Description = strings.Join(strings.Fields(txns[i].Description+" "+trailing[i]), " ")
|
|
}
|
|
return txns, nil
|
|
}
|
|
|
|
// parseNLBLine parses one transaction line, returning the transaction, the
|
|
// account number found among its columns, and the column at which the
|
|
// description starts, which is where its wrapped lines will sit.
|
|
func parseNLBLine(line string, digits int) (RawTxn, string, int, error) {
|
|
trimmed := strings.TrimRight(line, " \t\r")
|
|
loc := nlbLine.FindStringSubmatchIndex(trimmed)
|
|
if loc == nil {
|
|
return RawTxn{}, "", 0, fmt.Errorf("line starts with a date but has no amount and balance: %q", strings.TrimSpace(line))
|
|
}
|
|
group := func(name string) string {
|
|
i := nlbLine.SubexpIndex(name) * 2
|
|
if loc[i] < 0 {
|
|
return ""
|
|
}
|
|
return trimmed[loc[i]:loc[i+1]]
|
|
}
|
|
var (
|
|
date = group("date")
|
|
middle = group("middle")
|
|
amount = group("amount")
|
|
balance = group("balance")
|
|
descCol = loc[nlbLine.SubexpIndex("middle")*2]
|
|
)
|
|
|
|
d, err := time.Parse(nlbDateLayout, date)
|
|
if err != nil {
|
|
return RawTxn{}, "", 0, fmt.Errorf("date %q: %w", date, err)
|
|
}
|
|
|
|
amountMinor, err := parseNLBNumber(amount, digits)
|
|
if err != nil {
|
|
return RawTxn{}, "", 0, fmt.Errorf("amount %q: %w", amount, err)
|
|
}
|
|
balanceMinor, err := parseNLBNumber(balance, digits)
|
|
if err != nil {
|
|
return RawTxn{}, "", 0, fmt.Errorf("balance %q: %w", balance, err)
|
|
}
|
|
|
|
// The middle holds the description and, sometimes, the IBAN.
|
|
var (
|
|
account string
|
|
desc []string
|
|
)
|
|
for _, f := range nlbFieldSplit.Split(strings.TrimSpace(middle), -1) {
|
|
if f == "" {
|
|
continue
|
|
}
|
|
if nlbAccount.MatchString(f) {
|
|
account = f
|
|
continue
|
|
}
|
|
desc = append(desc, f)
|
|
}
|
|
|
|
return RawTxn{
|
|
Date: d.Format("2006-01-02"),
|
|
Description: strings.Join(desc, " "),
|
|
AmountMinor: amountMinor,
|
|
BalanceMinor: &balanceMinor,
|
|
}, account, descCol, nil
|
|
}
|
|
|
|
// parseNLBNumber reads the 1.234,56 format, where a trailing minus marks a
|
|
// negative balance.
|
|
func parseNLBNumber(s string, digits int) (int64, error) {
|
|
return ParseAmount(s, ",", ".", digits)
|
|
}
|