package parser import ( "fmt" "regexp" "strings" "time" "git.petrovv.com/nikola/money/internal/config" ) func init() { Register("nlb", func(acc *config.Account) (Parser, error) { return &nlbParser{digits: acc.Digits()}, nil }) } // nlbParser reads an NLB izpisek PDF. // // A transaction starts on a line beginning with a dd.mm.yy date and ends with // the signed amount and the running balance. Long descriptions and the other // side's account number wrap onto indented continuation lines below. type nlbParser struct { digits int } var ( // A transaction line starts with a two-digit date. nlbStart = regexp.MustCompile(`^\d{2}\.\d{2}\.\d{2}\s`) // The whole line: date, free-form middle, signed amount, running balance. // Both the ASCII hyphen and the typographic minus U+2212 count as a sign; // which one a PDF carries depends on the font it was produced with. nlbLine = regexp.MustCompile( `^(?P\d{2}\.\d{2}\.\d{2})\s+` + `(?P.*?)\s+` + `(?P[-+\x{2212}][\d.,]+)\s+` + `(?P[-\x{2212}]?[\d.,]+[-\x{2212}]?)\s*$`) // Columns inside a line are separated by four or more spaces. nlbFieldSplit = regexp.MustCompile(`\s{4,}`) // A Slovenian IBAN, which NLB prints in a column of its own. It is split // out only so that its wrapped fragments can be rejoined; the result goes // onto the end of the description, where every other parser keeps it. nlbAccount = regexp.MustCompile(`^SI\d{2}(?:\s?\d{4}){3}\s?\d{3}$`) ) // nlbMinContinuationIndent is the shallowest indent a wrapped description line // may have. The real threshold is the description column of the transaction // the line belongs to, measured per line rather than hardcoded: pdftotext // squeezes runs of spaces, so absolute columns shift with the font and page // size of the PDF. This floor only rejects flush-left page furniture. const nlbMinContinuationIndent = 2 // nlbDateLayout is the two-digit-year date NLB prints. const nlbDateLayout = "02.01.06" func (p *nlbParser) Parse(path string, acc *config.Account) ([]RawTxn, error) { text, err := pdfToText(path) if err != nil { return nil, err } return parseNLBText(text, p.digits) } // parseNLBText holds the whole parser, separated from PDF extraction so it can // be tested against captured pdftotext output. func parseNLBText(text string, digits int) ([]RawTxn, error) { var ( txns []RawTxn // The account column per transaction, built up alongside because it // wraps in fragments of its own and has to be rejoined before it can // be appended to the description. trailing []string descAt int // description column of the last transaction line ) for _, page := range pages(text) { // Continuation lines may only attach to a transaction started on the // same page, so a wrapped line at the top of a page is page furniture. pageStart := len(txns) for i, line := range strings.Split(page, "\n") { if strings.TrimSpace(line) == "" { continue } if nlbStart.MatchString(line) { txn, account, col, err := parseNLBLine(line, digits) if err != nil { return nil, fmt.Errorf("line %d: %w", i+1, err) } txns = append(txns, txn) trailing = append(trailing, account) descAt = col continue } if len(txns) <= pageStart { continue // header, before any transaction on this page } // A wrapped description sits under the description column of the // transaction it continues; anything to the left of that is a // footer or a column heading. threshold := max(descAt, nlbMinContinuationIndent) if indent(line) < threshold { continue } parts := nlbFieldSplit.Split(strings.TrimSpace(line), -1) last := len(txns) - 1 txns[last].Description += " " + parts[0] if len(parts) > 1 { trailing[last] = strings.TrimSpace(trailing[last] + " " + parts[1]) } } } // The account column goes last rather than in the position it occupied on // the page, so that an IBAN wrapped over several lines stays contiguous // and a glob can match it. for i := range txns { txns[i].Description = strings.Join(strings.Fields(txns[i].Description+" "+trailing[i]), " ") } return txns, nil } // parseNLBLine parses one transaction line, returning the transaction, the // account number found among its columns, and the column at which the // description starts, which is where its wrapped lines will sit. func parseNLBLine(line string, digits int) (RawTxn, string, int, error) { trimmed := strings.TrimRight(line, " \t\r") loc := nlbLine.FindStringSubmatchIndex(trimmed) if loc == nil { return RawTxn{}, "", 0, fmt.Errorf("line starts with a date but has no amount and balance: %q", strings.TrimSpace(line)) } group := func(name string) string { i := nlbLine.SubexpIndex(name) * 2 if loc[i] < 0 { return "" } return trimmed[loc[i]:loc[i+1]] } var ( date = group("date") middle = group("middle") amount = group("amount") balance = group("balance") descCol = loc[nlbLine.SubexpIndex("middle")*2] ) d, err := time.Parse(nlbDateLayout, date) if err != nil { return RawTxn{}, "", 0, fmt.Errorf("date %q: %w", date, err) } amountMinor, err := parseNLBNumber(amount, digits) if err != nil { return RawTxn{}, "", 0, fmt.Errorf("amount %q: %w", amount, err) } balanceMinor, err := parseNLBNumber(balance, digits) if err != nil { return RawTxn{}, "", 0, fmt.Errorf("balance %q: %w", balance, err) } // The middle holds the description and, sometimes, the IBAN. var ( account string desc []string ) for _, f := range nlbFieldSplit.Split(strings.TrimSpace(middle), -1) { if f == "" { continue } if nlbAccount.MatchString(f) { account = f continue } desc = append(desc, f) } return RawTxn{ Date: d.Format("2006-01-02"), Description: strings.Join(desc, " "), AmountMinor: amountMinor, BalanceMinor: &balanceMinor, }, account, descCol, nil } // parseNLBNumber reads the 1.234,56 format, where a trailing minus marks a // negative balance. func parseNLBNumber(s string, digits int) (int64, error) { return ParseAmount(s, ",", ".", digits) }