File order decided precedence, so a narrow rule had to be written above the broad one it carves an exception out of -- an ordering constraint the file cannot show and the user has to remember. *NIKOLA* below *NIK* silently matched nothing, and a catch-all * could only ever be the last line. Engine.New now sorts once and MatchIndex walks that order: most literal characters first, then fewest *, then account-scoped over unscoped. Literals are what a rule commits to and a * is what it gives up, so a bare * is tried last wherever it sits. The sort is stable, so equally specific rules keep file order and the earlier one wins -- which is all position decides now, and why AppendRule can keep appending without displacing a rule written by hand. The two orders must not be confused: Rules(), Usage and MatchIndex still speak in file positions, because that is what the rules screen numbers and what DeleteRules deletes by. A shadowed rule still reports zero usage, but a zero no longer says anything about where the rule sits. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
182 lines
5.9 KiB
Go
182 lines
5.9 KiB
Go
// Package rules applies the glob rules from rules.toml to transactions,
|
|
// deciding their tag. The most specific rule that fits wins, so a general rule
|
|
// and the narrower one carving an exception out of it can be written in either
|
|
// order. It is the only thing that decides a tag, so rule_tag is derived state
|
|
// and Retag can rewrite it wholesale at any time.
|
|
package rules
|
|
|
|
import (
|
|
"sort"
|
|
|
|
"git.petrovv.com/nikola/money/internal/config"
|
|
"git.petrovv.com/nikola/money/internal/glob"
|
|
"git.petrovv.com/nikola/money/internal/model"
|
|
"git.petrovv.com/nikola/money/internal/store"
|
|
)
|
|
|
|
// Engine evaluates rules most specific first; the first match wins.
|
|
type Engine struct {
|
|
// rules is file order, which is what Rules and Usage report and what the
|
|
// rules screen numbers and deletes by. Evaluation does not use it.
|
|
rules []config.Rule
|
|
// order indexes into rules, most specific first. Specificity decides
|
|
// precedence rather than position, so "*NIKOLA*" claims what it names
|
|
// wherever it sits relative to the "*NIK*" that would otherwise swallow it.
|
|
order []int
|
|
}
|
|
|
|
// New builds an engine from the parsed rules file, working out once which rule
|
|
// is tried before which.
|
|
func New(r *config.Rules) *Engine {
|
|
e := &Engine{rules: r.Rule, order: make([]int, len(r.Rule))}
|
|
for i := range e.order {
|
|
e.order[i] = i
|
|
}
|
|
// Stable, so rules of equal specificity keep file order between them and
|
|
// the earlier one still wins.
|
|
sort.SliceStable(e.order, func(a, b int) bool {
|
|
return moreSpecific(e.rules[e.order[a]], e.rules[e.order[b]])
|
|
})
|
|
return e
|
|
}
|
|
|
|
// moreSpecific reports whether a is tried before b.
|
|
//
|
|
// The literal characters a rule spells out are the evidence: they are what it
|
|
// commits to, and what it hands to a '*' is what it gives up. So "*NIKOLA*"
|
|
// beats "*NIK*", and a bare "*" sits last of all — the catch-all can be written
|
|
// anywhere in the file and still catch only what nothing else wanted.
|
|
//
|
|
// Two rules spelling out the same amount are separated by what else they pin
|
|
// down: fewer '*' first, since "NIKOLA" also fixes both ends where "*NIKOLA*"
|
|
// does not, and then an account-scoped rule over one that applies everywhere.
|
|
// Rules that tie on all three are left to file order by the stable sort.
|
|
func moreSpecific(a, b config.Rule) bool {
|
|
if x, y := literals(a), literals(b); x != y {
|
|
return x > y
|
|
}
|
|
if x, y := stars(a), stars(b); x != y {
|
|
return x < y
|
|
}
|
|
return a.Account != "" && b.Account == ""
|
|
}
|
|
|
|
// literals counts the characters a rule pins down exactly, across every pattern
|
|
// it sets: a rule with both match and type has to satisfy both, so both count.
|
|
// '?' is not one of them — it fixes a length, not a character.
|
|
func literals(r config.Rule) int {
|
|
n := 0
|
|
for _, c := range r.Match + r.Type {
|
|
if c != '*' && c != '?' {
|
|
n++
|
|
}
|
|
}
|
|
return n
|
|
}
|
|
|
|
// stars counts the unbounded wildcards, the only ones that let a pattern match
|
|
// a run of any length.
|
|
func stars(r config.Rule) int {
|
|
n := 0
|
|
for _, c := range r.Match + r.Type {
|
|
if c == '*' {
|
|
n++
|
|
}
|
|
}
|
|
return n
|
|
}
|
|
|
|
// Rules returns the rules in file order, as loaded from rules.toml. That is the
|
|
// order the rules screen shows and deletes by; it is no longer the order they
|
|
// are tried in.
|
|
func (e *Engine) Rules() []config.Rule { return e.rules }
|
|
|
|
// Match returns the first rule matching a transaction on the given account, or
|
|
// nil if none does.
|
|
func (e *Engine) Match(accountSlug string, t model.Transaction) *config.Rule {
|
|
if i := e.MatchIndex(accountSlug, t); i >= 0 {
|
|
return &e.rules[i]
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// MatchIndex returns the file position of the rule claiming a transaction, or
|
|
// -1 if none does. Every pattern a rule sets must match: a rule with both
|
|
// match and type is an "and", not an "or".
|
|
//
|
|
// Rules are tried most specific first, so the answer is the narrowest rule that
|
|
// fits and not merely the topmost. The returned index is still the position in
|
|
// rules.toml, because that is what the caller can point the user at — and it is
|
|
// what reveals a rule that is fully shadowed by a more specific one and so
|
|
// never applies to anything.
|
|
func (e *Engine) MatchIndex(accountSlug string, t model.Transaction) int {
|
|
var (
|
|
description = model.NormalizeDescription(t.Description)
|
|
kind = model.NormalizeDescription(t.Type)
|
|
)
|
|
for _, i := range e.order {
|
|
r := &e.rules[i]
|
|
if r.Account != "" && r.Account != accountSlug {
|
|
continue
|
|
}
|
|
if r.Match != "" && !glob.Match(r.Match, description) {
|
|
continue
|
|
}
|
|
if r.Type != "" && !glob.Match(r.Type, kind) {
|
|
continue
|
|
}
|
|
return i
|
|
}
|
|
return -1
|
|
}
|
|
|
|
// Usage counts how many transactions each rule actually claims, indexed by file
|
|
// position. A rule with a count of zero is dead: either nothing matches it, or
|
|
// a more specific rule takes everything it would have caught.
|
|
func (e *Engine) Usage(txns []model.Transaction) []int {
|
|
counts := make([]int, len(e.rules))
|
|
for _, t := range txns {
|
|
if i := e.MatchIndex(t.AccountSlug, t); i >= 0 {
|
|
counts[i]++
|
|
}
|
|
}
|
|
return counts
|
|
}
|
|
|
|
// ApplyTxn returns the tag for a transaction, empty if no rule matches.
|
|
func (e *Engine) ApplyTxn(accountSlug string, t model.Transaction) string {
|
|
if r := e.Match(accountSlug, t); r != nil {
|
|
return r.Tag
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// Apply is the description-only shorthand, for callers that have nothing else.
|
|
func (e *Engine) Apply(accountSlug, description string) string {
|
|
return e.ApplyTxn(accountSlug, model.Transaction{Description: description})
|
|
}
|
|
|
|
// Retag recomputes rule verdicts for every transaction in the index and
|
|
// writes them back. It returns how many rows changed.
|
|
func (e *Engine) Retag(db *store.DB) (int, error) {
|
|
txns, err := db.Transactions(store.Filter{})
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
var changed []store.RuleAssignment
|
|
for _, t := range txns {
|
|
tag := e.ApplyTxn(t.AccountSlug, t)
|
|
if tag == t.RuleTag {
|
|
continue
|
|
}
|
|
changed = append(changed, store.RuleAssignment{ID: t.ID, Tag: tag})
|
|
}
|
|
if len(changed) == 0 {
|
|
return 0, nil
|
|
}
|
|
if err := db.ApplyRuleResults(changed); err != nil {
|
|
return 0, err
|
|
}
|
|
return len(changed), nil
|
|
}
|