diff --git a/CLAUDE.md b/CLAUDE.md index 825e2ea..e7ff22d 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -313,15 +313,18 @@ description — at the end, and not in the position it held on the page, so an IBAN wrapped across continuation lines stays contiguous for a glob to match. A new parser must do the same rather than reintroduce a structured field. -**A parser may name its statements, and only uploads use it.** `parser.Namer` -is optional, like `parser.Warner`: `nlb` names an izpisek `izpisek_YYYY_MM_DD` -after its *Datum izpiska*, ported from the `rename_izpiski.py` it replaced. -`/api/upload` stages each file as a dotfile — invisible to import — so the -parser can read it, then numbers a chosen name that is taken by different -contents (`_2`, as the script did) where a name the user gave is refused with -409. Nothing renames a file already in a folder: `source_files` records -statements by path, so a rename would orphan its rows as `missing` until the -index is rebuilt. Covered by `TestUploadNamesStatementsTheParserCanName`. +**A parser may name its statements, and only uploads use it.** `parser.Namer` is +optional, like `parser.Warner`: `nlb` names an izpisek `izpisek_YYYY_MM_DD` +after its *Datum izpiska*, ported from the `rename_izpiski.py` it replaced, and +`traderepublic` names a statement `traderepublic_YYYY_MM_DD_YYYY_MM_DD` after +the period in its header. A name is derived from what the statement says, never +from the upload's own name, and comes back without an extension; the upload's is +kept, lowercased. `/api/upload` stages each file as a dotfile — invisible to +import — so the parser can read it, then numbers a chosen name that is taken by +different contents (`_2`, as the script did) where a name the user gave is +refused with 409. Nothing renames a file already in a folder: `source_files` +records statements by path, so a rename would orphan its rows as `missing` until +the index is rebuilt. Covered by `TestUploadNamesStatementsTheParserCanName`. ## Verifying diff --git a/README.md b/README.md index 8835548..2c01211 100644 --- a/README.md +++ b/README.md @@ -216,14 +216,16 @@ already exist with an `account.toml`: an upload adds statements to an account, it does not create one. Some banks name their downloads unhelpfully, so a parser may name the statement -instead. An NLB izpisek is saved as `izpisek_YYYY_MM_DD.pdf` after the *Datum -izpiska* in its header — lowercase throughout, extension included — so the -folder sorts by date and the status line says what each file became. Downloading -the same statement twice then lands on the same name and is recognised as -already there; a different statement with the same date (a reissue) is numbered -`_2`, `_3` rather than refused. A statement with no date it can find keeps its -own name. Only uploads are named — files already in a folder are never renamed, -since the index records them by path. +instead — lowercase throughout, extension included — and the status line says +what each file became. An NLB izpisek is saved as `izpisek_YYYY_MM_DD.pdf` after +the *Datum izpiska* in its header, and a Trade Republic statement, which +downloads as `document-N.pdf`, as `traderepublic_YYYY_MM_DD_YYYY_MM_DD.pdf` +after the period on its first page (`DATE 01 May 2025 - 31 Jul 2026`). Either +way the folder sorts by date. Downloading the same statement twice then lands on +the same name and is recognised as already there; a different statement with the +same name (a reissue) is numbered `_2`, `_3` rather than refused. A statement +with no date it can find keeps its own name. Only uploads are named — files +already in a folder are never renamed, since the index records them by path. ### Keys @@ -678,7 +680,7 @@ nothing else to configure — `parser` names one of these and that is all. | `parser` | Statement | Notes | | --- | --- | --- | | `nlb` | NLB izpisek PDF | Wrapped descriptions are folded in from continuation lines, and the IBAN column is appended to the description. Uploads are saved as `izpisek_YYYY_MM_DD.pdf` after the statement date. | -| `traderepublic` | Trade Republic PDF | Handles both the single-line and the stacked layout by measuring column positions. | +| `traderepublic` | Trade Republic PDF | Handles both the single-line and the stacked layout by measuring column positions. Uploads are saved as `traderepublic_YYYY_MM_DD_YYYY_MM_DD.pdf` after the statement period. | | `revolut` | `account-statement*.csv` | Skips non-COMPLETED rows, folds the fee into the amount. | ```toml diff --git a/internal/parser/traderepublic.go b/internal/parser/traderepublic.go index c494f24..d74c563 100644 --- a/internal/parser/traderepublic.go +++ b/internal/parser/traderepublic.go @@ -46,6 +46,12 @@ const trAmountSlack = 5 // trDateLayout is the date Trade Republic prints, e.g. "05 Jan 2026". const trDateLayout = "02 Jan 2006" +// trPeriod is the statement period in the address block at the top of the +// first page, e.g. "DATE 01 May 2025 - 31 Jul 2026". It names the +// statement: Trade Republic's downloads are all called document-N.pdf. The +// transaction table's own DATE heading never matches, since no dates follow it. +var trPeriod = regexp.MustCompile(`\bDATE\s+(\d{2} [A-Z][a-z]{2} \d{4})\s*[-\x{2013}\x{2212}]\s*(\d{2} [A-Z][a-z]{2} \d{4})`) + func (p *tradeRepublicParser) Parse(path string, acc *config.Account) ([]RawTxn, error) { text, err := pdfToText(path) if err != nil { @@ -54,6 +60,33 @@ func (p *tradeRepublicParser) Parse(path string, acc *config.Account) ([]RawTxn, return parseTradeRepublicText(text, p.digits) } +// StatementName names a statement after its period, +// traderepublic_YYYY_MM_DD_YYYY_MM_DD, so the folder sorts by when it starts +// and two downloads of one statement collide. +func (p *tradeRepublicParser) StatementName(path string) (string, error) { + text, err := pdfToText(path) + if err != nil { + return "", err + } + return tradeRepublicStatementName(text), nil +} + +func tradeRepublicStatementName(text string) string { + m := trPeriod.FindStringSubmatch(text) + if m == nil { + return "" + } + from, err := time.Parse(trDateLayout, m[1]) + if err != nil { + return "" + } + to, err := time.Parse(trDateLayout, m[2]) + if err != nil { + return "" + } + return "traderepublic_" + from.Format("2006_01_02") + "_" + to.Format("2006_01_02") +} + // parseTradeRepublicText holds the whole parser, separated from PDF extraction // so it can be tested against captured pdftotext output. func parseTradeRepublicText(text string, digits int) ([]RawTxn, error) { diff --git a/internal/parser/traderepublic_test.go b/internal/parser/traderepublic_test.go index 200003f..1aef482 100644 --- a/internal/parser/traderepublic_test.go +++ b/internal/parser/traderepublic_test.go @@ -162,3 +162,19 @@ func TestParseTradeRepublicIgnoresAmountsInDescription(t *testing.T) { t.Errorf("description = %q, want the inline figure kept", txns[0].Description) } } + +// A statement is named after the period in its header; the transaction +// table's DATE heading, with no dates after it, is not mistaken for it. +func TestTradeRepublicStatementName(t *testing.T) { + header := "NIKOLA PETROV DATE 01 May 2025 - 31 Jul 2026\n" + + "Ulica 21 IBAN DE00000000000000000000\n\n" + + "DATE TYPE DESCRIPTION MONEY IN MONEY OUT BALANCE\n" + if got := tradeRepublicStatementName(header); got != "traderepublic_2025_05_01_2026_07_31" { + t.Errorf("name = %q, want traderepublic_2025_05_01_2026_07_31", got) + } + table := "DATE TYPE DESCRIPTION MONEY IN MONEY OUT BALANCE\n" + + "01 Jul 2026 Interest Interest payment €1.17 €671.58\n" + if got := tradeRepublicStatementName(table); got != "" { + t.Errorf("name without a period = %q, want none", got) + } +}