Files
money/internal/parser/pdftext.go
T
nikolaandClaude Opus 5.5 5847245638 Bundle a static pdftotext into the binary
The PDF parsers shell out to pdftotext, so every machine running money needed
poppler-utils installed. scripts/build-bundled.sh now builds a deployable
binary that carries its own: pdftotext is compiled in a container from a
checksum-pinned poppler release as a fully static musl executable, then
embedded with `go build -tags bundled`. The result is one file that runs on
any Linux of that architecture with nothing installed alongside it.

It is still the real pdftotext, run as a subprocess. Linking poppler through
cgo would have cost the pure-Go build, and its C++ text API is not guaranteed
to space columns the way pdftotext -layout does, which is what the parsers
were tuned on. Only what text extraction needs is compiled in -- no
fontconfig, cairo or image codecs -- and its output is byte-identical to a
full distro build on the same PDF.

At runtime the embedded copy is written to the user cache directory, not
/tmp, which servers often mount noexec. It is named by content hash, so a
newer build never runs an older copy, and verified before reuse, so a write cut
short by a killed process is replaced rather than trusted. `money config` says
which pdftotext is in use.

The tag is opt-in: plain go build and go test never need the 5 MB executable,
which is gitignored rather than committed. Building with the tag for anything
but linux/amd64 or linux/arm64 fails with a message saying so.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-02 18:02:46 +02:00

66 lines
2.1 KiB
Go

package parser
import (
"bytes"
"context"
"fmt"
"os/exec"
"strings"
"time"
)
// pdfToTextTimeout bounds extraction so a malformed PDF cannot wedge an import.
const pdfToTextTimeout = 2 * time.Minute
// pdfToText renders a PDF as text with the original column positions
// preserved, which is what the statement parsers key off.
//
// This shells out to poppler's pdftotext rather than decoding the PDF in Go:
// the layout reconstruction it does is the whole reason the column-based
// parsers work, and no Go library matches it. A build made with -tags bundled
// carries its own copy, so a server needs nothing installed.
func pdfToText(path string) (string, error) {
bin, err := pdftotextCommand()
if err != nil {
return "", err
}
ctx, cancel := context.WithTimeout(context.Background(), pdfToTextTimeout)
defer cancel()
cmd := exec.CommandContext(ctx, bin, "-layout", path, "-")
var stdout, stderr bytes.Buffer
cmd.Stdout = &stdout
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
if ctx.Err() == context.DeadlineExceeded {
return "", fmt.Errorf("pdftotext timed out after %s on %s", pdfToTextTimeout, path)
}
if errors := strings.TrimSpace(stderr.String()); errors != "" {
return "", fmt.Errorf("pdftotext %s: %w: %s", path, err, errors)
}
if bin != "pdftotext" {
// The copy was installed, so failing to start it is about where it
// was put: a cache directory on a noexec mount is the usual cause.
return "", fmt.Errorf("bundled pdftotext at %s did not run (is that directory mounted noexec? "+
"point XDG_CACHE_HOME somewhere else): %w", bin, err)
}
if _, lookErr := exec.LookPath("pdftotext"); lookErr != nil {
return "", fmt.Errorf("pdftotext is not installed (it ships with poppler-utils): %w", lookErr)
}
return "", fmt.Errorf("pdftotext %s: %w", path, err)
}
return stdout.String(), nil
}
// pages splits pdftotext output on form feeds.
func pages(text string) []string {
return strings.Split(text, "\f")
}
// indent counts the leading spaces of a line.
func indent(line string) int {
return len(line) - len(strings.TrimLeft(line, " "))
}