package parser import ( "bytes" "context" "fmt" "os/exec" "strings" "time" ) // pdfToTextTimeout bounds extraction so a malformed PDF cannot wedge an import. const pdfToTextTimeout = 2 * time.Minute // pdfToText renders a PDF as text with the original column positions // preserved, which is what the statement parsers key off. // // This shells out to poppler's pdftotext rather than decoding the PDF in Go: // the layout reconstruction it does is the whole reason the column-based // parsers work, and no Go library matches it. func pdfToText(path string) (string, error) { ctx, cancel := context.WithTimeout(context.Background(), pdfToTextTimeout) defer cancel() cmd := exec.CommandContext(ctx, "pdftotext", "-layout", path, "-") var stdout, stderr bytes.Buffer cmd.Stdout = &stdout cmd.Stderr = &stderr if err := cmd.Run(); err != nil { if ctx.Err() == context.DeadlineExceeded { return "", fmt.Errorf("pdftotext timed out after %s on %s", pdfToTextTimeout, path) } if errors := strings.TrimSpace(stderr.String()); errors != "" { return "", fmt.Errorf("pdftotext %s: %w: %s", path, err, errors) } if _, lookErr := exec.LookPath("pdftotext"); lookErr != nil { return "", fmt.Errorf("pdftotext is not installed (it ships with poppler-utils): %w", lookErr) } return "", fmt.Errorf("pdftotext %s: %w", path, err) } return stdout.String(), nil } // pages splits pdftotext output on form feeds. func pages(text string) []string { return strings.Split(text, "\f") } // indent counts the leading spaces of a line. func indent(line string) int { return len(line) - len(strings.TrimLeft(line, " ")) }