feat(search): index embedded PDF attachment text (#42)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go stable) (pull_request) Successful in 25s
Tests / Test (Go 1.25) (pull_request) Successful in 28s

This commit is contained in:
2026-09-16 17:34:04 +00:00
parent ce4778bdf1
commit 3cc288d281
8 changed files with 616 additions and 11 deletions
+3
View File
@@ -5,6 +5,7 @@
// - pandoc — md↔docx
// - ocrmypdf — OCR into a searchable PDF
// - pdftotext — extract text layer
// - pdfdetach — list/save embedded PDF attachments
// - tesseract — OCR single images when ocrmypdf is unsuitable
// - ghostscript (gs) — PDF rewrite/optimize via PostScript (pdfwrite)
package docpipe
@@ -26,6 +27,7 @@ type Tools struct {
Pandoc string
OCRMyPDF string
PDFToText string
PDFDetach string
Tesseract string
Ghostscript string
}
@@ -44,6 +46,7 @@ func LookPath() Tools {
Pandoc: find("pandoc"),
OCRMyPDF: find("ocrmypdf"),
PDFToText: find("pdftotext"),
PDFDetach: find("pdfdetach"),
Tesseract: find("tesseract"),
Ghostscript: find("gs", "ghostscript"),
}
+231
View File
@@ -0,0 +1,231 @@
package docpipe
// Embedded PDF attachments (F6 #42). Digitised invoices often carry the
// original scan as a PDF attachment; the searchable body may hold only a
// summary. pdfdetach (poppler) lists/saves them; each saved attachment is run
// through the normal docpipe extraction (pdftotext/OCR).
import (
"bytes"
"encoding/xml"
"fmt"
"os"
"os/exec"
"path/filepath"
"strconv"
"strings"
)
// PDFAttachment is one embedded file in a PDF.
type PDFAttachment struct {
Index int // 1-based number as `pdfdetach -list` reports it
Name string // embedded file name
}
// AttachmentText is the extracted text of one embedded attachment.
type AttachmentText struct {
Name string
Text string
}
// parseAttachmentList parses `pdfdetach -list` output. The first line is a
// count ("N embedded files"); every following line is "<index>: <name>".
// Pure, so it is unit-tested.
func parseAttachmentList(out string) []PDFAttachment {
var atts []PDFAttachment
for _, line := range strings.Split(out, "\n") {
line = strings.TrimSpace(line)
if line == "" {
continue
}
colon := strings.Index(line, ":")
if colon <= 0 {
continue
}
n, err := strconv.Atoi(strings.TrimSpace(line[:colon]))
if err != nil {
continue
}
name := strings.TrimSpace(line[colon+1:])
if name == "" {
continue
}
atts = append(atts, PDFAttachment{Index: n, Name: name})
}
return atts
}
// safeAttachmentName strips directories and leading dots so a hostile
// attachment name cannot escape the extraction directory.
func safeAttachmentName(name string) string {
name = strings.ReplaceAll(strings.TrimSpace(name), "\\", "/")
name = filepath.Base(name)
name = strings.TrimLeft(name, ".")
if name == "" || name == "." || name == "/" {
return ""
}
return name
}
// JoinWithAttachments appends attachment text to the document body, each
// section preceded by an "[attachment: <name>]" marker so a search hit shows
// its source. Empty attachments are skipped. Pure, so it is unit-tested.
func JoinWithAttachments(body string, atts []AttachmentText) string {
var b strings.Builder
b.WriteString(strings.TrimRight(body, "\n"))
for _, a := range atts {
text := strings.TrimSpace(a.Text)
if text == "" {
continue
}
b.WriteString("\n\n[attachment: ")
b.WriteString(a.Name)
b.WriteString("]\n\n")
b.WriteString(text)
}
return b.String()
}
// ListAttachments returns the embedded files of a PDF. A PDF without
// attachments yields an empty slice and no error.
func (t Tools) ListAttachments(pdfPath string) ([]PDFAttachment, error) {
if t.PDFDetach == "" {
return nil, fmt.Errorf("pdfdetach not found on PATH")
}
cmd := exec.Command(t.PDFDetach, "-list", pdfPath)
var stderr bytes.Buffer
cmd.Stderr = &stderr
out, err := cmd.Output()
if err != nil {
return nil, fmt.Errorf("pdfdetach -list %s: %w (%s)", filepath.Base(pdfPath), err, strings.TrimSpace(stderr.String()))
}
return parseAttachmentList(string(out)), nil
}
// SaveAttachment writes the n-th embedded file (1-based) to outPath.
func (t Tools) SaveAttachment(pdfPath string, index int, outPath string) error {
if t.PDFDetach == "" {
return fmt.Errorf("pdfdetach not found on PATH")
}
if strings.TrimSpace(outPath) == "" {
return fmt.Errorf("output path required")
}
if err := EnsureDir(outPath); err != nil {
return err
}
cmd := exec.Command(t.PDFDetach, "-save", strconv.Itoa(index), "-o", outPath, pdfPath)
var stderr bytes.Buffer
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
return fmt.Errorf("pdfdetach -save %d: %w (%s)", index, err, strings.TrimSpace(stderr.String()))
}
return nil
}
// ToMarkdownWithAttachments extracts the file as ToMarkdown does, then — for
// PDFs — appends the text of every embedded attachment under an
// "[attachment: <name>]" marker. Attachment failures are non-fatal: the body
// is returned unchanged.
func (t Tools) ToMarkdownWithAttachments(path, workDir, lang string, minChars int) (string, error) {
res, err := t.ToMarkdown(path, workDir, lang, minChars)
if err != nil {
return "", err
}
if Ext(path) != ".pdf" {
return res.Markdown, nil
}
atts, err := t.attachmentTexts(path, workDir, lang, minChars)
if err != nil {
return res.Markdown, nil
}
return JoinWithAttachments(res.Markdown, atts), nil
}
// attachmentTexts saves and extracts every embedded attachment, skipping the
// ones that cannot be read. It returns an error only when the attachment list
// itself cannot be obtained.
func (t Tools) attachmentTexts(pdfPath, workDir, lang string, minChars int) ([]AttachmentText, error) {
list, err := t.ListAttachments(pdfPath)
if err != nil || len(list) == 0 {
return nil, err
}
if workDir == "" {
workDir = os.TempDir()
}
dir := filepath.Join(workDir, "att-"+trimExt(filepath.Base(pdfPath)))
if err := os.MkdirAll(dir, 0o755); err != nil {
return nil, err
}
defer os.RemoveAll(dir)
out := make([]AttachmentText, 0, len(list))
for _, a := range list {
name := safeAttachmentName(a.Name)
if name == "" {
continue
}
saved := filepath.Join(dir, fmt.Sprintf("%d-%s", a.Index, name))
if err := t.SaveAttachment(pdfPath, a.Index, saved); err != nil {
continue
}
text, err := t.attachmentMarkdown(saved, dir, lang, minChars)
if err != nil {
continue
}
out = append(out, AttachmentText{Name: a.Name, Text: text})
}
return out, nil
}
// attachmentMarkdown extracts a saved attachment with the regular pipeline.
// Structured attachments that docpipe does not convert (e-invoice XML,
// CuraSoft JSON, CSV/HTML) fall back to their text content, so the embedded
// original is still searchable. Other unreadable formats return an error and
// the caller skips them.
func (t Tools) attachmentMarkdown(path, workDir, lang string, minChars int) (string, error) {
if res, err := t.ToMarkdown(path, workDir, lang, minChars); err == nil {
return res.Markdown, nil
}
switch Ext(path) {
case ".xml", ".html", ".htm":
raw, err := os.ReadFile(path)
if err != nil {
return "", err
}
return xmlToText(raw), nil
case ".json", ".csv":
raw, err := os.ReadFile(path)
if err != nil {
return "", err
}
return string(raw), nil
default:
return "", fmt.Errorf("unsupported attachment type %q", Ext(path))
}
}
// xmlToText returns the character data of an XML/HTML document: element text
// values with decoded entities, one per line. Used for invoice XML (EN 16931
// CII / ZUGFeRD) and HTML attachments. Pure, so it is unit-tested.
func xmlToText(raw []byte) string {
dec := xml.NewDecoder(bytes.NewReader(raw))
dec.Strict = false
var b strings.Builder
for {
tok, err := dec.Token()
if err != nil {
break
}
cd, ok := tok.(xml.CharData)
if !ok {
continue
}
s := strings.TrimSpace(string(cd))
if s == "" {
continue
}
b.WriteString(s)
b.WriteByte('\n')
}
return b.String()
}
+163
View File
@@ -0,0 +1,163 @@
package docpipe
import (
"os"
"path/filepath"
"strings"
"testing"
)
func TestParseAttachmentList(t *testing.T) {
out := "2 embedded files\n1: original.pdf\n2: scan_001.png\n"
got := parseAttachmentList(out)
want := []PDFAttachment{{Index: 1, Name: "original.pdf"}, {Index: 2, Name: "scan_001.png"}}
if len(got) != len(want) {
t.Fatalf("got %+v, want %+v", got, want)
}
for i := range want {
if got[i] != want[i] {
t.Errorf("att[%d] = %+v, want %+v", i, got[i], want[i])
}
}
}
func TestParseAttachmentListEmptyAndMalformed(t *testing.T) {
for _, in := range []string{"", "0 embedded files\n", "garbage\n\n \n"} {
if got := parseAttachmentList(in); len(got) != 0 {
t.Errorf("parseAttachmentList(%q) = %+v, want empty", in, got)
}
}
}
func TestSafeAttachmentName(t *testing.T) {
cases := map[string]string{
"note.txt": "note.txt",
"../../evil.pdf": "evil.pdf",
`..\..\evil.pdf`: "evil.pdf",
"/abs/scan_001.pdf": "scan_001.pdf",
".hidden": "hidden",
" spaced name.txt ": "spaced name.txt",
"..": "",
"": "",
}
for in, want := range cases {
if got := safeAttachmentName(in); got != want {
t.Errorf("safeAttachmentName(%q) = %q, want %q", in, got, want)
}
}
}
func TestJoinWithAttachments(t *testing.T) {
body := "# scan.pdf\n\nbody token\n"
atts := []AttachmentText{
{Name: "original.pdf", Text: " original token "},
{Name: "empty.txt", Text: " "},
}
got := JoinWithAttachments(body, atts)
if !strings.Contains(got, "body token") {
t.Errorf("body text lost: %q", got)
}
if !strings.Contains(got, "[attachment: original.pdf]") {
t.Errorf("marker missing: %q", got)
}
if !strings.Contains(got, "original token") {
t.Errorf("attachment text missing: %q", got)
}
if strings.Contains(got, "empty.txt") {
t.Errorf("empty attachment must be skipped: %q", got)
}
}
func TestJoinWithAttachmentsNoAttachments(t *testing.T) {
got := JoinWithAttachments("# a.pdf\n\ntext\n\n", nil)
if got != "# a.pdf\n\ntext" {
t.Errorf("got %q, want trimmed body only", got)
}
}
func TestListAttachmentsWithoutTool(t *testing.T) {
if _, err := (Tools{}).ListAttachments("x.pdf"); err == nil || !strings.Contains(err.Error(), "pdfdetach") {
t.Fatalf("want pdfdetach error, got %v", err)
}
}
// TestToMarkdownWithAttachmentsFixture exercises the real pdfdetach + pdftotext
// pipeline on testdata/pdf-with-attachment.pdf (body token + embedded
// goo-note.txt). Skips when poppler is not installed.
func TestToMarkdownWithAttachmentsFixture(t *testing.T) {
tools := LookPath()
if tools.PDFDetach == "" || tools.PDFToText == "" {
t.Skip("pdfdetach/pdftotext not on PATH — skipping attachment extraction test")
}
fixture := filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf")
got, err := tools.ToMarkdownWithAttachments(fixture, t.TempDir(), "eng", 1)
if err != nil {
t.Fatalf("ToMarkdownWithAttachments: %v", err)
}
for _, want := range []string{"goobodytoken", "[attachment: goo-note.txt]", "gooattachmenttoken"} {
if !strings.Contains(got, want) {
t.Errorf("result missing %q:\n%s", want, got)
}
}
}
func TestXMLToText(t *testing.T) {
raw := []byte(`<?xml version="1.0" encoding="UTF-8"?>
<rsm:CrossIndustryInvoice><rsm:ExchangedDocument>
<ram:ID>S1063</ram:ID></rsm:ExchangedDocument>
<ram:Name>Edelweiss &amp; Co</ram:Name><ram:GrandTotalAmount>42.00</ram:GrandTotalAmount>
</rsm:CrossIndustryInvoice>`)
got := xmlToText(raw)
for _, want := range []string{"S1063", "Edelweiss & Co", "42.00"} {
if !strings.Contains(got, want) {
t.Errorf("xmlToText missing %q:\n%s", want, got)
}
}
if strings.ContainsAny(got, "<>") {
t.Errorf("xmlToText left markup: %q", got)
}
}
// TestAttachmentMarkdownFallback verifies structured attachments that docpipe
// cannot convert are still reduced to searchable text, and unknown binary
// formats error (so the caller skips them).
func TestAttachmentMarkdownFallback(t *testing.T) {
dir := t.TempDir()
xmlPath := filepath.Join(dir, "factur-x.xml")
if err := os.WriteFile(xmlPath, []byte(`<Invoice><Number>S1063</Number></Invoice>`), 0o644); err != nil {
t.Fatal(err)
}
got, err := (Tools{}).attachmentMarkdown(xmlPath, dir, "", 0)
if err != nil {
t.Fatalf("attachmentMarkdown(xml): %v", err)
}
if !strings.Contains(got, "S1063") {
t.Errorf("xml attachment text = %q, want S1063", got)
}
binPath := filepath.Join(dir, "data.bin")
if err := os.WriteFile(binPath, []byte{0, 1, 2, 3}, 0o644); err != nil {
t.Fatal(err)
}
if _, err := (Tools{}).attachmentMarkdown(binPath, dir, "", 0); err == nil {
t.Error("unsupported attachment: want error, got nil")
}
}
// TestToMarkdownWithAttachmentsPlainPDF ensures a PDF without attachments
// returns just the body (pdfdetach prints "0 embedded files").
func TestToMarkdownWithAttachmentsPlainPDF(t *testing.T) {
tools := LookPath()
if tools.PDFDetach == "" || tools.PDFToText == "" {
t.Skip("pdfdetach/pdftotext not on PATH")
}
// The fixture itself is a PDF with one attachment; strip it by extracting
// the body only through ToMarkdown and compare JoinWithAttachments(nil).
res, err := tools.ToMarkdown(filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf"), t.TempDir(), "eng", 1)
if err != nil {
t.Fatalf("ToMarkdown: %v", err)
}
if strings.Contains(res.Markdown, "gooattachmenttoken") {
t.Fatalf("body must not contain attachment text: %q", res.Markdown)
}
}