feat(search): index embedded PDF attachment text (#42)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go stable) (pull_request) Successful in 25s
Tests / Test (Go 1.25) (pull_request) Successful in 28s
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go stable) (pull_request) Successful in 25s
Tests / Test (Go 1.25) (pull_request) Successful in 28s
This commit is contained in:
@@ -5,6 +5,7 @@
|
||||
// - pandoc — md↔docx
|
||||
// - ocrmypdf — OCR into a searchable PDF
|
||||
// - pdftotext — extract text layer
|
||||
// - pdfdetach — list/save embedded PDF attachments
|
||||
// - tesseract — OCR single images when ocrmypdf is unsuitable
|
||||
// - ghostscript (gs) — PDF rewrite/optimize via PostScript (pdfwrite)
|
||||
package docpipe
|
||||
@@ -26,6 +27,7 @@ type Tools struct {
|
||||
Pandoc string
|
||||
OCRMyPDF string
|
||||
PDFToText string
|
||||
PDFDetach string
|
||||
Tesseract string
|
||||
Ghostscript string
|
||||
}
|
||||
@@ -44,6 +46,7 @@ func LookPath() Tools {
|
||||
Pandoc: find("pandoc"),
|
||||
OCRMyPDF: find("ocrmypdf"),
|
||||
PDFToText: find("pdftotext"),
|
||||
PDFDetach: find("pdfdetach"),
|
||||
Tesseract: find("tesseract"),
|
||||
Ghostscript: find("gs", "ghostscript"),
|
||||
}
|
||||
|
||||
@@ -0,0 +1,231 @@
|
||||
package docpipe
|
||||
|
||||
// Embedded PDF attachments (F6 #42). Digitised invoices often carry the
|
||||
// original scan as a PDF attachment; the searchable body may hold only a
|
||||
// summary. pdfdetach (poppler) lists/saves them; each saved attachment is run
|
||||
// through the normal docpipe extraction (pdftotext/OCR).
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/xml"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// PDFAttachment is one embedded file in a PDF.
|
||||
type PDFAttachment struct {
|
||||
Index int // 1-based number as `pdfdetach -list` reports it
|
||||
Name string // embedded file name
|
||||
}
|
||||
|
||||
// AttachmentText is the extracted text of one embedded attachment.
|
||||
type AttachmentText struct {
|
||||
Name string
|
||||
Text string
|
||||
}
|
||||
|
||||
// parseAttachmentList parses `pdfdetach -list` output. The first line is a
|
||||
// count ("N embedded files"); every following line is "<index>: <name>".
|
||||
// Pure, so it is unit-tested.
|
||||
func parseAttachmentList(out string) []PDFAttachment {
|
||||
var atts []PDFAttachment
|
||||
for _, line := range strings.Split(out, "\n") {
|
||||
line = strings.TrimSpace(line)
|
||||
if line == "" {
|
||||
continue
|
||||
}
|
||||
colon := strings.Index(line, ":")
|
||||
if colon <= 0 {
|
||||
continue
|
||||
}
|
||||
n, err := strconv.Atoi(strings.TrimSpace(line[:colon]))
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
name := strings.TrimSpace(line[colon+1:])
|
||||
if name == "" {
|
||||
continue
|
||||
}
|
||||
atts = append(atts, PDFAttachment{Index: n, Name: name})
|
||||
}
|
||||
return atts
|
||||
}
|
||||
|
||||
// safeAttachmentName strips directories and leading dots so a hostile
|
||||
// attachment name cannot escape the extraction directory.
|
||||
func safeAttachmentName(name string) string {
|
||||
name = strings.ReplaceAll(strings.TrimSpace(name), "\\", "/")
|
||||
name = filepath.Base(name)
|
||||
name = strings.TrimLeft(name, ".")
|
||||
if name == "" || name == "." || name == "/" {
|
||||
return ""
|
||||
}
|
||||
return name
|
||||
}
|
||||
|
||||
// JoinWithAttachments appends attachment text to the document body, each
|
||||
// section preceded by an "[attachment: <name>]" marker so a search hit shows
|
||||
// its source. Empty attachments are skipped. Pure, so it is unit-tested.
|
||||
func JoinWithAttachments(body string, atts []AttachmentText) string {
|
||||
var b strings.Builder
|
||||
b.WriteString(strings.TrimRight(body, "\n"))
|
||||
for _, a := range atts {
|
||||
text := strings.TrimSpace(a.Text)
|
||||
if text == "" {
|
||||
continue
|
||||
}
|
||||
b.WriteString("\n\n[attachment: ")
|
||||
b.WriteString(a.Name)
|
||||
b.WriteString("]\n\n")
|
||||
b.WriteString(text)
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// ListAttachments returns the embedded files of a PDF. A PDF without
|
||||
// attachments yields an empty slice and no error.
|
||||
func (t Tools) ListAttachments(pdfPath string) ([]PDFAttachment, error) {
|
||||
if t.PDFDetach == "" {
|
||||
return nil, fmt.Errorf("pdfdetach not found on PATH")
|
||||
}
|
||||
cmd := exec.Command(t.PDFDetach, "-list", pdfPath)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("pdfdetach -list %s: %w (%s)", filepath.Base(pdfPath), err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return parseAttachmentList(string(out)), nil
|
||||
}
|
||||
|
||||
// SaveAttachment writes the n-th embedded file (1-based) to outPath.
|
||||
func (t Tools) SaveAttachment(pdfPath string, index int, outPath string) error {
|
||||
if t.PDFDetach == "" {
|
||||
return fmt.Errorf("pdfdetach not found on PATH")
|
||||
}
|
||||
if strings.TrimSpace(outPath) == "" {
|
||||
return fmt.Errorf("output path required")
|
||||
}
|
||||
if err := EnsureDir(outPath); err != nil {
|
||||
return err
|
||||
}
|
||||
cmd := exec.Command(t.PDFDetach, "-save", strconv.Itoa(index), "-o", outPath, pdfPath)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return fmt.Errorf("pdfdetach -save %d: %w (%s)", index, err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ToMarkdownWithAttachments extracts the file as ToMarkdown does, then — for
|
||||
// PDFs — appends the text of every embedded attachment under an
|
||||
// "[attachment: <name>]" marker. Attachment failures are non-fatal: the body
|
||||
// is returned unchanged.
|
||||
func (t Tools) ToMarkdownWithAttachments(path, workDir, lang string, minChars int) (string, error) {
|
||||
res, err := t.ToMarkdown(path, workDir, lang, minChars)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
if Ext(path) != ".pdf" {
|
||||
return res.Markdown, nil
|
||||
}
|
||||
atts, err := t.attachmentTexts(path, workDir, lang, minChars)
|
||||
if err != nil {
|
||||
return res.Markdown, nil
|
||||
}
|
||||
return JoinWithAttachments(res.Markdown, atts), nil
|
||||
}
|
||||
|
||||
// attachmentTexts saves and extracts every embedded attachment, skipping the
|
||||
// ones that cannot be read. It returns an error only when the attachment list
|
||||
// itself cannot be obtained.
|
||||
func (t Tools) attachmentTexts(pdfPath, workDir, lang string, minChars int) ([]AttachmentText, error) {
|
||||
list, err := t.ListAttachments(pdfPath)
|
||||
if err != nil || len(list) == 0 {
|
||||
return nil, err
|
||||
}
|
||||
if workDir == "" {
|
||||
workDir = os.TempDir()
|
||||
}
|
||||
dir := filepath.Join(workDir, "att-"+trimExt(filepath.Base(pdfPath)))
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
out := make([]AttachmentText, 0, len(list))
|
||||
for _, a := range list {
|
||||
name := safeAttachmentName(a.Name)
|
||||
if name == "" {
|
||||
continue
|
||||
}
|
||||
saved := filepath.Join(dir, fmt.Sprintf("%d-%s", a.Index, name))
|
||||
if err := t.SaveAttachment(pdfPath, a.Index, saved); err != nil {
|
||||
continue
|
||||
}
|
||||
text, err := t.attachmentMarkdown(saved, dir, lang, minChars)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
out = append(out, AttachmentText{Name: a.Name, Text: text})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// attachmentMarkdown extracts a saved attachment with the regular pipeline.
|
||||
// Structured attachments that docpipe does not convert (e-invoice XML,
|
||||
// CuraSoft JSON, CSV/HTML) fall back to their text content, so the embedded
|
||||
// original is still searchable. Other unreadable formats return an error and
|
||||
// the caller skips them.
|
||||
func (t Tools) attachmentMarkdown(path, workDir, lang string, minChars int) (string, error) {
|
||||
if res, err := t.ToMarkdown(path, workDir, lang, minChars); err == nil {
|
||||
return res.Markdown, nil
|
||||
}
|
||||
switch Ext(path) {
|
||||
case ".xml", ".html", ".htm":
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return xmlToText(raw), nil
|
||||
case ".json", ".csv":
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return string(raw), nil
|
||||
default:
|
||||
return "", fmt.Errorf("unsupported attachment type %q", Ext(path))
|
||||
}
|
||||
}
|
||||
|
||||
// xmlToText returns the character data of an XML/HTML document: element text
|
||||
// values with decoded entities, one per line. Used for invoice XML (EN 16931
|
||||
// CII / ZUGFeRD) and HTML attachments. Pure, so it is unit-tested.
|
||||
func xmlToText(raw []byte) string {
|
||||
dec := xml.NewDecoder(bytes.NewReader(raw))
|
||||
dec.Strict = false
|
||||
var b strings.Builder
|
||||
for {
|
||||
tok, err := dec.Token()
|
||||
if err != nil {
|
||||
break
|
||||
}
|
||||
cd, ok := tok.(xml.CharData)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
s := strings.TrimSpace(string(cd))
|
||||
if s == "" {
|
||||
continue
|
||||
}
|
||||
b.WriteString(s)
|
||||
b.WriteByte('\n')
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
@@ -0,0 +1,163 @@
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestParseAttachmentList(t *testing.T) {
|
||||
out := "2 embedded files\n1: original.pdf\n2: scan_001.png\n"
|
||||
got := parseAttachmentList(out)
|
||||
want := []PDFAttachment{{Index: 1, Name: "original.pdf"}, {Index: 2, Name: "scan_001.png"}}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("got %+v, want %+v", got, want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("att[%d] = %+v, want %+v", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseAttachmentListEmptyAndMalformed(t *testing.T) {
|
||||
for _, in := range []string{"", "0 embedded files\n", "garbage\n\n \n"} {
|
||||
if got := parseAttachmentList(in); len(got) != 0 {
|
||||
t.Errorf("parseAttachmentList(%q) = %+v, want empty", in, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSafeAttachmentName(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"note.txt": "note.txt",
|
||||
"../../evil.pdf": "evil.pdf",
|
||||
`..\..\evil.pdf`: "evil.pdf",
|
||||
"/abs/scan_001.pdf": "scan_001.pdf",
|
||||
".hidden": "hidden",
|
||||
" spaced name.txt ": "spaced name.txt",
|
||||
"..": "",
|
||||
"": "",
|
||||
}
|
||||
for in, want := range cases {
|
||||
if got := safeAttachmentName(in); got != want {
|
||||
t.Errorf("safeAttachmentName(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestJoinWithAttachments(t *testing.T) {
|
||||
body := "# scan.pdf\n\nbody token\n"
|
||||
atts := []AttachmentText{
|
||||
{Name: "original.pdf", Text: " original token "},
|
||||
{Name: "empty.txt", Text: " "},
|
||||
}
|
||||
got := JoinWithAttachments(body, atts)
|
||||
if !strings.Contains(got, "body token") {
|
||||
t.Errorf("body text lost: %q", got)
|
||||
}
|
||||
if !strings.Contains(got, "[attachment: original.pdf]") {
|
||||
t.Errorf("marker missing: %q", got)
|
||||
}
|
||||
if !strings.Contains(got, "original token") {
|
||||
t.Errorf("attachment text missing: %q", got)
|
||||
}
|
||||
if strings.Contains(got, "empty.txt") {
|
||||
t.Errorf("empty attachment must be skipped: %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestJoinWithAttachmentsNoAttachments(t *testing.T) {
|
||||
got := JoinWithAttachments("# a.pdf\n\ntext\n\n", nil)
|
||||
if got != "# a.pdf\n\ntext" {
|
||||
t.Errorf("got %q, want trimmed body only", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestListAttachmentsWithoutTool(t *testing.T) {
|
||||
if _, err := (Tools{}).ListAttachments("x.pdf"); err == nil || !strings.Contains(err.Error(), "pdfdetach") {
|
||||
t.Fatalf("want pdfdetach error, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestToMarkdownWithAttachmentsFixture exercises the real pdfdetach + pdftotext
|
||||
// pipeline on testdata/pdf-with-attachment.pdf (body token + embedded
|
||||
// goo-note.txt). Skips when poppler is not installed.
|
||||
func TestToMarkdownWithAttachmentsFixture(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.PDFDetach == "" || tools.PDFToText == "" {
|
||||
t.Skip("pdfdetach/pdftotext not on PATH — skipping attachment extraction test")
|
||||
}
|
||||
fixture := filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf")
|
||||
got, err := tools.ToMarkdownWithAttachments(fixture, t.TempDir(), "eng", 1)
|
||||
if err != nil {
|
||||
t.Fatalf("ToMarkdownWithAttachments: %v", err)
|
||||
}
|
||||
for _, want := range []string{"goobodytoken", "[attachment: goo-note.txt]", "gooattachmenttoken"} {
|
||||
if !strings.Contains(got, want) {
|
||||
t.Errorf("result missing %q:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestXMLToText(t *testing.T) {
|
||||
raw := []byte(`<?xml version="1.0" encoding="UTF-8"?>
|
||||
<rsm:CrossIndustryInvoice><rsm:ExchangedDocument>
|
||||
<ram:ID>S1063</ram:ID></rsm:ExchangedDocument>
|
||||
<ram:Name>Edelweiss & Co</ram:Name><ram:GrandTotalAmount>42.00</ram:GrandTotalAmount>
|
||||
</rsm:CrossIndustryInvoice>`)
|
||||
got := xmlToText(raw)
|
||||
for _, want := range []string{"S1063", "Edelweiss & Co", "42.00"} {
|
||||
if !strings.Contains(got, want) {
|
||||
t.Errorf("xmlToText missing %q:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
if strings.ContainsAny(got, "<>") {
|
||||
t.Errorf("xmlToText left markup: %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAttachmentMarkdownFallback verifies structured attachments that docpipe
|
||||
// cannot convert are still reduced to searchable text, and unknown binary
|
||||
// formats error (so the caller skips them).
|
||||
func TestAttachmentMarkdownFallback(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
xmlPath := filepath.Join(dir, "factur-x.xml")
|
||||
if err := os.WriteFile(xmlPath, []byte(`<Invoice><Number>S1063</Number></Invoice>`), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := (Tools{}).attachmentMarkdown(xmlPath, dir, "", 0)
|
||||
if err != nil {
|
||||
t.Fatalf("attachmentMarkdown(xml): %v", err)
|
||||
}
|
||||
if !strings.Contains(got, "S1063") {
|
||||
t.Errorf("xml attachment text = %q, want S1063", got)
|
||||
}
|
||||
|
||||
binPath := filepath.Join(dir, "data.bin")
|
||||
if err := os.WriteFile(binPath, []byte{0, 1, 2, 3}, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := (Tools{}).attachmentMarkdown(binPath, dir, "", 0); err == nil {
|
||||
t.Error("unsupported attachment: want error, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
// TestToMarkdownWithAttachmentsPlainPDF ensures a PDF without attachments
|
||||
// returns just the body (pdfdetach prints "0 embedded files").
|
||||
func TestToMarkdownWithAttachmentsPlainPDF(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.PDFDetach == "" || tools.PDFToText == "" {
|
||||
t.Skip("pdfdetach/pdftotext not on PATH")
|
||||
}
|
||||
// The fixture itself is a PDF with one attachment; strip it by extracting
|
||||
// the body only through ToMarkdown and compare JoinWithAttachments(nil).
|
||||
res, err := tools.ToMarkdown(filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf"), t.TempDir(), "eng", 1)
|
||||
if err != nil {
|
||||
t.Fatalf("ToMarkdown: %v", err)
|
||||
}
|
||||
if strings.Contains(res.Markdown, "gooattachmenttoken") {
|
||||
t.Fatalf("body must not contain attachment text: %q", res.Markdown)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user