From 3cc288d281b22b06b734db33870c023b424bb944 Mon Sep 17 00:00:00 2001 From: Andriy Oblivantsev Date: Wed, 16 Sep 2026 17:34:04 +0000 Subject: [PATCH] feat(search): index embedded PDF attachment text (#42) --- README.md | 8 +- docs/elasticsearch.md | 28 +++- file_es_text_integration_test.go | 66 +++++++++ file_text_index.go | 14 +- internal/docpipe/docpipe.go | 3 + internal/docpipe/pdfattach.go | 231 +++++++++++++++++++++++++++++ internal/docpipe/pdfattach_test.go | 163 ++++++++++++++++++++ testdata/pdf-with-attachment.pdf | 114 ++++++++++++++ 8 files changed, 616 insertions(+), 11 deletions(-) create mode 100644 internal/docpipe/pdfattach.go create mode 100644 internal/docpipe/pdfattach_test.go create mode 100644 testdata/pdf-with-attachment.pdf diff --git a/README.md b/README.md index 6c12250..bd97800 100644 --- a/README.md +++ b/README.md @@ -701,9 +701,11 @@ Requires `ONLYOFFICE_ES_URL` (plus optional `ONLYOFFICE_ES_INDEX`, The OnlyOffice index covers Office formats only, so PDFs (`S1019`-style invoice numbers) are not searchable by content. `oo index` extracts PDF text with -`internal/docpipe` (pdftotext, OCR for scans) into a separate index -(`ONLYOFFICE_ES_TEXT_INDEX`, default `oo_docs_text`); the OnlyOffice server and -its index are **not** modified. Then search it with `--backend own`. +`internal/docpipe` (pdftotext, OCR for scans) — including the text of embedded +PDF attachments (`pdfdetach`: `.md`, `.xml`, covers the original/scan and +ZUGFeRD e-invoice XML) — into a separate index (`ONLYOFFICE_ES_TEXT_INDEX`, +default `oo_docs_text`); the OnlyOffice server and its index are **not** +modified. Then search it with `--backend own`. ```bash oo index folder 634 --recursive --exts pdf # populate (idempotent upsert) diff --git a/docs/elasticsearch.md b/docs/elasticsearch.md index 35fd1a5..a918b67 100644 --- a/docs/elasticsearch.md +++ b/docs/elasticsearch.md @@ -174,6 +174,22 @@ OnlyOffice PDF лежит только по имени. - CLI: `oo index folder|files` наполняет индекс; `oo search --backend own` ищет по нему. +### Встроенные вложения PDF + +Оцифрованные PDF несут вложения (`.md` — текст/таблицы скана, +`.yaml`/`.json` — метаданные, `.xml` — EN 16931 CII eRechnung, +`factur-x.xml` у ZUGFeRD; см. `office-assistant/docs/reference/document-metadata.md`). +`TextIndexer` обходит их: `pdfdetach -list` перечисляет, `-save` сохраняет, +каждое вложение проходит штатный `docpipe.ToMarkdown` (PDF/картинки → OCR, +`.md`/`.txt` — как есть). Форматы, которые docpipe не конвертирует +(`.xml`/`.html` — снимаются теги; `.json`/`.csv` — как текст), извлекаются +текстом; нечитаемые — пропускаются. + +Текст склеивается: тело, затем по секции на вложение с маркером +`[attachment: <имя>]` (функция `docpipe.JoinWithAttachments`). Индекс — тот же +`file_id`, upsert идемпотентен. Нет вложений или pdfdetach/формат нечитаем — +индексируется тело (без падения). + Поля `oo_docs_text`: | поле | тип | смысл | @@ -217,8 +233,12 @@ ONLYOFFICE_ES_URL=http://127.0.0.1:9200 \ ``` Интеграционный тест создаёт временный индекс, наполняет, ищет по контенту, -проверяет фильтры и удаление, затем удаляет индекс. Unit-тесты используют -fake-store/fake-extractor и не требуют pdftotext/OCR. +проверяет фильтры и удаление, затем удаляет индекс; +`TestIntegrationESTextIndexPDFAttachment` индексирует +`testdata/pdf-with-attachment.pdf` реальным конвейером (pdfdetach + pdftotext) +и ищет токен, лежащий только во вложении. Unit-тесты используют +fake-store/fake-extractor и не требуют pdftotext/OCR (парсер списка, склейка +`JoinWithAttachments`, снятие тегов `xmlToText` — чистые). ## Грабли @@ -229,3 +249,7 @@ fake-store/fake-extractor и не требуют pdftotext/OCR. - `folder` фильтруется как id папки, а не как путь. - Дубликаты (напр. `S1055.pdf` и `2026-08-20-S1055-…`) дадут несколько строк — это ожидаемо, дедуп — на стороне потребителя. +- Вложения: нужен `pdfdetach` (poppler); если его нет — индексируется только + тело. Вложенный PDF/картинка с плохим текстовым слоем проходит OCR, это + медленно. `.json`-метаданные (CuraSoft) индексируются как текст и могут + добавить шумовых токенов. diff --git a/file_es_text_integration_test.go b/file_es_text_integration_test.go index ed070e8..30d295e 100644 --- a/file_es_text_integration_test.go +++ b/file_es_text_integration_test.go @@ -9,6 +9,8 @@ import ( "strings" "testing" "time" + + "github.com/eslider/go-onlyoffice/internal/docpipe" ) // TestIntegrationESTextIndex verifies the own full-text index end to end @@ -90,3 +92,67 @@ func TestIntegrationESTextIndex(t *testing.T) { t.Errorf("after delete search returned %d hits, want 0", len(hits)) } } + +// TestIntegrationESTextIndexPDFAttachment indexes testdata/pdf-with-attachment.pdf +// through the real pipeline (TextIndexer + docpipe: pdfdetach + pdftotext) and +// verifies that text living only in the embedded attachment is searchable. +// +// Requires ONLYOFFICE_ES_URL plus poppler (pdfdetach/pdftotext). No OnlyOffice +// credentials are needed: a fixture FileStore serves the PDF bytes. +func TestIntegrationESTextIndexPDFAttachment(t *testing.T) { + esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")) + if esURL == "" { + t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test") + } + if docpipe.LookPath().PDFDetach == "" { + t.Skip("pdfdetach not on PATH — skipping PDF attachment integration test") + } + pdf, err := os.ReadFile("testdata/pdf-with-attachment.pdf") + if err != nil { + t.Fatalf("read fixture: %v", err) + } + + stamp := time.Now().UTC().Format("20060102150405") + idx, err := NewESTextIndex(ESTextConfig{URL: esURL, Index: "oo_docs_text_it_att_" + stamp}) + if err != nil { + t.Fatalf("NewESTextIndex: %v", err) + } + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute) + defer cancel() + t.Cleanup(func() { + cleanupCtx, done := context.WithTimeout(context.Background(), 30*time.Second) + defer done() + _, _, _ = idx.do(cleanupCtx, http.MethodDelete, "/"+idx.Index(), nil, "") + }) + if err := idx.Ensure(ctx); err != nil { + t.Fatalf("Ensure: %v", err) + } + + store := &textFakeStore{files: map[string][]byte{"9001": pdf}} + ix := NewTextIndexer(store, idx) + res, err := ix.IndexEntries(ctx, []Entry{{ID: "9001", Title: "scan.pdf", ParentID: "777", Kind: File}}, IndexOptions{MinChars: 1}) + if err != nil { + t.Fatalf("IndexEntries: %v", err) + } + if res.Indexed != 1 || res.Failed != 0 { + t.Fatalf("result = %+v, want one indexed doc", res) + } + + // Token appears only inside the embedded goo-note.txt attachment. + hits, err := idx.Search(ctx, SearchQuery{Text: "gooattachmenttoken"}) + if err != nil { + t.Fatalf("Search attachment token: %v", err) + } + if len(hits) != 1 || hits[0].ID != "9001" { + t.Fatalf("attachment-token hits = %+v, want doc 9001", hits) + } + if !strings.Contains(hits[0].Highlight, "gooattachmenttoken") { + t.Errorf("highlight = %q, want attachment token", hits[0].Highlight) + } + // Body text is indexed as before. + if hits, err := idx.Search(ctx, SearchQuery{Text: "goobodytoken"}); err != nil { + t.Fatalf("Search body token: %v", err) + } else if len(hits) != 1 { + t.Errorf("body-token hits = %d, want 1", len(hits)) + } +} diff --git a/file_text_index.go b/file_text_index.go index 005eacf..92b4818 100644 --- a/file_text_index.go +++ b/file_text_index.go @@ -3,9 +3,10 @@ package onlyoffice // Text extraction pipeline for the own full-text index (epic #34, F6 #42). // // TextIndexer downloads stored documents, extracts text through docpipe -// (pdftotext; OCR for scans) and writes the result to a TextIndex. It is the -// write side of ESTextIndex and never touches the OnlyOffice server's own ES -// index. +// (pdftotext; OCR for scans) and writes the result to a TextIndex. For PDFs it +// also indexes the text of embedded attachments (pdfdetach), so a scan filed +// as an attachment is searchable too. It is the write side of ESTextIndex and +// never touches the OnlyOffice server's own ES index. import ( "context" @@ -38,13 +39,14 @@ type TextExtractor interface { type docpipeExtractor struct{ tools docpipe.Tools } // Extract renders the file as Markdown, OCRing PDFs/images with a weak text -// layer first (docpipe.ToMarkdown). +// layer first and appending the text of embedded PDF attachments +// (docpipe.ToMarkdownWithAttachments). func (d docpipeExtractor) Extract(path, workDir, lang string, minChars int) (string, error) { - res, err := d.tools.ToMarkdown(path, workDir, lang, minChars) + text, err := d.tools.ToMarkdownWithAttachments(path, workDir, lang, minChars) if err != nil { return "", err } - return strings.TrimSpace(res.Markdown), nil + return strings.TrimSpace(text), nil } // IndexOptions controls a TextIndexer run. diff --git a/internal/docpipe/docpipe.go b/internal/docpipe/docpipe.go index c5038d6..ce2fafc 100644 --- a/internal/docpipe/docpipe.go +++ b/internal/docpipe/docpipe.go @@ -5,6 +5,7 @@ // - pandoc — md↔docx // - ocrmypdf — OCR into a searchable PDF // - pdftotext — extract text layer +// - pdfdetach — list/save embedded PDF attachments // - tesseract — OCR single images when ocrmypdf is unsuitable // - ghostscript (gs) — PDF rewrite/optimize via PostScript (pdfwrite) package docpipe @@ -26,6 +27,7 @@ type Tools struct { Pandoc string OCRMyPDF string PDFToText string + PDFDetach string Tesseract string Ghostscript string } @@ -44,6 +46,7 @@ func LookPath() Tools { Pandoc: find("pandoc"), OCRMyPDF: find("ocrmypdf"), PDFToText: find("pdftotext"), + PDFDetach: find("pdfdetach"), Tesseract: find("tesseract"), Ghostscript: find("gs", "ghostscript"), } diff --git a/internal/docpipe/pdfattach.go b/internal/docpipe/pdfattach.go new file mode 100644 index 0000000..11fd414 --- /dev/null +++ b/internal/docpipe/pdfattach.go @@ -0,0 +1,231 @@ +package docpipe + +// Embedded PDF attachments (F6 #42). Digitised invoices often carry the +// original scan as a PDF attachment; the searchable body may hold only a +// summary. pdfdetach (poppler) lists/saves them; each saved attachment is run +// through the normal docpipe extraction (pdftotext/OCR). + +import ( + "bytes" + "encoding/xml" + "fmt" + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" +) + +// PDFAttachment is one embedded file in a PDF. +type PDFAttachment struct { + Index int // 1-based number as `pdfdetach -list` reports it + Name string // embedded file name +} + +// AttachmentText is the extracted text of one embedded attachment. +type AttachmentText struct { + Name string + Text string +} + +// parseAttachmentList parses `pdfdetach -list` output. The first line is a +// count ("N embedded files"); every following line is ": ". +// Pure, so it is unit-tested. +func parseAttachmentList(out string) []PDFAttachment { + var atts []PDFAttachment + for _, line := range strings.Split(out, "\n") { + line = strings.TrimSpace(line) + if line == "" { + continue + } + colon := strings.Index(line, ":") + if colon <= 0 { + continue + } + n, err := strconv.Atoi(strings.TrimSpace(line[:colon])) + if err != nil { + continue + } + name := strings.TrimSpace(line[colon+1:]) + if name == "" { + continue + } + atts = append(atts, PDFAttachment{Index: n, Name: name}) + } + return atts +} + +// safeAttachmentName strips directories and leading dots so a hostile +// attachment name cannot escape the extraction directory. +func safeAttachmentName(name string) string { + name = strings.ReplaceAll(strings.TrimSpace(name), "\\", "/") + name = filepath.Base(name) + name = strings.TrimLeft(name, ".") + if name == "" || name == "." || name == "/" { + return "" + } + return name +} + +// JoinWithAttachments appends attachment text to the document body, each +// section preceded by an "[attachment: ]" marker so a search hit shows +// its source. Empty attachments are skipped. Pure, so it is unit-tested. +func JoinWithAttachments(body string, atts []AttachmentText) string { + var b strings.Builder + b.WriteString(strings.TrimRight(body, "\n")) + for _, a := range atts { + text := strings.TrimSpace(a.Text) + if text == "" { + continue + } + b.WriteString("\n\n[attachment: ") + b.WriteString(a.Name) + b.WriteString("]\n\n") + b.WriteString(text) + } + return b.String() +} + +// ListAttachments returns the embedded files of a PDF. A PDF without +// attachments yields an empty slice and no error. +func (t Tools) ListAttachments(pdfPath string) ([]PDFAttachment, error) { + if t.PDFDetach == "" { + return nil, fmt.Errorf("pdfdetach not found on PATH") + } + cmd := exec.Command(t.PDFDetach, "-list", pdfPath) + var stderr bytes.Buffer + cmd.Stderr = &stderr + out, err := cmd.Output() + if err != nil { + return nil, fmt.Errorf("pdfdetach -list %s: %w (%s)", filepath.Base(pdfPath), err, strings.TrimSpace(stderr.String())) + } + return parseAttachmentList(string(out)), nil +} + +// SaveAttachment writes the n-th embedded file (1-based) to outPath. +func (t Tools) SaveAttachment(pdfPath string, index int, outPath string) error { + if t.PDFDetach == "" { + return fmt.Errorf("pdfdetach not found on PATH") + } + if strings.TrimSpace(outPath) == "" { + return fmt.Errorf("output path required") + } + if err := EnsureDir(outPath); err != nil { + return err + } + cmd := exec.Command(t.PDFDetach, "-save", strconv.Itoa(index), "-o", outPath, pdfPath) + var stderr bytes.Buffer + cmd.Stderr = &stderr + if err := cmd.Run(); err != nil { + return fmt.Errorf("pdfdetach -save %d: %w (%s)", index, err, strings.TrimSpace(stderr.String())) + } + return nil +} + +// ToMarkdownWithAttachments extracts the file as ToMarkdown does, then — for +// PDFs — appends the text of every embedded attachment under an +// "[attachment: ]" marker. Attachment failures are non-fatal: the body +// is returned unchanged. +func (t Tools) ToMarkdownWithAttachments(path, workDir, lang string, minChars int) (string, error) { + res, err := t.ToMarkdown(path, workDir, lang, minChars) + if err != nil { + return "", err + } + if Ext(path) != ".pdf" { + return res.Markdown, nil + } + atts, err := t.attachmentTexts(path, workDir, lang, minChars) + if err != nil { + return res.Markdown, nil + } + return JoinWithAttachments(res.Markdown, atts), nil +} + +// attachmentTexts saves and extracts every embedded attachment, skipping the +// ones that cannot be read. It returns an error only when the attachment list +// itself cannot be obtained. +func (t Tools) attachmentTexts(pdfPath, workDir, lang string, minChars int) ([]AttachmentText, error) { + list, err := t.ListAttachments(pdfPath) + if err != nil || len(list) == 0 { + return nil, err + } + if workDir == "" { + workDir = os.TempDir() + } + dir := filepath.Join(workDir, "att-"+trimExt(filepath.Base(pdfPath))) + if err := os.MkdirAll(dir, 0o755); err != nil { + return nil, err + } + defer os.RemoveAll(dir) + + out := make([]AttachmentText, 0, len(list)) + for _, a := range list { + name := safeAttachmentName(a.Name) + if name == "" { + continue + } + saved := filepath.Join(dir, fmt.Sprintf("%d-%s", a.Index, name)) + if err := t.SaveAttachment(pdfPath, a.Index, saved); err != nil { + continue + } + text, err := t.attachmentMarkdown(saved, dir, lang, minChars) + if err != nil { + continue + } + out = append(out, AttachmentText{Name: a.Name, Text: text}) + } + return out, nil +} + +// attachmentMarkdown extracts a saved attachment with the regular pipeline. +// Structured attachments that docpipe does not convert (e-invoice XML, +// CuraSoft JSON, CSV/HTML) fall back to their text content, so the embedded +// original is still searchable. Other unreadable formats return an error and +// the caller skips them. +func (t Tools) attachmentMarkdown(path, workDir, lang string, minChars int) (string, error) { + if res, err := t.ToMarkdown(path, workDir, lang, minChars); err == nil { + return res.Markdown, nil + } + switch Ext(path) { + case ".xml", ".html", ".htm": + raw, err := os.ReadFile(path) + if err != nil { + return "", err + } + return xmlToText(raw), nil + case ".json", ".csv": + raw, err := os.ReadFile(path) + if err != nil { + return "", err + } + return string(raw), nil + default: + return "", fmt.Errorf("unsupported attachment type %q", Ext(path)) + } +} + +// xmlToText returns the character data of an XML/HTML document: element text +// values with decoded entities, one per line. Used for invoice XML (EN 16931 +// CII / ZUGFeRD) and HTML attachments. Pure, so it is unit-tested. +func xmlToText(raw []byte) string { + dec := xml.NewDecoder(bytes.NewReader(raw)) + dec.Strict = false + var b strings.Builder + for { + tok, err := dec.Token() + if err != nil { + break + } + cd, ok := tok.(xml.CharData) + if !ok { + continue + } + s := strings.TrimSpace(string(cd)) + if s == "" { + continue + } + b.WriteString(s) + b.WriteByte('\n') + } + return b.String() +} diff --git a/internal/docpipe/pdfattach_test.go b/internal/docpipe/pdfattach_test.go new file mode 100644 index 0000000..ad8ab6e --- /dev/null +++ b/internal/docpipe/pdfattach_test.go @@ -0,0 +1,163 @@ +package docpipe + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +func TestParseAttachmentList(t *testing.T) { + out := "2 embedded files\n1: original.pdf\n2: scan_001.png\n" + got := parseAttachmentList(out) + want := []PDFAttachment{{Index: 1, Name: "original.pdf"}, {Index: 2, Name: "scan_001.png"}} + if len(got) != len(want) { + t.Fatalf("got %+v, want %+v", got, want) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("att[%d] = %+v, want %+v", i, got[i], want[i]) + } + } +} + +func TestParseAttachmentListEmptyAndMalformed(t *testing.T) { + for _, in := range []string{"", "0 embedded files\n", "garbage\n\n \n"} { + if got := parseAttachmentList(in); len(got) != 0 { + t.Errorf("parseAttachmentList(%q) = %+v, want empty", in, got) + } + } +} + +func TestSafeAttachmentName(t *testing.T) { + cases := map[string]string{ + "note.txt": "note.txt", + "../../evil.pdf": "evil.pdf", + `..\..\evil.pdf`: "evil.pdf", + "/abs/scan_001.pdf": "scan_001.pdf", + ".hidden": "hidden", + " spaced name.txt ": "spaced name.txt", + "..": "", + "": "", + } + for in, want := range cases { + if got := safeAttachmentName(in); got != want { + t.Errorf("safeAttachmentName(%q) = %q, want %q", in, got, want) + } + } +} + +func TestJoinWithAttachments(t *testing.T) { + body := "# scan.pdf\n\nbody token\n" + atts := []AttachmentText{ + {Name: "original.pdf", Text: " original token "}, + {Name: "empty.txt", Text: " "}, + } + got := JoinWithAttachments(body, atts) + if !strings.Contains(got, "body token") { + t.Errorf("body text lost: %q", got) + } + if !strings.Contains(got, "[attachment: original.pdf]") { + t.Errorf("marker missing: %q", got) + } + if !strings.Contains(got, "original token") { + t.Errorf("attachment text missing: %q", got) + } + if strings.Contains(got, "empty.txt") { + t.Errorf("empty attachment must be skipped: %q", got) + } +} + +func TestJoinWithAttachmentsNoAttachments(t *testing.T) { + got := JoinWithAttachments("# a.pdf\n\ntext\n\n", nil) + if got != "# a.pdf\n\ntext" { + t.Errorf("got %q, want trimmed body only", got) + } +} + +func TestListAttachmentsWithoutTool(t *testing.T) { + if _, err := (Tools{}).ListAttachments("x.pdf"); err == nil || !strings.Contains(err.Error(), "pdfdetach") { + t.Fatalf("want pdfdetach error, got %v", err) + } +} + +// TestToMarkdownWithAttachmentsFixture exercises the real pdfdetach + pdftotext +// pipeline on testdata/pdf-with-attachment.pdf (body token + embedded +// goo-note.txt). Skips when poppler is not installed. +func TestToMarkdownWithAttachmentsFixture(t *testing.T) { + tools := LookPath() + if tools.PDFDetach == "" || tools.PDFToText == "" { + t.Skip("pdfdetach/pdftotext not on PATH — skipping attachment extraction test") + } + fixture := filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf") + got, err := tools.ToMarkdownWithAttachments(fixture, t.TempDir(), "eng", 1) + if err != nil { + t.Fatalf("ToMarkdownWithAttachments: %v", err) + } + for _, want := range []string{"goobodytoken", "[attachment: goo-note.txt]", "gooattachmenttoken"} { + if !strings.Contains(got, want) { + t.Errorf("result missing %q:\n%s", want, got) + } + } +} + +func TestXMLToText(t *testing.T) { + raw := []byte(` + +S1063 +Edelweiss & Co42.00 +`) + got := xmlToText(raw) + for _, want := range []string{"S1063", "Edelweiss & Co", "42.00"} { + if !strings.Contains(got, want) { + t.Errorf("xmlToText missing %q:\n%s", want, got) + } + } + if strings.ContainsAny(got, "<>") { + t.Errorf("xmlToText left markup: %q", got) + } +} + +// TestAttachmentMarkdownFallback verifies structured attachments that docpipe +// cannot convert are still reduced to searchable text, and unknown binary +// formats error (so the caller skips them). +func TestAttachmentMarkdownFallback(t *testing.T) { + dir := t.TempDir() + xmlPath := filepath.Join(dir, "factur-x.xml") + if err := os.WriteFile(xmlPath, []byte(`S1063`), 0o644); err != nil { + t.Fatal(err) + } + got, err := (Tools{}).attachmentMarkdown(xmlPath, dir, "", 0) + if err != nil { + t.Fatalf("attachmentMarkdown(xml): %v", err) + } + if !strings.Contains(got, "S1063") { + t.Errorf("xml attachment text = %q, want S1063", got) + } + + binPath := filepath.Join(dir, "data.bin") + if err := os.WriteFile(binPath, []byte{0, 1, 2, 3}, 0o644); err != nil { + t.Fatal(err) + } + if _, err := (Tools{}).attachmentMarkdown(binPath, dir, "", 0); err == nil { + t.Error("unsupported attachment: want error, got nil") + } +} + +// TestToMarkdownWithAttachmentsPlainPDF ensures a PDF without attachments +// returns just the body (pdfdetach prints "0 embedded files"). +func TestToMarkdownWithAttachmentsPlainPDF(t *testing.T) { + tools := LookPath() + if tools.PDFDetach == "" || tools.PDFToText == "" { + t.Skip("pdfdetach/pdftotext not on PATH") + } + // The fixture itself is a PDF with one attachment; strip it by extracting + // the body only through ToMarkdown and compare JoinWithAttachments(nil). + res, err := tools.ToMarkdown(filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf"), t.TempDir(), "eng", 1) + if err != nil { + t.Fatalf("ToMarkdown: %v", err) + } + if strings.Contains(res.Markdown, "gooattachmenttoken") { + t.Fatalf("body must not contain attachment text: %q", res.Markdown) + } +} diff --git a/testdata/pdf-with-attachment.pdf b/testdata/pdf-with-attachment.pdf new file mode 100644 index 0000000..34dac60 --- /dev/null +++ b/testdata/pdf-with-attachment.pdf @@ -0,0 +1,114 @@ +%PDF-1.7 +%Ç쏢 +%%Invocation: gs -q -dNOPAUSE -dBATCH -sDEVICE=pdfwrite ? ? ? +5 0 obj +<> +stream +xœ-ŠA +ƒ0÷ÿoÙnìO7q]èZ*ÿ�Iˆm J½½M ³f7 +\¨6‘žfRÿŠ*ñºõª…8:g}‡f†Dºø”ÞiØsúØ ÝôÝ;炱(Ùn.ly]ìUFz +½~Aë øendstream +endobj +6 0 obj +110 +endobj +4 0 obj +<> +/Contents 5 0 R +>> +endobj +3 0 obj +<< /Type /Pages /Kids [ +4 0 R +] /Count 1 +>> +endobj +1 0 obj +<> +endobj +8 0 obj +<> +endobj +7 0 obj +<> +endobj +9 0 obj +<>stream + + + + + +2026-09-16T17:29:12Z +2026-09-16T17:29:12Z +UnknownApplication + +Untitled + + + + + +endstream +endobj +2 0 obj +<>endobj +xref +0 10 +0000000000 65535 f +0000000476 00000 n +0000001882 00000 n +0000000417 00000 n +0000000276 00000 n +0000000077 00000 n +0000000257 00000 n +0000000569 00000 n +0000000540 00000 n +0000000633 00000 n +trailer +<< /Size 10 /Root 1 0 R /Info 2 0 R +/ID [<5A562476FBF11852133AA873452909F9><5A562476FBF11852133AA873452909F9>] +>> +startxref +2008 +%%EOF +1 0 obj +<> >> +endobj +10 0 obj +<> >> stream +gooattachmenttoken embedded attachment text + +endstream + +endobj +11 0 obj +< /EF <> >> +endobj +12 0 obj +< 11 0 R ] >> +endobj +xref +0 2 +0000000002 65535 f +0000002361 00000 n +10 3 +0000002463 00000 n +0000002586 00000 n +0000002705 00000 n +trailer +<> +startxref +2802 +%%EOF