feat(search): PDF-контент в поиске — свой индекс oo_docs_text (#42) #44
@@ -701,9 +701,11 @@ Requires `ONLYOFFICE_ES_URL` (plus optional `ONLYOFFICE_ES_INDEX`,
|
|||||||
|
|
||||||
The OnlyOffice index covers Office formats only, so PDFs (`S1019`-style invoice
|
The OnlyOffice index covers Office formats only, so PDFs (`S1019`-style invoice
|
||||||
numbers) are not searchable by content. `oo index` extracts PDF text with
|
numbers) are not searchable by content. `oo index` extracts PDF text with
|
||||||
`internal/docpipe` (pdftotext, OCR for scans) into a separate index
|
`internal/docpipe` (pdftotext, OCR for scans) — including the text of embedded
|
||||||
(`ONLYOFFICE_ES_TEXT_INDEX`, default `oo_docs_text`); the OnlyOffice server and
|
PDF attachments (`pdfdetach`: `<doc>.md`, `.xml`, covers the original/scan and
|
||||||
its index are **not** modified. Then search it with `--backend own`.
|
ZUGFeRD e-invoice XML) — into a separate index (`ONLYOFFICE_ES_TEXT_INDEX`,
|
||||||
|
default `oo_docs_text`); the OnlyOffice server and its index are **not**
|
||||||
|
modified. Then search it with `--backend own`.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
oo index folder 634 --recursive --exts pdf # populate (idempotent upsert)
|
oo index folder 634 --recursive --exts pdf # populate (idempotent upsert)
|
||||||
|
|||||||
+26
-2
@@ -174,6 +174,22 @@ OnlyOffice PDF лежит только по имени.
|
|||||||
- CLI: `oo index folder|files` наполняет индекс; `oo search --backend own`
|
- CLI: `oo index folder|files` наполняет индекс; `oo search --backend own`
|
||||||
ищет по нему.
|
ищет по нему.
|
||||||
|
|
||||||
|
### Встроенные вложения PDF
|
||||||
|
|
||||||
|
Оцифрованные PDF несут вложения (`<doc>.md` — текст/таблицы скана,
|
||||||
|
`<doc>.yaml`/`.json` — метаданные, `.xml` — EN 16931 CII eRechnung,
|
||||||
|
`factur-x.xml` у ZUGFeRD; см. `office-assistant/docs/reference/document-metadata.md`).
|
||||||
|
`TextIndexer` обходит их: `pdfdetach -list` перечисляет, `-save` сохраняет,
|
||||||
|
каждое вложение проходит штатный `docpipe.ToMarkdown` (PDF/картинки → OCR,
|
||||||
|
`.md`/`.txt` — как есть). Форматы, которые docpipe не конвертирует
|
||||||
|
(`.xml`/`.html` — снимаются теги; `.json`/`.csv` — как текст), извлекаются
|
||||||
|
текстом; нечитаемые — пропускаются.
|
||||||
|
|
||||||
|
Текст склеивается: тело, затем по секции на вложение с маркером
|
||||||
|
`[attachment: <имя>]` (функция `docpipe.JoinWithAttachments`). Индекс — тот же
|
||||||
|
`file_id`, upsert идемпотентен. Нет вложений или pdfdetach/формат нечитаем —
|
||||||
|
индексируется тело (без падения).
|
||||||
|
|
||||||
Поля `oo_docs_text`:
|
Поля `oo_docs_text`:
|
||||||
|
|
||||||
| поле | тип | смысл |
|
| поле | тип | смысл |
|
||||||
@@ -217,8 +233,12 @@ ONLYOFFICE_ES_URL=http://127.0.0.1:9200 \
|
|||||||
```
|
```
|
||||||
|
|
||||||
Интеграционный тест создаёт временный индекс, наполняет, ищет по контенту,
|
Интеграционный тест создаёт временный индекс, наполняет, ищет по контенту,
|
||||||
проверяет фильтры и удаление, затем удаляет индекс. Unit-тесты используют
|
проверяет фильтры и удаление, затем удаляет индекс;
|
||||||
fake-store/fake-extractor и не требуют pdftotext/OCR.
|
`TestIntegrationESTextIndexPDFAttachment` индексирует
|
||||||
|
`testdata/pdf-with-attachment.pdf` реальным конвейером (pdfdetach + pdftotext)
|
||||||
|
и ищет токен, лежащий только во вложении. Unit-тесты используют
|
||||||
|
fake-store/fake-extractor и не требуют pdftotext/OCR (парсер списка, склейка
|
||||||
|
`JoinWithAttachments`, снятие тегов `xmlToText` — чистые).
|
||||||
|
|
||||||
## Грабли
|
## Грабли
|
||||||
|
|
||||||
@@ -229,3 +249,7 @@ fake-store/fake-extractor и не требуют pdftotext/OCR.
|
|||||||
- `folder` фильтруется как id папки, а не как путь.
|
- `folder` фильтруется как id папки, а не как путь.
|
||||||
- Дубликаты (напр. `S1055.pdf` и `2026-08-20-S1055-…`) дадут несколько строк —
|
- Дубликаты (напр. `S1055.pdf` и `2026-08-20-S1055-…`) дадут несколько строк —
|
||||||
это ожидаемо, дедуп — на стороне потребителя.
|
это ожидаемо, дедуп — на стороне потребителя.
|
||||||
|
- Вложения: нужен `pdfdetach` (poppler); если его нет — индексируется только
|
||||||
|
тело. Вложенный PDF/картинка с плохим текстовым слоем проходит OCR, это
|
||||||
|
медленно. `.json`-метаданные (CuraSoft) индексируются как текст и могут
|
||||||
|
добавить шумовых токенов.
|
||||||
|
|||||||
@@ -9,6 +9,8 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"github.com/eslider/go-onlyoffice/internal/docpipe"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestIntegrationESTextIndex verifies the own full-text index end to end
|
// TestIntegrationESTextIndex verifies the own full-text index end to end
|
||||||
@@ -90,3 +92,67 @@ func TestIntegrationESTextIndex(t *testing.T) {
|
|||||||
t.Errorf("after delete search returned %d hits, want 0", len(hits))
|
t.Errorf("after delete search returned %d hits, want 0", len(hits))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestIntegrationESTextIndexPDFAttachment indexes testdata/pdf-with-attachment.pdf
|
||||||
|
// through the real pipeline (TextIndexer + docpipe: pdfdetach + pdftotext) and
|
||||||
|
// verifies that text living only in the embedded attachment is searchable.
|
||||||
|
//
|
||||||
|
// Requires ONLYOFFICE_ES_URL plus poppler (pdfdetach/pdftotext). No OnlyOffice
|
||||||
|
// credentials are needed: a fixture FileStore serves the PDF bytes.
|
||||||
|
func TestIntegrationESTextIndexPDFAttachment(t *testing.T) {
|
||||||
|
esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL"))
|
||||||
|
if esURL == "" {
|
||||||
|
t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test")
|
||||||
|
}
|
||||||
|
if docpipe.LookPath().PDFDetach == "" {
|
||||||
|
t.Skip("pdfdetach not on PATH — skipping PDF attachment integration test")
|
||||||
|
}
|
||||||
|
pdf, err := os.ReadFile("testdata/pdf-with-attachment.pdf")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read fixture: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
stamp := time.Now().UTC().Format("20060102150405")
|
||||||
|
idx, err := NewESTextIndex(ESTextConfig{URL: esURL, Index: "oo_docs_text_it_att_" + stamp})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("NewESTextIndex: %v", err)
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
t.Cleanup(func() {
|
||||||
|
cleanupCtx, done := context.WithTimeout(context.Background(), 30*time.Second)
|
||||||
|
defer done()
|
||||||
|
_, _, _ = idx.do(cleanupCtx, http.MethodDelete, "/"+idx.Index(), nil, "")
|
||||||
|
})
|
||||||
|
if err := idx.Ensure(ctx); err != nil {
|
||||||
|
t.Fatalf("Ensure: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
store := &textFakeStore{files: map[string][]byte{"9001": pdf}}
|
||||||
|
ix := NewTextIndexer(store, idx)
|
||||||
|
res, err := ix.IndexEntries(ctx, []Entry{{ID: "9001", Title: "scan.pdf", ParentID: "777", Kind: File}}, IndexOptions{MinChars: 1})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("IndexEntries: %v", err)
|
||||||
|
}
|
||||||
|
if res.Indexed != 1 || res.Failed != 0 {
|
||||||
|
t.Fatalf("result = %+v, want one indexed doc", res)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Token appears only inside the embedded goo-note.txt attachment.
|
||||||
|
hits, err := idx.Search(ctx, SearchQuery{Text: "gooattachmenttoken"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Search attachment token: %v", err)
|
||||||
|
}
|
||||||
|
if len(hits) != 1 || hits[0].ID != "9001" {
|
||||||
|
t.Fatalf("attachment-token hits = %+v, want doc 9001", hits)
|
||||||
|
}
|
||||||
|
if !strings.Contains(hits[0].Highlight, "gooattachmenttoken") {
|
||||||
|
t.Errorf("highlight = %q, want attachment token", hits[0].Highlight)
|
||||||
|
}
|
||||||
|
// Body text is indexed as before.
|
||||||
|
if hits, err := idx.Search(ctx, SearchQuery{Text: "goobodytoken"}); err != nil {
|
||||||
|
t.Fatalf("Search body token: %v", err)
|
||||||
|
} else if len(hits) != 1 {
|
||||||
|
t.Errorf("body-token hits = %d, want 1", len(hits))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+8
-6
@@ -3,9 +3,10 @@ package onlyoffice
|
|||||||
// Text extraction pipeline for the own full-text index (epic #34, F6 #42).
|
// Text extraction pipeline for the own full-text index (epic #34, F6 #42).
|
||||||
//
|
//
|
||||||
// TextIndexer downloads stored documents, extracts text through docpipe
|
// TextIndexer downloads stored documents, extracts text through docpipe
|
||||||
// (pdftotext; OCR for scans) and writes the result to a TextIndex. It is the
|
// (pdftotext; OCR for scans) and writes the result to a TextIndex. For PDFs it
|
||||||
// write side of ESTextIndex and never touches the OnlyOffice server's own ES
|
// also indexes the text of embedded attachments (pdfdetach), so a scan filed
|
||||||
// index.
|
// as an attachment is searchable too. It is the write side of ESTextIndex and
|
||||||
|
// never touches the OnlyOffice server's own ES index.
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
@@ -38,13 +39,14 @@ type TextExtractor interface {
|
|||||||
type docpipeExtractor struct{ tools docpipe.Tools }
|
type docpipeExtractor struct{ tools docpipe.Tools }
|
||||||
|
|
||||||
// Extract renders the file as Markdown, OCRing PDFs/images with a weak text
|
// Extract renders the file as Markdown, OCRing PDFs/images with a weak text
|
||||||
// layer first (docpipe.ToMarkdown).
|
// layer first and appending the text of embedded PDF attachments
|
||||||
|
// (docpipe.ToMarkdownWithAttachments).
|
||||||
func (d docpipeExtractor) Extract(path, workDir, lang string, minChars int) (string, error) {
|
func (d docpipeExtractor) Extract(path, workDir, lang string, minChars int) (string, error) {
|
||||||
res, err := d.tools.ToMarkdown(path, workDir, lang, minChars)
|
text, err := d.tools.ToMarkdownWithAttachments(path, workDir, lang, minChars)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return "", err
|
return "", err
|
||||||
}
|
}
|
||||||
return strings.TrimSpace(res.Markdown), nil
|
return strings.TrimSpace(text), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// IndexOptions controls a TextIndexer run.
|
// IndexOptions controls a TextIndexer run.
|
||||||
|
|||||||
@@ -5,6 +5,7 @@
|
|||||||
// - pandoc — md↔docx
|
// - pandoc — md↔docx
|
||||||
// - ocrmypdf — OCR into a searchable PDF
|
// - ocrmypdf — OCR into a searchable PDF
|
||||||
// - pdftotext — extract text layer
|
// - pdftotext — extract text layer
|
||||||
|
// - pdfdetach — list/save embedded PDF attachments
|
||||||
// - tesseract — OCR single images when ocrmypdf is unsuitable
|
// - tesseract — OCR single images when ocrmypdf is unsuitable
|
||||||
// - ghostscript (gs) — PDF rewrite/optimize via PostScript (pdfwrite)
|
// - ghostscript (gs) — PDF rewrite/optimize via PostScript (pdfwrite)
|
||||||
package docpipe
|
package docpipe
|
||||||
@@ -26,6 +27,7 @@ type Tools struct {
|
|||||||
Pandoc string
|
Pandoc string
|
||||||
OCRMyPDF string
|
OCRMyPDF string
|
||||||
PDFToText string
|
PDFToText string
|
||||||
|
PDFDetach string
|
||||||
Tesseract string
|
Tesseract string
|
||||||
Ghostscript string
|
Ghostscript string
|
||||||
}
|
}
|
||||||
@@ -44,6 +46,7 @@ func LookPath() Tools {
|
|||||||
Pandoc: find("pandoc"),
|
Pandoc: find("pandoc"),
|
||||||
OCRMyPDF: find("ocrmypdf"),
|
OCRMyPDF: find("ocrmypdf"),
|
||||||
PDFToText: find("pdftotext"),
|
PDFToText: find("pdftotext"),
|
||||||
|
PDFDetach: find("pdfdetach"),
|
||||||
Tesseract: find("tesseract"),
|
Tesseract: find("tesseract"),
|
||||||
Ghostscript: find("gs", "ghostscript"),
|
Ghostscript: find("gs", "ghostscript"),
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,231 @@
|
|||||||
|
package docpipe
|
||||||
|
|
||||||
|
// Embedded PDF attachments (F6 #42). Digitised invoices often carry the
|
||||||
|
// original scan as a PDF attachment; the searchable body may hold only a
|
||||||
|
// summary. pdfdetach (poppler) lists/saves them; each saved attachment is run
|
||||||
|
// through the normal docpipe extraction (pdftotext/OCR).
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/xml"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// PDFAttachment is one embedded file in a PDF.
|
||||||
|
type PDFAttachment struct {
|
||||||
|
Index int // 1-based number as `pdfdetach -list` reports it
|
||||||
|
Name string // embedded file name
|
||||||
|
}
|
||||||
|
|
||||||
|
// AttachmentText is the extracted text of one embedded attachment.
|
||||||
|
type AttachmentText struct {
|
||||||
|
Name string
|
||||||
|
Text string
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseAttachmentList parses `pdfdetach -list` output. The first line is a
|
||||||
|
// count ("N embedded files"); every following line is "<index>: <name>".
|
||||||
|
// Pure, so it is unit-tested.
|
||||||
|
func parseAttachmentList(out string) []PDFAttachment {
|
||||||
|
var atts []PDFAttachment
|
||||||
|
for _, line := range strings.Split(out, "\n") {
|
||||||
|
line = strings.TrimSpace(line)
|
||||||
|
if line == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
colon := strings.Index(line, ":")
|
||||||
|
if colon <= 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
n, err := strconv.Atoi(strings.TrimSpace(line[:colon]))
|
||||||
|
if err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
name := strings.TrimSpace(line[colon+1:])
|
||||||
|
if name == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
atts = append(atts, PDFAttachment{Index: n, Name: name})
|
||||||
|
}
|
||||||
|
return atts
|
||||||
|
}
|
||||||
|
|
||||||
|
// safeAttachmentName strips directories and leading dots so a hostile
|
||||||
|
// attachment name cannot escape the extraction directory.
|
||||||
|
func safeAttachmentName(name string) string {
|
||||||
|
name = strings.ReplaceAll(strings.TrimSpace(name), "\\", "/")
|
||||||
|
name = filepath.Base(name)
|
||||||
|
name = strings.TrimLeft(name, ".")
|
||||||
|
if name == "" || name == "." || name == "/" {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return name
|
||||||
|
}
|
||||||
|
|
||||||
|
// JoinWithAttachments appends attachment text to the document body, each
|
||||||
|
// section preceded by an "[attachment: <name>]" marker so a search hit shows
|
||||||
|
// its source. Empty attachments are skipped. Pure, so it is unit-tested.
|
||||||
|
func JoinWithAttachments(body string, atts []AttachmentText) string {
|
||||||
|
var b strings.Builder
|
||||||
|
b.WriteString(strings.TrimRight(body, "\n"))
|
||||||
|
for _, a := range atts {
|
||||||
|
text := strings.TrimSpace(a.Text)
|
||||||
|
if text == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
b.WriteString("\n\n[attachment: ")
|
||||||
|
b.WriteString(a.Name)
|
||||||
|
b.WriteString("]\n\n")
|
||||||
|
b.WriteString(text)
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ListAttachments returns the embedded files of a PDF. A PDF without
|
||||||
|
// attachments yields an empty slice and no error.
|
||||||
|
func (t Tools) ListAttachments(pdfPath string) ([]PDFAttachment, error) {
|
||||||
|
if t.PDFDetach == "" {
|
||||||
|
return nil, fmt.Errorf("pdfdetach not found on PATH")
|
||||||
|
}
|
||||||
|
cmd := exec.Command(t.PDFDetach, "-list", pdfPath)
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
cmd.Stderr = &stderr
|
||||||
|
out, err := cmd.Output()
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("pdfdetach -list %s: %w (%s)", filepath.Base(pdfPath), err, strings.TrimSpace(stderr.String()))
|
||||||
|
}
|
||||||
|
return parseAttachmentList(string(out)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// SaveAttachment writes the n-th embedded file (1-based) to outPath.
|
||||||
|
func (t Tools) SaveAttachment(pdfPath string, index int, outPath string) error {
|
||||||
|
if t.PDFDetach == "" {
|
||||||
|
return fmt.Errorf("pdfdetach not found on PATH")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(outPath) == "" {
|
||||||
|
return fmt.Errorf("output path required")
|
||||||
|
}
|
||||||
|
if err := EnsureDir(outPath); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
cmd := exec.Command(t.PDFDetach, "-save", strconv.Itoa(index), "-o", outPath, pdfPath)
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
cmd.Stderr = &stderr
|
||||||
|
if err := cmd.Run(); err != nil {
|
||||||
|
return fmt.Errorf("pdfdetach -save %d: %w (%s)", index, err, strings.TrimSpace(stderr.String()))
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ToMarkdownWithAttachments extracts the file as ToMarkdown does, then — for
|
||||||
|
// PDFs — appends the text of every embedded attachment under an
|
||||||
|
// "[attachment: <name>]" marker. Attachment failures are non-fatal: the body
|
||||||
|
// is returned unchanged.
|
||||||
|
func (t Tools) ToMarkdownWithAttachments(path, workDir, lang string, minChars int) (string, error) {
|
||||||
|
res, err := t.ToMarkdown(path, workDir, lang, minChars)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
if Ext(path) != ".pdf" {
|
||||||
|
return res.Markdown, nil
|
||||||
|
}
|
||||||
|
atts, err := t.attachmentTexts(path, workDir, lang, minChars)
|
||||||
|
if err != nil {
|
||||||
|
return res.Markdown, nil
|
||||||
|
}
|
||||||
|
return JoinWithAttachments(res.Markdown, atts), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// attachmentTexts saves and extracts every embedded attachment, skipping the
|
||||||
|
// ones that cannot be read. It returns an error only when the attachment list
|
||||||
|
// itself cannot be obtained.
|
||||||
|
func (t Tools) attachmentTexts(pdfPath, workDir, lang string, minChars int) ([]AttachmentText, error) {
|
||||||
|
list, err := t.ListAttachments(pdfPath)
|
||||||
|
if err != nil || len(list) == 0 {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if workDir == "" {
|
||||||
|
workDir = os.TempDir()
|
||||||
|
}
|
||||||
|
dir := filepath.Join(workDir, "att-"+trimExt(filepath.Base(pdfPath)))
|
||||||
|
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer os.RemoveAll(dir)
|
||||||
|
|
||||||
|
out := make([]AttachmentText, 0, len(list))
|
||||||
|
for _, a := range list {
|
||||||
|
name := safeAttachmentName(a.Name)
|
||||||
|
if name == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
saved := filepath.Join(dir, fmt.Sprintf("%d-%s", a.Index, name))
|
||||||
|
if err := t.SaveAttachment(pdfPath, a.Index, saved); err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
text, err := t.attachmentMarkdown(saved, dir, lang, minChars)
|
||||||
|
if err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
out = append(out, AttachmentText{Name: a.Name, Text: text})
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// attachmentMarkdown extracts a saved attachment with the regular pipeline.
|
||||||
|
// Structured attachments that docpipe does not convert (e-invoice XML,
|
||||||
|
// CuraSoft JSON, CSV/HTML) fall back to their text content, so the embedded
|
||||||
|
// original is still searchable. Other unreadable formats return an error and
|
||||||
|
// the caller skips them.
|
||||||
|
func (t Tools) attachmentMarkdown(path, workDir, lang string, minChars int) (string, error) {
|
||||||
|
if res, err := t.ToMarkdown(path, workDir, lang, minChars); err == nil {
|
||||||
|
return res.Markdown, nil
|
||||||
|
}
|
||||||
|
switch Ext(path) {
|
||||||
|
case ".xml", ".html", ".htm":
|
||||||
|
raw, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return xmlToText(raw), nil
|
||||||
|
case ".json", ".csv":
|
||||||
|
raw, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return string(raw), nil
|
||||||
|
default:
|
||||||
|
return "", fmt.Errorf("unsupported attachment type %q", Ext(path))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// xmlToText returns the character data of an XML/HTML document: element text
|
||||||
|
// values with decoded entities, one per line. Used for invoice XML (EN 16931
|
||||||
|
// CII / ZUGFeRD) and HTML attachments. Pure, so it is unit-tested.
|
||||||
|
func xmlToText(raw []byte) string {
|
||||||
|
dec := xml.NewDecoder(bytes.NewReader(raw))
|
||||||
|
dec.Strict = false
|
||||||
|
var b strings.Builder
|
||||||
|
for {
|
||||||
|
tok, err := dec.Token()
|
||||||
|
if err != nil {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
cd, ok := tok.(xml.CharData)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
s := strings.TrimSpace(string(cd))
|
||||||
|
if s == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
b.WriteString(s)
|
||||||
|
b.WriteByte('\n')
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
@@ -0,0 +1,163 @@
|
|||||||
|
package docpipe
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestParseAttachmentList(t *testing.T) {
|
||||||
|
out := "2 embedded files\n1: original.pdf\n2: scan_001.png\n"
|
||||||
|
got := parseAttachmentList(out)
|
||||||
|
want := []PDFAttachment{{Index: 1, Name: "original.pdf"}, {Index: 2, Name: "scan_001.png"}}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("got %+v, want %+v", got, want)
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("att[%d] = %+v, want %+v", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParseAttachmentListEmptyAndMalformed(t *testing.T) {
|
||||||
|
for _, in := range []string{"", "0 embedded files\n", "garbage\n\n \n"} {
|
||||||
|
if got := parseAttachmentList(in); len(got) != 0 {
|
||||||
|
t.Errorf("parseAttachmentList(%q) = %+v, want empty", in, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSafeAttachmentName(t *testing.T) {
|
||||||
|
cases := map[string]string{
|
||||||
|
"note.txt": "note.txt",
|
||||||
|
"../../evil.pdf": "evil.pdf",
|
||||||
|
`..\..\evil.pdf`: "evil.pdf",
|
||||||
|
"/abs/scan_001.pdf": "scan_001.pdf",
|
||||||
|
".hidden": "hidden",
|
||||||
|
" spaced name.txt ": "spaced name.txt",
|
||||||
|
"..": "",
|
||||||
|
"": "",
|
||||||
|
}
|
||||||
|
for in, want := range cases {
|
||||||
|
if got := safeAttachmentName(in); got != want {
|
||||||
|
t.Errorf("safeAttachmentName(%q) = %q, want %q", in, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestJoinWithAttachments(t *testing.T) {
|
||||||
|
body := "# scan.pdf\n\nbody token\n"
|
||||||
|
atts := []AttachmentText{
|
||||||
|
{Name: "original.pdf", Text: " original token "},
|
||||||
|
{Name: "empty.txt", Text: " "},
|
||||||
|
}
|
||||||
|
got := JoinWithAttachments(body, atts)
|
||||||
|
if !strings.Contains(got, "body token") {
|
||||||
|
t.Errorf("body text lost: %q", got)
|
||||||
|
}
|
||||||
|
if !strings.Contains(got, "[attachment: original.pdf]") {
|
||||||
|
t.Errorf("marker missing: %q", got)
|
||||||
|
}
|
||||||
|
if !strings.Contains(got, "original token") {
|
||||||
|
t.Errorf("attachment text missing: %q", got)
|
||||||
|
}
|
||||||
|
if strings.Contains(got, "empty.txt") {
|
||||||
|
t.Errorf("empty attachment must be skipped: %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestJoinWithAttachmentsNoAttachments(t *testing.T) {
|
||||||
|
got := JoinWithAttachments("# a.pdf\n\ntext\n\n", nil)
|
||||||
|
if got != "# a.pdf\n\ntext" {
|
||||||
|
t.Errorf("got %q, want trimmed body only", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestListAttachmentsWithoutTool(t *testing.T) {
|
||||||
|
if _, err := (Tools{}).ListAttachments("x.pdf"); err == nil || !strings.Contains(err.Error(), "pdfdetach") {
|
||||||
|
t.Fatalf("want pdfdetach error, got %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestToMarkdownWithAttachmentsFixture exercises the real pdfdetach + pdftotext
|
||||||
|
// pipeline on testdata/pdf-with-attachment.pdf (body token + embedded
|
||||||
|
// goo-note.txt). Skips when poppler is not installed.
|
||||||
|
func TestToMarkdownWithAttachmentsFixture(t *testing.T) {
|
||||||
|
tools := LookPath()
|
||||||
|
if tools.PDFDetach == "" || tools.PDFToText == "" {
|
||||||
|
t.Skip("pdfdetach/pdftotext not on PATH — skipping attachment extraction test")
|
||||||
|
}
|
||||||
|
fixture := filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf")
|
||||||
|
got, err := tools.ToMarkdownWithAttachments(fixture, t.TempDir(), "eng", 1)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ToMarkdownWithAttachments: %v", err)
|
||||||
|
}
|
||||||
|
for _, want := range []string{"goobodytoken", "[attachment: goo-note.txt]", "gooattachmenttoken"} {
|
||||||
|
if !strings.Contains(got, want) {
|
||||||
|
t.Errorf("result missing %q:\n%s", want, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestXMLToText(t *testing.T) {
|
||||||
|
raw := []byte(`<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<rsm:CrossIndustryInvoice><rsm:ExchangedDocument>
|
||||||
|
<ram:ID>S1063</ram:ID></rsm:ExchangedDocument>
|
||||||
|
<ram:Name>Edelweiss & Co</ram:Name><ram:GrandTotalAmount>42.00</ram:GrandTotalAmount>
|
||||||
|
</rsm:CrossIndustryInvoice>`)
|
||||||
|
got := xmlToText(raw)
|
||||||
|
for _, want := range []string{"S1063", "Edelweiss & Co", "42.00"} {
|
||||||
|
if !strings.Contains(got, want) {
|
||||||
|
t.Errorf("xmlToText missing %q:\n%s", want, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if strings.ContainsAny(got, "<>") {
|
||||||
|
t.Errorf("xmlToText left markup: %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAttachmentMarkdownFallback verifies structured attachments that docpipe
|
||||||
|
// cannot convert are still reduced to searchable text, and unknown binary
|
||||||
|
// formats error (so the caller skips them).
|
||||||
|
func TestAttachmentMarkdownFallback(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
xmlPath := filepath.Join(dir, "factur-x.xml")
|
||||||
|
if err := os.WriteFile(xmlPath, []byte(`<Invoice><Number>S1063</Number></Invoice>`), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
got, err := (Tools{}).attachmentMarkdown(xmlPath, dir, "", 0)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("attachmentMarkdown(xml): %v", err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(got, "S1063") {
|
||||||
|
t.Errorf("xml attachment text = %q, want S1063", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
binPath := filepath.Join(dir, "data.bin")
|
||||||
|
if err := os.WriteFile(binPath, []byte{0, 1, 2, 3}, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, err := (Tools{}).attachmentMarkdown(binPath, dir, "", 0); err == nil {
|
||||||
|
t.Error("unsupported attachment: want error, got nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestToMarkdownWithAttachmentsPlainPDF ensures a PDF without attachments
|
||||||
|
// returns just the body (pdfdetach prints "0 embedded files").
|
||||||
|
func TestToMarkdownWithAttachmentsPlainPDF(t *testing.T) {
|
||||||
|
tools := LookPath()
|
||||||
|
if tools.PDFDetach == "" || tools.PDFToText == "" {
|
||||||
|
t.Skip("pdfdetach/pdftotext not on PATH")
|
||||||
|
}
|
||||||
|
// The fixture itself is a PDF with one attachment; strip it by extracting
|
||||||
|
// the body only through ToMarkdown and compare JoinWithAttachments(nil).
|
||||||
|
res, err := tools.ToMarkdown(filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf"), t.TempDir(), "eng", 1)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ToMarkdown: %v", err)
|
||||||
|
}
|
||||||
|
if strings.Contains(res.Markdown, "gooattachmenttoken") {
|
||||||
|
t.Fatalf("body must not contain attachment text: %q", res.Markdown)
|
||||||
|
}
|
||||||
|
}
|
||||||
Vendored
+114
@@ -0,0 +1,114 @@
|
|||||||
|
%PDF-1.7
|
||||||
|
%Çì�¢
|
||||||
|
%%Invocation: gs -q -dNOPAUSE -dBATCH -sDEVICE=pdfwrite ? ? ?
|
||||||
|
5 0 obj
|
||||||
|
<</Length 6 0 R/Filter /FlateDecode>>
|
||||||
|
stream
|
||||||
|
xœ-ŠA
|
||||||
|
ƒ0÷ÿoÙnìO7q]èZ*ÿ�IˆmJ½½M
|
||||||
|
³f7
|
||||||
|
\¨6‘žfRÿŠ*ñºõª…8:g}‡f†Dºø”ÞiØsúØ ÝôÝ;炱(Ùn.ly]ìUFz
|
||||||
|
½~Aë øendstream
|
||||||
|
endobj
|
||||||
|
6 0 obj
|
||||||
|
110
|
||||||
|
endobj
|
||||||
|
4 0 obj
|
||||||
|
<</Type/Page/MediaBox [0 0 612 792]
|
||||||
|
/Rotate 0/Parent 3 0 R
|
||||||
|
/Resources<</ProcSet[/PDF /Text]
|
||||||
|
/Font 8 0 R
|
||||||
|
>>
|
||||||
|
/Contents 5 0 R
|
||||||
|
>>
|
||||||
|
endobj
|
||||||
|
3 0 obj
|
||||||
|
<< /Type /Pages /Kids [
|
||||||
|
4 0 R
|
||||||
|
] /Count 1
|
||||||
|
>>
|
||||||
|
endobj
|
||||||
|
1 0 obj
|
||||||
|
<</Type /Catalog /Pages 3 0 R
|
||||||
|
/Metadata 9 0 R
|
||||||
|
>>
|
||||||
|
endobj
|
||||||
|
8 0 obj
|
||||||
|
<</R7
|
||||||
|
7 0 R>>
|
||||||
|
endobj
|
||||||
|
7 0 obj
|
||||||
|
<</BaseFont/Helvetica/Type/Font
|
||||||
|
/Subtype/Type1>>
|
||||||
|
endobj
|
||||||
|
9 0 obj
|
||||||
|
<</Type/Metadata
|
||||||
|
/Subtype/XML/Length 1173>>stream
|
||||||
|
<?xpacket begin='' id='W5M0MpCehiHzreSzNTczkc9d'?>
|
||||||
|
<?adobe-xap-filters esc="CRLF"?>
|
||||||
|
<x:xmpmeta xmlns:x='adobe:ns:meta/' x:xmptk='XMP toolkit 2.9.1-13, framework 1.6'>
|
||||||
|
<rdf:RDF xmlns:rdf='http://www.w3.org/1999/02/22-rdf-syntax-ns#' xmlns:iX='http://ns.adobe.com/iX/1.0/'>
|
||||||
|
<rdf:Description rdf:about="" xmlns:pdf='http://ns.adobe.com/pdf/1.3/' pdf:Producer='GPL Ghostscript 10.02.1'/>
|
||||||
|
<rdf:Description rdf:about="" xmlns:xmp='http://ns.adobe.com/xap/1.0/'><xmp:ModifyDate>2026-09-16T17:29:12Z</xmp:ModifyDate>
|
||||||
|
<xmp:CreateDate>2026-09-16T17:29:12Z</xmp:CreateDate>
|
||||||
|
<xmp:CreatorTool>UnknownApplication</xmp:CreatorTool></rdf:Description>
|
||||||
|
<rdf:Description rdf:about="" xmlns:xapMM='http://ns.adobe.com/xap/1.0/mm/' xapMM:DocumentID='uuid:ae8d7575-ea10-11fc-0000-3a98742e0b71'/>
|
||||||
|
<rdf:Description rdf:about="" xmlns:dc='http://purl.org/dc/elements/1.1/' dc:format='application/pdf'><dc:title><rdf:Alt><rdf:li xml:lang='x-default'>Untitled</rdf:li></rdf:Alt></dc:title></rdf:Description>
|
||||||
|
</rdf:RDF>
|
||||||
|
</x:xmpmeta>
|
||||||
|
|
||||||
|
|
||||||
|
<?xpacket end='w'?>
|
||||||
|
endstream
|
||||||
|
endobj
|
||||||
|
2 0 obj
|
||||||
|
<</Producer(GPL Ghostscript 10.02.1)
|
||||||
|
/CreationDate(D:20260916172912Z00'00')
|
||||||
|
/ModDate(D:20260916172912Z00'00')>>endobj
|
||||||
|
xref
|
||||||
|
0 10
|
||||||
|
0000000000 65535 f
|
||||||
|
0000000476 00000 n
|
||||||
|
0000001882 00000 n
|
||||||
|
0000000417 00000 n
|
||||||
|
0000000276 00000 n
|
||||||
|
0000000077 00000 n
|
||||||
|
0000000257 00000 n
|
||||||
|
0000000569 00000 n
|
||||||
|
0000000540 00000 n
|
||||||
|
0000000633 00000 n
|
||||||
|
trailer
|
||||||
|
<< /Size 10 /Root 1 0 R /Info 2 0 R
|
||||||
|
/ID [<5A562476FBF11852133AA873452909F9><5A562476FBF11852133AA873452909F9>]
|
||||||
|
>>
|
||||||
|
startxref
|
||||||
|
2008
|
||||||
|
%%EOF
|
||||||
|
1 0 obj
|
||||||
|
<</Type /Catalog /Pages 3 0 R /Metadata 9 0 R /Names <</EmbeddedFiles 12 0 R >> >>
|
||||||
|
endobj
|
||||||
|
10 0 obj
|
||||||
|
<</Length 44 /Params <</Size 44 >> >> stream
|
||||||
|
gooattachmenttoken embedded attachment text
|
||||||
|
|
||||||
|
endstream
|
||||||
|
|
||||||
|
endobj
|
||||||
|
11 0 obj
|
||||||
|
<</Type /Filespec /UF <feff0067006f006f002d006e006f00740065002e007400780074> /EF <</F 10 0 R >> >>
|
||||||
|
endobj
|
||||||
|
12 0 obj
|
||||||
|
<</Names [<feff0067006f006f002d006e006f00740065002e007400780074> 11 0 R ] >>
|
||||||
|
endobj
|
||||||
|
xref
|
||||||
|
0 2
|
||||||
|
0000000002 65535 f
|
||||||
|
0000002361 00000 n
|
||||||
|
10 3
|
||||||
|
0000002463 00000 n
|
||||||
|
0000002586 00000 n
|
||||||
|
0000002705 00000 n
|
||||||
|
trailer
|
||||||
|
<</Size 13 /ID [(ZV$vûñR:¨sE\) ù) (P…õrˆ‚ Åð¡Ï) ] /Root 1 0 R /Prev 2008 /Info 2 0 R >>
|
||||||
|
startxref
|
||||||
|
2802
|
||||||
Reference in New Issue
Block a user