diff --git a/.env.example b/.env.example index df586af..fe4bb8b 100644 --- a/.env.example +++ b/.env.example @@ -55,3 +55,8 @@ ONLYOFFICE_PROJECT_ID=33 # ONLYOFFICE_PG_PASSWORD= # ONLYOFFICE_PG_DBNAME=onlyoffice # ONLYOFFICE_PG_SSLMODE=disable + +# oo index / oo search --backend own — own full-text index for PDF/scans, +# filled by `oo index` from internal/docpipe (pdftotext + OCR). Defaults to +# oo_docs_text. Uses the same ONLYOFFICE_ES_URL. +# ONLYOFFICE_ES_TEXT_INDEX=oo_docs_text diff --git a/README.md b/README.md index 1458a5f..bd97800 100644 --- a/README.md +++ b/README.md @@ -697,6 +697,27 @@ oo search "Rechnung" --json # shorthand for -o json Requires `ONLYOFFICE_ES_URL` (plus optional `ONLYOFFICE_ES_INDEX`, `ONLYOFFICE_TENANT`). +#### PDF/scans: own index (`oo index` + `--backend own`) + +The OnlyOffice index covers Office formats only, so PDFs (`S1019`-style invoice +numbers) are not searchable by content. `oo index` extracts PDF text with +`internal/docpipe` (pdftotext, OCR for scans) — including the text of embedded +PDF attachments (`pdfdetach`: `.md`, `.xml`, covers the original/scan and +ZUGFeRD e-invoice XML) — into a separate index (`ONLYOFFICE_ES_TEXT_INDEX`, +default `oo_docs_text`); the OnlyOffice server and its index are **not** +modified. Then search it with `--backend own`. + +```bash +oo index folder 634 --recursive --exts pdf # populate (idempotent upsert) +oo index files 3576 3578 # specific files +oo index folder 634 --dry-run # plan only +oo search "S1021" --content --backend own # finds the PDF +oo search "Rechnung" --backend own --folder 634 --json +``` + +See [`docs/elasticsearch.md`](docs/elasticsearch.md) for the decision and +trade-offs. + ### Bulk tools (`cmd/`) Small single-purpose binaries for bulk Documents work. All of them pace @@ -733,7 +754,8 @@ kontolink IN.xlsx oo-index.tsv OUT.xlsx [FILE_ID] [AMOUNTS_TSV] | `docs` | `tools`, `convert`, `optimize`, `ocr`, `hocr`, `as-md`, `put-md`, `put-txt`, `put-xlsx` | | `catalog` | `match`, `merge`, `apply`, `scan-contacts`, `scan-projects`, `scan-thunderbird` | | `dav` | `ls`, `move`, `copy`, `mkdir`, `rename-file`, `rename-folder`, `download`, `fileops` | -| `search` | `QUERY` (`--content`, `--folder ID`, `--limit N`, `--json`) | +| `search` | `QUERY` (`--content`, `--folder ID`, `--limit N`, `--backend oo\|own`, `--json`) | +| `index` | `folder FOLDER_ID`, `files FILE_ID...` (`--recursive`, `--exts pdf`, `--limit N`, `--dry-run`) | The CLI reads only `.env` from the current working directory (godotenv is a CLI-only concern — the library itself never loads dotfiles). diff --git a/cmd/oo/index.go b/cmd/oo/index.go new file mode 100644 index 0000000..82c13d1 --- /dev/null +++ b/cmd/oo/index.go @@ -0,0 +1,158 @@ +package main + +import ( + "strings" + + onlyoffice "github.com/eslider/go-onlyoffice" + "github.com/spf13/cobra" +) + +func init() { + rootCmd.AddCommand(indexCmd()) +} + +// indexFlags are shared by the `oo index folder` and `oo index files` verbs. +type indexFlags struct { + recursive bool + exts string + limit int + workers int + lang string + minChars int + workDir string + backend string + dryRun bool + asJSON bool +} + +// indexCmd populates the own full-text index (oo_docs_text) that makes PDF and +// scanned content searchable. The OnlyOffice index is left untouched. +func indexCmd() *cobra.Command { + f := &indexFlags{} + cmd := &cobra.Command{ + Use: "index", + Short: "Populate the own full-text index for PDF/scan content", + Long: "Index document text that the OnlyOffice Elasticsearch index does not\n" + + "cover (PDFs and scans) into a separate index (ONLYOFFICE_ES_TEXT_INDEX,\n" + + "default oo_docs_text). Text is extracted with docpipe (pdftotext, OCR)\n" + + "and the OnlyOffice server is never modified.\n\n" + + "Requires ONLYOFFICE_URL/USER/PASS (to download files) and ONLYOFFICE_ES_URL\n" + + "(to write the index). See docs/elasticsearch.md.", + } + cmd.PersistentFlags().BoolVar(&f.recursive, "recursive", false, "folder: descend into subfolders") + cmd.PersistentFlags().StringVar(&f.exts, "exts", "pdf", "comma-separated extensions to index") + cmd.PersistentFlags().IntVar(&f.limit, "limit", 0, "maximum number of files to index (0 = all)") + cmd.PersistentFlags().IntVar(&f.workers, "workers", 3, "parallel downloads/extractions") + cmd.PersistentFlags().StringVar(&f.lang, "lang", "deu+eng", "OCR language(s)") + cmd.PersistentFlags().IntVar(&f.minChars, "min-chars", 0, "text-layer threshold below which OCR runs") + cmd.PersistentFlags().StringVar(&f.workDir, "work-dir", "", "temp dir for downloads (default: system temp)") + cmd.PersistentFlags().StringVar(&f.backend, "backend", "rest", "file backend: rest|dav") + cmd.PersistentFlags().BoolVar(&f.dryRun, "dry-run", false, "list what would be indexed, without changes") + cmd.PersistentFlags().BoolVar(&f.asJSON, "json", false, "shorthand for --output json") + + cmd.AddCommand(indexFolderCmd(f), indexFilesCmd(f)) + return cmd +} + +func indexFolderCmd(f *indexFlags) *cobra.Command { + return &cobra.Command{ + Use: "folder FOLDER_ID", + Short: "Index every matching file in a Documents folder", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runIndex(cmd, f, args[0], nil) + }, + } +} + +func indexFilesCmd(f *indexFlags) *cobra.Command { + return &cobra.Command{ + Use: "files FILE_ID...", + Short: "Index specific Documents files", + Args: cobra.MinimumNArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runIndex(cmd, f, "", args) + }, + } +} + +func runIndex(cmd *cobra.Command, f *indexFlags, folderID string, ids []string) error { + if f.asJSON { + outputFormat = "json" + } + c, err := newOO(cmd) + if err != nil { + return err + } + idx, err := onlyoffice.NewESTextIndex(onlyoffice.ESTextConfigFromEnv()) + if err != nil { + return err + } + ti := onlyoffice.NewTextIndexer(c.FileStore(f.backend), idx) + ti.WorkDir = f.workDir + opts := onlyoffice.IndexOptions{ + Recursive: f.recursive, + Extensions: splitList(f.exts), + Limit: f.limit, + Lang: f.lang, + MinChars: f.minChars, + Workers: f.workers, + } + ctx := cmd.Context() + + if f.dryRun { + var entries []onlyoffice.Entry + if folderID != "" { + entries, err = ti.PlanFolder(ctx, folderID, opts) + } else { + entries, err = ti.PlanFiles(ctx, ids, opts) + } + if err != nil { + return err + } + rows := make([]map[string]any, 0, len(entries)) + for _, e := range entries { + rows = append(rows, map[string]any{ + "id": e.ID, + "title": e.Title, + "folder": e.ParentID, + }) + } + printTable([]string{"id", "title", "folder"}, rows) + return nil + } + + if err := ti.Ensure(ctx); err != nil { + return err + } + var res onlyoffice.IndexResult + if folderID != "" { + res, err = ti.IndexFolder(ctx, folderID, opts) + } else { + res, err = ti.IndexFiles(ctx, ids, opts) + } + if err != nil { + return err + } + printObject(map[string]any{ + "index": idx.Index(), + "scanned": res.Scanned, + "indexed": res.Indexed, + "skipped": res.Skipped, + "failed": res.Failed, + "errors": res.Errors, + }) + return nil +} + +// splitList parses a comma-separated flag value, dropping blanks. +func splitList(s string) []string { + parts := strings.Split(s, ",") + out := make([]string, 0, len(parts)) + for _, p := range parts { + if p = strings.TrimSpace(p); p != "" { + out = append(out, p) + } + } + return out +} diff --git a/cmd/oo/index_test.go b/cmd/oo/index_test.go new file mode 100644 index 0000000..b6f2805 --- /dev/null +++ b/cmd/oo/index_test.go @@ -0,0 +1,55 @@ +package main + +import ( + "reflect" + "strings" + "testing" +) + +func TestIndexCommandRegistered(t *testing.T) { + cmd, _, err := rootCmd.Find([]string{"index"}) + if err != nil { + t.Fatal(err) + } + if cmd.Name() != "index" { + t.Fatalf("index resolved to %q", cmd.Name()) + } + for _, name := range []string{"exts", "limit", "workers", "lang", "min-chars", "work-dir", "backend", "dry-run", "recursive", "json"} { + if cmd.PersistentFlags().Lookup(name) == nil { + t.Errorf("index: missing --%s flag", name) + } + } + if cmd.PersistentFlags().Lookup("exts").DefValue != "pdf" { + t.Errorf("--exts default = %q, want pdf", cmd.PersistentFlags().Lookup("exts").DefValue) + } + for _, verb := range []string{"index folder", "index files"} { + sub, _, err := rootCmd.Find(strings.Fields(verb)) + if err != nil { + t.Fatalf("%s: %v", verb, err) + } + if sub.Name() != strings.Fields(verb)[1] { + t.Errorf("%s resolved to %q", verb, sub.Name()) + } + } +} + +func TestIndexSearchBackendFlag(t *testing.T) { + cmd, _, err := rootCmd.Find([]string{"search"}) + if err != nil { + t.Fatal(err) + } + if cmd.Flags().Lookup("backend") == nil { + t.Fatal("search: missing --backend flag") + } + if cmd.Flags().Lookup("backend").DefValue != "oo" { + t.Errorf("--backend default = %q, want oo", cmd.Flags().Lookup("backend").DefValue) + } +} + +func TestSplitList(t *testing.T) { + got := splitList(" pdf , .PDF, docx ,, ") + want := []string{"pdf", ".PDF", "docx"} + if !reflect.DeepEqual(got, want) { + t.Errorf("splitList = %v, want %v", got, want) + } +} diff --git a/cmd/oo/main.go b/cmd/oo/main.go index 82ab8e8..1933d9d 100644 --- a/cmd/oo/main.go +++ b/cmd/oo/main.go @@ -18,7 +18,8 @@ // oo docs tools | convert | optimize | ocr | hocr | as-md | put-md | put-txt | put-xlsx // oo catalog match | merge | apply | scan-contacts | scan-projects | scan-thunderbird // oo dav ls | move | copy | mkdir | rename-file | rename-folder | download | fileops -// oo search QUERY [--content] [--folder ID] [--limit N] [--json] +// oo search QUERY [--content] [--folder ID] [--limit N] [--backend oo|own] [--json] +// oo index folder FOLDER_ID | files FILE_ID... [--recursive] [--exts pdf] [--dry-run] // // CRM association rules: docs/crm-associations.md // diff --git a/cmd/oo/search.go b/cmd/oo/search.go index 5da8309..433aac9 100644 --- a/cmd/oo/search.go +++ b/cmd/oo/search.go @@ -1,6 +1,9 @@ package main import ( + "fmt" + "strings" + onlyoffice "github.com/eslider/go-onlyoffice" "github.com/eslider/go-onlyoffice/cmd/internal/bootstrap" "github.com/spf13/cobra" @@ -18,6 +21,7 @@ func searchCmd() *cobra.Command { content bool folder string limit int + backend string asJSON bool ) cmd := &cobra.Command{ @@ -27,6 +31,9 @@ func searchCmd() *cobra.Command { "By default only file names are matched. With --content the query also\n" + "matches extracted document text (document.attachment.content); this covers\n" + "Office formats (docx/xlsx/pptx) and is slower.\n\n" + + "--backend own queries the separate index populated by `oo index`\n" + + "(ONLYOFFICE_ES_TEXT_INDEX, default oo_docs_text) instead, which also holds\n" + + "PDFs and scans (see docs/elasticsearch.md).\n\n" + "Requires ONLYOFFICE_ES_URL (and optionally ONLYOFFICE_ES_INDEX,\n" + "ONLYOFFICE_TENANT). See docs/elasticsearch.md for the tunnel setup.", Args: cobra.ExactArgs(1), @@ -36,7 +43,18 @@ func searchCmd() *cobra.Command { } bootstrap.LoadEnv() c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials()) - searcher, err := c.Files().Search() + var ( + searcher onlyoffice.Searcher + err error + ) + switch strings.ToLower(strings.TrimSpace(backend)) { + case "", "oo", "elasticsearch": + searcher, err = c.Files().Search() + case "own", "es-text": + searcher, err = onlyoffice.NewESTextIndex(onlyoffice.ESTextConfigFromEnv()) + default: + return fmt.Errorf("unknown search backend %q (want oo|own)", backend) + } if err != nil { return err } @@ -70,6 +88,7 @@ func searchCmd() *cobra.Command { cmd.Flags().BoolVar(&content, "content", false, "also match extracted document content") cmd.Flags().StringVar(&folder, "folder", "", "limit to a Documents folder id") cmd.Flags().IntVar(&limit, "limit", 20, "maximum number of results") + cmd.Flags().StringVar(&backend, "backend", "oo", "index to query: oo (OnlyOffice) | own (oo index)") cmd.Flags().BoolVar(&asJSON, "json", false, "shorthand for --output json") return cmd } diff --git a/docs/elasticsearch.md b/docs/elasticsearch.md index 05df5d6..a918b67 100644 --- a/docs/elasticsearch.md +++ b/docs/elasticsearch.md @@ -121,5 +121,135 @@ ONLYOFFICE_ES_URL=http://127.0.0.1:9200 ONLYOFFICE_TENANT=1 \ - `locale`/версия ES: 7.16.3, `_search` совместим с REST 7.x. - ES без auth и слушает только localhost — туннель обязателен. - Фильтр `tenantId` сузит выдачу; без него видны документы всех тенантов. -- Поиск по содержимому PDF, залитых через API, не работает (нет - `attachment.content`) — только Office-форматы. +- Поиск по содержимому PDF в индексе OnlyOffice не работает (для PDF нет + `attachment.content`) — только Office-форматы. Решение для PDF — свой индекс + `oo_docs_text` (F6 #42), см. ниже. + +# PDF и сканы — свой индекс (F6 #42) + +## Проблема + +`oo search --content "S1019"` не находил номер внутри PDF-счёта: в индексе +OnlyOffice PDF лежит только по имени. + +## Почему PDF исключён (исходники CommunityServer) + +Разобрано в `ONLYOFFICE/CommunityServer`: + +- `web/core/ASC.Web.Core/Files/FileUtility.cs` — `CanIndex(fileName)` читает + серверную настройку `files.index.formats` (в `web/studio/ASC.Web.Studio/web.appsettings.config` + значение по умолчанию `".pptx|.xlsx|.docx"`). +- `web/studio/ASC.Web.Studio/Products/Files/Core/Search/FilesWrapper.cs` — + `GetDocumentStream*` возвращает `null`, если `!FileUtility.CanIndex(Title)`, + файл зашифрован или больше `MaxFileSize`. +- `module/ASC.ElasticSearch/Core/WrapperWithDoc.cs` + mapping в `Wrapper.cs` — + маппинг `document.attachment.content` и ingest-pipeline `attachments` + формат-агностичны: они распарсят любой поток. + +Вывод: PDF исключён **только настройкой** `files.index.formats`; жёсткого +ограничения на формат в коде нет. + +## Варианты и решение + +| # | Вариант | Оценка | +|---|---------|--------| +| a | Включить `.pdf` в `files.index.formats` + reindex | Правка сервера OO; настройка может потеряться при обновлении; полный reindex 39k док-в; Tika **не OCR** — сканы без текстового слоя дадут пустой контент. Отклонён без решения PO. | +| b | Server-side ingest/attachment для PDF | По факту то же, что (a): сервер кормит поток только для `CanIndex`. | +| c | **Свой индекс** `oo_docs_text`, наполняемый `internal/docpipe` | **Выбран.** Сервер OO не трогаем; детерминированно; работает OCR для сканов; независимо от обновлений OO; любые форматы; фильтры папка/тип. | +| d | Локальный поиск без индекса | Отклонён как основной: качаем и извлекаем на каждый запрос, нет выдачи/ранжирования/highlight. | + +Итог: **вариант c**. Индекс OnlyOffice (`files_file`) не изменяется; наш +индекс живёт рядом. + +## Устройство + +- `file_es_text.go` — `ESTextIndex` (`Name() = "es-text"`): + `Ensure` (создаёт индекс с явным маппингом), `Put` (bulk, `refresh`), + `Delete` (по `id`), `Search` (`multi_match` по `title^2` + `content`, + фильтры `folder`/`ext`, highlight). +- `file_text_index.go` — `TextIndexer`: листает папки (`FileStore.List`), + качает файлы (`FileStore.Download`), извлекает текст через + `internal/docpipe` (`pdftotext`, для сканов — `ocrmypdf`/`tesseract`), + пишет в `TextIndex`. Пул воркеров (по умолчанию 3). +- CLI: `oo index folder|files` наполняет индекс; `oo search --backend own` + ищет по нему. + +### Встроенные вложения PDF + +Оцифрованные PDF несут вложения (`.md` — текст/таблицы скана, +`.yaml`/`.json` — метаданные, `.xml` — EN 16931 CII eRechnung, +`factur-x.xml` у ZUGFeRD; см. `office-assistant/docs/reference/document-metadata.md`). +`TextIndexer` обходит их: `pdfdetach -list` перечисляет, `-save` сохраняет, +каждое вложение проходит штатный `docpipe.ToMarkdown` (PDF/картинки → OCR, +`.md`/`.txt` — как есть). Форматы, которые docpipe не конвертирует +(`.xml`/`.html` — снимаются теги; `.json`/`.csv` — как текст), извлекаются +текстом; нечитаемые — пропускаются. + +Текст склеивается: тело, затем по секции на вложение с маркером +`[attachment: <имя>]` (функция `docpipe.JoinWithAttachments`). Индекс — тот же +`file_id`, upsert идемпотентен. Нет вложений или pdfdetach/формат нечитаем — +индексируется тело (без падения). + +Поля `oo_docs_text`: + +| поле | тип | смысл | +|------|-----|-------| +| `id` | keyword | id файла Documents | +| `title` | text (+`.keyword`) | имя файла | +| `folder` | keyword | id папки | +| `ext` | keyword | расширение | +| `content` | text | извлечённый текст (pdftotext/OCR) | + +## CLI + +```bash +set -a; . .env; set +a # ONLYOFFICE_URL/USER/PASS + ONLYOFFICE_ES_URL +oo index folder 634 # PDF в папке 634 +oo index folder 634 --recursive --exts pdf,png --limit 100 +oo index files 3576 3578 # точечно +oo index folder 634 --dry-run # показать план, ничего не менять + +oo search "S1021" --content --backend own +oo search "S1021" --backend own --folder 634 --json +``` + +`--backend` у `oo search`: `oo` (по умолчанию, индекс OnlyOffice) или `own` +(наш `ONLYOFFICE_ES_TEXT_INDEX`). + +## Переменные (дополнение) + +| env | default | смысл | +|-----|---------|-------| +| `ONLYOFFICE_ES_TEXT_INDEX` | `oo_docs_text` | индекс своего конвейера | + +`ONLYOFFICE_ES_URL` — общий для обоих индексов. + +## Тесты + +```bash +go test -run 'ESText|TextIndexer|Index' ./ ./cmd/oo/ # unit, без сети +ONLYOFFICE_ES_URL=http://127.0.0.1:9200 \ + go test -tags=integration -run TestIntegrationESTextIndex -v . +``` + +Интеграционный тест создаёт временный индекс, наполняет, ищет по контенту, +проверяет фильтры и удаление, затем удаляет индекс; +`TestIntegrationESTextIndexPDFAttachment` индексирует +`testdata/pdf-with-attachment.pdf` реальным конвейером (pdfdetach + pdftotext) +и ищет токен, лежащий только во вложении. Unit-тесты используют +fake-store/fake-extractor и не требуют pdftotext/OCR (парсер списка, склейка +`JoinWithAttachments`, снятие тегов `xmlToText` — чистые). + +## Грабли + +- Наполнение — ручное (`oo index`); после изменения/добавления PDF повтори. + Повтор идемпотентен (upsert по id файла). +- В индексе ищется только то, что проиндексировано; `oo index` качает каждый + файл и (для сканов) гоняет OCR — это медленно, отсюда `--limit`/`--exts`. +- `folder` фильтруется как id папки, а не как путь. +- Дубликаты (напр. `S1055.pdf` и `2026-08-20-S1055-…`) дадут несколько строк — + это ожидаемо, дедуп — на стороне потребителя. +- Вложения: нужен `pdfdetach` (poppler); если его нет — индексируется только + тело. Вложенный PDF/картинка с плохим текстовым слоем проходит OCR, это + медленно. `.json`-метаданные (CuraSoft) индексируются как текст и могут + добавить шумовых токенов. diff --git a/file_es.go b/file_es.go index 9c2db76..5c286c3 100644 --- a/file_es.go +++ b/file_es.go @@ -188,6 +188,7 @@ type esBool struct { type esClause struct { MultiMatch *esMultiMatch `json:"multi_match,omitempty"` Term map[string]any `json:"term,omitempty"` + Terms map[string]any `json:"terms,omitempty"` Wildcard map[string]any `json:"wildcard,omitempty"` } @@ -268,9 +269,10 @@ func parseESSearchResponse(raw []byte) ([]SearchHit, error) { var esHighlightTag = regexp.MustCompile(`]*>`) // esHighlightText flattens a highlight map into one plain-text snippet, -// preferring the content fragment over the title. +// preferring the content fragment over the title. It covers both the +// OnlyOffice content field and the own-index "content" field. func esHighlightText(hl map[string][]string) string { - for _, key := range []string{"document.attachment.content", "title"} { + for _, key := range []string{"document.attachment.content", "content", "title"} { frags := hl[key] if len(frags) == 0 { continue diff --git a/file_es_text.go b/file_es_text.go new file mode 100644 index 0000000..072f00c --- /dev/null +++ b/file_es_text.go @@ -0,0 +1,336 @@ +package onlyoffice + +// Own full-text index (epic #34, F6 #42). +// +// The OnlyOffice Elasticsearch index (files_file) only holds extracted content +// for Office formats. FileUtility.CanIndex gates extraction by the server +// setting files.index.formats, whose default is ".pptx|.xlsx|.docx", so PDFs +// are indexed by name only. Instead of patching the server (risky: lost on +// upgrade, forces a full reindex) this file implements a second, independent +// index (default oo_docs_text) that our own pipeline fills from +// internal/docpipe (pdftotext + OCR). The OnlyOffice index is never touched. +// +// See docs/elasticsearch.md for the decision and the trade-offs. + +import ( + "bytes" + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "os" + "strings" + "time" +) + +const defaultESTextIndex = "oo_docs_text" + +// ESTextConfig configures the own full-text index. +type ESTextConfig struct { + URL string // scheme://host:port of the ES HTTP endpoint + Index string // index name, default oo_docs_text + Tenant string // reserved for future multi-tenant data; unused for now +} + +// ESTextConfigFromEnv reads ONLYOFFICE_ES_URL and ONLYOFFICE_ES_TEXT_INDEX +// (default oo_docs_text). The library never loads dotfiles — the CLI does that. +func ESTextConfigFromEnv() ESTextConfig { + return ESTextConfig{ + URL: strings.TrimRight(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")), "/"), + Index: firstNonEmpty(os.Getenv("ONLYOFFICE_ES_TEXT_INDEX"), defaultESTextIndex), + Tenant: strings.TrimSpace(os.Getenv("ONLYOFFICE_TENANT")), + } +} + +// TextDoc is one document in the own full-text index. It is keyed by the +// OnlyOffice file id so hits map straight back to Documents entries. +type TextDoc struct { + ID string `json:"id"` + Title string `json:"title"` + FolderID string `json:"folder,omitempty"` + Ext string `json:"ext,omitempty"` + Content string `json:"content"` +} + +// TextIndex is the storage/search surface for locally extracted document text. +// It complements Searcher: ESSearcher reads OnlyOffice's index, ESTextIndex +// reads ours. +type TextIndex interface { + Put(ctx context.Context, docs []TextDoc) error + Delete(ctx context.Context, ids []string) error + Search(ctx context.Context, q SearchQuery) ([]SearchHit, error) + Name() string +} + +// ESTextIndex is a TextIndex (and Searcher) over a dedicated Elasticsearch +// index filled by TextIndexer. +type ESTextIndex struct { + cfg ESTextConfig + http *http.Client +} + +var ( + _ TextIndex = (*ESTextIndex)(nil) + _ Searcher = (*ESTextIndex)(nil) +) + +// NewESTextIndex returns a searcher/writer for the own full-text index. The URL +// is required; an empty index falls back to oo_docs_text. +func NewESTextIndex(cfg ESTextConfig) (*ESTextIndex, error) { + if strings.TrimSpace(cfg.URL) == "" { + return nil, fmt.Errorf("onlyoffice: elasticsearch URL is empty (set ONLYOFFICE_ES_URL)") + } + cfg.URL = strings.TrimRight(cfg.URL, "/") + if cfg.Index == "" { + cfg.Index = defaultESTextIndex + } + return &ESTextIndex{cfg: cfg, http: &http.Client{Timeout: 120 * time.Second}}, nil +} + +// Name implements Searcher and TextIndex. +func (x *ESTextIndex) Name() string { return "es-text" } + +// Index returns the configured index name. +func (x *ESTextIndex) Index() string { return x.cfg.Index } + +// esTextMapping pins explicit types: content must stay a plain text field (no +// keyword subfield) and folder/ext stay exact keywords for filters. +const esTextMapping = `{ + "mappings": { + "properties": { + "id": {"type": "keyword"}, + "title": {"type": "text", "fields": {"keyword": {"type": "keyword", "ignore_above": 512}}}, + "folder": {"type": "keyword"}, + "ext": {"type": "keyword"}, + "content": {"type": "text"} + } + } +}` + +// Ensure creates the index with the explicit mapping. A missing index is +// created; an already existing one is left untouched. +func (x *ESTextIndex) Ensure(ctx context.Context) error { + status, raw, err := x.do(ctx, http.MethodPut, "/"+x.cfg.Index, []byte(esTextMapping), "application/json") + if err != nil { + return err + } + if status == http.StatusOK { + return nil + } + if status == http.StatusBadRequest && bytes.Contains(raw, []byte("resource_already_exists_exception")) { + return nil + } + return fmt.Errorf("onlyoffice: create text index %s: %d %s", x.cfg.Index, status, truncate(string(raw), 300)) +} + +// Put upserts documents via the bulk API and refreshes so they are immediately +// searchable. +func (x *ESTextIndex) Put(ctx context.Context, docs []TextDoc) error { + if len(docs) == 0 { + return nil + } + status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_bulk?refresh=true", esTextBulkBody(docs), "application/x-ndjson") + if err != nil { + return err + } + if status >= 400 { + return fmt.Errorf("onlyoffice: bulk index %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400)) + } + var res esBulkResponse + if err := json.Unmarshal(raw, &res); err != nil { + return fmt.Errorf("onlyoffice: decode bulk response: %w", err) + } + if !res.Errors { + return nil + } + return fmt.Errorf("onlyoffice: bulk index %s: %s", x.cfg.Index, res.firstError()) +} + +// Delete removes documents by OnlyOffice file id. A missing index means there +// is nothing to delete. +func (x *ESTextIndex) Delete(ctx context.Context, ids []string) error { + if len(ids) == 0 { + return nil + } + body, err := json.Marshal(map[string]any{"query": map[string]any{"terms": map[string]any{"id": ids}}}) + if err != nil { + return err + } + status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_delete_by_query?refresh=true", body, "application/json") + if err != nil { + return err + } + if status == http.StatusNotFound { + return nil + } + if status >= 400 { + return fmt.Errorf("onlyoffice: delete from %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400)) + } + return nil +} + +// Search runs a multi_match over title (boosted) and content, with optional +// folder and extension filters. A missing index yields no hits, not an error. +func (x *ESTextIndex) Search(ctx context.Context, q SearchQuery) ([]SearchHit, error) { + q.Text = strings.TrimSpace(q.Text) + if q.Text == "" { + return nil, fmt.Errorf("onlyoffice: empty search query") + } + body, err := json.Marshal(esTextSearchRequest(q)) + if err != nil { + return nil, fmt.Errorf("onlyoffice: build elasticsearch query: %w", err) + } + status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_search", body, "application/json") + if err != nil { + return nil, err + } + if status == http.StatusNotFound { + return nil, nil + } + if status >= 400 { + return nil, fmt.Errorf("onlyoffice: search %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400)) + } + return parseESTextResponse(raw) +} + +// do sends one request and returns the status and body (bounded). The caller +// decides which statuses are errors. +func (x *ESTextIndex) do(ctx context.Context, method, path string, body []byte, contentType string) (int, []byte, error) { + var r io.Reader + if body != nil { + r = bytes.NewReader(body) + } + req, err := http.NewRequestWithContext(ctx, method, x.cfg.URL+path, r) + if err != nil { + return 0, nil, err + } + req.Header.Set("Accept", "application/json") + if contentType != "" { + req.Header.Set("Content-Type", contentType) + } + resp, err := x.http.Do(req) + if err != nil { + return 0, nil, fmt.Errorf("onlyoffice: elasticsearch %s: %w", method, err) + } + defer resp.Body.Close() + raw, err := io.ReadAll(io.LimitReader(resp.Body, maxESResponseSize)) + if err != nil { + return resp.StatusCode, nil, err + } + return resp.StatusCode, raw, nil +} + +// esTextBulkBody renders the NDJSON bulk payload. Pure, so it is unit-tested. +func esTextBulkBody(docs []TextDoc) []byte { + var b bytes.Buffer + enc := json.NewEncoder(&b) + enc.SetEscapeHTML(false) + for _, d := range docs { + _ = enc.Encode(map[string]any{"index": map[string]any{"_id": d.ID}}) + _ = enc.Encode(d) + } + return b.Bytes() +} + +// esTextSearchRequest builds the own-index query. Pure, so it is unit-tested. +func esTextSearchRequest(q SearchQuery) esRequest { + limit := q.Limit + if limit <= 0 { + limit = defaultESLimit + } + if limit > maxESLimit { + limit = maxESLimit + } + fields := []string{"title^2", "content"} + must := []esClause{{MultiMatch: &esMultiMatch{Query: q.Text, Fields: fields}}} + + var filter []esClause + if f := strings.TrimSpace(q.FolderID); f != "" { + filter = append(filter, esClause{Term: map[string]any{"folder": f}}) + } + if exts := normalizeExtensions(q.Extensions); len(exts) > 0 { + filter = append(filter, esClause{Terms: map[string]any{"ext": exts}}) + } + + return esRequest{ + Size: limit, + Source: []string{"id", "title", "folder", "ext"}, + Query: esQuery{Bool: esBool{Must: must, Filter: filter}}, + Highlight: esHighlight{PreTags: []string{""}, PostTags: []string{""}, Fields: map[string]struct{}{"title": {}, "content": {}}}, + } +} + +// esBulkResponse is the subset of an ES bulk response we consume. +type esBulkResponse struct { + Errors bool `json:"errors"` + Items []map[string]struct { + ID string `json:"_id"` + Status int `json:"status"` + Error *struct { + Type string `json:"type"` + Reason string `json:"reason"` + } `json:"error"` + } `json:"items"` +} + +// firstError returns a compact description of the first failed bulk item. +func (r esBulkResponse) firstError() string { + for _, item := range r.Items { + for op, res := range item { + if res.Error != nil { + return fmt.Sprintf("%s %s: %s %s", op, res.ID, res.Error.Type, res.Error.Reason) + } + } + } + return "unknown bulk error" +} + +// esTextResponse is the subset of an own-index search response we consume. +type esTextResponse struct { + Hits struct { + Total struct { + Value int `json:"value"` + } `json:"total"` + Hits []struct { + ID string `json:"_id"` + Score float64 `json:"_score"` + Source TextDoc `json:"_source"` + HL map[string][]string `json:"highlight"` + } `json:"hits"` + } `json:"hits"` +} + +// parseESTextResponse converts an own-index search response into SearchHit +// values. Pure, so it is unit-tested. +func parseESTextResponse(raw []byte) ([]SearchHit, error) { + var r esTextResponse + if err := json.Unmarshal(raw, &r); err != nil { + return nil, fmt.Errorf("onlyoffice: decode elasticsearch response: %w", err) + } + hits := make([]SearchHit, 0, len(r.Hits.Hits)) + for _, h := range r.Hits.Hits { + id := h.Source.ID + if id == "" { + id = h.ID + } + parent := h.Source.FolderID + var path []string + if parent != "" { + path = []string{parent} + } + hits = append(hits, SearchHit{ + Entry: Entry{ + ID: id, + ParentID: parent, + Title: h.Source.Title, + Kind: File, + Provider: "es-text", + }, + Score: h.Score, + Highlight: esHighlightText(h.HL), + Path: path, + }) + } + return hits, nil +} diff --git a/file_es_text_integration_test.go b/file_es_text_integration_test.go new file mode 100644 index 0000000..30d295e --- /dev/null +++ b/file_es_text_integration_test.go @@ -0,0 +1,158 @@ +//go:build integration + +package onlyoffice + +import ( + "context" + "net/http" + "os" + "strings" + "testing" + "time" + + "github.com/eslider/go-onlyoffice/internal/docpipe" +) + +// TestIntegrationESTextIndex verifies the own full-text index end to end +// against a live Elasticsearch: create the index with its mapping, index a +// document, find it by content (and reject it via a folder filter and after +// deletion), then drop the throwaway index. +// +// Requires ONLYOFFICE_ES_URL (a reachable ES endpoint — in the current setup a +// tunnel to the ES inside the OnlyOffice VM, see docs/elasticsearch.md). It +// does not need OnlyOffice credentials because no file is downloaded: the +// TextIndexer write path is covered by unit tests with a fake extractor. +func TestIntegrationESTextIndex(t *testing.T) { + esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")) + if esURL == "" { + t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test") + } + stamp := time.Now().UTC().Format("20060102150405") + idx, err := NewESTextIndex(ESTextConfig{URL: esURL, Index: "oo_docs_text_it_" + stamp}) + if err != nil { + t.Fatalf("NewESTextIndex: %v", err) + } + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute) + defer cancel() + + t.Cleanup(func() { + cleanupCtx, done := context.WithTimeout(context.Background(), 30*time.Second) + defer done() + _, _, _ = idx.do(cleanupCtx, http.MethodDelete, "/"+idx.Index(), nil, "") + }) + + if err := idx.Ensure(ctx); err != nil { + t.Fatalf("Ensure: %v", err) + } + // Ensure is idempotent. + if err := idx.Ensure(ctx); err != nil { + t.Fatalf("Ensure (second): %v", err) + } + + token := "gotes" + stamp + doc := TextDoc{ + ID: "3578", + Title: "2026-07-28-S1021-Edelweiss-rechnung.pdf", + FolderID: "634", + Ext: "pdf", + Content: "Begleitzettel SGB XI — Rechnung " + token, + } + if err := idx.Put(ctx, []TextDoc{doc}); err != nil { + t.Fatalf("Put: %v", err) + } + + hits, err := idx.Search(ctx, SearchQuery{Text: token}) + if err != nil { + t.Fatalf("Search: %v", err) + } + if len(hits) != 1 || hits[0].ID != "3578" { + t.Fatalf("content search hits = %+v, want doc 3578", hits) + } + if !strings.Contains(hits[0].Highlight, token) { + t.Errorf("highlight = %q, want token", hits[0].Highlight) + } + + if hits, err := idx.Search(ctx, SearchQuery{Text: token, FolderID: "999"}); err != nil { + t.Fatalf("Search with folder filter: %v", err) + } else if len(hits) != 0 { + t.Errorf("folder filter returned %d hits, want 0", len(hits)) + } + if hits, err := idx.Search(ctx, SearchQuery{Text: token, Extensions: []string{"docx"}}); err != nil { + t.Fatalf("Search with ext filter: %v", err) + } else if len(hits) != 0 { + t.Errorf("ext filter returned %d hits, want 0", len(hits)) + } + + if err := idx.Delete(ctx, []string{"3578"}); err != nil { + t.Fatalf("Delete: %v", err) + } + if hits, err := idx.Search(ctx, SearchQuery{Text: token}); err != nil { + t.Fatalf("Search after delete: %v", err) + } else if len(hits) != 0 { + t.Errorf("after delete search returned %d hits, want 0", len(hits)) + } +} + +// TestIntegrationESTextIndexPDFAttachment indexes testdata/pdf-with-attachment.pdf +// through the real pipeline (TextIndexer + docpipe: pdfdetach + pdftotext) and +// verifies that text living only in the embedded attachment is searchable. +// +// Requires ONLYOFFICE_ES_URL plus poppler (pdfdetach/pdftotext). No OnlyOffice +// credentials are needed: a fixture FileStore serves the PDF bytes. +func TestIntegrationESTextIndexPDFAttachment(t *testing.T) { + esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")) + if esURL == "" { + t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test") + } + if docpipe.LookPath().PDFDetach == "" { + t.Skip("pdfdetach not on PATH — skipping PDF attachment integration test") + } + pdf, err := os.ReadFile("testdata/pdf-with-attachment.pdf") + if err != nil { + t.Fatalf("read fixture: %v", err) + } + + stamp := time.Now().UTC().Format("20060102150405") + idx, err := NewESTextIndex(ESTextConfig{URL: esURL, Index: "oo_docs_text_it_att_" + stamp}) + if err != nil { + t.Fatalf("NewESTextIndex: %v", err) + } + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute) + defer cancel() + t.Cleanup(func() { + cleanupCtx, done := context.WithTimeout(context.Background(), 30*time.Second) + defer done() + _, _, _ = idx.do(cleanupCtx, http.MethodDelete, "/"+idx.Index(), nil, "") + }) + if err := idx.Ensure(ctx); err != nil { + t.Fatalf("Ensure: %v", err) + } + + store := &textFakeStore{files: map[string][]byte{"9001": pdf}} + ix := NewTextIndexer(store, idx) + res, err := ix.IndexEntries(ctx, []Entry{{ID: "9001", Title: "scan.pdf", ParentID: "777", Kind: File}}, IndexOptions{MinChars: 1}) + if err != nil { + t.Fatalf("IndexEntries: %v", err) + } + if res.Indexed != 1 || res.Failed != 0 { + t.Fatalf("result = %+v, want one indexed doc", res) + } + + // Token appears only inside the embedded goo-note.txt attachment. + hits, err := idx.Search(ctx, SearchQuery{Text: "gooattachmenttoken"}) + if err != nil { + t.Fatalf("Search attachment token: %v", err) + } + if len(hits) != 1 || hits[0].ID != "9001" { + t.Fatalf("attachment-token hits = %+v, want doc 9001", hits) + } + if !strings.Contains(hits[0].Highlight, "gooattachmenttoken") { + t.Errorf("highlight = %q, want attachment token", hits[0].Highlight) + } + // Body text is indexed as before. + if hits, err := idx.Search(ctx, SearchQuery{Text: "goobodytoken"}); err != nil { + t.Fatalf("Search body token: %v", err) + } else if len(hits) != 1 { + t.Errorf("body-token hits = %d, want 1", len(hits)) + } +} diff --git a/file_es_text_test.go b/file_es_text_test.go new file mode 100644 index 0000000..aaeeca3 --- /dev/null +++ b/file_es_text_test.go @@ -0,0 +1,275 @@ +package onlyoffice + +import ( + "context" + "fmt" + "io" + "os" + "reflect" + "strings" + "testing" +) + +func TestESTextSearchRequestShape(t *testing.T) { + got := esTextSearchRequest(SearchQuery{ + Text: "S1021", + FolderID: "634", + Extensions: []string{".PDF", "pdf"}, + Limit: 5, + }) + if got.Size != 5 { + t.Errorf("size = %d, want 5", got.Size) + } + mm := got.Query.Bool.Must[0].MultiMatch + if mm == nil || !reflect.DeepEqual(mm.Fields, []string{"title^2", "content"}) { + t.Fatalf("multi_match = %+v, want title^2 + content", mm) + } + if _, ok := got.Highlight.Fields["content"]; !ok { + t.Error("content highlight missing") + } + if _, ok := got.Highlight.Fields["title"]; !ok { + t.Error("title highlight missing") + } + var folder, exts int + for _, f := range got.Query.Bool.Filter { + switch { + case f.Term != nil && f.Term["folder"] != nil: + folder++ + if f.Term["folder"] != "634" { + t.Errorf("folder term = %+v", f.Term) + } + case f.Terms != nil: + exts++ + if !reflect.DeepEqual(f.Terms["ext"], []string{"pdf"}) { + t.Errorf("ext terms = %+v, want deduped pdf", f.Terms) + } + } + } + if folder != 1 || exts != 1 { + t.Errorf("filters folder=%d exts=%d, want 1 each", folder, exts) + } +} + +func TestESTextBulkBody(t *testing.T) { + body := esTextBulkBody([]TextDoc{ + {ID: "3578", Title: "S1021.pdf", FolderID: "634", Ext: "pdf", Content: "Begleitzettel & mehr"}, + {ID: "3579", Title: "S1023.pdf", Ext: "pdf", Content: "x"}, + }) + lines := strings.Split(strings.TrimRight(string(body), "\n"), "\n") + if len(lines) != 4 { + t.Fatalf("bulk body has %d lines, want 4:\n%s", len(lines), body) + } + if !strings.Contains(lines[0], `"index"`) || !strings.Contains(lines[0], `"_id":"3578"`) { + t.Errorf("action line = %q", lines[0]) + } + if !strings.Contains(lines[1], `"content":"Begleitzettel & mehr"`) { + t.Errorf("source line should keep HTML unescaped, got %q", lines[1]) + } + if !strings.Contains(lines[2], `"_id":"3579"`) { + t.Errorf("second action line = %q", lines[2]) + } +} + +func TestParseESTextResponse(t *testing.T) { + raw := []byte(`{ + "hits": { + "total": {"value": 1, "relation": "eq"}, + "hits": [ + { + "_id": "3578", + "_score": 3.21, + "_source": {"id": "3578", "title": "2026-07-28-S1021-Edelweiss-rechnung.pdf", "folder": "634", "ext": "pdf"}, + "highlight": {"content": ["Begleitzettel … S1021 …"]} + } + ] + } + }`) + hits, err := parseESTextResponse(raw) + if err != nil { + t.Fatalf("parseESTextResponse: %v", err) + } + if len(hits) != 1 { + t.Fatalf("hits = %d, want 1", len(hits)) + } + h := hits[0] + if h.ID != "3578" || h.Title != "2026-07-28-S1021-Edelweiss-rechnung.pdf" || h.Kind != File { + t.Errorf("entry = %+v", h.Entry) + } + if h.ParentID != "634" || !reflect.DeepEqual(h.Path, []string{"634"}) { + t.Errorf("path = %v parent = %q", h.Path, h.ParentID) + } + if h.Provider != "es-text" { + t.Errorf("provider = %q", h.Provider) + } + if h.Highlight != "Begleitzettel … S1021 …" { + t.Errorf("highlight = %q, want tags stripped", h.Highlight) + } +} + +func TestNewESTextIndexDefaults(t *testing.T) { + if _, err := NewESTextIndex(ESTextConfig{}); err == nil { + t.Error("empty URL: want error") + } + x, err := NewESTextIndex(ESTextConfig{URL: "http://es:9200/"}) + if err != nil { + t.Fatalf("NewESTextIndex: %v", err) + } + if x.Index() != defaultESTextIndex { + t.Errorf("index = %q, want %q", x.Index(), defaultESTextIndex) + } + if x.cfg.URL != "http://es:9200" { + t.Errorf("url = %q, want trimmed", x.cfg.URL) + } + if x.Name() != "es-text" { + t.Errorf("Name() = %q", x.Name()) + } +} + +func TestESTextConfigFromEnvIndexDefault(t *testing.T) { + t.Setenv("ONLYOFFICE_ES_URL", "http://es:9200/") + t.Setenv("ONLYOFFICE_ES_TEXT_INDEX", "") + cfg := ESTextConfigFromEnv() + if cfg.Index != defaultESTextIndex { + t.Errorf("index = %q, want %q", cfg.Index, defaultESTextIndex) + } +} + +func TestTextIndexerIndexEntries(t *testing.T) { + store := &textFakeStore{ + files: map[string][]byte{"1": []byte("PDFBYTES")}, + } + idx := &textFakeIndex{} + ix := NewTextIndexer(store, idx) + ix.Extractor = textFakeExtractor{prefix: "TEXT "} + + res, err := ix.IndexEntries(context.Background(), []Entry{ + {ID: "1", Title: "Rechnung.PDF", ParentID: "649", Kind: File}, + {ID: "2", Title: "Tabelle.xlsx", ParentID: "649", Kind: File}, + {ID: "3", Title: "Unterordner", Kind: Folder}, + }, IndexOptions{}) + if err != nil { + t.Fatalf("IndexEntries: %v", err) + } + if res.Scanned != 3 || res.Indexed != 1 || res.Skipped != 2 || res.Failed != 0 { + t.Errorf("result = %+v, want scanned=3 indexed=1 skipped=2 failed=0", res) + } + if len(idx.docs) != 1 { + t.Fatalf("indexed docs = %d, want 1", len(idx.docs)) + } + got := idx.docs[0] + want := TextDoc{ID: "1", Title: "Rechnung.PDF", FolderID: "649", Ext: "pdf", Content: "TEXT PDFBYTES"} + if !reflect.DeepEqual(got, want) { + t.Errorf("doc = %+v, want %+v", got, want) + } +} + +func TestTextIndexerRecordsExtractionFailure(t *testing.T) { + store := &textFakeStore{files: map[string][]byte{"1": []byte("x")}} + idx := &textFakeIndex{} + ix := NewTextIndexer(store, idx) + ix.Extractor = textFailingExtractor{} + + res, err := ix.IndexEntries(context.Background(), []Entry{{ID: "1", Title: "a.pdf", Kind: File}}, IndexOptions{}) + if err != nil { + t.Fatalf("IndexEntries: %v", err) + } + if res.Indexed != 0 || res.Failed != 1 || len(res.Errors) != 1 { + t.Errorf("result = %+v, want one failure recorded", res) + } +} + +func TestTextIndexerPlanFolder(t *testing.T) { + store := &textFakeStore{dirs: map[string][]Entry{ + "root": { + {ID: "10", Title: "a.pdf", Kind: File}, + {ID: "11", Title: "sub", Kind: Folder}, + }, + "11": { + {ID: "12", Title: "b.PDF", Kind: File}, + {ID: "13", Title: "c.xlsx", Kind: File}, + }, + }} + ix := NewTextIndexer(store, &textFakeIndex{}) + + flat, err := ix.PlanFolder(context.Background(), "root", IndexOptions{}) + if err != nil { + t.Fatalf("PlanFolder: %v", err) + } + if len(flat) != 1 || flat[0].ID != "10" { + t.Errorf("flat plan = %+v, want only a.pdf", flat) + } + deep, err := ix.PlanFolder(context.Background(), "root", IndexOptions{Recursive: true, Limit: 10}) + if err != nil { + t.Fatalf("PlanFolder recursive: %v", err) + } + if len(deep) != 2 { + t.Errorf("recursive plan = %d entries, want 2", len(deep)) + } +} + +// --- fakes ----------------------------------------------------------------- + +type textFakeStore struct { + dirs map[string][]Entry + files map[string][]byte + stat map[string]Entry +} + +func (f *textFakeStore) Name() string { return "fake" } + +func (f *textFakeStore) List(_ context.Context, parentID string) ([]Entry, error) { + return f.dirs[parentID], nil +} + +func (f *textFakeStore) Stat(_ context.Context, id string) (Entry, error) { + if e, ok := f.stat[id]; ok { + return e, nil + } + return Entry{}, fmt.Errorf("not found: %s", id) +} + +func (f *textFakeStore) Download(_ context.Context, id string, w io.Writer) (int64, error) { + b, ok := f.files[id] + if !ok { + return 0, fmt.Errorf("no bytes for %s", id) + } + n, err := w.Write(b) + return int64(n), err +} + +func (f *textFakeStore) CreateFolder(context.Context, string, string) (Entry, error) { + return Entry{}, nil +} +func (f *textFakeStore) Upload(context.Context, string, string, io.Reader) (Entry, error) { + return Entry{}, nil +} +func (f *textFakeStore) Move(context.Context, []string, string) error { return nil } +func (f *textFakeStore) Copy(context.Context, []string, string) error { return nil } +func (f *textFakeStore) Rename(context.Context, string, string) error { return nil } +func (f *textFakeStore) Delete(context.Context, []string) error { return nil } + +type textFakeIndex struct{ docs []TextDoc } + +func (f *textFakeIndex) Put(_ context.Context, docs []TextDoc) error { + f.docs = append(f.docs, docs...) + return nil +} +func (f *textFakeIndex) Delete(context.Context, []string) error { return nil } +func (f *textFakeIndex) Search(context.Context, SearchQuery) ([]SearchHit, error) { return nil, nil } +func (f *textFakeIndex) Name() string { return "fake" } + +type textFakeExtractor struct{ prefix string } + +func (f textFakeExtractor) Extract(path, _, _ string, _ int) (string, error) { + b, err := os.ReadFile(path) + if err != nil { + return "", err + } + return f.prefix + string(b), nil +} + +type textFailingExtractor struct{} + +func (textFailingExtractor) Extract(string, string, string, int) (string, error) { + return "", fmt.Errorf("boom") +} diff --git a/file_facade.go b/file_facade.go index bbc8533..b1b2db2 100644 --- a/file_facade.go +++ b/file_facade.go @@ -13,12 +13,10 @@ import ( "strings" ) -// Provider names for the composed backends. ProviderPG is reserved for the -// read-only PostgreSQL store (F2 #36); it is not registered until it exists. -const ( - ProviderPG = "postgres" - ProviderES = "elasticsearch" -) +// ProviderES is the composed Elasticsearch searcher. The SQL store owns +// ProviderPG/ProviderMySQL (file_pg.go); the facade references ProviderPG in +// readOrder. +const ProviderES = "elasticsearch" var ( errNoReadBackend = errors.New("onlyoffice: no file backend registered for reads") diff --git a/file_text_index.go b/file_text_index.go new file mode 100644 index 0000000..92b4818 --- /dev/null +++ b/file_text_index.go @@ -0,0 +1,337 @@ +package onlyoffice + +// Text extraction pipeline for the own full-text index (epic #34, F6 #42). +// +// TextIndexer downloads stored documents, extracts text through docpipe +// (pdftotext; OCR for scans) and writes the result to a TextIndex. For PDFs it +// also indexes the text of embedded attachments (pdfdetach), so a scan filed +// as an attachment is searchable too. It is the write side of ESTextIndex and +// never touches the OnlyOffice server's own ES index. + +import ( + "context" + "fmt" + "os" + "path/filepath" + "strings" + "sync" + + "github.com/eslider/go-onlyoffice/internal/docpipe" +) + +// defaultTextIndexExts are the formats extracted by default. The OnlyOffice +// index already covers docx/xlsx/pptx; F6 adds PDF. +var defaultTextIndexExts = []string{"pdf"} + +const ( + defaultTextIndexWorkers = 3 + defaultTextIndexLang = "deu+eng" + maxTextIndexErrors = 20 +) + +// TextExtractor turns a local file into indexable plain text. The default uses +// docpipe (pdftotext + OCR); tests inject a fake to stay offline. +type TextExtractor interface { + Extract(path, workDir, lang string, minChars int) (string, error) +} + +// docpipeExtractor is the production TextExtractor. +type docpipeExtractor struct{ tools docpipe.Tools } + +// Extract renders the file as Markdown, OCRing PDFs/images with a weak text +// layer first and appending the text of embedded PDF attachments +// (docpipe.ToMarkdownWithAttachments). +func (d docpipeExtractor) Extract(path, workDir, lang string, minChars int) (string, error) { + text, err := d.tools.ToMarkdownWithAttachments(path, workDir, lang, minChars) + if err != nil { + return "", err + } + return strings.TrimSpace(text), nil +} + +// IndexOptions controls a TextIndexer run. +type IndexOptions struct { + Recursive bool // IndexFolder: descend into subfolders + Extensions []string // empty = defaultTextIndexExts (pdf) + Limit int // max files to index, 0 = all + Lang string // OCR language(s), default deu+eng + MinChars int // OCR threshold, default docpipe.DefaultMinTextChars + Workers int // parallel downloads/extractions, default 3 +} + +// IndexResult summarises a run. +type IndexResult struct { + Scanned int + Indexed int + Skipped int + Failed int + Errors []string +} + +// TextIndexer wires a FileStore, a TextIndex and an extractor together. +type TextIndexer struct { + Store FileStore + Index TextIndex + Extractor TextExtractor // nil = local docpipe tools + WorkDir string // temp dir for downloads/extraction +} + +// NewTextIndexer returns a TextIndexer over the given store and index. +func NewTextIndexer(store FileStore, index TextIndex) *TextIndexer { + return &TextIndexer{Store: store, Index: index} +} + +// textIndexEnsurer is implemented by indexes that can be created up front. +type textIndexEnsurer interface { + Ensure(ctx context.Context) error +} + +// Ensure creates the backing index when the TextIndex supports it. +func (ix *TextIndexer) Ensure(ctx context.Context) error { + if e, ok := ix.Index.(textIndexEnsurer); ok { + return e.Ensure(ctx) + } + return nil +} + +// IndexFiles stats the given file ids and indexes them. +func (ix *TextIndexer) IndexFiles(ctx context.Context, ids []string, opts IndexOptions) (IndexResult, error) { + entries := make([]Entry, 0, len(ids)) + for _, id := range ids { + e, err := ix.Store.Stat(ctx, id) + if err != nil { + return IndexResult{}, fmt.Errorf("onlyoffice: stat %s: %w", id, err) + } + entries = append(entries, e) + } + return ix.IndexEntries(ctx, entries, opts) +} + +// IndexFolder lists a folder and indexes every matching file. +func (ix *TextIndexer) IndexFolder(ctx context.Context, folderID string, opts IndexOptions) (IndexResult, error) { + entries, err := ix.collect(ctx, folderID, opts.Recursive) + if err != nil { + return IndexResult{}, err + } + return ix.IndexEntries(ctx, entries, opts) +} + +// PlanFolder lists the files IndexFolder would process, without downloading or +// extracting anything. +func (ix *TextIndexer) PlanFolder(ctx context.Context, folderID string, opts IndexOptions) ([]Entry, error) { + entries, err := ix.collect(ctx, folderID, opts.Recursive) + if err != nil { + return nil, err + } + return selectEntries(entries, opts), nil +} + +// PlanFiles stats the ids and returns those that would be indexed. +func (ix *TextIndexer) PlanFiles(ctx context.Context, ids []string, opts IndexOptions) ([]Entry, error) { + entries := make([]Entry, 0, len(ids)) + for _, id := range ids { + e, err := ix.Store.Stat(ctx, id) + if err != nil { + return nil, fmt.Errorf("onlyoffice: stat %s: %w", id, err) + } + entries = append(entries, e) + } + return selectEntries(entries, opts), nil +} + +// IndexEntries extracts and indexes the given files (folders are ignored). +func (ix *TextIndexer) IndexEntries(ctx context.Context, entries []Entry, opts IndexOptions) (IndexResult, error) { + opts = opts.withDefaults() + var res IndexResult + work := selectEntries(entries, opts) + res.Scanned = len(entries) + res.Skipped = len(entries) - len(work) + if len(work) == 0 { + return res, nil + } + + workers := opts.Workers + if workers > len(work) { + workers = len(work) + } + if workers < 1 { + workers = 1 + } + + type outcome struct { + doc TextDoc + err error + } + jobs := make(chan Entry) + results := make(chan outcome, workers) + var wg sync.WaitGroup + for i := 0; i < workers; i++ { + wg.Add(1) + go func() { + defer wg.Done() + for e := range jobs { + if err := ctx.Err(); err != nil { + results <- outcome{err: err} + continue + } + doc, err := ix.indexOne(ctx, e, opts) + results <- outcome{doc: doc, err: err} + } + }() + } + go func() { + defer close(jobs) + for _, e := range work { + select { + case jobs <- e: + case <-ctx.Done(): + return + } + } + }() + go func() { + wg.Wait() + close(results) + }() + + var docs []TextDoc + for r := range results { + if r.err != nil { + res.Failed++ + if len(res.Errors) < maxTextIndexErrors { + res.Errors = append(res.Errors, r.err.Error()) + } + continue + } + docs = append(docs, r.doc) + } + if err := ctx.Err(); err != nil { + return res, err + } + if len(docs) > 0 { + if err := ix.Index.Put(ctx, docs); err != nil { + return res, fmt.Errorf("onlyoffice: index %d docs: %w", len(docs), err) + } + res.Indexed = len(docs) + } + return res, nil +} + +// indexOne downloads and extracts a single file. +func (ix *TextIndexer) indexOne(ctx context.Context, e Entry, opts IndexOptions) (TextDoc, error) { + ext := fileExt(e.Title) + dir := ix.WorkDir + if dir == "" { + dir = os.TempDir() + } + if err := os.MkdirAll(dir, 0o755); err != nil { + return TextDoc{}, err + } + tmp, err := os.CreateTemp(dir, "ooidx-*."+ext) + if err != nil { + return TextDoc{}, err + } + tmpPath := tmp.Name() + defer os.Remove(tmpPath) + + if _, err := ix.Store.Download(ctx, e.ID, tmp); err != nil { + tmp.Close() + return TextDoc{}, fmt.Errorf("download %s (%s): %w", e.ID, e.Title, err) + } + if err := tmp.Close(); err != nil { + return TextDoc{}, err + } + text, err := ix.extractor().Extract(tmpPath, dir, opts.Lang, opts.MinChars) + if err != nil { + return TextDoc{}, fmt.Errorf("extract %s: %w", e.Title, err) + } + return TextDoc{ID: e.ID, Title: e.Title, FolderID: e.ParentID, Ext: ext, Content: text}, nil +} + +// collect lists files under folderID, breadth-first when recursive. +func (ix *TextIndexer) collect(ctx context.Context, folderID string, recursive bool) ([]Entry, error) { + var files []Entry + queue := []string{folderID} + for len(queue) > 0 { + if err := ctx.Err(); err != nil { + return nil, err + } + id := queue[0] + queue = queue[1:] + entries, err := ix.Store.List(ctx, id) + if err != nil { + return nil, fmt.Errorf("onlyoffice: list folder %s: %w", id, err) + } + for _, e := range entries { + if e.Kind == Folder { + if recursive { + queue = append(queue, e.ID) + } + continue + } + if e.ParentID == "" { + e.ParentID = id + } + files = append(files, e) + } + } + return files, nil +} + +// selectEntries filters files by extension and applies the limit. +func selectEntries(entries []Entry, opts IndexOptions) []Entry { + allowed := extensionSet(opts.Extensions) + work := make([]Entry, 0, len(entries)) + for _, e := range entries { + if e.Kind != File { + continue + } + if opts.Limit > 0 && len(work) >= opts.Limit { + break + } + if !allowed[fileExt(e.Title)] { + continue + } + work = append(work, e) + } + return work +} + +// extensionSet normalises the extension allow-list (default: pdf). +func extensionSet(exts []string) map[string]bool { + if len(exts) == 0 { + exts = defaultTextIndexExts + } + set := make(map[string]bool, len(exts)) + for _, e := range normalizeExtensions(exts) { + set[e] = true + } + return set +} + +// fileExt returns the lower-case extension without the dot. +func fileExt(title string) string { + return strings.ToLower(strings.TrimPrefix(filepath.Ext(strings.TrimSpace(title)), ".")) +} + +// withDefaults fills zero-valued options. +func (o IndexOptions) withDefaults() IndexOptions { + if o.Workers <= 0 { + o.Workers = defaultTextIndexWorkers + } + if o.MinChars <= 0 { + o.MinChars = docpipe.DefaultMinTextChars + } + if strings.TrimSpace(o.Lang) == "" { + o.Lang = defaultTextIndexLang + } + return o +} + +// extractor returns the configured extractor or the local docpipe default. +func (ix *TextIndexer) extractor() TextExtractor { + if ix.Extractor != nil { + return ix.Extractor + } + return docpipeExtractor{tools: docpipe.LookPath()} +} diff --git a/internal/docpipe/docpipe.go b/internal/docpipe/docpipe.go index c5038d6..ce2fafc 100644 --- a/internal/docpipe/docpipe.go +++ b/internal/docpipe/docpipe.go @@ -5,6 +5,7 @@ // - pandoc — md↔docx // - ocrmypdf — OCR into a searchable PDF // - pdftotext — extract text layer +// - pdfdetach — list/save embedded PDF attachments // - tesseract — OCR single images when ocrmypdf is unsuitable // - ghostscript (gs) — PDF rewrite/optimize via PostScript (pdfwrite) package docpipe @@ -26,6 +27,7 @@ type Tools struct { Pandoc string OCRMyPDF string PDFToText string + PDFDetach string Tesseract string Ghostscript string } @@ -44,6 +46,7 @@ func LookPath() Tools { Pandoc: find("pandoc"), OCRMyPDF: find("ocrmypdf"), PDFToText: find("pdftotext"), + PDFDetach: find("pdfdetach"), Tesseract: find("tesseract"), Ghostscript: find("gs", "ghostscript"), } diff --git a/internal/docpipe/pdfattach.go b/internal/docpipe/pdfattach.go new file mode 100644 index 0000000..11fd414 --- /dev/null +++ b/internal/docpipe/pdfattach.go @@ -0,0 +1,231 @@ +package docpipe + +// Embedded PDF attachments (F6 #42). Digitised invoices often carry the +// original scan as a PDF attachment; the searchable body may hold only a +// summary. pdfdetach (poppler) lists/saves them; each saved attachment is run +// through the normal docpipe extraction (pdftotext/OCR). + +import ( + "bytes" + "encoding/xml" + "fmt" + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" +) + +// PDFAttachment is one embedded file in a PDF. +type PDFAttachment struct { + Index int // 1-based number as `pdfdetach -list` reports it + Name string // embedded file name +} + +// AttachmentText is the extracted text of one embedded attachment. +type AttachmentText struct { + Name string + Text string +} + +// parseAttachmentList parses `pdfdetach -list` output. The first line is a +// count ("N embedded files"); every following line is ": ". +// Pure, so it is unit-tested. +func parseAttachmentList(out string) []PDFAttachment { + var atts []PDFAttachment + for _, line := range strings.Split(out, "\n") { + line = strings.TrimSpace(line) + if line == "" { + continue + } + colon := strings.Index(line, ":") + if colon <= 0 { + continue + } + n, err := strconv.Atoi(strings.TrimSpace(line[:colon])) + if err != nil { + continue + } + name := strings.TrimSpace(line[colon+1:]) + if name == "" { + continue + } + atts = append(atts, PDFAttachment{Index: n, Name: name}) + } + return atts +} + +// safeAttachmentName strips directories and leading dots so a hostile +// attachment name cannot escape the extraction directory. +func safeAttachmentName(name string) string { + name = strings.ReplaceAll(strings.TrimSpace(name), "\\", "/") + name = filepath.Base(name) + name = strings.TrimLeft(name, ".") + if name == "" || name == "." || name == "/" { + return "" + } + return name +} + +// JoinWithAttachments appends attachment text to the document body, each +// section preceded by an "[attachment: ]" marker so a search hit shows +// its source. Empty attachments are skipped. Pure, so it is unit-tested. +func JoinWithAttachments(body string, atts []AttachmentText) string { + var b strings.Builder + b.WriteString(strings.TrimRight(body, "\n")) + for _, a := range atts { + text := strings.TrimSpace(a.Text) + if text == "" { + continue + } + b.WriteString("\n\n[attachment: ") + b.WriteString(a.Name) + b.WriteString("]\n\n") + b.WriteString(text) + } + return b.String() +} + +// ListAttachments returns the embedded files of a PDF. A PDF without +// attachments yields an empty slice and no error. +func (t Tools) ListAttachments(pdfPath string) ([]PDFAttachment, error) { + if t.PDFDetach == "" { + return nil, fmt.Errorf("pdfdetach not found on PATH") + } + cmd := exec.Command(t.PDFDetach, "-list", pdfPath) + var stderr bytes.Buffer + cmd.Stderr = &stderr + out, err := cmd.Output() + if err != nil { + return nil, fmt.Errorf("pdfdetach -list %s: %w (%s)", filepath.Base(pdfPath), err, strings.TrimSpace(stderr.String())) + } + return parseAttachmentList(string(out)), nil +} + +// SaveAttachment writes the n-th embedded file (1-based) to outPath. +func (t Tools) SaveAttachment(pdfPath string, index int, outPath string) error { + if t.PDFDetach == "" { + return fmt.Errorf("pdfdetach not found on PATH") + } + if strings.TrimSpace(outPath) == "" { + return fmt.Errorf("output path required") + } + if err := EnsureDir(outPath); err != nil { + return err + } + cmd := exec.Command(t.PDFDetach, "-save", strconv.Itoa(index), "-o", outPath, pdfPath) + var stderr bytes.Buffer + cmd.Stderr = &stderr + if err := cmd.Run(); err != nil { + return fmt.Errorf("pdfdetach -save %d: %w (%s)", index, err, strings.TrimSpace(stderr.String())) + } + return nil +} + +// ToMarkdownWithAttachments extracts the file as ToMarkdown does, then — for +// PDFs — appends the text of every embedded attachment under an +// "[attachment: ]" marker. Attachment failures are non-fatal: the body +// is returned unchanged. +func (t Tools) ToMarkdownWithAttachments(path, workDir, lang string, minChars int) (string, error) { + res, err := t.ToMarkdown(path, workDir, lang, minChars) + if err != nil { + return "", err + } + if Ext(path) != ".pdf" { + return res.Markdown, nil + } + atts, err := t.attachmentTexts(path, workDir, lang, minChars) + if err != nil { + return res.Markdown, nil + } + return JoinWithAttachments(res.Markdown, atts), nil +} + +// attachmentTexts saves and extracts every embedded attachment, skipping the +// ones that cannot be read. It returns an error only when the attachment list +// itself cannot be obtained. +func (t Tools) attachmentTexts(pdfPath, workDir, lang string, minChars int) ([]AttachmentText, error) { + list, err := t.ListAttachments(pdfPath) + if err != nil || len(list) == 0 { + return nil, err + } + if workDir == "" { + workDir = os.TempDir() + } + dir := filepath.Join(workDir, "att-"+trimExt(filepath.Base(pdfPath))) + if err := os.MkdirAll(dir, 0o755); err != nil { + return nil, err + } + defer os.RemoveAll(dir) + + out := make([]AttachmentText, 0, len(list)) + for _, a := range list { + name := safeAttachmentName(a.Name) + if name == "" { + continue + } + saved := filepath.Join(dir, fmt.Sprintf("%d-%s", a.Index, name)) + if err := t.SaveAttachment(pdfPath, a.Index, saved); err != nil { + continue + } + text, err := t.attachmentMarkdown(saved, dir, lang, minChars) + if err != nil { + continue + } + out = append(out, AttachmentText{Name: a.Name, Text: text}) + } + return out, nil +} + +// attachmentMarkdown extracts a saved attachment with the regular pipeline. +// Structured attachments that docpipe does not convert (e-invoice XML, +// CuraSoft JSON, CSV/HTML) fall back to their text content, so the embedded +// original is still searchable. Other unreadable formats return an error and +// the caller skips them. +func (t Tools) attachmentMarkdown(path, workDir, lang string, minChars int) (string, error) { + if res, err := t.ToMarkdown(path, workDir, lang, minChars); err == nil { + return res.Markdown, nil + } + switch Ext(path) { + case ".xml", ".html", ".htm": + raw, err := os.ReadFile(path) + if err != nil { + return "", err + } + return xmlToText(raw), nil + case ".json", ".csv": + raw, err := os.ReadFile(path) + if err != nil { + return "", err + } + return string(raw), nil + default: + return "", fmt.Errorf("unsupported attachment type %q", Ext(path)) + } +} + +// xmlToText returns the character data of an XML/HTML document: element text +// values with decoded entities, one per line. Used for invoice XML (EN 16931 +// CII / ZUGFeRD) and HTML attachments. Pure, so it is unit-tested. +func xmlToText(raw []byte) string { + dec := xml.NewDecoder(bytes.NewReader(raw)) + dec.Strict = false + var b strings.Builder + for { + tok, err := dec.Token() + if err != nil { + break + } + cd, ok := tok.(xml.CharData) + if !ok { + continue + } + s := strings.TrimSpace(string(cd)) + if s == "" { + continue + } + b.WriteString(s) + b.WriteByte('\n') + } + return b.String() +} diff --git a/internal/docpipe/pdfattach_test.go b/internal/docpipe/pdfattach_test.go new file mode 100644 index 0000000..ad8ab6e --- /dev/null +++ b/internal/docpipe/pdfattach_test.go @@ -0,0 +1,163 @@ +package docpipe + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +func TestParseAttachmentList(t *testing.T) { + out := "2 embedded files\n1: original.pdf\n2: scan_001.png\n" + got := parseAttachmentList(out) + want := []PDFAttachment{{Index: 1, Name: "original.pdf"}, {Index: 2, Name: "scan_001.png"}} + if len(got) != len(want) { + t.Fatalf("got %+v, want %+v", got, want) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("att[%d] = %+v, want %+v", i, got[i], want[i]) + } + } +} + +func TestParseAttachmentListEmptyAndMalformed(t *testing.T) { + for _, in := range []string{"", "0 embedded files\n", "garbage\n\n \n"} { + if got := parseAttachmentList(in); len(got) != 0 { + t.Errorf("parseAttachmentList(%q) = %+v, want empty", in, got) + } + } +} + +func TestSafeAttachmentName(t *testing.T) { + cases := map[string]string{ + "note.txt": "note.txt", + "../../evil.pdf": "evil.pdf", + `..\..\evil.pdf`: "evil.pdf", + "/abs/scan_001.pdf": "scan_001.pdf", + ".hidden": "hidden", + " spaced name.txt ": "spaced name.txt", + "..": "", + "": "", + } + for in, want := range cases { + if got := safeAttachmentName(in); got != want { + t.Errorf("safeAttachmentName(%q) = %q, want %q", in, got, want) + } + } +} + +func TestJoinWithAttachments(t *testing.T) { + body := "# scan.pdf\n\nbody token\n" + atts := []AttachmentText{ + {Name: "original.pdf", Text: " original token "}, + {Name: "empty.txt", Text: " "}, + } + got := JoinWithAttachments(body, atts) + if !strings.Contains(got, "body token") { + t.Errorf("body text lost: %q", got) + } + if !strings.Contains(got, "[attachment: original.pdf]") { + t.Errorf("marker missing: %q", got) + } + if !strings.Contains(got, "original token") { + t.Errorf("attachment text missing: %q", got) + } + if strings.Contains(got, "empty.txt") { + t.Errorf("empty attachment must be skipped: %q", got) + } +} + +func TestJoinWithAttachmentsNoAttachments(t *testing.T) { + got := JoinWithAttachments("# a.pdf\n\ntext\n\n", nil) + if got != "# a.pdf\n\ntext" { + t.Errorf("got %q, want trimmed body only", got) + } +} + +func TestListAttachmentsWithoutTool(t *testing.T) { + if _, err := (Tools{}).ListAttachments("x.pdf"); err == nil || !strings.Contains(err.Error(), "pdfdetach") { + t.Fatalf("want pdfdetach error, got %v", err) + } +} + +// TestToMarkdownWithAttachmentsFixture exercises the real pdfdetach + pdftotext +// pipeline on testdata/pdf-with-attachment.pdf (body token + embedded +// goo-note.txt). Skips when poppler is not installed. +func TestToMarkdownWithAttachmentsFixture(t *testing.T) { + tools := LookPath() + if tools.PDFDetach == "" || tools.PDFToText == "" { + t.Skip("pdfdetach/pdftotext not on PATH — skipping attachment extraction test") + } + fixture := filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf") + got, err := tools.ToMarkdownWithAttachments(fixture, t.TempDir(), "eng", 1) + if err != nil { + t.Fatalf("ToMarkdownWithAttachments: %v", err) + } + for _, want := range []string{"goobodytoken", "[attachment: goo-note.txt]", "gooattachmenttoken"} { + if !strings.Contains(got, want) { + t.Errorf("result missing %q:\n%s", want, got) + } + } +} + +func TestXMLToText(t *testing.T) { + raw := []byte(` + +S1063 +Edelweiss & Co42.00 +`) + got := xmlToText(raw) + for _, want := range []string{"S1063", "Edelweiss & Co", "42.00"} { + if !strings.Contains(got, want) { + t.Errorf("xmlToText missing %q:\n%s", want, got) + } + } + if strings.ContainsAny(got, "<>") { + t.Errorf("xmlToText left markup: %q", got) + } +} + +// TestAttachmentMarkdownFallback verifies structured attachments that docpipe +// cannot convert are still reduced to searchable text, and unknown binary +// formats error (so the caller skips them). +func TestAttachmentMarkdownFallback(t *testing.T) { + dir := t.TempDir() + xmlPath := filepath.Join(dir, "factur-x.xml") + if err := os.WriteFile(xmlPath, []byte(`S1063`), 0o644); err != nil { + t.Fatal(err) + } + got, err := (Tools{}).attachmentMarkdown(xmlPath, dir, "", 0) + if err != nil { + t.Fatalf("attachmentMarkdown(xml): %v", err) + } + if !strings.Contains(got, "S1063") { + t.Errorf("xml attachment text = %q, want S1063", got) + } + + binPath := filepath.Join(dir, "data.bin") + if err := os.WriteFile(binPath, []byte{0, 1, 2, 3}, 0o644); err != nil { + t.Fatal(err) + } + if _, err := (Tools{}).attachmentMarkdown(binPath, dir, "", 0); err == nil { + t.Error("unsupported attachment: want error, got nil") + } +} + +// TestToMarkdownWithAttachmentsPlainPDF ensures a PDF without attachments +// returns just the body (pdfdetach prints "0 embedded files"). +func TestToMarkdownWithAttachmentsPlainPDF(t *testing.T) { + tools := LookPath() + if tools.PDFDetach == "" || tools.PDFToText == "" { + t.Skip("pdfdetach/pdftotext not on PATH") + } + // The fixture itself is a PDF with one attachment; strip it by extracting + // the body only through ToMarkdown and compare JoinWithAttachments(nil). + res, err := tools.ToMarkdown(filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf"), t.TempDir(), "eng", 1) + if err != nil { + t.Fatalf("ToMarkdown: %v", err) + } + if strings.Contains(res.Markdown, "gooattachmenttoken") { + t.Fatalf("body must not contain attachment text: %q", res.Markdown) + } +} diff --git a/testdata/pdf-with-attachment.pdf b/testdata/pdf-with-attachment.pdf new file mode 100644 index 0000000..34dac60 --- /dev/null +++ b/testdata/pdf-with-attachment.pdf @@ -0,0 +1,114 @@ +%PDF-1.7 +%Ç쏢 +%%Invocation: gs -q -dNOPAUSE -dBATCH -sDEVICE=pdfwrite ? ? ? +5 0 obj +<> +stream +xœ-ŠA +ƒ0÷ÿoÙnìO7q]èZ*ÿ�Iˆm J½½M ³f7 +\¨6‘žfRÿŠ*ñºõª…8:g}‡f†Dºø”ÞiØsúØ ÝôÝ;炱(Ùn.ly]ìUFz +½~Aë øendstream +endobj +6 0 obj +110 +endobj +4 0 obj +<> +/Contents 5 0 R +>> +endobj +3 0 obj +<< /Type /Pages /Kids [ +4 0 R +] /Count 1 +>> +endobj +1 0 obj +<> +endobj +8 0 obj +<> +endobj +7 0 obj +<> +endobj +9 0 obj +<>stream + + + + + +2026-09-16T17:29:12Z +2026-09-16T17:29:12Z +UnknownApplication + +Untitled + + + + + +endstream +endobj +2 0 obj +<>endobj +xref +0 10 +0000000000 65535 f +0000000476 00000 n +0000001882 00000 n +0000000417 00000 n +0000000276 00000 n +0000000077 00000 n +0000000257 00000 n +0000000569 00000 n +0000000540 00000 n +0000000633 00000 n +trailer +<< /Size 10 /Root 1 0 R /Info 2 0 R +/ID [<5A562476FBF11852133AA873452909F9><5A562476FBF11852133AA873452909F9>] +>> +startxref +2008 +%%EOF +1 0 obj +<> >> +endobj +10 0 obj +<> >> stream +gooattachmenttoken embedded attachment text + +endstream + +endobj +11 0 obj +< /EF <> >> +endobj +12 0 obj +< 11 0 R ] >> +endobj +xref +0 2 +0000000002 65535 f +0000002361 00000 n +10 3 +0000002463 00000 n +0000002586 00000 n +0000002705 00000 n +trailer +<> +startxref +2802 +%%EOF