diff --git a/.env.example b/.env.example
index d5f5b5e..e11e86d 100644
--- a/.env.example
+++ b/.env.example
@@ -41,3 +41,8 @@ ONLYOFFICE_PROJECT_ID=33
# ONLYOFFICE_ES_URL=http://127.0.0.1:9200
# ONLYOFFICE_ES_INDEX=files_file
# ONLYOFFICE_TENANT=
+#
+# oo index / oo search --backend own — own full-text index for PDF/scans,
+# filled by `oo index` from internal/docpipe (pdftotext + OCR). Defaults to
+# oo_docs_text. Uses the same ONLYOFFICE_ES_URL.
+# ONLYOFFICE_ES_TEXT_INDEX=oo_docs_text
diff --git a/README.md b/README.md
index 1458a5f..6c12250 100644
--- a/README.md
+++ b/README.md
@@ -697,6 +697,25 @@ oo search "Rechnung" --json # shorthand for -o json
Requires `ONLYOFFICE_ES_URL` (plus optional `ONLYOFFICE_ES_INDEX`,
`ONLYOFFICE_TENANT`).
+#### PDF/scans: own index (`oo index` + `--backend own`)
+
+The OnlyOffice index covers Office formats only, so PDFs (`S1019`-style invoice
+numbers) are not searchable by content. `oo index` extracts PDF text with
+`internal/docpipe` (pdftotext, OCR for scans) into a separate index
+(`ONLYOFFICE_ES_TEXT_INDEX`, default `oo_docs_text`); the OnlyOffice server and
+its index are **not** modified. Then search it with `--backend own`.
+
+```bash
+oo index folder 634 --recursive --exts pdf # populate (idempotent upsert)
+oo index files 3576 3578 # specific files
+oo index folder 634 --dry-run # plan only
+oo search "S1021" --content --backend own # finds the PDF
+oo search "Rechnung" --backend own --folder 634 --json
+```
+
+See [`docs/elasticsearch.md`](docs/elasticsearch.md) for the decision and
+trade-offs.
+
### Bulk tools (`cmd/`)
Small single-purpose binaries for bulk Documents work. All of them pace
@@ -733,7 +752,8 @@ kontolink IN.xlsx oo-index.tsv OUT.xlsx [FILE_ID] [AMOUNTS_TSV]
| `docs` | `tools`, `convert`, `optimize`, `ocr`, `hocr`, `as-md`, `put-md`, `put-txt`, `put-xlsx` |
| `catalog` | `match`, `merge`, `apply`, `scan-contacts`, `scan-projects`, `scan-thunderbird` |
| `dav` | `ls`, `move`, `copy`, `mkdir`, `rename-file`, `rename-folder`, `download`, `fileops` |
-| `search` | `QUERY` (`--content`, `--folder ID`, `--limit N`, `--json`) |
+| `search` | `QUERY` (`--content`, `--folder ID`, `--limit N`, `--backend oo\|own`, `--json`) |
+| `index` | `folder FOLDER_ID`, `files FILE_ID...` (`--recursive`, `--exts pdf`, `--limit N`, `--dry-run`) |
The CLI reads only `.env` from the current working directory (godotenv is a
CLI-only concern — the library itself never loads dotfiles).
diff --git a/cmd/oo/index.go b/cmd/oo/index.go
new file mode 100644
index 0000000..82c13d1
--- /dev/null
+++ b/cmd/oo/index.go
@@ -0,0 +1,158 @@
+package main
+
+import (
+ "strings"
+
+ onlyoffice "github.com/eslider/go-onlyoffice"
+ "github.com/spf13/cobra"
+)
+
+func init() {
+ rootCmd.AddCommand(indexCmd())
+}
+
+// indexFlags are shared by the `oo index folder` and `oo index files` verbs.
+type indexFlags struct {
+ recursive bool
+ exts string
+ limit int
+ workers int
+ lang string
+ minChars int
+ workDir string
+ backend string
+ dryRun bool
+ asJSON bool
+}
+
+// indexCmd populates the own full-text index (oo_docs_text) that makes PDF and
+// scanned content searchable. The OnlyOffice index is left untouched.
+func indexCmd() *cobra.Command {
+ f := &indexFlags{}
+ cmd := &cobra.Command{
+ Use: "index",
+ Short: "Populate the own full-text index for PDF/scan content",
+ Long: "Index document text that the OnlyOffice Elasticsearch index does not\n" +
+ "cover (PDFs and scans) into a separate index (ONLYOFFICE_ES_TEXT_INDEX,\n" +
+ "default oo_docs_text). Text is extracted with docpipe (pdftotext, OCR)\n" +
+ "and the OnlyOffice server is never modified.\n\n" +
+ "Requires ONLYOFFICE_URL/USER/PASS (to download files) and ONLYOFFICE_ES_URL\n" +
+ "(to write the index). See docs/elasticsearch.md.",
+ }
+ cmd.PersistentFlags().BoolVar(&f.recursive, "recursive", false, "folder: descend into subfolders")
+ cmd.PersistentFlags().StringVar(&f.exts, "exts", "pdf", "comma-separated extensions to index")
+ cmd.PersistentFlags().IntVar(&f.limit, "limit", 0, "maximum number of files to index (0 = all)")
+ cmd.PersistentFlags().IntVar(&f.workers, "workers", 3, "parallel downloads/extractions")
+ cmd.PersistentFlags().StringVar(&f.lang, "lang", "deu+eng", "OCR language(s)")
+ cmd.PersistentFlags().IntVar(&f.minChars, "min-chars", 0, "text-layer threshold below which OCR runs")
+ cmd.PersistentFlags().StringVar(&f.workDir, "work-dir", "", "temp dir for downloads (default: system temp)")
+ cmd.PersistentFlags().StringVar(&f.backend, "backend", "rest", "file backend: rest|dav")
+ cmd.PersistentFlags().BoolVar(&f.dryRun, "dry-run", false, "list what would be indexed, without changes")
+ cmd.PersistentFlags().BoolVar(&f.asJSON, "json", false, "shorthand for --output json")
+
+ cmd.AddCommand(indexFolderCmd(f), indexFilesCmd(f))
+ return cmd
+}
+
+func indexFolderCmd(f *indexFlags) *cobra.Command {
+ return &cobra.Command{
+ Use: "folder FOLDER_ID",
+ Short: "Index every matching file in a Documents folder",
+ Args: cobra.ExactArgs(1),
+ RunE: func(cmd *cobra.Command, args []string) error {
+ return runIndex(cmd, f, args[0], nil)
+ },
+ }
+}
+
+func indexFilesCmd(f *indexFlags) *cobra.Command {
+ return &cobra.Command{
+ Use: "files FILE_ID...",
+ Short: "Index specific Documents files",
+ Args: cobra.MinimumNArgs(1),
+ RunE: func(cmd *cobra.Command, args []string) error {
+ return runIndex(cmd, f, "", args)
+ },
+ }
+}
+
+func runIndex(cmd *cobra.Command, f *indexFlags, folderID string, ids []string) error {
+ if f.asJSON {
+ outputFormat = "json"
+ }
+ c, err := newOO(cmd)
+ if err != nil {
+ return err
+ }
+ idx, err := onlyoffice.NewESTextIndex(onlyoffice.ESTextConfigFromEnv())
+ if err != nil {
+ return err
+ }
+ ti := onlyoffice.NewTextIndexer(c.FileStore(f.backend), idx)
+ ti.WorkDir = f.workDir
+ opts := onlyoffice.IndexOptions{
+ Recursive: f.recursive,
+ Extensions: splitList(f.exts),
+ Limit: f.limit,
+ Lang: f.lang,
+ MinChars: f.minChars,
+ Workers: f.workers,
+ }
+ ctx := cmd.Context()
+
+ if f.dryRun {
+ var entries []onlyoffice.Entry
+ if folderID != "" {
+ entries, err = ti.PlanFolder(ctx, folderID, opts)
+ } else {
+ entries, err = ti.PlanFiles(ctx, ids, opts)
+ }
+ if err != nil {
+ return err
+ }
+ rows := make([]map[string]any, 0, len(entries))
+ for _, e := range entries {
+ rows = append(rows, map[string]any{
+ "id": e.ID,
+ "title": e.Title,
+ "folder": e.ParentID,
+ })
+ }
+ printTable([]string{"id", "title", "folder"}, rows)
+ return nil
+ }
+
+ if err := ti.Ensure(ctx); err != nil {
+ return err
+ }
+ var res onlyoffice.IndexResult
+ if folderID != "" {
+ res, err = ti.IndexFolder(ctx, folderID, opts)
+ } else {
+ res, err = ti.IndexFiles(ctx, ids, opts)
+ }
+ if err != nil {
+ return err
+ }
+ printObject(map[string]any{
+ "index": idx.Index(),
+ "scanned": res.Scanned,
+ "indexed": res.Indexed,
+ "skipped": res.Skipped,
+ "failed": res.Failed,
+ "errors": res.Errors,
+ })
+ return nil
+}
+
+// splitList parses a comma-separated flag value, dropping blanks.
+func splitList(s string) []string {
+ parts := strings.Split(s, ",")
+ out := make([]string, 0, len(parts))
+ for _, p := range parts {
+ if p = strings.TrimSpace(p); p != "" {
+ out = append(out, p)
+ }
+ }
+ return out
+}
diff --git a/cmd/oo/index_test.go b/cmd/oo/index_test.go
new file mode 100644
index 0000000..b6f2805
--- /dev/null
+++ b/cmd/oo/index_test.go
@@ -0,0 +1,55 @@
+package main
+
+import (
+ "reflect"
+ "strings"
+ "testing"
+)
+
+func TestIndexCommandRegistered(t *testing.T) {
+ cmd, _, err := rootCmd.Find([]string{"index"})
+ if err != nil {
+ t.Fatal(err)
+ }
+ if cmd.Name() != "index" {
+ t.Fatalf("index resolved to %q", cmd.Name())
+ }
+ for _, name := range []string{"exts", "limit", "workers", "lang", "min-chars", "work-dir", "backend", "dry-run", "recursive", "json"} {
+ if cmd.PersistentFlags().Lookup(name) == nil {
+ t.Errorf("index: missing --%s flag", name)
+ }
+ }
+ if cmd.PersistentFlags().Lookup("exts").DefValue != "pdf" {
+ t.Errorf("--exts default = %q, want pdf", cmd.PersistentFlags().Lookup("exts").DefValue)
+ }
+ for _, verb := range []string{"index folder", "index files"} {
+ sub, _, err := rootCmd.Find(strings.Fields(verb))
+ if err != nil {
+ t.Fatalf("%s: %v", verb, err)
+ }
+ if sub.Name() != strings.Fields(verb)[1] {
+ t.Errorf("%s resolved to %q", verb, sub.Name())
+ }
+ }
+}
+
+func TestIndexSearchBackendFlag(t *testing.T) {
+ cmd, _, err := rootCmd.Find([]string{"search"})
+ if err != nil {
+ t.Fatal(err)
+ }
+ if cmd.Flags().Lookup("backend") == nil {
+ t.Fatal("search: missing --backend flag")
+ }
+ if cmd.Flags().Lookup("backend").DefValue != "oo" {
+ t.Errorf("--backend default = %q, want oo", cmd.Flags().Lookup("backend").DefValue)
+ }
+}
+
+func TestSplitList(t *testing.T) {
+ got := splitList(" pdf , .PDF, docx ,, ")
+ want := []string{"pdf", ".PDF", "docx"}
+ if !reflect.DeepEqual(got, want) {
+ t.Errorf("splitList = %v, want %v", got, want)
+ }
+}
diff --git a/cmd/oo/main.go b/cmd/oo/main.go
index 82ab8e8..1933d9d 100644
--- a/cmd/oo/main.go
+++ b/cmd/oo/main.go
@@ -18,7 +18,8 @@
// oo docs tools | convert | optimize | ocr | hocr | as-md | put-md | put-txt | put-xlsx
// oo catalog match | merge | apply | scan-contacts | scan-projects | scan-thunderbird
// oo dav ls | move | copy | mkdir | rename-file | rename-folder | download | fileops
-// oo search QUERY [--content] [--folder ID] [--limit N] [--json]
+// oo search QUERY [--content] [--folder ID] [--limit N] [--backend oo|own] [--json]
+// oo index folder FOLDER_ID | files FILE_ID... [--recursive] [--exts pdf] [--dry-run]
//
// CRM association rules: docs/crm-associations.md
//
diff --git a/cmd/oo/search.go b/cmd/oo/search.go
index 2ceab0e..9a244e4 100644
--- a/cmd/oo/search.go
+++ b/cmd/oo/search.go
@@ -1,6 +1,9 @@
package main
import (
+ "fmt"
+ "strings"
+
onlyoffice "github.com/eslider/go-onlyoffice"
"github.com/spf13/cobra"
)
@@ -17,6 +20,7 @@ func searchCmd() *cobra.Command {
content bool
folder string
limit int
+ backend string
asJSON bool
)
cmd := &cobra.Command{
@@ -26,6 +30,9 @@ func searchCmd() *cobra.Command {
"By default only file names are matched. With --content the query also\n" +
"matches extracted document text (document.attachment.content); this covers\n" +
"Office formats (docx/xlsx/pptx) and is slower.\n\n" +
+ "--backend own queries the separate index populated by `oo index`\n" +
+ "(ONLYOFFICE_ES_TEXT_INDEX, default oo_docs_text) instead, which also holds\n" +
+ "PDFs and scans (see docs/elasticsearch.md).\n\n" +
"Requires ONLYOFFICE_ES_URL (and optionally ONLYOFFICE_ES_INDEX,\n" +
"ONLYOFFICE_TENANT). See docs/elasticsearch.md for the tunnel setup.",
Args: cobra.ExactArgs(1),
@@ -33,11 +40,22 @@ func searchCmd() *cobra.Command {
if asJSON {
outputFormat = "json"
}
- es, err := onlyoffice.NewESSearcher(onlyoffice.ESConfigFromEnv())
+ var (
+ searcher onlyoffice.Searcher
+ err error
+ )
+ switch strings.ToLower(strings.TrimSpace(backend)) {
+ case "", "oo", "elasticsearch":
+ searcher, err = onlyoffice.NewESSearcher(onlyoffice.ESConfigFromEnv())
+ case "own", "es-text":
+ searcher, err = onlyoffice.NewESTextIndex(onlyoffice.ESTextConfigFromEnv())
+ default:
+ return fmt.Errorf("unknown search backend %q (want oo|own)", backend)
+ }
if err != nil {
return err
}
- hits, err := es.Search(cmd.Context(), onlyoffice.SearchQuery{
+ hits, err := searcher.Search(cmd.Context(), onlyoffice.SearchQuery{
Text: args[0],
InContent: content,
FolderID: folder,
@@ -67,6 +85,7 @@ func searchCmd() *cobra.Command {
cmd.Flags().BoolVar(&content, "content", false, "also match extracted document content")
cmd.Flags().StringVar(&folder, "folder", "", "limit to a Documents folder id")
cmd.Flags().IntVar(&limit, "limit", 20, "maximum number of results")
+ cmd.Flags().StringVar(&backend, "backend", "oo", "index to query: oo (OnlyOffice) | own (oo index)")
cmd.Flags().BoolVar(&asJSON, "json", false, "shorthand for --output json")
return cmd
}
diff --git a/docs/elasticsearch.md b/docs/elasticsearch.md
index 05df5d6..35fd1a5 100644
--- a/docs/elasticsearch.md
+++ b/docs/elasticsearch.md
@@ -121,5 +121,111 @@ ONLYOFFICE_ES_URL=http://127.0.0.1:9200 ONLYOFFICE_TENANT=1 \
- `locale`/версия ES: 7.16.3, `_search` совместим с REST 7.x.
- ES без auth и слушает только localhost — туннель обязателен.
- Фильтр `tenantId` сузит выдачу; без него видны документы всех тенантов.
-- Поиск по содержимому PDF, залитых через API, не работает (нет
- `attachment.content`) — только Office-форматы.
+- Поиск по содержимому PDF в индексе OnlyOffice не работает (для PDF нет
+ `attachment.content`) — только Office-форматы. Решение для PDF — свой индекс
+ `oo_docs_text` (F6 #42), см. ниже.
+
+# PDF и сканы — свой индекс (F6 #42)
+
+## Проблема
+
+`oo search --content "S1019"` не находил номер внутри PDF-счёта: в индексе
+OnlyOffice PDF лежит только по имени.
+
+## Почему PDF исключён (исходники CommunityServer)
+
+Разобрано в `ONLYOFFICE/CommunityServer`:
+
+- `web/core/ASC.Web.Core/Files/FileUtility.cs` — `CanIndex(fileName)` читает
+ серверную настройку `files.index.formats` (в `web/studio/ASC.Web.Studio/web.appsettings.config`
+ значение по умолчанию `".pptx|.xlsx|.docx"`).
+- `web/studio/ASC.Web.Studio/Products/Files/Core/Search/FilesWrapper.cs` —
+ `GetDocumentStream*` возвращает `null`, если `!FileUtility.CanIndex(Title)`,
+ файл зашифрован или больше `MaxFileSize`.
+- `module/ASC.ElasticSearch/Core/WrapperWithDoc.cs` + mapping в `Wrapper.cs` —
+ маппинг `document.attachment.content` и ingest-pipeline `attachments`
+ формат-агностичны: они распарсят любой поток.
+
+Вывод: PDF исключён **только настройкой** `files.index.formats`; жёсткого
+ограничения на формат в коде нет.
+
+## Варианты и решение
+
+| # | Вариант | Оценка |
+|---|---------|--------|
+| a | Включить `.pdf` в `files.index.formats` + reindex | Правка сервера OO; настройка может потеряться при обновлении; полный reindex 39k док-в; Tika **не OCR** — сканы без текстового слоя дадут пустой контент. Отклонён без решения PO. |
+| b | Server-side ingest/attachment для PDF | По факту то же, что (a): сервер кормит поток только для `CanIndex`. |
+| c | **Свой индекс** `oo_docs_text`, наполняемый `internal/docpipe` | **Выбран.** Сервер OO не трогаем; детерминированно; работает OCR для сканов; независимо от обновлений OO; любые форматы; фильтры папка/тип. |
+| d | Локальный поиск без индекса | Отклонён как основной: качаем и извлекаем на каждый запрос, нет выдачи/ранжирования/highlight. |
+
+Итог: **вариант c**. Индекс OnlyOffice (`files_file`) не изменяется; наш
+индекс живёт рядом.
+
+## Устройство
+
+- `file_es_text.go` — `ESTextIndex` (`Name() = "es-text"`):
+ `Ensure` (создаёт индекс с явным маппингом), `Put` (bulk, `refresh`),
+ `Delete` (по `id`), `Search` (`multi_match` по `title^2` + `content`,
+ фильтры `folder`/`ext`, highlight).
+- `file_text_index.go` — `TextIndexer`: листает папки (`FileStore.List`),
+ качает файлы (`FileStore.Download`), извлекает текст через
+ `internal/docpipe` (`pdftotext`, для сканов — `ocrmypdf`/`tesseract`),
+ пишет в `TextIndex`. Пул воркеров (по умолчанию 3).
+- CLI: `oo index folder|files` наполняет индекс; `oo search --backend own`
+ ищет по нему.
+
+Поля `oo_docs_text`:
+
+| поле | тип | смысл |
+|------|-----|-------|
+| `id` | keyword | id файла Documents |
+| `title` | text (+`.keyword`) | имя файла |
+| `folder` | keyword | id папки |
+| `ext` | keyword | расширение |
+| `content` | text | извлечённый текст (pdftotext/OCR) |
+
+## CLI
+
+```bash
+set -a; . .env; set +a # ONLYOFFICE_URL/USER/PASS + ONLYOFFICE_ES_URL
+oo index folder 634 # PDF в папке 634
+oo index folder 634 --recursive --exts pdf,png --limit 100
+oo index files 3576 3578 # точечно
+oo index folder 634 --dry-run # показать план, ничего не менять
+
+oo search "S1021" --content --backend own
+oo search "S1021" --backend own --folder 634 --json
+```
+
+`--backend` у `oo search`: `oo` (по умолчанию, индекс OnlyOffice) или `own`
+(наш `ONLYOFFICE_ES_TEXT_INDEX`).
+
+## Переменные (дополнение)
+
+| env | default | смысл |
+|-----|---------|-------|
+| `ONLYOFFICE_ES_TEXT_INDEX` | `oo_docs_text` | индекс своего конвейера |
+
+`ONLYOFFICE_ES_URL` — общий для обоих индексов.
+
+## Тесты
+
+```bash
+go test -run 'ESText|TextIndexer|Index' ./ ./cmd/oo/ # unit, без сети
+ONLYOFFICE_ES_URL=http://127.0.0.1:9200 \
+ go test -tags=integration -run TestIntegrationESTextIndex -v .
+```
+
+Интеграционный тест создаёт временный индекс, наполняет, ищет по контенту,
+проверяет фильтры и удаление, затем удаляет индекс. Unit-тесты используют
+fake-store/fake-extractor и не требуют pdftotext/OCR.
+
+## Грабли
+
+- Наполнение — ручное (`oo index`); после изменения/добавления PDF повтори.
+ Повтор идемпотентен (upsert по id файла).
+- В индексе ищется только то, что проиндексировано; `oo index` качает каждый
+ файл и (для сканов) гоняет OCR — это медленно, отсюда `--limit`/`--exts`.
+- `folder` фильтруется как id папки, а не как путь.
+- Дубликаты (напр. `S1055.pdf` и `2026-08-20-S1055-…`) дадут несколько строк —
+ это ожидаемо, дедуп — на стороне потребителя.
diff --git a/file_es.go b/file_es.go
index 9c2db76..5c286c3 100644
--- a/file_es.go
+++ b/file_es.go
@@ -188,6 +188,7 @@ type esBool struct {
type esClause struct {
MultiMatch *esMultiMatch `json:"multi_match,omitempty"`
Term map[string]any `json:"term,omitempty"`
+ Terms map[string]any `json:"terms,omitempty"`
Wildcard map[string]any `json:"wildcard,omitempty"`
}
@@ -268,9 +269,10 @@ func parseESSearchResponse(raw []byte) ([]SearchHit, error) {
var esHighlightTag = regexp.MustCompile(`?em[^>]*>`)
// esHighlightText flattens a highlight map into one plain-text snippet,
-// preferring the content fragment over the title.
+// preferring the content fragment over the title. It covers both the
+// OnlyOffice content field and the own-index "content" field.
func esHighlightText(hl map[string][]string) string {
- for _, key := range []string{"document.attachment.content", "title"} {
+ for _, key := range []string{"document.attachment.content", "content", "title"} {
frags := hl[key]
if len(frags) == 0 {
continue
diff --git a/file_es_text.go b/file_es_text.go
new file mode 100644
index 0000000..072f00c
--- /dev/null
+++ b/file_es_text.go
@@ -0,0 +1,336 @@
+package onlyoffice
+
+// Own full-text index (epic #34, F6 #42).
+//
+// The OnlyOffice Elasticsearch index (files_file) only holds extracted content
+// for Office formats. FileUtility.CanIndex gates extraction by the server
+// setting files.index.formats, whose default is ".pptx|.xlsx|.docx", so PDFs
+// are indexed by name only. Instead of patching the server (risky: lost on
+// upgrade, forces a full reindex) this file implements a second, independent
+// index (default oo_docs_text) that our own pipeline fills from
+// internal/docpipe (pdftotext + OCR). The OnlyOffice index is never touched.
+//
+// See docs/elasticsearch.md for the decision and the trade-offs.
+
+import (
+ "bytes"
+ "context"
+ "encoding/json"
+ "fmt"
+ "io"
+ "net/http"
+ "os"
+ "strings"
+ "time"
+)
+
+const defaultESTextIndex = "oo_docs_text"
+
+// ESTextConfig configures the own full-text index.
+type ESTextConfig struct {
+ URL string // scheme://host:port of the ES HTTP endpoint
+ Index string // index name, default oo_docs_text
+ Tenant string // reserved for future multi-tenant data; unused for now
+}
+
+// ESTextConfigFromEnv reads ONLYOFFICE_ES_URL and ONLYOFFICE_ES_TEXT_INDEX
+// (default oo_docs_text). The library never loads dotfiles — the CLI does that.
+func ESTextConfigFromEnv() ESTextConfig {
+ return ESTextConfig{
+ URL: strings.TrimRight(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")), "/"),
+ Index: firstNonEmpty(os.Getenv("ONLYOFFICE_ES_TEXT_INDEX"), defaultESTextIndex),
+ Tenant: strings.TrimSpace(os.Getenv("ONLYOFFICE_TENANT")),
+ }
+}
+
+// TextDoc is one document in the own full-text index. It is keyed by the
+// OnlyOffice file id so hits map straight back to Documents entries.
+type TextDoc struct {
+ ID string `json:"id"`
+ Title string `json:"title"`
+ FolderID string `json:"folder,omitempty"`
+ Ext string `json:"ext,omitempty"`
+ Content string `json:"content"`
+}
+
+// TextIndex is the storage/search surface for locally extracted document text.
+// It complements Searcher: ESSearcher reads OnlyOffice's index, ESTextIndex
+// reads ours.
+type TextIndex interface {
+ Put(ctx context.Context, docs []TextDoc) error
+ Delete(ctx context.Context, ids []string) error
+ Search(ctx context.Context, q SearchQuery) ([]SearchHit, error)
+ Name() string
+}
+
+// ESTextIndex is a TextIndex (and Searcher) over a dedicated Elasticsearch
+// index filled by TextIndexer.
+type ESTextIndex struct {
+ cfg ESTextConfig
+ http *http.Client
+}
+
+var (
+ _ TextIndex = (*ESTextIndex)(nil)
+ _ Searcher = (*ESTextIndex)(nil)
+)
+
+// NewESTextIndex returns a searcher/writer for the own full-text index. The URL
+// is required; an empty index falls back to oo_docs_text.
+func NewESTextIndex(cfg ESTextConfig) (*ESTextIndex, error) {
+ if strings.TrimSpace(cfg.URL) == "" {
+ return nil, fmt.Errorf("onlyoffice: elasticsearch URL is empty (set ONLYOFFICE_ES_URL)")
+ }
+ cfg.URL = strings.TrimRight(cfg.URL, "/")
+ if cfg.Index == "" {
+ cfg.Index = defaultESTextIndex
+ }
+ return &ESTextIndex{cfg: cfg, http: &http.Client{Timeout: 120 * time.Second}}, nil
+}
+
+// Name implements Searcher and TextIndex.
+func (x *ESTextIndex) Name() string { return "es-text" }
+
+// Index returns the configured index name.
+func (x *ESTextIndex) Index() string { return x.cfg.Index }
+
+// esTextMapping pins explicit types: content must stay a plain text field (no
+// keyword subfield) and folder/ext stay exact keywords for filters.
+const esTextMapping = `{
+ "mappings": {
+ "properties": {
+ "id": {"type": "keyword"},
+ "title": {"type": "text", "fields": {"keyword": {"type": "keyword", "ignore_above": 512}}},
+ "folder": {"type": "keyword"},
+ "ext": {"type": "keyword"},
+ "content": {"type": "text"}
+ }
+ }
+}`
+
+// Ensure creates the index with the explicit mapping. A missing index is
+// created; an already existing one is left untouched.
+func (x *ESTextIndex) Ensure(ctx context.Context) error {
+ status, raw, err := x.do(ctx, http.MethodPut, "/"+x.cfg.Index, []byte(esTextMapping), "application/json")
+ if err != nil {
+ return err
+ }
+ if status == http.StatusOK {
+ return nil
+ }
+ if status == http.StatusBadRequest && bytes.Contains(raw, []byte("resource_already_exists_exception")) {
+ return nil
+ }
+ return fmt.Errorf("onlyoffice: create text index %s: %d %s", x.cfg.Index, status, truncate(string(raw), 300))
+}
+
+// Put upserts documents via the bulk API and refreshes so they are immediately
+// searchable.
+func (x *ESTextIndex) Put(ctx context.Context, docs []TextDoc) error {
+ if len(docs) == 0 {
+ return nil
+ }
+ status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_bulk?refresh=true", esTextBulkBody(docs), "application/x-ndjson")
+ if err != nil {
+ return err
+ }
+ if status >= 400 {
+ return fmt.Errorf("onlyoffice: bulk index %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400))
+ }
+ var res esBulkResponse
+ if err := json.Unmarshal(raw, &res); err != nil {
+ return fmt.Errorf("onlyoffice: decode bulk response: %w", err)
+ }
+ if !res.Errors {
+ return nil
+ }
+ return fmt.Errorf("onlyoffice: bulk index %s: %s", x.cfg.Index, res.firstError())
+}
+
+// Delete removes documents by OnlyOffice file id. A missing index means there
+// is nothing to delete.
+func (x *ESTextIndex) Delete(ctx context.Context, ids []string) error {
+ if len(ids) == 0 {
+ return nil
+ }
+ body, err := json.Marshal(map[string]any{"query": map[string]any{"terms": map[string]any{"id": ids}}})
+ if err != nil {
+ return err
+ }
+ status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_delete_by_query?refresh=true", body, "application/json")
+ if err != nil {
+ return err
+ }
+ if status == http.StatusNotFound {
+ return nil
+ }
+ if status >= 400 {
+ return fmt.Errorf("onlyoffice: delete from %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400))
+ }
+ return nil
+}
+
+// Search runs a multi_match over title (boosted) and content, with optional
+// folder and extension filters. A missing index yields no hits, not an error.
+func (x *ESTextIndex) Search(ctx context.Context, q SearchQuery) ([]SearchHit, error) {
+ q.Text = strings.TrimSpace(q.Text)
+ if q.Text == "" {
+ return nil, fmt.Errorf("onlyoffice: empty search query")
+ }
+ body, err := json.Marshal(esTextSearchRequest(q))
+ if err != nil {
+ return nil, fmt.Errorf("onlyoffice: build elasticsearch query: %w", err)
+ }
+ status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_search", body, "application/json")
+ if err != nil {
+ return nil, err
+ }
+ if status == http.StatusNotFound {
+ return nil, nil
+ }
+ if status >= 400 {
+ return nil, fmt.Errorf("onlyoffice: search %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400))
+ }
+ return parseESTextResponse(raw)
+}
+
+// do sends one request and returns the status and body (bounded). The caller
+// decides which statuses are errors.
+func (x *ESTextIndex) do(ctx context.Context, method, path string, body []byte, contentType string) (int, []byte, error) {
+ var r io.Reader
+ if body != nil {
+ r = bytes.NewReader(body)
+ }
+ req, err := http.NewRequestWithContext(ctx, method, x.cfg.URL+path, r)
+ if err != nil {
+ return 0, nil, err
+ }
+ req.Header.Set("Accept", "application/json")
+ if contentType != "" {
+ req.Header.Set("Content-Type", contentType)
+ }
+ resp, err := x.http.Do(req)
+ if err != nil {
+ return 0, nil, fmt.Errorf("onlyoffice: elasticsearch %s: %w", method, err)
+ }
+ defer resp.Body.Close()
+ raw, err := io.ReadAll(io.LimitReader(resp.Body, maxESResponseSize))
+ if err != nil {
+ return resp.StatusCode, nil, err
+ }
+ return resp.StatusCode, raw, nil
+}
+
+// esTextBulkBody renders the NDJSON bulk payload. Pure, so it is unit-tested.
+func esTextBulkBody(docs []TextDoc) []byte {
+ var b bytes.Buffer
+ enc := json.NewEncoder(&b)
+ enc.SetEscapeHTML(false)
+ for _, d := range docs {
+ _ = enc.Encode(map[string]any{"index": map[string]any{"_id": d.ID}})
+ _ = enc.Encode(d)
+ }
+ return b.Bytes()
+}
+
+// esTextSearchRequest builds the own-index query. Pure, so it is unit-tested.
+func esTextSearchRequest(q SearchQuery) esRequest {
+ limit := q.Limit
+ if limit <= 0 {
+ limit = defaultESLimit
+ }
+ if limit > maxESLimit {
+ limit = maxESLimit
+ }
+ fields := []string{"title^2", "content"}
+ must := []esClause{{MultiMatch: &esMultiMatch{Query: q.Text, Fields: fields}}}
+
+ var filter []esClause
+ if f := strings.TrimSpace(q.FolderID); f != "" {
+ filter = append(filter, esClause{Term: map[string]any{"folder": f}})
+ }
+ if exts := normalizeExtensions(q.Extensions); len(exts) > 0 {
+ filter = append(filter, esClause{Terms: map[string]any{"ext": exts}})
+ }
+
+ return esRequest{
+ Size: limit,
+ Source: []string{"id", "title", "folder", "ext"},
+ Query: esQuery{Bool: esBool{Must: must, Filter: filter}},
+ Highlight: esHighlight{PreTags: []string{""}, PostTags: []string{""}, Fields: map[string]struct{}{"title": {}, "content": {}}},
+ }
+}
+
+// esBulkResponse is the subset of an ES bulk response we consume.
+type esBulkResponse struct {
+ Errors bool `json:"errors"`
+ Items []map[string]struct {
+ ID string `json:"_id"`
+ Status int `json:"status"`
+ Error *struct {
+ Type string `json:"type"`
+ Reason string `json:"reason"`
+ } `json:"error"`
+ } `json:"items"`
+}
+
+// firstError returns a compact description of the first failed bulk item.
+func (r esBulkResponse) firstError() string {
+ for _, item := range r.Items {
+ for op, res := range item {
+ if res.Error != nil {
+ return fmt.Sprintf("%s %s: %s %s", op, res.ID, res.Error.Type, res.Error.Reason)
+ }
+ }
+ }
+ return "unknown bulk error"
+}
+
+// esTextResponse is the subset of an own-index search response we consume.
+type esTextResponse struct {
+ Hits struct {
+ Total struct {
+ Value int `json:"value"`
+ } `json:"total"`
+ Hits []struct {
+ ID string `json:"_id"`
+ Score float64 `json:"_score"`
+ Source TextDoc `json:"_source"`
+ HL map[string][]string `json:"highlight"`
+ } `json:"hits"`
+ } `json:"hits"`
+}
+
+// parseESTextResponse converts an own-index search response into SearchHit
+// values. Pure, so it is unit-tested.
+func parseESTextResponse(raw []byte) ([]SearchHit, error) {
+ var r esTextResponse
+ if err := json.Unmarshal(raw, &r); err != nil {
+ return nil, fmt.Errorf("onlyoffice: decode elasticsearch response: %w", err)
+ }
+ hits := make([]SearchHit, 0, len(r.Hits.Hits))
+ for _, h := range r.Hits.Hits {
+ id := h.Source.ID
+ if id == "" {
+ id = h.ID
+ }
+ parent := h.Source.FolderID
+ var path []string
+ if parent != "" {
+ path = []string{parent}
+ }
+ hits = append(hits, SearchHit{
+ Entry: Entry{
+ ID: id,
+ ParentID: parent,
+ Title: h.Source.Title,
+ Kind: File,
+ Provider: "es-text",
+ },
+ Score: h.Score,
+ Highlight: esHighlightText(h.HL),
+ Path: path,
+ })
+ }
+ return hits, nil
+}
diff --git a/file_es_text_integration_test.go b/file_es_text_integration_test.go
new file mode 100644
index 0000000..ed070e8
--- /dev/null
+++ b/file_es_text_integration_test.go
@@ -0,0 +1,92 @@
+//go:build integration
+
+package onlyoffice
+
+import (
+ "context"
+ "net/http"
+ "os"
+ "strings"
+ "testing"
+ "time"
+)
+
+// TestIntegrationESTextIndex verifies the own full-text index end to end
+// against a live Elasticsearch: create the index with its mapping, index a
+// document, find it by content (and reject it via a folder filter and after
+// deletion), then drop the throwaway index.
+//
+// Requires ONLYOFFICE_ES_URL (a reachable ES endpoint — in the current setup a
+// tunnel to the ES inside the OnlyOffice VM, see docs/elasticsearch.md). It
+// does not need OnlyOffice credentials because no file is downloaded: the
+// TextIndexer write path is covered by unit tests with a fake extractor.
+func TestIntegrationESTextIndex(t *testing.T) {
+ esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL"))
+ if esURL == "" {
+ t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test")
+ }
+ stamp := time.Now().UTC().Format("20060102150405")
+ idx, err := NewESTextIndex(ESTextConfig{URL: esURL, Index: "oo_docs_text_it_" + stamp})
+ if err != nil {
+ t.Fatalf("NewESTextIndex: %v", err)
+ }
+ ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
+ defer cancel()
+
+ t.Cleanup(func() {
+ cleanupCtx, done := context.WithTimeout(context.Background(), 30*time.Second)
+ defer done()
+ _, _, _ = idx.do(cleanupCtx, http.MethodDelete, "/"+idx.Index(), nil, "")
+ })
+
+ if err := idx.Ensure(ctx); err != nil {
+ t.Fatalf("Ensure: %v", err)
+ }
+ // Ensure is idempotent.
+ if err := idx.Ensure(ctx); err != nil {
+ t.Fatalf("Ensure (second): %v", err)
+ }
+
+ token := "gotes" + stamp
+ doc := TextDoc{
+ ID: "3578",
+ Title: "2026-07-28-S1021-Edelweiss-rechnung.pdf",
+ FolderID: "634",
+ Ext: "pdf",
+ Content: "Begleitzettel SGB XI — Rechnung " + token,
+ }
+ if err := idx.Put(ctx, []TextDoc{doc}); err != nil {
+ t.Fatalf("Put: %v", err)
+ }
+
+ hits, err := idx.Search(ctx, SearchQuery{Text: token})
+ if err != nil {
+ t.Fatalf("Search: %v", err)
+ }
+ if len(hits) != 1 || hits[0].ID != "3578" {
+ t.Fatalf("content search hits = %+v, want doc 3578", hits)
+ }
+ if !strings.Contains(hits[0].Highlight, token) {
+ t.Errorf("highlight = %q, want token", hits[0].Highlight)
+ }
+
+ if hits, err := idx.Search(ctx, SearchQuery{Text: token, FolderID: "999"}); err != nil {
+ t.Fatalf("Search with folder filter: %v", err)
+ } else if len(hits) != 0 {
+ t.Errorf("folder filter returned %d hits, want 0", len(hits))
+ }
+ if hits, err := idx.Search(ctx, SearchQuery{Text: token, Extensions: []string{"docx"}}); err != nil {
+ t.Fatalf("Search with ext filter: %v", err)
+ } else if len(hits) != 0 {
+ t.Errorf("ext filter returned %d hits, want 0", len(hits))
+ }
+
+ if err := idx.Delete(ctx, []string{"3578"}); err != nil {
+ t.Fatalf("Delete: %v", err)
+ }
+ if hits, err := idx.Search(ctx, SearchQuery{Text: token}); err != nil {
+ t.Fatalf("Search after delete: %v", err)
+ } else if len(hits) != 0 {
+ t.Errorf("after delete search returned %d hits, want 0", len(hits))
+ }
+}
diff --git a/file_es_text_test.go b/file_es_text_test.go
new file mode 100644
index 0000000..7e72012
--- /dev/null
+++ b/file_es_text_test.go
@@ -0,0 +1,275 @@
+package onlyoffice
+
+import (
+ "context"
+ "fmt"
+ "io"
+ "os"
+ "reflect"
+ "strings"
+ "testing"
+)
+
+func TestESTextSearchRequestShape(t *testing.T) {
+ got := esTextSearchRequest(SearchQuery{
+ Text: "S1021",
+ FolderID: "634",
+ Extensions: []string{".PDF", "pdf"},
+ Limit: 5,
+ })
+ if got.Size != 5 {
+ t.Errorf("size = %d, want 5", got.Size)
+ }
+ mm := got.Query.Bool.Must[0].MultiMatch
+ if mm == nil || !reflect.DeepEqual(mm.Fields, []string{"title^2", "content"}) {
+ t.Fatalf("multi_match = %+v, want title^2 + content", mm)
+ }
+ if _, ok := got.Highlight.Fields["content"]; !ok {
+ t.Error("content highlight missing")
+ }
+ if _, ok := got.Highlight.Fields["title"]; !ok {
+ t.Error("title highlight missing")
+ }
+ var folder, exts int
+ for _, f := range got.Query.Bool.Filter {
+ switch {
+ case f.Term != nil && f.Term["folder"] != nil:
+ folder++
+ if f.Term["folder"] != "634" {
+ t.Errorf("folder term = %+v", f.Term)
+ }
+ case f.Terms != nil:
+ exts++
+ if !reflect.DeepEqual(f.Terms["ext"], []string{"pdf"}) {
+ t.Errorf("ext terms = %+v, want deduped pdf", f.Terms)
+ }
+ }
+ }
+ if folder != 1 || exts != 1 {
+ t.Errorf("filters folder=%d exts=%d, want 1 each", folder, exts)
+ }
+}
+
+func TestESTextBulkBody(t *testing.T) {
+ body := esTextBulkBody([]TextDoc{
+ {ID: "3578", Title: "S1021.pdf", FolderID: "634", Ext: "pdf", Content: "Begleitzettel & mehr"},
+ {ID: "3579", Title: "S1023.pdf", Ext: "pdf", Content: "x"},
+ })
+ lines := strings.Split(strings.TrimRight(string(body), "\n"), "\n")
+ if len(lines) != 4 {
+ t.Fatalf("bulk body has %d lines, want 4:\n%s", len(lines), body)
+ }
+ if !strings.Contains(lines[0], `"index"`) || !strings.Contains(lines[0], `"_id":"3578"`) {
+ t.Errorf("action line = %q", lines[0])
+ }
+ if !strings.Contains(lines[1], `"content":"Begleitzettel & mehr"`) {
+ t.Errorf("source line should keep HTML unescaped, got %q", lines[1])
+ }
+ if !strings.Contains(lines[2], `"_id":"3579"`) {
+ t.Errorf("second action line = %q", lines[2])
+ }
+}
+
+func TestParseESTextResponse(t *testing.T) {
+ raw := []byte(`{
+ "hits": {
+ "total": {"value": 1, "relation": "eq"},
+ "hits": [
+ {
+ "_id": "3578",
+ "_score": 3.21,
+ "_source": {"id": "3578", "title": "2026-07-28-S1021-Edelweiss-rechnung.pdf", "folder": "634", "ext": "pdf"},
+ "highlight": {"content": ["Begleitzettel … S1021 …"]}
+ }
+ ]
+ }
+ }`)
+ hits, err := parseESTextResponse(raw)
+ if err != nil {
+ t.Fatalf("parseESTextResponse: %v", err)
+ }
+ if len(hits) != 1 {
+ t.Fatalf("hits = %d, want 1", len(hits))
+ }
+ h := hits[0]
+ if h.ID != "3578" || h.Title != "2026-07-28-S1021-Edelweiss-rechnung.pdf" || h.Kind != File {
+ t.Errorf("entry = %+v", h.Entry)
+ }
+ if h.ParentID != "634" || !reflect.DeepEqual(h.Path, []string{"634"}) {
+ t.Errorf("path = %v parent = %q", h.Path, h.ParentID)
+ }
+ if h.Provider != "es-text" {
+ t.Errorf("provider = %q", h.Provider)
+ }
+ if h.Highlight != "Begleitzettel … S1021 …" {
+ t.Errorf("highlight = %q, want tags stripped", h.Highlight)
+ }
+}
+
+func TestNewESTextIndexDefaults(t *testing.T) {
+ if _, err := NewESTextIndex(ESTextConfig{}); err == nil {
+ t.Error("empty URL: want error")
+ }
+ x, err := NewESTextIndex(ESTextConfig{URL: "http://es:9200/"})
+ if err != nil {
+ t.Fatalf("NewESTextIndex: %v", err)
+ }
+ if x.Index() != defaultESTextIndex {
+ t.Errorf("index = %q, want %q", x.Index(), defaultESTextIndex)
+ }
+ if x.cfg.URL != "http://es:9200" {
+ t.Errorf("url = %q, want trimmed", x.cfg.URL)
+ }
+ if x.Name() != "es-text" {
+ t.Errorf("Name() = %q", x.Name())
+ }
+}
+
+func TestESTextConfigFromEnvIndexDefault(t *testing.T) {
+ t.Setenv("ONLYOFFICE_ES_URL", "http://es:9200/")
+ t.Setenv("ONLYOFFICE_ES_TEXT_INDEX", "")
+ cfg := ESTextConfigFromEnv()
+ if cfg.Index != defaultESTextIndex {
+ t.Errorf("index = %q, want %q", cfg.Index, defaultESTextIndex)
+ }
+}
+
+func TestTextIndexerIndexEntries(t *testing.T) {
+ store := &fakeStore{
+ files: map[string][]byte{"1": []byte("PDFBYTES")},
+ }
+ idx := &fakeIndex{}
+ ix := NewTextIndexer(store, idx)
+ ix.Extractor = fakeExtractor{prefix: "TEXT "}
+
+ res, err := ix.IndexEntries(context.Background(), []Entry{
+ {ID: "1", Title: "Rechnung.PDF", ParentID: "649", Kind: File},
+ {ID: "2", Title: "Tabelle.xlsx", ParentID: "649", Kind: File},
+ {ID: "3", Title: "Unterordner", Kind: Folder},
+ }, IndexOptions{})
+ if err != nil {
+ t.Fatalf("IndexEntries: %v", err)
+ }
+ if res.Scanned != 3 || res.Indexed != 1 || res.Skipped != 2 || res.Failed != 0 {
+ t.Errorf("result = %+v, want scanned=3 indexed=1 skipped=2 failed=0", res)
+ }
+ if len(idx.docs) != 1 {
+ t.Fatalf("indexed docs = %d, want 1", len(idx.docs))
+ }
+ got := idx.docs[0]
+ want := TextDoc{ID: "1", Title: "Rechnung.PDF", FolderID: "649", Ext: "pdf", Content: "TEXT PDFBYTES"}
+ if !reflect.DeepEqual(got, want) {
+ t.Errorf("doc = %+v, want %+v", got, want)
+ }
+}
+
+func TestTextIndexerRecordsExtractionFailure(t *testing.T) {
+ store := &fakeStore{files: map[string][]byte{"1": []byte("x")}}
+ idx := &fakeIndex{}
+ ix := NewTextIndexer(store, idx)
+ ix.Extractor = failingExtractor{}
+
+ res, err := ix.IndexEntries(context.Background(), []Entry{{ID: "1", Title: "a.pdf", Kind: File}}, IndexOptions{})
+ if err != nil {
+ t.Fatalf("IndexEntries: %v", err)
+ }
+ if res.Indexed != 0 || res.Failed != 1 || len(res.Errors) != 1 {
+ t.Errorf("result = %+v, want one failure recorded", res)
+ }
+}
+
+func TestTextIndexerPlanFolder(t *testing.T) {
+ store := &fakeStore{dirs: map[string][]Entry{
+ "root": {
+ {ID: "10", Title: "a.pdf", Kind: File},
+ {ID: "11", Title: "sub", Kind: Folder},
+ },
+ "11": {
+ {ID: "12", Title: "b.PDF", Kind: File},
+ {ID: "13", Title: "c.xlsx", Kind: File},
+ },
+ }}
+ ix := NewTextIndexer(store, &fakeIndex{})
+
+ flat, err := ix.PlanFolder(context.Background(), "root", IndexOptions{})
+ if err != nil {
+ t.Fatalf("PlanFolder: %v", err)
+ }
+ if len(flat) != 1 || flat[0].ID != "10" {
+ t.Errorf("flat plan = %+v, want only a.pdf", flat)
+ }
+ deep, err := ix.PlanFolder(context.Background(), "root", IndexOptions{Recursive: true, Limit: 10})
+ if err != nil {
+ t.Fatalf("PlanFolder recursive: %v", err)
+ }
+ if len(deep) != 2 {
+ t.Errorf("recursive plan = %d entries, want 2", len(deep))
+ }
+}
+
+// --- fakes -----------------------------------------------------------------
+
+type fakeStore struct {
+ dirs map[string][]Entry
+ files map[string][]byte
+ stat map[string]Entry
+}
+
+func (f *fakeStore) Name() string { return "fake" }
+
+func (f *fakeStore) List(_ context.Context, parentID string) ([]Entry, error) {
+ return f.dirs[parentID], nil
+}
+
+func (f *fakeStore) Stat(_ context.Context, id string) (Entry, error) {
+ if e, ok := f.stat[id]; ok {
+ return e, nil
+ }
+ return Entry{}, fmt.Errorf("not found: %s", id)
+}
+
+func (f *fakeStore) Download(_ context.Context, id string, w io.Writer) (int64, error) {
+ b, ok := f.files[id]
+ if !ok {
+ return 0, fmt.Errorf("no bytes for %s", id)
+ }
+ n, err := w.Write(b)
+ return int64(n), err
+}
+
+func (f *fakeStore) CreateFolder(context.Context, string, string) (Entry, error) {
+ return Entry{}, nil
+}
+func (f *fakeStore) Upload(context.Context, string, string, io.Reader) (Entry, error) {
+ return Entry{}, nil
+}
+func (f *fakeStore) Move(context.Context, []string, string) error { return nil }
+func (f *fakeStore) Copy(context.Context, []string, string) error { return nil }
+func (f *fakeStore) Rename(context.Context, string, string) error { return nil }
+func (f *fakeStore) Delete(context.Context, []string) error { return nil }
+
+type fakeIndex struct{ docs []TextDoc }
+
+func (f *fakeIndex) Put(_ context.Context, docs []TextDoc) error {
+ f.docs = append(f.docs, docs...)
+ return nil
+}
+func (f *fakeIndex) Delete(context.Context, []string) error { return nil }
+func (f *fakeIndex) Search(context.Context, SearchQuery) ([]SearchHit, error) { return nil, nil }
+func (f *fakeIndex) Name() string { return "fake" }
+
+type fakeExtractor struct{ prefix string }
+
+func (f fakeExtractor) Extract(path, _, _ string, _ int) (string, error) {
+ b, err := os.ReadFile(path)
+ if err != nil {
+ return "", err
+ }
+ return f.prefix + string(b), nil
+}
+
+type failingExtractor struct{}
+
+func (failingExtractor) Extract(string, string, string, int) (string, error) {
+ return "", fmt.Errorf("boom")
+}
diff --git a/file_text_index.go b/file_text_index.go
new file mode 100644
index 0000000..005eacf
--- /dev/null
+++ b/file_text_index.go
@@ -0,0 +1,335 @@
+package onlyoffice
+
+// Text extraction pipeline for the own full-text index (epic #34, F6 #42).
+//
+// TextIndexer downloads stored documents, extracts text through docpipe
+// (pdftotext; OCR for scans) and writes the result to a TextIndex. It is the
+// write side of ESTextIndex and never touches the OnlyOffice server's own ES
+// index.
+
+import (
+ "context"
+ "fmt"
+ "os"
+ "path/filepath"
+ "strings"
+ "sync"
+
+ "github.com/eslider/go-onlyoffice/internal/docpipe"
+)
+
+// defaultTextIndexExts are the formats extracted by default. The OnlyOffice
+// index already covers docx/xlsx/pptx; F6 adds PDF.
+var defaultTextIndexExts = []string{"pdf"}
+
+const (
+ defaultTextIndexWorkers = 3
+ defaultTextIndexLang = "deu+eng"
+ maxTextIndexErrors = 20
+)
+
+// TextExtractor turns a local file into indexable plain text. The default uses
+// docpipe (pdftotext + OCR); tests inject a fake to stay offline.
+type TextExtractor interface {
+ Extract(path, workDir, lang string, minChars int) (string, error)
+}
+
+// docpipeExtractor is the production TextExtractor.
+type docpipeExtractor struct{ tools docpipe.Tools }
+
+// Extract renders the file as Markdown, OCRing PDFs/images with a weak text
+// layer first (docpipe.ToMarkdown).
+func (d docpipeExtractor) Extract(path, workDir, lang string, minChars int) (string, error) {
+ res, err := d.tools.ToMarkdown(path, workDir, lang, minChars)
+ if err != nil {
+ return "", err
+ }
+ return strings.TrimSpace(res.Markdown), nil
+}
+
+// IndexOptions controls a TextIndexer run.
+type IndexOptions struct {
+ Recursive bool // IndexFolder: descend into subfolders
+ Extensions []string // empty = defaultTextIndexExts (pdf)
+ Limit int // max files to index, 0 = all
+ Lang string // OCR language(s), default deu+eng
+ MinChars int // OCR threshold, default docpipe.DefaultMinTextChars
+ Workers int // parallel downloads/extractions, default 3
+}
+
+// IndexResult summarises a run.
+type IndexResult struct {
+ Scanned int
+ Indexed int
+ Skipped int
+ Failed int
+ Errors []string
+}
+
+// TextIndexer wires a FileStore, a TextIndex and an extractor together.
+type TextIndexer struct {
+ Store FileStore
+ Index TextIndex
+ Extractor TextExtractor // nil = local docpipe tools
+ WorkDir string // temp dir for downloads/extraction
+}
+
+// NewTextIndexer returns a TextIndexer over the given store and index.
+func NewTextIndexer(store FileStore, index TextIndex) *TextIndexer {
+ return &TextIndexer{Store: store, Index: index}
+}
+
+// textIndexEnsurer is implemented by indexes that can be created up front.
+type textIndexEnsurer interface {
+ Ensure(ctx context.Context) error
+}
+
+// Ensure creates the backing index when the TextIndex supports it.
+func (ix *TextIndexer) Ensure(ctx context.Context) error {
+ if e, ok := ix.Index.(textIndexEnsurer); ok {
+ return e.Ensure(ctx)
+ }
+ return nil
+}
+
+// IndexFiles stats the given file ids and indexes them.
+func (ix *TextIndexer) IndexFiles(ctx context.Context, ids []string, opts IndexOptions) (IndexResult, error) {
+ entries := make([]Entry, 0, len(ids))
+ for _, id := range ids {
+ e, err := ix.Store.Stat(ctx, id)
+ if err != nil {
+ return IndexResult{}, fmt.Errorf("onlyoffice: stat %s: %w", id, err)
+ }
+ entries = append(entries, e)
+ }
+ return ix.IndexEntries(ctx, entries, opts)
+}
+
+// IndexFolder lists a folder and indexes every matching file.
+func (ix *TextIndexer) IndexFolder(ctx context.Context, folderID string, opts IndexOptions) (IndexResult, error) {
+ entries, err := ix.collect(ctx, folderID, opts.Recursive)
+ if err != nil {
+ return IndexResult{}, err
+ }
+ return ix.IndexEntries(ctx, entries, opts)
+}
+
+// PlanFolder lists the files IndexFolder would process, without downloading or
+// extracting anything.
+func (ix *TextIndexer) PlanFolder(ctx context.Context, folderID string, opts IndexOptions) ([]Entry, error) {
+ entries, err := ix.collect(ctx, folderID, opts.Recursive)
+ if err != nil {
+ return nil, err
+ }
+ return selectEntries(entries, opts), nil
+}
+
+// PlanFiles stats the ids and returns those that would be indexed.
+func (ix *TextIndexer) PlanFiles(ctx context.Context, ids []string, opts IndexOptions) ([]Entry, error) {
+ entries := make([]Entry, 0, len(ids))
+ for _, id := range ids {
+ e, err := ix.Store.Stat(ctx, id)
+ if err != nil {
+ return nil, fmt.Errorf("onlyoffice: stat %s: %w", id, err)
+ }
+ entries = append(entries, e)
+ }
+ return selectEntries(entries, opts), nil
+}
+
+// IndexEntries extracts and indexes the given files (folders are ignored).
+func (ix *TextIndexer) IndexEntries(ctx context.Context, entries []Entry, opts IndexOptions) (IndexResult, error) {
+ opts = opts.withDefaults()
+ var res IndexResult
+ work := selectEntries(entries, opts)
+ res.Scanned = len(entries)
+ res.Skipped = len(entries) - len(work)
+ if len(work) == 0 {
+ return res, nil
+ }
+
+ workers := opts.Workers
+ if workers > len(work) {
+ workers = len(work)
+ }
+ if workers < 1 {
+ workers = 1
+ }
+
+ type outcome struct {
+ doc TextDoc
+ err error
+ }
+ jobs := make(chan Entry)
+ results := make(chan outcome, workers)
+ var wg sync.WaitGroup
+ for i := 0; i < workers; i++ {
+ wg.Add(1)
+ go func() {
+ defer wg.Done()
+ for e := range jobs {
+ if err := ctx.Err(); err != nil {
+ results <- outcome{err: err}
+ continue
+ }
+ doc, err := ix.indexOne(ctx, e, opts)
+ results <- outcome{doc: doc, err: err}
+ }
+ }()
+ }
+ go func() {
+ defer close(jobs)
+ for _, e := range work {
+ select {
+ case jobs <- e:
+ case <-ctx.Done():
+ return
+ }
+ }
+ }()
+ go func() {
+ wg.Wait()
+ close(results)
+ }()
+
+ var docs []TextDoc
+ for r := range results {
+ if r.err != nil {
+ res.Failed++
+ if len(res.Errors) < maxTextIndexErrors {
+ res.Errors = append(res.Errors, r.err.Error())
+ }
+ continue
+ }
+ docs = append(docs, r.doc)
+ }
+ if err := ctx.Err(); err != nil {
+ return res, err
+ }
+ if len(docs) > 0 {
+ if err := ix.Index.Put(ctx, docs); err != nil {
+ return res, fmt.Errorf("onlyoffice: index %d docs: %w", len(docs), err)
+ }
+ res.Indexed = len(docs)
+ }
+ return res, nil
+}
+
+// indexOne downloads and extracts a single file.
+func (ix *TextIndexer) indexOne(ctx context.Context, e Entry, opts IndexOptions) (TextDoc, error) {
+ ext := fileExt(e.Title)
+ dir := ix.WorkDir
+ if dir == "" {
+ dir = os.TempDir()
+ }
+ if err := os.MkdirAll(dir, 0o755); err != nil {
+ return TextDoc{}, err
+ }
+ tmp, err := os.CreateTemp(dir, "ooidx-*."+ext)
+ if err != nil {
+ return TextDoc{}, err
+ }
+ tmpPath := tmp.Name()
+ defer os.Remove(tmpPath)
+
+ if _, err := ix.Store.Download(ctx, e.ID, tmp); err != nil {
+ tmp.Close()
+ return TextDoc{}, fmt.Errorf("download %s (%s): %w", e.ID, e.Title, err)
+ }
+ if err := tmp.Close(); err != nil {
+ return TextDoc{}, err
+ }
+ text, err := ix.extractor().Extract(tmpPath, dir, opts.Lang, opts.MinChars)
+ if err != nil {
+ return TextDoc{}, fmt.Errorf("extract %s: %w", e.Title, err)
+ }
+ return TextDoc{ID: e.ID, Title: e.Title, FolderID: e.ParentID, Ext: ext, Content: text}, nil
+}
+
+// collect lists files under folderID, breadth-first when recursive.
+func (ix *TextIndexer) collect(ctx context.Context, folderID string, recursive bool) ([]Entry, error) {
+ var files []Entry
+ queue := []string{folderID}
+ for len(queue) > 0 {
+ if err := ctx.Err(); err != nil {
+ return nil, err
+ }
+ id := queue[0]
+ queue = queue[1:]
+ entries, err := ix.Store.List(ctx, id)
+ if err != nil {
+ return nil, fmt.Errorf("onlyoffice: list folder %s: %w", id, err)
+ }
+ for _, e := range entries {
+ if e.Kind == Folder {
+ if recursive {
+ queue = append(queue, e.ID)
+ }
+ continue
+ }
+ if e.ParentID == "" {
+ e.ParentID = id
+ }
+ files = append(files, e)
+ }
+ }
+ return files, nil
+}
+
+// selectEntries filters files by extension and applies the limit.
+func selectEntries(entries []Entry, opts IndexOptions) []Entry {
+ allowed := extensionSet(opts.Extensions)
+ work := make([]Entry, 0, len(entries))
+ for _, e := range entries {
+ if e.Kind != File {
+ continue
+ }
+ if opts.Limit > 0 && len(work) >= opts.Limit {
+ break
+ }
+ if !allowed[fileExt(e.Title)] {
+ continue
+ }
+ work = append(work, e)
+ }
+ return work
+}
+
+// extensionSet normalises the extension allow-list (default: pdf).
+func extensionSet(exts []string) map[string]bool {
+ if len(exts) == 0 {
+ exts = defaultTextIndexExts
+ }
+ set := make(map[string]bool, len(exts))
+ for _, e := range normalizeExtensions(exts) {
+ set[e] = true
+ }
+ return set
+}
+
+// fileExt returns the lower-case extension without the dot.
+func fileExt(title string) string {
+ return strings.ToLower(strings.TrimPrefix(filepath.Ext(strings.TrimSpace(title)), "."))
+}
+
+// withDefaults fills zero-valued options.
+func (o IndexOptions) withDefaults() IndexOptions {
+ if o.Workers <= 0 {
+ o.Workers = defaultTextIndexWorkers
+ }
+ if o.MinChars <= 0 {
+ o.MinChars = docpipe.DefaultMinTextChars
+ }
+ if strings.TrimSpace(o.Lang) == "" {
+ o.Lang = defaultTextIndexLang
+ }
+ return o
+}
+
+// extractor returns the configured extractor or the local docpipe default.
+func (ix *TextIndexer) extractor() TextExtractor {
+ if ix.Extractor != nil {
+ return ix.Extractor
+ }
+ return docpipeExtractor{tools: docpipe.LookPath()}
+}