Merge pull request 'feat(search): PDF-контент в поиске — свой индекс oo_docs_text (#42)' (#44) from feat/pdf-content#42 into main
This commit was merged in pull request #44.
This commit is contained in:
@@ -55,3 +55,8 @@ ONLYOFFICE_PROJECT_ID=33
|
|||||||
# ONLYOFFICE_PG_PASSWORD=
|
# ONLYOFFICE_PG_PASSWORD=
|
||||||
# ONLYOFFICE_PG_DBNAME=onlyoffice
|
# ONLYOFFICE_PG_DBNAME=onlyoffice
|
||||||
# ONLYOFFICE_PG_SSLMODE=disable
|
# ONLYOFFICE_PG_SSLMODE=disable
|
||||||
|
|
||||||
|
# oo index / oo search --backend own — own full-text index for PDF/scans,
|
||||||
|
# filled by `oo index` from internal/docpipe (pdftotext + OCR). Defaults to
|
||||||
|
# oo_docs_text. Uses the same ONLYOFFICE_ES_URL.
|
||||||
|
# ONLYOFFICE_ES_TEXT_INDEX=oo_docs_text
|
||||||
|
|||||||
@@ -697,6 +697,27 @@ oo search "Rechnung" --json # shorthand for -o json
|
|||||||
Requires `ONLYOFFICE_ES_URL` (plus optional `ONLYOFFICE_ES_INDEX`,
|
Requires `ONLYOFFICE_ES_URL` (plus optional `ONLYOFFICE_ES_INDEX`,
|
||||||
`ONLYOFFICE_TENANT`).
|
`ONLYOFFICE_TENANT`).
|
||||||
|
|
||||||
|
#### PDF/scans: own index (`oo index` + `--backend own`)
|
||||||
|
|
||||||
|
The OnlyOffice index covers Office formats only, so PDFs (`S1019`-style invoice
|
||||||
|
numbers) are not searchable by content. `oo index` extracts PDF text with
|
||||||
|
`internal/docpipe` (pdftotext, OCR for scans) — including the text of embedded
|
||||||
|
PDF attachments (`pdfdetach`: `<doc>.md`, `.xml`, covers the original/scan and
|
||||||
|
ZUGFeRD e-invoice XML) — into a separate index (`ONLYOFFICE_ES_TEXT_INDEX`,
|
||||||
|
default `oo_docs_text`); the OnlyOffice server and its index are **not**
|
||||||
|
modified. Then search it with `--backend own`.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
oo index folder 634 --recursive --exts pdf # populate (idempotent upsert)
|
||||||
|
oo index files 3576 3578 # specific files
|
||||||
|
oo index folder 634 --dry-run # plan only
|
||||||
|
oo search "S1021" --content --backend own # finds the PDF
|
||||||
|
oo search "Rechnung" --backend own --folder 634 --json
|
||||||
|
```
|
||||||
|
|
||||||
|
See [`docs/elasticsearch.md`](docs/elasticsearch.md) for the decision and
|
||||||
|
trade-offs.
|
||||||
|
|
||||||
### Bulk tools (`cmd/`)
|
### Bulk tools (`cmd/`)
|
||||||
|
|
||||||
Small single-purpose binaries for bulk Documents work. All of them pace
|
Small single-purpose binaries for bulk Documents work. All of them pace
|
||||||
@@ -733,7 +754,8 @@ kontolink IN.xlsx oo-index.tsv OUT.xlsx [FILE_ID] [AMOUNTS_TSV]
|
|||||||
| `docs` | `tools`, `convert`, `optimize`, `ocr`, `hocr`, `as-md`, `put-md`, `put-txt`, `put-xlsx` |
|
| `docs` | `tools`, `convert`, `optimize`, `ocr`, `hocr`, `as-md`, `put-md`, `put-txt`, `put-xlsx` |
|
||||||
| `catalog` | `match`, `merge`, `apply`, `scan-contacts`, `scan-projects`, `scan-thunderbird` |
|
| `catalog` | `match`, `merge`, `apply`, `scan-contacts`, `scan-projects`, `scan-thunderbird` |
|
||||||
| `dav` | `ls`, `move`, `copy`, `mkdir`, `rename-file`, `rename-folder`, `download`, `fileops` |
|
| `dav` | `ls`, `move`, `copy`, `mkdir`, `rename-file`, `rename-folder`, `download`, `fileops` |
|
||||||
| `search` | `QUERY` (`--content`, `--folder ID`, `--limit N`, `--json`) |
|
| `search` | `QUERY` (`--content`, `--folder ID`, `--limit N`, `--backend oo\|own`, `--json`) |
|
||||||
|
| `index` | `folder FOLDER_ID`, `files FILE_ID...` (`--recursive`, `--exts pdf`, `--limit N`, `--dry-run`) |
|
||||||
|
|
||||||
The CLI reads only `.env` from the current working directory (godotenv is a
|
The CLI reads only `.env` from the current working directory (godotenv is a
|
||||||
CLI-only concern — the library itself never loads dotfiles).
|
CLI-only concern — the library itself never loads dotfiles).
|
||||||
|
|||||||
+158
@@ -0,0 +1,158 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||||
|
"github.com/spf13/cobra"
|
||||||
|
)
|
||||||
|
|
||||||
|
func init() {
|
||||||
|
rootCmd.AddCommand(indexCmd())
|
||||||
|
}
|
||||||
|
|
||||||
|
// indexFlags are shared by the `oo index folder` and `oo index files` verbs.
|
||||||
|
type indexFlags struct {
|
||||||
|
recursive bool
|
||||||
|
exts string
|
||||||
|
limit int
|
||||||
|
workers int
|
||||||
|
lang string
|
||||||
|
minChars int
|
||||||
|
workDir string
|
||||||
|
backend string
|
||||||
|
dryRun bool
|
||||||
|
asJSON bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// indexCmd populates the own full-text index (oo_docs_text) that makes PDF and
|
||||||
|
// scanned content searchable. The OnlyOffice index is left untouched.
|
||||||
|
func indexCmd() *cobra.Command {
|
||||||
|
f := &indexFlags{}
|
||||||
|
cmd := &cobra.Command{
|
||||||
|
Use: "index",
|
||||||
|
Short: "Populate the own full-text index for PDF/scan content",
|
||||||
|
Long: "Index document text that the OnlyOffice Elasticsearch index does not\n" +
|
||||||
|
"cover (PDFs and scans) into a separate index (ONLYOFFICE_ES_TEXT_INDEX,\n" +
|
||||||
|
"default oo_docs_text). Text is extracted with docpipe (pdftotext, OCR)\n" +
|
||||||
|
"and the OnlyOffice server is never modified.\n\n" +
|
||||||
|
"Requires ONLYOFFICE_URL/USER/PASS (to download files) and ONLYOFFICE_ES_URL\n" +
|
||||||
|
"(to write the index). See docs/elasticsearch.md.",
|
||||||
|
}
|
||||||
|
cmd.PersistentFlags().BoolVar(&f.recursive, "recursive", false, "folder: descend into subfolders")
|
||||||
|
cmd.PersistentFlags().StringVar(&f.exts, "exts", "pdf", "comma-separated extensions to index")
|
||||||
|
cmd.PersistentFlags().IntVar(&f.limit, "limit", 0, "maximum number of files to index (0 = all)")
|
||||||
|
cmd.PersistentFlags().IntVar(&f.workers, "workers", 3, "parallel downloads/extractions")
|
||||||
|
cmd.PersistentFlags().StringVar(&f.lang, "lang", "deu+eng", "OCR language(s)")
|
||||||
|
cmd.PersistentFlags().IntVar(&f.minChars, "min-chars", 0, "text-layer threshold below which OCR runs")
|
||||||
|
cmd.PersistentFlags().StringVar(&f.workDir, "work-dir", "", "temp dir for downloads (default: system temp)")
|
||||||
|
cmd.PersistentFlags().StringVar(&f.backend, "backend", "rest", "file backend: rest|dav")
|
||||||
|
cmd.PersistentFlags().BoolVar(&f.dryRun, "dry-run", false, "list what would be indexed, without changes")
|
||||||
|
cmd.PersistentFlags().BoolVar(&f.asJSON, "json", false, "shorthand for --output json")
|
||||||
|
|
||||||
|
cmd.AddCommand(indexFolderCmd(f), indexFilesCmd(f))
|
||||||
|
return cmd
|
||||||
|
}
|
||||||
|
|
||||||
|
func indexFolderCmd(f *indexFlags) *cobra.Command {
|
||||||
|
return &cobra.Command{
|
||||||
|
Use: "folder FOLDER_ID",
|
||||||
|
Short: "Index every matching file in a Documents folder",
|
||||||
|
Args: cobra.ExactArgs(1),
|
||||||
|
RunE: func(cmd *cobra.Command, args []string) error {
|
||||||
|
return runIndex(cmd, f, args[0], nil)
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func indexFilesCmd(f *indexFlags) *cobra.Command {
|
||||||
|
return &cobra.Command{
|
||||||
|
Use: "files FILE_ID...",
|
||||||
|
Short: "Index specific Documents files",
|
||||||
|
Args: cobra.MinimumNArgs(1),
|
||||||
|
RunE: func(cmd *cobra.Command, args []string) error {
|
||||||
|
return runIndex(cmd, f, "", args)
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func runIndex(cmd *cobra.Command, f *indexFlags, folderID string, ids []string) error {
|
||||||
|
if f.asJSON {
|
||||||
|
outputFormat = "json"
|
||||||
|
}
|
||||||
|
c, err := newOO(cmd)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
idx, err := onlyoffice.NewESTextIndex(onlyoffice.ESTextConfigFromEnv())
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
ti := onlyoffice.NewTextIndexer(c.FileStore(f.backend), idx)
|
||||||
|
ti.WorkDir = f.workDir
|
||||||
|
opts := onlyoffice.IndexOptions{
|
||||||
|
Recursive: f.recursive,
|
||||||
|
Extensions: splitList(f.exts),
|
||||||
|
Limit: f.limit,
|
||||||
|
Lang: f.lang,
|
||||||
|
MinChars: f.minChars,
|
||||||
|
Workers: f.workers,
|
||||||
|
}
|
||||||
|
ctx := cmd.Context()
|
||||||
|
|
||||||
|
if f.dryRun {
|
||||||
|
var entries []onlyoffice.Entry
|
||||||
|
if folderID != "" {
|
||||||
|
entries, err = ti.PlanFolder(ctx, folderID, opts)
|
||||||
|
} else {
|
||||||
|
entries, err = ti.PlanFiles(ctx, ids, opts)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
rows := make([]map[string]any, 0, len(entries))
|
||||||
|
for _, e := range entries {
|
||||||
|
rows = append(rows, map[string]any{
|
||||||
|
"id": e.ID,
|
||||||
|
"title": e.Title,
|
||||||
|
"folder": e.ParentID,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
printTable([]string{"id", "title", "folder"}, rows)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := ti.Ensure(ctx); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
var res onlyoffice.IndexResult
|
||||||
|
if folderID != "" {
|
||||||
|
res, err = ti.IndexFolder(ctx, folderID, opts)
|
||||||
|
} else {
|
||||||
|
res, err = ti.IndexFiles(ctx, ids, opts)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
printObject(map[string]any{
|
||||||
|
"index": idx.Index(),
|
||||||
|
"scanned": res.Scanned,
|
||||||
|
"indexed": res.Indexed,
|
||||||
|
"skipped": res.Skipped,
|
||||||
|
"failed": res.Failed,
|
||||||
|
"errors": res.Errors,
|
||||||
|
})
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// splitList parses a comma-separated flag value, dropping blanks.
|
||||||
|
func splitList(s string) []string {
|
||||||
|
parts := strings.Split(s, ",")
|
||||||
|
out := make([]string, 0, len(parts))
|
||||||
|
for _, p := range parts {
|
||||||
|
if p = strings.TrimSpace(p); p != "" {
|
||||||
|
out = append(out, p)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"reflect"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestIndexCommandRegistered(t *testing.T) {
|
||||||
|
cmd, _, err := rootCmd.Find([]string{"index"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if cmd.Name() != "index" {
|
||||||
|
t.Fatalf("index resolved to %q", cmd.Name())
|
||||||
|
}
|
||||||
|
for _, name := range []string{"exts", "limit", "workers", "lang", "min-chars", "work-dir", "backend", "dry-run", "recursive", "json"} {
|
||||||
|
if cmd.PersistentFlags().Lookup(name) == nil {
|
||||||
|
t.Errorf("index: missing --%s flag", name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if cmd.PersistentFlags().Lookup("exts").DefValue != "pdf" {
|
||||||
|
t.Errorf("--exts default = %q, want pdf", cmd.PersistentFlags().Lookup("exts").DefValue)
|
||||||
|
}
|
||||||
|
for _, verb := range []string{"index folder", "index files"} {
|
||||||
|
sub, _, err := rootCmd.Find(strings.Fields(verb))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s: %v", verb, err)
|
||||||
|
}
|
||||||
|
if sub.Name() != strings.Fields(verb)[1] {
|
||||||
|
t.Errorf("%s resolved to %q", verb, sub.Name())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestIndexSearchBackendFlag(t *testing.T) {
|
||||||
|
cmd, _, err := rootCmd.Find([]string{"search"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if cmd.Flags().Lookup("backend") == nil {
|
||||||
|
t.Fatal("search: missing --backend flag")
|
||||||
|
}
|
||||||
|
if cmd.Flags().Lookup("backend").DefValue != "oo" {
|
||||||
|
t.Errorf("--backend default = %q, want oo", cmd.Flags().Lookup("backend").DefValue)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSplitList(t *testing.T) {
|
||||||
|
got := splitList(" pdf , .PDF, docx ,, ")
|
||||||
|
want := []string{"pdf", ".PDF", "docx"}
|
||||||
|
if !reflect.DeepEqual(got, want) {
|
||||||
|
t.Errorf("splitList = %v, want %v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
+2
-1
@@ -18,7 +18,8 @@
|
|||||||
// oo docs tools | convert | optimize | ocr | hocr | as-md | put-md | put-txt | put-xlsx
|
// oo docs tools | convert | optimize | ocr | hocr | as-md | put-md | put-txt | put-xlsx
|
||||||
// oo catalog match | merge | apply | scan-contacts | scan-projects | scan-thunderbird
|
// oo catalog match | merge | apply | scan-contacts | scan-projects | scan-thunderbird
|
||||||
// oo dav ls | move | copy | mkdir | rename-file | rename-folder | download | fileops
|
// oo dav ls | move | copy | mkdir | rename-file | rename-folder | download | fileops
|
||||||
// oo search QUERY [--content] [--folder ID] [--limit N] [--json]
|
// oo search QUERY [--content] [--folder ID] [--limit N] [--backend oo|own] [--json]
|
||||||
|
// oo index folder FOLDER_ID | files FILE_ID... [--recursive] [--exts pdf] [--dry-run]
|
||||||
//
|
//
|
||||||
// CRM association rules: docs/crm-associations.md
|
// CRM association rules: docs/crm-associations.md
|
||||||
//
|
//
|
||||||
|
|||||||
+20
-1
@@ -1,6 +1,9 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||||
"github.com/eslider/go-onlyoffice/cmd/internal/bootstrap"
|
"github.com/eslider/go-onlyoffice/cmd/internal/bootstrap"
|
||||||
"github.com/spf13/cobra"
|
"github.com/spf13/cobra"
|
||||||
@@ -18,6 +21,7 @@ func searchCmd() *cobra.Command {
|
|||||||
content bool
|
content bool
|
||||||
folder string
|
folder string
|
||||||
limit int
|
limit int
|
||||||
|
backend string
|
||||||
asJSON bool
|
asJSON bool
|
||||||
)
|
)
|
||||||
cmd := &cobra.Command{
|
cmd := &cobra.Command{
|
||||||
@@ -27,6 +31,9 @@ func searchCmd() *cobra.Command {
|
|||||||
"By default only file names are matched. With --content the query also\n" +
|
"By default only file names are matched. With --content the query also\n" +
|
||||||
"matches extracted document text (document.attachment.content); this covers\n" +
|
"matches extracted document text (document.attachment.content); this covers\n" +
|
||||||
"Office formats (docx/xlsx/pptx) and is slower.\n\n" +
|
"Office formats (docx/xlsx/pptx) and is slower.\n\n" +
|
||||||
|
"--backend own queries the separate index populated by `oo index`\n" +
|
||||||
|
"(ONLYOFFICE_ES_TEXT_INDEX, default oo_docs_text) instead, which also holds\n" +
|
||||||
|
"PDFs and scans (see docs/elasticsearch.md).\n\n" +
|
||||||
"Requires ONLYOFFICE_ES_URL (and optionally ONLYOFFICE_ES_INDEX,\n" +
|
"Requires ONLYOFFICE_ES_URL (and optionally ONLYOFFICE_ES_INDEX,\n" +
|
||||||
"ONLYOFFICE_TENANT). See docs/elasticsearch.md for the tunnel setup.",
|
"ONLYOFFICE_TENANT). See docs/elasticsearch.md for the tunnel setup.",
|
||||||
Args: cobra.ExactArgs(1),
|
Args: cobra.ExactArgs(1),
|
||||||
@@ -36,7 +43,18 @@ func searchCmd() *cobra.Command {
|
|||||||
}
|
}
|
||||||
bootstrap.LoadEnv()
|
bootstrap.LoadEnv()
|
||||||
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
|
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
|
||||||
searcher, err := c.Files().Search()
|
var (
|
||||||
|
searcher onlyoffice.Searcher
|
||||||
|
err error
|
||||||
|
)
|
||||||
|
switch strings.ToLower(strings.TrimSpace(backend)) {
|
||||||
|
case "", "oo", "elasticsearch":
|
||||||
|
searcher, err = c.Files().Search()
|
||||||
|
case "own", "es-text":
|
||||||
|
searcher, err = onlyoffice.NewESTextIndex(onlyoffice.ESTextConfigFromEnv())
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("unknown search backend %q (want oo|own)", backend)
|
||||||
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
@@ -70,6 +88,7 @@ func searchCmd() *cobra.Command {
|
|||||||
cmd.Flags().BoolVar(&content, "content", false, "also match extracted document content")
|
cmd.Flags().BoolVar(&content, "content", false, "also match extracted document content")
|
||||||
cmd.Flags().StringVar(&folder, "folder", "", "limit to a Documents folder id")
|
cmd.Flags().StringVar(&folder, "folder", "", "limit to a Documents folder id")
|
||||||
cmd.Flags().IntVar(&limit, "limit", 20, "maximum number of results")
|
cmd.Flags().IntVar(&limit, "limit", 20, "maximum number of results")
|
||||||
|
cmd.Flags().StringVar(&backend, "backend", "oo", "index to query: oo (OnlyOffice) | own (oo index)")
|
||||||
cmd.Flags().BoolVar(&asJSON, "json", false, "shorthand for --output json")
|
cmd.Flags().BoolVar(&asJSON, "json", false, "shorthand for --output json")
|
||||||
return cmd
|
return cmd
|
||||||
}
|
}
|
||||||
|
|||||||
+132
-2
@@ -121,5 +121,135 @@ ONLYOFFICE_ES_URL=http://127.0.0.1:9200 ONLYOFFICE_TENANT=1 \
|
|||||||
- `locale`/версия ES: 7.16.3, `_search` совместим с REST 7.x.
|
- `locale`/версия ES: 7.16.3, `_search` совместим с REST 7.x.
|
||||||
- ES без auth и слушает только localhost — туннель обязателен.
|
- ES без auth и слушает только localhost — туннель обязателен.
|
||||||
- Фильтр `tenantId` сузит выдачу; без него видны документы всех тенантов.
|
- Фильтр `tenantId` сузит выдачу; без него видны документы всех тенантов.
|
||||||
- Поиск по содержимому PDF, залитых через API, не работает (нет
|
- Поиск по содержимому PDF в индексе OnlyOffice не работает (для PDF нет
|
||||||
`attachment.content`) — только Office-форматы.
|
`attachment.content`) — только Office-форматы. Решение для PDF — свой индекс
|
||||||
|
`oo_docs_text` (F6 #42), см. ниже.
|
||||||
|
|
||||||
|
# PDF и сканы — свой индекс (F6 #42)
|
||||||
|
|
||||||
|
## Проблема
|
||||||
|
|
||||||
|
`oo search --content "S1019"` не находил номер внутри PDF-счёта: в индексе
|
||||||
|
OnlyOffice PDF лежит только по имени.
|
||||||
|
|
||||||
|
## Почему PDF исключён (исходники CommunityServer)
|
||||||
|
|
||||||
|
Разобрано в `ONLYOFFICE/CommunityServer`:
|
||||||
|
|
||||||
|
- `web/core/ASC.Web.Core/Files/FileUtility.cs` — `CanIndex(fileName)` читает
|
||||||
|
серверную настройку `files.index.formats` (в `web/studio/ASC.Web.Studio/web.appsettings.config`
|
||||||
|
значение по умолчанию `".pptx|.xlsx|.docx"`).
|
||||||
|
- `web/studio/ASC.Web.Studio/Products/Files/Core/Search/FilesWrapper.cs` —
|
||||||
|
`GetDocumentStream*` возвращает `null`, если `!FileUtility.CanIndex(Title)`,
|
||||||
|
файл зашифрован или больше `MaxFileSize`.
|
||||||
|
- `module/ASC.ElasticSearch/Core/WrapperWithDoc.cs` + mapping в `Wrapper.cs` —
|
||||||
|
маппинг `document.attachment.content` и ingest-pipeline `attachments`
|
||||||
|
формат-агностичны: они распарсят любой поток.
|
||||||
|
|
||||||
|
Вывод: PDF исключён **только настройкой** `files.index.formats`; жёсткого
|
||||||
|
ограничения на формат в коде нет.
|
||||||
|
|
||||||
|
## Варианты и решение
|
||||||
|
|
||||||
|
| # | Вариант | Оценка |
|
||||||
|
|---|---------|--------|
|
||||||
|
| a | Включить `.pdf` в `files.index.formats` + reindex | Правка сервера OO; настройка может потеряться при обновлении; полный reindex 39k док-в; Tika **не OCR** — сканы без текстового слоя дадут пустой контент. Отклонён без решения PO. |
|
||||||
|
| b | Server-side ingest/attachment для PDF | По факту то же, что (a): сервер кормит поток только для `CanIndex`. |
|
||||||
|
| c | **Свой индекс** `oo_docs_text`, наполняемый `internal/docpipe` | **Выбран.** Сервер OO не трогаем; детерминированно; работает OCR для сканов; независимо от обновлений OO; любые форматы; фильтры папка/тип. |
|
||||||
|
| d | Локальный поиск без индекса | Отклонён как основной: качаем и извлекаем на каждый запрос, нет выдачи/ранжирования/highlight. |
|
||||||
|
|
||||||
|
Итог: **вариант c**. Индекс OnlyOffice (`files_file`) не изменяется; наш
|
||||||
|
индекс живёт рядом.
|
||||||
|
|
||||||
|
## Устройство
|
||||||
|
|
||||||
|
- `file_es_text.go` — `ESTextIndex` (`Name() = "es-text"`):
|
||||||
|
`Ensure` (создаёт индекс с явным маппингом), `Put` (bulk, `refresh`),
|
||||||
|
`Delete` (по `id`), `Search` (`multi_match` по `title^2` + `content`,
|
||||||
|
фильтры `folder`/`ext`, highlight).
|
||||||
|
- `file_text_index.go` — `TextIndexer`: листает папки (`FileStore.List`),
|
||||||
|
качает файлы (`FileStore.Download`), извлекает текст через
|
||||||
|
`internal/docpipe` (`pdftotext`, для сканов — `ocrmypdf`/`tesseract`),
|
||||||
|
пишет в `TextIndex`. Пул воркеров (по умолчанию 3).
|
||||||
|
- CLI: `oo index folder|files` наполняет индекс; `oo search --backend own`
|
||||||
|
ищет по нему.
|
||||||
|
|
||||||
|
### Встроенные вложения PDF
|
||||||
|
|
||||||
|
Оцифрованные PDF несут вложения (`<doc>.md` — текст/таблицы скана,
|
||||||
|
`<doc>.yaml`/`.json` — метаданные, `.xml` — EN 16931 CII eRechnung,
|
||||||
|
`factur-x.xml` у ZUGFeRD; см. `office-assistant/docs/reference/document-metadata.md`).
|
||||||
|
`TextIndexer` обходит их: `pdfdetach -list` перечисляет, `-save` сохраняет,
|
||||||
|
каждое вложение проходит штатный `docpipe.ToMarkdown` (PDF/картинки → OCR,
|
||||||
|
`.md`/`.txt` — как есть). Форматы, которые docpipe не конвертирует
|
||||||
|
(`.xml`/`.html` — снимаются теги; `.json`/`.csv` — как текст), извлекаются
|
||||||
|
текстом; нечитаемые — пропускаются.
|
||||||
|
|
||||||
|
Текст склеивается: тело, затем по секции на вложение с маркером
|
||||||
|
`[attachment: <имя>]` (функция `docpipe.JoinWithAttachments`). Индекс — тот же
|
||||||
|
`file_id`, upsert идемпотентен. Нет вложений или pdfdetach/формат нечитаем —
|
||||||
|
индексируется тело (без падения).
|
||||||
|
|
||||||
|
Поля `oo_docs_text`:
|
||||||
|
|
||||||
|
| поле | тип | смысл |
|
||||||
|
|------|-----|-------|
|
||||||
|
| `id` | keyword | id файла Documents |
|
||||||
|
| `title` | text (+`.keyword`) | имя файла |
|
||||||
|
| `folder` | keyword | id папки |
|
||||||
|
| `ext` | keyword | расширение |
|
||||||
|
| `content` | text | извлечённый текст (pdftotext/OCR) |
|
||||||
|
|
||||||
|
## CLI
|
||||||
|
|
||||||
|
```bash
|
||||||
|
set -a; . .env; set +a # ONLYOFFICE_URL/USER/PASS + ONLYOFFICE_ES_URL
|
||||||
|
oo index folder 634 # PDF в папке 634
|
||||||
|
oo index folder 634 --recursive --exts pdf,png --limit 100
|
||||||
|
oo index files 3576 3578 # точечно
|
||||||
|
oo index folder 634 --dry-run # показать план, ничего не менять
|
||||||
|
|
||||||
|
oo search "S1021" --content --backend own
|
||||||
|
oo search "S1021" --backend own --folder 634 --json
|
||||||
|
```
|
||||||
|
|
||||||
|
`--backend` у `oo search`: `oo` (по умолчанию, индекс OnlyOffice) или `own`
|
||||||
|
(наш `ONLYOFFICE_ES_TEXT_INDEX`).
|
||||||
|
|
||||||
|
## Переменные (дополнение)
|
||||||
|
|
||||||
|
| env | default | смысл |
|
||||||
|
|-----|---------|-------|
|
||||||
|
| `ONLYOFFICE_ES_TEXT_INDEX` | `oo_docs_text` | индекс своего конвейера |
|
||||||
|
|
||||||
|
`ONLYOFFICE_ES_URL` — общий для обоих индексов.
|
||||||
|
|
||||||
|
## Тесты
|
||||||
|
|
||||||
|
```bash
|
||||||
|
go test -run 'ESText|TextIndexer|Index' ./ ./cmd/oo/ # unit, без сети
|
||||||
|
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 \
|
||||||
|
go test -tags=integration -run TestIntegrationESTextIndex -v .
|
||||||
|
```
|
||||||
|
|
||||||
|
Интеграционный тест создаёт временный индекс, наполняет, ищет по контенту,
|
||||||
|
проверяет фильтры и удаление, затем удаляет индекс;
|
||||||
|
`TestIntegrationESTextIndexPDFAttachment` индексирует
|
||||||
|
`testdata/pdf-with-attachment.pdf` реальным конвейером (pdfdetach + pdftotext)
|
||||||
|
и ищет токен, лежащий только во вложении. Unit-тесты используют
|
||||||
|
fake-store/fake-extractor и не требуют pdftotext/OCR (парсер списка, склейка
|
||||||
|
`JoinWithAttachments`, снятие тегов `xmlToText` — чистые).
|
||||||
|
|
||||||
|
## Грабли
|
||||||
|
|
||||||
|
- Наполнение — ручное (`oo index`); после изменения/добавления PDF повтори.
|
||||||
|
Повтор идемпотентен (upsert по id файла).
|
||||||
|
- В индексе ищется только то, что проиндексировано; `oo index` качает каждый
|
||||||
|
файл и (для сканов) гоняет OCR — это медленно, отсюда `--limit`/`--exts`.
|
||||||
|
- `folder` фильтруется как id папки, а не как путь.
|
||||||
|
- Дубликаты (напр. `S1055.pdf` и `2026-08-20-S1055-…`) дадут несколько строк —
|
||||||
|
это ожидаемо, дедуп — на стороне потребителя.
|
||||||
|
- Вложения: нужен `pdfdetach` (poppler); если его нет — индексируется только
|
||||||
|
тело. Вложенный PDF/картинка с плохим текстовым слоем проходит OCR, это
|
||||||
|
медленно. `.json`-метаданные (CuraSoft) индексируются как текст и могут
|
||||||
|
добавить шумовых токенов.
|
||||||
|
|||||||
+4
-2
@@ -188,6 +188,7 @@ type esBool struct {
|
|||||||
type esClause struct {
|
type esClause struct {
|
||||||
MultiMatch *esMultiMatch `json:"multi_match,omitempty"`
|
MultiMatch *esMultiMatch `json:"multi_match,omitempty"`
|
||||||
Term map[string]any `json:"term,omitempty"`
|
Term map[string]any `json:"term,omitempty"`
|
||||||
|
Terms map[string]any `json:"terms,omitempty"`
|
||||||
Wildcard map[string]any `json:"wildcard,omitempty"`
|
Wildcard map[string]any `json:"wildcard,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -268,9 +269,10 @@ func parseESSearchResponse(raw []byte) ([]SearchHit, error) {
|
|||||||
var esHighlightTag = regexp.MustCompile(`</?em[^>]*>`)
|
var esHighlightTag = regexp.MustCompile(`</?em[^>]*>`)
|
||||||
|
|
||||||
// esHighlightText flattens a highlight map into one plain-text snippet,
|
// esHighlightText flattens a highlight map into one plain-text snippet,
|
||||||
// preferring the content fragment over the title.
|
// preferring the content fragment over the title. It covers both the
|
||||||
|
// OnlyOffice content field and the own-index "content" field.
|
||||||
func esHighlightText(hl map[string][]string) string {
|
func esHighlightText(hl map[string][]string) string {
|
||||||
for _, key := range []string{"document.attachment.content", "title"} {
|
for _, key := range []string{"document.attachment.content", "content", "title"} {
|
||||||
frags := hl[key]
|
frags := hl[key]
|
||||||
if len(frags) == 0 {
|
if len(frags) == 0 {
|
||||||
continue
|
continue
|
||||||
|
|||||||
+336
@@ -0,0 +1,336 @@
|
|||||||
|
package onlyoffice
|
||||||
|
|
||||||
|
// Own full-text index (epic #34, F6 #42).
|
||||||
|
//
|
||||||
|
// The OnlyOffice Elasticsearch index (files_file) only holds extracted content
|
||||||
|
// for Office formats. FileUtility.CanIndex gates extraction by the server
|
||||||
|
// setting files.index.formats, whose default is ".pptx|.xlsx|.docx", so PDFs
|
||||||
|
// are indexed by name only. Instead of patching the server (risky: lost on
|
||||||
|
// upgrade, forces a full reindex) this file implements a second, independent
|
||||||
|
// index (default oo_docs_text) that our own pipeline fills from
|
||||||
|
// internal/docpipe (pdftotext + OCR). The OnlyOffice index is never touched.
|
||||||
|
//
|
||||||
|
// See docs/elasticsearch.md for the decision and the trade-offs.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"net/http"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
const defaultESTextIndex = "oo_docs_text"
|
||||||
|
|
||||||
|
// ESTextConfig configures the own full-text index.
|
||||||
|
type ESTextConfig struct {
|
||||||
|
URL string // scheme://host:port of the ES HTTP endpoint
|
||||||
|
Index string // index name, default oo_docs_text
|
||||||
|
Tenant string // reserved for future multi-tenant data; unused for now
|
||||||
|
}
|
||||||
|
|
||||||
|
// ESTextConfigFromEnv reads ONLYOFFICE_ES_URL and ONLYOFFICE_ES_TEXT_INDEX
|
||||||
|
// (default oo_docs_text). The library never loads dotfiles — the CLI does that.
|
||||||
|
func ESTextConfigFromEnv() ESTextConfig {
|
||||||
|
return ESTextConfig{
|
||||||
|
URL: strings.TrimRight(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")), "/"),
|
||||||
|
Index: firstNonEmpty(os.Getenv("ONLYOFFICE_ES_TEXT_INDEX"), defaultESTextIndex),
|
||||||
|
Tenant: strings.TrimSpace(os.Getenv("ONLYOFFICE_TENANT")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TextDoc is one document in the own full-text index. It is keyed by the
|
||||||
|
// OnlyOffice file id so hits map straight back to Documents entries.
|
||||||
|
type TextDoc struct {
|
||||||
|
ID string `json:"id"`
|
||||||
|
Title string `json:"title"`
|
||||||
|
FolderID string `json:"folder,omitempty"`
|
||||||
|
Ext string `json:"ext,omitempty"`
|
||||||
|
Content string `json:"content"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// TextIndex is the storage/search surface for locally extracted document text.
|
||||||
|
// It complements Searcher: ESSearcher reads OnlyOffice's index, ESTextIndex
|
||||||
|
// reads ours.
|
||||||
|
type TextIndex interface {
|
||||||
|
Put(ctx context.Context, docs []TextDoc) error
|
||||||
|
Delete(ctx context.Context, ids []string) error
|
||||||
|
Search(ctx context.Context, q SearchQuery) ([]SearchHit, error)
|
||||||
|
Name() string
|
||||||
|
}
|
||||||
|
|
||||||
|
// ESTextIndex is a TextIndex (and Searcher) over a dedicated Elasticsearch
|
||||||
|
// index filled by TextIndexer.
|
||||||
|
type ESTextIndex struct {
|
||||||
|
cfg ESTextConfig
|
||||||
|
http *http.Client
|
||||||
|
}
|
||||||
|
|
||||||
|
var (
|
||||||
|
_ TextIndex = (*ESTextIndex)(nil)
|
||||||
|
_ Searcher = (*ESTextIndex)(nil)
|
||||||
|
)
|
||||||
|
|
||||||
|
// NewESTextIndex returns a searcher/writer for the own full-text index. The URL
|
||||||
|
// is required; an empty index falls back to oo_docs_text.
|
||||||
|
func NewESTextIndex(cfg ESTextConfig) (*ESTextIndex, error) {
|
||||||
|
if strings.TrimSpace(cfg.URL) == "" {
|
||||||
|
return nil, fmt.Errorf("onlyoffice: elasticsearch URL is empty (set ONLYOFFICE_ES_URL)")
|
||||||
|
}
|
||||||
|
cfg.URL = strings.TrimRight(cfg.URL, "/")
|
||||||
|
if cfg.Index == "" {
|
||||||
|
cfg.Index = defaultESTextIndex
|
||||||
|
}
|
||||||
|
return &ESTextIndex{cfg: cfg, http: &http.Client{Timeout: 120 * time.Second}}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Name implements Searcher and TextIndex.
|
||||||
|
func (x *ESTextIndex) Name() string { return "es-text" }
|
||||||
|
|
||||||
|
// Index returns the configured index name.
|
||||||
|
func (x *ESTextIndex) Index() string { return x.cfg.Index }
|
||||||
|
|
||||||
|
// esTextMapping pins explicit types: content must stay a plain text field (no
|
||||||
|
// keyword subfield) and folder/ext stay exact keywords for filters.
|
||||||
|
const esTextMapping = `{
|
||||||
|
"mappings": {
|
||||||
|
"properties": {
|
||||||
|
"id": {"type": "keyword"},
|
||||||
|
"title": {"type": "text", "fields": {"keyword": {"type": "keyword", "ignore_above": 512}}},
|
||||||
|
"folder": {"type": "keyword"},
|
||||||
|
"ext": {"type": "keyword"},
|
||||||
|
"content": {"type": "text"}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}`
|
||||||
|
|
||||||
|
// Ensure creates the index with the explicit mapping. A missing index is
|
||||||
|
// created; an already existing one is left untouched.
|
||||||
|
func (x *ESTextIndex) Ensure(ctx context.Context) error {
|
||||||
|
status, raw, err := x.do(ctx, http.MethodPut, "/"+x.cfg.Index, []byte(esTextMapping), "application/json")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if status == http.StatusOK {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if status == http.StatusBadRequest && bytes.Contains(raw, []byte("resource_already_exists_exception")) {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return fmt.Errorf("onlyoffice: create text index %s: %d %s", x.cfg.Index, status, truncate(string(raw), 300))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Put upserts documents via the bulk API and refreshes so they are immediately
|
||||||
|
// searchable.
|
||||||
|
func (x *ESTextIndex) Put(ctx context.Context, docs []TextDoc) error {
|
||||||
|
if len(docs) == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_bulk?refresh=true", esTextBulkBody(docs), "application/x-ndjson")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if status >= 400 {
|
||||||
|
return fmt.Errorf("onlyoffice: bulk index %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400))
|
||||||
|
}
|
||||||
|
var res esBulkResponse
|
||||||
|
if err := json.Unmarshal(raw, &res); err != nil {
|
||||||
|
return fmt.Errorf("onlyoffice: decode bulk response: %w", err)
|
||||||
|
}
|
||||||
|
if !res.Errors {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return fmt.Errorf("onlyoffice: bulk index %s: %s", x.cfg.Index, res.firstError())
|
||||||
|
}
|
||||||
|
|
||||||
|
// Delete removes documents by OnlyOffice file id. A missing index means there
|
||||||
|
// is nothing to delete.
|
||||||
|
func (x *ESTextIndex) Delete(ctx context.Context, ids []string) error {
|
||||||
|
if len(ids) == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
body, err := json.Marshal(map[string]any{"query": map[string]any{"terms": map[string]any{"id": ids}}})
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_delete_by_query?refresh=true", body, "application/json")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if status == http.StatusNotFound {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if status >= 400 {
|
||||||
|
return fmt.Errorf("onlyoffice: delete from %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400))
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Search runs a multi_match over title (boosted) and content, with optional
|
||||||
|
// folder and extension filters. A missing index yields no hits, not an error.
|
||||||
|
func (x *ESTextIndex) Search(ctx context.Context, q SearchQuery) ([]SearchHit, error) {
|
||||||
|
q.Text = strings.TrimSpace(q.Text)
|
||||||
|
if q.Text == "" {
|
||||||
|
return nil, fmt.Errorf("onlyoffice: empty search query")
|
||||||
|
}
|
||||||
|
body, err := json.Marshal(esTextSearchRequest(q))
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("onlyoffice: build elasticsearch query: %w", err)
|
||||||
|
}
|
||||||
|
status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_search", body, "application/json")
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if status == http.StatusNotFound {
|
||||||
|
return nil, nil
|
||||||
|
}
|
||||||
|
if status >= 400 {
|
||||||
|
return nil, fmt.Errorf("onlyoffice: search %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400))
|
||||||
|
}
|
||||||
|
return parseESTextResponse(raw)
|
||||||
|
}
|
||||||
|
|
||||||
|
// do sends one request and returns the status and body (bounded). The caller
|
||||||
|
// decides which statuses are errors.
|
||||||
|
func (x *ESTextIndex) do(ctx context.Context, method, path string, body []byte, contentType string) (int, []byte, error) {
|
||||||
|
var r io.Reader
|
||||||
|
if body != nil {
|
||||||
|
r = bytes.NewReader(body)
|
||||||
|
}
|
||||||
|
req, err := http.NewRequestWithContext(ctx, method, x.cfg.URL+path, r)
|
||||||
|
if err != nil {
|
||||||
|
return 0, nil, err
|
||||||
|
}
|
||||||
|
req.Header.Set("Accept", "application/json")
|
||||||
|
if contentType != "" {
|
||||||
|
req.Header.Set("Content-Type", contentType)
|
||||||
|
}
|
||||||
|
resp, err := x.http.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
return 0, nil, fmt.Errorf("onlyoffice: elasticsearch %s: %w", method, err)
|
||||||
|
}
|
||||||
|
defer resp.Body.Close()
|
||||||
|
raw, err := io.ReadAll(io.LimitReader(resp.Body, maxESResponseSize))
|
||||||
|
if err != nil {
|
||||||
|
return resp.StatusCode, nil, err
|
||||||
|
}
|
||||||
|
return resp.StatusCode, raw, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// esTextBulkBody renders the NDJSON bulk payload. Pure, so it is unit-tested.
|
||||||
|
func esTextBulkBody(docs []TextDoc) []byte {
|
||||||
|
var b bytes.Buffer
|
||||||
|
enc := json.NewEncoder(&b)
|
||||||
|
enc.SetEscapeHTML(false)
|
||||||
|
for _, d := range docs {
|
||||||
|
_ = enc.Encode(map[string]any{"index": map[string]any{"_id": d.ID}})
|
||||||
|
_ = enc.Encode(d)
|
||||||
|
}
|
||||||
|
return b.Bytes()
|
||||||
|
}
|
||||||
|
|
||||||
|
// esTextSearchRequest builds the own-index query. Pure, so it is unit-tested.
|
||||||
|
func esTextSearchRequest(q SearchQuery) esRequest {
|
||||||
|
limit := q.Limit
|
||||||
|
if limit <= 0 {
|
||||||
|
limit = defaultESLimit
|
||||||
|
}
|
||||||
|
if limit > maxESLimit {
|
||||||
|
limit = maxESLimit
|
||||||
|
}
|
||||||
|
fields := []string{"title^2", "content"}
|
||||||
|
must := []esClause{{MultiMatch: &esMultiMatch{Query: q.Text, Fields: fields}}}
|
||||||
|
|
||||||
|
var filter []esClause
|
||||||
|
if f := strings.TrimSpace(q.FolderID); f != "" {
|
||||||
|
filter = append(filter, esClause{Term: map[string]any{"folder": f}})
|
||||||
|
}
|
||||||
|
if exts := normalizeExtensions(q.Extensions); len(exts) > 0 {
|
||||||
|
filter = append(filter, esClause{Terms: map[string]any{"ext": exts}})
|
||||||
|
}
|
||||||
|
|
||||||
|
return esRequest{
|
||||||
|
Size: limit,
|
||||||
|
Source: []string{"id", "title", "folder", "ext"},
|
||||||
|
Query: esQuery{Bool: esBool{Must: must, Filter: filter}},
|
||||||
|
Highlight: esHighlight{PreTags: []string{"<em>"}, PostTags: []string{"</em>"}, Fields: map[string]struct{}{"title": {}, "content": {}}},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// esBulkResponse is the subset of an ES bulk response we consume.
|
||||||
|
type esBulkResponse struct {
|
||||||
|
Errors bool `json:"errors"`
|
||||||
|
Items []map[string]struct {
|
||||||
|
ID string `json:"_id"`
|
||||||
|
Status int `json:"status"`
|
||||||
|
Error *struct {
|
||||||
|
Type string `json:"type"`
|
||||||
|
Reason string `json:"reason"`
|
||||||
|
} `json:"error"`
|
||||||
|
} `json:"items"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// firstError returns a compact description of the first failed bulk item.
|
||||||
|
func (r esBulkResponse) firstError() string {
|
||||||
|
for _, item := range r.Items {
|
||||||
|
for op, res := range item {
|
||||||
|
if res.Error != nil {
|
||||||
|
return fmt.Sprintf("%s %s: %s %s", op, res.ID, res.Error.Type, res.Error.Reason)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "unknown bulk error"
|
||||||
|
}
|
||||||
|
|
||||||
|
// esTextResponse is the subset of an own-index search response we consume.
|
||||||
|
type esTextResponse struct {
|
||||||
|
Hits struct {
|
||||||
|
Total struct {
|
||||||
|
Value int `json:"value"`
|
||||||
|
} `json:"total"`
|
||||||
|
Hits []struct {
|
||||||
|
ID string `json:"_id"`
|
||||||
|
Score float64 `json:"_score"`
|
||||||
|
Source TextDoc `json:"_source"`
|
||||||
|
HL map[string][]string `json:"highlight"`
|
||||||
|
} `json:"hits"`
|
||||||
|
} `json:"hits"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseESTextResponse converts an own-index search response into SearchHit
|
||||||
|
// values. Pure, so it is unit-tested.
|
||||||
|
func parseESTextResponse(raw []byte) ([]SearchHit, error) {
|
||||||
|
var r esTextResponse
|
||||||
|
if err := json.Unmarshal(raw, &r); err != nil {
|
||||||
|
return nil, fmt.Errorf("onlyoffice: decode elasticsearch response: %w", err)
|
||||||
|
}
|
||||||
|
hits := make([]SearchHit, 0, len(r.Hits.Hits))
|
||||||
|
for _, h := range r.Hits.Hits {
|
||||||
|
id := h.Source.ID
|
||||||
|
if id == "" {
|
||||||
|
id = h.ID
|
||||||
|
}
|
||||||
|
parent := h.Source.FolderID
|
||||||
|
var path []string
|
||||||
|
if parent != "" {
|
||||||
|
path = []string{parent}
|
||||||
|
}
|
||||||
|
hits = append(hits, SearchHit{
|
||||||
|
Entry: Entry{
|
||||||
|
ID: id,
|
||||||
|
ParentID: parent,
|
||||||
|
Title: h.Source.Title,
|
||||||
|
Kind: File,
|
||||||
|
Provider: "es-text",
|
||||||
|
},
|
||||||
|
Score: h.Score,
|
||||||
|
Highlight: esHighlightText(h.HL),
|
||||||
|
Path: path,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
return hits, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,158 @@
|
|||||||
|
//go:build integration
|
||||||
|
|
||||||
|
package onlyoffice
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"net/http"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/eslider/go-onlyoffice/internal/docpipe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestIntegrationESTextIndex verifies the own full-text index end to end
|
||||||
|
// against a live Elasticsearch: create the index with its mapping, index a
|
||||||
|
// document, find it by content (and reject it via a folder filter and after
|
||||||
|
// deletion), then drop the throwaway index.
|
||||||
|
//
|
||||||
|
// Requires ONLYOFFICE_ES_URL (a reachable ES endpoint — in the current setup a
|
||||||
|
// tunnel to the ES inside the OnlyOffice VM, see docs/elasticsearch.md). It
|
||||||
|
// does not need OnlyOffice credentials because no file is downloaded: the
|
||||||
|
// TextIndexer write path is covered by unit tests with a fake extractor.
|
||||||
|
func TestIntegrationESTextIndex(t *testing.T) {
|
||||||
|
esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL"))
|
||||||
|
if esURL == "" {
|
||||||
|
t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test")
|
||||||
|
}
|
||||||
|
stamp := time.Now().UTC().Format("20060102150405")
|
||||||
|
idx, err := NewESTextIndex(ESTextConfig{URL: esURL, Index: "oo_docs_text_it_" + stamp})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("NewESTextIndex: %v", err)
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
|
||||||
|
t.Cleanup(func() {
|
||||||
|
cleanupCtx, done := context.WithTimeout(context.Background(), 30*time.Second)
|
||||||
|
defer done()
|
||||||
|
_, _, _ = idx.do(cleanupCtx, http.MethodDelete, "/"+idx.Index(), nil, "")
|
||||||
|
})
|
||||||
|
|
||||||
|
if err := idx.Ensure(ctx); err != nil {
|
||||||
|
t.Fatalf("Ensure: %v", err)
|
||||||
|
}
|
||||||
|
// Ensure is idempotent.
|
||||||
|
if err := idx.Ensure(ctx); err != nil {
|
||||||
|
t.Fatalf("Ensure (second): %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
token := "gotes" + stamp
|
||||||
|
doc := TextDoc{
|
||||||
|
ID: "3578",
|
||||||
|
Title: "2026-07-28-S1021-Edelweiss-rechnung.pdf",
|
||||||
|
FolderID: "634",
|
||||||
|
Ext: "pdf",
|
||||||
|
Content: "Begleitzettel SGB XI — Rechnung " + token,
|
||||||
|
}
|
||||||
|
if err := idx.Put(ctx, []TextDoc{doc}); err != nil {
|
||||||
|
t.Fatalf("Put: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
hits, err := idx.Search(ctx, SearchQuery{Text: token})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Search: %v", err)
|
||||||
|
}
|
||||||
|
if len(hits) != 1 || hits[0].ID != "3578" {
|
||||||
|
t.Fatalf("content search hits = %+v, want doc 3578", hits)
|
||||||
|
}
|
||||||
|
if !strings.Contains(hits[0].Highlight, token) {
|
||||||
|
t.Errorf("highlight = %q, want token", hits[0].Highlight)
|
||||||
|
}
|
||||||
|
|
||||||
|
if hits, err := idx.Search(ctx, SearchQuery{Text: token, FolderID: "999"}); err != nil {
|
||||||
|
t.Fatalf("Search with folder filter: %v", err)
|
||||||
|
} else if len(hits) != 0 {
|
||||||
|
t.Errorf("folder filter returned %d hits, want 0", len(hits))
|
||||||
|
}
|
||||||
|
if hits, err := idx.Search(ctx, SearchQuery{Text: token, Extensions: []string{"docx"}}); err != nil {
|
||||||
|
t.Fatalf("Search with ext filter: %v", err)
|
||||||
|
} else if len(hits) != 0 {
|
||||||
|
t.Errorf("ext filter returned %d hits, want 0", len(hits))
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := idx.Delete(ctx, []string{"3578"}); err != nil {
|
||||||
|
t.Fatalf("Delete: %v", err)
|
||||||
|
}
|
||||||
|
if hits, err := idx.Search(ctx, SearchQuery{Text: token}); err != nil {
|
||||||
|
t.Fatalf("Search after delete: %v", err)
|
||||||
|
} else if len(hits) != 0 {
|
||||||
|
t.Errorf("after delete search returned %d hits, want 0", len(hits))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestIntegrationESTextIndexPDFAttachment indexes testdata/pdf-with-attachment.pdf
|
||||||
|
// through the real pipeline (TextIndexer + docpipe: pdfdetach + pdftotext) and
|
||||||
|
// verifies that text living only in the embedded attachment is searchable.
|
||||||
|
//
|
||||||
|
// Requires ONLYOFFICE_ES_URL plus poppler (pdfdetach/pdftotext). No OnlyOffice
|
||||||
|
// credentials are needed: a fixture FileStore serves the PDF bytes.
|
||||||
|
func TestIntegrationESTextIndexPDFAttachment(t *testing.T) {
|
||||||
|
esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL"))
|
||||||
|
if esURL == "" {
|
||||||
|
t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test")
|
||||||
|
}
|
||||||
|
if docpipe.LookPath().PDFDetach == "" {
|
||||||
|
t.Skip("pdfdetach not on PATH — skipping PDF attachment integration test")
|
||||||
|
}
|
||||||
|
pdf, err := os.ReadFile("testdata/pdf-with-attachment.pdf")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read fixture: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
stamp := time.Now().UTC().Format("20060102150405")
|
||||||
|
idx, err := NewESTextIndex(ESTextConfig{URL: esURL, Index: "oo_docs_text_it_att_" + stamp})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("NewESTextIndex: %v", err)
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
t.Cleanup(func() {
|
||||||
|
cleanupCtx, done := context.WithTimeout(context.Background(), 30*time.Second)
|
||||||
|
defer done()
|
||||||
|
_, _, _ = idx.do(cleanupCtx, http.MethodDelete, "/"+idx.Index(), nil, "")
|
||||||
|
})
|
||||||
|
if err := idx.Ensure(ctx); err != nil {
|
||||||
|
t.Fatalf("Ensure: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
store := &textFakeStore{files: map[string][]byte{"9001": pdf}}
|
||||||
|
ix := NewTextIndexer(store, idx)
|
||||||
|
res, err := ix.IndexEntries(ctx, []Entry{{ID: "9001", Title: "scan.pdf", ParentID: "777", Kind: File}}, IndexOptions{MinChars: 1})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("IndexEntries: %v", err)
|
||||||
|
}
|
||||||
|
if res.Indexed != 1 || res.Failed != 0 {
|
||||||
|
t.Fatalf("result = %+v, want one indexed doc", res)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Token appears only inside the embedded goo-note.txt attachment.
|
||||||
|
hits, err := idx.Search(ctx, SearchQuery{Text: "gooattachmenttoken"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Search attachment token: %v", err)
|
||||||
|
}
|
||||||
|
if len(hits) != 1 || hits[0].ID != "9001" {
|
||||||
|
t.Fatalf("attachment-token hits = %+v, want doc 9001", hits)
|
||||||
|
}
|
||||||
|
if !strings.Contains(hits[0].Highlight, "gooattachmenttoken") {
|
||||||
|
t.Errorf("highlight = %q, want attachment token", hits[0].Highlight)
|
||||||
|
}
|
||||||
|
// Body text is indexed as before.
|
||||||
|
if hits, err := idx.Search(ctx, SearchQuery{Text: "goobodytoken"}); err != nil {
|
||||||
|
t.Fatalf("Search body token: %v", err)
|
||||||
|
} else if len(hits) != 1 {
|
||||||
|
t.Errorf("body-token hits = %d, want 1", len(hits))
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,275 @@
|
|||||||
|
package onlyoffice
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"os"
|
||||||
|
"reflect"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestESTextSearchRequestShape(t *testing.T) {
|
||||||
|
got := esTextSearchRequest(SearchQuery{
|
||||||
|
Text: "S1021",
|
||||||
|
FolderID: "634",
|
||||||
|
Extensions: []string{".PDF", "pdf"},
|
||||||
|
Limit: 5,
|
||||||
|
})
|
||||||
|
if got.Size != 5 {
|
||||||
|
t.Errorf("size = %d, want 5", got.Size)
|
||||||
|
}
|
||||||
|
mm := got.Query.Bool.Must[0].MultiMatch
|
||||||
|
if mm == nil || !reflect.DeepEqual(mm.Fields, []string{"title^2", "content"}) {
|
||||||
|
t.Fatalf("multi_match = %+v, want title^2 + content", mm)
|
||||||
|
}
|
||||||
|
if _, ok := got.Highlight.Fields["content"]; !ok {
|
||||||
|
t.Error("content highlight missing")
|
||||||
|
}
|
||||||
|
if _, ok := got.Highlight.Fields["title"]; !ok {
|
||||||
|
t.Error("title highlight missing")
|
||||||
|
}
|
||||||
|
var folder, exts int
|
||||||
|
for _, f := range got.Query.Bool.Filter {
|
||||||
|
switch {
|
||||||
|
case f.Term != nil && f.Term["folder"] != nil:
|
||||||
|
folder++
|
||||||
|
if f.Term["folder"] != "634" {
|
||||||
|
t.Errorf("folder term = %+v", f.Term)
|
||||||
|
}
|
||||||
|
case f.Terms != nil:
|
||||||
|
exts++
|
||||||
|
if !reflect.DeepEqual(f.Terms["ext"], []string{"pdf"}) {
|
||||||
|
t.Errorf("ext terms = %+v, want deduped pdf", f.Terms)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if folder != 1 || exts != 1 {
|
||||||
|
t.Errorf("filters folder=%d exts=%d, want 1 each", folder, exts)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestESTextBulkBody(t *testing.T) {
|
||||||
|
body := esTextBulkBody([]TextDoc{
|
||||||
|
{ID: "3578", Title: "S1021.pdf", FolderID: "634", Ext: "pdf", Content: "Begleitzettel <S1021> & mehr"},
|
||||||
|
{ID: "3579", Title: "S1023.pdf", Ext: "pdf", Content: "x"},
|
||||||
|
})
|
||||||
|
lines := strings.Split(strings.TrimRight(string(body), "\n"), "\n")
|
||||||
|
if len(lines) != 4 {
|
||||||
|
t.Fatalf("bulk body has %d lines, want 4:\n%s", len(lines), body)
|
||||||
|
}
|
||||||
|
if !strings.Contains(lines[0], `"index"`) || !strings.Contains(lines[0], `"_id":"3578"`) {
|
||||||
|
t.Errorf("action line = %q", lines[0])
|
||||||
|
}
|
||||||
|
if !strings.Contains(lines[1], `"content":"Begleitzettel <S1021> & mehr"`) {
|
||||||
|
t.Errorf("source line should keep HTML unescaped, got %q", lines[1])
|
||||||
|
}
|
||||||
|
if !strings.Contains(lines[2], `"_id":"3579"`) {
|
||||||
|
t.Errorf("second action line = %q", lines[2])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParseESTextResponse(t *testing.T) {
|
||||||
|
raw := []byte(`{
|
||||||
|
"hits": {
|
||||||
|
"total": {"value": 1, "relation": "eq"},
|
||||||
|
"hits": [
|
||||||
|
{
|
||||||
|
"_id": "3578",
|
||||||
|
"_score": 3.21,
|
||||||
|
"_source": {"id": "3578", "title": "2026-07-28-S1021-Edelweiss-rechnung.pdf", "folder": "634", "ext": "pdf"},
|
||||||
|
"highlight": {"content": ["Begleitzettel … <em>S1021</em> …"]}
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}`)
|
||||||
|
hits, err := parseESTextResponse(raw)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parseESTextResponse: %v", err)
|
||||||
|
}
|
||||||
|
if len(hits) != 1 {
|
||||||
|
t.Fatalf("hits = %d, want 1", len(hits))
|
||||||
|
}
|
||||||
|
h := hits[0]
|
||||||
|
if h.ID != "3578" || h.Title != "2026-07-28-S1021-Edelweiss-rechnung.pdf" || h.Kind != File {
|
||||||
|
t.Errorf("entry = %+v", h.Entry)
|
||||||
|
}
|
||||||
|
if h.ParentID != "634" || !reflect.DeepEqual(h.Path, []string{"634"}) {
|
||||||
|
t.Errorf("path = %v parent = %q", h.Path, h.ParentID)
|
||||||
|
}
|
||||||
|
if h.Provider != "es-text" {
|
||||||
|
t.Errorf("provider = %q", h.Provider)
|
||||||
|
}
|
||||||
|
if h.Highlight != "Begleitzettel … S1021 …" {
|
||||||
|
t.Errorf("highlight = %q, want tags stripped", h.Highlight)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNewESTextIndexDefaults(t *testing.T) {
|
||||||
|
if _, err := NewESTextIndex(ESTextConfig{}); err == nil {
|
||||||
|
t.Error("empty URL: want error")
|
||||||
|
}
|
||||||
|
x, err := NewESTextIndex(ESTextConfig{URL: "http://es:9200/"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("NewESTextIndex: %v", err)
|
||||||
|
}
|
||||||
|
if x.Index() != defaultESTextIndex {
|
||||||
|
t.Errorf("index = %q, want %q", x.Index(), defaultESTextIndex)
|
||||||
|
}
|
||||||
|
if x.cfg.URL != "http://es:9200" {
|
||||||
|
t.Errorf("url = %q, want trimmed", x.cfg.URL)
|
||||||
|
}
|
||||||
|
if x.Name() != "es-text" {
|
||||||
|
t.Errorf("Name() = %q", x.Name())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestESTextConfigFromEnvIndexDefault(t *testing.T) {
|
||||||
|
t.Setenv("ONLYOFFICE_ES_URL", "http://es:9200/")
|
||||||
|
t.Setenv("ONLYOFFICE_ES_TEXT_INDEX", "")
|
||||||
|
cfg := ESTextConfigFromEnv()
|
||||||
|
if cfg.Index != defaultESTextIndex {
|
||||||
|
t.Errorf("index = %q, want %q", cfg.Index, defaultESTextIndex)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestTextIndexerIndexEntries(t *testing.T) {
|
||||||
|
store := &textFakeStore{
|
||||||
|
files: map[string][]byte{"1": []byte("PDFBYTES")},
|
||||||
|
}
|
||||||
|
idx := &textFakeIndex{}
|
||||||
|
ix := NewTextIndexer(store, idx)
|
||||||
|
ix.Extractor = textFakeExtractor{prefix: "TEXT "}
|
||||||
|
|
||||||
|
res, err := ix.IndexEntries(context.Background(), []Entry{
|
||||||
|
{ID: "1", Title: "Rechnung.PDF", ParentID: "649", Kind: File},
|
||||||
|
{ID: "2", Title: "Tabelle.xlsx", ParentID: "649", Kind: File},
|
||||||
|
{ID: "3", Title: "Unterordner", Kind: Folder},
|
||||||
|
}, IndexOptions{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("IndexEntries: %v", err)
|
||||||
|
}
|
||||||
|
if res.Scanned != 3 || res.Indexed != 1 || res.Skipped != 2 || res.Failed != 0 {
|
||||||
|
t.Errorf("result = %+v, want scanned=3 indexed=1 skipped=2 failed=0", res)
|
||||||
|
}
|
||||||
|
if len(idx.docs) != 1 {
|
||||||
|
t.Fatalf("indexed docs = %d, want 1", len(idx.docs))
|
||||||
|
}
|
||||||
|
got := idx.docs[0]
|
||||||
|
want := TextDoc{ID: "1", Title: "Rechnung.PDF", FolderID: "649", Ext: "pdf", Content: "TEXT PDFBYTES"}
|
||||||
|
if !reflect.DeepEqual(got, want) {
|
||||||
|
t.Errorf("doc = %+v, want %+v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestTextIndexerRecordsExtractionFailure(t *testing.T) {
|
||||||
|
store := &textFakeStore{files: map[string][]byte{"1": []byte("x")}}
|
||||||
|
idx := &textFakeIndex{}
|
||||||
|
ix := NewTextIndexer(store, idx)
|
||||||
|
ix.Extractor = textFailingExtractor{}
|
||||||
|
|
||||||
|
res, err := ix.IndexEntries(context.Background(), []Entry{{ID: "1", Title: "a.pdf", Kind: File}}, IndexOptions{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("IndexEntries: %v", err)
|
||||||
|
}
|
||||||
|
if res.Indexed != 0 || res.Failed != 1 || len(res.Errors) != 1 {
|
||||||
|
t.Errorf("result = %+v, want one failure recorded", res)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestTextIndexerPlanFolder(t *testing.T) {
|
||||||
|
store := &textFakeStore{dirs: map[string][]Entry{
|
||||||
|
"root": {
|
||||||
|
{ID: "10", Title: "a.pdf", Kind: File},
|
||||||
|
{ID: "11", Title: "sub", Kind: Folder},
|
||||||
|
},
|
||||||
|
"11": {
|
||||||
|
{ID: "12", Title: "b.PDF", Kind: File},
|
||||||
|
{ID: "13", Title: "c.xlsx", Kind: File},
|
||||||
|
},
|
||||||
|
}}
|
||||||
|
ix := NewTextIndexer(store, &textFakeIndex{})
|
||||||
|
|
||||||
|
flat, err := ix.PlanFolder(context.Background(), "root", IndexOptions{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("PlanFolder: %v", err)
|
||||||
|
}
|
||||||
|
if len(flat) != 1 || flat[0].ID != "10" {
|
||||||
|
t.Errorf("flat plan = %+v, want only a.pdf", flat)
|
||||||
|
}
|
||||||
|
deep, err := ix.PlanFolder(context.Background(), "root", IndexOptions{Recursive: true, Limit: 10})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("PlanFolder recursive: %v", err)
|
||||||
|
}
|
||||||
|
if len(deep) != 2 {
|
||||||
|
t.Errorf("recursive plan = %d entries, want 2", len(deep))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- fakes -----------------------------------------------------------------
|
||||||
|
|
||||||
|
type textFakeStore struct {
|
||||||
|
dirs map[string][]Entry
|
||||||
|
files map[string][]byte
|
||||||
|
stat map[string]Entry
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *textFakeStore) Name() string { return "fake" }
|
||||||
|
|
||||||
|
func (f *textFakeStore) List(_ context.Context, parentID string) ([]Entry, error) {
|
||||||
|
return f.dirs[parentID], nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *textFakeStore) Stat(_ context.Context, id string) (Entry, error) {
|
||||||
|
if e, ok := f.stat[id]; ok {
|
||||||
|
return e, nil
|
||||||
|
}
|
||||||
|
return Entry{}, fmt.Errorf("not found: %s", id)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *textFakeStore) Download(_ context.Context, id string, w io.Writer) (int64, error) {
|
||||||
|
b, ok := f.files[id]
|
||||||
|
if !ok {
|
||||||
|
return 0, fmt.Errorf("no bytes for %s", id)
|
||||||
|
}
|
||||||
|
n, err := w.Write(b)
|
||||||
|
return int64(n), err
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *textFakeStore) CreateFolder(context.Context, string, string) (Entry, error) {
|
||||||
|
return Entry{}, nil
|
||||||
|
}
|
||||||
|
func (f *textFakeStore) Upload(context.Context, string, string, io.Reader) (Entry, error) {
|
||||||
|
return Entry{}, nil
|
||||||
|
}
|
||||||
|
func (f *textFakeStore) Move(context.Context, []string, string) error { return nil }
|
||||||
|
func (f *textFakeStore) Copy(context.Context, []string, string) error { return nil }
|
||||||
|
func (f *textFakeStore) Rename(context.Context, string, string) error { return nil }
|
||||||
|
func (f *textFakeStore) Delete(context.Context, []string) error { return nil }
|
||||||
|
|
||||||
|
type textFakeIndex struct{ docs []TextDoc }
|
||||||
|
|
||||||
|
func (f *textFakeIndex) Put(_ context.Context, docs []TextDoc) error {
|
||||||
|
f.docs = append(f.docs, docs...)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
func (f *textFakeIndex) Delete(context.Context, []string) error { return nil }
|
||||||
|
func (f *textFakeIndex) Search(context.Context, SearchQuery) ([]SearchHit, error) { return nil, nil }
|
||||||
|
func (f *textFakeIndex) Name() string { return "fake" }
|
||||||
|
|
||||||
|
type textFakeExtractor struct{ prefix string }
|
||||||
|
|
||||||
|
func (f textFakeExtractor) Extract(path, _, _ string, _ int) (string, error) {
|
||||||
|
b, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return f.prefix + string(b), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type textFailingExtractor struct{}
|
||||||
|
|
||||||
|
func (textFailingExtractor) Extract(string, string, string, int) (string, error) {
|
||||||
|
return "", fmt.Errorf("boom")
|
||||||
|
}
|
||||||
+4
-6
@@ -13,12 +13,10 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Provider names for the composed backends. ProviderPG is reserved for the
|
// ProviderES is the composed Elasticsearch searcher. The SQL store owns
|
||||||
// read-only PostgreSQL store (F2 #36); it is not registered until it exists.
|
// ProviderPG/ProviderMySQL (file_pg.go); the facade references ProviderPG in
|
||||||
const (
|
// readOrder.
|
||||||
ProviderPG = "postgres"
|
const ProviderES = "elasticsearch"
|
||||||
ProviderES = "elasticsearch"
|
|
||||||
)
|
|
||||||
|
|
||||||
var (
|
var (
|
||||||
errNoReadBackend = errors.New("onlyoffice: no file backend registered for reads")
|
errNoReadBackend = errors.New("onlyoffice: no file backend registered for reads")
|
||||||
|
|||||||
@@ -0,0 +1,337 @@
|
|||||||
|
package onlyoffice
|
||||||
|
|
||||||
|
// Text extraction pipeline for the own full-text index (epic #34, F6 #42).
|
||||||
|
//
|
||||||
|
// TextIndexer downloads stored documents, extracts text through docpipe
|
||||||
|
// (pdftotext; OCR for scans) and writes the result to a TextIndex. For PDFs it
|
||||||
|
// also indexes the text of embedded attachments (pdfdetach), so a scan filed
|
||||||
|
// as an attachment is searchable too. It is the write side of ESTextIndex and
|
||||||
|
// never touches the OnlyOffice server's own ES index.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"sync"
|
||||||
|
|
||||||
|
"github.com/eslider/go-onlyoffice/internal/docpipe"
|
||||||
|
)
|
||||||
|
|
||||||
|
// defaultTextIndexExts are the formats extracted by default. The OnlyOffice
|
||||||
|
// index already covers docx/xlsx/pptx; F6 adds PDF.
|
||||||
|
var defaultTextIndexExts = []string{"pdf"}
|
||||||
|
|
||||||
|
const (
|
||||||
|
defaultTextIndexWorkers = 3
|
||||||
|
defaultTextIndexLang = "deu+eng"
|
||||||
|
maxTextIndexErrors = 20
|
||||||
|
)
|
||||||
|
|
||||||
|
// TextExtractor turns a local file into indexable plain text. The default uses
|
||||||
|
// docpipe (pdftotext + OCR); tests inject a fake to stay offline.
|
||||||
|
type TextExtractor interface {
|
||||||
|
Extract(path, workDir, lang string, minChars int) (string, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// docpipeExtractor is the production TextExtractor.
|
||||||
|
type docpipeExtractor struct{ tools docpipe.Tools }
|
||||||
|
|
||||||
|
// Extract renders the file as Markdown, OCRing PDFs/images with a weak text
|
||||||
|
// layer first and appending the text of embedded PDF attachments
|
||||||
|
// (docpipe.ToMarkdownWithAttachments).
|
||||||
|
func (d docpipeExtractor) Extract(path, workDir, lang string, minChars int) (string, error) {
|
||||||
|
text, err := d.tools.ToMarkdownWithAttachments(path, workDir, lang, minChars)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return strings.TrimSpace(text), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// IndexOptions controls a TextIndexer run.
|
||||||
|
type IndexOptions struct {
|
||||||
|
Recursive bool // IndexFolder: descend into subfolders
|
||||||
|
Extensions []string // empty = defaultTextIndexExts (pdf)
|
||||||
|
Limit int // max files to index, 0 = all
|
||||||
|
Lang string // OCR language(s), default deu+eng
|
||||||
|
MinChars int // OCR threshold, default docpipe.DefaultMinTextChars
|
||||||
|
Workers int // parallel downloads/extractions, default 3
|
||||||
|
}
|
||||||
|
|
||||||
|
// IndexResult summarises a run.
|
||||||
|
type IndexResult struct {
|
||||||
|
Scanned int
|
||||||
|
Indexed int
|
||||||
|
Skipped int
|
||||||
|
Failed int
|
||||||
|
Errors []string
|
||||||
|
}
|
||||||
|
|
||||||
|
// TextIndexer wires a FileStore, a TextIndex and an extractor together.
|
||||||
|
type TextIndexer struct {
|
||||||
|
Store FileStore
|
||||||
|
Index TextIndex
|
||||||
|
Extractor TextExtractor // nil = local docpipe tools
|
||||||
|
WorkDir string // temp dir for downloads/extraction
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewTextIndexer returns a TextIndexer over the given store and index.
|
||||||
|
func NewTextIndexer(store FileStore, index TextIndex) *TextIndexer {
|
||||||
|
return &TextIndexer{Store: store, Index: index}
|
||||||
|
}
|
||||||
|
|
||||||
|
// textIndexEnsurer is implemented by indexes that can be created up front.
|
||||||
|
type textIndexEnsurer interface {
|
||||||
|
Ensure(ctx context.Context) error
|
||||||
|
}
|
||||||
|
|
||||||
|
// Ensure creates the backing index when the TextIndex supports it.
|
||||||
|
func (ix *TextIndexer) Ensure(ctx context.Context) error {
|
||||||
|
if e, ok := ix.Index.(textIndexEnsurer); ok {
|
||||||
|
return e.Ensure(ctx)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// IndexFiles stats the given file ids and indexes them.
|
||||||
|
func (ix *TextIndexer) IndexFiles(ctx context.Context, ids []string, opts IndexOptions) (IndexResult, error) {
|
||||||
|
entries := make([]Entry, 0, len(ids))
|
||||||
|
for _, id := range ids {
|
||||||
|
e, err := ix.Store.Stat(ctx, id)
|
||||||
|
if err != nil {
|
||||||
|
return IndexResult{}, fmt.Errorf("onlyoffice: stat %s: %w", id, err)
|
||||||
|
}
|
||||||
|
entries = append(entries, e)
|
||||||
|
}
|
||||||
|
return ix.IndexEntries(ctx, entries, opts)
|
||||||
|
}
|
||||||
|
|
||||||
|
// IndexFolder lists a folder and indexes every matching file.
|
||||||
|
func (ix *TextIndexer) IndexFolder(ctx context.Context, folderID string, opts IndexOptions) (IndexResult, error) {
|
||||||
|
entries, err := ix.collect(ctx, folderID, opts.Recursive)
|
||||||
|
if err != nil {
|
||||||
|
return IndexResult{}, err
|
||||||
|
}
|
||||||
|
return ix.IndexEntries(ctx, entries, opts)
|
||||||
|
}
|
||||||
|
|
||||||
|
// PlanFolder lists the files IndexFolder would process, without downloading or
|
||||||
|
// extracting anything.
|
||||||
|
func (ix *TextIndexer) PlanFolder(ctx context.Context, folderID string, opts IndexOptions) ([]Entry, error) {
|
||||||
|
entries, err := ix.collect(ctx, folderID, opts.Recursive)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return selectEntries(entries, opts), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// PlanFiles stats the ids and returns those that would be indexed.
|
||||||
|
func (ix *TextIndexer) PlanFiles(ctx context.Context, ids []string, opts IndexOptions) ([]Entry, error) {
|
||||||
|
entries := make([]Entry, 0, len(ids))
|
||||||
|
for _, id := range ids {
|
||||||
|
e, err := ix.Store.Stat(ctx, id)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("onlyoffice: stat %s: %w", id, err)
|
||||||
|
}
|
||||||
|
entries = append(entries, e)
|
||||||
|
}
|
||||||
|
return selectEntries(entries, opts), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// IndexEntries extracts and indexes the given files (folders are ignored).
|
||||||
|
func (ix *TextIndexer) IndexEntries(ctx context.Context, entries []Entry, opts IndexOptions) (IndexResult, error) {
|
||||||
|
opts = opts.withDefaults()
|
||||||
|
var res IndexResult
|
||||||
|
work := selectEntries(entries, opts)
|
||||||
|
res.Scanned = len(entries)
|
||||||
|
res.Skipped = len(entries) - len(work)
|
||||||
|
if len(work) == 0 {
|
||||||
|
return res, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
workers := opts.Workers
|
||||||
|
if workers > len(work) {
|
||||||
|
workers = len(work)
|
||||||
|
}
|
||||||
|
if workers < 1 {
|
||||||
|
workers = 1
|
||||||
|
}
|
||||||
|
|
||||||
|
type outcome struct {
|
||||||
|
doc TextDoc
|
||||||
|
err error
|
||||||
|
}
|
||||||
|
jobs := make(chan Entry)
|
||||||
|
results := make(chan outcome, workers)
|
||||||
|
var wg sync.WaitGroup
|
||||||
|
for i := 0; i < workers; i++ {
|
||||||
|
wg.Add(1)
|
||||||
|
go func() {
|
||||||
|
defer wg.Done()
|
||||||
|
for e := range jobs {
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
results <- outcome{err: err}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
doc, err := ix.indexOne(ctx, e, opts)
|
||||||
|
results <- outcome{doc: doc, err: err}
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
}
|
||||||
|
go func() {
|
||||||
|
defer close(jobs)
|
||||||
|
for _, e := range work {
|
||||||
|
select {
|
||||||
|
case jobs <- e:
|
||||||
|
case <-ctx.Done():
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
go func() {
|
||||||
|
wg.Wait()
|
||||||
|
close(results)
|
||||||
|
}()
|
||||||
|
|
||||||
|
var docs []TextDoc
|
||||||
|
for r := range results {
|
||||||
|
if r.err != nil {
|
||||||
|
res.Failed++
|
||||||
|
if len(res.Errors) < maxTextIndexErrors {
|
||||||
|
res.Errors = append(res.Errors, r.err.Error())
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
docs = append(docs, r.doc)
|
||||||
|
}
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return res, err
|
||||||
|
}
|
||||||
|
if len(docs) > 0 {
|
||||||
|
if err := ix.Index.Put(ctx, docs); err != nil {
|
||||||
|
return res, fmt.Errorf("onlyoffice: index %d docs: %w", len(docs), err)
|
||||||
|
}
|
||||||
|
res.Indexed = len(docs)
|
||||||
|
}
|
||||||
|
return res, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// indexOne downloads and extracts a single file.
|
||||||
|
func (ix *TextIndexer) indexOne(ctx context.Context, e Entry, opts IndexOptions) (TextDoc, error) {
|
||||||
|
ext := fileExt(e.Title)
|
||||||
|
dir := ix.WorkDir
|
||||||
|
if dir == "" {
|
||||||
|
dir = os.TempDir()
|
||||||
|
}
|
||||||
|
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||||
|
return TextDoc{}, err
|
||||||
|
}
|
||||||
|
tmp, err := os.CreateTemp(dir, "ooidx-*."+ext)
|
||||||
|
if err != nil {
|
||||||
|
return TextDoc{}, err
|
||||||
|
}
|
||||||
|
tmpPath := tmp.Name()
|
||||||
|
defer os.Remove(tmpPath)
|
||||||
|
|
||||||
|
if _, err := ix.Store.Download(ctx, e.ID, tmp); err != nil {
|
||||||
|
tmp.Close()
|
||||||
|
return TextDoc{}, fmt.Errorf("download %s (%s): %w", e.ID, e.Title, err)
|
||||||
|
}
|
||||||
|
if err := tmp.Close(); err != nil {
|
||||||
|
return TextDoc{}, err
|
||||||
|
}
|
||||||
|
text, err := ix.extractor().Extract(tmpPath, dir, opts.Lang, opts.MinChars)
|
||||||
|
if err != nil {
|
||||||
|
return TextDoc{}, fmt.Errorf("extract %s: %w", e.Title, err)
|
||||||
|
}
|
||||||
|
return TextDoc{ID: e.ID, Title: e.Title, FolderID: e.ParentID, Ext: ext, Content: text}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// collect lists files under folderID, breadth-first when recursive.
|
||||||
|
func (ix *TextIndexer) collect(ctx context.Context, folderID string, recursive bool) ([]Entry, error) {
|
||||||
|
var files []Entry
|
||||||
|
queue := []string{folderID}
|
||||||
|
for len(queue) > 0 {
|
||||||
|
if err := ctx.Err(); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
id := queue[0]
|
||||||
|
queue = queue[1:]
|
||||||
|
entries, err := ix.Store.List(ctx, id)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("onlyoffice: list folder %s: %w", id, err)
|
||||||
|
}
|
||||||
|
for _, e := range entries {
|
||||||
|
if e.Kind == Folder {
|
||||||
|
if recursive {
|
||||||
|
queue = append(queue, e.ID)
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if e.ParentID == "" {
|
||||||
|
e.ParentID = id
|
||||||
|
}
|
||||||
|
files = append(files, e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return files, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// selectEntries filters files by extension and applies the limit.
|
||||||
|
func selectEntries(entries []Entry, opts IndexOptions) []Entry {
|
||||||
|
allowed := extensionSet(opts.Extensions)
|
||||||
|
work := make([]Entry, 0, len(entries))
|
||||||
|
for _, e := range entries {
|
||||||
|
if e.Kind != File {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if opts.Limit > 0 && len(work) >= opts.Limit {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
if !allowed[fileExt(e.Title)] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
work = append(work, e)
|
||||||
|
}
|
||||||
|
return work
|
||||||
|
}
|
||||||
|
|
||||||
|
// extensionSet normalises the extension allow-list (default: pdf).
|
||||||
|
func extensionSet(exts []string) map[string]bool {
|
||||||
|
if len(exts) == 0 {
|
||||||
|
exts = defaultTextIndexExts
|
||||||
|
}
|
||||||
|
set := make(map[string]bool, len(exts))
|
||||||
|
for _, e := range normalizeExtensions(exts) {
|
||||||
|
set[e] = true
|
||||||
|
}
|
||||||
|
return set
|
||||||
|
}
|
||||||
|
|
||||||
|
// fileExt returns the lower-case extension without the dot.
|
||||||
|
func fileExt(title string) string {
|
||||||
|
return strings.ToLower(strings.TrimPrefix(filepath.Ext(strings.TrimSpace(title)), "."))
|
||||||
|
}
|
||||||
|
|
||||||
|
// withDefaults fills zero-valued options.
|
||||||
|
func (o IndexOptions) withDefaults() IndexOptions {
|
||||||
|
if o.Workers <= 0 {
|
||||||
|
o.Workers = defaultTextIndexWorkers
|
||||||
|
}
|
||||||
|
if o.MinChars <= 0 {
|
||||||
|
o.MinChars = docpipe.DefaultMinTextChars
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(o.Lang) == "" {
|
||||||
|
o.Lang = defaultTextIndexLang
|
||||||
|
}
|
||||||
|
return o
|
||||||
|
}
|
||||||
|
|
||||||
|
// extractor returns the configured extractor or the local docpipe default.
|
||||||
|
func (ix *TextIndexer) extractor() TextExtractor {
|
||||||
|
if ix.Extractor != nil {
|
||||||
|
return ix.Extractor
|
||||||
|
}
|
||||||
|
return docpipeExtractor{tools: docpipe.LookPath()}
|
||||||
|
}
|
||||||
@@ -5,6 +5,7 @@
|
|||||||
// - pandoc — md↔docx
|
// - pandoc — md↔docx
|
||||||
// - ocrmypdf — OCR into a searchable PDF
|
// - ocrmypdf — OCR into a searchable PDF
|
||||||
// - pdftotext — extract text layer
|
// - pdftotext — extract text layer
|
||||||
|
// - pdfdetach — list/save embedded PDF attachments
|
||||||
// - tesseract — OCR single images when ocrmypdf is unsuitable
|
// - tesseract — OCR single images when ocrmypdf is unsuitable
|
||||||
// - ghostscript (gs) — PDF rewrite/optimize via PostScript (pdfwrite)
|
// - ghostscript (gs) — PDF rewrite/optimize via PostScript (pdfwrite)
|
||||||
package docpipe
|
package docpipe
|
||||||
@@ -26,6 +27,7 @@ type Tools struct {
|
|||||||
Pandoc string
|
Pandoc string
|
||||||
OCRMyPDF string
|
OCRMyPDF string
|
||||||
PDFToText string
|
PDFToText string
|
||||||
|
PDFDetach string
|
||||||
Tesseract string
|
Tesseract string
|
||||||
Ghostscript string
|
Ghostscript string
|
||||||
}
|
}
|
||||||
@@ -44,6 +46,7 @@ func LookPath() Tools {
|
|||||||
Pandoc: find("pandoc"),
|
Pandoc: find("pandoc"),
|
||||||
OCRMyPDF: find("ocrmypdf"),
|
OCRMyPDF: find("ocrmypdf"),
|
||||||
PDFToText: find("pdftotext"),
|
PDFToText: find("pdftotext"),
|
||||||
|
PDFDetach: find("pdfdetach"),
|
||||||
Tesseract: find("tesseract"),
|
Tesseract: find("tesseract"),
|
||||||
Ghostscript: find("gs", "ghostscript"),
|
Ghostscript: find("gs", "ghostscript"),
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,231 @@
|
|||||||
|
package docpipe
|
||||||
|
|
||||||
|
// Embedded PDF attachments (F6 #42). Digitised invoices often carry the
|
||||||
|
// original scan as a PDF attachment; the searchable body may hold only a
|
||||||
|
// summary. pdfdetach (poppler) lists/saves them; each saved attachment is run
|
||||||
|
// through the normal docpipe extraction (pdftotext/OCR).
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/xml"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// PDFAttachment is one embedded file in a PDF.
|
||||||
|
type PDFAttachment struct {
|
||||||
|
Index int // 1-based number as `pdfdetach -list` reports it
|
||||||
|
Name string // embedded file name
|
||||||
|
}
|
||||||
|
|
||||||
|
// AttachmentText is the extracted text of one embedded attachment.
|
||||||
|
type AttachmentText struct {
|
||||||
|
Name string
|
||||||
|
Text string
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseAttachmentList parses `pdfdetach -list` output. The first line is a
|
||||||
|
// count ("N embedded files"); every following line is "<index>: <name>".
|
||||||
|
// Pure, so it is unit-tested.
|
||||||
|
func parseAttachmentList(out string) []PDFAttachment {
|
||||||
|
var atts []PDFAttachment
|
||||||
|
for _, line := range strings.Split(out, "\n") {
|
||||||
|
line = strings.TrimSpace(line)
|
||||||
|
if line == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
colon := strings.Index(line, ":")
|
||||||
|
if colon <= 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
n, err := strconv.Atoi(strings.TrimSpace(line[:colon]))
|
||||||
|
if err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
name := strings.TrimSpace(line[colon+1:])
|
||||||
|
if name == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
atts = append(atts, PDFAttachment{Index: n, Name: name})
|
||||||
|
}
|
||||||
|
return atts
|
||||||
|
}
|
||||||
|
|
||||||
|
// safeAttachmentName strips directories and leading dots so a hostile
|
||||||
|
// attachment name cannot escape the extraction directory.
|
||||||
|
func safeAttachmentName(name string) string {
|
||||||
|
name = strings.ReplaceAll(strings.TrimSpace(name), "\\", "/")
|
||||||
|
name = filepath.Base(name)
|
||||||
|
name = strings.TrimLeft(name, ".")
|
||||||
|
if name == "" || name == "." || name == "/" {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return name
|
||||||
|
}
|
||||||
|
|
||||||
|
// JoinWithAttachments appends attachment text to the document body, each
|
||||||
|
// section preceded by an "[attachment: <name>]" marker so a search hit shows
|
||||||
|
// its source. Empty attachments are skipped. Pure, so it is unit-tested.
|
||||||
|
func JoinWithAttachments(body string, atts []AttachmentText) string {
|
||||||
|
var b strings.Builder
|
||||||
|
b.WriteString(strings.TrimRight(body, "\n"))
|
||||||
|
for _, a := range atts {
|
||||||
|
text := strings.TrimSpace(a.Text)
|
||||||
|
if text == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
b.WriteString("\n\n[attachment: ")
|
||||||
|
b.WriteString(a.Name)
|
||||||
|
b.WriteString("]\n\n")
|
||||||
|
b.WriteString(text)
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ListAttachments returns the embedded files of a PDF. A PDF without
|
||||||
|
// attachments yields an empty slice and no error.
|
||||||
|
func (t Tools) ListAttachments(pdfPath string) ([]PDFAttachment, error) {
|
||||||
|
if t.PDFDetach == "" {
|
||||||
|
return nil, fmt.Errorf("pdfdetach not found on PATH")
|
||||||
|
}
|
||||||
|
cmd := exec.Command(t.PDFDetach, "-list", pdfPath)
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
cmd.Stderr = &stderr
|
||||||
|
out, err := cmd.Output()
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("pdfdetach -list %s: %w (%s)", filepath.Base(pdfPath), err, strings.TrimSpace(stderr.String()))
|
||||||
|
}
|
||||||
|
return parseAttachmentList(string(out)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// SaveAttachment writes the n-th embedded file (1-based) to outPath.
|
||||||
|
func (t Tools) SaveAttachment(pdfPath string, index int, outPath string) error {
|
||||||
|
if t.PDFDetach == "" {
|
||||||
|
return fmt.Errorf("pdfdetach not found on PATH")
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(outPath) == "" {
|
||||||
|
return fmt.Errorf("output path required")
|
||||||
|
}
|
||||||
|
if err := EnsureDir(outPath); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
cmd := exec.Command(t.PDFDetach, "-save", strconv.Itoa(index), "-o", outPath, pdfPath)
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
cmd.Stderr = &stderr
|
||||||
|
if err := cmd.Run(); err != nil {
|
||||||
|
return fmt.Errorf("pdfdetach -save %d: %w (%s)", index, err, strings.TrimSpace(stderr.String()))
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ToMarkdownWithAttachments extracts the file as ToMarkdown does, then — for
|
||||||
|
// PDFs — appends the text of every embedded attachment under an
|
||||||
|
// "[attachment: <name>]" marker. Attachment failures are non-fatal: the body
|
||||||
|
// is returned unchanged.
|
||||||
|
func (t Tools) ToMarkdownWithAttachments(path, workDir, lang string, minChars int) (string, error) {
|
||||||
|
res, err := t.ToMarkdown(path, workDir, lang, minChars)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
if Ext(path) != ".pdf" {
|
||||||
|
return res.Markdown, nil
|
||||||
|
}
|
||||||
|
atts, err := t.attachmentTexts(path, workDir, lang, minChars)
|
||||||
|
if err != nil {
|
||||||
|
return res.Markdown, nil
|
||||||
|
}
|
||||||
|
return JoinWithAttachments(res.Markdown, atts), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// attachmentTexts saves and extracts every embedded attachment, skipping the
|
||||||
|
// ones that cannot be read. It returns an error only when the attachment list
|
||||||
|
// itself cannot be obtained.
|
||||||
|
func (t Tools) attachmentTexts(pdfPath, workDir, lang string, minChars int) ([]AttachmentText, error) {
|
||||||
|
list, err := t.ListAttachments(pdfPath)
|
||||||
|
if err != nil || len(list) == 0 {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if workDir == "" {
|
||||||
|
workDir = os.TempDir()
|
||||||
|
}
|
||||||
|
dir := filepath.Join(workDir, "att-"+trimExt(filepath.Base(pdfPath)))
|
||||||
|
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer os.RemoveAll(dir)
|
||||||
|
|
||||||
|
out := make([]AttachmentText, 0, len(list))
|
||||||
|
for _, a := range list {
|
||||||
|
name := safeAttachmentName(a.Name)
|
||||||
|
if name == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
saved := filepath.Join(dir, fmt.Sprintf("%d-%s", a.Index, name))
|
||||||
|
if err := t.SaveAttachment(pdfPath, a.Index, saved); err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
text, err := t.attachmentMarkdown(saved, dir, lang, minChars)
|
||||||
|
if err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
out = append(out, AttachmentText{Name: a.Name, Text: text})
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// attachmentMarkdown extracts a saved attachment with the regular pipeline.
|
||||||
|
// Structured attachments that docpipe does not convert (e-invoice XML,
|
||||||
|
// CuraSoft JSON, CSV/HTML) fall back to their text content, so the embedded
|
||||||
|
// original is still searchable. Other unreadable formats return an error and
|
||||||
|
// the caller skips them.
|
||||||
|
func (t Tools) attachmentMarkdown(path, workDir, lang string, minChars int) (string, error) {
|
||||||
|
if res, err := t.ToMarkdown(path, workDir, lang, minChars); err == nil {
|
||||||
|
return res.Markdown, nil
|
||||||
|
}
|
||||||
|
switch Ext(path) {
|
||||||
|
case ".xml", ".html", ".htm":
|
||||||
|
raw, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return xmlToText(raw), nil
|
||||||
|
case ".json", ".csv":
|
||||||
|
raw, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return string(raw), nil
|
||||||
|
default:
|
||||||
|
return "", fmt.Errorf("unsupported attachment type %q", Ext(path))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// xmlToText returns the character data of an XML/HTML document: element text
|
||||||
|
// values with decoded entities, one per line. Used for invoice XML (EN 16931
|
||||||
|
// CII / ZUGFeRD) and HTML attachments. Pure, so it is unit-tested.
|
||||||
|
func xmlToText(raw []byte) string {
|
||||||
|
dec := xml.NewDecoder(bytes.NewReader(raw))
|
||||||
|
dec.Strict = false
|
||||||
|
var b strings.Builder
|
||||||
|
for {
|
||||||
|
tok, err := dec.Token()
|
||||||
|
if err != nil {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
cd, ok := tok.(xml.CharData)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
s := strings.TrimSpace(string(cd))
|
||||||
|
if s == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
b.WriteString(s)
|
||||||
|
b.WriteByte('\n')
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
@@ -0,0 +1,163 @@
|
|||||||
|
package docpipe
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestParseAttachmentList(t *testing.T) {
|
||||||
|
out := "2 embedded files\n1: original.pdf\n2: scan_001.png\n"
|
||||||
|
got := parseAttachmentList(out)
|
||||||
|
want := []PDFAttachment{{Index: 1, Name: "original.pdf"}, {Index: 2, Name: "scan_001.png"}}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("got %+v, want %+v", got, want)
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("att[%d] = %+v, want %+v", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParseAttachmentListEmptyAndMalformed(t *testing.T) {
|
||||||
|
for _, in := range []string{"", "0 embedded files\n", "garbage\n\n \n"} {
|
||||||
|
if got := parseAttachmentList(in); len(got) != 0 {
|
||||||
|
t.Errorf("parseAttachmentList(%q) = %+v, want empty", in, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSafeAttachmentName(t *testing.T) {
|
||||||
|
cases := map[string]string{
|
||||||
|
"note.txt": "note.txt",
|
||||||
|
"../../evil.pdf": "evil.pdf",
|
||||||
|
`..\..\evil.pdf`: "evil.pdf",
|
||||||
|
"/abs/scan_001.pdf": "scan_001.pdf",
|
||||||
|
".hidden": "hidden",
|
||||||
|
" spaced name.txt ": "spaced name.txt",
|
||||||
|
"..": "",
|
||||||
|
"": "",
|
||||||
|
}
|
||||||
|
for in, want := range cases {
|
||||||
|
if got := safeAttachmentName(in); got != want {
|
||||||
|
t.Errorf("safeAttachmentName(%q) = %q, want %q", in, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestJoinWithAttachments(t *testing.T) {
|
||||||
|
body := "# scan.pdf\n\nbody token\n"
|
||||||
|
atts := []AttachmentText{
|
||||||
|
{Name: "original.pdf", Text: " original token "},
|
||||||
|
{Name: "empty.txt", Text: " "},
|
||||||
|
}
|
||||||
|
got := JoinWithAttachments(body, atts)
|
||||||
|
if !strings.Contains(got, "body token") {
|
||||||
|
t.Errorf("body text lost: %q", got)
|
||||||
|
}
|
||||||
|
if !strings.Contains(got, "[attachment: original.pdf]") {
|
||||||
|
t.Errorf("marker missing: %q", got)
|
||||||
|
}
|
||||||
|
if !strings.Contains(got, "original token") {
|
||||||
|
t.Errorf("attachment text missing: %q", got)
|
||||||
|
}
|
||||||
|
if strings.Contains(got, "empty.txt") {
|
||||||
|
t.Errorf("empty attachment must be skipped: %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestJoinWithAttachmentsNoAttachments(t *testing.T) {
|
||||||
|
got := JoinWithAttachments("# a.pdf\n\ntext\n\n", nil)
|
||||||
|
if got != "# a.pdf\n\ntext" {
|
||||||
|
t.Errorf("got %q, want trimmed body only", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestListAttachmentsWithoutTool(t *testing.T) {
|
||||||
|
if _, err := (Tools{}).ListAttachments("x.pdf"); err == nil || !strings.Contains(err.Error(), "pdfdetach") {
|
||||||
|
t.Fatalf("want pdfdetach error, got %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestToMarkdownWithAttachmentsFixture exercises the real pdfdetach + pdftotext
|
||||||
|
// pipeline on testdata/pdf-with-attachment.pdf (body token + embedded
|
||||||
|
// goo-note.txt). Skips when poppler is not installed.
|
||||||
|
func TestToMarkdownWithAttachmentsFixture(t *testing.T) {
|
||||||
|
tools := LookPath()
|
||||||
|
if tools.PDFDetach == "" || tools.PDFToText == "" {
|
||||||
|
t.Skip("pdfdetach/pdftotext not on PATH — skipping attachment extraction test")
|
||||||
|
}
|
||||||
|
fixture := filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf")
|
||||||
|
got, err := tools.ToMarkdownWithAttachments(fixture, t.TempDir(), "eng", 1)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ToMarkdownWithAttachments: %v", err)
|
||||||
|
}
|
||||||
|
for _, want := range []string{"goobodytoken", "[attachment: goo-note.txt]", "gooattachmenttoken"} {
|
||||||
|
if !strings.Contains(got, want) {
|
||||||
|
t.Errorf("result missing %q:\n%s", want, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestXMLToText(t *testing.T) {
|
||||||
|
raw := []byte(`<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<rsm:CrossIndustryInvoice><rsm:ExchangedDocument>
|
||||||
|
<ram:ID>S1063</ram:ID></rsm:ExchangedDocument>
|
||||||
|
<ram:Name>Edelweiss & Co</ram:Name><ram:GrandTotalAmount>42.00</ram:GrandTotalAmount>
|
||||||
|
</rsm:CrossIndustryInvoice>`)
|
||||||
|
got := xmlToText(raw)
|
||||||
|
for _, want := range []string{"S1063", "Edelweiss & Co", "42.00"} {
|
||||||
|
if !strings.Contains(got, want) {
|
||||||
|
t.Errorf("xmlToText missing %q:\n%s", want, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if strings.ContainsAny(got, "<>") {
|
||||||
|
t.Errorf("xmlToText left markup: %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAttachmentMarkdownFallback verifies structured attachments that docpipe
|
||||||
|
// cannot convert are still reduced to searchable text, and unknown binary
|
||||||
|
// formats error (so the caller skips them).
|
||||||
|
func TestAttachmentMarkdownFallback(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
xmlPath := filepath.Join(dir, "factur-x.xml")
|
||||||
|
if err := os.WriteFile(xmlPath, []byte(`<Invoice><Number>S1063</Number></Invoice>`), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
got, err := (Tools{}).attachmentMarkdown(xmlPath, dir, "", 0)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("attachmentMarkdown(xml): %v", err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(got, "S1063") {
|
||||||
|
t.Errorf("xml attachment text = %q, want S1063", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
binPath := filepath.Join(dir, "data.bin")
|
||||||
|
if err := os.WriteFile(binPath, []byte{0, 1, 2, 3}, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, err := (Tools{}).attachmentMarkdown(binPath, dir, "", 0); err == nil {
|
||||||
|
t.Error("unsupported attachment: want error, got nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestToMarkdownWithAttachmentsPlainPDF ensures a PDF without attachments
|
||||||
|
// returns just the body (pdfdetach prints "0 embedded files").
|
||||||
|
func TestToMarkdownWithAttachmentsPlainPDF(t *testing.T) {
|
||||||
|
tools := LookPath()
|
||||||
|
if tools.PDFDetach == "" || tools.PDFToText == "" {
|
||||||
|
t.Skip("pdfdetach/pdftotext not on PATH")
|
||||||
|
}
|
||||||
|
// The fixture itself is a PDF with one attachment; strip it by extracting
|
||||||
|
// the body only through ToMarkdown and compare JoinWithAttachments(nil).
|
||||||
|
res, err := tools.ToMarkdown(filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf"), t.TempDir(), "eng", 1)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ToMarkdown: %v", err)
|
||||||
|
}
|
||||||
|
if strings.Contains(res.Markdown, "gooattachmenttoken") {
|
||||||
|
t.Fatalf("body must not contain attachment text: %q", res.Markdown)
|
||||||
|
}
|
||||||
|
}
|
||||||
Vendored
+114
@@ -0,0 +1,114 @@
|
|||||||
|
%PDF-1.7
|
||||||
|
%Çì�¢
|
||||||
|
%%Invocation: gs -q -dNOPAUSE -dBATCH -sDEVICE=pdfwrite ? ? ?
|
||||||
|
5 0 obj
|
||||||
|
<</Length 6 0 R/Filter /FlateDecode>>
|
||||||
|
stream
|
||||||
|
xœ-ŠA
|
||||||
|
ƒ0÷ÿoÙnìO7q]èZ*ÿ�IˆmJ½½M
|
||||||
|
³f7
|
||||||
|
\¨6‘žfRÿŠ*ñºõª…8:g}‡f†Dºø”ÞiØsúØ ÝôÝ;炱(Ùn.ly]ìUFz
|
||||||
|
½~Aë øendstream
|
||||||
|
endobj
|
||||||
|
6 0 obj
|
||||||
|
110
|
||||||
|
endobj
|
||||||
|
4 0 obj
|
||||||
|
<</Type/Page/MediaBox [0 0 612 792]
|
||||||
|
/Rotate 0/Parent 3 0 R
|
||||||
|
/Resources<</ProcSet[/PDF /Text]
|
||||||
|
/Font 8 0 R
|
||||||
|
>>
|
||||||
|
/Contents 5 0 R
|
||||||
|
>>
|
||||||
|
endobj
|
||||||
|
3 0 obj
|
||||||
|
<< /Type /Pages /Kids [
|
||||||
|
4 0 R
|
||||||
|
] /Count 1
|
||||||
|
>>
|
||||||
|
endobj
|
||||||
|
1 0 obj
|
||||||
|
<</Type /Catalog /Pages 3 0 R
|
||||||
|
/Metadata 9 0 R
|
||||||
|
>>
|
||||||
|
endobj
|
||||||
|
8 0 obj
|
||||||
|
<</R7
|
||||||
|
7 0 R>>
|
||||||
|
endobj
|
||||||
|
7 0 obj
|
||||||
|
<</BaseFont/Helvetica/Type/Font
|
||||||
|
/Subtype/Type1>>
|
||||||
|
endobj
|
||||||
|
9 0 obj
|
||||||
|
<</Type/Metadata
|
||||||
|
/Subtype/XML/Length 1173>>stream
|
||||||
|
<?xpacket begin='' id='W5M0MpCehiHzreSzNTczkc9d'?>
|
||||||
|
<?adobe-xap-filters esc="CRLF"?>
|
||||||
|
<x:xmpmeta xmlns:x='adobe:ns:meta/' x:xmptk='XMP toolkit 2.9.1-13, framework 1.6'>
|
||||||
|
<rdf:RDF xmlns:rdf='http://www.w3.org/1999/02/22-rdf-syntax-ns#' xmlns:iX='http://ns.adobe.com/iX/1.0/'>
|
||||||
|
<rdf:Description rdf:about="" xmlns:pdf='http://ns.adobe.com/pdf/1.3/' pdf:Producer='GPL Ghostscript 10.02.1'/>
|
||||||
|
<rdf:Description rdf:about="" xmlns:xmp='http://ns.adobe.com/xap/1.0/'><xmp:ModifyDate>2026-09-16T17:29:12Z</xmp:ModifyDate>
|
||||||
|
<xmp:CreateDate>2026-09-16T17:29:12Z</xmp:CreateDate>
|
||||||
|
<xmp:CreatorTool>UnknownApplication</xmp:CreatorTool></rdf:Description>
|
||||||
|
<rdf:Description rdf:about="" xmlns:xapMM='http://ns.adobe.com/xap/1.0/mm/' xapMM:DocumentID='uuid:ae8d7575-ea10-11fc-0000-3a98742e0b71'/>
|
||||||
|
<rdf:Description rdf:about="" xmlns:dc='http://purl.org/dc/elements/1.1/' dc:format='application/pdf'><dc:title><rdf:Alt><rdf:li xml:lang='x-default'>Untitled</rdf:li></rdf:Alt></dc:title></rdf:Description>
|
||||||
|
</rdf:RDF>
|
||||||
|
</x:xmpmeta>
|
||||||
|
|
||||||
|
|
||||||
|
<?xpacket end='w'?>
|
||||||
|
endstream
|
||||||
|
endobj
|
||||||
|
2 0 obj
|
||||||
|
<</Producer(GPL Ghostscript 10.02.1)
|
||||||
|
/CreationDate(D:20260916172912Z00'00')
|
||||||
|
/ModDate(D:20260916172912Z00'00')>>endobj
|
||||||
|
xref
|
||||||
|
0 10
|
||||||
|
0000000000 65535 f
|
||||||
|
0000000476 00000 n
|
||||||
|
0000001882 00000 n
|
||||||
|
0000000417 00000 n
|
||||||
|
0000000276 00000 n
|
||||||
|
0000000077 00000 n
|
||||||
|
0000000257 00000 n
|
||||||
|
0000000569 00000 n
|
||||||
|
0000000540 00000 n
|
||||||
|
0000000633 00000 n
|
||||||
|
trailer
|
||||||
|
<< /Size 10 /Root 1 0 R /Info 2 0 R
|
||||||
|
/ID [<5A562476FBF11852133AA873452909F9><5A562476FBF11852133AA873452909F9>]
|
||||||
|
>>
|
||||||
|
startxref
|
||||||
|
2008
|
||||||
|
%%EOF
|
||||||
|
1 0 obj
|
||||||
|
<</Type /Catalog /Pages 3 0 R /Metadata 9 0 R /Names <</EmbeddedFiles 12 0 R >> >>
|
||||||
|
endobj
|
||||||
|
10 0 obj
|
||||||
|
<</Length 44 /Params <</Size 44 >> >> stream
|
||||||
|
gooattachmenttoken embedded attachment text
|
||||||
|
|
||||||
|
endstream
|
||||||
|
|
||||||
|
endobj
|
||||||
|
11 0 obj
|
||||||
|
<</Type /Filespec /UF <feff0067006f006f002d006e006f00740065002e007400780074> /EF <</F 10 0 R >> >>
|
||||||
|
endobj
|
||||||
|
12 0 obj
|
||||||
|
<</Names [<feff0067006f006f002d006e006f00740065002e007400780074> 11 0 R ] >>
|
||||||
|
endobj
|
||||||
|
xref
|
||||||
|
0 2
|
||||||
|
0000000002 65535 f
|
||||||
|
0000002361 00000 n
|
||||||
|
10 3
|
||||||
|
0000002463 00000 n
|
||||||
|
0000002586 00000 n
|
||||||
|
0000002705 00000 n
|
||||||
|
trailer
|
||||||
|
<</Size 13 /ID [(ZV$vûñR:¨sE\) ù) (P…õrˆ‚ Åð¡Ï) ] /Root 1 0 R /Prev 2008 /Info 2 0 R >>
|
||||||
|
startxref
|
||||||
|
2802
|
||||||
Reference in New Issue
Block a user