diff --git a/.env.example b/.env.example index bfcff31..d5f5b5e 100644 --- a/.env.example +++ b/.env.example @@ -34,3 +34,10 @@ ONLYOFFICE_PROJECT_ID=33 # MINIO_BUCKET=office # MINIO_ACCESS_KEY= # MINIO_SECRET_KEY= + +# oo search — direct Elasticsearch access for name + content search. ES lives +# inside the OnlyOffice VM on localhost:9200; expose it with an SSH tunnel +# (see docs/elasticsearch.md). ONLYOFFICE_ES_INDEX defaults to files_file. +# ONLYOFFICE_ES_URL=http://127.0.0.1:9200 +# ONLYOFFICE_ES_INDEX=files_file +# ONLYOFFICE_TENANT= diff --git a/README.md b/README.md index 50870c4..1458a5f 100644 --- a/README.md +++ b/README.md @@ -678,6 +678,25 @@ oo dav download 22881 --to ./copy.pdf # default path: ./ oo dav fileops # active move/copy operations (status polling) ``` +### Search (`oo search`) + +Full-text search over the Documents index. The REST endpoint +`/api/2.0/files/@search/{query}` only searches file names in the database, so +`oo search` talks to the OnlyOffice **Elasticsearch** directly (index +`files_file`). Name search is default; `--content` also matches extracted +document text (`document.attachment.content`, Office formats only). +See [`docs/elasticsearch.md`](docs/elasticsearch.md) for the tunnel setup. + +```bash +oo search "Rechnung" # names only +oo search "Mahngebühr" --content # names + document text +oo search "Rechnung" --folder 649 --limit 50 +oo search "Rechnung" --json # shorthand for -o json +``` + +Requires `ONLYOFFICE_ES_URL` (plus optional `ONLYOFFICE_ES_INDEX`, +`ONLYOFFICE_TENANT`). + ### Bulk tools (`cmd/`) Small single-purpose binaries for bulk Documents work. All of them pace @@ -714,6 +733,7 @@ kontolink IN.xlsx oo-index.tsv OUT.xlsx [FILE_ID] [AMOUNTS_TSV] | `docs` | `tools`, `convert`, `optimize`, `ocr`, `hocr`, `as-md`, `put-md`, `put-txt`, `put-xlsx` | | `catalog` | `match`, `merge`, `apply`, `scan-contacts`, `scan-projects`, `scan-thunderbird` | | `dav` | `ls`, `move`, `copy`, `mkdir`, `rename-file`, `rename-folder`, `download`, `fileops` | +| `search` | `QUERY` (`--content`, `--folder ID`, `--limit N`, `--json`) | The CLI reads only `.env` from the current working directory (godotenv is a CLI-only concern — the library itself never loads dotfiles). diff --git a/cmd/oo/main.go b/cmd/oo/main.go index c91baaa..82ab8e8 100644 --- a/cmd/oo/main.go +++ b/cmd/oo/main.go @@ -18,6 +18,7 @@ // oo docs tools | convert | optimize | ocr | hocr | as-md | put-md | put-txt | put-xlsx // oo catalog match | merge | apply | scan-contacts | scan-projects | scan-thunderbird // oo dav ls | move | copy | mkdir | rename-file | rename-folder | download | fileops +// oo search QUERY [--content] [--folder ID] [--limit N] [--json] // // CRM association rules: docs/crm-associations.md // diff --git a/cmd/oo/search.go b/cmd/oo/search.go new file mode 100644 index 0000000..2ceab0e --- /dev/null +++ b/cmd/oo/search.go @@ -0,0 +1,72 @@ +package main + +import ( + onlyoffice "github.com/eslider/go-onlyoffice" + "github.com/spf13/cobra" +) + +func init() { + rootCmd.AddCommand(searchCmd()) +} + +// searchCmd queries the OnlyOffice Elasticsearch index directly. The REST +// /api/2.0/files/@search endpoint only searches file names in the database; +// content search needs ES (see docs/elasticsearch.md). +func searchCmd() *cobra.Command { + var ( + content bool + folder string + limit int + asJSON bool + ) + cmd := &cobra.Command{ + Use: "search QUERY", + Short: "Full-text search over documents by name, optionally by content (Elasticsearch)", + Long: "Search the OnlyOffice Documents index.\n\n" + + "By default only file names are matched. With --content the query also\n" + + "matches extracted document text (document.attachment.content); this covers\n" + + "Office formats (docx/xlsx/pptx) and is slower.\n\n" + + "Requires ONLYOFFICE_ES_URL (and optionally ONLYOFFICE_ES_INDEX,\n" + + "ONLYOFFICE_TENANT). See docs/elasticsearch.md for the tunnel setup.", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + if asJSON { + outputFormat = "json" + } + es, err := onlyoffice.NewESSearcher(onlyoffice.ESConfigFromEnv()) + if err != nil { + return err + } + hits, err := es.Search(cmd.Context(), onlyoffice.SearchQuery{ + Text: args[0], + InContent: content, + FolderID: folder, + Limit: limit, + }) + if err != nil { + return err + } + rows := make([]map[string]any, 0, len(hits)) + for _, h := range hits { + rows = append(rows, map[string]any{ + "id": h.ID, + "title": h.Title, + "folder": h.ParentID, + "score": h.Score, + "highlight": h.Highlight, + }) + } + if outputFormat == "json" { + printJSON(rows) + return nil + } + printTable([]string{"id", "title", "folder", "score", "highlight"}, rows) + return nil + }, + } + cmd.Flags().BoolVar(&content, "content", false, "also match extracted document content") + cmd.Flags().StringVar(&folder, "folder", "", "limit to a Documents folder id") + cmd.Flags().IntVar(&limit, "limit", 20, "maximum number of results") + cmd.Flags().BoolVar(&asJSON, "json", false, "shorthand for --output json") + return cmd +} diff --git a/cmd/oo/search_test.go b/cmd/oo/search_test.go new file mode 100644 index 0000000..ea1716c --- /dev/null +++ b/cmd/oo/search_test.go @@ -0,0 +1,46 @@ +package main + +import ( + "bytes" + "strings" + "testing" +) + +func TestSearchCommandRegisteredWithFlags(t *testing.T) { + cmd, _, err := rootCmd.Find([]string{"search"}) + if err != nil { + t.Fatal(err) + } + if cmd.Name() != "search" { + t.Fatalf("search resolved to %q", cmd.Name()) + } + for _, name := range []string{"content", "folder", "limit", "json"} { + if cmd.Flags().Lookup(name) == nil { + t.Errorf("search: missing --%s flag", name) + } + } + if got := cmd.Flags().Lookup("limit").DefValue; got != "20" { + t.Errorf("--limit default = %q, want 20", got) + } +} + +func TestSearchWithoutESURLIsClearError(t *testing.T) { + clearEnv(t, "ONLYOFFICE_ES_URL", "ONLYOFFICE_ES_INDEX", "ONLYOFFICE_TENANT") + errBuf := &bytes.Buffer{} + rootCmd.SetErr(errBuf) + rootCmd.SetOut(&bytes.Buffer{}) + rootCmd.SetArgs([]string{"search", "Rechnung"}) + t.Cleanup(func() { + rootCmd.SetArgs(nil) + rootCmd.SetOut(nil) + rootCmd.SetErr(nil) + }) + + err := rootCmd.Execute() + if err == nil { + t.Fatal("expected error without ONLYOFFICE_ES_URL") + } + if !strings.Contains(err.Error(), "ONLYOFFICE_ES_URL") { + t.Fatalf("error %q missing ONLYOFFICE_ES_URL", err.Error()) + } +} diff --git a/docs/elasticsearch.md b/docs/elasticsearch.md new file mode 100644 index 0000000..05df5d6 --- /dev/null +++ b/docs/elasticsearch.md @@ -0,0 +1,125 @@ +--- +type: reference +status: current +related: + - README.md + - file_es.go +--- + +# Elasticsearch — полнотекстовый поиск OnlyOffice + +## Что это + +Полнотекстовый поиск OnlyOffice Workspace работает на **Elasticsearch**. +Клиент на сервере — NEST. Индекс — имя таблицы. + +Для файлов индекс `files_file`: + +| поле | тип | смысл | +|------|-----|-------| +| `id` | integer | id файла (тот же, что в REST/Documents) | +| `title` | text (`whitespacecustom`) | имя файла | +| `tenantId` | integer | тенант (портал) | +| `folders` | nested | список папок: `folderId` (строка), `id`, `tenantId` | +| `document.attachment.content` | text (`document`) | извлеченный текст (ingest-attachment) | +| `document.attachment.content_type` | text | MIME | + +Важно: +- Живой сервер — **Elasticsearch 7.16.3**, кластер `elasticsearch`. +- REST `GET /api/2.0/files/@search/{query}` ищет **только по имени в БД** + (`fileDao.Search`), ES не задействует. Для поиска по содержимому нужен + прямой ES — это и делает `oo search`. +- `title` analyzer `whitespacecustom` режет по пробелам и lower-case. Полное + имя файла — один токен (`Rechnung-4711.pdf`), поэтому поиск по имени ищет + слово целиком, а не подстроку. +- `document.attachment.content` заполняется **только для Office-форматов** + (docx / xlsx / pptx). У PDF/txt, залитых через API, контент не извлекается. +- Индексация асинхронная (TeamLabSvc) — файл появляется в ES не мгновенно. + +## Доступ + +ES слушает `127.0.0.1:9200` **внутри** VM OnlyOffice. Снаружи порт закрыт, +SSH в VM открыт на хосте как `127.0.0.1:32` (контейнер `onlyoffice-v2`, +QEMU). Схема — SSH-туннель. + +```bash +# из корня go-onlyoffice (ключ и хост — как в infra-доках) +ssh -f -N -o ControlMaster=no -o ControlPath=none \ + -p 32 -i ~/.ssh/id_ed25519 \ + -L 9200:127.0.0.1:9200 root@127.0.0.1 + +curl -s http://127.0.0.1:9200/ | head # tagline + version +curl -s 'http://127.0.0.1:9200/_cat/indices?h=index,docs.count' +``` + +`-o ControlMaster=no -o ControlPath=none` обязательны: иначе forward уходит +в persistent master-соединение из `~/.ssh/config` и порт остаётся занят. + +Проверить, что туннель жив: + +```bash +curl -s http://127.0.0.1:9200/files_file/_count +``` + +## Переменные + +| env | default | смысл | +|-----|---------|-------| +| `ONLYOFFICE_ES_URL` | — (обязателен) | `scheme://host:port` ES | +| `ONLYOFFICE_ES_INDEX` | `files_file` | индекс | +| `ONLYOFFICE_TENANT` | пусто (все) | фильтр `tenantId` | + +Имена — в [`.env.example`](../.env.example). Секретов нет: ES без пароля. + +## CLI + +```bash +ONLYOFFICE_ES_URL=http://127.0.0.1:9200 oo search "Rechnung" +ONLYOFFICE_ES_URL=http://127.0.0.1:9200 oo search "Mahngebühr" --content +oo search "Rechnung" --folder 649 --limit 50 --json +``` + +Флаги: `--content` (искать и по тексту), `--folder ID` (папка +`folders.folderId`), `--limit N` (по умолчанию 20, максимум 200), +`--json` = `-o json`. + +## Библиотека + +`file_es.go` — `ESSearcher` (`Name() = "elasticsearch"`), прямой ES REST на +stdlib `net/http`: + +```go +es, _ := onlyoffice.NewESSearcher(onlyoffice.ESConfigFromEnv()) +hits, _ := es.Search(ctx, onlyoffice.SearchQuery{ + Text: "Rechnung", InContent: true, Limit: 20, +}) +``` + +Запрос: `multi_match` по `title^2` (+ `document.attachment.content` при +`InContent`), фильтры `tenantId` и `folders.folderId`, `_source` +id/title/folders, `highlight` для фрагмента. Ответ → `[]SearchHit` (модель из +эпика #34; пока объявлена в `file_es.go`, переедет в `file_core.go` с F1 #35). + +## Тесты + +```bash +# unit — чистые builders/парсеры, без сети +go test ./ -run ES + +# integration — нужен ONLYOFFICE_ES_URL (+ креды REST для залива) +set -a; . .env; set +a +ONLYOFFICE_ES_URL=http://127.0.0.1:9200 ONLYOFFICE_TENANT=1 \ + go test -tags=integration -run TestIntegrationESSearch -v . +``` + +Интеграционный тест заливает временный xlsx (в имени и в ячейке — уникальные +токены), ждёт индексации, проверяет поиск по имени и по содержимому, затем +удаляет проект. + +## Грабли + +- `locale`/версия ES: 7.16.3, `_search` совместим с REST 7.x. +- ES без auth и слушает только localhost — туннель обязателен. +- Фильтр `tenantId` сузит выдачу; без него видны документы всех тенантов. +- Поиск по содержимому PDF, залитых через API, не работает (нет + `attachment.content`) — только Office-форматы. diff --git a/file_es.go b/file_es.go new file mode 100644 index 0000000..9c2db76 --- /dev/null +++ b/file_es.go @@ -0,0 +1,285 @@ +package onlyoffice + +// Elasticsearch backend of the unified file client (epic #34, F3 #37). +// +// OnlyOffice full-text search runs on Elasticsearch (index `files_file`, NEST +// client on the server). The REST endpoint GET /api/2.0/files/@search/{query} +// only searches file names in the database, so content search needs a direct +// ES query. The live server is Elasticsearch 7.16.3; the request shape below +// is plain REST and stays stdlib-only, matching the repo's no-extra-deps rule. + +import ( + "bytes" + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "os" + "regexp" + "strconv" + "strings" + "time" +) + +// The canonical model (Kind, Entry, SearchQuery, SearchHit, Searcher) lives in +// file_core.go (F1 #35). + +const ( + defaultESIndex = "files_file" + defaultESLimit = 20 + maxESLimit = 200 + maxESResponseSize = 8 << 20 +) + +// ESConfig configures the direct Elasticsearch searcher. +type ESConfig struct { + URL string // scheme://host:port of the ES HTTP endpoint + Index string // index name, default files_file + Tenant string // tenantId filter, empty means all tenants +} + +// ESConfigFromEnv reads ONLYOFFICE_ES_URL, ONLYOFFICE_ES_INDEX (default +// files_file) and ONLYOFFICE_TENANT. The library never loads dotfiles — the +// CLI does that. +func ESConfigFromEnv() ESConfig { + return ESConfig{ + URL: strings.TrimRight(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")), "/"), + Index: firstNonEmpty(os.Getenv("ONLYOFFICE_ES_INDEX"), defaultESIndex), + Tenant: strings.TrimSpace(os.Getenv("ONLYOFFICE_TENANT")), + } +} + +// ESSearcher queries OnlyOffice's Elasticsearch index directly for file name +// and document content. +type ESSearcher struct { + cfg ESConfig + http *http.Client +} + +// NewESSearcher returns a searcher for the OnlyOffice Elasticsearch index. +// The URL is required; an empty index falls back to files_file. +func NewESSearcher(cfg ESConfig) (*ESSearcher, error) { + if strings.TrimSpace(cfg.URL) == "" { + return nil, fmt.Errorf("onlyoffice: elasticsearch URL is empty (set ONLYOFFICE_ES_URL)") + } + cfg.URL = strings.TrimRight(cfg.URL, "/") + if cfg.Index == "" { + cfg.Index = defaultESIndex + } + return &ESSearcher{cfg: cfg, http: &http.Client{Timeout: 30 * time.Second}}, nil +} + +// Name implements Searcher. +func (s *ESSearcher) Name() string { return "elasticsearch" } + +// Search runs a multi_match over title (and, when q.InContent is set, +// document.attachment.content), filtered by tenant and optional folder. +func (s *ESSearcher) Search(ctx context.Context, q SearchQuery) ([]SearchHit, error) { + q.Text = strings.TrimSpace(q.Text) + if q.Text == "" { + return nil, fmt.Errorf("onlyoffice: empty search query") + } + body, err := json.Marshal(esSearchRequest(q, s.cfg.Tenant)) + if err != nil { + return nil, fmt.Errorf("onlyoffice: build elasticsearch query: %w", err) + } + endpoint := s.cfg.URL + "/" + s.cfg.Index + "/_search" + req, err := http.NewRequestWithContext(ctx, http.MethodPost, endpoint, bytes.NewReader(body)) + if err != nil { + return nil, err + } + req.Header.Set("Content-Type", "application/json") + req.Header.Set("Accept", "application/json") + resp, err := s.http.Do(req) + if err != nil { + return nil, fmt.Errorf("onlyoffice: elasticsearch search: %w", err) + } + defer resp.Body.Close() + raw, err := io.ReadAll(io.LimitReader(resp.Body, maxESResponseSize)) + if err != nil { + return nil, err + } + if resp.StatusCode >= 400 { + return nil, fmt.Errorf("onlyoffice: elasticsearch search: %d %s", resp.StatusCode, truncate(string(raw), 400)) + } + return parseESSearchResponse(raw) +} + +// esSearchRequest builds the ES query body. Pure, so it is unit-tested. +func esSearchRequest(q SearchQuery, tenant string) esRequest { + limit := q.Limit + if limit <= 0 { + limit = defaultESLimit + } + if limit > maxESLimit { + limit = maxESLimit + } + fields := []string{"title^2"} + if q.InContent { + fields = append(fields, "document.attachment.content") + } + must := []esClause{{MultiMatch: &esMultiMatch{Query: q.Text, Fields: fields}}} + + var filter []esClause + if t := strings.TrimSpace(tenant); t != "" { + filter = append(filter, esClause{Term: map[string]any{"tenantId": numericOrString(t)}}) + } + if f := strings.TrimSpace(q.FolderID); f != "" { + filter = append(filter, esClause{Term: map[string]any{"folders.folderId": f}}) + } + for _, ext := range normalizeExtensions(q.Extensions) { + filter = append(filter, esClause{Wildcard: map[string]any{"title": "*." + ext}}) + } + + highlightFields := map[string]struct{}{"title": {}} + if q.InContent { + highlightFields["document.attachment.content"] = struct{}{} + } + return esRequest{ + Size: limit, + Source: []string{"id", "title", "folders"}, + Query: esQuery{Bool: esBool{Must: must, Filter: filter}}, + Highlight: esHighlight{PreTags: []string{""}, PostTags: []string{""}, Fields: highlightFields}, + } +} + +// normalizeExtensions lowercases, trims leading dots and drops empties. +func normalizeExtensions(exts []string) []string { + out := make([]string, 0, len(exts)) + seen := map[string]bool{} + for _, e := range exts { + e = strings.ToLower(strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(e), "."))) + if e == "" || seen[e] { + continue + } + seen[e] = true + out = append(out, e) + } + return out +} + +// numericOrString keeps an integer-looking filter value numeric (tenantId is +// a long) and leaves anything else as a string (folderId is a text token). +func numericOrString(s string) any { + if n, err := strconv.ParseInt(s, 10, 64); err == nil { + return n + } + return s +} + +// esRequest is the subset of the ES query DSL this client emits. +type esRequest struct { + Size int `json:"size"` + Source []string `json:"_source"` + Query esQuery `json:"query"` + Highlight esHighlight `json:"highlight"` +} + +type esQuery struct { + Bool esBool `json:"bool"` +} + +type esBool struct { + Must []esClause `json:"must,omitempty"` + Filter []esClause `json:"filter,omitempty"` +} + +type esClause struct { + MultiMatch *esMultiMatch `json:"multi_match,omitempty"` + Term map[string]any `json:"term,omitempty"` + Wildcard map[string]any `json:"wildcard,omitempty"` +} + +type esMultiMatch struct { + Query string `json:"query"` + Fields []string `json:"fields"` +} + +type esHighlight struct { + PreTags []string `json:"pre_tags,omitempty"` + PostTags []string `json:"post_tags,omitempty"` + Fields map[string]struct{} `json:"fields"` +} + +// esResponse is the subset of an ES search response we consume. +type esResponse struct { + Took int `json:"took"` + Hits struct { + Total struct { + Value int `json:"value"` + Relation string `json:"relation"` + } `json:"total"` + Hits []esResponseHit `json:"hits"` + } `json:"hits"` +} + +type esResponseHit struct { + ID string `json:"_id"` + Score float64 `json:"_score"` + Source struct { + ID int `json:"id"` + Title string `json:"title"` + Folders []struct { + FolderID string `json:"folderId"` + ID int `json:"id"` + } `json:"folders"` + } `json:"_source"` + Highlight map[string][]string `json:"highlight"` +} + +// parseESSearchResponse converts an ES search response into SearchHit values. +// Pure, so it is unit-tested. +func parseESSearchResponse(raw []byte) ([]SearchHit, error) { + var r esResponse + if err := json.Unmarshal(raw, &r); err != nil { + return nil, fmt.Errorf("onlyoffice: decode elasticsearch response: %w", err) + } + hits := make([]SearchHit, 0, len(r.Hits.Hits)) + for _, h := range r.Hits.Hits { + id := strconv.Itoa(h.Source.ID) + if h.Source.ID == 0 { + id = h.ID + } + var parent string + path := make([]string, 0, len(h.Source.Folders)) + for i, f := range h.Source.Folders { + path = append(path, f.FolderID) + if i == 0 { + parent = f.FolderID + } + } + hits = append(hits, SearchHit{ + Entry: Entry{ + ID: id, + ParentID: parent, + Title: h.Source.Title, + Kind: File, + Provider: "elasticsearch", + }, + Score: h.Score, + Highlight: esHighlightText(h.Highlight), + Path: path, + }) + } + return hits, nil +} + +var esHighlightTag = regexp.MustCompile(`]*>`) + +// esHighlightText flattens a highlight map into one plain-text snippet, +// preferring the content fragment over the title. +func esHighlightText(hl map[string][]string) string { + for _, key := range []string{"document.attachment.content", "title"} { + frags := hl[key] + if len(frags) == 0 { + continue + } + clean := make([]string, 0, len(frags)) + for _, f := range frags { + clean = append(clean, esHighlightTag.ReplaceAllString(f, "")) + } + return strings.Join(clean, " … ") + } + return "" +} diff --git a/file_es_integration_test.go b/file_es_integration_test.go new file mode 100644 index 0000000..ce4e891 --- /dev/null +++ b/file_es_integration_test.go @@ -0,0 +1,135 @@ +//go:build integration + +package onlyoffice + +import ( + "context" + "os" + "path/filepath" + "strconv" + "strings" + "testing" + "time" + + "github.com/xuri/excelize/v2" +) + +// TestIntegrationESSearch uploads a throwaway workbook and verifies that the +// direct Elasticsearch search finds it by file name and by content. +// +// The content index (document.attachment.content) is only populated for Office +// formats (docx/xlsx/pptx), so the fixture is an xlsx whose cell carries a +// unique token. Requires ONLYOFFICE_ES_URL (a reachable ES endpoint — in the +// current setup a tunnel to the ES inside the OnlyOffice VM, see +// docs/elasticsearch.md) plus the regular REST credentials for the upload. +// Skips when either is missing. +func TestIntegrationESSearch(t *testing.T) { + esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")) + if esURL == "" { + t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test") + } + c := liveClient(t) + t.Cleanup(func() { cleanupTestProjects(t, c) }) + + stamp := time.Now().UTC().Format("20060102-150405") + nameToken := "goesname" + stamp + contentToken := "goescontent" + stamp + + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute) + defer cancel() + + project, err := c.CreateProject(NewProjectRequest{ + Title: testProjectPrefix + "es-" + stamp, + Description: "go-onlyoffice elasticsearch integration", + }) + if err != nil { + t.Fatalf("CreateProject: %v", err) + } + if project.ID == nil { + t.Fatal("created project without id") + } + pid := strconv.Itoa(*project.ID) + + title := nameToken + ".xlsx" + localPath := filepath.Join(t.TempDir(), title) + book := excelize.NewFile() + if err := book.SetCellValue("Sheet1", "A1", "OnlyOffice Elasticsearch content fixture "+contentToken); err != nil { + t.Fatalf("SetCellValue: %v", err) + } + if err := book.SaveAs(localPath); err != nil { + t.Fatalf("SaveAs: %v", err) + } + + entry, err := c.UploadProjectFile(ctx, pid, localPath) + if err != nil { + t.Fatalf("UploadProjectFile: %v", err) + } + fileID := strconv.Itoa(int(FileEntryNumericID(entry))) + if fileID == "0" { + t.Fatalf("upload returned no file id: %+v", entry) + } + + es, err := NewESSearcher(ESConfig{ + URL: esURL, + Index: os.Getenv("ONLYOFFICE_ES_INDEX"), + Tenant: os.Getenv("ONLYOFFICE_TENANT"), + }) + if err != nil { + t.Fatalf("NewESSearcher: %v", err) + } + + // Indexing is asynchronous on the server; poll until the file shows up. + // The server's title analyzer splits on whitespace, so the name query is + // the full file name token (including extension), as a user would type it. + nameHit := waitForHit(t, ctx, es, SearchQuery{Text: title}, fileID) + if nameHit.Title != title { + t.Errorf("name hit title = %q, want %q", nameHit.Title, title) + } + contentHit := waitForHit(t, ctx, es, SearchQuery{Text: contentToken, InContent: true}, fileID) + if contentHit.Highlight == "" { + t.Error("content hit has no highlight fragment") + } + if !strings.Contains(contentHit.Title, nameToken) { + t.Errorf("content hit title = %q, want the uploaded workbook", contentHit.Title) + } + + // The content token is absent from the title, so a name-only search must + // not return the file — this proves the content field is really queried. + if hits := searchQuiet(t, es, SearchQuery{Text: contentToken}); len(hits) != 0 { + t.Errorf("name-only search for content token returned %d hits, want 0", len(hits)) + } +} + +// waitForHit polls ES until the file with fileID appears and returns that hit. +func waitForHit(t *testing.T, ctx context.Context, s *ESSearcher, q SearchQuery, fileID string) SearchHit { + t.Helper() + var lastErr error + for { + hits, err := s.Search(ctx, q) + if err != nil { + lastErr = err + } else { + for _, h := range hits { + if h.ID == fileID { + return h + } + } + } + select { + case <-ctx.Done(): + t.Fatalf("search %q: file %s not indexed in time (last err: %v)", q.Text, fileID, lastErr) + case <-time.After(3 * time.Second): + } + } +} + +func searchQuiet(t *testing.T, s *ESSearcher, q SearchQuery) []SearchHit { + t.Helper() + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + hits, err := s.Search(ctx, q) + if err != nil { + t.Fatalf("Search(%q): %v", q.Text, err) + } + return hits +} diff --git a/file_es_test.go b/file_es_test.go new file mode 100644 index 0000000..ee60ef1 --- /dev/null +++ b/file_es_test.go @@ -0,0 +1,188 @@ +package onlyoffice + +import ( + "encoding/json" + "reflect" + "testing" +) + +func TestESSearchRequestNameOnly(t *testing.T) { + got := esSearchRequest(SearchQuery{Text: "Rechnung"}, "1") + if got.Size != defaultESLimit { + t.Errorf("size = %d, want %d", got.Size, defaultESLimit) + } + if !reflect.DeepEqual(got.Source, []string{"id", "title", "folders"}) { + t.Errorf("_source = %v", got.Source) + } + if len(got.Query.Bool.Must) != 1 || got.Query.Bool.Must[0].MultiMatch == nil { + t.Fatalf("must = %+v, want one multi_match", got.Query.Bool.Must) + } + mm := got.Query.Bool.Must[0].MultiMatch + if mm.Query != "Rechnung" { + t.Errorf("query = %q", mm.Query) + } + if !reflect.DeepEqual(mm.Fields, []string{"title^2"}) { + t.Errorf("fields = %v, want title only", mm.Fields) + } + if _, ok := got.Highlight.Fields["document.attachment.content"]; ok { + t.Error("content highlight present without InContent") + } + if _, ok := got.Highlight.Fields["title"]; !ok { + t.Error("title highlight missing") + } + if len(got.Query.Bool.Filter) != 1 || got.Query.Bool.Filter[0].Term["tenantId"] != int64(1) { + t.Errorf("tenant filter = %+v, want numeric tenantId=1", got.Query.Bool.Filter) + } +} + +func TestESSearchRequestContentFields(t *testing.T) { + got := esSearchRequest(SearchQuery{Text: "Mahnung", InContent: true}, "") + mm := got.Query.Bool.Must[0].MultiMatch + want := []string{"title^2", "document.attachment.content"} + if !reflect.DeepEqual(mm.Fields, want) { + t.Errorf("fields = %v, want %v", mm.Fields, want) + } + if _, ok := got.Highlight.Fields["document.attachment.content"]; !ok { + t.Error("content highlight missing with InContent") + } + if len(got.Query.Bool.Filter) != 0 { + t.Errorf("filter = %+v, want none without tenant/folder", got.Query.Bool.Filter) + } +} + +func TestESSearchRequestFiltersAndLimit(t *testing.T) { + got := esSearchRequest(SearchQuery{ + Text: "Storchen", + FolderID: "649", + Extensions: []string{".PDF", "pdf", "docx"}, + Limit: 999, + }, "42") + if got.Size != maxESLimit { + t.Errorf("size = %d, want cap %d", got.Size, maxESLimit) + } + var tenant, folder, wildcards int + for _, f := range got.Query.Bool.Filter { + switch { + case f.Term != nil && f.Term["tenantId"] != nil: + tenant++ + case f.Term != nil && f.Term["folders.folderId"] != nil: + folder++ + if f.Term["folders.folderId"] != "649" { + t.Errorf("folder filter = %+v", f.Term) + } + case f.Wildcard != nil: + wildcards++ + } + } + if tenant != 1 || folder != 1 { + t.Errorf("term filters tenant=%d folder=%d, want 1 each", tenant, folder) + } + if wildcards != 2 { + t.Errorf("wildcard filters = %d, want deduped PDF+docx", wildcards) + } +} + +func TestESSearchRequestRejectsEmptyTextAtSearch(t *testing.T) { + s, err := NewESSearcher(ESConfig{URL: "http://localhost:9200"}) + if err != nil { + t.Fatalf("NewESSearcher: %v", err) + } + if _, err := s.Search(t.Context(), SearchQuery{Text: " "}); err == nil { + t.Error("empty query: want error") + } +} + +func TestNewESSearcherRequiresURL(t *testing.T) { + if _, err := NewESSearcher(ESConfig{}); err == nil { + t.Error("empty URL: want error") + } + s, err := NewESSearcher(ESConfig{URL: "http://es:9200/"}) + if err != nil { + t.Fatalf("NewESSearcher: %v", err) + } + if s.cfg.Index != defaultESIndex { + t.Errorf("index = %q, want %q", s.cfg.Index, defaultESIndex) + } + if s.cfg.URL != "http://es:9200" { + t.Errorf("url = %q, want trimmed", s.cfg.URL) + } + if s.Name() != "elasticsearch" { + t.Errorf("Name() = %q", s.Name()) + } +} + +func TestNormalizeExtensions(t *testing.T) { + got := normalizeExtensions([]string{" .PDF ", "pdf", "", "xlsx"}) + want := []string{"pdf", "xlsx"} + if !reflect.DeepEqual(got, want) { + t.Errorf("normalizeExtensions = %v, want %v", got, want) + } +} + +func TestParseESSearchResponse(t *testing.T) { + raw := []byte(`{ + "took": 12, + "hits": { + "total": {"value": 2, "relation": "eq"}, + "hits": [ + { + "_id": "2395", + "_score": 7.31, + "_source": {"id": 2395, "title": "Rechnung-4711.pdf", + "folders": [{"folderId": "438", "id": 0}, {"folderId": "11", "id": 0}]}, + "highlight": { + "title": ["Rechnung-4711.pdf"], + "document.attachment.content": ["… Zahlung der Rechnung …"] + } + }, + { + "_id": "2318", + "_score": 6.02, + "_source": {"id": 2318, "title": "Mahnung.pdf", "folders": []}, + "highlight": {"title": ["Mahnung.pdf"]} + } + ] + } + }`) + hits, err := parseESSearchResponse(raw) + if err != nil { + t.Fatalf("parseESSearchResponse: %v", err) + } + if len(hits) != 2 { + t.Fatalf("hits = %d, want 2", len(hits)) + } + h0 := hits[0] + if h0.ID != "2395" || h0.Title != "Rechnung-4711.pdf" || h0.Kind != File { + t.Errorf("hit0 entry = %+v", h0.Entry) + } + if h0.ParentID != "438" || !reflect.DeepEqual(h0.Path, []string{"438", "11"}) { + t.Errorf("hit0 path = %v parent = %q", h0.Path, h0.ParentID) + } + if h0.Score != 7.31 { + t.Errorf("hit0 score = %v", h0.Score) + } + if h0.Highlight != "… Zahlung der Rechnung …" { + t.Errorf("hit0 highlight = %q, want content fragment", h0.Highlight) + } + if hits[1].Highlight != "Mahnung.pdf" { + t.Errorf("hit1 highlight = %q, want title without tags", hits[1].Highlight) + } + if hits[1].ParentID != "" || len(hits[1].Path) != 0 { + t.Errorf("hit1 path = %v", hits[1].Path) + } +} + +func TestESSearchRequestJSONShape(t *testing.T) { + got := esSearchRequest(SearchQuery{Text: "Rechnung", InContent: true}, "1") + b, err := json.Marshal(got) + if err != nil { + t.Fatalf("marshal: %v", err) + } + var back map[string]any + if err := json.Unmarshal(b, &back); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if _, ok := back["query"].(map[string]any)["bool"]; !ok { + t.Errorf("query.bool missing: %s", b) + } +}