feat(search): PDF content via own ES index and oo index (#42)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go stable) (pull_request) Successful in 18s
Tests / Test (Go 1.25) (pull_request) Successful in 18s

This commit is contained in:
2026-09-16 17:20:54 +00:00
parent 72c7cd5173
commit 7c3a0b8b6e
12 changed files with 1412 additions and 8 deletions
+4 -2
View File
@@ -188,6 +188,7 @@ type esBool struct {
type esClause struct {
MultiMatch *esMultiMatch `json:"multi_match,omitempty"`
Term map[string]any `json:"term,omitempty"`
Terms map[string]any `json:"terms,omitempty"`
Wildcard map[string]any `json:"wildcard,omitempty"`
}
@@ -268,9 +269,10 @@ func parseESSearchResponse(raw []byte) ([]SearchHit, error) {
var esHighlightTag = regexp.MustCompile(`</?em[^>]*>`)
// esHighlightText flattens a highlight map into one plain-text snippet,
// preferring the content fragment over the title.
// preferring the content fragment over the title. It covers both the
// OnlyOffice content field and the own-index "content" field.
func esHighlightText(hl map[string][]string) string {
for _, key := range []string{"document.attachment.content", "title"} {
for _, key := range []string{"document.attachment.content", "content", "title"} {
frags := hl[key]
if len(frags) == 0 {
continue