Compare commits

...
23 Commits
Author SHA1 Message Date
Yftyr b3dfff1a4c refactor(search): use canonical model from file_core.go (#37) 2026-09-16 16:43:20 +00:00
eSlider b5a61ad420 feat(search): Elasticsearch searcher (name+content) and oo search (#37) 2026-09-16 16:42:57 +00:00
eSlider 1aba545e1d Merge pull request 'feat(files): канонический Entry + FileStore (REST/DAV) (#35)' (#40) from feat/file-store#35 into main
Release / GoReleaser (push) Skipped
Release Please / Release Please (push) Skipped
Tests / Secret scan (gitleaks) (push) Successful in 3s
Tests / Test (Go stable) (push) Successful in 18s
Tests / Test (Go 1.25) (push) Successful in 18s
2026-09-16 17:42:49 +01:00
eSlider 94951fc6c5 feat(files): canonical Entry + FileStore REST/DAV adapters (#35)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go 1.25) (pull_request) Successful in 18s
Tests / Test (Go stable) (pull_request) Successful in 18s
2026-09-16 16:30:07 +00:00
eSlider d95985e6d0 Merge pull request 'fix(pdfamount): ставка НДС не сумма; итог DKV (#29)' (#33) from fix/pdfamount-tax#29 into main
Release / GoReleaser (push) Skipped
Release Please / Release Please (push) Skipped
Tests / Secret scan (gitleaks) (push) Successful in 5s
Tests / Test (Go 1.25) (push) Successful in 19s
Tests / Test (Go stable) (push) Successful in 20s
2026-09-15 12:58:30 +01:00
eSlider 57b383d579 fix(pdfamount): не считать ставку НДС суммой; итог DKV (#29)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go 1.25) (pull_request) Successful in 20s
Tests / Test (Go stable) (pull_request) Successful in 23s
- Строки с %/MwSt/USt/Prozent/Steuer больше не дают сумму (было: Storchen
  19,00 % -> amount 19.00 и ложная отбраковка кандидатов в матчере).
- Приоритет меток: zu zahlender betrag > rechnungsbetrag > rechnungsendbetrag
  > gesamtbetrag > gesamtsumme (inkl. Steuern); low-priority не перебивает.
- DKV: секция Gesamtsummenaufstellung (значение после «»») важнее повторяющихся
  TOTAL-строк по машинам.
- Тесты: 19,00 % не сумма; fallback на usable-метку; DKV-итог.
2026-09-15 11:57:31 +00:00
eSlider a73f6c172d Merge pull request 'feat(oo): dav rm — удаление папок/файлов Documents (#30)' (#31) from feat/dav-rm#30 into main
Release / GoReleaser (push) Skipped
Release Please / Release Please (push) Skipped
Tests / Secret scan (gitleaks) (push) Successful in 7s
Tests / Test (Go stable) (push) Successful in 29s
Tests / Test (Go 1.25) (push) Successful in 42s
2026-09-15 09:44:02 +01:00
eSlider e49693ad2c feat(oo): dav rm — удаление папок/файлов Documents (#30)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 6s
Tests / Test (Go stable) (pull_request) Successful in 22s
Tests / Test (Go 1.25) (pull_request) Successful in 25s
DeleteDavItems был только в клиенте; добавлена команда
`oo dav rm [FILE_ID...] --folders F1,F2`.
2026-09-15 08:31:47 +00:00
eSlider b86d71211f merge: GitHub release-please v0.18+ into Gitea main
Release / GoReleaser (push) Skipped
Release Please / Release Please (push) Skipped
Tests / Secret scan (gitleaks) (push) Successful in 5s
Tests / Test (Go stable) (push) Successful in 23s
Tests / Test (Go 1.25) (push) Successful in 30s
2026-09-15 09:29:13 +01:00
eSlider a2b919478e Merge pull request 'fix(pdfamount): суммы DKV/Diashop и др. (метки и форматы) (#27)' (#28) from fix/pdfamount-amounts#27 into main
Release / GoReleaser (push) Skipped
Release Please / Release Please (push) Skipped
Tests / Secret scan (gitleaks) (push) Successful in 4s
Tests / Test (Go 1.25) (push) Successful in 18s
Tests / Test (Go stable) (push) Successful in 19s
2026-09-15 08:55:40 +01:00
eSlider ccabc11153 fix(pdfamount): widen amount labels and formats (#27)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go 1.25) (pull_request) Successful in 19s
Tests / Test (Go stable) (pull_request) Successful in 18s
Match more payable-amount labels by priority and accept DE/EN/plain
number formats, normalising to a dot-separated 2-decimal string. Adds
unit tests with synthetic DKV-style and Diashop-style lines.
2026-09-15 07:51:23 +00:00
eSlider 9a6da1a4f7 Merge pull request 'fix(files): UpdateFile uses PUT (#25)' (#26) from fix/update-file-put#25 into main
Release / GoReleaser (push) Skipped
Release Please / Release Please (push) Skipped
Tests / Secret scan (gitleaks) (push) Successful in 4s
Tests / Test (Go stable) (push) Successful in 18s
Tests / Test (Go 1.25) (push) Successful in 23s
2026-09-15 08:22:50 +01:00
eSlider 504d13ed08 fix(files): UpdateFile uses PUT /api/2.0/files/{id}/update (#25)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go stable) (pull_request) Successful in 18s
Tests / Test (Go 1.25) (pull_request) Successful in 20s
2026-09-15 07:19:42 +00:00
eSlider efc0864d42 Merge pull request 'docs(oo): dav, documents files api, bulk tools, fix verbs (#22)' (#24) from docs/oo-reference#22 into main
Release / GoReleaser (push) Skipped
Release Please / Release Please (push) Skipped
Tests / Secret scan (gitleaks) (push) Successful in 4s
Tests / Test (Go stable) (push) Successful in 19s
Tests / Test (Go 1.25) (push) Successful in 19s
Reviewed-on: #24
2026-09-14 22:44:25 +01:00
eSlider 59caf5e560 docs(oo): dav, documents files api, bulk tools, fix verbs (#22)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go 1.25) (pull_request) Successful in 17s
Tests / Test (Go stable) (pull_request) Successful in 18s
2026-09-14 21:43:37 +00:00
eSlider e7803c5269 Merge pull request 'feat(files): Documents Dav ops + UpdateFile, fileops errors (#152)' (#20) from feat/oo-automation#152 into main
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Successful in 4s
Tests / Test (Go stable) (push) Successful in 14s
Tests / Test (Go 1.25) (push) Successful in 1m6s
2026-09-14 22:35:09 +01:00
eSlider a08e7c49ad Merge branch 'main' into feat/oo-automation#152
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go stable) (pull_request) Successful in 1m49s
Tests / Test (Go 1.25) (pull_request) Successful in 2m20s
2026-09-14 22:31:44 +01:00
eSlider 4310aa7002 Merge pull request 'feat(files): MinIO fallback for stale S3 downloads (#152)' (#21) from feat/oo-minio-fallback#152 into main
Release / GoReleaser (push) Skipped
Release Please / Release Please (push) Skipped
Tests / Secret scan (gitleaks) (push) Successful in 12s
Tests / Test (Go 1.25) (push) Successful in 23s
Tests / Test (Go stable) (push) Successful in 2m19s
Reviewed-on: #21
2026-09-14 22:31:32 +01:00
eSlider 1a7a962183 Merge pull request 'feat(oo): kontoblatt bulk tools, oo dav, update, retry (#22)' (#23) from feat/kontoblatt-tools#22 into main
Release / GoReleaser (push) Skipped
Release Please / Release Please (push) Skipped
Tests / Secret scan (gitleaks) (push) Successful in 5s
Tests / Test (Go 1.25) (push) Successful in 16s
Tests / Test (Go stable) (push) Successful in 24s
Reviewed-on: #23
2026-09-14 22:31:22 +01:00
eSlider af8e3b053a feat(oo): kontoblatt bulk tools, oo dav, update, retry (#22)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go 1.25) (pull_request) Successful in 1m22s
Tests / Test (Go stable) (pull_request) Successful in 1m21s
2026-09-14 21:03:39 +00:00
eSlider 8ac777c031 feat(files): MinIO fallback for stale S3 downloads (#152)
Release Please / Release Please (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go stable) (pull_request) Successful in 1m15s
Tests / Test (Go 1.25) (pull_request) Successful in 1m22s
2026-09-14 15:41:39 +00:00
eSliderandGitHub 47a2c256f3 Merge pull request #48 from eSlider/release-please--branches--main--components--go-onlyoffice
chore(main): release 0.18.0
2026-09-14 09:18:11 +01:00
github-actions[bot]andGitHub c1881e1e61 chore(main): release 0.18.0 2026-09-04 14:47:27 +00:00
32 changed files with 3853 additions and 93 deletions
+16
View File
@@ -25,3 +25,19 @@ ONLYOFFICE_PROJECT_ID=33
# cmd/office TUI — optional Document Server for DOCX→HTML preview:
# ONLYOFFICE_DOCS_URL=https://docs.example.com
# ONLYOFFICE_DOCS_SECRET=
# MinIO download fallback for the portal's stale AWS S3 consumer (older
# Documents folders). When the portal redirects to amazonaws.com with access
# key "minio" (403 InvalidAccessKeyId), files are fetched from the local MinIO
# store instead. Without a key/secret the fallback is disabled.
# MINIO_ENDPOINT=http://192.168.188.10:9000
# MINIO_BUCKET=office
# MINIO_ACCESS_KEY=
# MINIO_SECRET_KEY=
# oo search — direct Elasticsearch access for name + content search. ES lives
# inside the OnlyOffice VM on localhost:9200; expose it with an SSH tunnel
# (see docs/elasticsearch.md). ONLYOFFICE_ES_INDEX defaults to files_file.
# ONLYOFFICE_ES_URL=http://127.0.0.1:9200
# ONLYOFFICE_ES_INDEX=files_file
# ONLYOFFICE_TENANT=
+1 -1
View File
@@ -1,3 +1,3 @@
{
".": "0.17.0"
".": "0.18.0"
}
+4 -3
View File
@@ -9,16 +9,17 @@ Canonical Go client for OnlyOffice Workspace (Projects + Calendar + CRM) and the
- `request.go` — `Request`, `Query`, `Time`, `Token`, `MetaResponse`, `Permissions`.
- `auth.go` — `Authenticate`, `AuthenticateContext`, `InvalidateToken`, `Auth`, token lifecycle.
- `http.go` — transport + DRY response decoders (`ResponseArray`/`ResponseObject`/`postFormObject`/`putFormObject`/`deleteObject`).
- `projects.go`, `tasks.go`, `users.go`, `calendar.go`, `crm.go`, `files.go`, `mails.go`, `invoices.go` — typed / untyped domain methods. **`files.go`** — CRM opportunity upload plus **project/task Documents**. **`mails.go`** — OnlyOffice Workspace Mail. **`invoices.go`** — CRM invoices, PDF regen/cleanup, status. Association rules: [`docs/crm-associations.md`](docs/crm-associations.md).
- `projects.go`, `tasks.go`, `users.go`, `calendar.go`, `crm.go`, `files.go`, `files_webdav.go`, `files_stem.go`, `retry.go`, `mails.go`, `invoices.go` — typed / untyped domain methods. **`files.go`** — CRM opportunity upload plus **project/task Documents** (`UpdateFile`, `UploadToFolderReplacing`). **`files_webdav.go`** — Documents module by id (`ListDavFolder`, `MoveDavItems`/`CopyDavItems` with per-operation error surfacing, `ListFileOps`). **`retry.go`** — `DoRetry`: deterministic linear backoff (no jitter) on 429/502/503/504; every bulk tool routes API calls through it. **`mails.go`** — OnlyOffice Workspace Mail. **`invoices.go`** — CRM invoices, PDF regen/cleanup, status. Association rules: [`docs/crm-associations.md`](docs/crm-associations.md).
- Pure stdlib + `google/go-querystring`; no UI, no dotenv.
- **CLI — `cmd/oo/` as `package main`.** Cobra wrapper that loads `.env` via `godotenv` at startup. **Subject-based command tree** mirroring [`tea`](https://gitea.com/gitea/tea):
- `main.go` — entry point (docstring lists the command tree).
- `common.go` — `rootCmd`, `newOO`, `printTable`/`printObject`, `--output table|json` flag.
- `calendar.go`, `projects.go`, `projects_files.go`, `tasks.go`, `tasks_files.go`, `users.go`, `contacts.go`, `opportunities.go`, `cases.go`, `crm_tasks.go`, `mails.go`, `invoices.go` — one file per subject (or per subject facet), each registers in `init()`.
- `calendar.go`, `projects.go`, `projects_files.go`, `tasks.go`, `tasks_files.go`, `users.go`, `contacts.go`, `opportunities.go`, `cases.go`, `crm.go`, `crm_tasks.go`, `catalog.go`, `docs.go`, `dav.go`, `mails.go`, `invoices.go` — one file per subject (or per subject facet), each registers in `init()`. `dav.go` exposes the Documents module by id (`oo dav ls|move|copy|mkdir|rename-file|rename-folder|download|fileops`).
- CLI-only deps (`spf13/cobra`, `joho/godotenv`) stay out of the library.
- **TUI — `cmd/office/` as `package main`.** Bubble Tea three-pane browser (module tree, selectable list, markdown preview). Reuses `cmd/internal/bootstrap` for env/auth and the root `onlyoffice` library for all API calls. UI logic in `cmd/office/ui/`; preview/formatting in `cmd/office/preview/`; list loaders in `cmd/office/fetch/`.
- **List table (`DataTable`)** — `cmd/office/ui/table*.go`. Column layout policies live in `cmd/office/model/table_layout.go` (`TableFlexLayoutFor`); cell rendering uses the bubbles/table inline pattern in `table_render.go` (`renderTableCell`, `padANSIWidth`). See `.cursor/skills/office-tui-table/SKILL.md` before changing center-pane tables.
- **Shared bootstrap — `cmd/internal/bootstrap/`.** `LoadEnv()` + `NewClient(ctx)` extracted from `oo`; both binaries import it.
- **Bulk Documents tools — `cmd/ooscan/`, `cmd/pdfamount/`, `cmd/kontoblatt/`, `cmd/kontolink/`.** Single-purpose binaries (folder index, PDF amounts, Kontoblatt summary/linking). Pace requests, route API calls through `DoRetry`; usage in README.
- **Personal ops tooling** (disk inventory, dossier→CRM sync, SearXNG) lives in private [`eSlider/oo-workspace`](https://git.produktor.io/eSlider/oo-workspace) (`oow`), not in this public tree.
## Rules
@@ -27,7 +28,7 @@ Canonical Go client for OnlyOffice Workspace (Projects + Calendar + CRM) and the
- New endpoints go into the library first; CLI commands are thin wrappers.
- Prefer `ResponseObject` / `postFormObject` / `putFormObject` / `deleteObject` over hand-rolled `json.Unmarshal(responseField(...))` blocks — they exist for DRY, use them.
- Domain split is by file, **not** by subpackage. Don't introduce `internal/` or `pkg/*` subpackages inside the library — it flattens the `*Client` call surface for a reason.
- CLI commands follow **subject → verb** structure (`oo <subject> <verb>`), never `oo <verb>-<subject>`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/<subject>.go` for a genuinely new domain.
- CLI commands follow **subject → verb** structure (`oo <subject> <verb>`), never `oo <verb>-<subject>`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/<subject>.go` for a genuinely new domain. The subject→verb tree in `cmd/oo/main.go` and the README table are documentation — update them with the code.
- **Documents for agents:** prefer Markdown in git; OnlyOffice UI is weak for `.md`/`.txt`. Use `oo docs put-md` (md→docx) and `oo docs put-txt` (txt→docx, preserves line breaks). All upload paths default to **upsert** by `stem|ext` (`--replace`, default true); `--no-replace` fails on conflict; `--allow-duplicate` opts into raw OO append. `oo projects files dedupe PROJECT_ID` reports/removes duplicate stem|ext copies (`--apply`, `--cross`; includes project root folder).
- Every table output goes through `printTable(headers, rows)`; every single-object through `printObject(v)`. Do not `fmt.Println` rows ad-hoc or the `--output json` flag breaks for that command.
- No secrets in the repo; use `.env` (gitignored). Commit `.env.example` only.
+17
View File
@@ -6,6 +6,23 @@ adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
## Unreleased
## [0.18.0](https://github.com/eSlider/go-onlyoffice/compare/v0.17.0...v0.18.0) (2026-09-04)
### Features
* **crm:** add UpdateContactName and CloseCRMTask helpers ([4d8af7a](https://github.com/eSlider/go-onlyoffice/commit/4d8af7a2fef1fcfd96ad913cefc6497e3389963b))
### Bug Fixes
* **crm:** deterministic sortBy=id in contact paged lists ([4d91726](https://github.com/eSlider/go-onlyoffice/commit/4d917261792203732ac739644c9c82124b2166eb))
### Documentation
* **funding:** eSlider support links (reverse-import GitHub e9c969a) ([d650a16](https://github.com/eSlider/go-onlyoffice/commit/d650a16a36037949505eb017c92bd3283e6e2f0f))
## [0.17.0](https://github.com/eSlider/go-onlyoffice/compare/v0.16.0...v0.17.0) (2026-08-31)
+93 -5
View File
@@ -503,6 +503,29 @@ type Task struct {
|---|---|
| `GetUsers()` | List all users with profiles |
### Documents Files
| Method | Description |
|---|---|
| `ListDavFolder(ctx, id)` | List a Documents folder (`@root` for virtual sections) |
| `ListDavSections(ctx)` | Virtual sections (Documents, Projects, …) |
| `CreateDavFolder(ctx, parentID, title)` | Create a subfolder |
| `RenameDavFolder(ctx, id, title)` / `RenameDavFile(ctx, id, title)` | Rename folder / file |
| `DownloadFile(ctx, id, dst)` / `DownloadDavFile(ctx, id, w)` | Download file bytes |
| `UploadDavFile(ctx, folderID, fileName, src)` | Upload from a reader |
| `UploadToFolder(ctx, folderID, localPath)` | Upload a local file into a folder |
| `UploadToFolderReplacing(ctx, folderID, localPath)` | Upsert by `stem\|ext`; returns replaced ids |
| `UpdateFile(ctx, fileID, localPath)` | New version of an existing file (same id, no copy) |
| `MoveDavItems(ctx, folderIDs, fileIDs, dest)` | Move (`resolveType=Skip`); per-operation errors surfaced, not silent nil |
| `CopyDavItems(ctx, folderIDs, fileIDs, dest)` | Copy (`conflictResolveType=Skip`); errors surfaced |
| `MoveFiles(ctx, destFolderID, fileIDs)` | Move with `resolveType=Skip` + `holdResult`; errors surfaced |
| `ListFileOps(ctx)` | Active file operations (move/copy status polling) |
| `FolderFiles(ctx, folderID)` | Flat file list of a folder (stem helpers) |
| `DeleteFilesByStem(ctx, folderID, stem)` | Remove `stem\|ext` copies |
| `DoRetry(ctx, policy, fn)` | Deterministic linear backoff (N·Base, no jitter) on 429/502/503/504 |
| `DefaultRetryPolicy()` | 5 attempts, 1s·2s·3s·4s waits, 30s cap |
| `Transient(err)` | True for retriable OnlyOffice answers |
### Helper Types
| Type | Description |
@@ -622,6 +645,8 @@ oo docs convert ./note.docx # → note.md
oo docs ocr ./scan.jpg --md ./scan.md # searchable PDF + markdown
oo docs hocr ./scan.jpg --lang spa --md ./scan.hocr.md --yaml ./scan.yml
oo docs put-md 7 ./OO-HONDA-7-INDEX.md --folder 490
oo docs put-txt 7 ./notes.txt --folder 490
oo docs put-xlsx 7 ./table.xlsx --folder 490
oo docs as-md 2815 --to ./parte.md # download OO file as MD (OCR if needed)
oo docs as-md 307 --hocr --lang spa # OO download via go-hocr structure
oo projects files put-md 7 ./note.md # alias
@@ -631,21 +656,84 @@ oo tasks files upload 208 ./notes.pdf
oo tasks files detach 208 12345
```
### Documents module (`oo dav`)
Direct access to the Documents module by folder/file id — the same calls that
back `oo-webdav` and the project/task file commands. `move` sends
`resolveType=Skip` + `holdResult=true`: without those params the legacy
`fileops/move` endpoint answers 200 without moving anything, and the library
surfaces such per-operation errors instead of a silent nil
(`MoveDavItems` / `CopyDavItems` / `MoveFiles`).
```bash
oo dav ls 659
oo dav ls @root # virtual sections (Documents, Projects, …)
oo dav mkdir 659 "2026 inbox"
oo dav move 659 22881 22882 # DEST_FOLDER_ID FILE_ID…
oo dav move 659 22881 --folders 670 # move folders along with files
oo dav copy 659 22881
oo dav rename-file 22881 invoice-v2.pdf
oo dav rename-folder 671 o2-archive
oo dav download 22881 --to ./copy.pdf # default path: ./<server title>
oo dav fileops # active move/copy operations (status polling)
```
### Search (`oo search`)
Full-text search over the Documents index. The REST endpoint
`/api/2.0/files/@search/{query}` only searches file names in the database, so
`oo search` talks to the OnlyOffice **Elasticsearch** directly (index
`files_file`). Name search is default; `--content` also matches extracted
document text (`document.attachment.content`, Office formats only).
See [`docs/elasticsearch.md`](docs/elasticsearch.md) for the tunnel setup.
```bash
oo search "Rechnung" # names only
oo search "Mahngebühr" --content # names + document text
oo search "Rechnung" --folder 649 --limit 50
oo search "Rechnung" --json # shorthand for -o json
```
Requires `ONLYOFFICE_ES_URL` (plus optional `ONLYOFFICE_ES_INDEX`,
`ONLYOFFICE_TENANT`).
### Bulk tools (`cmd/`)
Small single-purpose binaries for bulk Documents work. All of them pace
requests and retry transient OnlyOffice answers (429/502/503/504) with a
deterministic linear backoff — no jitter, same waits on every run
(see `DoRetry` below). Build with `go build ./cmd/<tool>`.
```bash
ooscan 659 # recursive index → TSV: file_id, folder_id, path, title
ooscan 659 666 > oo-index.tsv # several roots into one index
pdfamount 671 # "Zu zahlender Betrag" per PDF → TSV: file_id, title, amount
kontoblatt 3906 ./kontoblatt.xlsx # summary (Gegenkonto/Monat) uploaded next to source
kontolink IN.xlsx oo-index.tsv OUT.xlsx [FILE_ID] [AMOUNTS_TSV]
# kontolink writes DocEditor links into the Link column: Beleg → supplier+month
# → amount+date (5th arg = pdfamount output); with FILE_ID it updates the
# source file in place, else uploads an "(links)" copy next to it.
```
| Subject | Verbs |
|---|---|
| `calendar` | `list`, `events`, `add`, `delete` |
| `projects` | `list`, `get`, `milestones`, `create`, `update`, `delete`, **`files`** (`list`, `upload`, `download`, `rename`, `delete`) |
| `projects` | `list`, `get`, `milestones`, `milestone-create`, `create`, `update`, `delete`, `contacts` (`add`, `remove`), `link-authors`, `link-git`, **`files`** (`list`, `upload`, `download`, `rename`, `delete`, `dedupe`, `as-md`, `put-md`, `put-txt`, `put-xlsx`) |
| `tasks` | `list`, `get`, `create`, `update`, `delete`, `subtask add`, **`files`** (`list`, `upload`, `detach`) |
| `users` | `list`, `self` (alias: `oo whoami`) |
| `contacts` | `list`, `get`, `delete`, `info-add`, `merge`, `dedupe-info` |
| `contacts` | `list`, `get`, `delete`, `info-add`, `merge`, `dedupe-info`, `tags`, `tag-add`, `tag-create`, `tag-remove` |
| `persons` | `list`, `create`, `delete`, `dedupe` |
| `companies` | `list`, `create`, `delete`, `dedupe`, `dedupe-persons` |
| `opportunities` | `list`, `get`, `create`, `delete`, `stages`, `member-add`, `dedupe`, `dedupe-members`, `fix-titles` |
| `opportunities` | `list`, `get`, `create`, `update`, `delete`, `stages`, `member-add`, `dedupe`, `dedupe-members`, `fix-titles` |
| `invoices` | `list`, `get`, `create`, `update`, `pdf`, `pdf-cleanup`, `status`, `delete`, `items …` |
| `crm` | `cleanup` |
| `mails` | `accounts`, `folders`, `list`, `get`, `draft`, `attach`, `draft-invoice`, `delete` |
| `mails` | `accounts`, `folders`, `list`, `get`, `download-attachment`, `draft`, `attach`, `draft-invoice`, `send`, `delete` |
| `cases` | `list`, `create`, `delete`, `member-add` |
| `crm-tasks` | `list`, `create`, `delete`, `categories` |
| `crm-tasks` | `list`, `create`, `delete`, `categories`, `reassign-self` |
| `docs` | `tools`, `convert`, `optimize`, `ocr`, `hocr`, `as-md`, `put-md`, `put-txt`, `put-xlsx` |
| `catalog` | `match`, `merge`, `apply`, `scan-contacts`, `scan-projects`, `scan-thunderbird` |
| `dav` | `ls`, `move`, `copy`, `mkdir`, `rename-file`, `rename-folder`, `download`, `fileops` |
| `search` | `QUERY` (`--content`, `--folder ID`, `--limit N`, `--json`) |
The CLI reads only `.env` from the current working directory (godotenv is a
CLI-only concern — the library itself never loads dotfiles).
+220
View File
@@ -0,0 +1,220 @@
// Command kontoblatt builds a summary ("сводная таблица") of a Kontoblatt XLSX
// (Datum, Gegenkonto, Buchungstext, Beleg, Soll, Haben, Bemerkung) and uploads
// it back to the same OnlyOffice folder as the source file.
//
// Usage: kontoblatt <FILE_ID> <LOCAL_XLSX>
package main
import (
"context"
"fmt"
"os"
"regexp"
"sort"
"strconv"
"strings"
onlyoffice "github.com/eslider/go-onlyoffice"
"github.com/xuri/excelize/v2"
)
type agg struct {
count int
soll float64
haben float64
reFehlt int
}
type rec struct {
date, month, konto, text string
soll, haben float64
reFehlt bool
}
var dateRe = regexp.MustCompile(`^\d{2}\.\d{2}\.\d{4}$`)
func parseAmount(s string) float64 {
s = strings.TrimSpace(s)
s = strings.ReplaceAll(s, "€", "")
s = strings.ReplaceAll(s, " ", "")
s = strings.ReplaceAll(s, ",", "") // German thousands separator
s = strings.TrimSpace(s)
if s == "" {
return 0
}
v, err := strconv.ParseFloat(s, 64)
if err != nil {
return 0
}
return v
}
func cell(row []string, i int) string {
if i < len(row) {
return strings.TrimSpace(row[i])
}
return ""
}
func main() {
if len(os.Args) < 3 {
fmt.Fprintln(os.Stderr, "usage: kontoblatt <FILE_ID> <LOCAL_XLSX>")
os.Exit(2)
}
fileID, path := os.Args[1], os.Args[2]
ctx := context.Background()
f, err := excelize.OpenFile(path)
if err != nil {
panic(err)
}
defer f.Close()
var recs []rec
for _, sh := range f.GetSheetList() {
rows, err := f.GetRows(sh)
if err != nil {
continue
}
for _, r := range rows {
d := cell(r, 0)
if !dateRe.MatchString(d) {
continue
}
text := cell(r, 2)
recs = append(recs, rec{
date: d,
month: d[3:10],
konto: cell(r, 1),
text: text,
soll: parseAmount(cell(r, 4)),
haben: parseAmount(cell(r, 5)),
reFehlt: strings.Contains(strings.ToUpper(text), "FEHLT"),
})
}
}
byKonto := map[string]*agg{}
byMonth := map[string]*agg{}
getK := func(k string) *agg {
if byKonto[k] == nil {
byKonto[k] = &agg{}
}
return byKonto[k]
}
getM := func(k string) *agg {
if byMonth[k] == nil {
byMonth[k] = &agg{}
}
return byMonth[k]
}
var tot agg
for _, r := range recs {
k := getK(r.konto)
k.count++
k.soll += r.soll
k.haben += r.haben
if r.reFehlt {
k.reFehlt++
}
m := getM(r.month)
m.count++
m.soll += r.soll
m.haben += r.haben
if r.reFehlt {
m.reFehlt++
}
tot.count++
tot.soll += r.soll
tot.haben += r.haben
if r.reFehlt {
tot.reFehlt++
}
}
out := excelize.NewFile()
defer out.Close()
writeSheet(out, "Nach Gegenkonto", "Gegenkonto", byKonto, tot)
writeSheet(out, "Nach Monat", "Monat", byMonth, tot)
outPath := "/tmp/opencode/kontoblatt-zusammenfassung.xlsx"
if err := out.SaveAs(outPath); err != nil {
panic(err)
}
// upload next to the source file
creds := onlyoffice.GetEnvironmentCredentials()
c := onlyoffice.NewClient(creds)
var src *onlyoffice.FileEntry
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
var err error
src, err = c.GetFile(ctx, fileID)
return err
}); derr != nil {
panic(derr)
}
folder := ""
if src.FolderID != nil {
folder = src.FolderID.String()
}
title := ""
if src.Title != nil {
title = *src.Title
}
fmt.Printf("source: id=%s title=%q folder=%s\n", fileID, title, folder)
name := "Kontoblatt-1591-2025-Zusammenfassung.xlsx"
tmp := "/tmp/opencode/" + name
data, _ := os.ReadFile(outPath)
if err := os.WriteFile(tmp, data, 0o600); err != nil {
panic(err)
}
var entry *onlyoffice.FileEntry
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
var err error
entry, _, err = c.UploadToFolderReplacing(ctx, folder, tmp)
return err
}); derr != nil {
panic(derr)
}
fmt.Printf("uploaded: %s -> folder %s (id %v)\n", name, folder, entry.ID)
// print the summary
printAgg("Nach Gegenkonto", byKonto, tot)
printAgg("Nach Monat", byMonth, tot)
}
func writeSheet(f *excelize.File, sheet, key string, m map[string]*agg, tot agg) {
f.NewSheet(sheet)
rows := [][]any{{key, "Anzahl", "Soll", "Haben", "Saldo", `davon "fehlt"`}}
keys := make([]string, 0, len(m))
for k := range m {
keys = append(keys, k)
}
sort.Strings(keys)
for _, k := range keys {
a := m[k]
rows = append(rows, []any{k, a.count, a.soll, a.haben, a.soll - a.haben, a.reFehlt})
}
rows = append(rows, []any{"GESAMT", tot.count, tot.soll, tot.haben, tot.soll - tot.haben, tot.reFehlt})
for i, row := range rows {
for j, v := range row {
cellRef, _ := excelize.CoordinatesToCellName(j+1, i+1)
_ = f.SetCellValue(sheet, cellRef, v)
}
}
}
func printAgg(title string, m map[string]*agg, tot agg) {
fmt.Printf("\n== %s ==\n", title)
keys := make([]string, 0, len(m))
for k := range m {
keys = append(keys, k)
}
sort.Strings(keys)
fmt.Printf("%-12s %6s %12s %12s %12s %7s\n", "key", "count", "soll", "haben", "saldo", "fehlt")
for _, k := range keys {
a := m[k]
fmt.Printf("%-12s %6d %12.2f %12.2f %12.2f %7d\n", k, a.count, a.soll, a.haben, a.soll-a.haben, a.reFehlt)
}
fmt.Printf("%-12s %6d %12.2f %12.2f %12.2f %7d\n", "GESAMT", tot.count, tot.soll, tot.haben, tot.soll-tot.haben, tot.reFehlt)
}
+478
View File
@@ -0,0 +1,478 @@
// Command kontolink fills the "Link" column of a Kontoblatt ("ungeklärte
// Posten") XLSX by matching each row to an OnlyOffice document.
//
// Strategy (deterministic, conservative — no LLM):
// 1. Beleg token (letters/digits from the "Beleg" column) appears in the file
// title; among candidates prefer (a) the row's month, (b) real invoices over
// copies/dupes, and require the result to be unique;
// 2. else supplier + row month + "rechnung", again unique.
//
// A file is linked at most once (rows already carrying a link are kept and their
// file counts as used). Ambiguous rows are left UNLINKED for manual review.
//
// Usage: kontolink <IN_XLSX> <INDEX_TSV> <OUT_XLSX>
package main
import (
"context"
"fmt"
"os"
"path/filepath"
"regexp"
"strconv"
"strings"
"time"
onlyoffice "github.com/eslider/go-onlyoffice"
"github.com/xuri/excelize/v2"
)
var (
dateRe = regexp.MustCompile(`^\d{2}\.\d{2}\.\d{4}$`)
nonAln = regexp.MustCompile(`[^0-9a-z]+`)
fileID = regexp.MustCompile(`fileid=(\d+)`)
)
func parseDay(s string) (time.Time, bool) {
t, err := time.Parse("02.01.2006", strings.TrimSpace(s))
return t, err == nil
}
func titleDay(title string) (time.Time, bool) {
if len(title) >= 10 {
if t, err := time.Parse("2006-01-02", title[:10]); err == nil {
return t, true
}
}
return time.Time{}, false
}
// nearest picks the candidate whose title date is closest to rd. Ties and
// undated candidates (when >1) are rejected.
func nearest(cands []entry, rd time.Time) (entry, bool) {
if len(cands) == 1 {
return cands[0], true
}
best, bestD, tie := -1, 0.0, false
for i, e := range cands {
td, ok := titleDay(e.title)
if !ok {
continue
}
d := td.Sub(rd).Hours() / 24
if d < 0 {
d = -d
}
if best < 0 || d < bestD {
best, bestD, tie = i, d, false
} else if d == bestD {
tie = true
}
}
if best < 0 || tie {
return entry{}, false
}
return cands[best], true
}
const linkPrefix = "https://office.pro-dukt.de/Products/Files/DocEditor.aspx?fileid="
type entry struct {
id, path, title, norm string
}
func norm(s string) string { return nonAln.ReplaceAllString(strings.ToLower(s), "") }
func main() {
if len(os.Args) < 4 {
fmt.Fprintln(os.Stderr, "usage: kontolink <IN_XLSX> <INDEX_TSV> <OUT_XLSX>")
os.Exit(2)
}
in, idxPath, out := os.Args[1], os.Args[2], os.Args[3]
idxRaw, err := os.ReadFile(idxPath)
if err != nil {
panic(err)
}
var entries []entry
for _, line := range strings.Split(string(idxRaw), "\n") {
parts := strings.Split(line, "\t")
if len(parts) < 4 || parts[0] == "" {
continue
}
entries = append(entries, entry{id: parts[0], path: parts[2], title: parts[3], norm: norm(parts[3])})
}
f, err := excelize.OpenFile(in)
if err != nil {
panic(err)
}
defer f.Close()
sheet := f.GetSheetList()[0]
rows, err := f.GetRows(sheet)
if err != nil {
panic(err)
}
// optional 5th arg: amounts TSV "file_id\ttitle\tamount" (see cmd/pdfamount)
var amts []amtEntry
if len(os.Args) >= 6 && os.Args[5] != "" {
amts = loadAmounts(os.Args[5])
}
used := map[string]bool{}
for _, r := range rows {
if m := fileID.FindStringSubmatch(cell(r, 7)); m != nil {
used[m[1]] = true
}
}
var linked, byBeleg, bySupplier, byAmount, unmatched, ambiguous int
for i, r := range rows {
if i == 0 || !dateRe.MatchString(cell(r, 0)) || strings.TrimSpace(cell(r, 7)) != "" {
continue
}
beleg := norm(cell(r, 3))
supplier := supplierNorm(cell(r, 2))
month := monthYear(cell(r, 0))
rd, _ := parseDay(cell(r, 0))
e, kind, ok := pick(entries, used, beleg, supplier, month, rd)
if !ok {
if ae, aok := amountPick(amts, used, supplier, rowAmount(r), rd); aok {
e, kind, ok = entry{id: ae.id, title: ae.title}, "amount", true
}
}
if !ok {
if beleg != "" {
ambiguous++
} else {
unmatched++
}
continue
}
ref, _ := excelize.CoordinatesToCellName(8, i+1)
if err := f.SetCellValue(sheet, ref, linkPrefix+e.id); err != nil {
panic(err)
}
used[e.id] = true
linked++
switch kind {
case "beleg":
byBeleg++
case "supplier":
bySupplier++
case "amount":
byAmount++
}
fmt.Printf("row %3d %-30s -> %s [%s]\n", i+1, cell(r, 2), e.title, kind)
}
if err := f.SaveAs(out); err != nil {
panic(err)
}
fmt.Printf("\nlinked=%d (beleg=%d, supplier=%d, amount=%d), ambiguous=%d, no-candidate=%d\n",
linked, byBeleg, bySupplier, byAmount, ambiguous, unmatched)
// Optional 4th arg: source OnlyOffice file id. Try to update it in place;
// if it is locked (OnlyOffice 500), upload a "(links)" copy next to it.
if len(os.Args) >= 5 && os.Args[4] != "" {
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
ctx, cancel := context.WithTimeout(context.Background(), 120*time.Second)
defer cancel()
var src *onlyoffice.FileEntry
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
var err error
src, err = c.GetFile(ctx, os.Args[4])
return err
}); derr != nil {
panic(derr)
}
folder, title := "", ""
if src.FolderID != nil {
folder = src.FolderID.String()
}
if src.Title != nil {
title = *src.Title
}
uderr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
_, err := c.UpdateFile(ctx, os.Args[4], out)
return err
})
if uderr == nil {
fmt.Printf("updated file %s in place\n", os.Args[4])
return
}
fmt.Printf("in-place update failed (locked?); uploading a copy to folder %s\n", folder)
ext := filepath.Ext(title)
name := strings.TrimSuffix(title, ext) + " (links)" + ext
tmp := filepath.Join(os.TempDir(), name)
data, _ := os.ReadFile(out)
if err := os.WriteFile(tmp, data, 0o600); err != nil {
panic(err)
}
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
_, _, err := c.UploadToFolderReplacing(ctx, folder, tmp)
return err
}); derr != nil {
panic(derr)
}
fmt.Printf("uploaded copy: %s -> folder %s\n", name, folder)
}
}
func cell(r []string, i int) string {
if i < len(r) {
return strings.TrimSpace(r[i])
}
return ""
}
func supplierNorm(s string) string {
s = strings.ToUpper(s)
if i := strings.Index(s, ","); i >= 0 {
s = s[:i]
}
for _, w := range []string{"RE FEHLT", "GS FEHLT", "WOFR", "WOFÜR"} {
s = strings.ReplaceAll(s, w, "")
}
return norm(s)
}
func monthYear(date string) string {
if len(date) == 10 {
return date[6:10] + "-" + date[3:5]
}
return ""
}
// pick returns an unused candidate. Beleg match wins; supplier+month is a
// fallback. When several candidates qualify, the one closest in time to the row
// date wins; a tie is rejected (ambiguous) rather than guessed.
func pick(entries []entry, used map[string]bool, beleg, supplier, month string, rd time.Time) (entry, string, bool) {
free := func(e entry) bool { return !used[e.id] }
if len(beleg) >= 5 {
var inMonth []entry
for _, e := range entries {
if free(e) && belegMatches(e.norm, beleg) &&
(month == "" || strings.Contains(e.title, month)) {
inMonth = append(inMonth, e)
}
}
if supplier != "" {
var s []entry
for _, e := range inMonth {
if strings.Contains(e.norm, supplier) {
s = append(s, e)
}
}
if len(s) > 0 {
inMonth = s
}
}
inMonth = topRank(inMonth)
if e, ok := nearest(inMonth, rd); ok {
return e, "beleg", true
}
// A Beleg is present but no file carries it: do NOT fall back to a
// supplier guess (that links the wrong invoice).
return entry{}, "", false
}
if supplier != "" && month != "" {
var c []entry
for _, e := range entries {
if free(e) && strings.Contains(e.norm, supplier) &&
strings.Contains(e.title, month) && strings.Contains(e.norm, "rechnung") {
c = append(c, e)
}
}
c = topRank(c)
if e, ok := nearest(c, rd); ok {
return e, "supplier", true
}
}
return entry{}, "", false
}
// topRank keeps only the highest-ranked candidates (real invoice over copy /
// dupe / op), so a tie with a duplicate does not mask the real file.
func topRank(cands []entry) []entry {
if len(cands) < 2 {
return cands
}
best := 0
for _, e := range cands {
if rank(e) > best {
best = rank(e)
}
}
out := cands[:0]
for _, e := range cands {
if rank(e) == best {
out = append(out, e)
}
}
return out
}
func rank(e entry) int {
s := 0
if strings.Contains(e.path, "/2025") || strings.Contains(e.path, "/2024") {
s += 4
}
if strings.Contains(e.norm, "rechnung") {
s += 2
}
if strings.Contains(e.norm, "dupe") || strings.Contains(e.norm, "copy") ||
strings.Contains(e.norm, "op") {
s--
}
return s
}
// belegMatches reports whether a Beleg identifies the file: the whole normalized
// Beleg appears, or (for long numeric Belege, e.g. "24/641393110") an 8-digit
// window of its longest digit run appears.
func belegMatches(titleNorm, beleg string) bool {
if strings.Contains(titleNorm, beleg) {
return true
}
run := longestDigitRun(beleg)
for i := 0; i+8 <= len(run); i++ {
if strings.Contains(titleNorm, run[i:i+8]) {
return true
}
}
return false
}
func longestDigitRun(s string) string {
var best, cur strings.Builder
for _, r := range s {
if r >= '0' && r <= '9' {
cur.WriteRune(r)
if cur.Len() > best.Len() {
best.Reset()
best.WriteString(cur.String())
}
} else {
cur.Reset()
}
}
return best.String()
}
type amtEntry struct {
id string
title string
norm string
amount float64
date time.Time
hasDate bool
}
func loadAmounts(path string) []amtEntry {
raw, err := os.ReadFile(path)
if err != nil {
return nil
}
var out []amtEntry
for _, line := range strings.Split(string(raw), "\n") {
p := strings.Split(line, "\t")
if len(p) < 3 {
continue
}
v, err := strconv.ParseFloat(strings.TrimSpace(p[2]), 64)
if err != nil {
continue
}
e := amtEntry{id: p[0], title: p[1], norm: norm(p[1]), amount: v}
if len(p[1]) >= 10 {
if t, err := time.Parse("2006-01-02", p[1][:10]); err == nil {
e.date, e.hasDate = t, true
}
}
out = append(out, e)
}
return out
}
func rowAmount(r []string) float64 {
if v := parseAmount(cell(r, 4)); v != 0 {
return v
}
return parseAmount(cell(r, 5))
}
func parseAmount(s string) float64 {
s = strings.ReplaceAll(s, "€", "")
s = strings.ReplaceAll(s, " ", "")
s = strings.ReplaceAll(s, ",", ".")
if s == "" {
return 0
}
v, err := strconv.ParseFloat(s, 64)
if err != nil {
return 0
}
return v
}
// amountPick matches a row to an O2 invoice by amount + nearest date. Scoped to
// Telefonica/O2 rows and O2 files, so it cannot cross-link other suppliers.
func amountPick(amts []amtEntry, used map[string]bool, supplier string, amt float64, rd time.Time) (amtEntry, bool) {
if amt <= 0 || len(amts) == 0 {
return amtEntry{}, false
}
if !strings.Contains(supplier, "telefonica") && !strings.Contains(supplier, "o2") {
return amtEntry{}, false
}
var cands []amtEntry
for _, a := range amts {
if used[a.id] || !strings.Contains(a.norm, "o2") {
continue
}
d := a.amount - amt
if d < 0 {
d = -d
}
if d > 0.005 {
continue
}
if a.hasDate && !rd.IsZero() {
days := a.date.Sub(rd).Hours() / 24
if days < 0 {
days = -days
}
if days > 75 {
continue
}
}
cands = append(cands, a)
}
if len(cands) == 1 {
return cands[0], true
}
best, bestD, tie := -1, 0.0, false
for i, a := range cands {
if !a.hasDate {
continue
}
d := a.date.Sub(rd).Hours() / 24
if d < 0 {
d = -d
}
if best < 0 || d < bestD {
best, bestD, tie = i, d, false
} else if d == bestD {
tie = true
}
}
if best < 0 || tie {
return amtEntry{}, false
}
return cands[best], true
}
+27
View File
@@ -25,6 +25,7 @@ func davCmd() *cobra.Command {
cmd.AddCommand(davMoveCmd())
cmd.AddCommand(davCopyCmd())
cmd.AddCommand(davMkdirCmd())
cmd.AddCommand(davRemoveCmd())
cmd.AddCommand(davRenameFileCmd())
cmd.AddCommand(davRenameFolderCmd())
cmd.AddCommand(davDownloadCmd())
@@ -184,6 +185,32 @@ func davMkdirCmd() *cobra.Command {
}
}
func davRemoveCmd() *cobra.Command {
var folderIDs []string
cmd := &cobra.Command{
Use: "rm [FILE_ID...]",
Aliases: []string{"delete"},
Short: "Permanently delete file(s) and/or folder(s) from Documents",
Args: cobra.ArbitraryArgs,
RunE: func(cmd *cobra.Command, args []string) error {
if len(args) == 0 && len(folderIDs) == 0 {
return fmt.Errorf("dav rm: give at least one FILE_ID or --folders")
}
c, err := newOO(cmd)
if err != nil {
return err
}
if err := c.DeleteDavItems(cmd.Context(), folderIDs, args); err != nil {
return err
}
printObject(map[string]any{"deleted_files": args, "deleted_folders": folderIDs})
return nil
},
}
cmd.Flags().StringSliceVar(&folderIDs, "folders", nil, "folder ids to delete")
return cmd
}
func davRenameFileCmd() *cobra.Command {
return &cobra.Command{
Use: "rename-file FILE_ID NEW_TITLE",
+8 -6
View File
@@ -3,20 +3,22 @@
// Command tree is subject-based (mirrors the library split and the `tea` CLI):
//
// oo calendar list | events | add | delete
// oo projects list | get | milestones | create | update | delete | files (list|upload|download|rename|delete|as-md|put-md)
// oo projects list | get | milestones | milestone-create | create | update | delete | contacts (add|remove) | link-authors | link-git | files (list|upload|download|rename|delete|dedupe|as-md|put-md|put-txt|put-xlsx)
// oo tasks list | get | create | update | delete | subtask add | files (list|upload|detach)
// oo users list | self (alias: oo whoami)
// oo contacts list | get | delete | info-add | merge | dedupe-info
// oo contacts list | get | delete | info-add | merge | dedupe-info | tags | tag-add | tag-create | tag-remove
// oo persons list | create | delete | dedupe
// oo companies list | create | delete | dedupe | dedupe-persons
// oo opportunities list | get | create | delete | stages | member-add | dedupe | dedupe-members | fix-titles
// oo opportunities list | get | create | update | delete | stages | member-add | dedupe | dedupe-members | fix-titles
// oo cases list | create | delete | member-add
// oo crm-tasks list | create | delete | categories
// oo crm-tasks list | create | delete | categories | reassign-self
// oo crm cleanup
// oo mails accounts | folders | list | get | download-attachment | draft | attach | draft-invoice | delete
// oo mails accounts | folders | list | get | download-attachment | draft | attach | draft-invoice | send | delete
// oo invoices list | get | create | update | pdf | pdf-cleanup | status | delete | items …
// oo docs tools | convert | ocr | as-md | put-md
// oo docs tools | convert | optimize | ocr | hocr | as-md | put-md | put-txt | put-xlsx
// oo catalog match | merge | apply | scan-contacts | scan-projects | scan-thunderbird
// oo dav ls | move | copy | mkdir | rename-file | rename-folder | download | fileops
// oo search QUERY [--content] [--folder ID] [--limit N] [--json]
//
// CRM association rules: docs/crm-associations.md
//
+72
View File
@@ -0,0 +1,72 @@
package main
import (
onlyoffice "github.com/eslider/go-onlyoffice"
"github.com/spf13/cobra"
)
func init() {
rootCmd.AddCommand(searchCmd())
}
// searchCmd queries the OnlyOffice Elasticsearch index directly. The REST
// /api/2.0/files/@search endpoint only searches file names in the database;
// content search needs ES (see docs/elasticsearch.md).
func searchCmd() *cobra.Command {
var (
content bool
folder string
limit int
asJSON bool
)
cmd := &cobra.Command{
Use: "search QUERY",
Short: "Full-text search over documents by name, optionally by content (Elasticsearch)",
Long: "Search the OnlyOffice Documents index.\n\n" +
"By default only file names are matched. With --content the query also\n" +
"matches extracted document text (document.attachment.content); this covers\n" +
"Office formats (docx/xlsx/pptx) and is slower.\n\n" +
"Requires ONLYOFFICE_ES_URL (and optionally ONLYOFFICE_ES_INDEX,\n" +
"ONLYOFFICE_TENANT). See docs/elasticsearch.md for the tunnel setup.",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
if asJSON {
outputFormat = "json"
}
es, err := onlyoffice.NewESSearcher(onlyoffice.ESConfigFromEnv())
if err != nil {
return err
}
hits, err := es.Search(cmd.Context(), onlyoffice.SearchQuery{
Text: args[0],
InContent: content,
FolderID: folder,
Limit: limit,
})
if err != nil {
return err
}
rows := make([]map[string]any, 0, len(hits))
for _, h := range hits {
rows = append(rows, map[string]any{
"id": h.ID,
"title": h.Title,
"folder": h.ParentID,
"score": h.Score,
"highlight": h.Highlight,
})
}
if outputFormat == "json" {
printJSON(rows)
return nil
}
printTable([]string{"id", "title", "folder", "score", "highlight"}, rows)
return nil
},
}
cmd.Flags().BoolVar(&content, "content", false, "also match extracted document content")
cmd.Flags().StringVar(&folder, "folder", "", "limit to a Documents folder id")
cmd.Flags().IntVar(&limit, "limit", 20, "maximum number of results")
cmd.Flags().BoolVar(&asJSON, "json", false, "shorthand for --output json")
return cmd
}
+46
View File
@@ -0,0 +1,46 @@
package main
import (
"bytes"
"strings"
"testing"
)
func TestSearchCommandRegisteredWithFlags(t *testing.T) {
cmd, _, err := rootCmd.Find([]string{"search"})
if err != nil {
t.Fatal(err)
}
if cmd.Name() != "search" {
t.Fatalf("search resolved to %q", cmd.Name())
}
for _, name := range []string{"content", "folder", "limit", "json"} {
if cmd.Flags().Lookup(name) == nil {
t.Errorf("search: missing --%s flag", name)
}
}
if got := cmd.Flags().Lookup("limit").DefValue; got != "20" {
t.Errorf("--limit default = %q, want 20", got)
}
}
func TestSearchWithoutESURLIsClearError(t *testing.T) {
clearEnv(t, "ONLYOFFICE_ES_URL", "ONLYOFFICE_ES_INDEX", "ONLYOFFICE_TENANT")
errBuf := &bytes.Buffer{}
rootCmd.SetErr(errBuf)
rootCmd.SetOut(&bytes.Buffer{})
rootCmd.SetArgs([]string{"search", "Rechnung"})
t.Cleanup(func() {
rootCmd.SetArgs(nil)
rootCmd.SetOut(nil)
rootCmd.SetErr(nil)
})
err := rootCmd.Execute()
if err == nil {
t.Fatal("expected error without ONLYOFFICE_ES_URL")
}
if !strings.Contains(err.Error(), "ONLYOFFICE_ES_URL") {
t.Fatalf("error %q missing ONLYOFFICE_ES_URL", err.Error())
}
}
+50
View File
@@ -0,0 +1,50 @@
// Command ooscan recursively lists OnlyOffice Documents folders into a TSV
// index: file_id, folder_id, path, title.
//
// Usage: ooscan <FOLDER_ID> [<FOLDER_ID>...]
package main
import (
"context"
"fmt"
"os"
"time"
onlyoffice "github.com/eslider/go-onlyoffice"
)
func main() {
ctx := context.Background()
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
seen := map[string]bool{}
for _, root := range os.Args[1:] {
walk(ctx, c, root, "", 0, seen)
}
}
func walk(ctx context.Context, c *onlyoffice.Client, folderID, path string, depth int, seen map[string]bool) {
if depth > 8 || seen[folderID] {
return
}
seen[folderID] = true
// Throttle: OnlyOffice rate-limits (429) and the host must not be flooded.
time.Sleep(350 * time.Millisecond)
ctx, cancel := context.WithTimeout(ctx, 60*time.Second)
defer cancel()
var l *onlyoffice.DavListing
derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
var err error
l, err = c.ListDavFolder(ctx, folderID)
return err
})
if derr != nil {
fmt.Fprintf(os.Stderr, "list %s (%s): %v\n", path, folderID, derr)
return
}
for _, f := range l.Files {
fmt.Printf("%s\t%s\t%s\t%s\n", f.ID, folderID, path, f.Title)
}
for _, sub := range l.Folders {
walk(ctx, c, sub.ID, path+"/"+sub.Title, depth+1, seen)
}
}
+308
View File
@@ -0,0 +1,308 @@
// Command pdfamount walks a Documents folder, downloads matching PDFs and
// extracts the payable amount, printing "file_id\ttitle\tamount".
//
// Usage: pdfamount <FOLDER_ID> [TITLE_FILTER_REGEX]
package main
import (
"bytes"
"context"
"fmt"
"os"
"os/exec"
"regexp"
"strconv"
"strings"
"time"
onlyoffice "github.com/eslider/go-onlyoffice"
)
// amountPat is the amount capture shared by every amount regex.
const amountPat = `([0-9]+(?:[.,][0-9]+)*)`
// amountRE builds "<label> [optional (comment)] [: -] <number>".
func amountRE(label string) *regexp.Regexp {
return regexp.MustCompile(
`(?i)\b` + regexp.QuoteMeta(label) + `\b\s*(?:\([^)]*\))?\s*[:\-]?\s*` + amountPat)
}
// amountRes lists the payable-amount patterns in strict priority order: the
// first pattern with a usable amount wins, and a lower-priority label can never
// override a higher-priority one ("zu zahlender betrag" > "rechnungsbetrag" >
// "rechnungsendbetrag" > "gesamtbetrag" > "gesamtsumme (inkl. steuern)").
//
// "gesamtbetrag" and "gesamtsumme" are not in the original set but are the real
// labels on Diashop invoices ("Gesamtsumme (inkl. Steuern)"). The inclusive
// variant is matched before a plain "gesamtsumme". Everything after those
// primary labels is the broader fallback set, consulted only when no primary
// label yields an amount. Within one pattern the last usable amount is taken,
// because totals usually come last.
var amountRes = []*regexp.Regexp{
amountRE("zu zahlender betrag"),
amountRE("rechnungsbetrag"),
amountRE("rechnungsendbetrag"),
amountRE("gesamtbetrag"),
regexp.MustCompile(`(?i)\bgesamtsumme\b\s*\(\s*inkl\.?\s*steuern\s*\)\s*[:\-]?\s*` + amountPat),
amountRE("gesamtsumme"),
amountRE("endbetrag"),
amountRE("zahlbetrag"),
amountRE("bruttobetrag"),
amountRE("betrag"),
amountRE("total"),
amountRE("summe"),
}
// taxLineRe marks a line whose number is a tax rate/percentage: an explicit
// percent sign or a VAT/tax keyword. "Steuern" (plural, as in "inkl. Steuern")
// is handled separately so the inclusive total stays usable.
var taxLineRe = regexp.MustCompile(`(?i)%|\bMwSt\b|\bUSt\b|\bProzent\b`)
// steuerRe finds "Steuer"/"Umsatzsteuer" etc. RE2 has no lookahead, so the
// plural "Steuern" is excluded in isTaxLine.
var steuerRe = regexp.MustCompile(`(?i)steuer`)
// percentAfterRe detects a percent sign directly after a number (spaces ok).
var percentAfterRe = regexp.MustCompile(`^\s*%`)
func main() {
if len(os.Args) < 2 {
fmt.Fprintln(os.Stderr, "usage: pdfamount <FOLDER_ID> [TITLE_FILTER_REGEX]")
os.Exit(2)
}
folder := os.Args[1]
filter := regexp.MustCompile(`(?i)rechnung`)
if len(os.Args) >= 3 {
filter = regexp.MustCompile(os.Args[2])
}
ctx := context.Background()
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
files := listAll(ctx, c, folder)
for _, f := range files {
if !filter.MatchString(f.title) {
continue
}
if !strings.HasSuffix(strings.ToLower(f.title), ".pdf") {
continue
}
amount, err := pdfAmount(ctx, c, f.id)
if err != nil {
fmt.Fprintf(os.Stderr, "%s: %v\n", f.title, err)
continue
}
if amount == "" {
continue
}
fmt.Printf("%s\t%s\t%s\n", f.id, f.title, amount)
}
}
type file struct{ id, title string }
func listAll(ctx context.Context, c *onlyoffice.Client, folder string) []file {
seen := map[string]bool{}
var out []file
var walk func(string)
walk = func(id string) {
if seen[id] {
return
}
seen[id] = true
time.Sleep(300 * time.Millisecond)
l, err := c.ListDavFolder(ctx, id)
if err != nil {
fmt.Fprintf(os.Stderr, "list %s: %v\n", id, err)
return
}
for _, f := range l.Files {
out = append(out, file{f.ID, f.Title})
}
for _, sub := range l.Folders {
walk(sub.ID)
}
}
walk(folder)
return out
}
func pdfAmount(ctx context.Context, c *onlyoffice.Client, id string) (string, error) {
time.Sleep(time.Second)
tmp, err := os.CreateTemp("", "pdf-*.pdf")
if err != nil {
return "", err
}
defer os.Remove(tmp.Name())
derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
_ = tmp.Truncate(0)
_, _ = tmp.Seek(0, 0)
_, err := c.DownloadFile(ctx, id, tmp)
return err
})
if derr != nil {
tmp.Close()
return "", derr
}
tmp.Close()
var buf bytes.Buffer
cmd := exec.CommandContext(ctx, "pdftotext", "-layout", tmp.Name(), "-")
cmd.Stdout = &buf
if err := cmd.Run(); err != nil {
return "", err
}
return extractAmount(buf.String()), nil
}
// extractAmount returns the normalised ("1234.56") payable amount found in
// text, or "" if no usable amount matches.
//
// DKV invoices are special-cased first: they repeat a per-vehicle "TOTAL:" line
// and carry the real total only in the "Gesamtsummenaufstellung" section.
func extractAmount(text string) string {
if v, ok := dkvGrandTotal(text); ok {
return v
}
for _, re := range amountRes {
if v, ok := lastUsableAmount(text, re); ok {
return v
}
}
return ""
}
// dkvGrandTotal extracts the total of a DKV "Gesamtsummenaufstellung" section.
//
// Rule: DKV invoices repeat a per-vehicle "TOTAL:" line, so the last TOTAL is
// not the invoice total. When a "Gesamtsummenaufstellung" section exists, its
// total wins over every "TOTAL:" line: the first amount after the "»" marker,
// or, if there is none, the last amount in the section. The section ends at the
// page break (form feed) or end of text.
func dkvGrandTotal(text string) (string, bool) {
idx := strings.Index(strings.ToLower(text), "gesamtsummenaufstellung")
if idx < 0 {
return "", false
}
section := text[idx:]
if ff := strings.IndexByte(section, '\f'); ff >= 0 {
section = section[:ff]
}
if m := strings.Index(section, "»"); m >= 0 {
if v, ok := firstAmount(section[m:]); ok {
return v, true
}
}
return lastAmount(section)
}
// lastUsableAmount returns the last amount matched by re that is not a tax rate
// or percentage. Within one label the last usable amount wins.
func lastUsableAmount(text string, re *regexp.Regexp) (string, bool) {
ms := re.FindAllStringSubmatchIndex(text, -1)
for i := len(ms) - 1; i >= 0; i-- {
m := ms[i]
if isTaxRate(text, m[2], m[3]) {
continue
}
if v, ok := normalizeAmount(text[m[2]:m[3]]); ok {
return v, true
}
}
return "", false
}
// isTaxRate reports whether the number at text[start:end] is a tax rate or a
// percentage instead of a payable amount. A candidate is rejected when the
// token right after the number is "%" or the number's line carries a percent
// sign or a tax keyword. Rejecting is deliberate: office matching treats a
// known-but-different amount as a hard disqualifier, so an empty result is
// safer than the VAT rate.
func isTaxRate(text string, start, end int) bool {
if percentAfterRe.MatchString(text[end:]) {
return true
}
lineStart := strings.LastIndexByte(text[:start], '\n') + 1
line := text[lineStart:]
if n := strings.IndexByte(text[end:], '\n'); n >= 0 {
line = text[lineStart : end+n]
}
return isTaxLine(line)
}
// isTaxLine reports whether a line looks like a tax rate rather than a payable
// amount. "Steuern" is treated as a qualifier ("inkl. Steuern"), not a rate.
func isTaxLine(line string) bool {
if taxLineRe.MatchString(line) {
return true
}
for _, loc := range steuerRe.FindAllStringIndex(line, -1) {
if loc[1] >= len(line) || (line[loc[1]] != 'n' && line[loc[1]] != 'N') {
return true
}
}
return false
}
// numberRe finds bare numbers (with optional thousands/decimal separators).
var numberRe = regexp.MustCompile(`[0-9]+(?:[.,][0-9]+)*`)
func firstAmount(s string) (string, bool) {
for _, m := range numberRe.FindAllString(s, -1) {
if v, ok := normalizeAmount(m); ok {
return v, true
}
}
return "", false
}
func lastAmount(s string) (string, bool) {
ms := numberRe.FindAllString(s, -1)
for i := len(ms) - 1; i >= 0; i-- {
if v, ok := normalizeAmount(ms[i]); ok {
return v, true
}
}
return "", false
}
// normalizeAmount turns "1.234,56" (DE), "1,234.56" (EN) or "1234.56" into
// "1234.56". The rightmost separator is decimal only when followed by one or
// two digits; otherwise every separator is a thousands separator.
func normalizeAmount(s string) (string, bool) {
last := -1
for i := 0; i < len(s); i++ {
if s[i] == '.' || s[i] == ',' {
last = i
}
}
var dec byte
if last >= 0 {
digits := 0
for i := last + 1; i < len(s); i++ {
if s[i] < '0' || s[i] > '9' {
return "", false
}
digits++
}
if digits == 1 || digits == 2 {
dec = s[last]
}
}
var b strings.Builder
for i := 0; i < len(s); i++ {
switch c := s[i]; {
case c >= '0' && c <= '9':
b.WriteByte(c)
case (c == '.' || c == ',') && c == dec:
b.WriteByte('.')
case c == '.' || c == ',':
// thousands separator
default:
return "", false
}
}
v, err := strconv.ParseFloat(b.String(), 64)
if err != nil {
return "", false
}
return strconv.FormatFloat(v, 'f', 2, 64), true
}
+175
View File
@@ -0,0 +1,175 @@
package main
import "testing"
func TestExtractAmount(t *testing.T) {
tests := []struct {
name, text, want string
}{
{
name: "rechnungsbetrag de format",
text: "Rechnungsbetrag: 1.234,56 €",
want: "1234.56",
},
{
name: "rechnungsbetrag en thousands and dot",
text: "Rechnungsbetrag: 1,234.56",
want: "1234.56",
},
{
name: "rechnungsbetrag plain dot",
text: "Rechnungsbetrag: 1234.56",
want: "1234.56",
},
{
name: "rechnungsbetrag de comma only",
text: "Rechnungsbetrag: 1234,56",
want: "1234.56",
},
{
name: "currency suffix eur",
text: "Rechnungsbetrag: 1.234,56 EUR",
want: "1234.56",
},
{
name: "zu zahlender betrag wins over rechnungsbetrag",
text: "Zu zahlender Betrag: 10,00\nRechnungsbetrag: 99,00",
want: "10.00",
},
{
name: "rechnungsbetrag wins over endbetrag",
text: "Endbetrag: 20,00\nRechnungsbetrag: 30,00",
want: "30.00",
},
{
name: "bruttobetrag wins over bare betrag",
text: "Bruttobetrag: 50,00\nBetrag: 10,00",
want: "50.00",
},
{
name: "gesamtbetrag wins over bare betrag",
text: "Gesamtbetrag: 80,00\nBetrag: 10,00",
want: "80.00",
},
{
name: "last occurrence of same label wins",
text: "Rechnungsbetrag: 10,00\nRechnungsbetrag: 20,00",
want: "20.00",
},
{
name: "endbetrag fallback",
text: "Endbetrag: 42,00",
want: "42.00",
},
{
name: "zahlbetrag fallback without colon",
text: "Zahlbetrag 7,50 €",
want: "7.50",
},
{
name: "rechnungsendbetrag beats endbetrag",
text: "Rechnungsendbetrag: 12,00\nEndbetrag: 13,00",
want: "12.00",
},
{
name: "dkv style total line",
text: "Kundenbezogene Daten\n» TOTAL: 123,45 100,00 23,45 123,45\n",
want: "123.45",
},
{
name: "dkv gesamtsummenaufstellung grand total after marker",
text: "» TOTAL: 111,11 100,00 11,11 111,11\n" +
"» TOTAL: 222,22 200,00 22,22 222,22\n" +
"Gesamtsummenaufstellung\n" +
"Netto 240,00\n" +
"MwSt 47,25\n" +
"» 287,25\n",
want: "287.25",
},
{
name: "dkv gesamtsummenaufstellung total on next line",
text: "» TOTAL: 111,11\nGesamtsummenaufstellung\n»\n287,25\n",
want: "287.25",
},
{
name: "tax rate with percent sign is not an amount",
text: "Betrag: 19,00 % MwSt",
want: "",
},
{
name: "mehrwertsteuer rate is not an amount",
text: "Gesamtsumme: 19,00% MwSt",
want: "",
},
{
name: "steuer word on the number line rejects it",
text: "Betrag: 2,83 Steuer",
want: "",
},
{
name: "rejected primary falls back to a usable label",
text: "Gesamtsumme: 19,00 % MwSt\nEndbetrag: 42,00",
want: "42.00",
},
{
name: "labeled zu zahlender betrag beats unlabeled larger number",
text: "unlabeled 999,99\nZu zahlender Betrag: 10,00",
want: "10.00",
},
{
name: "labeled zu zahlender betrag beats lower label larger number",
text: "Endbetrag: 999,99\nZu zahlender Betrag: 10,00",
want: "10.00",
},
{
name: "diashop style gesamtsumme with comment",
text: "Zwischensumme\n12,34 €\nZwischensumme\n12,34 €\nVersand & Bearbeitung\n4,95 €\nGesamtsumme (inkl. Steuern)\n17,29 €\n",
want: "17.29",
},
{
name: "diashop picks inclusive total last",
text: "Gesamtsumme (exkl. Steuern)\n12,34 €\nGesamtsumme (inkl. Steuern)\n17,29 €",
want: "17.29",
},
{
name: "no label",
text: "some text without any amount label 12,34",
want: "",
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
if got := extractAmount(tc.text); got != tc.want {
t.Fatalf("extractAmount()=%q want %q", got, tc.want)
}
})
}
}
func TestNormalizeAmount(t *testing.T) {
tests := []struct {
in string
want string
ok bool
}{
{"1.234,56", "1234.56", true},
{"1,234.56", "1234.56", true},
{"1234.56", "1234.56", true},
{"1234,56", "1234.56", true},
{"1.234.567,89", "1234567.89", true},
{"1,234,567.89", "1234567.89", true},
{"1.234", "1234.00", true},
{"12,5", "12.50", true},
{"12", "12.00", true},
{"", "0.00", false},
}
for _, tc := range tests {
got, ok := normalizeAmount(tc.in)
if ok != tc.ok {
t.Fatalf("normalizeAmount(%q) ok=%v want %v", tc.in, ok, tc.ok)
}
if ok && got != tc.want {
t.Fatalf("normalizeAmount(%q)=%q want %q", tc.in, got, tc.want)
}
}
}
+125
View File
@@ -0,0 +1,125 @@
---
type: reference
status: current
related:
- README.md
- file_es.go
---
# Elasticsearch — полнотекстовый поиск OnlyOffice
## Что это
Полнотекстовый поиск OnlyOffice Workspace работает на **Elasticsearch**.
Клиент на сервере — NEST. Индекс — имя таблицы.
Для файлов индекс `files_file`:
| поле | тип | смысл |
|------|-----|-------|
| `id` | integer | id файла (тот же, что в REST/Documents) |
| `title` | text (`whitespacecustom`) | имя файла |
| `tenantId` | integer | тенант (портал) |
| `folders` | nested | список папок: `folderId` (строка), `id`, `tenantId` |
| `document.attachment.content` | text (`document`) | извлеченный текст (ingest-attachment) |
| `document.attachment.content_type` | text | MIME |
Важно:
- Живой сервер — **Elasticsearch 7.16.3**, кластер `elasticsearch`.
- REST `GET /api/2.0/files/@search/{query}` ищет **только по имени в БД**
(`fileDao.Search`), ES не задействует. Для поиска по содержимому нужен
прямой ES — это и делает `oo search`.
- `title` analyzer `whitespacecustom` режет по пробелам и lower-case. Полное
имя файла — один токен (`Rechnung-4711.pdf`), поэтому поиск по имени ищет
слово целиком, а не подстроку.
- `document.attachment.content` заполняется **только для Office-форматов**
(docx / xlsx / pptx). У PDF/txt, залитых через API, контент не извлекается.
- Индексация асинхронная (TeamLabSvc) — файл появляется в ES не мгновенно.
## Доступ
ES слушает `127.0.0.1:9200` **внутри** VM OnlyOffice. Снаружи порт закрыт,
SSH в VM открыт на хосте как `127.0.0.1:32` (контейнер `onlyoffice-v2`,
QEMU). Схема — SSH-туннель.
```bash
# из корня go-onlyoffice (ключ и хост — как в infra-доках)
ssh -f -N -o ControlMaster=no -o ControlPath=none \
-p 32 -i ~/.ssh/id_ed25519 \
-L 9200:127.0.0.1:9200 root@127.0.0.1
curl -s http://127.0.0.1:9200/ | head # tagline + version
curl -s 'http://127.0.0.1:9200/_cat/indices?h=index,docs.count'
```
`-o ControlMaster=no -o ControlPath=none` обязательны: иначе forward уходит
в persistent master-соединение из `~/.ssh/config` и порт остаётся занят.
Проверить, что туннель жив:
```bash
curl -s http://127.0.0.1:9200/files_file/_count
```
## Переменные
| env | default | смысл |
|-----|---------|-------|
| `ONLYOFFICE_ES_URL` | — (обязателен) | `scheme://host:port` ES |
| `ONLYOFFICE_ES_INDEX` | `files_file` | индекс |
| `ONLYOFFICE_TENANT` | пусто (все) | фильтр `tenantId` |
Имена — в [`.env.example`](../.env.example). Секретов нет: ES без пароля.
## CLI
```bash
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 oo search "Rechnung"
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 oo search "Mahngebühr" --content
oo search "Rechnung" --folder 649 --limit 50 --json
```
Флаги: `--content` (искать и по тексту), `--folder ID` (папка
`folders.folderId`), `--limit N` (по умолчанию 20, максимум 200),
`--json` = `-o json`.
## Библиотека
`file_es.go` — `ESSearcher` (`Name() = "elasticsearch"`), прямой ES REST на
stdlib `net/http`:
```go
es, _ := onlyoffice.NewESSearcher(onlyoffice.ESConfigFromEnv())
hits, _ := es.Search(ctx, onlyoffice.SearchQuery{
Text: "Rechnung", InContent: true, Limit: 20,
})
```
Запрос: `multi_match` по `title^2` (+ `document.attachment.content` при
`InContent`), фильтры `tenantId` и `folders.folderId`, `_source`
id/title/folders, `highlight` для фрагмента. Ответ → `[]SearchHit` (модель из
эпика #34; пока объявлена в `file_es.go`, переедет в `file_core.go` с F1 #35).
## Тесты
```bash
# unit — чистые builders/парсеры, без сети
go test ./ -run ES
# integration — нужен ONLYOFFICE_ES_URL (+ креды REST для залива)
set -a; . .env; set +a
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 ONLYOFFICE_TENANT=1 \
go test -tags=integration -run TestIntegrationESSearch -v .
```
Интеграционный тест заливает временный xlsx (в имени и в ячейке — уникальные
токены), ждёт индексации, проверяет поиск по имени и по содержимому, затем
удаляет проект.
## Грабли
- `locale`/версия ES: 7.16.3, `_search` совместим с REST 7.x.
- ES без auth и слушает только localhost — туннель обязателен.
- Фильтр `tenantId` сузит выдачу; без него видны документы всех тенантов.
- Поиск по содержимому PDF, залитых через API, не работает (нет
`attachment.content`) — только Office-форматы.
+199
View File
@@ -0,0 +1,199 @@
package onlyoffice
// Canonical file model and the backend-agnostic store interface. REST
// (files.go), WebDAV (files_webdav.go) and future backends (PostgreSQL,
// Elasticsearch) implement FileStore/Searcher so callers stop depending on a
// concrete transport. This file holds only types and pure conversions — no IO.
import (
"context"
"io"
"mime"
"path/filepath"
"strconv"
"strings"
"time"
)
// Kind distinguishes files from folders in the canonical model.
type Kind int
const (
File Kind = iota
Folder
)
// String renders the kind for logs and table output.
func (k Kind) String() string {
switch k {
case File:
return "file"
case Folder:
return "folder"
default:
return "unknown"
}
}
// Provider names for the FileStore adapters.
const (
ProviderREST = "rest"
ProviderDAV = "dav"
)
// Entry is the backend-independent representation of a document or folder.
// Fields that a backend cannot supply stay at their zero value.
type Entry struct {
ID string
ParentID string
Title string
Kind Kind
Size int64
MIME string
Created time.Time
Modified time.Time
Version int
Provider string
}
// FileStore is the operation surface every file backend implements.
type FileStore interface {
Name() string
List(ctx context.Context, parentID string) ([]Entry, error)
Stat(ctx context.Context, id string) (Entry, error)
CreateFolder(ctx context.Context, parentID, title string) (Entry, error)
Upload(ctx context.Context, parentID, title string, r io.Reader) (Entry, error)
Download(ctx context.Context, id string, w io.Writer) (int64, error)
Move(ctx context.Context, ids []string, parentID string) error
Copy(ctx context.Context, ids []string, parentID string) error
Rename(ctx context.Context, id, title string) error
Delete(ctx context.Context, ids []string) error
}
// SearchQuery narrows a Searcher request. InContent asks the backend to match
// document bodies, not just titles.
type SearchQuery struct {
Text string
InContent bool
FolderID string
Extensions []string
Limit int
}
// SearchHit is one Searcher result: the matching entry plus backend-specific
// ranking metadata.
type SearchHit struct {
Entry
Score float64
Highlight string
Path []string
}
// Searcher is the optional content/name search surface. Only some backends
// (for example Elasticsearch) provide it.
type Searcher interface {
Search(ctx context.Context, q SearchQuery) ([]SearchHit, error)
Name() string
}
// FileStore returns the adapter for a backend name: ProviderREST (default) or
// ProviderDAV. Unknown or empty names select the REST backend. The full facade
// (backend composition) is deliberately left to a later change.
func (c *Client) FileStore(backend string) FileStore {
switch strings.ToLower(strings.TrimSpace(backend)) {
case ProviderDAV, "webdav":
return &davStore{c: c}
default:
return &restStore{c: c}
}
}
// Files returns the default (REST) file store.
func (c *Client) Files() FileStore { return c.FileStore(ProviderREST) }
// retryStoreOp runs one store operation under the shared deterministic
// transient-error policy (429/502/503/504).
func retryStoreOp(ctx context.Context, fn func() error) error {
return DoRetry(ctx, DefaultRetryPolicy(), fn)
}
// FileEntryToEntry converts a Files-module file row to the canonical model.
func FileEntryToEntry(f *FileEntry, provider string) Entry {
e := Entry{Kind: File, Provider: provider}
if f == nil {
return e
}
if f.ID != nil {
e.ID = f.ID.String()
}
e.ParentID = FileFolderID(f)
if f.Title != nil {
e.Title = *f.Title
}
if f.ContentLength != nil {
e.Size = parseContentLength(*f.ContentLength)
}
exst := ""
if f.FileExst != nil {
exst = *f.FileExst
}
e.MIME = mimeForTitle(e.Title, exst)
if f.Updated != nil {
e.Modified = *f.Updated
}
return e
}
// DavFileToEntry converts a WebDAV file row to the canonical model.
func DavFileToEntry(f DavFile, provider string) Entry {
return Entry{
ID: f.ID,
Title: f.Title,
Kind: File,
Size: f.Size,
MIME: mimeForTitle(f.Title, ""),
Modified: f.ModTime(),
Provider: provider,
}
}
// DavFolderToEntry converts a WebDAV folder row to the canonical model.
func DavFolderToEntry(f DavFolder, provider string) Entry {
return Entry{
ID: f.ID,
ParentID: f.ParentID,
Title: f.Title,
Kind: Folder,
Modified: f.ModTime(),
Provider: provider,
}
}
// parseContentLength reads the leading integer of an OnlyOffice contentLength
// string (the API sometimes appends a unit, e.g. "12345 b").
func parseContentLength(s string) int64 {
fields := strings.Fields(s)
if len(fields) == 0 {
return 0
}
n, err := strconv.ParseInt(fields[0], 10, 64)
if err != nil {
return 0
}
return n
}
// mimeForTitle derives a MIME type from an explicit extension or the title.
func mimeForTitle(title, exst string) string {
ext := strings.TrimSpace(exst)
if ext == "" {
ext = filepath.Ext(title)
}
if ext == "" {
return ""
}
if !strings.HasPrefix(ext, ".") {
ext = "." + ext
}
return mime.TypeByExtension(strings.ToLower(ext))
}
+193
View File
@@ -0,0 +1,193 @@
package onlyoffice
import (
"encoding/json"
"strings"
"testing"
"time"
)
func TestFileEntryToEntry(t *testing.T) {
id := json.Number("42")
title := "invoice.pdf"
exst := ".pdf"
size := "12345"
parent := json.Number("7")
updated := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
f := &FileEntry{
ID: &id,
Title: &title,
FileExst: &exst,
ContentLength: &size,
FolderID: &parent,
Updated: &updated,
}
e := FileEntryToEntry(f, ProviderREST)
if e.ID != "42" {
t.Errorf("ID = %q, want 42", e.ID)
}
if e.ParentID != "7" {
t.Errorf("ParentID = %q, want 7", e.ParentID)
}
if e.Title != title {
t.Errorf("Title = %q, want %q", e.Title, title)
}
if e.Kind != File {
t.Errorf("Kind = %v, want file", e.Kind)
}
if e.Size != 12345 {
t.Errorf("Size = %d, want 12345", e.Size)
}
if e.MIME != "application/pdf" {
t.Errorf("MIME = %q, want application/pdf", e.MIME)
}
if !e.Modified.Equal(updated) {
t.Errorf("Modified = %v, want %v", e.Modified, updated)
}
if e.Provider != ProviderREST {
t.Errorf("Provider = %q, want %q", e.Provider, ProviderREST)
}
}
func TestFileEntryToEntryNil(t *testing.T) {
e := FileEntryToEntry(nil, ProviderDAV)
if e.Kind != File {
t.Errorf("Kind = %v, want file", e.Kind)
}
if e.ID != "" || e.Title != "" {
t.Errorf("nil entry should be empty: %+v", e)
}
if e.Provider != ProviderDAV {
t.Errorf("Provider = %q, want %q", e.Provider, ProviderDAV)
}
}
func TestFileEntryToEntrySizeFormats(t *testing.T) {
cases := map[string]int64{
"12345": 12345,
"12345 b": 12345,
"0": 0,
"": 0,
"notanum": 0,
}
for in, want := range cases {
got := parseContentLength(in)
if got != want {
t.Errorf("parseContentLength(%q) = %d, want %d", in, got, want)
}
}
}
func TestDavFileToEntry(t *testing.T) {
f := DavFile{
ID: "9",
Title: "note.txt",
Size: 10,
Updated: "2026-01-02T03:04:05.0000000+01:00",
}
e := DavFileToEntry(f, ProviderDAV)
if e.ID != "9" || e.Title != "note.txt" {
t.Errorf("identity mismatch: %+v", e)
}
if e.Kind != File {
t.Errorf("Kind = %v, want file", e.Kind)
}
if e.Size != 10 {
t.Errorf("Size = %d, want 10", e.Size)
}
if !strings.HasPrefix(e.MIME, "text/plain") {
t.Errorf("MIME = %q, want text/plain*", e.MIME)
}
if e.Modified.IsZero() {
t.Error("Modified not parsed")
}
if e.Provider != ProviderDAV {
t.Errorf("Provider = %q, want %q", e.Provider, ProviderDAV)
}
}
func TestDavFolderToEntry(t *testing.T) {
f := DavFolder{
ID: "5",
Title: "inbox",
ParentID: "1",
Updated: "2026-01-02T03:04:05.0000000+01:00",
}
e := DavFolderToEntry(f, ProviderDAV)
if e.ID != "5" || e.Title != "inbox" || e.ParentID != "1" {
t.Errorf("identity mismatch: %+v", e)
}
if e.Kind != Folder {
t.Errorf("Kind = %v, want folder", e.Kind)
}
if e.MIME != "" {
t.Errorf("folder MIME = %q, want empty", e.MIME)
}
if e.Modified.IsZero() {
t.Error("Modified not parsed")
}
}
func TestEntriesFromFolderMap(t *testing.T) {
m := map[string]any{
"files": []any{
map[string]any{"id": float64(42), "title": "a.pdf", "pureContentLength": float64(7)},
},
"folders": []any{
map[string]any{"id": float64(7), "title": "sub", "parentId": float64(1)},
},
}
entries, err := entriesFromFolderMap(m, ProviderREST)
if err != nil {
t.Fatalf("entriesFromFolderMap: %v", err)
}
if len(entries) != 2 {
t.Fatalf("got %d entries, want 2: %+v", len(entries), entries)
}
byID := map[string]Entry{}
for _, e := range entries {
byID[e.ID] = e
}
if got := byID["42"]; got.Kind != File || got.Size != 7 || got.Title != "a.pdf" {
t.Errorf("file entry = %+v", got)
}
if got := byID["7"]; got.Kind != Folder || got.ParentID != "1" || got.Title != "sub" {
t.Errorf("folder entry = %+v", got)
}
}
func TestEntriesFromFolderMapNil(t *testing.T) {
entries, err := entriesFromFolderMap(nil, ProviderREST)
if err != nil || entries != nil {
t.Fatalf("got %v, %v; want nil, nil", entries, err)
}
}
func TestKindString(t *testing.T) {
if File.String() != "file" || Folder.String() != "folder" {
t.Errorf("kind strings: %q %q", File.String(), Folder.String())
}
if Kind(9).String() != "unknown" {
t.Errorf("unknown kind = %q", Kind(9).String())
}
}
func TestClientFileStoreSelection(t *testing.T) {
c := NewClient(Credentials{})
if got := c.FileStore(ProviderDAV).Name(); got != ProviderDAV {
t.Errorf("FileStore(dav).Name() = %q", got)
}
if got := c.FileStore("webdav").Name(); got != ProviderDAV {
t.Errorf("FileStore(webdav).Name() = %q", got)
}
if got := c.FileStore(ProviderREST).Name(); got != ProviderREST {
t.Errorf("FileStore(rest).Name() = %q", got)
}
if got := c.FileStore("").Name(); got != ProviderREST {
t.Errorf("FileStore(\"\").Name() = %q", got)
}
if got := c.Files().Name(); got != ProviderREST {
t.Errorf("Files().Name() = %q", got)
}
}
+189
View File
@@ -0,0 +1,189 @@
package onlyoffice
// davStore implements FileStore on top of the Documents/WebDAV methods in
// files_webdav.go. The Documents fileops calls need folder and file ids
// separated, so ids are classified through Stat before move/copy/rename/delete.
import (
"bytes"
"context"
"fmt"
"io"
)
// davStore is a FileStore over the WebDAV-oriented Documents API.
type davStore struct{ c *Client }
// Name reports the backend name.
func (s *davStore) Name() string { return ProviderDAV }
// List returns the files and folders directly below parentID.
func (s *davStore) List(ctx context.Context, parentID string) ([]Entry, error) {
var out []Entry
err := retryStoreOp(ctx, func() error {
l, err := s.c.ListDavFolder(ctx, parentID)
if err != nil {
return err
}
entries := make([]Entry, 0, len(l.Folders)+len(l.Files))
for _, f := range l.Folders {
entries = append(entries, DavFolderToEntry(f, ProviderDAV))
}
for _, f := range l.Files {
entries = append(entries, DavFileToEntry(f, ProviderDAV))
}
out = entries
return nil
})
return out, err
}
// Stat resolves a folder or file entry by id. A folder answers ListDavFolder
// with its own metadata in Current; otherwise the file metadata API is used.
func (s *davStore) Stat(ctx context.Context, id string) (Entry, error) {
return s.stat(ctx, id)
}
// CreateFolder creates a subfolder under parentID.
func (s *davStore) CreateFolder(ctx context.Context, parentID, title string) (Entry, error) {
var out Entry
err := retryStoreOp(ctx, func() error {
f, err := s.c.CreateDavFolder(ctx, parentID, title)
if err != nil {
return err
}
if f == nil {
return fmt.Errorf("onlyoffice: dav store: empty create-folder response")
}
out = DavFolderToEntry(*f, ProviderDAV)
return nil
})
return out, err
}
// Upload streams r into parentID as title. The reader is buffered once so a
// retry re-sends the same bytes instead of an exhausted stream.
func (s *davStore) Upload(ctx context.Context, parentID, title string, r io.Reader) (Entry, error) {
data, err := io.ReadAll(r)
if err != nil {
return Entry{}, err
}
var out Entry
err = retryStoreOp(ctx, func() error {
f, err := s.c.UploadDavFile(ctx, parentID, title, bytes.NewReader(data))
if err != nil {
return err
}
if f == nil {
return fmt.Errorf("onlyoffice: dav store: empty upload response")
}
out = DavFileToEntry(*f, ProviderDAV)
return nil
})
return out, err
}
// Download streams the file bytes into w.
func (s *davStore) Download(ctx context.Context, id string, w io.Writer) (int64, error) {
var n int64
err := retryStoreOp(ctx, func() error {
var e error
n, e = s.c.DownloadDavFile(ctx, id, w)
return e
})
return n, err
}
// Move moves ids into parentID, splitting folders from files.
func (s *davStore) Move(ctx context.Context, ids []string, parentID string) error {
folders, files, err := s.split(ctx, ids)
if err != nil {
return err
}
if len(folders) == 0 && len(files) == 0 {
return nil
}
return retryStoreOp(ctx, func() error {
return s.c.MoveDavItems(ctx, folders, files, parentID)
})
}
// Copy copies ids into parentID, splitting folders from files.
func (s *davStore) Copy(ctx context.Context, ids []string, parentID string) error {
folders, files, err := s.split(ctx, ids)
if err != nil {
return err
}
if len(folders) == 0 && len(files) == 0 {
return nil
}
return retryStoreOp(ctx, func() error {
return s.c.CopyDavItems(ctx, folders, files, parentID)
})
}
// Rename renames a folder or file.
func (s *davStore) Rename(ctx context.Context, id, title string) error {
e, err := s.stat(ctx, id)
if err != nil {
return err
}
return retryStoreOp(ctx, func() error {
if e.Kind == Folder {
return s.c.RenameDavFolder(ctx, id, title)
}
return s.c.RenameDavFile(ctx, id, title)
})
}
// Delete removes ids, splitting folders from files.
func (s *davStore) Delete(ctx context.Context, ids []string) error {
folders, files, err := s.split(ctx, ids)
if err != nil {
return err
}
if len(folders) == 0 && len(files) == 0 {
return nil
}
return retryStoreOp(ctx, func() error {
return s.c.DeleteDavItems(ctx, folders, files)
})
}
// stat resolves a single id to a folder or file Entry.
func (s *davStore) stat(ctx context.Context, id string) (Entry, error) {
var out Entry
err := retryStoreOp(ctx, func() error {
if l, err := s.c.ListDavFolder(ctx, id); err == nil {
if l != nil && l.Current.ID != "" && l.Current.ID == id {
out = DavFolderToEntry(l.Current, ProviderDAV)
return nil
}
} else if Transient(err) {
return err
}
f, err := s.c.GetFile(ctx, id)
if err != nil {
return err
}
out = FileEntryToEntry(f, ProviderDAV)
return nil
})
return out, err
}
// split classifies ids into folder and file id lists.
func (s *davStore) split(ctx context.Context, ids []string) (folders, files []string, err error) {
for _, id := range ids {
e, err := s.stat(ctx, id)
if err != nil {
return nil, nil, err
}
if e.Kind == Folder {
folders = append(folders, id)
} else {
files = append(files, id)
}
}
return folders, files, nil
}
+285
View File
@@ -0,0 +1,285 @@
package onlyoffice
// Elasticsearch backend of the unified file client (epic #34, F3 #37).
//
// OnlyOffice full-text search runs on Elasticsearch (index `files_file`, NEST
// client on the server). The REST endpoint GET /api/2.0/files/@search/{query}
// only searches file names in the database, so content search needs a direct
// ES query. The live server is Elasticsearch 7.16.3; the request shape below
// is plain REST and stays stdlib-only, matching the repo's no-extra-deps rule.
import (
"bytes"
"context"
"encoding/json"
"fmt"
"io"
"net/http"
"os"
"regexp"
"strconv"
"strings"
"time"
)
// The canonical model (Kind, Entry, SearchQuery, SearchHit, Searcher) lives in
// file_core.go (F1 #35).
const (
defaultESIndex = "files_file"
defaultESLimit = 20
maxESLimit = 200
maxESResponseSize = 8 << 20
)
// ESConfig configures the direct Elasticsearch searcher.
type ESConfig struct {
URL string // scheme://host:port of the ES HTTP endpoint
Index string // index name, default files_file
Tenant string // tenantId filter, empty means all tenants
}
// ESConfigFromEnv reads ONLYOFFICE_ES_URL, ONLYOFFICE_ES_INDEX (default
// files_file) and ONLYOFFICE_TENANT. The library never loads dotfiles — the
// CLI does that.
func ESConfigFromEnv() ESConfig {
return ESConfig{
URL: strings.TrimRight(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")), "/"),
Index: firstNonEmpty(os.Getenv("ONLYOFFICE_ES_INDEX"), defaultESIndex),
Tenant: strings.TrimSpace(os.Getenv("ONLYOFFICE_TENANT")),
}
}
// ESSearcher queries OnlyOffice's Elasticsearch index directly for file name
// and document content.
type ESSearcher struct {
cfg ESConfig
http *http.Client
}
// NewESSearcher returns a searcher for the OnlyOffice Elasticsearch index.
// The URL is required; an empty index falls back to files_file.
func NewESSearcher(cfg ESConfig) (*ESSearcher, error) {
if strings.TrimSpace(cfg.URL) == "" {
return nil, fmt.Errorf("onlyoffice: elasticsearch URL is empty (set ONLYOFFICE_ES_URL)")
}
cfg.URL = strings.TrimRight(cfg.URL, "/")
if cfg.Index == "" {
cfg.Index = defaultESIndex
}
return &ESSearcher{cfg: cfg, http: &http.Client{Timeout: 30 * time.Second}}, nil
}
// Name implements Searcher.
func (s *ESSearcher) Name() string { return "elasticsearch" }
// Search runs a multi_match over title (and, when q.InContent is set,
// document.attachment.content), filtered by tenant and optional folder.
func (s *ESSearcher) Search(ctx context.Context, q SearchQuery) ([]SearchHit, error) {
q.Text = strings.TrimSpace(q.Text)
if q.Text == "" {
return nil, fmt.Errorf("onlyoffice: empty search query")
}
body, err := json.Marshal(esSearchRequest(q, s.cfg.Tenant))
if err != nil {
return nil, fmt.Errorf("onlyoffice: build elasticsearch query: %w", err)
}
endpoint := s.cfg.URL + "/" + s.cfg.Index + "/_search"
req, err := http.NewRequestWithContext(ctx, http.MethodPost, endpoint, bytes.NewReader(body))
if err != nil {
return nil, err
}
req.Header.Set("Content-Type", "application/json")
req.Header.Set("Accept", "application/json")
resp, err := s.http.Do(req)
if err != nil {
return nil, fmt.Errorf("onlyoffice: elasticsearch search: %w", err)
}
defer resp.Body.Close()
raw, err := io.ReadAll(io.LimitReader(resp.Body, maxESResponseSize))
if err != nil {
return nil, err
}
if resp.StatusCode >= 400 {
return nil, fmt.Errorf("onlyoffice: elasticsearch search: %d %s", resp.StatusCode, truncate(string(raw), 400))
}
return parseESSearchResponse(raw)
}
// esSearchRequest builds the ES query body. Pure, so it is unit-tested.
func esSearchRequest(q SearchQuery, tenant string) esRequest {
limit := q.Limit
if limit <= 0 {
limit = defaultESLimit
}
if limit > maxESLimit {
limit = maxESLimit
}
fields := []string{"title^2"}
if q.InContent {
fields = append(fields, "document.attachment.content")
}
must := []esClause{{MultiMatch: &esMultiMatch{Query: q.Text, Fields: fields}}}
var filter []esClause
if t := strings.TrimSpace(tenant); t != "" {
filter = append(filter, esClause{Term: map[string]any{"tenantId": numericOrString(t)}})
}
if f := strings.TrimSpace(q.FolderID); f != "" {
filter = append(filter, esClause{Term: map[string]any{"folders.folderId": f}})
}
for _, ext := range normalizeExtensions(q.Extensions) {
filter = append(filter, esClause{Wildcard: map[string]any{"title": "*." + ext}})
}
highlightFields := map[string]struct{}{"title": {}}
if q.InContent {
highlightFields["document.attachment.content"] = struct{}{}
}
return esRequest{
Size: limit,
Source: []string{"id", "title", "folders"},
Query: esQuery{Bool: esBool{Must: must, Filter: filter}},
Highlight: esHighlight{PreTags: []string{"<em>"}, PostTags: []string{"</em>"}, Fields: highlightFields},
}
}
// normalizeExtensions lowercases, trims leading dots and drops empties.
func normalizeExtensions(exts []string) []string {
out := make([]string, 0, len(exts))
seen := map[string]bool{}
for _, e := range exts {
e = strings.ToLower(strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(e), ".")))
if e == "" || seen[e] {
continue
}
seen[e] = true
out = append(out, e)
}
return out
}
// numericOrString keeps an integer-looking filter value numeric (tenantId is
// a long) and leaves anything else as a string (folderId is a text token).
func numericOrString(s string) any {
if n, err := strconv.ParseInt(s, 10, 64); err == nil {
return n
}
return s
}
// esRequest is the subset of the ES query DSL this client emits.
type esRequest struct {
Size int `json:"size"`
Source []string `json:"_source"`
Query esQuery `json:"query"`
Highlight esHighlight `json:"highlight"`
}
type esQuery struct {
Bool esBool `json:"bool"`
}
type esBool struct {
Must []esClause `json:"must,omitempty"`
Filter []esClause `json:"filter,omitempty"`
}
type esClause struct {
MultiMatch *esMultiMatch `json:"multi_match,omitempty"`
Term map[string]any `json:"term,omitempty"`
Wildcard map[string]any `json:"wildcard,omitempty"`
}
type esMultiMatch struct {
Query string `json:"query"`
Fields []string `json:"fields"`
}
type esHighlight struct {
PreTags []string `json:"pre_tags,omitempty"`
PostTags []string `json:"post_tags,omitempty"`
Fields map[string]struct{} `json:"fields"`
}
// esResponse is the subset of an ES search response we consume.
type esResponse struct {
Took int `json:"took"`
Hits struct {
Total struct {
Value int `json:"value"`
Relation string `json:"relation"`
} `json:"total"`
Hits []esResponseHit `json:"hits"`
} `json:"hits"`
}
type esResponseHit struct {
ID string `json:"_id"`
Score float64 `json:"_score"`
Source struct {
ID int `json:"id"`
Title string `json:"title"`
Folders []struct {
FolderID string `json:"folderId"`
ID int `json:"id"`
} `json:"folders"`
} `json:"_source"`
Highlight map[string][]string `json:"highlight"`
}
// parseESSearchResponse converts an ES search response into SearchHit values.
// Pure, so it is unit-tested.
func parseESSearchResponse(raw []byte) ([]SearchHit, error) {
var r esResponse
if err := json.Unmarshal(raw, &r); err != nil {
return nil, fmt.Errorf("onlyoffice: decode elasticsearch response: %w", err)
}
hits := make([]SearchHit, 0, len(r.Hits.Hits))
for _, h := range r.Hits.Hits {
id := strconv.Itoa(h.Source.ID)
if h.Source.ID == 0 {
id = h.ID
}
var parent string
path := make([]string, 0, len(h.Source.Folders))
for i, f := range h.Source.Folders {
path = append(path, f.FolderID)
if i == 0 {
parent = f.FolderID
}
}
hits = append(hits, SearchHit{
Entry: Entry{
ID: id,
ParentID: parent,
Title: h.Source.Title,
Kind: File,
Provider: "elasticsearch",
},
Score: h.Score,
Highlight: esHighlightText(h.Highlight),
Path: path,
})
}
return hits, nil
}
var esHighlightTag = regexp.MustCompile(`</?em[^>]*>`)
// esHighlightText flattens a highlight map into one plain-text snippet,
// preferring the content fragment over the title.
func esHighlightText(hl map[string][]string) string {
for _, key := range []string{"document.attachment.content", "title"} {
frags := hl[key]
if len(frags) == 0 {
continue
}
clean := make([]string, 0, len(frags))
for _, f := range frags {
clean = append(clean, esHighlightTag.ReplaceAllString(f, ""))
}
return strings.Join(clean, " … ")
}
return ""
}
+135
View File
@@ -0,0 +1,135 @@
//go:build integration
package onlyoffice
import (
"context"
"os"
"path/filepath"
"strconv"
"strings"
"testing"
"time"
"github.com/xuri/excelize/v2"
)
// TestIntegrationESSearch uploads a throwaway workbook and verifies that the
// direct Elasticsearch search finds it by file name and by content.
//
// The content index (document.attachment.content) is only populated for Office
// formats (docx/xlsx/pptx), so the fixture is an xlsx whose cell carries a
// unique token. Requires ONLYOFFICE_ES_URL (a reachable ES endpoint — in the
// current setup a tunnel to the ES inside the OnlyOffice VM, see
// docs/elasticsearch.md) plus the regular REST credentials for the upload.
// Skips when either is missing.
func TestIntegrationESSearch(t *testing.T) {
esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL"))
if esURL == "" {
t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test")
}
c := liveClient(t)
t.Cleanup(func() { cleanupTestProjects(t, c) })
stamp := time.Now().UTC().Format("20060102-150405")
nameToken := "goesname" + stamp
contentToken := "goescontent" + stamp
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
defer cancel()
project, err := c.CreateProject(NewProjectRequest{
Title: testProjectPrefix + "es-" + stamp,
Description: "go-onlyoffice elasticsearch integration",
})
if err != nil {
t.Fatalf("CreateProject: %v", err)
}
if project.ID == nil {
t.Fatal("created project without id")
}
pid := strconv.Itoa(*project.ID)
title := nameToken + ".xlsx"
localPath := filepath.Join(t.TempDir(), title)
book := excelize.NewFile()
if err := book.SetCellValue("Sheet1", "A1", "OnlyOffice Elasticsearch content fixture "+contentToken); err != nil {
t.Fatalf("SetCellValue: %v", err)
}
if err := book.SaveAs(localPath); err != nil {
t.Fatalf("SaveAs: %v", err)
}
entry, err := c.UploadProjectFile(ctx, pid, localPath)
if err != nil {
t.Fatalf("UploadProjectFile: %v", err)
}
fileID := strconv.Itoa(int(FileEntryNumericID(entry)))
if fileID == "0" {
t.Fatalf("upload returned no file id: %+v", entry)
}
es, err := NewESSearcher(ESConfig{
URL: esURL,
Index: os.Getenv("ONLYOFFICE_ES_INDEX"),
Tenant: os.Getenv("ONLYOFFICE_TENANT"),
})
if err != nil {
t.Fatalf("NewESSearcher: %v", err)
}
// Indexing is asynchronous on the server; poll until the file shows up.
// The server's title analyzer splits on whitespace, so the name query is
// the full file name token (including extension), as a user would type it.
nameHit := waitForHit(t, ctx, es, SearchQuery{Text: title}, fileID)
if nameHit.Title != title {
t.Errorf("name hit title = %q, want %q", nameHit.Title, title)
}
contentHit := waitForHit(t, ctx, es, SearchQuery{Text: contentToken, InContent: true}, fileID)
if contentHit.Highlight == "" {
t.Error("content hit has no highlight fragment")
}
if !strings.Contains(contentHit.Title, nameToken) {
t.Errorf("content hit title = %q, want the uploaded workbook", contentHit.Title)
}
// The content token is absent from the title, so a name-only search must
// not return the file — this proves the content field is really queried.
if hits := searchQuiet(t, es, SearchQuery{Text: contentToken}); len(hits) != 0 {
t.Errorf("name-only search for content token returned %d hits, want 0", len(hits))
}
}
// waitForHit polls ES until the file with fileID appears and returns that hit.
func waitForHit(t *testing.T, ctx context.Context, s *ESSearcher, q SearchQuery, fileID string) SearchHit {
t.Helper()
var lastErr error
for {
hits, err := s.Search(ctx, q)
if err != nil {
lastErr = err
} else {
for _, h := range hits {
if h.ID == fileID {
return h
}
}
}
select {
case <-ctx.Done():
t.Fatalf("search %q: file %s not indexed in time (last err: %v)", q.Text, fileID, lastErr)
case <-time.After(3 * time.Second):
}
}
}
func searchQuiet(t *testing.T, s *ESSearcher, q SearchQuery) []SearchHit {
t.Helper()
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
defer cancel()
hits, err := s.Search(ctx, q)
if err != nil {
t.Fatalf("Search(%q): %v", q.Text, err)
}
return hits
}
+188
View File
@@ -0,0 +1,188 @@
package onlyoffice
import (
"encoding/json"
"reflect"
"testing"
)
func TestESSearchRequestNameOnly(t *testing.T) {
got := esSearchRequest(SearchQuery{Text: "Rechnung"}, "1")
if got.Size != defaultESLimit {
t.Errorf("size = %d, want %d", got.Size, defaultESLimit)
}
if !reflect.DeepEqual(got.Source, []string{"id", "title", "folders"}) {
t.Errorf("_source = %v", got.Source)
}
if len(got.Query.Bool.Must) != 1 || got.Query.Bool.Must[0].MultiMatch == nil {
t.Fatalf("must = %+v, want one multi_match", got.Query.Bool.Must)
}
mm := got.Query.Bool.Must[0].MultiMatch
if mm.Query != "Rechnung" {
t.Errorf("query = %q", mm.Query)
}
if !reflect.DeepEqual(mm.Fields, []string{"title^2"}) {
t.Errorf("fields = %v, want title only", mm.Fields)
}
if _, ok := got.Highlight.Fields["document.attachment.content"]; ok {
t.Error("content highlight present without InContent")
}
if _, ok := got.Highlight.Fields["title"]; !ok {
t.Error("title highlight missing")
}
if len(got.Query.Bool.Filter) != 1 || got.Query.Bool.Filter[0].Term["tenantId"] != int64(1) {
t.Errorf("tenant filter = %+v, want numeric tenantId=1", got.Query.Bool.Filter)
}
}
func TestESSearchRequestContentFields(t *testing.T) {
got := esSearchRequest(SearchQuery{Text: "Mahnung", InContent: true}, "")
mm := got.Query.Bool.Must[0].MultiMatch
want := []string{"title^2", "document.attachment.content"}
if !reflect.DeepEqual(mm.Fields, want) {
t.Errorf("fields = %v, want %v", mm.Fields, want)
}
if _, ok := got.Highlight.Fields["document.attachment.content"]; !ok {
t.Error("content highlight missing with InContent")
}
if len(got.Query.Bool.Filter) != 0 {
t.Errorf("filter = %+v, want none without tenant/folder", got.Query.Bool.Filter)
}
}
func TestESSearchRequestFiltersAndLimit(t *testing.T) {
got := esSearchRequest(SearchQuery{
Text: "Storchen",
FolderID: "649",
Extensions: []string{".PDF", "pdf", "docx"},
Limit: 999,
}, "42")
if got.Size != maxESLimit {
t.Errorf("size = %d, want cap %d", got.Size, maxESLimit)
}
var tenant, folder, wildcards int
for _, f := range got.Query.Bool.Filter {
switch {
case f.Term != nil && f.Term["tenantId"] != nil:
tenant++
case f.Term != nil && f.Term["folders.folderId"] != nil:
folder++
if f.Term["folders.folderId"] != "649" {
t.Errorf("folder filter = %+v", f.Term)
}
case f.Wildcard != nil:
wildcards++
}
}
if tenant != 1 || folder != 1 {
t.Errorf("term filters tenant=%d folder=%d, want 1 each", tenant, folder)
}
if wildcards != 2 {
t.Errorf("wildcard filters = %d, want deduped PDF+docx", wildcards)
}
}
func TestESSearchRequestRejectsEmptyTextAtSearch(t *testing.T) {
s, err := NewESSearcher(ESConfig{URL: "http://localhost:9200"})
if err != nil {
t.Fatalf("NewESSearcher: %v", err)
}
if _, err := s.Search(t.Context(), SearchQuery{Text: " "}); err == nil {
t.Error("empty query: want error")
}
}
func TestNewESSearcherRequiresURL(t *testing.T) {
if _, err := NewESSearcher(ESConfig{}); err == nil {
t.Error("empty URL: want error")
}
s, err := NewESSearcher(ESConfig{URL: "http://es:9200/"})
if err != nil {
t.Fatalf("NewESSearcher: %v", err)
}
if s.cfg.Index != defaultESIndex {
t.Errorf("index = %q, want %q", s.cfg.Index, defaultESIndex)
}
if s.cfg.URL != "http://es:9200" {
t.Errorf("url = %q, want trimmed", s.cfg.URL)
}
if s.Name() != "elasticsearch" {
t.Errorf("Name() = %q", s.Name())
}
}
func TestNormalizeExtensions(t *testing.T) {
got := normalizeExtensions([]string{" .PDF ", "pdf", "", "xlsx"})
want := []string{"pdf", "xlsx"}
if !reflect.DeepEqual(got, want) {
t.Errorf("normalizeExtensions = %v, want %v", got, want)
}
}
func TestParseESSearchResponse(t *testing.T) {
raw := []byte(`{
"took": 12,
"hits": {
"total": {"value": 2, "relation": "eq"},
"hits": [
{
"_id": "2395",
"_score": 7.31,
"_source": {"id": 2395, "title": "Rechnung-4711.pdf",
"folders": [{"folderId": "438", "id": 0}, {"folderId": "11", "id": 0}]},
"highlight": {
"title": ["<em>Rechnung</em>-4711.pdf"],
"document.attachment.content": ["… Zahlung der <em>Rechnung</em> …"]
}
},
{
"_id": "2318",
"_score": 6.02,
"_source": {"id": 2318, "title": "Mahnung.pdf", "folders": []},
"highlight": {"title": ["<em>Mahnung</em>.pdf"]}
}
]
}
}`)
hits, err := parseESSearchResponse(raw)
if err != nil {
t.Fatalf("parseESSearchResponse: %v", err)
}
if len(hits) != 2 {
t.Fatalf("hits = %d, want 2", len(hits))
}
h0 := hits[0]
if h0.ID != "2395" || h0.Title != "Rechnung-4711.pdf" || h0.Kind != File {
t.Errorf("hit0 entry = %+v", h0.Entry)
}
if h0.ParentID != "438" || !reflect.DeepEqual(h0.Path, []string{"438", "11"}) {
t.Errorf("hit0 path = %v parent = %q", h0.Path, h0.ParentID)
}
if h0.Score != 7.31 {
t.Errorf("hit0 score = %v", h0.Score)
}
if h0.Highlight != "… Zahlung der Rechnung …" {
t.Errorf("hit0 highlight = %q, want content fragment", h0.Highlight)
}
if hits[1].Highlight != "Mahnung.pdf" {
t.Errorf("hit1 highlight = %q, want title without tags", hits[1].Highlight)
}
if hits[1].ParentID != "" || len(hits[1].Path) != 0 {
t.Errorf("hit1 path = %v", hits[1].Path)
}
}
func TestESSearchRequestJSONShape(t *testing.T) {
got := esSearchRequest(SearchQuery{Text: "Rechnung", InContent: true}, "1")
b, err := json.Marshal(got)
if err != nil {
t.Fatalf("marshal: %v", err)
}
var back map[string]any
if err := json.Unmarshal(b, &back); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if _, ok := back["query"].(map[string]any)["bool"]; !ok {
t.Errorf("query.bool missing: %s", b)
}
}
+230
View File
@@ -0,0 +1,230 @@
package onlyoffice
// restStore implements FileStore on top of the REST Documents methods in
// files.go. It is a thin adapter: no endpoint logic lives here, and every call
// is wrapped in DoRetry.
import (
"context"
"encoding/json"
"fmt"
"io"
"os"
"path/filepath"
"strconv"
"strings"
)
// restStore is a FileStore over the REST Documents API.
type restStore struct{ c *Client }
// Name reports the backend name.
func (s *restStore) Name() string { return ProviderREST }
// List returns the files and folders directly below parentID.
func (s *restStore) List(ctx context.Context, parentID string) ([]Entry, error) {
var out []Entry
err := retryStoreOp(ctx, func() error {
raw, err := s.c.ListFolder(ctx, parentID)
if err != nil {
return err
}
entries, err := entriesFromFolderMap(raw, ProviderREST)
if err != nil {
return err
}
out = entries
return nil
})
return out, err
}
// Stat returns file metadata. The REST adapter resolves files only; folders
// are listed by their parent (use List).
func (s *restStore) Stat(ctx context.Context, id string) (Entry, error) {
var out Entry
err := retryStoreOp(ctx, func() error {
f, err := s.c.GetFile(ctx, id)
if err != nil {
return err
}
out = FileEntryToEntry(f, ProviderREST)
return nil
})
return out, err
}
// CreateFolder creates a subfolder under parentID.
func (s *restStore) CreateFolder(ctx context.Context, parentID, title string) (Entry, error) {
var out Entry
err := retryStoreOp(ctx, func() error {
m, err := s.c.CreateFolder(ctx, parentID, title)
if err != nil {
return err
}
e, err := folderEntryFromMap(m, parentID, ProviderREST)
if err != nil {
return err
}
if e.ParentID == "" {
e.ParentID = parentID
}
if e.Title == "" {
e.Title = title
}
out = e
return nil
})
return out, err
}
// Upload streams r into parentID as title. UploadToFolder is path based, so
// the reader is spooled to a temporary file first (ponytail: OnlyOffice
// multipart upload buffers the whole body anyway).
func (s *restStore) Upload(ctx context.Context, parentID, title string, r io.Reader) (Entry, error) {
dir, err := os.MkdirTemp("", "oo-rest-upload-")
if err != nil {
return Entry{}, err
}
defer os.RemoveAll(dir)
local := filepath.Join(dir, SafeLocalFileName(title))
f, err := os.Create(local)
if err != nil {
return Entry{}, err
}
if _, err := io.Copy(f, r); err != nil {
f.Close()
return Entry{}, err
}
if err := f.Close(); err != nil {
return Entry{}, err
}
var out Entry
err = retryStoreOp(ctx, func() error {
fe, err := s.c.UploadToFolder(ctx, parentID, local)
if err != nil {
return err
}
out = FileEntryToEntry(fe, ProviderREST)
return nil
})
return out, err
}
// Download streams the file bytes into w.
func (s *restStore) Download(ctx context.Context, id string, w io.Writer) (int64, error) {
var n int64
err := retryStoreOp(ctx, func() error {
var e error
n, e = s.c.DownloadFile(ctx, id, w)
return e
})
return n, err
}
// Move moves file ids into parentID. The REST MoveFiles endpoint handles files
// only; folder moves are not exposed by this adapter.
func (s *restStore) Move(ctx context.Context, ids []string, parentID string) error {
dest, err := strconv.Atoi(strings.TrimSpace(parentID))
if err != nil {
return fmt.Errorf("onlyoffice: rest store: move: non-numeric destination folder id %q", parentID)
}
fileIDs, err := numericIDs(ids)
if err != nil {
return err
}
return retryStoreOp(ctx, func() error {
_, err := s.c.MoveFiles(ctx, dest, fileIDs)
return err
})
}
// Copy copies file ids into parentID. files.go has no copy method, so the
// shared REST fileops copy endpoint (CopyDavItems) is used.
func (s *restStore) Copy(ctx context.Context, ids []string, parentID string) error {
if len(ids) == 0 {
return nil
}
return retryStoreOp(ctx, func() error {
return s.c.CopyDavItems(ctx, nil, ids, parentID)
})
}
// Rename sets a new title (including extension) for a file.
func (s *restStore) Rename(ctx context.Context, id, title string) error {
return retryStoreOp(ctx, func() error {
_, err := s.c.RenameFile(ctx, id, title)
return err
})
}
// Delete permanently deletes file ids.
func (s *restStore) Delete(ctx context.Context, ids []string) error {
fileIDs, err := numericIDs(ids)
if err != nil {
return err
}
if len(fileIDs) == 0 {
return nil
}
return retryStoreOp(ctx, func() error {
return s.c.DeleteFiles(ctx, fileIDs)
})
}
// entriesFromFolderMap converts a ListFolder response map into canonical
// entries, reusing the DavFile/DavFolder decoders for robust size handling.
func entriesFromFolderMap(m map[string]any, provider string) ([]Entry, error) {
if m == nil {
return nil, nil
}
b, err := json.Marshal(m)
if err != nil {
return nil, err
}
var listing DavListing
if err := json.Unmarshal(b, &listing); err != nil {
return nil, err
}
out := make([]Entry, 0, len(listing.Folders)+len(listing.Files))
for _, f := range listing.Folders {
out = append(out, DavFolderToEntry(f, provider))
}
for _, f := range listing.Files {
out = append(out, DavFileToEntry(f, provider))
}
return out, nil
}
// folderEntryFromMap converts a CreateFolder response map into a folder Entry.
func folderEntryFromMap(m map[string]any, parentID, provider string) (Entry, error) {
e := Entry{Kind: Folder, Provider: provider, ParentID: parentID}
if m == nil {
return e, nil
}
b, err := json.Marshal(m)
if err != nil {
return e, err
}
var f DavFolder
if err := json.Unmarshal(b, &f); err != nil {
return e, err
}
e = DavFolderToEntry(f, provider)
return e, nil
}
// numericIDs parses Documents numeric ids from strings.
func numericIDs(ids []string) ([]int, error) {
out := make([]int, 0, len(ids))
for _, id := range ids {
n, err := strconv.Atoi(strings.TrimSpace(id))
if err != nil {
return nil, fmt.Errorf("onlyoffice: rest store: non-numeric id %q", id)
}
out = append(out, n)
}
return out, nil
}
+209
View File
@@ -0,0 +1,209 @@
//go:build integration
package onlyoffice
import (
"bytes"
"context"
"strconv"
"testing"
"time"
)
// TestIntegrationFileStores runs the same operation set (create folder, upload,
// list, stat, download, move, copy, rename, delete) through the REST and DAV
// FileStore adapters against a throwaway project Documents folder. Destructive
// — only run against instances you own.
//
// The Documents fileops API is asynchronous: a move/copy/delete is accepted
// immediately and becomes visible a moment later, so effects are polled.
func TestIntegrationFileStores(t *testing.T) {
c := liveClient(t)
t.Cleanup(func() { cleanupTestProjects(t, c) })
ctx := context.Background()
suffix := time.Now().UTC().Format("20060102-150405")
project, err := c.CreateProject(NewProjectRequest{
Title: testProjectPrefix + "store-" + suffix,
Description: "go-onlyoffice file store integration",
})
if err != nil {
t.Fatalf("CreateProject: %v", err)
}
if project.ID == nil {
t.Fatal("created project without id")
}
root, err := c.projectFolderID(ctx, strconv.Itoa(*project.ID))
if err != nil {
t.Fatalf("projectFolderID: %v", err)
}
for _, backend := range []string{ProviderREST, ProviderDAV} {
t.Run(backend, func(t *testing.T) {
testFileStoreOps(t, ctx, c, c.FileStore(backend), root, suffix)
})
}
}
func testFileStoreOps(t *testing.T, ctx context.Context, c *Client, store FileStore, root, suffix string) {
t.Helper()
content := []byte("file store " + store.Name() + " " + suffix + "\n")
src, err := store.CreateFolder(ctx, root, "fs-src-"+suffix)
if err != nil {
t.Fatalf("CreateFolder src: %v", err)
}
if src.Kind != Folder || src.ID == "" {
t.Fatalf("created src folder: %+v", src)
}
dst, err := store.CreateFolder(ctx, root, "fs-dst-"+suffix)
if err != nil {
t.Fatalf("CreateFolder dst: %v", err)
}
if dst.Kind != Folder || dst.ID == "" {
t.Fatalf("created dst folder: %+v", dst)
}
t.Cleanup(func() {
if err := c.DeleteDavItems(ctx, []string{src.ID, dst.ID}, nil); err != nil {
t.Logf("cleanup folders: %v", err)
}
})
up, err := store.Upload(ctx, src.ID, "doc-"+suffix+".txt", bytes.NewReader(content))
if err != nil {
t.Fatalf("Upload: %v", err)
}
if up.Kind != File || up.ID == "" {
t.Fatalf("uploaded entry: %+v", up)
}
if !waitEntry(ctx, store, src.ID, up.ID, 15*time.Second) {
t.Fatalf("uploaded %s not listed in src", up.ID)
}
st, err := store.Stat(ctx, up.ID)
if err != nil {
t.Fatalf("Stat: %v", err)
}
if st.ID != up.ID || st.Kind != File {
t.Fatalf("stat = %+v", st)
}
var buf bytes.Buffer
n, err := store.Download(ctx, up.ID, &buf)
if err != nil {
t.Fatalf("Download: %v", err)
}
if n != int64(len(content)) || !bytes.Equal(buf.Bytes(), content) {
t.Fatalf("download mismatch: got %d bytes %q want %d", n, buf.String(), len(content))
}
moveEventually(t, ctx, store, up.ID, dst.ID)
if !waitEntry(ctx, store, dst.ID, up.ID, 20*time.Second) {
t.Fatalf("moved file %s not in dst", up.ID)
}
if err := store.Copy(ctx, []string{up.ID}, src.ID); err != nil {
t.Fatalf("Copy: %v", err)
}
copied := waitOtherFile(ctx, store, src.ID, up.ID, 20*time.Second)
if copied == nil {
t.Fatalf("no copy found in src after Copy")
}
newTitle := "renamed-" + suffix + ".txt"
renameEventually(t, ctx, store, up.ID, newTitle)
if err := store.Delete(ctx, []string{up.ID, copied.ID}); err != nil {
t.Fatalf("Delete: %v", err)
}
if !waitNoEntry(ctx, store, dst.ID, up.ID, 20*time.Second) {
t.Fatalf("file %s still present in dst after delete", up.ID)
}
if !waitNoEntry(ctx, store, src.ID, copied.ID, 20*time.Second) {
t.Fatalf("copy %s still present in src after delete", copied.ID)
}
}
// moveEventually issues Move and retries while the operation is not visible yet
// (the fileops API accepts asynchronously and occasionally rejects a move that
// raced the just-finished upload).
func moveEventually(t *testing.T, ctx context.Context, store FileStore, id, dstID string) {
t.Helper()
var lastErr error
for attempt := 0; attempt < 5; attempt++ {
if lastErr = store.Move(ctx, []string{id}, dstID); lastErr == nil {
if waitEntry(ctx, store, dstID, id, 6*time.Second) {
return
}
}
time.Sleep(time.Second)
}
t.Fatalf("Move %s -> %s: %v", id, dstID, lastErr)
}
func renameEventually(t *testing.T, ctx context.Context, store FileStore, id, title string) {
t.Helper()
var lastErr error
for attempt := 0; attempt < 5; attempt++ {
if lastErr = store.Rename(ctx, id, title); lastErr == nil {
if e, err := store.Stat(ctx, id); err == nil && e.Title == title {
return
}
}
time.Sleep(time.Second)
}
t.Fatalf("Rename %s -> %q: %v", id, title, lastErr)
}
func waitEntry(ctx context.Context, store FileStore, parentID, id string, d time.Duration) bool {
deadline := time.Now().Add(d)
for time.Now().Before(deadline) {
if list, err := store.List(ctx, parentID); err == nil && entryByID(list, id) != nil {
return true
}
time.Sleep(500 * time.Millisecond)
}
return false
}
func waitNoEntry(ctx context.Context, store FileStore, parentID, id string, d time.Duration) bool {
deadline := time.Now().Add(d)
for time.Now().Before(deadline) {
if list, err := store.List(ctx, parentID); err == nil && entryByID(list, id) == nil {
return true
}
time.Sleep(500 * time.Millisecond)
}
return false
}
func waitOtherFile(ctx context.Context, store FileStore, parentID, id string, d time.Duration) *Entry {
deadline := time.Now().Add(d)
for time.Now().Before(deadline) {
if list, err := store.List(ctx, parentID); err == nil {
if e := firstFileOtherThan(list, id); e != nil {
return e
}
}
time.Sleep(500 * time.Millisecond)
}
return nil
}
func entryByID(entries []Entry, id string) *Entry {
for i := range entries {
if entries[i].ID == id {
return &entries[i]
}
}
return nil
}
func firstFileOtherThan(entries []Entry, id string) *Entry {
for i := range entries {
if entries[i].Kind == File && entries[i].ID != id {
return &entries[i]
}
}
return nil
}
+37 -44
View File
@@ -324,19 +324,22 @@ func (c *Client) MoveFiles(ctx context.Context, destFolderID int, fileIDs []int)
"resolveType": "Skip",
"holdResult": true,
}
out, err := c.putJSONObject(ctx, "/api/2.0/files/fileops/move.json", body)
// fileops/move answers an operations envelope (like MoveDavItems), not a
// single object, so parse the raw body before unwrapping and surface any
// per-operation error. Unwrapping first (putJSONObject) made fileopsError
// look for a "response" key that was already stripped.
raw, err := c.putJSON(ctx, "/api/2.0/files/fileops/move", body)
if err != nil {
out, err = c.putJSONObject(ctx, "/api/2.0/files/fileops/move", body)
}
if err != nil {
return nil, err
}
if raw, merr := json.Marshal(out); merr == nil {
if ferr := fileopsError(raw); ferr != nil {
return nil, ferr
raw, err = c.putJSON(ctx, "/api/2.0/files/fileops/move.json", body)
if err != nil {
return nil, err
}
}
return out, err
if ferr := fileopsError(raw); ferr != nil {
return nil, ferr
}
out, _ := unmarshalResponseObject(raw)
return out, nil
}
// UploadToFolder uploads a local file into an arbitrary Documents folder id.
@@ -358,20 +361,31 @@ func (c *Client) UploadToFolder(ctx context.Context, folderID, localPath string)
// UpdateFile uploads a new version of an existing file (same id, name and
// folder). It does not delete and does not create a second file.
//
// The Documents API method is PUT /api/2.0/files/{id}/update; POST is kept as
// a fallback for older servers. The path is tried with and without .json.
func (c *Client) UpdateFile(ctx context.Context, fileID, localPath string) (*FileEntry, error) {
if fileID == "" || localPath == "" {
return nil, fmt.Errorf("file id and local path are required")
}
uploadPath := fmt.Sprintf("/api/2.0/files/%s/update", url.PathEscape(fileID))
raw, err := c.uploadMultipart(ctx, uploadPath, "file", localPath)
if err != nil {
uploadPath = fmt.Sprintf("/api/2.0/files/%s/update.json", url.PathEscape(fileID))
raw, err = c.uploadMultipart(ctx, uploadPath, "file", localPath)
if err != nil {
return nil, err
}
base := fmt.Sprintf("/api/2.0/files/%s/update", url.PathEscape(fileID))
attempts := []struct {
method, path string
}{
{http.MethodPut, base},
{http.MethodPut, base + ".json"},
{http.MethodPost, base},
{http.MethodPost, base + ".json"},
}
return decodeResponseFileEntry(raw)
var lastErr error
for _, a := range attempts {
raw, err := c.uploadMultipartMethod(ctx, a.method, a.path, "file", localPath)
if err == nil {
return decodeResponseFileEntry(raw)
}
lastErr = err
}
return nil, lastErr
}
// FileFolderID returns the parent folder id string for a file entry, if known.
@@ -383,36 +397,15 @@ func FileFolderID(f *FileEntry) string {
}
// DownloadFile streams file bytes from the file's viewUrl using the same auth
// as API calls. Writes into dst.
// as API calls. Writes into dst. When the portal serves the file from its stale
// AWS S3 consumer, the bytes are fetched from the local MinIO store instead
// (see storage_fallback.go).
func (c *Client) DownloadFile(ctx context.Context, fileID string, dst io.Writer) (int64, error) {
f, err := c.GetFile(ctx, fileID)
if err != nil {
return 0, err
}
if f.ViewURL == nil || *f.ViewURL == "" {
return 0, fmt.Errorf("file %s has no viewUrl", fileID)
}
downloadURL := c.resolveAPIURL(*f.ViewURL)
auth, err := c.authHeader()
if err != nil {
return 0, err
}
req, err := http.NewRequestWithContext(ctx, http.MethodGet, downloadURL, nil)
if err != nil {
return 0, err
}
req.Header.Set("Authorization", auth)
resp, err := c.client.Do(req)
if err != nil {
return 0, err
}
defer resp.Body.Close()
if resp.StatusCode >= 400 {
b, _ := io.ReadAll(io.LimitReader(resp.Body, 512))
return 0, fmt.Errorf("GET viewUrl: %d %s", resp.StatusCode, truncate(string(b), 400))
}
n, err := io.Copy(dst, resp.Body)
return n, err
return c.downloadFileEntry(ctx, f, dst)
}
func (c *Client) resolveAPIURL(ref string) string {
+9 -33
View File
@@ -259,34 +259,10 @@ func (c *Client) UploadDavFile(ctx context.Context, folderID, fileName string, s
return env.Response, nil
}
// DownloadDavFile streams the file identified by id to w, returning bytes copied.
// DownloadDavFile streams the file identified by id to w, returning bytes
// copied. It shares the MinIO stale-S3 fallback with DownloadFile.
func (c *Client) DownloadDavFile(ctx context.Context, id string, w io.Writer) (int64, error) {
file, err := c.GetFile(ctx, id)
if err != nil {
return 0, err
}
if file.ViewURL == nil || *file.ViewURL == "" {
return 0, fmt.Errorf("onlyoffice: file %s has no viewUrl", id)
}
u := c.resolveAPIURL(*file.ViewURL)
auth, err := c.authHeader()
if err != nil {
return 0, err
}
req, err := http.NewRequestWithContext(ctx, http.MethodGet, u, nil)
if err != nil {
return 0, err
}
req.Header.Set("Authorization", auth)
resp, err := c.client.Do(req)
if err != nil {
return 0, err
}
defer resp.Body.Close()
if resp.StatusCode >= 400 {
return 0, fmt.Errorf("onlyoffice: download: %d", resp.StatusCode)
}
return io.Copy(w, resp.Body)
return c.DownloadFile(ctx, id, w)
}
// --- internal helpers -------------------------------------------------------
@@ -434,12 +410,12 @@ func (f *DavFolder) UnmarshalJSON(b []byte) error {
// UnmarshalJSON decodes a file row, capturing size and timestamps.
func (f *DavFile) UnmarshalJSON(b []byte) error {
var raw struct {
ID *json.Number `json:"id"`
Title *string `json:"title"`
PureSize *int64 `json:"pureContentLength"`
SizeStr *string `json:"contentLength"`
Updated *string `json:"updated"`
ViewURL *string `json:"viewUrl"`
ID *json.Number `json:"id"`
Title *string `json:"title"`
PureSize *int64 `json:"pureContentLength"`
SizeStr *string `json:"contentLength"`
Updated *string `json:"updated"`
ViewURL *string `json:"viewUrl"`
}
if err := json.Unmarshal(b, &raw); err != nil {
return err
+2
View File
@@ -4,6 +4,7 @@ go 1.25.0
require (
github.com/JohannesKaufmann/html-to-markdown/v2 v2.5.2
github.com/aws/aws-sdk-go-v2 v1.41.1
github.com/charmbracelet/bubbles v0.18.0
github.com/charmbracelet/bubbletea v0.25.0
github.com/charmbracelet/glamour v0.8.0
@@ -26,6 +27,7 @@ require (
github.com/JohannesKaufmann/dom v0.3.1 // indirect
github.com/alecthomas/chroma/v2 v2.14.0 // indirect
github.com/atotto/clipboard v0.1.4 // indirect
github.com/aws/smithy-go v1.24.0 // indirect
github.com/aymanbagabas/go-osc52/v2 v2.0.1 // indirect
github.com/aymerick/douceur v0.2.0 // indirect
github.com/containerd/console v1.0.4-0.20230313162750-1ae8d489ac81 // indirect
+4
View File
@@ -10,6 +10,10 @@ github.com/alecthomas/repr v0.4.0 h1:GhI2A8MACjfegCPVq9f1FLvIBS+DrQ2KQBFZP1iFzXc
github.com/alecthomas/repr v0.4.0/go.mod h1:Fr0507jx4eOXV7AlPV6AVZLYrLIuIeSOWtW57eE/O/4=
github.com/atotto/clipboard v0.1.4 h1:EH0zSVneZPSuFR11BlR9YppQTVDbh5+16AmcJi4g1z4=
github.com/atotto/clipboard v0.1.4/go.mod h1:ZY9tmq7sm5xIbd9bOK4onWV4S6X0u6GY7Vn0Yu86PYI=
github.com/aws/aws-sdk-go-v2 v1.41.1 h1:ABlyEARCDLN034NhxlRUSZr4l71mh+T5KAeGh6cerhU=
github.com/aws/aws-sdk-go-v2 v1.41.1/go.mod h1:MayyLB8y+buD9hZqkCW3kX1AKq07Y5pXxtgB+rRFhz0=
github.com/aws/smithy-go v1.24.0 h1:LpilSUItNPFr1eY85RYgTIg5eIEPtvFbskaFcmmIUnk=
github.com/aws/smithy-go v1.24.0/go.mod h1:LEj2LM3rBRQJxPZTB4KuzZkaZYnZPnvgIhb4pu07mx0=
github.com/aymanbagabas/go-osc52/v2 v2.0.1 h1:HwpRHbFMcZLEVr42D4p7XBqjyuxQH5SMiErDT4WkJ2k=
github.com/aymanbagabas/go-osc52/v2 v2.0.1/go.mod h1:uYgXzlJ7ZpABp8OJ+exZzJJhRNQ2ASbcXHWsFqH8hp8=
github.com/aymanbagabas/go-udiff v0.2.0 h1:TK0fH4MteXUDspT88n8CKzvK0X9O2xu9yQjWpi6yML8=
+9 -1
View File
@@ -346,6 +346,14 @@ func (c *Client) putJSON(ctx context.Context, path string, body any) (json.RawMe
// uploadMultipart posts a single file to path under the given form field name.
func (c *Client) uploadMultipart(ctx context.Context, path, fieldName, filePath string) (json.RawMessage, error) {
return c.uploadMultipartMethod(ctx, http.MethodPost, path, fieldName, filePath)
}
// uploadMultipartMethod sends a single-file multipart request with the given
// HTTP method. The OnlyOffice Documents API needs PUT for /update (a new
// version) and POST for /upload (a new file); sending POST to /update answers
// 500 on current servers.
func (c *Client) uploadMultipartMethod(ctx context.Context, method, path, fieldName, filePath string) (json.RawMessage, error) {
auth, err := c.authHeader()
if err != nil {
return nil, err
@@ -368,7 +376,7 @@ func (c *Client) uploadMultipart(ctx context.Context, path, fieldName, filePath
if err := mw.Close(); err != nil {
return nil, err
}
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.baseURL()+path, &buf)
req, err := http.NewRequestWithContext(ctx, method, c.baseURL()+path, &buf)
if err != nil {
return nil, err
}
+60
View File
@@ -0,0 +1,60 @@
package onlyoffice
import (
"context"
"regexp"
"time"
)
// RetryPolicy controls deterministic retries against OnlyOffice: fixed linear
// backoff without jitter, so repeated runs wait exactly the same schedule.
// OnlyOffice throttles bulk reads/writes with 429 (and occasional 502/503/504
// from openresty), so every bulk tool routes API calls through DoRetry.
type RetryPolicy struct {
Attempts int // total attempts, including the first try
Base time.Duration // wait before retry N is N*Base
Max time.Duration // per-wait cap
}
// DefaultRetryPolicy retries up to 5 times with 1s, 2s, 3s, 4s waits.
func DefaultRetryPolicy() RetryPolicy {
return RetryPolicy{Attempts: 5, Base: time.Second, Max: 30 * time.Second}
}
var transientRe = regexp.MustCompile(`:\s*(429|502|503|504)\b`)
// Transient reports whether err looks like a transient OnlyOffice answer
// (an HTTP 429/502/503/504 surfaced as "...: <code> ...").
func Transient(err error) bool {
if err == nil {
return false
}
return transientRe.MatchString(err.Error())
}
// DoRetry runs fn until it succeeds, fails non-transiently, or attempts run
// out. Waits are deterministic: N*Base capped at Max, no jitter.
func DoRetry(ctx context.Context, p RetryPolicy, fn func() error) error {
if p.Attempts < 1 {
p.Attempts = 1
}
var err error
for attempt := 1; attempt <= p.Attempts; attempt++ {
if ctx.Err() != nil {
return ctx.Err()
}
if err = fn(); err == nil || !Transient(err) || attempt == p.Attempts {
return err
}
wait := time.Duration(attempt) * p.Base
if wait > p.Max {
wait = p.Max
}
select {
case <-ctx.Done():
return ctx.Err()
case <-time.After(wait):
}
}
return err
}
+200
View File
@@ -0,0 +1,200 @@
package onlyoffice
// MinIO download fallback for the portal's stale AWS S3 consumer.
//
// On the Fibu EDL portal some older Documents files live in S3/MinIO, but the
// portal's storage consumer still points at s3.us-east-1.amazonaws.com with
// access key "minio". Downloads of those files answer 403 InvalidAccessKeyId.
// The bytes are present in the local MinIO store under a deterministic object
// key, so the client retries the GET there.
import (
"context"
"crypto/sha256"
"encoding/hex"
"fmt"
"io"
"net/http"
"net/url"
"os"
"strings"
"time"
"github.com/aws/aws-sdk-go-v2/aws"
"github.com/aws/aws-sdk-go-v2/aws/signer/v4"
)
const (
defaultMinioEndpoint = "http://192.168.188.10:9000"
defaultMinioBucket = "office"
minioRegion = "us-east-1"
)
// minioObjectKey is the fallback object key layout the portal's S3 consumer
// writes for Documents files: 00/00/01/files/folder_<folderId>/file_<fileId>/v1/content.pdf.
// Prefer minioObjectKeyFromURL: the portal stores all files below its storage
// root folder, which is not the API folderId returned by GetFile.
func minioObjectKey(fileID, folderID string) string {
return "00/00/01/files/folder_" + folderID + "/file_" + fileID + "/v1/content.pdf"
}
// minioObjectKeyFromURL extracts the object key from an S3 download URL. Path
// style URLs (bucket as first path segment) have that segment removed; virtual
// host style URLs are returned as-is. This is authoritative: the portal signs
// the exact key, so no folder-id guessing is needed.
func minioObjectKeyFromURL(rawURL, bucket string) (string, bool) {
u, err := url.Parse(strings.TrimSpace(rawURL))
if err != nil || u.Path == "" {
return "", false
}
segs := strings.Split(strings.Trim(u.Path, "/"), "/")
// Path-style URLs carry the bucket as leading segment; the portal's S3
// consumer can emit it twice (serviceurl already includes the bucket), so
// strip every leading segment equal to the bucket.
for len(segs) > 0 && bucket != "" && segs[0] == bucket {
segs = segs[1:]
}
if len(segs) == 0 {
return "", false
}
for _, s := range segs {
if s == "" || s == "." || s == ".." {
return "", false
}
}
return strings.Join(segs, "/"), true
}
// isStaleS3Redirect reports whether a download landed on the portal's stale AWS
// S3 consumer. Such responses either carry an S3 InvalidAccessKeyId XML body or
// point at amazonaws.com with the "minio" access key id in the query.
func isStaleS3Redirect(rawURL string, body []byte) bool {
if strings.Contains(strings.ToLower(string(body)), "invalidaccesskeyid") {
return true
}
u, err := url.Parse(strings.TrimSpace(rawURL))
if err != nil || u.Host == "" {
return false
}
host := strings.ToLower(u.Host)
if !strings.Contains(host, "amazonaws.com") {
return false
}
q := strings.ToLower(u.RawQuery)
return strings.Contains(q, "accesskeyid=minio") || strings.Contains(q, "x-amz-credential=minio")
}
// minioConfig is the runtime configuration for the local MinIO fallback.
type minioConfig struct {
Endpoint string
Bucket string
AccessKey string
SecretKey string
}
// loadMinioConfig reads the fallback configuration from the environment.
// Secrets are never defaulted; without access/secret keys the fallback is off.
func loadMinioConfig() minioConfig {
return minioConfig{
Endpoint: strings.TrimRight(firstNonEmpty(os.Getenv("MINIO_ENDPOINT"), defaultMinioEndpoint), "/"),
Bucket: firstNonEmpty(os.Getenv("MINIO_BUCKET"), defaultMinioBucket),
AccessKey: os.Getenv("MINIO_ACCESS_KEY"),
SecretKey: os.Getenv("MINIO_SECRET_KEY"),
}
}
// downloadFileEntry downloads f's bytes to dst. It transparently falls back to
// the local MinIO store when the portal redirects the download to its stale AWS
// S3 consumer.
func (c *Client) downloadFileEntry(ctx context.Context, f *FileEntry, dst io.Writer) (int64, error) {
if f == nil {
return 0, fmt.Errorf("onlyoffice: download: nil file entry")
}
if f.ViewURL == nil || *f.ViewURL == "" {
return 0, fmt.Errorf("onlyoffice: file has no viewUrl")
}
downloadURL := c.resolveAPIURL(*f.ViewURL)
auth, err := c.authHeader()
if err != nil {
return 0, err
}
req, err := http.NewRequestWithContext(ctx, http.MethodGet, downloadURL, nil)
if err != nil {
return 0, err
}
req.Header.Set("Authorization", auth)
resp, err := c.client.Do(req)
if err != nil {
return 0, err
}
defer resp.Body.Close()
if resp.StatusCode >= 400 {
b, _ := io.ReadAll(io.LimitReader(resp.Body, 4096))
finalURL := downloadURL
if resp.Request != nil && resp.Request.URL != nil {
finalURL = resp.Request.URL.String()
}
if isStaleS3Redirect(finalURL, b) {
key, ok := minioObjectKeyFromURL(finalURL, loadMinioConfig().Bucket)
if !ok {
fileID := ""
if f.ID != nil {
fileID = f.ID.String()
}
key = minioObjectKey(fileID, FileFolderID(f))
}
n, merr := c.downloadFromMinio(ctx, key, dst)
if merr == nil {
return n, nil
}
return 0, fmt.Errorf("GET viewUrl: %d (stale S3) and minio fallback: %w", resp.StatusCode, merr)
}
return 0, fmt.Errorf("GET viewUrl: %d %s", resp.StatusCode, truncate(string(b), 400))
}
return io.Copy(dst, resp.Body)
}
// downloadFromMinio streams objectKey from the configured MinIO bucket.
func (c *Client) downloadFromMinio(ctx context.Context, objectKey string, dst io.Writer) (int64, error) {
if objectKey == "" {
return 0, fmt.Errorf("onlyoffice: minio fallback: empty object key")
}
cfg := loadMinioConfig()
if cfg.AccessKey == "" || cfg.SecretKey == "" {
return 0, fmt.Errorf("onlyoffice: minio fallback: MINIO_ACCESS_KEY/MINIO_SECRET_KEY not set")
}
base, err := url.Parse(cfg.Endpoint)
if err != nil || base.Host == "" {
return 0, fmt.Errorf("onlyoffice: minio fallback: bad MINIO_ENDPOINT %q", cfg.Endpoint)
}
u := *base
u.Path = "/" + cfg.Bucket + "/" + objectKey
req, err := http.NewRequestWithContext(ctx, http.MethodGet, u.String(), nil)
if err != nil {
return 0, err
}
if err := signMinioRequest(ctx, cfg, req); err != nil {
return 0, err
}
resp, err := c.client.Do(req)
if err != nil {
return 0, fmt.Errorf("onlyoffice: minio fallback: %w", err)
}
defer resp.Body.Close()
if resp.StatusCode >= 400 {
b, _ := io.ReadAll(io.LimitReader(resp.Body, 512))
return 0, fmt.Errorf("onlyoffice: minio fallback: %d %s", resp.StatusCode, truncate(string(b), 300))
}
return io.Copy(dst, resp.Body)
}
// signMinioRequest signs req with AWS Signature V4 for the S3 service.
func signMinioRequest(ctx context.Context, cfg minioConfig, req *http.Request) error {
sum := sha256.Sum256(nil)
creds := aws.Credentials{AccessKeyID: cfg.AccessKey, SecretAccessKey: cfg.SecretKey}
if err := v4.NewSigner().SignHTTP(ctx, creds, req, hex.EncodeToString(sum[:]), "s3", minioRegion, time.Now()); err != nil {
return fmt.Errorf("onlyoffice: minio fallback: sign: %w", err)
}
return nil
}
+45
View File
@@ -0,0 +1,45 @@
//go:build integration
package onlyoffice
import (
"context"
"io"
"os"
"strings"
"testing"
"time"
)
// TestIntegrationMinioFallback downloads known stale-S3 files through the local
// MinIO fallback. Requires ONLYOFFICE_URL/USER/PASS (as all integration tests),
// MINIO_ACCESS_KEY/MINIO_SECRET_KEY and MINIO_TEST_FILE_IDS="3785,3859,3666";
// skips when any of those are missing.
func TestIntegrationMinioFallback(t *testing.T) {
if os.Getenv("MINIO_ACCESS_KEY") == "" || os.Getenv("MINIO_SECRET_KEY") == "" {
t.Skip("MINIO_ACCESS_KEY/MINIO_SECRET_KEY not set — skipping integration test")
}
raw := strings.TrimSpace(os.Getenv("MINIO_TEST_FILE_IDS"))
if raw == "" {
t.Skip("MINIO_TEST_FILE_IDS not set — skipping integration test")
}
c := liveClient(t)
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
defer cancel()
for _, id := range strings.Split(raw, ",") {
id = strings.TrimSpace(id)
if id == "" {
continue
}
n, err := c.DownloadFile(ctx, id, io.Discard)
if err != nil {
t.Errorf("DownloadFile(%s): %v", id, err)
continue
}
if n == 0 {
t.Errorf("DownloadFile(%s): 0 bytes", id)
} else {
t.Logf("DownloadFile(%s): %d bytes", id, n)
}
}
}
+219
View File
@@ -0,0 +1,219 @@
package onlyoffice
import (
"bytes"
"context"
"io"
"net/http"
"net/http/httptest"
"strings"
"testing"
"time"
)
func TestMinioObjectKey(t *testing.T) {
cases := []struct {
fileID string
folderID string
want string
}{
{"3785", "652", "00/00/01/files/folder_652/file_3785/v1/content.pdf"},
{"1", "2", "00/00/01/files/folder_2/file_1/v1/content.pdf"},
{"3666", "4000", "00/00/01/files/folder_4000/file_3666/v1/content.pdf"},
}
for _, tc := range cases {
if got := minioObjectKey(tc.fileID, tc.folderID); got != tc.want {
t.Errorf("minioObjectKey(%q, %q) = %q, want %q", tc.fileID, tc.folderID, got, tc.want)
}
}
}
func TestMinioObjectKeyFromURL(t *testing.T) {
cases := []struct {
name string
url string
bucket string
want string
ok bool
}{
{
name: "path style drops bucket segment",
url: "https://s3.us-east-1.amazonaws.com/office/00/00/01/files/folder_4000/file_3785/v1/content.pdf?AWSAccessKeyId=minio",
bucket: "office",
want: "00/00/01/files/folder_4000/file_3785/v1/content.pdf",
ok: true,
},
{
name: "doubled bucket segment (portal serviceurl includes bucket)",
url: "https://s3.us-east-1.amazonaws.com/office/office/00/00/01/files/folder_4000/file_3785/v1/content.pdf?AWSAccessKeyId=minio",
bucket: "office",
want: "00/00/01/files/folder_4000/file_3785/v1/content.pdf",
ok: true,
},
{
name: "virtual host style keeps path",
url: "https://office.s3.us-east-1.amazonaws.com/00/00/01/files/folder_4000/file_3785/v1/content.pdf",
bucket: "office",
want: "00/00/01/files/folder_4000/file_3785/v1/content.pdf",
ok: true,
},
{
name: "foreign first segment kept",
url: "https://example.com/other/file_1/v1/content.pdf",
bucket: "office",
want: "other/file_1/v1/content.pdf",
ok: true,
},
{name: "empty path", url: "https://example.com", bucket: "office", ok: false},
{name: "traversal", url: "https://example.com/office/../etc/passwd", bucket: "office", ok: false},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
got, ok := minioObjectKeyFromURL(tc.url, tc.bucket)
if ok != tc.ok || got != tc.want {
t.Errorf("minioObjectKeyFromURL(%q, %q) = (%q, %v), want (%q, %v)", tc.url, tc.bucket, got, ok, tc.want, tc.ok)
}
})
}
}
func TestIsStaleS3Redirect(t *testing.T) {
cases := []struct {
name string
url string
body []byte
want bool
}{
{
name: "aws redirect with minio access key",
url: "https://s3.us-east-1.amazonaws.com/office/x/file_1?AWSAccessKeyId=minio&Expires=1",
want: true,
},
{
name: "aws redirect with minio x-amz-credential",
url: "https://office.s3.us-east-1.amazonaws.com/00/00/01/files/folder_1/file_1?X-Amz-Credential=minio%2F20260914",
want: true,
},
{
name: "invalid access key xml body",
url: "https://portal.internal/download/1",
body: []byte(`<?xml version="1.0"?><Error><Code>InvalidAccessKeyId</Code><AWSAccessKeyId>minio</AWSAccessKeyId></Error>`),
want: true,
},
{
name: "regular pdf from portal",
url: "https://portal.internal/download/1",
body: []byte("%PDF-1.7 data"),
want: false,
},
{
name: "aws redirect with foreign key",
url: "https://s3.us-east-1.amazonaws.com/office/x?AWSAccessKeyId=other",
want: false,
},
{
name: "amazonaws in path but foreign host",
url: "https://example.com/amazonaws.com/file?AWSAccessKeyId=minio",
want: false,
},
{
name: "empty",
want: false,
},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
if got := isStaleS3Redirect(tc.url, tc.body); got != tc.want {
t.Errorf("isStaleS3Redirect(%q, %q) = %v, want %v", tc.url, tc.body, got, tc.want)
}
})
}
}
const staleS3Body = `<?xml version="1.0" encoding="UTF-8"?>` +
`<Error><Code>InvalidAccessKeyId</Code>` +
`<Message>The AWS Access Key Id you provided does not exist in our records.</Message>` +
`<AWSAccessKeyId>minio</AWSAccessKeyId></Error>`
func TestDownloadFileMinioFallback(t *testing.T) {
const payload = "PDFDATA-3785"
portal := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
switch r.URL.Path {
case "/api/2.0/files/file/3785.json":
w.Header().Set("Content-Type", "application/json")
io.WriteString(w, `{"response":{"id":3785,"title":"04.pdf","folderId":655,"viewUrl":"/download/3785"}}`)
case "/download/3785":
http.Redirect(w, r, "/office/00/00/01/files/folder_4000/file_3785/v1/content.pdf?AWSAccessKeyId=minio", http.StatusTemporaryRedirect)
case "/office/00/00/01/files/folder_4000/file_3785/v1/content.pdf":
w.WriteHeader(http.StatusForbidden)
io.WriteString(w, staleS3Body)
default:
http.NotFound(w, r)
}
}))
defer portal.Close()
var minioPath, minioAuth string
minio := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
minioPath, minioAuth = r.URL.Path, r.Header.Get("Authorization")
io.WriteString(w, payload)
}))
defer minio.Close()
t.Setenv("MINIO_ENDPOINT", minio.URL)
t.Setenv("MINIO_BUCKET", "office")
t.Setenv("MINIO_ACCESS_KEY", "testkey")
t.Setenv("MINIO_SECRET_KEY", "testsecret")
c := &Client{
client: portal.Client(),
credentials: &Credentials{Url: portal.URL},
token: &Token{Value: "Bearer test", Expires: Time(time.Now().Add(time.Hour))},
}
var buf bytes.Buffer
n, err := c.DownloadFile(context.Background(), "3785", &buf)
if err != nil {
t.Fatalf("DownloadFile: %v", err)
}
if n != int64(len(payload)) || buf.String() != payload {
t.Fatalf("got %d bytes %q, want %d bytes %q", n, buf.String(), len(payload), payload)
}
if want := "/office/00/00/01/files/folder_4000/file_3785/v1/content.pdf"; minioPath != want {
t.Errorf("minio path = %q, want %q", minioPath, want)
}
if !strings.HasPrefix(minioAuth, "AWS4-HMAC-SHA256") {
t.Errorf("minio request not SigV4-signed; Authorization=%q", minioAuth)
}
}
func TestDownloadFileMinioFallbackWithoutCreds(t *testing.T) {
portal := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
switch r.URL.Path {
case "/api/2.0/files/file/3785.json":
io.WriteString(w, `{"response":{"id":3785,"folderId":652,"viewUrl":"/download/3785"}}`)
default:
w.WriteHeader(http.StatusForbidden)
io.WriteString(w, staleS3Body)
}
}))
defer portal.Close()
t.Setenv("MINIO_ACCESS_KEY", "")
t.Setenv("MINIO_SECRET_KEY", "")
c := &Client{
client: portal.Client(),
credentials: &Credentials{Url: portal.URL},
token: &Token{Value: "Bearer test", Expires: Time(time.Now().Add(time.Hour))},
}
_, err := c.DownloadFile(context.Background(), "3785", io.Discard)
if err == nil {
t.Fatal("expected error without minio credentials")
}
if !strings.Contains(err.Error(), "MINIO_ACCESS_KEY/MINIO_SECRET_KEY not set") {
t.Fatalf("unexpected error: %v", err)
}
}