Compare commits
141
Commits
v0.9.0
...
5588c2d3f7
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5588c2d3f7 | ||
|
|
f8783b86a8 | ||
|
|
d9db1c5e56 | ||
|
|
c551634e55 | ||
|
|
a88fd5b40c | ||
|
|
9b149281a8 | ||
|
|
d6b8eb7777 | ||
|
|
8a17a32269 | ||
|
|
43a1a47faa | ||
|
|
620505cb24 | ||
|
|
a6bba30438 | ||
|
|
6549fbf7e2 | ||
|
|
1a8bf0f83c | ||
|
|
046882b8ef | ||
|
|
a3ef83a961 | ||
|
|
54f825f8e8 | ||
|
|
6a4940ba82 | ||
|
|
1da368fd5d | ||
|
|
66a6b57401 | ||
|
|
075c66d4d7 | ||
|
|
c59ad645ea | ||
|
|
c76ab7267d | ||
|
|
3cc288d281 | ||
|
|
ce4778bdf1 | ||
|
|
3fe43ee82c | ||
|
|
6d93ab5b1f | ||
|
|
95d79925ea | ||
|
|
bf3ef025aa | ||
|
|
ba29738482 | ||
|
|
72c7cd5173 | ||
|
|
b3dfff1a4c | ||
|
|
b5a61ad420 | ||
|
|
1aba545e1d | ||
|
|
94951fc6c5 | ||
|
|
d95985e6d0 | ||
|
|
57b383d579 | ||
|
|
a73f6c172d | ||
|
|
e49693ad2c | ||
|
|
b86d71211f | ||
|
|
a2b919478e | ||
|
|
ccabc11153 | ||
|
|
9a6da1a4f7 | ||
|
|
504d13ed08 | ||
|
|
efc0864d42 | ||
|
|
59caf5e560 | ||
|
|
e7803c5269 | ||
|
|
a08e7c49ad | ||
|
|
4310aa7002 | ||
|
|
1a7a962183 | ||
|
|
af8e3b053a | ||
|
|
8ac777c031 | ||
|
|
c576bfe2ee | ||
|
|
47a2c256f3 | ||
|
|
c1881e1e61 | ||
|
|
7f56332285 | ||
|
|
a5807b9c31 | ||
|
|
d650a16a36 | ||
|
|
3da11586f9 | ||
|
|
e9c969a89c | ||
|
|
4d91726179 | ||
|
|
4d8af7a2fe | ||
|
|
34745e349e | ||
|
|
dfea57a57b | ||
|
|
8595c17f25 | ||
|
|
61da2fb88b | ||
|
|
24ca144b22 | ||
|
|
93828ee19d | ||
|
|
35f0cb8d20 | ||
|
|
ca5c85a4e2 | ||
|
|
220c7e8265 | ||
|
|
ff4c0481ba | ||
|
|
68445b0dfd | ||
|
|
8739d45b1a | ||
|
|
f3150436fd | ||
|
|
a30ae20f32 | ||
|
|
5a0b2ab412 | ||
|
|
69844c9122 | ||
|
|
b04d610275 | ||
|
|
6b3f40e1a1 | ||
|
|
ba8148a9e1 | ||
|
|
f1739dc9dc | ||
|
|
5b812ce346 | ||
|
|
309ae44ee4 | ||
|
|
ff2bc539f5 | ||
|
|
d9a7adcd94 | ||
|
|
3713ed6c51 | ||
|
|
ac06d86363 | ||
|
|
7d397ac680 | ||
|
|
e52cb31bb7 | ||
|
|
a8cb4e9780 | ||
|
|
622dcdb7bf | ||
|
|
2b267d36a8 | ||
|
|
50bd47570b | ||
|
|
096a996726 | ||
|
|
625dd0a60a | ||
|
|
e531de280f | ||
|
|
129e3a58cc | ||
|
|
70e605b4ca | ||
|
|
a264cbd61c | ||
|
|
a97a160c44 | ||
|
|
df447f2673 | ||
|
|
db9ef12ff4 | ||
|
|
2c0df0d55f | ||
|
|
239c5ea672 | ||
|
|
6ea2fbadf7 | ||
|
|
50d8974154 | ||
|
|
3fd3f797dc | ||
|
|
0e299d0692 | ||
|
|
9ead554f5c | ||
|
|
77c188c640 | ||
|
|
2a59817f51 | ||
|
|
9c450a0352 | ||
|
|
ab7dfbd859 | ||
|
|
dda2bd3ca5 | ||
|
|
ceea6d909e | ||
|
|
ecf34ba51a | ||
|
|
ebbd5d5373 | ||
|
|
20a09530cd | ||
|
|
666be883dc | ||
|
|
65cd3f5c74 | ||
|
|
1ef0670624 | ||
|
|
10c1cfc6a7 | ||
|
|
8ced804a6f | ||
|
|
256864f613 | ||
|
|
a3a6ad3543 | ||
|
|
0161a92c52 | ||
|
|
0e80a1a058 | ||
|
|
8b27539145 | ||
|
|
a0add3b771 | ||
|
|
b58940fcf2 | ||
|
|
11aa23f1a1 | ||
|
|
42c31d4b48 | ||
|
|
f95830961d | ||
|
|
6b5a82bef8 | ||
|
|
490426c277 | ||
|
|
31c556d34c | ||
|
|
7f35b0e910 | ||
|
|
73b1050a83 | ||
|
|
d81e7de911 | ||
|
|
3201a58df7 | ||
|
|
19a7e9cbf8 |
+42
-1
@@ -11,7 +11,7 @@ ONLYOFFICE_PASS=
|
||||
# ONLYOFFICE_NAME=
|
||||
# ONLYOFFICE_PASSWORD=
|
||||
#
|
||||
# produktor.io operator aliases (CLI-only):
|
||||
# Optional CLI-only aliases:
|
||||
# OO_URL=
|
||||
# OO_USER=
|
||||
# OO_PASS=
|
||||
@@ -22,6 +22,47 @@ ONLYOFFICE_PROJECT_ID=33
|
||||
|
||||
# oo mails uses ONLYOFFICE_URL/USER/PASS above (Workspace Mail addon).
|
||||
|
||||
# deploy/docker-compose.rclone-webdav.yml — rclone FUSE mount of the oo-webdav
|
||||
# sidecar. Reuses ONLYOFFICE_USER and accepts ONLYOFFICE_PASSWORD (alias
|
||||
# ONLYOFFICE_PASS). Set ONLYOFFICE_WEBDAV_URL only if the sidecar is not on
|
||||
# the default docker bridge address. See docs/rclone-webdav.md.
|
||||
# ONLYOFFICE_WEBDAV_URL=http://172.17.0.1:8098/webdav
|
||||
|
||||
# cmd/office TUI — optional Document Server for DOCX→HTML preview:
|
||||
# ONLYOFFICE_DOCS_URL=https://docs.example.com
|
||||
# ONLYOFFICE_DOCS_SECRET=
|
||||
|
||||
# MinIO download fallback for the portal's stale AWS S3 consumer (older
|
||||
# Documents folders). When the portal redirects to amazonaws.com with access
|
||||
# key "minio" (403 InvalidAccessKeyId), files are fetched from the local MinIO
|
||||
# store instead. Without a key/secret the fallback is disabled.
|
||||
# MINIO_ENDPOINT=http://192.168.188.10:9000
|
||||
# MINIO_BUCKET=office
|
||||
# MINIO_ACCESS_KEY=
|
||||
# MINIO_SECRET_KEY=
|
||||
|
||||
# oo search — direct Elasticsearch access for name + content search. ES lives
|
||||
# inside the OnlyOffice VM on localhost:9200; expose it with an SSH tunnel
|
||||
# (see docs/elasticsearch.md). ONLYOFFICE_ES_INDEX defaults to files_file.
|
||||
# ONLYOFFICE_ES_URL=http://127.0.0.1:9200
|
||||
# ONLYOFFICE_ES_INDEX=files_file
|
||||
# ONLYOFFICE_TENANT=
|
||||
|
||||
# Read-only SQL file store over the Community Server database (see
|
||||
# docs/community-server-db.md). The live portal runs MySQL; ONLYOFFICE_DSN is
|
||||
# `user:pass@tcp(host:port)/onlyoffice?parseTime=true`, or a `postgres://` URL.
|
||||
# ONLYOFFICE_DSN=
|
||||
# ONLYOFFICE_PG_DRIVER= # postgres | mysql (auto-detected from DSN)
|
||||
# ONLYOFFICE_PG_TENANT=1 # falls back to ONLYOFFICE_TENANT
|
||||
# Alternatively build a PostgreSQL DSN from parts:
|
||||
# ONLYOFFICE_PG_HOST=
|
||||
# ONLYOFFICE_PG_PORT=5432
|
||||
# ONLYOFFICE_PG_USER=
|
||||
# ONLYOFFICE_PG_PASSWORD=
|
||||
# ONLYOFFICE_PG_DBNAME=onlyoffice
|
||||
# ONLYOFFICE_PG_SSLMODE=disable
|
||||
|
||||
# oo index / oo search --backend own — own full-text index for PDF/scans,
|
||||
# filled by `oo index` from internal/docpipe (pdftotext + OCR). Defaults to
|
||||
# oo_docs_text. Uses the same ONLYOFFICE_ES_URL.
|
||||
# ONLYOFFICE_ES_TEXT_INDEX=oo_docs_text
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
github: eSlider
|
||||
ko_fi: eslider
|
||||
liberapay: eslider
|
||||
patreon: eslider
|
||||
custom:
|
||||
- https://polar.sh/eslider
|
||||
@@ -14,6 +14,7 @@ permissions:
|
||||
jobs:
|
||||
release-please:
|
||||
name: Release Please
|
||||
if: github.server_url == 'https://github.com'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Run release-please
|
||||
|
||||
@@ -16,6 +16,7 @@ permissions:
|
||||
jobs:
|
||||
goreleaser:
|
||||
name: GoReleaser
|
||||
if: github.server_url == 'https://github.com'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
# Always clone default branch first. workflow_dispatch often races with
|
||||
|
||||
@@ -10,6 +10,51 @@ permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
secret-scan:
|
||||
name: Secret scan (gitleaks)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Compute scan range (diff of new commits only)
|
||||
id: range
|
||||
run: |
|
||||
if [ "$GITHUB_EVENT_NAME" = "pull_request" ]; then
|
||||
RANGE="${{ github.event.pull_request.base.sha }}...${{ github.event.pull_request.head.sha }}"
|
||||
else
|
||||
BEFORE="${{ github.event.before }}"
|
||||
if [ "$BEFORE" = "0000000000000000000000000000000000000000" ]; then
|
||||
RANGE="$(git rev-list --max-parents=0 HEAD | tail -1)..$GITHUB_SHA"
|
||||
else
|
||||
RANGE="$BEFORE..$GITHUB_SHA"
|
||||
fi
|
||||
fi
|
||||
echo "RANGE=$RANGE" >> "$GITHUB_ENV"
|
||||
echo "Scanning range: $RANGE"
|
||||
|
||||
# Install the gitleaks binary instead of a docker action: the
|
||||
# docker://zricethezav/gitleaks action hardcodes /github/workspace,
|
||||
# which does not exist on the Gitea (act) runner. $GITHUB_WORKSPACE is
|
||||
# the checkout dir on BOTH runners (GitHub and Gitea act). Mirrors the
|
||||
# fix applied to 2dph (issue #142).
|
||||
- name: Gitleaks (diff-only, fail on leak)
|
||||
env:
|
||||
GITLEAKS_RANGE: ${{ env.RANGE }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
curl -fsSLo /tmp/gitleaks.tar.gz \
|
||||
https://github.com/gitleaks/gitleaks/releases/download/v8.30.1/gitleaks_8.30.1_linux_x64.tar.gz
|
||||
tar -xzf /tmp/gitleaks.tar.gz -C /tmp gitleaks
|
||||
chmod +x /tmp/gitleaks
|
||||
/tmp/gitleaks detect \
|
||||
--source "$GITHUB_WORKSPACE" \
|
||||
--log-opts="$GITLEAKS_RANGE" \
|
||||
--redact \
|
||||
--verbose
|
||||
|
||||
test:
|
||||
name: Test (Go ${{ matrix.go }})
|
||||
runs-on: ubuntu-latest
|
||||
@@ -22,7 +67,7 @@ jobs:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
uses: actions/setup-go@v4
|
||||
with:
|
||||
go-version: ${{ matrix.go }}
|
||||
cache: true
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
{
|
||||
".": "0.9.0"
|
||||
".": "0.18.0"
|
||||
}
|
||||
|
||||
@@ -9,18 +9,19 @@ Canonical Go client for OnlyOffice Workspace (Projects + Calendar + CRM) and the
|
||||
- `request.go` — `Request`, `Query`, `Time`, `Token`, `MetaResponse`, `Permissions`.
|
||||
- `auth.go` — `Authenticate`, `AuthenticateContext`, `InvalidateToken`, `Auth`, token lifecycle.
|
||||
- `http.go` — transport + DRY response decoders (`ResponseArray`/`ResponseObject`/`postFormObject`/`putFormObject`/`deleteObject`).
|
||||
- `projects.go`, `tasks.go`, `users.go`, `calendar.go`, `crm.go`, `files.go`, `mails.go` — typed / untyped domain methods. **`files.go`** — CRM opportunity upload plus **project/task Documents** (`GetProjectFiles`, `UploadProjectFile`, `GetTaskFiles`, `AttachFilesToTask`, `UploadTaskFile`, `DetachTaskFile`, `GetFile`, `RenameFile`, `DeleteFiles`, `DownloadFile`). **`mails.go`** — OnlyOffice Workspace Mail addon (`ListMailAccounts`, `ListMailFolders`, `ListMailMessages`, `GetMailMessage`, `RemoveMailMessages`).
|
||||
- `projects.go`, `tasks.go`, `users.go`, `calendar.go`, `crm.go`, `files.go`, `files_webdav.go`, `files_stem.go`, `retry.go`, `mails.go`, `invoices.go` — typed / untyped domain methods. **`files.go`** — CRM opportunity upload plus **project/task Documents** (`UpdateFile`, `UploadToFolderReplacing`). **`files_webdav.go`** — Documents module by id (`ListDavFolder`, `MoveDavItems`/`CopyDavItems` with per-operation error surfacing, `ListFileOps`). **`retry.go`** — `DoRetry`: deterministic linear backoff (no jitter) on 429/502/503/504; every bulk tool routes API calls through it, and the HTTP transport + auth (`retryRaw`, `AuthenticateContext`) retry transient answers centrally. **`mails.go`** — OnlyOffice Workspace Mail. **`invoices.go`** — CRM invoices, PDF regen/cleanup, status. Association rules: [`docs/crm-associations.md`](docs/crm-associations.md).
|
||||
- **Unified file client (epic #34) — `file_core.go`, `file_rest.go`, `file_dav.go`, `file_pg.go`, `file_es.go`, `file_es_text.go`, `file_text_index.go`, `file_facade.go`.** `file_core.go` — model (`Entry`, `Kind`) + `FileStore`/`Searcher`; `file_rest.go`/`file_dav.go` — REST/WebDAV adapters; `file_pg.go` — **read-only** SQL store (PostgreSQL/MySQL, `ErrReadOnly` on writes); `file_es.go` — OnlyOffice Elasticsearch searcher; `file_es_text.go`/`file_text_index.go` — own PDF/scan index (`oo_docs_text`, PDF attachments via pdfdetach); `file_facade.go` — `FileClient` with read/write/search order and transient fallback. Use `c.Files()` (facade), `c.FileStore("rest"|"dav"|"pg"|"sql")` or `c.SQLFileStore()`; contract and how to add a backend: [`docs/unified-file-client.md`](docs/unified-file-client.md).
|
||||
- Pure stdlib + `google/go-querystring`; no UI, no dotenv.
|
||||
- **CLI — `cmd/oo/` as `package main`.** Cobra wrapper that loads `.env` via `godotenv` at startup. **Subject-based command tree** mirroring [`tea`](https://gitea.com/gitea/tea):
|
||||
- `main.go` — entry point (docstring lists the command tree).
|
||||
- `common.go` — `rootCmd`, `newOO`, `printTable`/`printObject`, `--output table|json` flag.
|
||||
- `calendar.go`, `projects.go`, `projects_files.go`, `tasks.go`, `tasks_files.go`, `users.go`, `contacts.go`, `opportunities.go`, `cases.go`, `crm_tasks.go`, `apps.go`, `catalog.go` — one file per subject (or per subject facet), each registers in `init()`.
|
||||
- `calendar.go`, `projects.go`, `projects_files.go`, `tasks.go`, `tasks_files.go`, `users.go`, `contacts.go`, `opportunities.go`, `cases.go`, `crm.go`, `crm_tasks.go`, `catalog.go`, `docs.go`, `dav.go`, `search.go`, `index.go`, `mails.go`, `invoices.go` — one file per subject (or per subject facet), each registers in `init()`. `dav.go` exposes the Documents module by id (`oo dav ls|move|copy|mkdir|rename-file|rename-folder|download|fileops`); `search.go` runs `oo search QUERY` (name/content, `--backend oo|own`); `index.go` fills the own full-text index (`oo index folder|files`, see [`docs/unified-file-client.md`](docs/unified-file-client.md)).
|
||||
- CLI-only deps (`spf13/cobra`, `joho/godotenv`) stay out of the library.
|
||||
- **Catalog inventory — `catalog/` package.** Filesystem VCF/project scan → YAML → match/apply against CRM. Used by `oo catalog`; not part of the flat `*Client` surface (uses Client as a dependency).
|
||||
- **TUI — `cmd/office/` as `package main`.** Bubble Tea three-pane browser (module tree, selectable list, markdown preview). Reuses `cmd/internal/bootstrap` for env/auth and the root `onlyoffice` library for all API calls. UI logic in `cmd/office/ui/`; preview/formatting in `cmd/office/preview/`; list loaders in `cmd/office/fetch/`.
|
||||
- **List table (`DataTable`)** — `cmd/office/ui/table*.go`. Column layout policies live in `cmd/office/model/table_layout.go` (`TableFlexLayoutFor`); cell rendering uses the bubbles/table inline pattern in `table_render.go` (`renderTableCell`, `padANSIWidth`). See `.cursor/skills/office-tui-table/SKILL.md` before changing center-pane tables.
|
||||
- **Shared bootstrap — `cmd/internal/bootstrap/`.** `LoadEnv()` + `NewClient(ctx)` extracted from `oo`; both binaries import it.
|
||||
- **Applications sync — `cmd/oo/applications/`.** README→CRM bridge, CV-specific; kept under `cmd/oo/` so it's clear it's internal to the binary, not a library feature.
|
||||
- **Bulk Documents tools — `cmd/ooscan/`, `cmd/pdfamount/`, `cmd/kontoblatt/`, `cmd/kontolink/`.** Single-purpose binaries (folder index, PDF amounts, Kontoblatt summary/linking). Pace requests, route API calls through `DoRetry`; usage in README.
|
||||
- **Personal ops tooling** (disk inventory, dossier→CRM sync, SearXNG) lives in private [`eSlider/oo-workspace`](https://git.produktor.io/eSlider/oo-workspace) (`oow`), not in this public tree.
|
||||
|
||||
## Rules
|
||||
|
||||
@@ -28,7 +29,8 @@ Canonical Go client for OnlyOffice Workspace (Projects + Calendar + CRM) and the
|
||||
- New endpoints go into the library first; CLI commands are thin wrappers.
|
||||
- Prefer `ResponseObject` / `postFormObject` / `putFormObject` / `deleteObject` over hand-rolled `json.Unmarshal(responseField(...))` blocks — they exist for DRY, use them.
|
||||
- Domain split is by file, **not** by subpackage. Don't introduce `internal/` or `pkg/*` subpackages inside the library — it flattens the `*Client` call surface for a reason.
|
||||
- CLI commands follow **subject → verb** structure (`oo <subject> <verb>`), never `oo <verb>-<subject>`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/<subject>.go` for a genuinely new domain.
|
||||
- CLI commands follow **subject → verb** structure (`oo <subject> <verb>`), never `oo <verb>-<subject>`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/<subject>.go` for a genuinely new domain. The subject→verb tree in `cmd/oo/main.go` and the README table are documentation — update them with the code.
|
||||
- **Documents for agents:** prefer Markdown in git; OnlyOffice UI is weak for `.md`/`.txt`. Use `oo docs put-md` (md→docx) and `oo docs put-txt` (txt→docx, preserves line breaks). All upload paths default to **upsert** by `stem|ext` (`--replace`, default true); `--no-replace` fails on conflict; `--allow-duplicate` opts into raw OO append. `oo projects files dedupe PROJECT_ID` reports/removes duplicate stem|ext copies (`--apply`, `--cross`; includes project root folder).
|
||||
- Every table output goes through `printTable(headers, rows)`; every single-object through `printObject(v)`. Do not `fmt.Println` rows ad-hoc or the `--output json` flag breaks for that command.
|
||||
- No secrets in the repo; use `.env` (gitignored). Commit `.env.example` only.
|
||||
- Follow SemVer on tags; this repo is tagged at GitHub under `git@github.com:eSlider/go-onlyoffice.git`.
|
||||
@@ -54,6 +56,7 @@ write `mux.HandleFunc("/api/2.0/...")` to emulate OnlyOffice, we write an
|
||||
|
||||
## Related
|
||||
|
||||
- [`docs/README.md`](docs/README.md) — reference index (file client, ES, SQL, rclone).
|
||||
- [`eSlider/inventar`](https://git.produktor.io/eSlider/inventar) — ASR/ADR (see ASR-0008 Go library module conventions).
|
||||
- [`eSlider/inventar-sync`](https://git.produktor.io/eSlider/inventar-sync) — OnlyOffice → Gitea issue sync, consumes this library.
|
||||
- [`produktor.io/vidarr`](https://git.produktor.io/produktor.io/vidarr) — legacy consumer being migrated from `pkg/onlyoffice` to this module.
|
||||
|
||||
+176
@@ -4,6 +4,182 @@ All notable changes to this project are documented here. The format is based on
|
||||
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project
|
||||
adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||
|
||||
## Unreleased
|
||||
|
||||
## [0.18.0](https://github.com/eSlider/go-onlyoffice/compare/v0.17.0...v0.18.0) (2026-09-04)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **crm:** add UpdateContactName and CloseCRMTask helpers ([4d8af7a](https://github.com/eSlider/go-onlyoffice/commit/4d8af7a2fef1fcfd96ad913cefc6497e3389963b))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **crm:** deterministic sortBy=id in contact paged lists ([4d91726](https://github.com/eSlider/go-onlyoffice/commit/4d917261792203732ac739644c9c82124b2166eb))
|
||||
|
||||
|
||||
### Documentation
|
||||
|
||||
* **funding:** eSlider support links (reverse-import GitHub e9c969a) ([d650a16](https://github.com/eSlider/go-onlyoffice/commit/d650a16a36037949505eb017c92bd3283e6e2f0f))
|
||||
|
||||
## [0.17.0](https://github.com/eSlider/go-onlyoffice/compare/v0.16.0...v0.17.0) (2026-08-31)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **mailsync:** FetchMailFolder — integration-layer walk for ETL consumers ([35f0cb8](https://github.com/eSlider/go-onlyoffice/commit/35f0cb8d20076244141065c07e203e633bc3612a))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **files:** upsert uploads by default and dedupe project root ([24ca144](https://github.com/eSlider/go-onlyoffice/commit/24ca144b22abc5d056a5fd1ed9a1887f26a79d15))
|
||||
|
||||
## [0.16.0](https://github.com/eSlider/go-onlyoffice/compare/v0.15.0...v0.16.0) (2026-08-30)
|
||||
|
||||
### Features
|
||||
|
||||
* **docs:** `put-xlsx` — multi-sheet бюджеты с named inputs, SUM/AVG/MIN
|
||||
формулами, cross-sheet ссылками и cell comments (`internal/xlspipe`,
|
||||
excelize) ([68445b0](https://github.com/eSlider/go-onlyoffice/commit/68445b0))
|
||||
|
||||
### Added
|
||||
|
||||
* **docs:** CRM association graph and OO quirks (`docs/crm-associations.md`)
|
||||
* **crm:** `ForceRegenerateInvoicePDF`, `SetInvoiceStatus`, `PurgeStaleInvoicePDFs`, contact/opportunity file list helpers
|
||||
* **oo:** `invoices pdf`, `pdf-cleanup`, `status`; create `--consignee`; draft-invoice force-regens PDF
|
||||
|
||||
### Fixed
|
||||
|
||||
* Document that invoice→deal must be set at create (`update --opportunity` often HTTP 400)
|
||||
|
||||
## [0.15.0](https://github.com/eSlider/go-onlyoffice/compare/v0.14.0...v0.15.0) (2026-08-29)
|
||||
|
||||
### Features
|
||||
|
||||
* **files:** `ListFolder`, `CreateFolder`, `MoveFiles`, `UploadToFolder` — Documents folder helpers for OO Documents ingestion ([6ea2fba](https://github.com/eSlider/go-onlyoffice/commit/6ea2fba))
|
||||
* **files:** dedupe by stem|ext (`files_stem.go`) + `oo projects files dedupe` ([ac06d86](https://github.com/eSlider/go-onlyoffice/commit/ac06d86))
|
||||
* **docs:** `internal/docpipe` — md↔docx convert, OCR→PDF, hOCR→Markdown via go-hocr, as-md/put-md for agents ([6ea2fba](https://github.com/eSlider/go-onlyoffice/commit/6ea2fba), [a264cbd](https://github.com/eSlider/go-onlyoffice/commit/a264cbd))
|
||||
* **docs:** optimize PDF via Ghostscript pdfwrite ([6b3f40e](https://github.com/eSlider/go-onlyoffice/commit/6b3f40e))
|
||||
* **security:** gitleaks secret-scan in CI + pre-push/pre-commit hooks ([9ead554](https://github.com/eSlider/go-onlyoffice/commit/9ead554))
|
||||
|
||||
### Fixes
|
||||
|
||||
* **files:** `DeleteFiles` via per-file DELETE API; `DeleteDavItems` with `Immediately` true ([622dcdb](https://github.com/eSlider/go-onlyoffice/commit/622dcdb), [d9a7adc](https://github.com/eSlider/go-onlyoffice/commit/d9a7adc))
|
||||
* **docs:** put-txt preserves line breaks in DOCX; fixed-width extracts in code block; put-md upsert by stem ([309ae44](https://github.com/eSlider/go-onlyoffice/commit/309ae44), [f1739dc](https://github.com/eSlider/go-onlyoffice/commit/f1739dc), [50bd475](https://github.com/eSlider/go-onlyoffice/commit/50bd475))
|
||||
* **ci:** point gitleaks at `/github/workspace` in docker action ([2c0df0d](https://github.com/eSlider/go-onlyoffice/commit/2c0df0d))
|
||||
|
||||
## [0.14.0](https://github.com/eSlider/go-onlyoffice/compare/v0.13.0...v0.14.0) (2026-08-27)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **crm:** contact email & person-opportunity indexes, history entity whitelist ([9c450a0](https://github.com/eSlider/go-onlyoffice/commit/9c450a0352c25f12ac16d91062744ab026ffe660))
|
||||
* **docs:** hOCR → Markdown via go-hocr ([e531de2](https://github.com/eSlider/go-onlyoffice/commit/e531de280f2c521a7411f75212803649b177db58))
|
||||
* **docs:** md↔docx convert, OCR→PDF, as-md/put-md for agents ([6ea2fba](https://github.com/eSlider/go-onlyoffice/commit/6ea2fbadf768d7d85f6c7a49a9bc1050e642bc56))
|
||||
* **docs:** md↔docx, OCR→PDF, as-md/put-md ([db9ef12](https://github.com/eSlider/go-onlyoffice/commit/db9ef12ff4096f8aadb557ca128b6edcb503028c))
|
||||
* **docs:** oo docs hocr — tesseract hOCR → go-hocr Markdown/YAML ([a264cbd](https://github.com/eSlider/go-onlyoffice/commit/a264cbd61c12be9098c8d16e7c1c1253eb460816))
|
||||
* **mail:** oo mails send — SendMail + guard empty-by-id send ([dda2bd3](https://github.com/eSlider/go-onlyoffice/commit/dda2bd3ca5cea6ebb89e1ac02b785b9838727bda))
|
||||
* **mail:** oo mails send — SendMail client + guard empty-by-id ([ab7dfbd](https://github.com/eSlider/go-onlyoffice/commit/ab7dfbd8594de5724286e2b460e789f4acb114bc))
|
||||
* **security:** secret-scan via gitleaks in CI + pre-push/pre-commit hooks ([#142](https://github.com/eSlider/go-onlyoffice/issues/142)) ([9ead554](https://github.com/eSlider/go-onlyoffice/commit/9ead554f5cc229687d3a27934a023fb1dddaef15))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **ci:** point gitleaks at /github/workspace in docker action ([2c0df0d](https://github.com/eSlider/go-onlyoffice/commit/2c0df0d55fb03555b60481695554d75dceffce48))
|
||||
* **docs:** put-md upsert — no duplicate folder files ([2b267d3](https://github.com/eSlider/go-onlyoffice/commit/2b267d36a8086b14a94706468f7e92e9179e62d4))
|
||||
* **docs:** put-md upsert by stem to avoid duplicate folder files ([50bd475](https://github.com/eSlider/go-onlyoffice/commit/50bd47570b6430cebde43641a7b1d2f8d54d32f2))
|
||||
* **files:** DeleteFiles actually removes files on produktor OO ([a8cb4e9](https://github.com/eSlider/go-onlyoffice/commit/a8cb4e97805226e8af694ec6a64ee3ac6a5e9fdd))
|
||||
* **files:** DeleteFiles via per-file DELETE API ([622dcdb](https://github.com/eSlider/go-onlyoffice/commit/622dcdb7bffd49acbea5ef07f5d5d009f44b1358))
|
||||
|
||||
|
||||
### Documentation
|
||||
|
||||
* **readme:** document oo docs hocr ([70e605b](https://github.com/eSlider/go-onlyoffice/commit/70e605b4ca8ee005e4351b2fc2d2eec081604998))
|
||||
* **readme:** document oo docs md↔docx / OCR agent workflow ([239c5ea](https://github.com/eSlider/go-onlyoffice/commit/239c5ea67201e62de6ba067f0e7963c7c9bda188))
|
||||
|
||||
## [0.13.0](https://github.com/eSlider/go-onlyoffice/compare/v0.12.0...v0.13.0) (2026-08-27)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **crm:** contact email & person-opportunity indexes, history entity whitelist ([9c450a0](https://github.com/eSlider/go-onlyoffice/commit/9c450a0352c25f12ac16d91062744ab026ffe660))
|
||||
* **docs:** hOCR → Markdown via go-hocr ([e531de2](https://github.com/eSlider/go-onlyoffice/commit/e531de280f2c521a7411f75212803649b177db58))
|
||||
* **docs:** md↔docx convert, OCR→PDF, as-md/put-md for agents ([6ea2fba](https://github.com/eSlider/go-onlyoffice/commit/6ea2fbadf768d7d85f6c7a49a9bc1050e642bc56))
|
||||
* **docs:** md↔docx, OCR→PDF, as-md/put-md ([db9ef12](https://github.com/eSlider/go-onlyoffice/commit/db9ef12ff4096f8aadb557ca128b6edcb503028c))
|
||||
* **docs:** oo docs hocr — tesseract hOCR → go-hocr Markdown/YAML ([a264cbd](https://github.com/eSlider/go-onlyoffice/commit/a264cbd61c12be9098c8d16e7c1c1253eb460816))
|
||||
* **mail:** oo mails send — SendMail + guard empty-by-id send ([dda2bd3](https://github.com/eSlider/go-onlyoffice/commit/dda2bd3ca5cea6ebb89e1ac02b785b9838727bda))
|
||||
* **mail:** oo mails send — SendMail client + guard empty-by-id ([ab7dfbd](https://github.com/eSlider/go-onlyoffice/commit/ab7dfbd8594de5724286e2b460e789f4acb114bc))
|
||||
* **security:** secret-scan via gitleaks in CI + pre-push/pre-commit hooks ([#142](https://github.com/eSlider/go-onlyoffice/issues/142)) ([9ead554](https://github.com/eSlider/go-onlyoffice/commit/9ead554f5cc229687d3a27934a023fb1dddaef15))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **ci:** point gitleaks at /github/workspace in docker action ([2c0df0d](https://github.com/eSlider/go-onlyoffice/commit/2c0df0d55fb03555b60481695554d75dceffce48))
|
||||
* **files:** rewrite viewUrl host to API base on download ([ecf34ba](https://github.com/eSlider/go-onlyoffice/commit/ecf34ba51aae77f1626a60795f7329f25d9344e5))
|
||||
|
||||
|
||||
### Documentation
|
||||
|
||||
* **readme:** document oo docs hocr ([70e605b](https://github.com/eSlider/go-onlyoffice/commit/70e605b4ca8ee005e4351b2fc2d2eec081604998))
|
||||
* **readme:** document oo docs md↔docx / OCR agent workflow ([239c5ea](https://github.com/eSlider/go-onlyoffice/commit/239c5ea67201e62de6ba067f0e7963c7c9bda188))
|
||||
|
||||
## [0.12.0](https://github.com/eSlider/go-onlyoffice/compare/v0.11.0...v0.12.0) (2026-08-27)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **crm:** contact email & person-opportunity indexes, history entity whitelist ([9c450a0](https://github.com/eSlider/go-onlyoffice/commit/9c450a0352c25f12ac16d91062744ab026ffe660))
|
||||
* **docs:** md↔docx convert, OCR→PDF, as-md/put-md for agents ([6ea2fba](https://github.com/eSlider/go-onlyoffice/commit/6ea2fbadf768d7d85f6c7a49a9bc1050e642bc56))
|
||||
* **docs:** md↔docx, OCR→PDF, as-md/put-md ([db9ef12](https://github.com/eSlider/go-onlyoffice/commit/db9ef12ff4096f8aadb557ca128b6edcb503028c))
|
||||
* **files:** add WebDAV-oriented Files operations ([ebbd5d5](https://github.com/eSlider/go-onlyoffice/commit/ebbd5d5373abfeca5f160312f201cbd683f55e42))
|
||||
* **mail:** oo mails send — SendMail + guard empty-by-id send ([dda2bd3](https://github.com/eSlider/go-onlyoffice/commit/dda2bd3ca5cea6ebb89e1ac02b785b9838727bda))
|
||||
* **mail:** oo mails send — SendMail client + guard empty-by-id ([ab7dfbd](https://github.com/eSlider/go-onlyoffice/commit/ab7dfbd8594de5724286e2b460e789f4acb114bc))
|
||||
* **security:** secret-scan via gitleaks in CI + pre-push/pre-commit hooks ([#142](https://github.com/eSlider/go-onlyoffice/issues/142)) ([9ead554](https://github.com/eSlider/go-onlyoffice/commit/9ead554f5cc229687d3a27934a023fb1dddaef15))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **ci:** point gitleaks at /github/workspace in docker action ([2c0df0d](https://github.com/eSlider/go-onlyoffice/commit/2c0df0d55fb03555b60481695554d75dceffce48))
|
||||
* **files:** rewrite viewUrl host to API base on download ([ecf34ba](https://github.com/eSlider/go-onlyoffice/commit/ecf34ba51aae77f1626a60795f7329f25d9344e5))
|
||||
|
||||
|
||||
### Documentation
|
||||
|
||||
* **readme:** document oo docs md↔docx / OCR agent workflow ([239c5ea](https://github.com/eSlider/go-onlyoffice/commit/239c5ea67201e62de6ba067f0e7963c7c9bda188))
|
||||
|
||||
## [0.11.0](https://github.com/eSlider/go-onlyoffice/compare/v0.10.0...v0.11.0) (2026-08-18)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **catalog:** port oo catalog (scan/merge/match/apply) from legacy branch ([256864f](https://github.com/eSlider/go-onlyoffice/commit/256864f613f31570cf995dddc51255a9a91a38f8))
|
||||
|
||||
## [0.10.0](https://github.com/eSlider/go-onlyoffice/compare/v0.9.0...v0.10.0) (2026-08-13)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **crm:** add invoices CLI and opportunity update ([19a7e9c](https://github.com/eSlider/go-onlyoffice/commit/19a7e9cbf8f7bd335c4e4cfd4d3b8a84a350cc41))
|
||||
* **crm:** allow updating invoice terms via oo ([31c556d](https://github.com/eSlider/go-onlyoffice/commit/31c556d34cedd65d6957441f448520af9a7e380b))
|
||||
* **crm:** invoice update notes and PO fields ([d81e7de](https://github.com/eSlider/go-onlyoffice/commit/d81e7de9114a29439142efba67f205653f562730))
|
||||
* **crm:** invoices CLI and opportunity update ([f958309](https://github.com/eSlider/go-onlyoffice/commit/f95830961d3aecba65f3cb074d33c9c98f6c6cf2))
|
||||
* **mail:** draft, attach, and draft-invoice CLI ([73b1050](https://github.com/eSlider/go-onlyoffice/commit/73b1050a83e9e49dcf99d8a966a84746376d7dd9))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **crm:** JSON person update and oo persons update CLI ([42c31d4](https://github.com/eSlider/go-onlyoffice/commit/42c31d4b485bfab6903aeed0dd6fb2c43f61b4f2))
|
||||
* **crm:** link invoices to opportunities ([3201a58](https://github.com/eSlider/go-onlyoffice/commit/3201a58df75bceb7d41034970cf384a9169163ed))
|
||||
* **mail:** German spacing for invoice draft template ([7f35b0e](https://github.com/eSlider/go-onlyoffice/commit/7f35b0e91075d871f87de95b6c84b2353e02f9cf))
|
||||
|
||||
|
||||
### Documentation
|
||||
|
||||
* **crm:** capture association graph and invoice/mail quirks ([490426c](https://github.com/eSlider/go-onlyoffice/commit/490426c2779160f1a9b168f6ccae9dd8dccf89e9))
|
||||
* **crm:** clarify mail API send vs signature for chat links ([6b5a82b](https://github.com/eSlider/go-onlyoffice/commit/6b5a82bef831341aee16b0b0ea1fe6db2c24567d))
|
||||
* **crm:** Team vs Contacts and persons update note ([11aa23f](https://github.com/eSlider/go-onlyoffice/commit/11aa23f1a17471bad2bd95fddb769a94afbcf97a))
|
||||
|
||||
## [0.9.0](https://github.com/eSlider/go-onlyoffice/compare/v0.8.3...v0.9.0) (2026-07-26)
|
||||
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Example Author
|
||||
Copyright (c) 2026 Andriy Oblivantsev
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
|
||||
@@ -503,6 +503,29 @@ type Task struct {
|
||||
|---|---|
|
||||
| `GetUsers()` | List all users with profiles |
|
||||
|
||||
### Documents Files
|
||||
|
||||
| Method | Description |
|
||||
|---|---|
|
||||
| `ListDavFolder(ctx, id)` | List a Documents folder (`@root` for virtual sections) |
|
||||
| `ListDavSections(ctx)` | Virtual sections (Documents, Projects, …) |
|
||||
| `CreateDavFolder(ctx, parentID, title)` | Create a subfolder |
|
||||
| `RenameDavFolder(ctx, id, title)` / `RenameDavFile(ctx, id, title)` | Rename folder / file |
|
||||
| `DownloadFile(ctx, id, dst)` / `DownloadDavFile(ctx, id, w)` | Download file bytes |
|
||||
| `UploadDavFile(ctx, folderID, fileName, src)` | Upload from a reader |
|
||||
| `UploadToFolder(ctx, folderID, localPath)` | Upload a local file into a folder |
|
||||
| `UploadToFolderReplacing(ctx, folderID, localPath)` | Upsert by `stem\|ext`; returns replaced ids |
|
||||
| `UpdateFile(ctx, fileID, localPath)` | New version of an existing file (same id, no copy) |
|
||||
| `MoveDavItems(ctx, folderIDs, fileIDs, dest)` | Move (`resolveType=Skip`); per-operation errors surfaced, not silent nil |
|
||||
| `CopyDavItems(ctx, folderIDs, fileIDs, dest)` | Copy (`conflictResolveType=Skip`); errors surfaced |
|
||||
| `MoveFiles(ctx, destFolderID, fileIDs)` | Move with `resolveType=Skip` + `holdResult`; errors surfaced |
|
||||
| `ListFileOps(ctx)` | Active file operations (move/copy status polling) |
|
||||
| `FolderFiles(ctx, folderID)` | Flat file list of a folder (stem helpers) |
|
||||
| `DeleteFilesByStem(ctx, folderID, stem)` | Remove `stem\|ext` copies |
|
||||
| `DoRetry(ctx, policy, fn)` | Deterministic linear backoff (N·Base, no jitter) on 429/502/503/504 |
|
||||
| `DefaultRetryPolicy()` | 5 attempts, 1s·2s·3s·4s waits, 30s cap |
|
||||
| `Transient(err)` | True for retriable OnlyOffice answers |
|
||||
|
||||
### Helper Types
|
||||
|
||||
| Type | Description |
|
||||
@@ -533,7 +556,7 @@ deals, total, _ := client.ListOpportunities(ctx, 50, 0)
|
||||
company, _ := client.FindCompany(ctx, "ACME")
|
||||
|
||||
// Subtasks (form-encoded)
|
||||
client.AddSubtask(ctx, "4242", "Prepare CV")
|
||||
client.AddSubtask(ctx, "4242", "Prepare notes")
|
||||
```
|
||||
|
||||
### oo (bundled CLI)
|
||||
@@ -552,14 +575,13 @@ oo calendar events --start 2026-04-24 --end 2026-05-01
|
||||
oo projects list
|
||||
oo projects get 33
|
||||
oo tasks list --all --verbose
|
||||
oo tasks subtask add 4242 "Prepare CV"
|
||||
oo tasks subtask add 4242 "Prepare notes"
|
||||
oo persons create --first Jane --last Doe --email jane@example.com
|
||||
oo companies create --name "Acme GmbH" --website https://acme.com
|
||||
oo opportunities list
|
||||
oo opportunities stages
|
||||
oo cases list
|
||||
oo crm-tasks categories
|
||||
oo applications sync --path ./applications/2026 --apply
|
||||
```
|
||||
|
||||
### office (TUI)
|
||||
@@ -608,49 +630,156 @@ go test -tags=integration ./cmd/office/fetch/... ./cmd/office/preview/...
|
||||
```bash
|
||||
# Project Documents (files module)
|
||||
oo projects files list 33
|
||||
oo projects files upload 33 ./notes.md
|
||||
oo projects files download 12345 --to ./copy.md
|
||||
oo projects files rename 12345 notes-v2.md
|
||||
oo projects files upload 33 ./notes.docx
|
||||
oo projects files download 12345 --to ./copy.docx
|
||||
oo projects files rename 12345 notes-v2.docx
|
||||
oo projects files delete 12345
|
||||
oo projects files dedupe 7 # dry-run duplicate report
|
||||
oo projects files dedupe 7 --apply # remove older stem|ext copies per folder
|
||||
oo projects files dedupe 7 --cross --apply # cross-folder; keep non-_trash
|
||||
|
||||
# Agent document pipeline (md in git ↔ docx in OO; OCR scans)
|
||||
oo docs tools
|
||||
oo docs convert ./note.md # → note.docx
|
||||
oo docs convert ./note.docx # → note.md
|
||||
oo docs ocr ./scan.jpg --md ./scan.md # searchable PDF + markdown
|
||||
oo docs hocr ./scan.jpg --lang spa --md ./scan.hocr.md --yaml ./scan.yml
|
||||
oo docs put-md 7 ./OO-HONDA-7-INDEX.md --folder 490
|
||||
oo docs put-txt 7 ./notes.txt --folder 490
|
||||
oo docs put-xlsx 7 ./table.xlsx --folder 490
|
||||
oo docs as-md 2815 --to ./parte.md # download OO file as MD (OCR if needed)
|
||||
oo docs as-md 307 --hocr --lang spa # OO download via go-hocr structure
|
||||
oo projects files put-md 7 ./note.md # alias
|
||||
oo projects files as-md 2815 # alias
|
||||
oo tasks files list 208
|
||||
oo tasks files upload 208 ./cv.pdf
|
||||
oo tasks files upload 208 ./notes.pdf
|
||||
oo tasks files detach 208 12345
|
||||
```
|
||||
|
||||
### Documents module (`oo dav`)
|
||||
|
||||
Direct access to the Documents module by folder/file id — the same calls that
|
||||
back `oo-webdav` and the project/task file commands. `move` sends
|
||||
`resolveType=Skip` + `holdResult=true`: without those params the legacy
|
||||
`fileops/move` endpoint answers 200 without moving anything, and the library
|
||||
surfaces such per-operation errors instead of a silent nil
|
||||
(`MoveDavItems` / `CopyDavItems` / `MoveFiles`).
|
||||
|
||||
```bash
|
||||
oo dav ls 659
|
||||
oo dav ls @root # virtual sections (Documents, Projects, …)
|
||||
oo dav mkdir 659 "2026 inbox"
|
||||
oo dav move 659 22881 22882 # DEST_FOLDER_ID FILE_ID…
|
||||
oo dav move 659 22881 --folders 670 # move folders along with files
|
||||
oo dav copy 659 22881
|
||||
oo dav rename-file 22881 invoice-v2.pdf
|
||||
oo dav rename-folder 671 o2-archive
|
||||
oo dav download 22881 --to ./copy.pdf # default path: ./<server title>
|
||||
oo dav fileops # active move/copy operations (status polling)
|
||||
```
|
||||
|
||||
### Search and index (`oo search`, `oo index`)
|
||||
|
||||
Full-text search over the Documents index. The REST endpoint
|
||||
`/api/2.0/files/@search/{query}` only searches file names in the database, so
|
||||
`oo search` talks to the OnlyOffice **Elasticsearch** directly (index
|
||||
`files_file`). Name search is default; `--content` also matches extracted
|
||||
document text (`document.attachment.content`, Office formats only).
|
||||
See [`docs/elasticsearch.md`](docs/elasticsearch.md) for the tunnel setup.
|
||||
|
||||
```bash
|
||||
oo search "Rechnung" # names only
|
||||
oo search "Mahngebühr" --content # names + document text
|
||||
oo search "Rechnung" --folder 649 --limit 50
|
||||
oo search "Rechnung" --json # shorthand for -o json
|
||||
```
|
||||
|
||||
Requires `ONLYOFFICE_ES_URL` (plus optional `ONLYOFFICE_ES_INDEX`,
|
||||
`ONLYOFFICE_TENANT`).
|
||||
|
||||
#### PDF/scans: own index (`oo index` + `--backend own`)
|
||||
|
||||
The OnlyOffice index covers Office formats only, so PDFs (`S1019`-style invoice
|
||||
numbers) are not searchable by content. `oo index` extracts PDF text with
|
||||
`internal/docpipe` (pdftotext, OCR for scans) — including the text of embedded
|
||||
PDF attachments (`pdfdetach`: `<doc>.md`, `.xml`, covers the original/scan and
|
||||
ZUGFeRD e-invoice XML) — into a separate index (`ONLYOFFICE_ES_TEXT_INDEX`,
|
||||
default `oo_docs_text`); the OnlyOffice server and its index are **not**
|
||||
modified. Then search it with `--backend own`.
|
||||
|
||||
```bash
|
||||
oo index folder 634 --recursive --exts pdf # populate (idempotent upsert)
|
||||
oo index files 3576 3578 # specific files
|
||||
oo index folder 634 --dry-run # plan only
|
||||
oo search "S1021" --content --backend own # finds the PDF
|
||||
oo search "Rechnung" --backend own --folder 634 --json
|
||||
```
|
||||
|
||||
See [`docs/elasticsearch.md`](docs/elasticsearch.md) for the decision and
|
||||
trade-offs.
|
||||
|
||||
### Unified file client
|
||||
|
||||
All file backends (REST, WebDAV, read-only SQL, Elasticsearch) sit behind one
|
||||
facade: `c.Files()` returns a `*FileClient` that also implements `FileStore`,
|
||||
so old call sites keep working. Pick a transport per call with
|
||||
`c.FileStore("rest"|"dav"|"pg"|"sql")` (SQL is read-only), open the SQL store
|
||||
with `c.SQLFileStore()`, or register a backend on the facade
|
||||
(`RegisterStore`/`RegisterSearcher`). Contract, model (`Entry`/`Kind`),
|
||||
fallback rules, env names and how to add a backend:
|
||||
[`docs/unified-file-client.md`](docs/unified-file-client.md).
|
||||
|
||||
### Bulk tools (`cmd/`)
|
||||
|
||||
Small single-purpose binaries for bulk Documents work. All of them pace
|
||||
requests and retry transient OnlyOffice answers (429/502/503/504) with a
|
||||
deterministic linear backoff — no jitter, same waits on every run
|
||||
(see `DoRetry` below). Build with `go build ./cmd/<tool>`.
|
||||
|
||||
```bash
|
||||
ooscan 659 # recursive index → TSV: file_id, folder_id, path, title
|
||||
ooscan 659 666 > oo-index.tsv # several roots into one index
|
||||
pdfamount 671 # "Zu zahlender Betrag" per PDF → TSV: file_id, title, amount
|
||||
kontoblatt 3906 ./kontoblatt.xlsx # summary (Gegenkonto/Monat) uploaded next to source
|
||||
kontolink IN.xlsx oo-index.tsv OUT.xlsx [FILE_ID] [AMOUNTS_TSV]
|
||||
# kontolink writes DocEditor links into the Link column: Beleg → supplier+month
|
||||
# → amount+date (5th arg = pdfamount output); with FILE_ID it updates the
|
||||
# source file in place, else uploads an "(links)" copy next to it.
|
||||
```
|
||||
|
||||
| Subject | Verbs |
|
||||
|---|---|
|
||||
| `calendar` | `list`, `events`, `add`, `delete` |
|
||||
| `projects` | `list`, `get`, `milestones`, `create`, `update`, `delete`, **`files`** (`list`, `upload`, `download`, `rename`, `delete`) |
|
||||
| `projects` | `list`, `get`, `milestones`, `milestone-create`, `create`, `update`, `delete`, `contacts` (`add`, `remove`), `link-authors`, `link-git`, **`files`** (`list`, `upload`, `download`, `rename`, `delete`, `dedupe`, `as-md`, `put-md`, `put-txt`, `put-xlsx`) |
|
||||
| `tasks` | `list`, `get`, `create`, `update`, `delete`, `subtask add`, **`files`** (`list`, `upload`, `detach`) |
|
||||
| `users` | `list`, `self` (alias: `oo whoami`) |
|
||||
| `contacts` | `list`, `get`, `delete`, `info-add` |
|
||||
| `persons` | `list` (filtered), `create`, `delete` |
|
||||
| `companies` | `list` (filtered), `create`, `delete` |
|
||||
| `contacts` | `list`, `get`, `delete`, `info-add`, `dedupe-info` |
|
||||
| `contacts` | `list`, `get`, `delete`, `info-add`, `merge`, `dedupe-info`, `tags`, `tag-add`, `tag-create`, `tag-remove` |
|
||||
| `persons` | `list`, `create`, `delete`, `dedupe` |
|
||||
| `companies` | `list`, `create`, `delete`, `dedupe`, `dedupe-persons` |
|
||||
| `opportunities` | `list`, `get`, `create`, `delete`, `stages`, `member-add`, `dedupe`, `dedupe-members`, `fix-titles` |
|
||||
| `opportunities` | `list`, `get`, `create`, `update`, `delete`, `stages`, `member-add`, `dedupe`, `dedupe-members`, `fix-titles` |
|
||||
| `invoices` | `list`, `get`, `create`, `update`, `pdf`, `pdf-cleanup`, `status`, `delete`, `items …` |
|
||||
| `crm` | `cleanup` |
|
||||
| `mails` | `accounts`, `folders`, `list`, `get`, `delete` |
|
||||
| `mails` | `accounts`, `folders`, `list`, `get`, `download-attachment`, `draft`, `attach`, `draft-invoice`, `send`, `delete` |
|
||||
| `cases` | `list`, `create`, `delete`, `member-add` |
|
||||
| `crm-tasks` | `list`, `create`, `delete`, `categories` |
|
||||
| `applications` | `sync` |
|
||||
| `catalog` | `scan-contacts`, `scan-projects`, `scan-thunderbird`, `merge`, `match`, `apply` |
|
||||
| `crm-tasks` | `list`, `create`, `delete`, `categories`, `reassign-self` |
|
||||
| `docs` | `tools`, `convert`, `optimize`, `ocr`, `hocr`, `as-md`, `put-md`, `put-txt`, `put-xlsx` |
|
||||
| `catalog` | `match`, `merge`, `apply`, `scan-contacts`, `scan-projects`, `scan-thunderbird` |
|
||||
| `dav` | `ls`, `move`, `copy`, `mkdir`, `rename-file`, `rename-folder`, `download`, `fileops` |
|
||||
| `search` | `QUERY` (`--content`, `--folder ID`, `--limit N`, `--backend oo\|own`, `--json`) |
|
||||
| `index` | `folder FOLDER_ID`, `files FILE_ID...` (`--recursive`, `--exts pdf`, `--limit N`, `--dry-run`) |
|
||||
|
||||
The CLI reads only `.env` from the current working directory (godotenv is a
|
||||
CLI-only concern — the library itself never loads dotfiles).
|
||||
|
||||
Canonical `ONLYOFFICE_*` variables win over aliases. For produktor.io operator
|
||||
files, `OO_URL` / `OO_USER` / `OO_PASS` are accepted as CLI-only aliases for
|
||||
`ONLYOFFICE_URL` / `ONLYOFFICE_USER` / `ONLYOFFICE_PASS`.
|
||||
Canonical `ONLYOFFICE_*` variables win over aliases. Optional CLI-only aliases:
|
||||
`OO_URL` / `OO_USER` / `OO_PASS` → `ONLYOFFICE_URL` / `ONLYOFFICE_USER` / `ONLYOFFICE_PASS`.
|
||||
|
||||
Run `oo --help` or `oo <subject> --help` for the full command reference.
|
||||
|
||||
> **0.5.0 migration note:** the command tree was flattened per-subject. Old
|
||||
> flat names (`oo cal-events`, `oo task-list`, `oo crm-contacts`,
|
||||
> `oo applications-sync`, …) were replaced by subject-based equivalents
|
||||
> (`oo calendar events`, `oo tasks list`, `oo contacts list`,
|
||||
> `oo applications sync`). Flags on leaf commands are unchanged.
|
||||
> flat names (`oo cal-events`, `oo task-list`, `oo crm-contacts`, …) were
|
||||
> replaced by subject-based equivalents (`oo calendar events`, `oo tasks list`,
|
||||
> `oo contacts list`). Flags on leaf commands are unchanged.
|
||||
|
||||
## oo CLI use cases
|
||||
|
||||
@@ -672,7 +801,7 @@ Every list command accepts `-o table` (default) or `-o json` for scripting.
|
||||
### CRM cleanup after imports or sync drift
|
||||
|
||||
**Problem:** Duplicate companies (`Acme` / `ACME GmbH`), persons created twice,
|
||||
the same email on a contact three times, deals titled ` @ contoso`, or the same
|
||||
the same email on a contact three times, deals titled ` @ Acme`, or the same
|
||||
HR contact linked to a deal twice.
|
||||
|
||||
**One-shot fix** — runs every dedupe pass in order:
|
||||
@@ -711,13 +840,39 @@ oo contacts dedupe-info
|
||||
# Two deals with the same title
|
||||
oo opportunities dedupe
|
||||
|
||||
# Same contact attached twice to one deal (common after applications sync)
|
||||
# Same contact attached twice to one deal
|
||||
oo opportunities dedupe-members
|
||||
|
||||
# Titles like " @ contoso" or extra whitespace
|
||||
# Titles like " @ Acme" or extra whitespace
|
||||
oo opportunities fix-titles
|
||||
```
|
||||
|
||||
Merge two known company ids (keeps `INTO`):
|
||||
|
||||
```bash
|
||||
oo contacts merge FROM_ID INTO_ID
|
||||
```
|
||||
|
||||
Company ↔ person ↔ deal ↔ project ↔ invoice ↔ mail rules and OO quirks:
|
||||
[docs/crm-associations.md](docs/crm-associations.md).
|
||||
|
||||
### Invoices (`oo invoices`)
|
||||
|
||||
**Problem:** Bill a client deal as Draft, regenerate PDF, prune duplicate PDF
|
||||
attachments, prepare OnlyOffice Mail — without inventing a second bill-to company.
|
||||
|
||||
```bash
|
||||
# Always pass --opportunity at create (update --opportunity often HTTP 400)
|
||||
oo invoices create --number INV-2026-01 --contact COMPANY_ID --item ITEM_ID \
|
||||
--price 300 --opportunity DEAL_ID --language de-DE \
|
||||
--line-description "…" --terms $'…' --description $'…' --po "Deal #DEAL_ID"
|
||||
|
||||
oo invoices pdf INVOICE_ID --force
|
||||
oo invoices pdf-cleanup INVOICE_ID
|
||||
oo invoices status INVOICE_ID --status draft
|
||||
oo mails draft-invoice --invoice INVOICE_ID --to billing@example.com
|
||||
```
|
||||
|
||||
**Deal grouping flag** — when the same role at the same company created
|
||||
separate deals (`Engineer @ Acme` vs `Engineer`):
|
||||
|
||||
@@ -729,66 +884,11 @@ oo crm cleanup --ignore-company-suffix
|
||||
**Inspect before/after:**
|
||||
|
||||
```bash
|
||||
oo opportunities list --count 200 | grep -i contoso
|
||||
oo contacts get 857 -o json
|
||||
oo opportunities list --count 200
|
||||
oo contacts get CONTACT_ID -o json
|
||||
oo crm cleanup -o json
|
||||
```
|
||||
|
||||
### Clients/contacts inventory (`catalog`)
|
||||
|
||||
**Problem:** Contacts live as VCF/folders on disk and project trees under
|
||||
`~/work` / ops-host; OnlyOffice CRM is the SSOT for persons/companies but
|
||||
starts sparse. You need a reviewable inventory before writes.
|
||||
|
||||
```bash
|
||||
oo catalog scan-contacts --root /path/to/contacts -O /tmp/contacts.yaml
|
||||
oo catalog scan-projects --root ~/work -O /tmp/projects.yaml
|
||||
oo catalog scan-thunderbird --root /path/to/thunderbird-profile -O /tmp/thunderbird.yaml
|
||||
oo catalog merge -i /tmp/contacts.yaml -i /tmp/projects.yaml -i /tmp/thunderbird.yaml -O clients-contacts.yaml
|
||||
oo catalog match -i clients-contacts.yaml
|
||||
# set approve: true on pilot rows in the YAML
|
||||
oo catalog apply --dry-run -i clients-contacts.yaml
|
||||
oo catalog apply --apply -i clients-contacts.yaml
|
||||
oo crm cleanup
|
||||
```
|
||||
|
||||
YAML schema and ops runbook live in the inventar repo under `docs/catalog/` and
|
||||
`docs/ops/oo-clients-contacts-sync.md`. Library helper: `FindPersonByEmail`.
|
||||
|
||||
### Job applications → CRM (`applications sync`)
|
||||
|
||||
**Problem:** You keep CVs in a folder tree (`applications/2026/Acme/README.md`)
|
||||
and want companies, persons, deals, and history notes in OnlyOffice without
|
||||
re-typing.
|
||||
|
||||
**Dry-run first** (default — prints what would happen, writes nothing):
|
||||
|
||||
```bash
|
||||
oo applications sync --path ./applications/2026 --verbose
|
||||
```
|
||||
|
||||
**Apply** when the preview looks right:
|
||||
|
||||
```bash
|
||||
oo applications sync --path ./applications/2026 --apply --verbose
|
||||
```
|
||||
|
||||
Each dossier `README.md` is parsed for company, role, email, phone, LinkedIn, etc.
|
||||
Discovery walks `--path` and **skips** junk trees (`node_modules`, `tools`, `.venv`,
|
||||
`pdfs`, …). Only folders that look like CV application slugs (`source-NNN-…`, or
|
||||
long hyphenated dossiers) are synced — not npm package READMEs.
|
||||
|
||||
The sync creates or finds contacts, opens a deal, adds members, and appends a
|
||||
history note. Re-running is safe: duplicate members and duplicate deal titles
|
||||
are skipped when already present.
|
||||
|
||||
**After a large sync**, run CRM cleanup to collapse duplicates introduced by
|
||||
repeated runs or manual edits:
|
||||
|
||||
```bash
|
||||
oo applications sync --path ./applications/2026 --apply
|
||||
oo crm cleanup -o json
|
||||
```
|
||||
|
||||
### Workspace mail (`oo mails`)
|
||||
|
||||
@@ -821,8 +921,21 @@ oo mails get 5664 -o json | jq '{subject, from, to, date}'
|
||||
# Remove one or more messages (server moves to trash or deletes per Mail rules)
|
||||
oo mails delete 5664
|
||||
oo mails delete 5664 5663 5661
|
||||
|
||||
# Create / update a draft (HTML body field is API "body"; plain text is wrapped)
|
||||
oo mails draft --to client@example.com --subject "Rechnung INV-2026-01" \
|
||||
--body "Guten Tag,\n\nanbei die Rechnung.\n\nMit freundlichen Grüßen"
|
||||
|
||||
# Attach an OnlyOffice Files document (e.g. invoice PDF file id) to a draft
|
||||
oo mails attach 7301 --file-id 12345
|
||||
|
||||
# Regenerate invoice PDF (force) + draft + attach (does not send)
|
||||
oo mails draft-invoice --invoice 16 --to info@example.com
|
||||
```
|
||||
|
||||
Matrix / chat URLs in signatures: use plain text
|
||||
`chat: https://matrix.to/#/@user:server` — HTML `<a href="…#…">` truncates at `#`.
|
||||
|
||||
**Table output** splits the `from` header into `fromName` and `fromAddress`
|
||||
(e.g. `Bitfinex` + `no-reply@bitfinex.com`). **JSON output** returns the raw
|
||||
API payload.
|
||||
@@ -862,8 +975,8 @@ oo calendar events --start 2026-06-24 --end 2026-07-01
|
||||
# Schedule interview block
|
||||
oo calendar add "Technical interview" 2026-06-26T10:00:00Z 2026-06-26T11:00:00Z
|
||||
|
||||
# Attach CV to a hiring task
|
||||
oo tasks files upload 208 ./cv.pdf
|
||||
# Attach a file to a task
|
||||
oo tasks files upload 208 ./notes.pdf
|
||||
oo projects files list 33
|
||||
```
|
||||
|
||||
@@ -871,7 +984,7 @@ oo projects files list 33
|
||||
|
||||
| When | Command |
|
||||
|------|---------|
|
||||
| After `applications sync --apply` | `oo crm cleanup` |
|
||||
| After bulk CRM imports | `oo crm cleanup` |
|
||||
| After bulk CSV import into CRM | `oo crm cleanup` |
|
||||
| Weekly inbox triage | `oo mails list --folder inbox --limit 100` |
|
||||
| Before exec reporting | `oo opportunities list` + `oo projects list` |
|
||||
@@ -885,9 +998,51 @@ oo projects files list 33
|
||||
| `ONLYOFFICE_PASS` (or `ONLYOFFICE_PASSWORD`) | Password |
|
||||
| `ONLYOFFICE_CALENDAR_ID` | Default calendar id used when omitted (default `1`) |
|
||||
| `ONLYOFFICE_PROJECT_ID` | Default project id used when omitted (default `33`) |
|
||||
| `OO_URL`, `OO_USER`, `OO_PASS` | CLI-only produktor.io aliases mapped to `ONLYOFFICE_URL`, `ONLYOFFICE_USER`, `ONLYOFFICE_PASS` |
|
||||
| `OO_URL`, `OO_USER`, `OO_PASS` | Optional CLI-only aliases for `ONLYOFFICE_*` |
|
||||
| `ONLYOFFICE_ES_URL` | Elasticsearch base URL (`oo search`, own index); see [`docs/elasticsearch.md`](docs/elasticsearch.md) |
|
||||
| `ONLYOFFICE_ES_INDEX` | OnlyOffice index (default `files_file`) |
|
||||
| `ONLYOFFICE_ES_TEXT_INDEX` | Own PDF/scan index (default `oo_docs_text`) |
|
||||
| `ONLYOFFICE_TENANT` | `tenantId` filter for ES/SQL (empty = all) |
|
||||
| `ONLYOFFICE_DSN` | Read-only SQL DSN (MySQL or `postgres://`); see [`docs/community-server-db.md`](docs/community-server-db.md) |
|
||||
| `ONLYOFFICE_PG_DRIVER`, `ONLYOFFICE_PG_TENANT`, `ONLYOFFICE_PG_HOST/_PORT/_USER/_PASSWORD/_DBNAME/_SSLMODE` | SQL store override / DSN by parts (PostgreSQL) |
|
||||
| `MINIO_ENDPOINT`, `MINIO_BUCKET`, `MINIO_ACCESS_KEY`, `MINIO_SECRET_KEY` | Object-store layout for SQL `Download` |
|
||||
| `ONLYOFFICE_WEBDAV_URL` | rclone WebDAV sidecar URL (default `http://172.17.0.1:8098/webdav`) |
|
||||
|
||||
Mail, CRM cleanup, and applications sync are documented in [oo CLI use cases](#oo-cli-use-cases) above.
|
||||
Mail and CRM cleanup are documented in [oo CLI use cases](#oo-cli-use-cases) above. Personal disk inventory / dossier sync lives in the private `oo-workspace` (`oow`) tooling.
|
||||
|
||||
## Testing
|
||||
|
||||
```bash
|
||||
go build ./... && go vet ./...
|
||||
go test ./... # unit — no network, no vendor mocks
|
||||
go test -race ./...
|
||||
go test -tags=integration ./... # live OnlyOffice (skip without creds)
|
||||
```
|
||||
|
||||
Unit tests are pure Go (parsers, encoders, conversions). Integration tests
|
||||
(`//go:build integration`) hit a live instance and **skip** cleanly when the
|
||||
env is missing, so `go test ./...` stays green offline. New endpoints ship with
|
||||
an integration test before merge (policy in [`AGENTS.md`](AGENTS.md)).
|
||||
|
||||
Live runs need credentials (`ONLYOFFICE_URL`, `ONLYOFFICE_USER`,
|
||||
`ONLYOFFICE_PASS`) and, per backend:
|
||||
|
||||
- **Elasticsearch** (`oo search`, own index) — ES lives on `127.0.0.1:9200`
|
||||
inside the OnlyOffice VM; expose it over SSH
|
||||
(`-L 9200:127.0.0.1:9200`) and set `ONLYOFFICE_ES_URL`
|
||||
(see [`docs/elasticsearch.md`](docs/elasticsearch.md)).
|
||||
- **SQL backend** (`FileClient`, `SQLFileStore`) — MySQL on `127.0.0.1:3306`
|
||||
in the same VM; tunnel `-L 3306:127.0.0.1:3306`, then set `ONLYOFFICE_DSN`
|
||||
(see [`docs/community-server-db.md`](docs/community-server-db.md)).
|
||||
`ONLYOFFICE_PG_TEST_FILE_ID` / `ONLYOFFICE_PG_TEST_FOLDER_ID` select a real
|
||||
file for the REST cross-check; `MINIO_*` enable the download check.
|
||||
|
||||
### rclone WebDAV mount
|
||||
|
||||
`deploy/docker-compose.rclone-webdav.yml` mounts the Documents tree as a
|
||||
filesystem (compose, not systemd; container `rclone-webdav`) — read/write like
|
||||
a normal FS over the `oo-webdav` sidecar. Setup, smoke log and limitations:
|
||||
[`docs/rclone-webdav.md`](docs/rclone-webdav.md).
|
||||
|
||||
### CI / releases
|
||||
|
||||
|
||||
@@ -43,6 +43,11 @@ func (c *Client) Authenticate() error { return c.ensureToken() }
|
||||
// cached token is still valid it returns immediately; otherwise it performs
|
||||
// a POST to /api/2.0/authentication.json that is cancellable via ctx.
|
||||
//
|
||||
// Transient answers from the edge (openresty 429/502/503/504) are retried with
|
||||
// the same deterministic policy as every other request (see retry.go), because
|
||||
// the server rate-limits authentication and the integration suite otherwise
|
||||
// fails with a raw HTML 429 page.
|
||||
//
|
||||
// This is the recommended entry point for long-running syncs (cron,
|
||||
// watchers) because it guarantees that a stalled auth call will not block
|
||||
// the caller past its deadline.
|
||||
@@ -50,6 +55,14 @@ func (c *Client) AuthenticateContext(ctx context.Context) error {
|
||||
if c.tokenValid() {
|
||||
return nil
|
||||
}
|
||||
return DoRetry(ctx, DefaultRetryPolicy(), func() error {
|
||||
return c.authenticateOnce(ctx)
|
||||
})
|
||||
}
|
||||
|
||||
// authenticateOnce performs a single authentication POST. Callers must handle
|
||||
// retries; use AuthenticateContext.
|
||||
func (c *Client) authenticateOnce(ctx context.Context) error {
|
||||
body, err := json.Marshal(c.credentials)
|
||||
if err != nil {
|
||||
return fmt.Errorf("marshal credentials: %w", err)
|
||||
@@ -99,17 +112,13 @@ func (c *Client) tokenValid() bool {
|
||||
|
||||
// ensureToken refreshes the authentication token when missing or expired.
|
||||
// Mirrors the logic inline in Query() but is safe to call from helpers that
|
||||
// bypass the typed Request abstraction.
|
||||
// bypass the typed Request abstraction. It shares AuthenticateContext so the
|
||||
// transient-retry policy applies to every code path.
|
||||
func (c *Client) ensureToken() error {
|
||||
if c.tokenValid() {
|
||||
return nil
|
||||
}
|
||||
tok, err := c.Auth(c.credentials)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
c.token = tok
|
||||
return nil
|
||||
return c.AuthenticateContext(context.Background())
|
||||
}
|
||||
|
||||
// authHeader returns the value for the Authorization header, ensuring a token.
|
||||
|
||||
@@ -0,0 +1,199 @@
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
)
|
||||
|
||||
// ApplyResult summarizes one apply pass.
|
||||
type ApplyResult struct {
|
||||
Created int `json:"created"`
|
||||
Updated int `json:"updated"`
|
||||
Skipped int `json:"skipped"`
|
||||
Errors []string `json:"errors,omitempty"`
|
||||
DryRun bool `json:"dry_run"`
|
||||
}
|
||||
|
||||
// ApplyApproved creates/updates OO contacts for entries with approve=true.
|
||||
// Companies are applied before persons so Org links resolve in the same run.
|
||||
func ApplyApproved(ctx context.Context, client *onlyoffice.Client, doc *Document, dryRun bool) (*ApplyResult, error) {
|
||||
res := &ApplyResult{DryRun: dryRun}
|
||||
order := make([]int, 0, len(doc.Entries))
|
||||
for i, e := range doc.Entries {
|
||||
if e.Kind == "company" {
|
||||
order = append(order, i)
|
||||
}
|
||||
}
|
||||
for i, e := range doc.Entries {
|
||||
if e.Kind != "company" {
|
||||
order = append(order, i)
|
||||
}
|
||||
}
|
||||
for _, i := range order {
|
||||
e := &doc.Entries[i]
|
||||
if !e.Approve {
|
||||
res.Skipped++
|
||||
continue
|
||||
}
|
||||
if e.Status == "conflict" {
|
||||
res.Skipped++
|
||||
res.Errors = append(res.Errors, e.ID+": conflict — resolve manually")
|
||||
continue
|
||||
}
|
||||
if dryRun {
|
||||
if e.OOID != "" || e.Status == "exists" {
|
||||
res.Updated++
|
||||
} else {
|
||||
res.Created++
|
||||
}
|
||||
continue
|
||||
}
|
||||
created, err := applyOne(ctx, client, e)
|
||||
if err != nil {
|
||||
res.Errors = append(res.Errors, e.ID+": "+err.Error())
|
||||
continue
|
||||
}
|
||||
if created {
|
||||
res.Created++
|
||||
} else {
|
||||
res.Updated++
|
||||
}
|
||||
}
|
||||
return res, nil
|
||||
}
|
||||
|
||||
func applyOne(ctx context.Context, client *onlyoffice.Client, e *Entry) (created bool, err error) {
|
||||
switch e.Kind {
|
||||
case "company":
|
||||
return applyCompany(ctx, client, e)
|
||||
case "person":
|
||||
return applyPerson(ctx, client, e)
|
||||
default:
|
||||
return false, fmt.Errorf("unknown kind %q", e.Kind)
|
||||
}
|
||||
}
|
||||
|
||||
func applyCompany(ctx context.Context, client *onlyoffice.Client, e *Entry) (bool, error) {
|
||||
name := strings.TrimSpace(e.Name)
|
||||
if name == "" {
|
||||
return false, fmt.Errorf("company missing name")
|
||||
}
|
||||
var co map[string]any
|
||||
var err error
|
||||
if e.OOID != "" {
|
||||
co, err = client.GetContact(ctx, e.OOID)
|
||||
} else {
|
||||
co, err = client.FindCompany(ctx, name)
|
||||
}
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
created := false
|
||||
if co == nil {
|
||||
co, err = client.CreateCompany(ctx, name)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
created = true
|
||||
}
|
||||
id := contactIDString(co)
|
||||
e.OOID = id
|
||||
if err := ensureContactInfos(ctx, client, id, e); err != nil {
|
||||
return created, err
|
||||
}
|
||||
e.Status = "applied"
|
||||
if created {
|
||||
e.Notes = "created"
|
||||
} else {
|
||||
e.Notes = "updated"
|
||||
}
|
||||
return created, nil
|
||||
}
|
||||
|
||||
func applyPerson(ctx context.Context, client *onlyoffice.Client, e *Entry) (bool, error) {
|
||||
first := strings.TrimSpace(e.First)
|
||||
last := strings.TrimSpace(e.Last)
|
||||
if first == "" && last == "" {
|
||||
first, last = SplitDisplayName(e.Name)
|
||||
}
|
||||
if first == "" {
|
||||
first = strings.TrimSpace(e.Name)
|
||||
}
|
||||
if first == "" {
|
||||
return false, fmt.Errorf("person missing name")
|
||||
}
|
||||
if last == "" {
|
||||
last = "-"
|
||||
}
|
||||
|
||||
var p map[string]any
|
||||
var err error
|
||||
if e.OOID != "" {
|
||||
p, err = client.GetContact(ctx, e.OOID)
|
||||
} else if len(e.Emails) > 0 {
|
||||
p, err = client.FindPersonByEmail(ctx, e.Emails[0])
|
||||
} else {
|
||||
p, err = client.FindPerson(ctx, first, last)
|
||||
}
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
created := false
|
||||
companyID := 0
|
||||
if e.Org != "" {
|
||||
if co, ferr := client.FindCompany(ctx, e.Org); ferr == nil && co != nil {
|
||||
companyID, _ = strconv.Atoi(contactIDString(co))
|
||||
}
|
||||
}
|
||||
if p == nil {
|
||||
about := ""
|
||||
if e.Org != "" {
|
||||
about = "org: " + e.Org
|
||||
}
|
||||
p, err = client.CreatePerson(ctx, first, last, companyID, "", about)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
created = true
|
||||
}
|
||||
id := contactIDString(p)
|
||||
e.OOID = id
|
||||
if err := ensureContactInfos(ctx, client, id, e); err != nil {
|
||||
return created, err
|
||||
}
|
||||
e.Status = "applied"
|
||||
if created {
|
||||
e.Notes = "created"
|
||||
} else {
|
||||
e.Notes = "updated"
|
||||
}
|
||||
return created, nil
|
||||
}
|
||||
|
||||
func ensureContactInfos(ctx context.Context, client *onlyoffice.Client, contactID string, e *Entry) error {
|
||||
existing, err := client.GetContact(ctx, contactID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for i, em := range e.Emails {
|
||||
if onlyoffice.HasContactInfo(existing, "Email", em) {
|
||||
continue
|
||||
}
|
||||
if _, err := client.AddContactInfo(ctx, contactID, "Email", em, "Work", i == 0); err != nil {
|
||||
return fmt.Errorf("add email %s: %w", em, err)
|
||||
}
|
||||
}
|
||||
for _, ph := range e.Phones {
|
||||
if onlyoffice.HasContactInfo(existing, "Phone", ph) {
|
||||
continue
|
||||
}
|
||||
if _, err := client.AddContactInfo(ctx, contactID, "Phone", ph, "Work", false); err != nil {
|
||||
return fmt.Errorf("add phone %s: %w", ph, err)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,132 @@
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestParseVCFFile(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "sample.vcf")
|
||||
body := `BEGIN:VCARD
|
||||
VERSION:3.0
|
||||
FN:Ada Lovelace
|
||||
N:Lovelace;Ada;;;
|
||||
EMAIL;TYPE=INTERNET:ada@example.com
|
||||
TEL;TYPE=CELL:+1-555-0100
|
||||
ORG:Analytical Engines
|
||||
END:VCARD
|
||||
`
|
||||
if err := os.WriteFile(path, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
entries, err := ParseVCFFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(entries) != 1 {
|
||||
t.Fatalf("got %d entries", len(entries))
|
||||
}
|
||||
e := entries[0]
|
||||
if e.Kind != "person" || e.First != "Ada" || e.Last != "Lovelace" {
|
||||
t.Fatalf("unexpected entry: %+v", e)
|
||||
}
|
||||
if len(e.Emails) != 1 || e.Emails[0] != "ada@example.com" {
|
||||
t.Fatalf("emails: %v", e.Emails)
|
||||
}
|
||||
if e.Org != "Analytical Engines" {
|
||||
t.Fatalf("org: %q", e.Org)
|
||||
}
|
||||
if e.Approve {
|
||||
t.Fatal("approve should default false")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMergeDocs(t *testing.T) {
|
||||
a := &Document{Entries: []Entry{{
|
||||
ID: "person:ada@example.com", Kind: "person", First: "Ada", Emails: []string{"ada@example.com"},
|
||||
Sources: []string{"a.vcf"}, Zone: "private", Role: "unknown",
|
||||
}}}
|
||||
b := &Document{Entries: []Entry{{
|
||||
ID: "person:ada@example.com", Kind: "person", Last: "Lovelace", Phones: []string{"+1"},
|
||||
Sources: []string{"folder/Ada"}, Zone: "warm", Role: "work", Approve: true,
|
||||
}}}
|
||||
m := MergeDocs(a, b)
|
||||
if len(m.Entries) != 1 {
|
||||
t.Fatalf("len=%d", len(m.Entries))
|
||||
}
|
||||
e := m.Entries[0]
|
||||
if e.First != "Ada" || e.Last != "Lovelace" {
|
||||
t.Fatalf("%+v", e)
|
||||
}
|
||||
if len(e.Sources) != 2 || len(e.Phones) != 1 || !e.Approve {
|
||||
t.Fatalf("%+v", e)
|
||||
}
|
||||
if e.Zone != "warm" || e.Role != "work" {
|
||||
t.Fatalf("zone/role: %s/%s", e.Zone, e.Role)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScanContactsRoot(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
vcfDir := filepath.Join(root, "Contacts VCF's")
|
||||
if err := os.MkdirAll(vcfDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(vcfDir, "bob.vcf"), []byte(`BEGIN:VCARD
|
||||
VERSION:3.0
|
||||
FN:Bob Builder
|
||||
EMAIL:bob@work.test
|
||||
END:VCARD
|
||||
`), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
personDir := filepath.Join(root, "Carol Smith")
|
||||
if err := os.MkdirAll(personDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(personDir, "carol@example.com.txt"), []byte(""), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
doc, err := ScanContactsRoot(root)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(doc.Entries) < 2 {
|
||||
t.Fatalf("expected >=2 entries, got %d: %+v", len(doc.Entries), doc.Entries)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScanProjectsRoot(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
repo := filepath.Join(root, "produktor-demo")
|
||||
if err := os.MkdirAll(filepath.Join(repo, ".git"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
doc, err := ScanProjectsRoot(root, 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
found := false
|
||||
for _, e := range doc.Entries {
|
||||
if e.Kind == "company" && e.Name == "produktor-demo" {
|
||||
found = true
|
||||
if e.Role != "work" {
|
||||
t.Fatalf("role=%q", e.Role)
|
||||
}
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Fatalf("missing company: %+v", doc.Entries)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEntryID(t *testing.T) {
|
||||
if got := EntryID("person", "A@B.COM", ""); got != "person:a@b.com" {
|
||||
t.Fatal(got)
|
||||
}
|
||||
if got := EntryID("company", "", "Acme Corp"); got != "company:acme corp" {
|
||||
t.Fatal(got)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
)
|
||||
|
||||
// MatchAgainstOO sets status/oo_id by email then normalized name.
|
||||
func MatchAgainstOO(ctx context.Context, client *onlyoffice.Client, doc *Document) error {
|
||||
all, err := client.ListAllContacts(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
byEmail := map[string]map[string]any{}
|
||||
byPersonName := map[string]map[string]any{}
|
||||
byCompanyName := map[string]map[string]any{}
|
||||
|
||||
for _, c := range all {
|
||||
id := contactIDString(c)
|
||||
isCo, _ := c["isCompany"].(bool)
|
||||
for _, em := range contactEmails(c) {
|
||||
byEmail[NormalizeEmail(em)] = c
|
||||
}
|
||||
if isCo {
|
||||
key := NormalizeName(fmt.Sprint(c["displayName"]))
|
||||
if key != "" {
|
||||
byCompanyName[key] = c
|
||||
}
|
||||
continue
|
||||
}
|
||||
first := strings.TrimSpace(fmt.Sprint(c["firstName"]))
|
||||
last := strings.TrimSpace(fmt.Sprint(c["lastName"]))
|
||||
key := NormalizeName(first + " " + last)
|
||||
if key == "" {
|
||||
key = NormalizeName(fmt.Sprint(c["displayName"]))
|
||||
}
|
||||
if key != "" {
|
||||
byPersonName[key] = c
|
||||
}
|
||||
_ = id
|
||||
}
|
||||
|
||||
for i := range doc.Entries {
|
||||
e := &doc.Entries[i]
|
||||
// preserve approve
|
||||
matched := false
|
||||
conflict := false
|
||||
var oo map[string]any
|
||||
|
||||
for _, em := range e.Emails {
|
||||
if c, ok := byEmail[NormalizeEmail(em)]; ok {
|
||||
if oo != nil && contactIDString(oo) != contactIDString(c) {
|
||||
conflict = true
|
||||
}
|
||||
oo = c
|
||||
matched = true
|
||||
}
|
||||
}
|
||||
if !matched {
|
||||
if e.Kind == "company" {
|
||||
if c, ok := byCompanyName[NormalizeName(e.Name)]; ok {
|
||||
oo = c
|
||||
matched = true
|
||||
}
|
||||
} else {
|
||||
key := NormalizeName(strings.TrimSpace(e.First + " " + e.Last))
|
||||
if key == "" {
|
||||
key = NormalizeName(e.Name)
|
||||
}
|
||||
if c, ok := byPersonName[key]; ok {
|
||||
oo = c
|
||||
matched = true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if conflict {
|
||||
e.Status = "conflict"
|
||||
e.Notes = strings.TrimSpace(e.Notes + " email_matches_multiple_oo")
|
||||
if oo != nil {
|
||||
e.OOID = contactIDString(oo)
|
||||
}
|
||||
continue
|
||||
}
|
||||
if matched && oo != nil {
|
||||
e.Status = "exists"
|
||||
e.OOID = contactIDString(oo)
|
||||
continue
|
||||
}
|
||||
e.Status = "new"
|
||||
e.OOID = ""
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func contactIDString(c map[string]any) string {
|
||||
switch v := c["id"].(type) {
|
||||
case float64:
|
||||
return strconv.FormatInt(int64(v), 10)
|
||||
case int:
|
||||
return strconv.Itoa(v)
|
||||
case string:
|
||||
return v
|
||||
default:
|
||||
return fmt.Sprint(v)
|
||||
}
|
||||
}
|
||||
|
||||
func contactEmails(c map[string]any) []string {
|
||||
var out []string
|
||||
if em := strings.TrimSpace(fmt.Sprint(c["email"])); em != "" && em != "<nil>" {
|
||||
out = append(out, em)
|
||||
}
|
||||
if em := strings.TrimSpace(fmt.Sprint(c["primaryEmail"])); em != "" && em != "<nil>" {
|
||||
out = append(out, em)
|
||||
}
|
||||
for _, row := range onlyoffice.ContactInfoRows(c) {
|
||||
t := strings.ToLower(fmt.Sprint(row["infoType"]))
|
||||
if t != "email" {
|
||||
continue
|
||||
}
|
||||
data := strings.TrimSpace(fmt.Sprint(row["data"]))
|
||||
if data != "" {
|
||||
out = append(out, data)
|
||||
}
|
||||
}
|
||||
return uniqueEmails(out)
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestMatchAgainstOO_emailThenName(t *testing.T) {
|
||||
// MatchAgainstOO needs a live client; unit-test the helpers used by matching.
|
||||
c := map[string]any{
|
||||
"id": float64(42),
|
||||
"isCompany": false,
|
||||
"firstName": "Ada",
|
||||
"lastName": "Lovelace",
|
||||
"email": "ada@example.com",
|
||||
"commonData": []any{
|
||||
map[string]any{"infoType": "Email", "data": "ada.alt@example.com"},
|
||||
},
|
||||
}
|
||||
if contactIDString(c) != "42" {
|
||||
t.Fatal(contactIDString(c))
|
||||
}
|
||||
emails := contactEmails(c)
|
||||
if len(emails) < 2 {
|
||||
t.Fatalf("%v", emails)
|
||||
}
|
||||
}
|
||||
+238
@@ -0,0 +1,238 @@
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ScanOptions controls Thunderbird / mbox extraction depth.
|
||||
type ScanOptions struct {
|
||||
// MboxHeaders, when true, also walks mbox-like files under the root and
|
||||
// extracts From/To/Cc/Reply-To addresses (headers only, no bodies).
|
||||
MboxHeaders bool
|
||||
// MboxMaxBytes skips individual mbox files larger than this (0 = 256 MiB default).
|
||||
MboxMaxBytes int64
|
||||
}
|
||||
|
||||
// ScanThunderbirdRoot finds Thunderbird profiles under root and emits person rows
|
||||
// from address books (*.mab) and Gloda global-messages-db.sqlite.
|
||||
func ScanThunderbirdRoot(root string) (*Document, error) {
|
||||
return ScanThunderbirdRootOpts(root, ScanOptions{})
|
||||
}
|
||||
|
||||
// ScanThunderbirdRootOpts is ScanThunderbirdRoot with optional mbox header pass.
|
||||
func ScanThunderbirdRootOpts(root string, opts ScanOptions) (*Document, error) {
|
||||
root = filepath.Clean(root)
|
||||
st, err := os.Stat(root)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !st.IsDir() {
|
||||
return nil, fmt.Errorf("not a directory: %s", root)
|
||||
}
|
||||
if opts.MboxMaxBytes <= 0 {
|
||||
opts.MboxMaxBytes = 256 << 20
|
||||
}
|
||||
|
||||
var entries []Entry
|
||||
seenDB := map[string]struct{}{}
|
||||
seenMAB := map[string]struct{}{}
|
||||
seenMbox := map[string]struct{}{}
|
||||
|
||||
err = filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error {
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
name := d.Name()
|
||||
if d.IsDir() {
|
||||
switch name {
|
||||
case "Cache", "cache2", "startupCache", "OfflineCache", "minidumps",
|
||||
"crashes", "safebrowsing", "thumbnails", "chrome", "extensions",
|
||||
"node_modules", ".git":
|
||||
return filepath.SkipDir
|
||||
}
|
||||
return nil
|
||||
}
|
||||
lower := strings.ToLower(name)
|
||||
switch {
|
||||
case lower == "global-messages-db.sqlite":
|
||||
if _, ok := seenDB[path]; ok {
|
||||
return nil
|
||||
}
|
||||
seenDB[path] = struct{}{}
|
||||
parsed, perr := parseGlodaContacts(path)
|
||||
if perr != nil {
|
||||
entries = append(entries, Entry{
|
||||
ID: EntryID("person", "", filepath.Base(path)),
|
||||
Kind: "person",
|
||||
Name: "gloda",
|
||||
Sources: []string{path},
|
||||
Zone: "private",
|
||||
Role: "unknown",
|
||||
Notes: "gloda_parse_error: " + perr.Error(),
|
||||
Status: "new",
|
||||
})
|
||||
return nil
|
||||
}
|
||||
entries = append(entries, parsed...)
|
||||
case strings.HasSuffix(lower, ".mab"):
|
||||
if _, ok := seenMAB[path]; ok {
|
||||
return nil
|
||||
}
|
||||
seenMAB[path] = struct{}{}
|
||||
parsed, perr := parseMABEmails(path)
|
||||
if perr != nil {
|
||||
return nil
|
||||
}
|
||||
entries = append(entries, parsed...)
|
||||
case opts.MboxHeaders && isLikelyMboxFile(name, path):
|
||||
if _, ok := seenMbox[path]; ok {
|
||||
return nil
|
||||
}
|
||||
seenMbox[path] = struct{}{}
|
||||
info, ierr := d.Info()
|
||||
if ierr != nil {
|
||||
return nil
|
||||
}
|
||||
if info.Size() > opts.MboxMaxBytes {
|
||||
return nil
|
||||
}
|
||||
parsed, perr := parseMboxHeaderEmails(path)
|
||||
if perr != nil {
|
||||
return nil
|
||||
}
|
||||
entries = append(entries, parsed...)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
return MergeDocs(&Document{Entries: entries}), nil
|
||||
}
|
||||
|
||||
func isLikelyMboxFile(name, path string) bool {
|
||||
lower := strings.ToLower(name)
|
||||
if strings.HasSuffix(lower, ".msf") || strings.HasSuffix(lower, ".sqlite") ||
|
||||
strings.HasSuffix(lower, ".mab") || strings.HasSuffix(lower, ".json") ||
|
||||
strings.HasSuffix(lower, ".dat") || strings.HasSuffix(lower, ".ini") ||
|
||||
strings.HasSuffix(lower, ".log") {
|
||||
return false
|
||||
}
|
||||
if strings.HasSuffix(lower, ".mbox") || strings.HasSuffix(lower, ".mbx") {
|
||||
return true
|
||||
}
|
||||
// Thunderbird: …/ImapMail/<server>/INBOX or …/Mail/Local Folders/Inbox
|
||||
// or …/INBOX.sbd/<folder>
|
||||
sep := string(filepath.Separator)
|
||||
norm := filepath.ToSlash(path)
|
||||
if strings.Contains(norm, "/ImapMail/") || strings.Contains(norm, "/Mail/") {
|
||||
if !strings.Contains(name, ".") {
|
||||
return true
|
||||
}
|
||||
known := map[string]struct{}{
|
||||
"inbox": {}, "sent": {}, "drafts": {}, "trash": {}, "archives": {}, "junk": {},
|
||||
}
|
||||
if _, ok := known[lower]; ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
if strings.Contains(filepath.Base(filepath.Dir(path)), ".sbd") && !strings.Contains(name, ".") {
|
||||
return true
|
||||
}
|
||||
_ = sep
|
||||
return false
|
||||
}
|
||||
|
||||
// parseMboxHeaderEmails extracts addresses from From/To/Cc/Reply-To headers only.
|
||||
func parseMboxHeaderEmails(path string) ([]Entry, error) {
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
emails := map[string]struct{}{}
|
||||
br := bufio.NewReaderSize(f, 1<<20)
|
||||
inHeaders := false
|
||||
var headerBuf strings.Builder
|
||||
|
||||
flushHeaders := func() {
|
||||
if headerBuf.Len() == 0 {
|
||||
return
|
||||
}
|
||||
block := headerBuf.String()
|
||||
headerBuf.Reset()
|
||||
for _, line := range strings.Split(block, "\n") {
|
||||
lower := strings.ToLower(strings.TrimSpace(line))
|
||||
if strings.HasPrefix(lower, "from:") || strings.HasPrefix(lower, "to:") ||
|
||||
strings.HasPrefix(lower, "cc:") || strings.HasPrefix(lower, "reply-to:") ||
|
||||
strings.HasPrefix(lower, "sender:") {
|
||||
for _, m := range emailRE.FindAllString(line, -1) {
|
||||
em := NormalizeEmail(m)
|
||||
if !noisyEmail(em) {
|
||||
emails[em] = struct{}{}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for {
|
||||
line, err := br.ReadString('\n')
|
||||
if len(line) > 0 {
|
||||
// Classic mbox separator
|
||||
if strings.HasPrefix(line, "From ") && (len(line) == 5 || line[5] != ':') {
|
||||
flushHeaders()
|
||||
inHeaders = true
|
||||
continue
|
||||
}
|
||||
if inHeaders {
|
||||
trimmed := strings.TrimRight(line, "\r\n")
|
||||
if trimmed == "" {
|
||||
flushHeaders()
|
||||
inHeaders = false
|
||||
continue
|
||||
}
|
||||
headerBuf.WriteString(trimmed)
|
||||
headerBuf.WriteByte('\n')
|
||||
// cap pathological headers
|
||||
if headerBuf.Len() > 64<<10 {
|
||||
flushHeaders()
|
||||
inHeaders = false
|
||||
}
|
||||
}
|
||||
}
|
||||
if err == io.EOF {
|
||||
flushHeaders()
|
||||
break
|
||||
}
|
||||
if err != nil {
|
||||
flushHeaders()
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
|
||||
var out []Entry
|
||||
for em := range emails {
|
||||
org, zone, role := classifyMailIdentity("", em)
|
||||
out = append(out, Entry{
|
||||
ID: EntryID("person", em, ""),
|
||||
Kind: "person",
|
||||
Emails: []string{em},
|
||||
Org: org,
|
||||
Sources: []string{path},
|
||||
Zone: zone,
|
||||
Role: role,
|
||||
Approve: false,
|
||||
Status: "new",
|
||||
Notes: "thunderbird_mbox_header",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestParseMboxHeaderEmails(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "INBOX")
|
||||
body := `From - Mon Jul 1 00:00:00 2016
|
||||
From: Axel Schaefer <axel.schaefer@wheregroup.com>
|
||||
To: Andriy Oblivantsev <andriy.oblivantsev@wheregroup.com>
|
||||
Cc: noreply@example.com, client@stadt-example.de
|
||||
Subject: test
|
||||
|
||||
Body line ignored
|
||||
From - Mon Jul 2 00:00:00 2016
|
||||
From: Someone <paul.schmidt@wheregroup.com>
|
||||
To: list@wheregroup.com
|
||||
|
||||
more body
|
||||
`
|
||||
if err := os.WriteFile(path, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ents, err := parseMboxHeaderEmails(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got := map[string]bool{}
|
||||
for _, e := range ents {
|
||||
if len(e.Emails) > 0 {
|
||||
got[e.Emails[0]] = true
|
||||
}
|
||||
}
|
||||
if !got["axel.schaefer@wheregroup.com"] || !got["andriy.oblivantsev@wheregroup.com"] {
|
||||
t.Fatalf("%v", got)
|
||||
}
|
||||
if got["noreply@example.com"] {
|
||||
t.Fatal("noreply should be filtered")
|
||||
}
|
||||
if !got["client@stadt-example.de"] {
|
||||
t.Fatal("missing client")
|
||||
}
|
||||
}
|
||||
|
||||
func TestScanThunderbirdRootOptsMbox(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
imap := filepath.Join(root, "ImapMail", "mail.example.com")
|
||||
if err := os.MkdirAll(imap, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(imap, "INBOX"), []byte(
|
||||
"From - x\nFrom: a@wheregroup.com\nTo: b@wheregroup.com\n\nbody\n",
|
||||
), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
doc, err := ScanThunderbirdRootOpts(root, ScanOptions{MboxHeaders: true})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(doc.Entries) < 2 {
|
||||
t.Fatalf("%+v", doc.Entries)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ScanProjectsRoot finds git roots under root (max depth) and emits company rows.
|
||||
func ScanProjectsRoot(root string, maxDepth int) (*Document, error) {
|
||||
root = filepath.Clean(root)
|
||||
if maxDepth <= 0 {
|
||||
maxDepth = 4
|
||||
}
|
||||
st, err := os.Stat(root)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !st.IsDir() {
|
||||
return nil, fmt.Errorf("not a directory: %s", root)
|
||||
}
|
||||
|
||||
var entries []Entry
|
||||
err = walkGitRoots(root, root, 0, maxDepth, &entries)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// Also top-level dirs as company stubs (even without git) — coverage 1B.
|
||||
dents, _ := os.ReadDir(root)
|
||||
for _, d := range dents {
|
||||
if !d.IsDir() || strings.HasPrefix(d.Name(), ".") {
|
||||
continue
|
||||
}
|
||||
name := d.Name()
|
||||
skip := map[string]struct{}{
|
||||
"node_modules": {}, "vendor": {}, ".cache": {},
|
||||
}
|
||||
if _, ok := skip[name]; ok {
|
||||
continue
|
||||
}
|
||||
path := filepath.Join(root, name)
|
||||
id := EntryID("company", "", name)
|
||||
role, zone := classifyProjectName(name, "")
|
||||
entries = append(entries, Entry{
|
||||
ID: id,
|
||||
Kind: "company",
|
||||
Name: name,
|
||||
Sources: []string{path},
|
||||
Zone: zone,
|
||||
Role: role,
|
||||
Status: "new",
|
||||
Notes: "top_level_dir",
|
||||
})
|
||||
}
|
||||
|
||||
return MergeDocs(&Document{Entries: entries}), nil
|
||||
}
|
||||
|
||||
func walkGitRoots(root, dir string, depth, maxDepth int, out *[]Entry) error {
|
||||
if depth > maxDepth {
|
||||
return nil
|
||||
}
|
||||
// is this a git root?
|
||||
if st, err := os.Stat(filepath.Join(dir, ".git")); err == nil && (st.IsDir() || st.Mode().IsRegular()) {
|
||||
name := filepath.Base(dir)
|
||||
remote := gitRemoteOrigin(dir)
|
||||
role, zone := classifyProjectName(name, remote)
|
||||
*out = append(*out, Entry{
|
||||
ID: EntryID("company", "", name),
|
||||
Kind: "company",
|
||||
Name: name,
|
||||
Sources: []string{dir},
|
||||
Remote: remote,
|
||||
GitRoot: dir,
|
||||
Zone: zone,
|
||||
Role: role,
|
||||
Status: "new",
|
||||
})
|
||||
return nil // do not descend into nested repos from a git root
|
||||
}
|
||||
dents, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
for _, d := range dents {
|
||||
if !d.IsDir() {
|
||||
continue
|
||||
}
|
||||
name := d.Name()
|
||||
if name == ".git" || name == "node_modules" || name == "vendor" || name == ".venv" || name == "dist" {
|
||||
continue
|
||||
}
|
||||
_ = walkGitRoots(root, filepath.Join(dir, name), depth+1, maxDepth, out)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func gitRemoteOrigin(dir string) string {
|
||||
cmd := exec.Command("git", "-C", dir, "remote", "get-url", "origin")
|
||||
b, err := cmd.Output()
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
return strings.TrimSpace(string(b))
|
||||
}
|
||||
|
||||
func classifyProjectName(name, remote string) (role, zone string) {
|
||||
lower := strings.ToLower(name)
|
||||
remoteL := strings.ToLower(remote)
|
||||
switch {
|
||||
case strings.Contains(lower, "experiment") || strings.HasPrefix(lower, "test"):
|
||||
return "experiment", "cold"
|
||||
case lower == "mama" || lower == "personal" || strings.Contains(lower, "private"):
|
||||
return "personal", "private"
|
||||
case strings.Contains(remoteL, "git.produktor.io") || strings.Contains(remoteL, "github.com/eslider"):
|
||||
return "work", "hot"
|
||||
case strings.Contains(lower, "produktor") || strings.Contains(lower, "eslider") ||
|
||||
strings.Contains(lower, "asesoria") || strings.Contains(lower, "dyvenia") ||
|
||||
strings.Contains(lower, "onlyoffice"):
|
||||
return "work", "warm"
|
||||
default:
|
||||
return "unknown", "warm"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,191 @@
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"database/sql"
|
||||
"os"
|
||||
"regexp"
|
||||
"strings"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
var emailRE = regexp.MustCompile(`(?i)[a-z0-9._%+\-]+@[a-z0-9.\-]+\.[a-z]{2,}`)
|
||||
|
||||
// noisyEmail reports addresses that should not become CRM catalog persons.
|
||||
func noisyEmail(email string) bool {
|
||||
e := NormalizeEmail(email)
|
||||
if e == "" || !strings.Contains(e, "@") {
|
||||
return true
|
||||
}
|
||||
local, domain, ok := strings.Cut(e, "@")
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
noiseLocal := []string{
|
||||
"noreply", "no-reply", "donotreply", "mailer-daemon", "postmaster",
|
||||
"bounce", "notifications", "newsletter", "unsubscribe",
|
||||
"root", "admin", "abuse", "webmaster", "hostmaster",
|
||||
}
|
||||
for _, p := range noiseLocal {
|
||||
if local == p || strings.Contains(local, p) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
// role / shared mailboxes (keep human dotted names)
|
||||
roleExact := map[string]struct{}{
|
||||
"info": {}, "alle": {}, "all": {}, "office": {}, "office@": {},
|
||||
"wartung": {}, "starface": {}, "gitlab": {}, "pl": {}, "gf": {},
|
||||
"umsetzung": {}, "support": {}, "sales": {}, "billing": {},
|
||||
}
|
||||
if _, ok := roleExact[local]; ok {
|
||||
return true
|
||||
}
|
||||
if strings.HasSuffix(local, "-request") || strings.HasSuffix(local, "-owner") {
|
||||
return true
|
||||
}
|
||||
noiseDomain := []string{
|
||||
"marketplace.amazon.", "reply.github.com", "users.noreply.github.com",
|
||||
"groups.facebook.com", "googlegroups.com", "yahoogroups.",
|
||||
"email.apple.com", "amazonses.com", "mailchimp.com",
|
||||
"sendgrid.net", "mailgun.org", "postmarkapp.com",
|
||||
"property.booking.com", "transvascular.com",
|
||||
}
|
||||
for _, p := range noiseDomain {
|
||||
if strings.Contains(domain, p) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
// random marketplace / tracking locals
|
||||
if len(local) >= 16 && !strings.Contains(local, ".") && !strings.Contains(local, "-") {
|
||||
if strings.Contains(domain, "amazon.") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func parseMABEmails(path string) ([]Entry, error) {
|
||||
b, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
found := emailRE.FindAllString(string(b), -1)
|
||||
var out []Entry
|
||||
for _, raw := range found {
|
||||
em := NormalizeEmail(raw)
|
||||
if noisyEmail(em) {
|
||||
continue
|
||||
}
|
||||
org, zone, role := classifyMailIdentity("", em)
|
||||
out = append(out, Entry{
|
||||
ID: EntryID("person", em, ""),
|
||||
Kind: "person",
|
||||
Emails: []string{em},
|
||||
Org: org,
|
||||
Sources: []string{path},
|
||||
Zone: zone,
|
||||
Role: role,
|
||||
Approve: false,
|
||||
Status: "new",
|
||||
Notes: "thunderbird_mab",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func parseGlodaContacts(dbPath string) ([]Entry, error) {
|
||||
// read-only URI; immutable=1 helps when WAL/shm are missing
|
||||
dsn := "file:" + dbPath + "?mode=ro&_pragma=query_only(1)"
|
||||
db, err := sql.Open("sqlite", dsn)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer db.Close()
|
||||
|
||||
rows, err := db.Query(`
|
||||
SELECT COALESCE(c.name, ''), i.value
|
||||
FROM identities i
|
||||
LEFT JOIN contacts c ON c.id = i.contactID
|
||||
WHERE lower(i.kind) = 'email' AND i.value IS NOT NULL AND i.value != ''
|
||||
`)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
var out []Entry
|
||||
for rows.Next() {
|
||||
var name, value string
|
||||
if err := rows.Scan(&name, &value); err != nil {
|
||||
return out, err
|
||||
}
|
||||
em := NormalizeEmail(value)
|
||||
// Gloda sometimes stores "Name <email>" in value
|
||||
if m := emailRE.FindString(em); m != "" {
|
||||
em = NormalizeEmail(m)
|
||||
}
|
||||
if noisyEmail(em) {
|
||||
continue
|
||||
}
|
||||
display := strings.TrimSpace(name)
|
||||
// strip wrapping quotes / email leftovers
|
||||
display = strings.Trim(display, `"'`)
|
||||
if idx := strings.Index(display, "<"); idx > 0 {
|
||||
display = strings.TrimSpace(display[:idx])
|
||||
}
|
||||
first, last := "", ""
|
||||
if display != "" {
|
||||
first, last = SplitDisplayName(display)
|
||||
}
|
||||
org, zone, role := classifyMailIdentity(display, em)
|
||||
out = append(out, Entry{
|
||||
ID: EntryID("person", em, display),
|
||||
Kind: "person",
|
||||
Name: display,
|
||||
First: first,
|
||||
Last: last,
|
||||
Emails: []string{em},
|
||||
Org: org,
|
||||
Sources: []string{dbPath},
|
||||
Zone: zone,
|
||||
Role: role,
|
||||
Approve: false,
|
||||
Status: "new",
|
||||
Notes: "thunderbird_gloda",
|
||||
})
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
func classifyMailIdentity(name, email string) (org, zone, role string) {
|
||||
em := NormalizeEmail(email)
|
||||
_, domain, _ := strings.Cut(em, "@")
|
||||
switch {
|
||||
case domain == "wheregroup.com" || strings.Contains(strings.ToLower(name), "wheregroup"):
|
||||
return "WhereGroup", "warm", "work"
|
||||
case domain == "produktor.io" || domain == "eslider.de" || strings.HasSuffix(domain, ".produktor.io"):
|
||||
return "produktor.io", "hot", "work"
|
||||
case domain == "dyvenia.com":
|
||||
return "Dyvenia", "warm", "work"
|
||||
case domain == "immowelt.de" || domain == "immowelt.com":
|
||||
return "Immowelt", "warm", "work"
|
||||
case strings.HasSuffix(domain, ".de") && looksPublicSector(domain):
|
||||
return domain, "warm", "work"
|
||||
default:
|
||||
return "", "private", "unknown"
|
||||
}
|
||||
}
|
||||
|
||||
func looksPublicSector(domain string) bool {
|
||||
d := strings.ToLower(domain)
|
||||
hints := []string{
|
||||
"stadt-", "stadt.", "kreis-", "gemeinde", "landkreis",
|
||||
"bund.", "bundes", "lvermgeo", "rvr.", "eba.",
|
||||
}
|
||||
for _, h := range hints {
|
||||
if strings.Contains(d, h) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"database/sql"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
func TestNoisyEmail(t *testing.T) {
|
||||
if !noisyEmail("noreply@example.com") {
|
||||
t.Fatal("expected noisy")
|
||||
}
|
||||
if !noisyEmail("x@marketplace.amazon.de") {
|
||||
t.Fatal("amazon marketplace")
|
||||
}
|
||||
if noisyEmail("andriy.oblivantsev@wheregroup.com") {
|
||||
t.Fatal("should keep wheregroup")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseMABEmails(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "abook.mab")
|
||||
body := `// mork junk
|
||||
PrimaryEmail=andriy.oblivantsev@wheregroup.com
|
||||
noreply@github.com
|
||||
axel.schaefer@wheregroup.com
|
||||
`
|
||||
if err := os.WriteFile(path, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ents, err := parseMABEmails(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(ents) != 2 {
|
||||
t.Fatalf("got %d: %+v", len(ents), ents)
|
||||
}
|
||||
for _, e := range ents {
|
||||
if e.Org != "WhereGroup" || e.Role != "work" {
|
||||
t.Fatalf("%+v", e)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseGlodaContacts(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
dbPath := filepath.Join(dir, "global-messages-db.sqlite")
|
||||
db, err := sql.Open("sqlite", dbPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, err = db.Exec(`
|
||||
CREATE TABLE contacts (id INTEGER PRIMARY KEY, name TEXT);
|
||||
CREATE TABLE identities (id INTEGER PRIMARY KEY, contactID INTEGER, kind TEXT, value TEXT);
|
||||
INSERT INTO contacts VALUES (1, 'Axel Schaefer');
|
||||
INSERT INTO identities VALUES (1, 1, 'email', 'axel.schaefer@wheregroup.com');
|
||||
INSERT INTO contacts VALUES (2, 'Noise Bot');
|
||||
INSERT INTO identities VALUES (2, 2, 'email', 'noreply@example.com');
|
||||
`)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_ = db.Close()
|
||||
|
||||
ents, err := parseGlodaContacts(dbPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(ents) != 1 {
|
||||
t.Fatalf("got %d %+v", len(ents), ents)
|
||||
}
|
||||
if ents[0].First != "Axel" || ents[0].Emails[0] != "axel.schaefer@wheregroup.com" {
|
||||
t.Fatalf("%+v", ents[0])
|
||||
}
|
||||
}
|
||||
|
||||
func TestScanThunderbirdRoot(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
prof := filepath.Join(root, "profile")
|
||||
if err := os.MkdirAll(prof, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(prof, "abook.mab"), []byte("mail=paul.schmidt@wheregroup.com\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
doc, err := ScanThunderbirdRoot(root)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(doc.Entries) < 1 {
|
||||
t.Fatal(doc.Entries)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,211 @@
|
||||
// Package catalog builds a YAML inventory of CRM persons/companies from
|
||||
// filesystem contacts (VCF, folders) and project trees, then matches/applies
|
||||
// against OnlyOffice CRM.
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
// Entry is one catalog row (person or company).
|
||||
type Entry struct {
|
||||
ID string `yaml:"id"`
|
||||
Kind string `yaml:"kind"` // person | company
|
||||
Name string `yaml:"name,omitempty"`
|
||||
First string `yaml:"first,omitempty"`
|
||||
Last string `yaml:"last,omitempty"`
|
||||
Emails []string `yaml:"emails,omitempty"`
|
||||
Phones []string `yaml:"phones,omitempty"`
|
||||
Org string `yaml:"org,omitempty"`
|
||||
Sources []string `yaml:"sources,omitempty"`
|
||||
Zone string `yaml:"zone"`
|
||||
Role string `yaml:"role"`
|
||||
OOID string `yaml:"oo_id,omitempty"`
|
||||
Approve bool `yaml:"approve"`
|
||||
Status string `yaml:"status,omitempty"` // new | exists | conflict | applied | skipped
|
||||
Notes string `yaml:"notes,omitempty"`
|
||||
Remote string `yaml:"remote,omitempty"`
|
||||
GitRoot string `yaml:"git_root,omitempty"`
|
||||
}
|
||||
|
||||
// Document is the on-disk catalog file.
|
||||
type Document struct {
|
||||
GeneratedAt string `yaml:"generated_at"`
|
||||
Entries []Entry `yaml:"entries"`
|
||||
}
|
||||
|
||||
// LoadYAML reads a catalog document from path.
|
||||
func LoadYAML(path string) (*Document, error) {
|
||||
b, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var doc Document
|
||||
if err := yaml.Unmarshal(b, &doc); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &doc, nil
|
||||
}
|
||||
|
||||
// SaveYAML writes the catalog document.
|
||||
func SaveYAML(path string, doc *Document) error {
|
||||
doc.GeneratedAt = time.Now().UTC().Format(time.RFC3339)
|
||||
b, err := yaml.Marshal(doc)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return os.WriteFile(path, b, 0o644)
|
||||
}
|
||||
|
||||
// NormalizeEmail lowercases and trims.
|
||||
func NormalizeEmail(s string) string {
|
||||
return strings.ToLower(strings.TrimSpace(s))
|
||||
}
|
||||
|
||||
// NormalizeName collapses whitespace and lowercases for matching.
|
||||
func NormalizeName(s string) string {
|
||||
fields := strings.Fields(strings.ToLower(strings.TrimSpace(s)))
|
||||
return strings.Join(fields, " ")
|
||||
}
|
||||
|
||||
// SplitDisplayName splits "First Last …" into first/last (last = remainder).
|
||||
func SplitDisplayName(name string) (first, last string) {
|
||||
parts := strings.Fields(strings.TrimSpace(name))
|
||||
if len(parts) == 0 {
|
||||
return "", ""
|
||||
}
|
||||
if len(parts) == 1 {
|
||||
return parts[0], ""
|
||||
}
|
||||
return parts[0], strings.Join(parts[1:], " ")
|
||||
}
|
||||
|
||||
// EntryID builds a stable-ish id from kind + email or name.
|
||||
func EntryID(kind, email, name string) string {
|
||||
if e := NormalizeEmail(email); e != "" {
|
||||
return fmt.Sprintf("%s:%s", kind, e)
|
||||
}
|
||||
return fmt.Sprintf("%s:%s", kind, NormalizeName(name))
|
||||
}
|
||||
|
||||
// MergeDocs unions entries by id; later sources append sources[] and fill empties.
|
||||
func MergeDocs(docs ...*Document) *Document {
|
||||
byID := map[string]*Entry{}
|
||||
order := []string{}
|
||||
for _, d := range docs {
|
||||
if d == nil {
|
||||
continue
|
||||
}
|
||||
for i := range d.Entries {
|
||||
e := d.Entries[i]
|
||||
id := e.ID
|
||||
if id == "" {
|
||||
primary := ""
|
||||
if len(e.Emails) > 0 {
|
||||
primary = e.Emails[0]
|
||||
}
|
||||
name := e.Name
|
||||
if name == "" {
|
||||
name = strings.TrimSpace(e.First + " " + e.Last)
|
||||
}
|
||||
id = EntryID(e.Kind, primary, name)
|
||||
e.ID = id
|
||||
}
|
||||
if prev, ok := byID[id]; ok {
|
||||
mergeEntry(prev, &e)
|
||||
} else {
|
||||
cp := e
|
||||
byID[id] = &cp
|
||||
order = append(order, id)
|
||||
}
|
||||
}
|
||||
}
|
||||
out := &Document{Entries: make([]Entry, 0, len(order))}
|
||||
for _, id := range order {
|
||||
out.Entries = append(out.Entries, *byID[id])
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func mergeEntry(dst, src *Entry) {
|
||||
dst.Sources = uniqueStrings(append(dst.Sources, src.Sources...))
|
||||
dst.Emails = uniqueEmails(append(dst.Emails, src.Emails...))
|
||||
dst.Phones = uniqueStrings(append(dst.Phones, src.Phones...))
|
||||
if dst.First == "" {
|
||||
dst.First = src.First
|
||||
}
|
||||
if dst.Last == "" {
|
||||
dst.Last = src.Last
|
||||
}
|
||||
if dst.Name == "" {
|
||||
dst.Name = src.Name
|
||||
}
|
||||
if dst.Org == "" {
|
||||
dst.Org = src.Org
|
||||
}
|
||||
if dst.Remote == "" {
|
||||
dst.Remote = src.Remote
|
||||
}
|
||||
if dst.GitRoot == "" {
|
||||
dst.GitRoot = src.GitRoot
|
||||
}
|
||||
if dst.Notes == "" {
|
||||
dst.Notes = src.Notes
|
||||
}
|
||||
// Prefer more specific zone/role from project scans over default private.
|
||||
if dst.Zone == "private" && src.Zone != "" && src.Zone != "private" {
|
||||
dst.Zone = src.Zone
|
||||
}
|
||||
if dst.Role == "" || (dst.Role == "unknown" && src.Role != "" && src.Role != "unknown") {
|
||||
dst.Role = src.Role
|
||||
}
|
||||
// Preserve approve/oo_id/status from whichever already set.
|
||||
if !dst.Approve && src.Approve {
|
||||
dst.Approve = true
|
||||
}
|
||||
if dst.OOID == "" {
|
||||
dst.OOID = src.OOID
|
||||
}
|
||||
if dst.Status == "" {
|
||||
dst.Status = src.Status
|
||||
}
|
||||
}
|
||||
|
||||
func uniqueEmails(in []string) []string {
|
||||
seen := map[string]struct{}{}
|
||||
var out []string
|
||||
for _, e := range in {
|
||||
e = NormalizeEmail(e)
|
||||
if e == "" {
|
||||
continue
|
||||
}
|
||||
if _, ok := seen[e]; ok {
|
||||
continue
|
||||
}
|
||||
seen[e] = struct{}{}
|
||||
out = append(out, e)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func uniqueStrings(in []string) []string {
|
||||
seen := map[string]struct{}{}
|
||||
var out []string
|
||||
for _, s := range in {
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
continue
|
||||
}
|
||||
if _, ok := seen[s]; ok {
|
||||
continue
|
||||
}
|
||||
seen[s] = struct{}{}
|
||||
out = append(out, s)
|
||||
}
|
||||
return out
|
||||
}
|
||||
+218
@@ -0,0 +1,218 @@
|
||||
package catalog
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"github.com/emersion/go-vcard"
|
||||
)
|
||||
|
||||
// ParseVCFFile extracts person candidates from a vCard file (may contain many cards).
|
||||
func ParseVCFFile(path string) ([]Entry, error) {
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
dec := vcard.NewDecoder(f)
|
||||
var out []Entry
|
||||
for {
|
||||
card, err := dec.Decode()
|
||||
if err != nil {
|
||||
if errors.Is(err, io.EOF) {
|
||||
break
|
||||
}
|
||||
return out, fmt.Errorf("%s: %w", path, err)
|
||||
}
|
||||
e := entryFromCard(card, path)
|
||||
if e.Name == "" && e.First == "" && len(e.Emails) == 0 {
|
||||
continue
|
||||
}
|
||||
out = append(out, e)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func entryFromCard(card vcard.Card, source string) Entry {
|
||||
fn := strings.TrimSpace(card.PreferredValue(vcard.FieldFormattedName))
|
||||
n := card.Name()
|
||||
first, last := "", ""
|
||||
if n != nil {
|
||||
first = strings.TrimSpace(n.GivenName)
|
||||
last = strings.TrimSpace(n.FamilyName)
|
||||
}
|
||||
if fn == "" {
|
||||
fn = strings.TrimSpace(first + " " + last)
|
||||
}
|
||||
if first == "" && last == "" && fn != "" {
|
||||
first, last = SplitDisplayName(fn)
|
||||
}
|
||||
var emails, phones []string
|
||||
for _, v := range card.Values(vcard.FieldEmail) {
|
||||
if e := NormalizeEmail(v); e != "" {
|
||||
emails = append(emails, e)
|
||||
}
|
||||
}
|
||||
for _, v := range card.Values(vcard.FieldTelephone) {
|
||||
p := strings.TrimSpace(v)
|
||||
if p != "" {
|
||||
phones = append(phones, p)
|
||||
}
|
||||
}
|
||||
org := strings.TrimSpace(card.PreferredValue(vcard.FieldOrganization))
|
||||
primary := ""
|
||||
if len(emails) > 0 {
|
||||
primary = emails[0]
|
||||
}
|
||||
return Entry{
|
||||
ID: EntryID("person", primary, fn),
|
||||
Kind: "person",
|
||||
Name: fn,
|
||||
First: first,
|
||||
Last: last,
|
||||
Emails: uniqueEmails(emails),
|
||||
Phones: uniqueStrings(phones),
|
||||
Org: org,
|
||||
Sources: []string{source},
|
||||
Zone: "private",
|
||||
Role: "unknown",
|
||||
Approve: false,
|
||||
Status: "new",
|
||||
}
|
||||
}
|
||||
|
||||
// ScanContactsRoot walks a contacts directory: *.vcf anywhere, person-named
|
||||
// subdirs, and *@*.txt email hint files.
|
||||
func ScanContactsRoot(root string) (*Document, error) {
|
||||
root = filepath.Clean(root)
|
||||
st, err := os.Stat(root)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !st.IsDir() {
|
||||
return nil, fmt.Errorf("not a directory: %s", root)
|
||||
}
|
||||
|
||||
var entries []Entry
|
||||
seenVCF := map[string]struct{}{}
|
||||
|
||||
err = filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error {
|
||||
if err != nil {
|
||||
return nil // skip unreadable
|
||||
}
|
||||
if d.IsDir() {
|
||||
base := d.Name()
|
||||
if base == ".git" || base == "node_modules" {
|
||||
return filepath.SkipDir
|
||||
}
|
||||
return nil
|
||||
}
|
||||
name := d.Name()
|
||||
lower := strings.ToLower(name)
|
||||
if strings.HasSuffix(lower, ".vcf") {
|
||||
if _, ok := seenVCF[path]; ok {
|
||||
return nil
|
||||
}
|
||||
seenVCF[path] = struct{}{}
|
||||
parsed, perr := ParseVCFFile(path)
|
||||
if perr != nil {
|
||||
entries = append(entries, Entry{
|
||||
ID: EntryID("person", "", filepath.Base(path)),
|
||||
Kind: "person",
|
||||
Name: strings.TrimSuffix(filepath.Base(path), filepath.Ext(path)),
|
||||
Sources: []string{path},
|
||||
Zone: "private",
|
||||
Role: "unknown",
|
||||
Notes: "vcf_parse_error: " + perr.Error(),
|
||||
Status: "new",
|
||||
})
|
||||
return nil
|
||||
}
|
||||
entries = append(entries, parsed...)
|
||||
return nil
|
||||
}
|
||||
// email hint files: name contains @ and ends with .txt
|
||||
if strings.Contains(name, "@") && strings.HasSuffix(lower, ".txt") {
|
||||
email := NormalizeEmail(strings.TrimSuffix(name, filepath.Ext(name)))
|
||||
if !strings.Contains(email, "@") {
|
||||
return nil
|
||||
}
|
||||
parent := filepath.Base(filepath.Dir(path))
|
||||
first, last := SplitDisplayName(parent)
|
||||
entries = append(entries, Entry{
|
||||
ID: EntryID("person", email, parent),
|
||||
Kind: "person",
|
||||
Name: parent,
|
||||
First: first,
|
||||
Last: last,
|
||||
Emails: []string{email},
|
||||
Sources: []string{path},
|
||||
Zone: "private",
|
||||
Role: "unknown",
|
||||
Status: "new",
|
||||
})
|
||||
}
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// Top-level person folders without VCF still become stub persons.
|
||||
dents, _ := os.ReadDir(root)
|
||||
for _, d := range dents {
|
||||
if !d.IsDir() {
|
||||
continue
|
||||
}
|
||||
name := d.Name()
|
||||
if name == "Contacts VCF's" || strings.HasPrefix(name, ".") {
|
||||
continue
|
||||
}
|
||||
// skip obvious non-person dumps
|
||||
lower := strings.ToLower(name)
|
||||
if strings.Contains(lower, "vcf") {
|
||||
continue
|
||||
}
|
||||
first, last := SplitDisplayName(name)
|
||||
id := EntryID("person", "", name)
|
||||
entries = append(entries, Entry{
|
||||
ID: id,
|
||||
Kind: "person",
|
||||
Name: name,
|
||||
First: first,
|
||||
Last: last,
|
||||
Sources: []string{filepath.Join(root, name)},
|
||||
Zone: "private",
|
||||
Role: "unknown",
|
||||
Status: "new",
|
||||
Notes: "folder_stub",
|
||||
})
|
||||
}
|
||||
|
||||
doc := MergeDocs(&Document{Entries: entries})
|
||||
return doc, nil
|
||||
}
|
||||
|
||||
// ReadEmailsFromTxt reads loose emails from a text file (one per line or free text).
|
||||
func ReadEmailsFromTxt(path string) ([]string, error) {
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer f.Close()
|
||||
var out []string
|
||||
sc := bufio.NewScanner(f)
|
||||
for sc.Scan() {
|
||||
line := strings.TrimSpace(sc.Text())
|
||||
if strings.Contains(line, "@") && !strings.Contains(line, " ") {
|
||||
out = append(out, NormalizeEmail(line))
|
||||
}
|
||||
}
|
||||
return uniqueEmails(out), sc.Err()
|
||||
}
|
||||
@@ -11,8 +11,10 @@ package onlyoffice
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/cookiejar"
|
||||
"os"
|
||||
"strings"
|
||||
"sync"
|
||||
)
|
||||
|
||||
// Client of OnlyOffice API uses credentials to get a token and query the API
|
||||
@@ -31,12 +33,16 @@ type Client struct {
|
||||
defaults Defaults // optional fallbacks for calendar/project IDs
|
||||
selfID string // cached /api/2.0/people/@self id
|
||||
noteCatID int // cached CRM history category id for "note"
|
||||
|
||||
folderTitles map[string]string // cached Documents folder id -> title (F9)
|
||||
folderTitlesMu sync.Mutex
|
||||
}
|
||||
|
||||
// NewClient returns a new Client backed by http.DefaultClient.
|
||||
func NewClient(c Credentials) *Client {
|
||||
jar, _ := cookiejar.New(nil)
|
||||
return &Client{
|
||||
client: http.DefaultClient,
|
||||
client: &http.Client{Jar: jar},
|
||||
credentials: &c,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -139,6 +139,23 @@ func TestIntegrationProjectAndTaskLifecycle(t *testing.T) {
|
||||
|
||||
start := Time(time.Now().AddDate(0, 0, -2))
|
||||
deadline := Time(time.Now().AddDate(0, 0, 2))
|
||||
ms, err := c.CreateMilestone(NewMilestoneRequest{
|
||||
ProjectID: *project.ID,
|
||||
Title: "integration milestone",
|
||||
Deadline: deadline,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("CreateMilestone: %v", err)
|
||||
}
|
||||
if ms == nil || ms.ID == nil {
|
||||
t.Fatal("created milestone without id")
|
||||
}
|
||||
t.Cleanup(func() {
|
||||
if err := c.DeleteMilestone(*ms.ID); err != nil {
|
||||
t.Logf("cleanup DeleteMilestone: %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
task, err := c.CreateProjectTask(NewProjectTaskRequest{
|
||||
ProjectId: *project.ID,
|
||||
Title: "integration parent task",
|
||||
@@ -146,6 +163,7 @@ func TestIntegrationProjectAndTaskLifecycle(t *testing.T) {
|
||||
StartDate: start,
|
||||
Deadline: deadline,
|
||||
Priority: int(TaskPriorityNormal),
|
||||
MilestoneId: int(*ms.ID),
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("CreateProjectTask: %v", err)
|
||||
|
||||
@@ -0,0 +1,220 @@
|
||||
// Command kontoblatt builds a summary ("сводная таблица") of a Kontoblatt XLSX
|
||||
// (Datum, Gegenkonto, Buchungstext, Beleg, Soll, Haben, Bemerkung) and uploads
|
||||
// it back to the same OnlyOffice folder as the source file.
|
||||
//
|
||||
// Usage: kontoblatt <FILE_ID> <LOCAL_XLSX>
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
type agg struct {
|
||||
count int
|
||||
soll float64
|
||||
haben float64
|
||||
reFehlt int
|
||||
}
|
||||
|
||||
type rec struct {
|
||||
date, month, konto, text string
|
||||
soll, haben float64
|
||||
reFehlt bool
|
||||
}
|
||||
|
||||
var dateRe = regexp.MustCompile(`^\d{2}\.\d{2}\.\d{4}$`)
|
||||
|
||||
func parseAmount(s string) float64 {
|
||||
s = strings.TrimSpace(s)
|
||||
s = strings.ReplaceAll(s, "€", "")
|
||||
s = strings.ReplaceAll(s, " ", "")
|
||||
s = strings.ReplaceAll(s, ",", "") // German thousands separator
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return 0
|
||||
}
|
||||
v, err := strconv.ParseFloat(s, 64)
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
func cell(row []string, i int) string {
|
||||
if i < len(row) {
|
||||
return strings.TrimSpace(row[i])
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func main() {
|
||||
if len(os.Args) < 3 {
|
||||
fmt.Fprintln(os.Stderr, "usage: kontoblatt <FILE_ID> <LOCAL_XLSX>")
|
||||
os.Exit(2)
|
||||
}
|
||||
fileID, path := os.Args[1], os.Args[2]
|
||||
ctx := context.Background()
|
||||
|
||||
f, err := excelize.OpenFile(path)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
var recs []rec
|
||||
for _, sh := range f.GetSheetList() {
|
||||
rows, err := f.GetRows(sh)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
for _, r := range rows {
|
||||
d := cell(r, 0)
|
||||
if !dateRe.MatchString(d) {
|
||||
continue
|
||||
}
|
||||
text := cell(r, 2)
|
||||
recs = append(recs, rec{
|
||||
date: d,
|
||||
month: d[3:10],
|
||||
konto: cell(r, 1),
|
||||
text: text,
|
||||
soll: parseAmount(cell(r, 4)),
|
||||
haben: parseAmount(cell(r, 5)),
|
||||
reFehlt: strings.Contains(strings.ToUpper(text), "FEHLT"),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
byKonto := map[string]*agg{}
|
||||
byMonth := map[string]*agg{}
|
||||
getK := func(k string) *agg {
|
||||
if byKonto[k] == nil {
|
||||
byKonto[k] = &agg{}
|
||||
}
|
||||
return byKonto[k]
|
||||
}
|
||||
getM := func(k string) *agg {
|
||||
if byMonth[k] == nil {
|
||||
byMonth[k] = &agg{}
|
||||
}
|
||||
return byMonth[k]
|
||||
}
|
||||
var tot agg
|
||||
for _, r := range recs {
|
||||
k := getK(r.konto)
|
||||
k.count++
|
||||
k.soll += r.soll
|
||||
k.haben += r.haben
|
||||
if r.reFehlt {
|
||||
k.reFehlt++
|
||||
}
|
||||
m := getM(r.month)
|
||||
m.count++
|
||||
m.soll += r.soll
|
||||
m.haben += r.haben
|
||||
if r.reFehlt {
|
||||
m.reFehlt++
|
||||
}
|
||||
tot.count++
|
||||
tot.soll += r.soll
|
||||
tot.haben += r.haben
|
||||
if r.reFehlt {
|
||||
tot.reFehlt++
|
||||
}
|
||||
}
|
||||
|
||||
out := excelize.NewFile()
|
||||
defer out.Close()
|
||||
writeSheet(out, "Nach Gegenkonto", "Gegenkonto", byKonto, tot)
|
||||
writeSheet(out, "Nach Monat", "Monat", byMonth, tot)
|
||||
outPath := "/tmp/opencode/kontoblatt-zusammenfassung.xlsx"
|
||||
if err := out.SaveAs(outPath); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
|
||||
// upload next to the source file
|
||||
creds := onlyoffice.GetEnvironmentCredentials()
|
||||
c := onlyoffice.NewClient(creds)
|
||||
var src *onlyoffice.FileEntry
|
||||
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
var err error
|
||||
src, err = c.GetFile(ctx, fileID)
|
||||
return err
|
||||
}); derr != nil {
|
||||
panic(derr)
|
||||
}
|
||||
folder := ""
|
||||
if src.FolderID != nil {
|
||||
folder = src.FolderID.String()
|
||||
}
|
||||
title := ""
|
||||
if src.Title != nil {
|
||||
title = *src.Title
|
||||
}
|
||||
fmt.Printf("source: id=%s title=%q folder=%s\n", fileID, title, folder)
|
||||
|
||||
name := "Kontoblatt-1591-2025-Zusammenfassung.xlsx"
|
||||
tmp := "/tmp/opencode/" + name
|
||||
data, _ := os.ReadFile(outPath)
|
||||
if err := os.WriteFile(tmp, data, 0o600); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
var entry *onlyoffice.FileEntry
|
||||
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
var err error
|
||||
entry, _, err = c.UploadToFolderReplacing(ctx, folder, tmp)
|
||||
return err
|
||||
}); derr != nil {
|
||||
panic(derr)
|
||||
}
|
||||
fmt.Printf("uploaded: %s -> folder %s (id %v)\n", name, folder, entry.ID)
|
||||
|
||||
// print the summary
|
||||
printAgg("Nach Gegenkonto", byKonto, tot)
|
||||
printAgg("Nach Monat", byMonth, tot)
|
||||
}
|
||||
|
||||
func writeSheet(f *excelize.File, sheet, key string, m map[string]*agg, tot agg) {
|
||||
f.NewSheet(sheet)
|
||||
rows := [][]any{{key, "Anzahl", "Soll", "Haben", "Saldo", `davon "fehlt"`}}
|
||||
keys := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
for _, k := range keys {
|
||||
a := m[k]
|
||||
rows = append(rows, []any{k, a.count, a.soll, a.haben, a.soll - a.haben, a.reFehlt})
|
||||
}
|
||||
rows = append(rows, []any{"GESAMT", tot.count, tot.soll, tot.haben, tot.soll - tot.haben, tot.reFehlt})
|
||||
for i, row := range rows {
|
||||
for j, v := range row {
|
||||
cellRef, _ := excelize.CoordinatesToCellName(j+1, i+1)
|
||||
_ = f.SetCellValue(sheet, cellRef, v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func printAgg(title string, m map[string]*agg, tot agg) {
|
||||
fmt.Printf("\n== %s ==\n", title)
|
||||
keys := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
fmt.Printf("%-12s %6s %12s %12s %12s %7s\n", "key", "count", "soll", "haben", "saldo", "fehlt")
|
||||
for _, k := range keys {
|
||||
a := m[k]
|
||||
fmt.Printf("%-12s %6d %12.2f %12.2f %12.2f %7d\n", k, a.count, a.soll, a.haben, a.soll-a.haben, a.reFehlt)
|
||||
}
|
||||
fmt.Printf("%-12s %6d %12.2f %12.2f %12.2f %7d\n", "GESAMT", tot.count, tot.soll, tot.haben, tot.soll-tot.haben, tot.reFehlt)
|
||||
}
|
||||
@@ -0,0 +1,478 @@
|
||||
// Command kontolink fills the "Link" column of a Kontoblatt ("ungeklärte
|
||||
// Posten") XLSX by matching each row to an OnlyOffice document.
|
||||
//
|
||||
// Strategy (deterministic, conservative — no LLM):
|
||||
// 1. Beleg token (letters/digits from the "Beleg" column) appears in the file
|
||||
// title; among candidates prefer (a) the row's month, (b) real invoices over
|
||||
// copies/dupes, and require the result to be unique;
|
||||
// 2. else supplier + row month + "rechnung", again unique.
|
||||
//
|
||||
// A file is linked at most once (rows already carrying a link are kept and their
|
||||
// file counts as used). Ambiguous rows are left UNLINKED for manual review.
|
||||
//
|
||||
// Usage: kontolink <IN_XLSX> <INDEX_TSV> <OUT_XLSX>
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
var (
|
||||
dateRe = regexp.MustCompile(`^\d{2}\.\d{2}\.\d{4}$`)
|
||||
nonAln = regexp.MustCompile(`[^0-9a-z]+`)
|
||||
fileID = regexp.MustCompile(`fileid=(\d+)`)
|
||||
)
|
||||
|
||||
func parseDay(s string) (time.Time, bool) {
|
||||
t, err := time.Parse("02.01.2006", strings.TrimSpace(s))
|
||||
return t, err == nil
|
||||
}
|
||||
|
||||
func titleDay(title string) (time.Time, bool) {
|
||||
if len(title) >= 10 {
|
||||
if t, err := time.Parse("2006-01-02", title[:10]); err == nil {
|
||||
return t, true
|
||||
}
|
||||
}
|
||||
return time.Time{}, false
|
||||
}
|
||||
|
||||
// nearest picks the candidate whose title date is closest to rd. Ties and
|
||||
// undated candidates (when >1) are rejected.
|
||||
func nearest(cands []entry, rd time.Time) (entry, bool) {
|
||||
if len(cands) == 1 {
|
||||
return cands[0], true
|
||||
}
|
||||
best, bestD, tie := -1, 0.0, false
|
||||
for i, e := range cands {
|
||||
td, ok := titleDay(e.title)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
d := td.Sub(rd).Hours() / 24
|
||||
if d < 0 {
|
||||
d = -d
|
||||
}
|
||||
if best < 0 || d < bestD {
|
||||
best, bestD, tie = i, d, false
|
||||
} else if d == bestD {
|
||||
tie = true
|
||||
}
|
||||
}
|
||||
if best < 0 || tie {
|
||||
return entry{}, false
|
||||
}
|
||||
return cands[best], true
|
||||
}
|
||||
|
||||
const linkPrefix = "https://office.pro-dukt.de/Products/Files/DocEditor.aspx?fileid="
|
||||
|
||||
type entry struct {
|
||||
id, path, title, norm string
|
||||
}
|
||||
|
||||
func norm(s string) string { return nonAln.ReplaceAllString(strings.ToLower(s), "") }
|
||||
|
||||
func main() {
|
||||
if len(os.Args) < 4 {
|
||||
fmt.Fprintln(os.Stderr, "usage: kontolink <IN_XLSX> <INDEX_TSV> <OUT_XLSX>")
|
||||
os.Exit(2)
|
||||
}
|
||||
in, idxPath, out := os.Args[1], os.Args[2], os.Args[3]
|
||||
|
||||
idxRaw, err := os.ReadFile(idxPath)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
var entries []entry
|
||||
for _, line := range strings.Split(string(idxRaw), "\n") {
|
||||
parts := strings.Split(line, "\t")
|
||||
if len(parts) < 4 || parts[0] == "" {
|
||||
continue
|
||||
}
|
||||
entries = append(entries, entry{id: parts[0], path: parts[2], title: parts[3], norm: norm(parts[3])})
|
||||
}
|
||||
|
||||
f, err := excelize.OpenFile(in)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
defer f.Close()
|
||||
sheet := f.GetSheetList()[0]
|
||||
rows, err := f.GetRows(sheet)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
|
||||
// optional 5th arg: amounts TSV "file_id\ttitle\tamount" (see cmd/pdfamount)
|
||||
var amts []amtEntry
|
||||
if len(os.Args) >= 6 && os.Args[5] != "" {
|
||||
amts = loadAmounts(os.Args[5])
|
||||
}
|
||||
|
||||
used := map[string]bool{}
|
||||
for _, r := range rows {
|
||||
if m := fileID.FindStringSubmatch(cell(r, 7)); m != nil {
|
||||
used[m[1]] = true
|
||||
}
|
||||
}
|
||||
|
||||
var linked, byBeleg, bySupplier, byAmount, unmatched, ambiguous int
|
||||
for i, r := range rows {
|
||||
if i == 0 || !dateRe.MatchString(cell(r, 0)) || strings.TrimSpace(cell(r, 7)) != "" {
|
||||
continue
|
||||
}
|
||||
beleg := norm(cell(r, 3))
|
||||
supplier := supplierNorm(cell(r, 2))
|
||||
month := monthYear(cell(r, 0))
|
||||
rd, _ := parseDay(cell(r, 0))
|
||||
|
||||
e, kind, ok := pick(entries, used, beleg, supplier, month, rd)
|
||||
if !ok {
|
||||
if ae, aok := amountPick(amts, used, supplier, rowAmount(r), rd); aok {
|
||||
e, kind, ok = entry{id: ae.id, title: ae.title}, "amount", true
|
||||
}
|
||||
}
|
||||
if !ok {
|
||||
if beleg != "" {
|
||||
ambiguous++
|
||||
} else {
|
||||
unmatched++
|
||||
}
|
||||
continue
|
||||
}
|
||||
ref, _ := excelize.CoordinatesToCellName(8, i+1)
|
||||
if err := f.SetCellValue(sheet, ref, linkPrefix+e.id); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
used[e.id] = true
|
||||
linked++
|
||||
switch kind {
|
||||
case "beleg":
|
||||
byBeleg++
|
||||
case "supplier":
|
||||
bySupplier++
|
||||
case "amount":
|
||||
byAmount++
|
||||
}
|
||||
fmt.Printf("row %3d %-30s -> %s [%s]\n", i+1, cell(r, 2), e.title, kind)
|
||||
}
|
||||
|
||||
if err := f.SaveAs(out); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
fmt.Printf("\nlinked=%d (beleg=%d, supplier=%d, amount=%d), ambiguous=%d, no-candidate=%d\n",
|
||||
linked, byBeleg, bySupplier, byAmount, ambiguous, unmatched)
|
||||
|
||||
// Optional 4th arg: source OnlyOffice file id. Try to update it in place;
|
||||
// if it is locked (OnlyOffice 500), upload a "(links)" copy next to it.
|
||||
if len(os.Args) >= 5 && os.Args[4] != "" {
|
||||
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 120*time.Second)
|
||||
defer cancel()
|
||||
var src *onlyoffice.FileEntry
|
||||
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
var err error
|
||||
src, err = c.GetFile(ctx, os.Args[4])
|
||||
return err
|
||||
}); derr != nil {
|
||||
panic(derr)
|
||||
}
|
||||
folder, title := "", ""
|
||||
if src.FolderID != nil {
|
||||
folder = src.FolderID.String()
|
||||
}
|
||||
if src.Title != nil {
|
||||
title = *src.Title
|
||||
}
|
||||
uderr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
_, err := c.UpdateFile(ctx, os.Args[4], out)
|
||||
return err
|
||||
})
|
||||
if uderr == nil {
|
||||
fmt.Printf("updated file %s in place\n", os.Args[4])
|
||||
return
|
||||
}
|
||||
fmt.Printf("in-place update failed (locked?); uploading a copy to folder %s\n", folder)
|
||||
ext := filepath.Ext(title)
|
||||
name := strings.TrimSuffix(title, ext) + " (links)" + ext
|
||||
tmp := filepath.Join(os.TempDir(), name)
|
||||
data, _ := os.ReadFile(out)
|
||||
if err := os.WriteFile(tmp, data, 0o600); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
_, _, err := c.UploadToFolderReplacing(ctx, folder, tmp)
|
||||
return err
|
||||
}); derr != nil {
|
||||
panic(derr)
|
||||
}
|
||||
fmt.Printf("uploaded copy: %s -> folder %s\n", name, folder)
|
||||
}
|
||||
}
|
||||
|
||||
func cell(r []string, i int) string {
|
||||
if i < len(r) {
|
||||
return strings.TrimSpace(r[i])
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func supplierNorm(s string) string {
|
||||
s = strings.ToUpper(s)
|
||||
if i := strings.Index(s, ","); i >= 0 {
|
||||
s = s[:i]
|
||||
}
|
||||
for _, w := range []string{"RE FEHLT", "GS FEHLT", "WOFR", "WOFÜR"} {
|
||||
s = strings.ReplaceAll(s, w, "")
|
||||
}
|
||||
return norm(s)
|
||||
}
|
||||
|
||||
func monthYear(date string) string {
|
||||
if len(date) == 10 {
|
||||
return date[6:10] + "-" + date[3:5]
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// pick returns an unused candidate. Beleg match wins; supplier+month is a
|
||||
// fallback. When several candidates qualify, the one closest in time to the row
|
||||
// date wins; a tie is rejected (ambiguous) rather than guessed.
|
||||
func pick(entries []entry, used map[string]bool, beleg, supplier, month string, rd time.Time) (entry, string, bool) {
|
||||
free := func(e entry) bool { return !used[e.id] }
|
||||
|
||||
if len(beleg) >= 5 {
|
||||
var inMonth []entry
|
||||
for _, e := range entries {
|
||||
if free(e) && belegMatches(e.norm, beleg) &&
|
||||
(month == "" || strings.Contains(e.title, month)) {
|
||||
inMonth = append(inMonth, e)
|
||||
}
|
||||
}
|
||||
if supplier != "" {
|
||||
var s []entry
|
||||
for _, e := range inMonth {
|
||||
if strings.Contains(e.norm, supplier) {
|
||||
s = append(s, e)
|
||||
}
|
||||
}
|
||||
if len(s) > 0 {
|
||||
inMonth = s
|
||||
}
|
||||
}
|
||||
inMonth = topRank(inMonth)
|
||||
if e, ok := nearest(inMonth, rd); ok {
|
||||
return e, "beleg", true
|
||||
}
|
||||
// A Beleg is present but no file carries it: do NOT fall back to a
|
||||
// supplier guess (that links the wrong invoice).
|
||||
return entry{}, "", false
|
||||
}
|
||||
|
||||
if supplier != "" && month != "" {
|
||||
var c []entry
|
||||
for _, e := range entries {
|
||||
if free(e) && strings.Contains(e.norm, supplier) &&
|
||||
strings.Contains(e.title, month) && strings.Contains(e.norm, "rechnung") {
|
||||
c = append(c, e)
|
||||
}
|
||||
}
|
||||
c = topRank(c)
|
||||
if e, ok := nearest(c, rd); ok {
|
||||
return e, "supplier", true
|
||||
}
|
||||
}
|
||||
return entry{}, "", false
|
||||
}
|
||||
|
||||
// topRank keeps only the highest-ranked candidates (real invoice over copy /
|
||||
// dupe / op), so a tie with a duplicate does not mask the real file.
|
||||
func topRank(cands []entry) []entry {
|
||||
if len(cands) < 2 {
|
||||
return cands
|
||||
}
|
||||
best := 0
|
||||
for _, e := range cands {
|
||||
if rank(e) > best {
|
||||
best = rank(e)
|
||||
}
|
||||
}
|
||||
out := cands[:0]
|
||||
for _, e := range cands {
|
||||
if rank(e) == best {
|
||||
out = append(out, e)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func rank(e entry) int {
|
||||
s := 0
|
||||
if strings.Contains(e.path, "/2025") || strings.Contains(e.path, "/2024") {
|
||||
s += 4
|
||||
}
|
||||
if strings.Contains(e.norm, "rechnung") {
|
||||
s += 2
|
||||
}
|
||||
if strings.Contains(e.norm, "dupe") || strings.Contains(e.norm, "copy") ||
|
||||
strings.Contains(e.norm, "op") {
|
||||
s--
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// belegMatches reports whether a Beleg identifies the file: the whole normalized
|
||||
// Beleg appears, or (for long numeric Belege, e.g. "24/641393110") an 8-digit
|
||||
// window of its longest digit run appears.
|
||||
func belegMatches(titleNorm, beleg string) bool {
|
||||
if strings.Contains(titleNorm, beleg) {
|
||||
return true
|
||||
}
|
||||
run := longestDigitRun(beleg)
|
||||
for i := 0; i+8 <= len(run); i++ {
|
||||
if strings.Contains(titleNorm, run[i:i+8]) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func longestDigitRun(s string) string {
|
||||
var best, cur strings.Builder
|
||||
for _, r := range s {
|
||||
if r >= '0' && r <= '9' {
|
||||
cur.WriteRune(r)
|
||||
if cur.Len() > best.Len() {
|
||||
best.Reset()
|
||||
best.WriteString(cur.String())
|
||||
}
|
||||
} else {
|
||||
cur.Reset()
|
||||
}
|
||||
}
|
||||
return best.String()
|
||||
}
|
||||
|
||||
type amtEntry struct {
|
||||
id string
|
||||
title string
|
||||
norm string
|
||||
amount float64
|
||||
date time.Time
|
||||
hasDate bool
|
||||
}
|
||||
|
||||
func loadAmounts(path string) []amtEntry {
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
var out []amtEntry
|
||||
for _, line := range strings.Split(string(raw), "\n") {
|
||||
p := strings.Split(line, "\t")
|
||||
if len(p) < 3 {
|
||||
continue
|
||||
}
|
||||
v, err := strconv.ParseFloat(strings.TrimSpace(p[2]), 64)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
e := amtEntry{id: p[0], title: p[1], norm: norm(p[1]), amount: v}
|
||||
if len(p[1]) >= 10 {
|
||||
if t, err := time.Parse("2006-01-02", p[1][:10]); err == nil {
|
||||
e.date, e.hasDate = t, true
|
||||
}
|
||||
}
|
||||
out = append(out, e)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func rowAmount(r []string) float64 {
|
||||
if v := parseAmount(cell(r, 4)); v != 0 {
|
||||
return v
|
||||
}
|
||||
return parseAmount(cell(r, 5))
|
||||
}
|
||||
|
||||
func parseAmount(s string) float64 {
|
||||
s = strings.ReplaceAll(s, "€", "")
|
||||
s = strings.ReplaceAll(s, " ", "")
|
||||
s = strings.ReplaceAll(s, ",", ".")
|
||||
if s == "" {
|
||||
return 0
|
||||
}
|
||||
v, err := strconv.ParseFloat(s, 64)
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// amountPick matches a row to an O2 invoice by amount + nearest date. Scoped to
|
||||
// Telefonica/O2 rows and O2 files, so it cannot cross-link other suppliers.
|
||||
func amountPick(amts []amtEntry, used map[string]bool, supplier string, amt float64, rd time.Time) (amtEntry, bool) {
|
||||
if amt <= 0 || len(amts) == 0 {
|
||||
return amtEntry{}, false
|
||||
}
|
||||
if !strings.Contains(supplier, "telefonica") && !strings.Contains(supplier, "o2") {
|
||||
return amtEntry{}, false
|
||||
}
|
||||
var cands []amtEntry
|
||||
for _, a := range amts {
|
||||
if used[a.id] || !strings.Contains(a.norm, "o2") {
|
||||
continue
|
||||
}
|
||||
d := a.amount - amt
|
||||
if d < 0 {
|
||||
d = -d
|
||||
}
|
||||
if d > 0.005 {
|
||||
continue
|
||||
}
|
||||
if a.hasDate && !rd.IsZero() {
|
||||
days := a.date.Sub(rd).Hours() / 24
|
||||
if days < 0 {
|
||||
days = -days
|
||||
}
|
||||
if days > 75 {
|
||||
continue
|
||||
}
|
||||
}
|
||||
cands = append(cands, a)
|
||||
}
|
||||
if len(cands) == 1 {
|
||||
return cands[0], true
|
||||
}
|
||||
best, bestD, tie := -1, 0.0, false
|
||||
for i, a := range cands {
|
||||
if !a.hasDate {
|
||||
continue
|
||||
}
|
||||
d := a.date.Sub(rd).Hours() / 24
|
||||
if d < 0 {
|
||||
d = -d
|
||||
}
|
||||
if best < 0 || d < bestD {
|
||||
best, bestD, tie = i, d, false
|
||||
} else if d == bestD {
|
||||
tie = true
|
||||
}
|
||||
}
|
||||
if best < 0 || tie {
|
||||
return amtEntry{}, false
|
||||
}
|
||||
return cands[best], true
|
||||
}
|
||||
@@ -17,6 +17,18 @@ const MailListPageSize = 25
|
||||
// Loader fetches list items for a menu subject using the OnlyOffice client.
|
||||
type Loader struct {
|
||||
Client *onlyoffice.Client
|
||||
|
||||
// Files is the backend-agnostic file store used for file download, preview
|
||||
// and delete. When nil it falls back to Client.FileStore(ProviderREST).
|
||||
Files onlyoffice.FileStore
|
||||
}
|
||||
|
||||
// fileStore returns the configured file store, defaulting to REST.
|
||||
func (l *Loader) fileStore() onlyoffice.FileStore {
|
||||
if l.Files != nil {
|
||||
return l.Files
|
||||
}
|
||||
return l.Client.FileStore(onlyoffice.ProviderREST)
|
||||
}
|
||||
|
||||
// List returns items for the given list spec (nav leaf).
|
||||
@@ -171,11 +183,7 @@ func (l *Loader) executeDelete(ctx context.Context, item model.Item) (string, er
|
||||
}
|
||||
return fmt.Sprintf("Deleted message %s", item.Title), nil
|
||||
case model.KindFile:
|
||||
id, err := strconv.Atoi(item.ID)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
if err := l.Client.DeleteFiles(ctx, []int{id}); err != nil {
|
||||
if err := l.fileStore().Delete(ctx, []string{item.ID}); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return fmt.Sprintf("Deleted file %s", item.Title), nil
|
||||
@@ -199,7 +207,7 @@ func (l *Loader) executeDownload(ctx context.Context, item model.Item, destPath
|
||||
return "", err
|
||||
}
|
||||
defer f.Close()
|
||||
if _, err := l.Client.DownloadFile(ctx, item.ID, f); err != nil {
|
||||
if _, err := l.fileStore().Download(ctx, item.ID, f); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return fmt.Sprintf("Downloaded to %s", destPath), nil
|
||||
|
||||
@@ -5,7 +5,6 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/eslider/go-onlyoffice/cmd/office/model"
|
||||
"github.com/eslider/go-onlyoffice/cmd/office/preview"
|
||||
)
|
||||
@@ -30,13 +29,11 @@ func (l *Loader) filePreviewMarkdown(ctx context.Context, item model.Item) (stri
|
||||
return "", fmt.Errorf("file id missing")
|
||||
}
|
||||
name := item.Title
|
||||
if meta, err := l.Client.GetFile(ctx, item.ID); err == nil && meta != nil {
|
||||
if t := onlyoffice.FileEntryTitle(meta); t != "" {
|
||||
name = t
|
||||
}
|
||||
if e, err := l.fileStore().Stat(ctx, item.ID); err == nil && e.Title != "" {
|
||||
name = e.Title
|
||||
}
|
||||
var buf bytes.Buffer
|
||||
if _, err := l.Client.DownloadFile(ctx, item.ID, &buf); err != nil {
|
||||
if _, err := l.fileStore().Download(ctx, item.ID, &buf); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return preview.FileBytesToMarkdown(name, buf.Bytes())
|
||||
|
||||
@@ -4,7 +4,6 @@ package fetch_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"testing"
|
||||
|
||||
"github.com/eslider/go-onlyoffice/cmd/office/model"
|
||||
@@ -46,9 +45,6 @@ func TestIntegrationUpdateTaskTitleDescription(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestIntegrationTaskFieldsFromLiveAPI(t *testing.T) {
|
||||
if os.Getenv("ONLYOFFICE_URL") == "" && os.Getenv("ONLYOFFICE_HOST") == "" {
|
||||
t.Skip("ONLYOFFICE_URL not set")
|
||||
}
|
||||
loader, ctx := liveLoader(t)
|
||||
items, err := loader.List(ctx, model.ListSpec{Subject: model.SubjectTasks})
|
||||
if err != nil {
|
||||
@@ -57,12 +53,12 @@ func TestIntegrationTaskFieldsFromLiveAPI(t *testing.T) {
|
||||
if len(items) == 0 {
|
||||
t.Skip("no tasks")
|
||||
}
|
||||
title, desc, err := loader.TaskFields(ctx, items[0])
|
||||
fields, err := loader.DetailForm(ctx, items[0])
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if title == "" {
|
||||
if fields.Primary == "" {
|
||||
t.Fatal("empty title")
|
||||
}
|
||||
_ = desc
|
||||
_ = fields.Secondary
|
||||
}
|
||||
|
||||
@@ -0,0 +1,298 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
"github.com/eslider/go-onlyoffice/catalog"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
var catalogCmd = &cobra.Command{
|
||||
Use: "catalog",
|
||||
Short: "Inventory clients/contacts from disk; match and apply to OO CRM",
|
||||
Long: `Build a YAML catalog of persons/companies from contacts folders and project
|
||||
trees, match against live OnlyOffice CRM, then create/update only rows with
|
||||
approve: true.
|
||||
|
||||
Typical flow:
|
||||
oo catalog scan-contacts --root PATH -O /tmp/contacts.yaml
|
||||
oo catalog scan-projects --root ~/projects -O /tmp/projects.yaml
|
||||
oo catalog scan-thunderbird --root PATH -O /tmp/thunderbird.yaml
|
||||
oo catalog merge -i /tmp/contacts.yaml -i /tmp/projects.yaml -i /tmp/thunderbird.yaml -O docs/catalog/clients-contacts.yaml
|
||||
oo catalog match -i docs/catalog/clients-contacts.yaml
|
||||
# edit approve: true on pilot rows
|
||||
oo catalog apply --dry-run -i docs/catalog/clients-contacts.yaml
|
||||
oo catalog apply --apply -i docs/catalog/clients-contacts.yaml
|
||||
`}
|
||||
|
||||
func init() {
|
||||
rootCmd.AddCommand(catalogCmd)
|
||||
catalogCmd.AddCommand(catalogScanContactsCmd())
|
||||
catalogCmd.AddCommand(catalogScanProjectsCmd())
|
||||
catalogCmd.AddCommand(catalogScanThunderbirdCmd())
|
||||
catalogCmd.AddCommand(catalogMergeCmd())
|
||||
catalogCmd.AddCommand(catalogMatchCmd())
|
||||
catalogCmd.AddCommand(catalogApplyCmd())
|
||||
}
|
||||
|
||||
func catalogScanContactsCmd() *cobra.Command {
|
||||
var outPath string
|
||||
cmd := &cobra.Command{
|
||||
Use: "scan-contacts",
|
||||
Short: "Parse VCF + folder stubs + email txt under a contacts root",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
root, _ := cmd.Flags().GetString("root")
|
||||
if root == "" {
|
||||
return fmt.Errorf("--root is required")
|
||||
}
|
||||
doc, err := catalog.ScanContactsRoot(root)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return writeCatalog(cmd, doc, outPath)
|
||||
},
|
||||
}
|
||||
cmd.Flags().String("root", "", "contacts directory (local path)")
|
||||
cmd.Flags().StringVarP(&outPath, "out", "O", "", "write YAML to this path (default: stdout summary)")
|
||||
_ = cmd.MarkFlagRequired("root")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func catalogScanProjectsCmd() *cobra.Command {
|
||||
var outPath string
|
||||
var maxDepth int
|
||||
cmd := &cobra.Command{
|
||||
Use: "scan-projects",
|
||||
Short: "Git roots / remotes / top-level dirs → company rows",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
root, _ := cmd.Flags().GetString("root")
|
||||
if root == "" {
|
||||
return fmt.Errorf("--root is required")
|
||||
}
|
||||
doc, err := catalog.ScanProjectsRoot(root, maxDepth)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return writeCatalog(cmd, doc, outPath)
|
||||
},
|
||||
}
|
||||
cmd.Flags().String("root", "", "projects directory (local path)")
|
||||
cmd.Flags().IntVar(&maxDepth, "max-depth", 4, "max directory depth for git roots")
|
||||
cmd.Flags().StringVarP(&outPath, "out", "O", "", "write YAML to this path")
|
||||
_ = cmd.MarkFlagRequired("root")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func catalogScanThunderbirdCmd() *cobra.Command {
|
||||
var outPath string
|
||||
var mboxHeaders bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "scan-thunderbird",
|
||||
Short: "Thunderbird profiles: abook/history.mab + Gloda SQLite contacts",
|
||||
Long: `Walk a directory tree for Thunderbird profiles and extract person rows from:
|
||||
- *.mab address books (email regex)
|
||||
- global-messages-db.sqlite Gloda contacts/identities
|
||||
- optional --mbox-headers: From/To/Cc/Reply-To from mbox folder files (no bodies)
|
||||
|
||||
Noisy senders (noreply, Amazon marketplace, GitHub reply, …) are skipped.
|
||||
Default zone is private; known work domains (e.g. wheregroup.com) get zone=warm role=work.`,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
root, _ := cmd.Flags().GetString("root")
|
||||
if root == "" {
|
||||
return fmt.Errorf("--root is required")
|
||||
}
|
||||
doc, err := catalog.ScanThunderbirdRootOpts(root, catalog.ScanOptions{MboxHeaders: mboxHeaders})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return writeCatalog(cmd, doc, outPath)
|
||||
},
|
||||
}
|
||||
cmd.Flags().String("root", "", "Thunderbird profile or parent directory (local path)")
|
||||
cmd.Flags().BoolVar(&mboxHeaders, "mbox-headers", false, "also extract emails from mbox From/To/Cc headers")
|
||||
cmd.Flags().StringVarP(&outPath, "out", "O", "", "write YAML to this path")
|
||||
_ = cmd.MarkFlagRequired("root")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func catalogMergeCmd() *cobra.Command {
|
||||
var inputs []string
|
||||
var outPath string
|
||||
cmd := &cobra.Command{
|
||||
Use: "merge",
|
||||
Short: "Union + normalize catalog YAML files by entry id",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if len(inputs) == 0 {
|
||||
return fmt.Errorf("at least one -i/--input required")
|
||||
}
|
||||
var docs []*catalog.Document
|
||||
for _, p := range inputs {
|
||||
d, err := catalog.LoadYAML(p)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
docs = append(docs, d)
|
||||
}
|
||||
merged := catalog.MergeDocs(docs...)
|
||||
if outPath == "" {
|
||||
return fmt.Errorf("--out is required for merge")
|
||||
}
|
||||
if err := ensureParent(outPath); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := catalog.SaveYAML(outPath, merged); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "merged %d entries → %s\n", len(merged.Entries), outPath)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringArrayVarP(&inputs, "input", "i", nil, "input catalog YAML (repeatable)")
|
||||
cmd.Flags().StringVarP(&outPath, "out", "O", "", "output catalog YAML")
|
||||
_ = cmd.MarkFlagRequired("out")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func catalogMatchCmd() *cobra.Command {
|
||||
var inPath, outPath string
|
||||
cmd := &cobra.Command{
|
||||
Use: "match",
|
||||
Short: "Diff catalog vs live OO; set status new|exists|conflict and oo_id",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if inPath == "" {
|
||||
return fmt.Errorf("--input is required")
|
||||
}
|
||||
doc, err := catalog.LoadYAML(inPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
client, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := catalog.MatchAgainstOO(cmd.Context(), client, doc); err != nil {
|
||||
return err
|
||||
}
|
||||
if outPath == "" {
|
||||
outPath = inPath
|
||||
}
|
||||
if err := catalog.SaveYAML(outPath, doc); err != nil {
|
||||
return err
|
||||
}
|
||||
counts := map[string]int{}
|
||||
for _, e := range doc.Entries {
|
||||
counts[e.Status]++
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "matched → %s new=%d exists=%d conflict=%d\n",
|
||||
outPath, counts["new"], counts["exists"], counts["conflict"])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVarP(&inPath, "input", "i", "", "catalog YAML")
|
||||
cmd.Flags().StringVarP(&outPath, "out", "O", "", "output path (default: overwrite input)")
|
||||
_ = cmd.MarkFlagRequired("input")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func catalogApplyCmd() *cobra.Command {
|
||||
var inPath, outPath string
|
||||
var dryRun, doApply bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "apply",
|
||||
Short: "Create/update OO contacts for approve:true rows",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if !dryRun && !doApply {
|
||||
return fmt.Errorf("pass --dry-run or --apply")
|
||||
}
|
||||
if dryRun && doApply {
|
||||
return fmt.Errorf("use either --dry-run or --apply, not both")
|
||||
}
|
||||
if inPath == "" {
|
||||
return fmt.Errorf("--input is required")
|
||||
}
|
||||
doc, err := catalog.LoadYAML(inPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
client, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
res, err := catalog.ApplyApproved(cmd.Context(), client, doc, dryRun)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if !dryRun {
|
||||
if outPath == "" {
|
||||
outPath = inPath
|
||||
}
|
||||
if err := catalog.SaveYAML(outPath, doc); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if outputFormat == "json" {
|
||||
printJSON(res)
|
||||
return nil
|
||||
}
|
||||
mode := "apply"
|
||||
if dryRun {
|
||||
mode = "dry-run"
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "%s: created=%d updated=%d skipped=%d errors=%d\n",
|
||||
mode, res.Created, res.Updated, res.Skipped, len(res.Errors))
|
||||
for _, e := range res.Errors {
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), " error: %s\n", e)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVarP(&inPath, "input", "i", "", "catalog YAML")
|
||||
cmd.Flags().StringVarP(&outPath, "out", "O", "", "write updated catalog after apply (default: overwrite input)")
|
||||
cmd.Flags().BoolVar(&dryRun, "dry-run", false, "count actions without writing to OO")
|
||||
cmd.Flags().BoolVar(&doApply, "apply", false, "perform OO creates/updates")
|
||||
_ = cmd.MarkFlagRequired("input")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func writeCatalog(cmd *cobra.Command, doc *catalog.Document, outPath string) error {
|
||||
if outPath != "" {
|
||||
if err := ensureParent(outPath); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := catalog.SaveYAML(outPath, doc); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "wrote %d entries → %s\n", len(doc.Entries), outPath)
|
||||
return nil
|
||||
}
|
||||
if outputFormat == "json" {
|
||||
b, err := json.MarshalIndent(doc, "", " ")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintln(cmd.OutOrStdout(), string(b))
|
||||
return nil
|
||||
}
|
||||
persons, companies := 0, 0
|
||||
for _, e := range doc.Entries {
|
||||
if e.Kind == "company" {
|
||||
companies++
|
||||
} else {
|
||||
persons++
|
||||
}
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "entries=%d persons=%d companies=%d (pass -O PATH to write YAML)\n",
|
||||
len(doc.Entries), persons, companies)
|
||||
return nil
|
||||
}
|
||||
|
||||
func ensureParent(path string) error {
|
||||
dir := filepath.Dir(path)
|
||||
if dir == "" || dir == "." {
|
||||
return nil
|
||||
}
|
||||
return os.MkdirAll(dir, 0o755)
|
||||
}
|
||||
+9
-2
@@ -11,7 +11,7 @@ func TestRootRegistersSubjects(t *testing.T) {
|
||||
want := []string{
|
||||
"calendar", "projects", "tasks", "users", "whoami",
|
||||
"contacts", "persons", "companies",
|
||||
"opportunities", "cases", "crm-tasks", "applications", "crm", "mails", "catalog",
|
||||
"opportunities", "cases", "crm-tasks", "crm", "mails", "invoices",
|
||||
}
|
||||
got := make(map[string]bool, len(rootCmd.Commands()))
|
||||
for _, c := range rootCmd.Commands() {
|
||||
@@ -38,7 +38,7 @@ func TestRootHelpListsSubjects(t *testing.T) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
help := out.String()
|
||||
for _, snippet := range []string{"calendar", "projects", "tasks", "users", "opportunities"} {
|
||||
for _, snippet := range []string{"calendar", "projects", "tasks", "users", "opportunities", "invoices"} {
|
||||
if !strings.Contains(help, snippet) {
|
||||
t.Fatalf("help missing %q", snippet)
|
||||
}
|
||||
@@ -53,6 +53,13 @@ func TestProjectsAlias(t *testing.T) {
|
||||
if cmd.Name() != "projects" {
|
||||
t.Fatalf("prj alias resolved to %q", cmd.Name())
|
||||
}
|
||||
ms, _, err := rootCmd.Find([]string{"projects", "milestone-create"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if ms.Name() != "milestone-create" {
|
||||
t.Fatalf("milestone-create resolved to %q", ms.Name())
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewOOReturnsErrorWithoutCredentials(t *testing.T) {
|
||||
|
||||
+1
-1
@@ -20,7 +20,7 @@ var outputFormat = "table"
|
||||
var rootCmd = &cobra.Command{
|
||||
Use: "oo",
|
||||
Short: "OnlyOffice Workspace CLI — subject-based command tree",
|
||||
Long: "oo is a thin CLI over github.com/eslider/go-onlyoffice.\nCommands are grouped by OnlyOffice subject (calendar, projects, tasks, users, persons, companies, opportunities, cases, crm-tasks, applications, catalog).",
|
||||
Long: "oo is a thin CLI over github.com/eslider/go-onlyoffice.\nCommands are grouped by OnlyOffice subject (calendar, projects, tasks, users, persons, companies, opportunities, cases, crm-tasks, mails, invoices).",
|
||||
Version: version,
|
||||
SilenceUsage: true,
|
||||
SilenceErrors: false,
|
||||
|
||||
+197
-5
@@ -6,7 +6,6 @@ import (
|
||||
"strings"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/eslider/go-onlyoffice/catalog"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
@@ -44,10 +43,16 @@ func init() {
|
||||
contactsCmd.AddCommand(contactsMergeCmd())
|
||||
contactsCmd.AddCommand(contactsInfoAddCmd())
|
||||
contactsCmd.AddCommand(contactsDedupeInfoCmd())
|
||||
contactsCmd.AddCommand(contactsTagsCmd())
|
||||
contactsCmd.AddCommand(contactsTagCreateCmd())
|
||||
contactsCmd.AddCommand(contactsTagAddCmd())
|
||||
contactsCmd.AddCommand(contactsTagRemoveCmd())
|
||||
contactsCmd.AddCommand(contactsByTagCmd())
|
||||
|
||||
only := true
|
||||
personsCmd.AddCommand(contactsListCmd(&only)) // persons only
|
||||
personsCmd.AddCommand(personsCreateCmd())
|
||||
personsCmd.AddCommand(personsUpdateCmd())
|
||||
personsCmd.AddCommand(personsFixNamesCmd())
|
||||
personsCmd.AddCommand(contactsDeleteCmd())
|
||||
personsCmd.AddCommand(personsDedupeCmd())
|
||||
@@ -201,7 +206,7 @@ func contactsInfoAddCmd() *cobra.Command {
|
||||
}
|
||||
|
||||
func personsCreateCmd() *cobra.Command {
|
||||
var first, last, email, linkedin string
|
||||
var first, last, email, linkedin, phone string
|
||||
var companyID int
|
||||
var jobTitle, about string
|
||||
cmd := &cobra.Command{
|
||||
@@ -224,6 +229,9 @@ func personsCreateCmd() *cobra.Command {
|
||||
if email != "" {
|
||||
_, _ = c.AddContactInfo(cmd.Context(), pid, "Email", email, "Work", true)
|
||||
}
|
||||
if phone != "" {
|
||||
_, _ = c.AddContactInfo(cmd.Context(), pid, "Phone", phone, "Work", true)
|
||||
}
|
||||
if linkedin != "" {
|
||||
_, _ = c.AddContactInfo(cmd.Context(), pid, "LinkedIn", linkedin, "Work", false)
|
||||
}
|
||||
@@ -237,10 +245,62 @@ func personsCreateCmd() *cobra.Command {
|
||||
cmd.Flags().StringVar(&jobTitle, "job-title", "", "")
|
||||
cmd.Flags().StringVar(&about, "about", "", "about / bio")
|
||||
cmd.Flags().StringVar(&email, "email", "", "primary email (adds ContactInfo)")
|
||||
cmd.Flags().StringVar(&phone, "phone", "", "primary phone (adds ContactInfo)")
|
||||
cmd.Flags().StringVar(&linkedin, "linkedin", "", "linkedin url (adds ContactInfo)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func personsUpdateCmd() *cobra.Command {
|
||||
var first, last string
|
||||
var companyID int
|
||||
var jobTitle, about string
|
||||
cmd := &cobra.Command{
|
||||
Use: "update PERSON_ID",
|
||||
Short: "Update a person (name, company, job title, about)",
|
||||
Long: `Updates CRM person fields via PUT /crm/contact/person/{id}.
|
||||
|
||||
--company-id 0 leaves the employer association unchanged.
|
||||
Omit --first/--last to keep current names (fetched first).`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
pid := args[0]
|
||||
cur, err := c.GetContact(cmd.Context(), pid)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if first == "" {
|
||||
if v, ok := cur["firstName"].(string); ok {
|
||||
first = v
|
||||
}
|
||||
}
|
||||
if last == "" {
|
||||
if v, ok := cur["lastName"].(string); ok {
|
||||
last = v
|
||||
}
|
||||
}
|
||||
if first == "" || last == "" {
|
||||
return fmt.Errorf("first and last name required (pass --first/--last or ensure contact has them)")
|
||||
}
|
||||
out, err := c.UpdatePerson(cmd.Context(), pid, first, last, companyID, jobTitle, about)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&first, "first", "", "first name (default: current)")
|
||||
cmd.Flags().StringVar(&last, "last", "", "last name (default: current)")
|
||||
cmd.Flags().IntVar(&companyID, "company-id", 0, "employer company id (0 = leave unchanged)")
|
||||
cmd.Flags().StringVar(&jobTitle, "job-title", "", "job title")
|
||||
cmd.Flags().StringVar(&about, "about", "", "about / bio")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func personsFixNamesCmd() *cobra.Command {
|
||||
var companyID int
|
||||
var dryRun bool
|
||||
@@ -248,7 +308,7 @@ func personsFixNamesCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "fix-names",
|
||||
Short: "Strip company annotations from last names; keep companyId link",
|
||||
Long: `Repairs CRM persons whose lastName embeds a company ("Thomsen (Acme)")
|
||||
Long: `Repairs CRM persons whose lastName embeds a company ("Doe (Acme)")
|
||||
or whose firstName is an email. Company belongs on companyId, not in the name.
|
||||
|
||||
With --company-id, only persons linked to that company are scanned.`,
|
||||
@@ -300,7 +360,7 @@ With --company-id, only persons linked to that company are scanned.`,
|
||||
}
|
||||
}
|
||||
}
|
||||
cf, cl := catalog.CleanPersonNames(fn, ln, dn, org, emails)
|
||||
cf, cl := onlyoffice.CleanPersonNames(fn, ln, dn, org, emails)
|
||||
needName := cf != fn || cl != ln
|
||||
linkID := companyID
|
||||
if linkID == 0 {
|
||||
@@ -347,7 +407,7 @@ func contactEmailsFromMap(p map[string]any) []string {
|
||||
}
|
||||
|
||||
func companiesCreateCmd() *cobra.Command {
|
||||
var name, email, website string
|
||||
var name, email, website, phone, about, street, city, state, zip, country string
|
||||
cmd := &cobra.Command{
|
||||
Use: "create",
|
||||
Aliases: []string{"add"},
|
||||
@@ -365,12 +425,34 @@ func companiesCreateCmd() *cobra.Command {
|
||||
return err
|
||||
}
|
||||
cid := strconv.Itoa(int(flexIDFloat(out["id"])))
|
||||
if about != "" || street != "" {
|
||||
aboutText := about
|
||||
if street != "" {
|
||||
addr := strings.TrimSpace(strings.Join([]string{street, zip, city, state, country}, ", "))
|
||||
if aboutText != "" {
|
||||
aboutText = aboutText + "\n" + addr
|
||||
} else {
|
||||
aboutText = "Billing address: " + addr
|
||||
}
|
||||
}
|
||||
if updated, err := c.UpdateCompany(cmd.Context(), cid, name, aboutText); err == nil {
|
||||
out = updated
|
||||
}
|
||||
}
|
||||
if email != "" {
|
||||
_, _ = c.AddContactInfo(cmd.Context(), cid, "Email", email, "Work", true)
|
||||
}
|
||||
if website != "" {
|
||||
_, _ = c.AddContactInfo(cmd.Context(), cid, "Website", website, "Work", false)
|
||||
}
|
||||
if phone != "" {
|
||||
_, _ = c.AddContactInfo(cmd.Context(), cid, "Phone", phone, "Work", true)
|
||||
}
|
||||
if street != "" {
|
||||
if _, err := c.AddContactAddress(cmd.Context(), cid, street, city, state, zip, country, "Billing", true); err != nil {
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "warning: address API failed (stored in about): %v\n", err)
|
||||
}
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
@@ -378,6 +460,13 @@ func companiesCreateCmd() *cobra.Command {
|
||||
cmd.Flags().StringVar(&name, "name", "", "company name")
|
||||
cmd.Flags().StringVar(&email, "email", "", "primary email (adds ContactInfo)")
|
||||
cmd.Flags().StringVar(&website, "website", "", "website url (adds ContactInfo)")
|
||||
cmd.Flags().StringVar(&phone, "phone", "", "primary phone (adds ContactInfo)")
|
||||
cmd.Flags().StringVar(&about, "about", "", "about / notes")
|
||||
cmd.Flags().StringVar(&street, "street", "", "billing street")
|
||||
cmd.Flags().StringVar(&city, "city", "", "billing city")
|
||||
cmd.Flags().StringVar(&state, "state", "", "billing state")
|
||||
cmd.Flags().StringVar(&zip, "zip", "", "billing zip")
|
||||
cmd.Flags().StringVar(&country, "country", "", "billing country")
|
||||
return cmd
|
||||
}
|
||||
|
||||
@@ -440,3 +529,106 @@ func contactsDedupeInfoCmd() *cobra.Command {
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
func contactsTagsCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "tags",
|
||||
Short: "List CRM contact tags",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
list, err := c.ListContactTags(cmd.Context())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printTable([]string{"title", "relativeItemsCount"}, list)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func contactsTagCreateCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "tag-create TAG",
|
||||
Short: "Create a CRM contact tag",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := c.CreateContactTag(cmd.Context(), args[0]); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Println(args[0])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func contactsTagAddCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "tag-add CONTACT_ID TAG",
|
||||
Short: "Attach a tag to a contact",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := c.AddContactTag(cmd.Context(), args[0], args[1]); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Printf("%s ← %s\n", args[0], args[1])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func contactsTagRemoveCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "tag-remove CONTACT_ID TAG",
|
||||
Short: "Remove a tag from a contact",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := c.RemoveContactTag(cmd.Context(), args[0], args[1]); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Printf("%s ✕ %s\n", args[0], args[1])
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func contactsByTagCmd() *cobra.Command {
|
||||
var count, offset int
|
||||
cmd := &cobra.Command{
|
||||
Use: "by-tag TAG",
|
||||
Short: "List contacts with a given CRM tag (ignore-list = tag ignore)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
list, total, err := c.ListContactsByTag(cmd.Context(), args[0], count, offset)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if outputFormat == "table" {
|
||||
fmt.Printf("tag=%s total: %d (shown: %d)\n", args[0], total, len(list))
|
||||
}
|
||||
printTable([]string{"id", "displayName", "isCompany", "title", "about"}, list)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().IntVar(&count, "count", 50, "")
|
||||
cmd.Flags().IntVar(&offset, "offset", 0, "")
|
||||
return cmd
|
||||
}
|
||||
|
||||
+322
@@ -0,0 +1,322 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
func init() {
|
||||
rootCmd.AddCommand(davCmd())
|
||||
}
|
||||
|
||||
// davCmd exposes the Documents module through the backend-agnostic FileStore
|
||||
// (DAV backend). The underlying Dav calls are the oo-webdav proven path:
|
||||
// MoveDavItems sends resolveType=Skip + holdResult=true, which the legacy
|
||||
// fileops/move call without those params silently ignores (200 without move).
|
||||
func davCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "dav",
|
||||
Short: "Documents module by folder/file id (oo-webdav proven path)",
|
||||
}
|
||||
cmd.AddCommand(davLsCmd())
|
||||
cmd.AddCommand(davMoveCmd())
|
||||
cmd.AddCommand(davCopyCmd())
|
||||
cmd.AddCommand(davMkdirCmd())
|
||||
cmd.AddCommand(davRemoveCmd())
|
||||
cmd.AddCommand(davRenameFileCmd())
|
||||
cmd.AddCommand(davRenameFolderCmd())
|
||||
cmd.AddCommand(davDownloadCmd())
|
||||
cmd.AddCommand(davFileOpsCmd())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func davLsCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "ls FOLDER_ID",
|
||||
Short: "List a Documents folder (@root for virtual sections)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
if args[0] == "@root" {
|
||||
sections, err := c.ListDavSections(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(sections))
|
||||
for _, s := range sections {
|
||||
rows = append(rows, map[string]any{
|
||||
"id": s.ID,
|
||||
"title": s.Title,
|
||||
"filesCount": s.FilesCount,
|
||||
"foldersCount": s.FoldersCount,
|
||||
})
|
||||
}
|
||||
printTable([]string{"id", "title", "filesCount", "foldersCount"}, rows)
|
||||
return nil
|
||||
}
|
||||
entries, err := c.FileStore(onlyoffice.ProviderDAV).List(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
folders := make([]onlyoffice.Entry, 0, len(entries))
|
||||
files := make([]onlyoffice.Entry, 0, len(entries))
|
||||
for _, e := range entries {
|
||||
if e.Kind == onlyoffice.Folder {
|
||||
folders = append(folders, e)
|
||||
} else {
|
||||
files = append(files, e)
|
||||
}
|
||||
}
|
||||
frows := make([]map[string]any, 0, len(folders))
|
||||
for _, f := range folders {
|
||||
frows = append(frows, map[string]any{
|
||||
"id": f.ID,
|
||||
"title": f.Title,
|
||||
"filesCount": f.FilesCount,
|
||||
"foldersCount": f.FoldersCount,
|
||||
})
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(files))
|
||||
for _, f := range files {
|
||||
rows = append(rows, map[string]any{
|
||||
"id": f.ID,
|
||||
"title": f.Title,
|
||||
"size": f.Size,
|
||||
"updated": entryUpdated(f),
|
||||
})
|
||||
}
|
||||
if outputFormat == "json" {
|
||||
printObject(map[string]any{"folders": frows, "files": rows})
|
||||
return nil
|
||||
}
|
||||
if len(frows) > 0 {
|
||||
fmt.Println("folders:")
|
||||
printTable([]string{"id", "title", "filesCount", "foldersCount"}, frows)
|
||||
}
|
||||
fmt.Println("files:")
|
||||
printTable([]string{"id", "title", "size", "updated"}, rows)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
return cmd
|
||||
}
|
||||
|
||||
// entryUpdated prefers the backend-native timestamp string so table/JSON output
|
||||
// round-trips what the API returned.
|
||||
func entryUpdated(e onlyoffice.Entry) string {
|
||||
if e.Updated != "" {
|
||||
return e.Updated
|
||||
}
|
||||
if e.Modified.IsZero() {
|
||||
return ""
|
||||
}
|
||||
return e.Modified.Format(time.RFC3339)
|
||||
}
|
||||
|
||||
func davMoveCmd() *cobra.Command {
|
||||
var folderIDs []string
|
||||
cmd := &cobra.Command{
|
||||
Use: "move DEST_FOLDER_ID FILE_ID [FILE_ID...]",
|
||||
Short: "Move file(s) into a Documents folder (resolveType=Skip, holdResult)",
|
||||
Args: cobra.MinimumNArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ids := append(append([]string{}, folderIDs...), args[1:]...)
|
||||
if err := c.FileStore(onlyoffice.ProviderDAV).Move(cmd.Context(), ids, args[0]); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"moved_files": args[1:], "moved_folders": folderIDs, "dest": args[0]})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringSliceVar(&folderIDs, "folders", nil, "folder ids to move along with the files")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func davCopyCmd() *cobra.Command {
|
||||
var folderIDs []string
|
||||
cmd := &cobra.Command{
|
||||
Use: "copy DEST_FOLDER_ID FILE_ID [FILE_ID...]",
|
||||
Short: "Copy file(s) into a Documents folder (conflictResolveType=Skip)",
|
||||
Args: cobra.MinimumNArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ids := append(append([]string{}, folderIDs...), args[1:]...)
|
||||
if err := c.FileStore(onlyoffice.ProviderDAV).Copy(cmd.Context(), ids, args[0]); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"copied_files": args[1:], "copied_folders": folderIDs, "dest": args[0]})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringSliceVar(&folderIDs, "folders", nil, "folder ids to copy along with the files")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func davMkdirCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "mkdir PARENT_FOLDER_ID TITLE",
|
||||
Short: "Create a subfolder in Documents",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
f, err := c.FileStore(onlyoffice.ProviderDAV).CreateFolder(cmd.Context(), args[0], args[1])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"id": f.ID, "title": f.Title, "parent": args[0]})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func davRemoveCmd() *cobra.Command {
|
||||
var folderIDs []string
|
||||
cmd := &cobra.Command{
|
||||
Use: "rm [FILE_ID...]",
|
||||
Aliases: []string{"delete"},
|
||||
Short: "Permanently delete file(s) and/or folder(s) from Documents",
|
||||
Args: cobra.ArbitraryArgs,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if len(args) == 0 && len(folderIDs) == 0 {
|
||||
return fmt.Errorf("dav rm: give at least one FILE_ID or --folders")
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ids := append(append([]string{}, args...), folderIDs...)
|
||||
if err := c.FileStore(onlyoffice.ProviderDAV).Delete(cmd.Context(), ids); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"deleted_files": args, "deleted_folders": folderIDs})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringSliceVar(&folderIDs, "folders", nil, "folder ids to delete")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func davRenameFileCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "rename-file FILE_ID NEW_TITLE",
|
||||
Short: "Rename a Documents file (include extension in NEW_TITLE)",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := c.FileStore(onlyoffice.ProviderDAV).Rename(cmd.Context(), args[0], args[1]); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"id": args[0], "title": args[1]})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func davRenameFolderCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "rename-folder FOLDER_ID NEW_TITLE",
|
||||
Short: "Rename a Documents folder",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := c.FileStore(onlyoffice.ProviderDAV).Rename(cmd.Context(), args[0], args[1]); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"id": args[0], "title": args[1]})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func davDownloadCmd() *cobra.Command {
|
||||
var to string
|
||||
cmd := &cobra.Command{
|
||||
Use: "download FILE_ID",
|
||||
Short: "Download Documents file bytes (default path: ./<title>)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
store := c.FileStore(onlyoffice.ProviderDAV)
|
||||
e, err := store.Stat(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
path := to
|
||||
if path == "" {
|
||||
path = onlyoffice.SafeLocalFileName(e.Title)
|
||||
}
|
||||
out, err := os.Create(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer out.Close()
|
||||
n, err := store.Download(ctx, args[0], out)
|
||||
if err != nil {
|
||||
_ = os.Remove(path)
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"path": path, "bytes": n})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&to, "to", "", "output path (default: ./<server title>)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func davFileOpsCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "fileops",
|
||||
Short: "List active file operations (move/copy status polling)",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ops, err := c.ListFileOps(cmd.Context())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(ops))
|
||||
for _, op := range ops {
|
||||
rows = append(rows, map[string]any{
|
||||
"id": fmt.Sprint(op["id"]),
|
||||
"operation": fmt.Sprint(op["operation"]),
|
||||
"progress": fmt.Sprint(op["progress"]),
|
||||
"finished": fmt.Sprint(op["finished"]),
|
||||
"error": fmt.Sprint(op["error"]),
|
||||
})
|
||||
}
|
||||
printTable([]string{"id", "operation", "progress", "finished", "error"}, rows)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
+634
@@ -0,0 +1,634 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/eslider/go-onlyoffice/internal/docpipe"
|
||||
"github.com/eslider/go-onlyoffice/internal/xlspipe"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
func init() {
|
||||
rootCmd.AddCommand(docsCmd())
|
||||
}
|
||||
|
||||
func docsCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "docs",
|
||||
Short: "Local document pipeline: md↔docx, OCR→PDF, extract Markdown",
|
||||
Long: `Agent-friendly conversions (requires pandoc / ocrmypdf / pdftotext on PATH).
|
||||
|
||||
OnlyOffice Documents UI is poor for .md/.txt — keep sources in git, store .docx in OO.
|
||||
Upload Markdown as DOCX: oo docs put-md PROJECT_ID file.md
|
||||
Upload plain text: oo docs put-txt PROJECT_ID file.txt (preserves line breaks)
|
||||
Upload/generate XLSX: oo docs put-xlsx PROJECT_ID [--template cutover-portugal | FILE.xlsx]
|
||||
Read an OO file as MD: oo docs as-md FILE_ID
|
||||
OCR a scan locally: oo docs ocr scan.pdf --md out.md
|
||||
Structured OCR (hOCR→MD): oo docs hocr scan.jpg --md out.md --yaml out.yml`,
|
||||
}
|
||||
cmd.AddCommand(docsConvertCmd())
|
||||
cmd.AddCommand(docsOptimizeCmd())
|
||||
cmd.AddCommand(docsOCRCmd())
|
||||
cmd.AddCommand(docsHOCRCmd())
|
||||
cmd.AddCommand(docsAsMDCmd())
|
||||
cmd.AddCommand(docsPutMDCmd())
|
||||
cmd.AddCommand(docsPutTxtCmd())
|
||||
cmd.AddCommand(docsPutXlsxCmd())
|
||||
cmd.AddCommand(docsToolsCmd())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsToolsCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "tools",
|
||||
Short: "Show which converter binaries are on PATH",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
t := docpipe.LookPath()
|
||||
printObject(map[string]any{
|
||||
"pandoc": strOrNil(t.Pandoc),
|
||||
"ocrmypdf": strOrNil(t.OCRMyPDF),
|
||||
"pdftotext": strOrNil(t.PDFToText),
|
||||
"tesseract": strOrNil(t.Tesseract),
|
||||
"ghostscript": strOrNil(t.Ghostscript),
|
||||
})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func strOrNil(s string) any {
|
||||
if s == "" {
|
||||
return nil
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func docsConvertCmd() *cobra.Command {
|
||||
var to string
|
||||
cmd := &cobra.Command{
|
||||
Use: "convert PATH",
|
||||
Short: "Convert a local file with pandoc (md↔docx by default)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
in := args[0]
|
||||
t := docpipe.LookPath()
|
||||
out := to
|
||||
if out == "" {
|
||||
switch docpipe.Ext(in) {
|
||||
case ".md", ".markdown":
|
||||
out = docpipe.SiblingDOCX(in)
|
||||
case ".docx":
|
||||
out = strings.TrimSuffix(in, docpipe.Ext(in)) + ".md"
|
||||
default:
|
||||
return fmt.Errorf("--to required for input type %s", docpipe.Ext(in))
|
||||
}
|
||||
}
|
||||
if err := docpipe.EnsureDir(out); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := t.ConvertFile(in, out); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"in": in, "out": out})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&to, "to", "", "output path (default: sibling .docx or .md)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsOptimizeCmd() *cobra.Command {
|
||||
var out string
|
||||
cmd := &cobra.Command{
|
||||
Use: "optimize PDF_PATH",
|
||||
Short: "Rewrite PDF via Ghostscript (PostScript pdfwrite) for OO-friendly size/text",
|
||||
Long: `Use for InDesign/iText PDFs with a good text layer — avoids ocrmypdf invisible
|
||||
text overlays that break OnlyOffice DocEditor. Skips OCR; rewrites via gs pdfwrite.`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
in := args[0]
|
||||
t := docpipe.LookPath()
|
||||
if out == "" {
|
||||
base := strings.TrimSuffix(filepath.Base(in), filepath.Ext(in))
|
||||
out = filepath.Join(filepath.Dir(in), base+".optimized.pdf")
|
||||
}
|
||||
if err := docpipe.EnsureDir(out); err != nil {
|
||||
return err
|
||||
}
|
||||
chars, _ := t.PDFTextLayerChars(in)
|
||||
if err := t.OptimizePDF(in, out); err != nil {
|
||||
return err
|
||||
}
|
||||
outChars, _ := t.PDFTextLayerChars(out)
|
||||
printObject(map[string]any{
|
||||
"in": in,
|
||||
"pdf": out,
|
||||
"text_chars_in": chars,
|
||||
"text_chars_out": outChars,
|
||||
"note": "native text layer preserved; no OCR overlay",
|
||||
})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&out, "out", "", "output PDF (default: <name>.optimized.pdf)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsOCRCmd() *cobra.Command {
|
||||
var out, mdOut, lang string
|
||||
var force bool
|
||||
var writeMD bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "ocr PATH",
|
||||
Short: "OCR image/PDF → searchable PDF (and optional Markdown)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
in := args[0]
|
||||
t := docpipe.LookPath()
|
||||
if out == "" {
|
||||
base := strings.TrimSuffix(filepath.Base(in), filepath.Ext(in))
|
||||
out = filepath.Join(filepath.Dir(in), base+".ocr.pdf")
|
||||
}
|
||||
if err := docpipe.EnsureDir(out); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := t.OCRToPDF(in, out, force, lang); err != nil {
|
||||
return err
|
||||
}
|
||||
res := map[string]any{"in": in, "pdf": out}
|
||||
if writeMD || mdOut != "" {
|
||||
if mdOut == "" {
|
||||
mdOut = strings.TrimSuffix(out, filepath.Ext(out)) + ".md"
|
||||
}
|
||||
text, err := t.ExtractPDFText(out)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
body := "# " + filepath.Base(in) + "\n\n" + strings.TrimSpace(text) + "\n"
|
||||
if err := os.WriteFile(mdOut, []byte(body), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
res["md"] = mdOut
|
||||
}
|
||||
printObject(res)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&out, "out", "", "output searchable PDF (default: <name>.ocr.pdf)")
|
||||
cmd.Flags().StringVar(&mdOut, "md", "", "write Markdown extraction to this path")
|
||||
cmd.Flags().BoolVar(&writeMD, "markdown", false, "also write sibling .md next to OCR PDF")
|
||||
cmd.Flags().StringVar(&lang, "lang", "eng", "OCR language(s) for tesseract/ocrmypdf")
|
||||
cmd.Flags().BoolVar(&force, "force", false, "force OCR even if a text layer exists (default: skip when pdftotext finds enough text)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsHOCRCmd() *cobra.Command {
|
||||
var mdOut, yamlOut, hocrOut, lang string
|
||||
var dpi int
|
||||
var minConf float32
|
||||
cmd := &cobra.Command{
|
||||
Use: "hocr PATH",
|
||||
Short: "Tesseract hOCR → structured Markdown/YAML (via go-hocr)",
|
||||
Long: `Runs tesseract with hOCR output, parses with go-hocr, writes Markdown
|
||||
(and optional YAML). Better reading order / confidence than plain pdftotext.
|
||||
|
||||
For Spanish scans: --lang spa or spa+eng (needs tesseract-ocr-spa / TESSDATA_PREFIX).
|
||||
Phone photos: --dpi 200..300.`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
in := args[0]
|
||||
t := docpipe.LookPath()
|
||||
dir, err := os.MkdirTemp("", "oo-docs-hocr-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
res, err := t.ToHOCRMarkdown(in, dir, lang, dpi, minConf)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if mdOut == "" {
|
||||
base := strings.TrimSuffix(filepath.Base(in), filepath.Ext(in))
|
||||
mdOut = filepath.Join(filepath.Dir(in), base+".hocr.md")
|
||||
}
|
||||
if err := docpipe.EnsureDir(mdOut); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(mdOut, []byte(res.Markdown), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
obj := map[string]any{
|
||||
"in": in,
|
||||
"md": mdOut,
|
||||
"did_ocr": res.DidOCR,
|
||||
}
|
||||
if hocrOut != "" {
|
||||
if err := docpipe.EnsureDir(hocrOut); err != nil {
|
||||
return err
|
||||
}
|
||||
b, err := os.ReadFile(res.HOCRPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(hocrOut, b, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
obj["hocr"] = hocrOut
|
||||
} else {
|
||||
obj["hocr_tmp"] = res.HOCRPath
|
||||
}
|
||||
if yamlOut != "" {
|
||||
if err := docpipe.EnsureDir(yamlOut); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(yamlOut, []byte(res.YAML), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
obj["yaml"] = yamlOut
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&mdOut, "md", "", "Markdown output (default: <name>.hocr.md)")
|
||||
cmd.Flags().StringVar(&yamlOut, "yaml", "", "also write structured YAML from go-hocr")
|
||||
cmd.Flags().StringVar(&hocrOut, "hocr", "", "also keep raw .hocr file at this path")
|
||||
cmd.Flags().StringVar(&lang, "lang", "eng", "tesseract language(s), e.g. spa+eng")
|
||||
cmd.Flags().IntVar(&dpi, "dpi", 220, "hint DPI for phone photos / scans (0 = tesseract default)")
|
||||
cmd.Flags().Float32Var(&minConf, "min-conf", 0, "drop words with OCR confidence below this (0 = keep all)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsAsMDCmd() *cobra.Command {
|
||||
var to, lang string
|
||||
var minChars int
|
||||
var uploadOCR bool
|
||||
var useHOCR bool
|
||||
var dpi int
|
||||
var minConf float32
|
||||
cmd := &cobra.Command{
|
||||
Use: "as-md FILE_ID",
|
||||
Short: "Download an OO Documents file and emit Markdown (OCR PDF/image if needed)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
meta, err := c.GetFile(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
title := onlyoffice.FileEntryTitle(meta)
|
||||
dir, err := os.MkdirTemp("", "oo-docs-as-md-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
local := filepath.Join(dir, onlyoffice.SafeLocalFileName(title))
|
||||
f, err := os.Create(local)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := c.DownloadFile(ctx, args[0], f); err != nil {
|
||||
_ = f.Close()
|
||||
return err
|
||||
}
|
||||
_ = f.Close()
|
||||
|
||||
tools := docpipe.LookPath()
|
||||
var res docpipe.Result
|
||||
if useHOCR {
|
||||
hr, err := tools.ToHOCRMarkdown(local, dir, lang, dpi, minConf)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
res = docpipe.Result{Markdown: hr.Markdown, DidOCR: hr.DidOCR, Source: hr.Source}
|
||||
} else {
|
||||
res, err = tools.ToMarkdown(local, dir, lang, minChars)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
outPath := to
|
||||
if outPath == "" {
|
||||
base := strings.TrimSuffix(onlyoffice.SafeLocalFileName(title), filepath.Ext(onlyoffice.SafeLocalFileName(title)))
|
||||
if base == "" || base == "download" {
|
||||
base = "file-" + args[0]
|
||||
}
|
||||
outPath = base + ".md"
|
||||
}
|
||||
if err := docpipe.EnsureDir(outPath); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(outPath, []byte(res.Markdown), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
obj := map[string]any{
|
||||
"file_id": args[0],
|
||||
"title": title,
|
||||
"md": outPath,
|
||||
"did_ocr": res.DidOCR,
|
||||
}
|
||||
if res.OCRPDFPath != "" && uploadOCR {
|
||||
folderID := onlyoffice.FileFolderID(meta)
|
||||
upName := strings.TrimSuffix(onlyoffice.SafeLocalFileName(title), filepath.Ext(onlyoffice.SafeLocalFileName(title))) + ".ocr.pdf"
|
||||
tmpUp := filepath.Join(dir, upName)
|
||||
data, err := os.ReadFile(res.OCRPDFPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(tmpUp, data, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
if folderID == "" {
|
||||
obj["ocr_pdf_local"] = res.OCRPDFPath
|
||||
obj["note"] = "file has no folderId; OCR PDF left local — pass after moving into a folder"
|
||||
} else {
|
||||
ent, err := c.UploadToFolder(ctx, folderID, tmpUp)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
obj["ocr_pdf_file_id"] = fileIDStr(ent)
|
||||
obj["ocr_pdf_title"] = onlyoffice.FileEntryTitle(ent)
|
||||
}
|
||||
} else if res.OCRPDFPath != "" {
|
||||
// Keep OCR PDF outside temp by copying beside md if requested via env-less default:
|
||||
kept := strings.TrimSuffix(outPath, filepath.Ext(outPath)) + ".ocr.pdf"
|
||||
if b, err := os.ReadFile(res.OCRPDFPath); err == nil {
|
||||
_ = os.WriteFile(kept, b, 0o644)
|
||||
obj["ocr_pdf_local"] = kept
|
||||
} else {
|
||||
obj["ocr_pdf_local"] = res.OCRPDFPath
|
||||
}
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&to, "to", "", "write Markdown to this path (default: ./<title>.md)")
|
||||
cmd.Flags().StringVar(&lang, "lang", "eng", "OCR language")
|
||||
cmd.Flags().IntVar(&minChars, "min-chars", docpipe.DefaultMinTextChars, "OCR PDF if text layer shorter than this")
|
||||
cmd.Flags().BoolVar(&uploadOCR, "upload-ocr", false, "upload searchable OCR PDF back into the same OO folder")
|
||||
cmd.Flags().BoolVar(&useHOCR, "hocr", false, "use tesseract hOCR + go-hocr instead of ocrmypdf/pdftotext")
|
||||
cmd.Flags().IntVar(&dpi, "dpi", 220, "DPI hint when --hocr (phone photos)")
|
||||
cmd.Flags().Float32Var(&minConf, "min-conf", 0, "drop low-confidence words when --hocr")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsPutMDCmd() *cobra.Command {
|
||||
var folderID string
|
||||
var keepLocalDOCX string
|
||||
var replace bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "put-md PROJECT_ID MARKDOWN_PATH",
|
||||
Short: "Convert Markdown→DOCX and upload DOCX into a project (OO-friendly)",
|
||||
Long: `Agents edit .md locally; this uploads .docx so OnlyOffice can open/version it.`,
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
pid, mdPath := args[0], args[1]
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
tools := docpipe.LookPath()
|
||||
dir, err := os.MkdirTemp("", "oo-docs-put-md-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
docxName := strings.TrimSuffix(filepath.Base(mdPath), filepath.Ext(mdPath)) + ".docx"
|
||||
docxPath := filepath.Join(dir, docxName)
|
||||
if err := tools.MDToDOCX(mdPath, docxPath); err != nil {
|
||||
return err
|
||||
}
|
||||
if keepLocalDOCX != "" {
|
||||
if err := docpipe.EnsureDir(keepLocalDOCX); err != nil {
|
||||
return err
|
||||
}
|
||||
b, err := os.ReadFile(docxPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(keepLocalDOCX, b, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
ent, deleted, err := uploadProjectDoc(ctx, c, pid, docxPath, folderID, replace)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
obj := map[string]any{
|
||||
"project_id": pid,
|
||||
"md": mdPath,
|
||||
"uploaded": fileEntryToMap(ent),
|
||||
}
|
||||
if folderID != "" {
|
||||
obj["folder_id"] = folderID
|
||||
}
|
||||
if len(deleted) > 0 {
|
||||
obj["replaced_file_ids"] = deleted
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&folderID, "folder", "", "Documents folder id (default: project root)")
|
||||
cmd.Flags().StringVar(&keepLocalDOCX, "keep-docx", "", "also write the generated DOCX to this local path")
|
||||
cmd.Flags().BoolVar(&replace, "replace", true, "replace same stem|ext before upload (default); false = fail if name taken")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsPutTxtCmd() *cobra.Command {
|
||||
var folderID string
|
||||
var keepLocalDOCX string
|
||||
var replace bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "put-txt PROJECT_ID TEXT_PATH",
|
||||
Short: "Convert plain text→DOCX (preserve line breaks) and upload into a project",
|
||||
Long: `OnlyOffice cannot render .txt well. This keeps each source line on its own DOCX line.`,
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
pid, txtPath := args[0], args[1]
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
tools := docpipe.LookPath()
|
||||
dir, err := os.MkdirTemp("", "oo-docs-put-txt-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
docxName := strings.TrimSuffix(filepath.Base(txtPath), filepath.Ext(txtPath)) + ".docx"
|
||||
docxPath := filepath.Join(dir, docxName)
|
||||
if err := tools.TXTToDOCX(txtPath, docxPath); err != nil {
|
||||
return err
|
||||
}
|
||||
if keepLocalDOCX != "" {
|
||||
if err := docpipe.EnsureDir(keepLocalDOCX); err != nil {
|
||||
return err
|
||||
}
|
||||
b, err := os.ReadFile(docxPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(keepLocalDOCX, b, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
ent, deleted, err := uploadProjectDoc(ctx, c, pid, docxPath, folderID, replace)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
obj := map[string]any{
|
||||
"project_id": pid,
|
||||
"txt": txtPath,
|
||||
"uploaded": fileEntryToMap(ent),
|
||||
}
|
||||
if folderID != "" {
|
||||
obj["folder_id"] = folderID
|
||||
}
|
||||
if len(deleted) > 0 {
|
||||
obj["replaced_file_ids"] = deleted
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&folderID, "folder", "", "Documents folder id (default: project root)")
|
||||
cmd.Flags().StringVar(&keepLocalDOCX, "keep-docx", "", "also write the generated DOCX to this local path")
|
||||
cmd.Flags().BoolVar(&replace, "replace", true, "replace same stem|ext before upload (default); false = fail if name taken")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsPutXlsxCmd() *cobra.Command {
|
||||
var folderID, template, title, keepLocal string
|
||||
var replace bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "put-xlsx PROJECT_ID [LOCAL_XLSX]",
|
||||
Short: "Upload or generate an XLSX workbook into a project (excelize templates with formulas)",
|
||||
Long: `Spreadsheets live in OnlyOffice — not in git. Generate multi-sheet workbooks with
|
||||
formulas (SUM/AVG, cross-sheet refs, named inputs) via --template, or upload an existing .xlsx.
|
||||
|
||||
oo docs put-xlsx 218 --template cutover-portugal
|
||||
oo docs put-xlsx 218 --template cutover-portugal --title 2026-08-28-cutover-budget.xlsx
|
||||
oo docs put-xlsx 218 ./my.xlsx`,
|
||||
Args: cobra.RangeArgs(1, 2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
pid := args[0]
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dir, err := os.MkdirTemp("", "oo-docs-put-xlsx-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
var xlsxPath string
|
||||
var srcLabel string
|
||||
switch {
|
||||
case template != "":
|
||||
wb, err := xlspipe.BuildTemplate(template)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
name := title
|
||||
if name == "" {
|
||||
name = "cutover-budget.xlsx"
|
||||
if template == xlspipe.TemplateCutoverPortugal {
|
||||
name = "2026-08-28-cutover-budget.xlsx"
|
||||
}
|
||||
}
|
||||
if !strings.HasSuffix(strings.ToLower(name), ".xlsx") {
|
||||
name += ".xlsx"
|
||||
}
|
||||
xlsxPath = filepath.Join(dir, name)
|
||||
if err := xlspipe.Save(wb, xlsxPath); err != nil {
|
||||
wb.Close()
|
||||
return err
|
||||
}
|
||||
wb.Close()
|
||||
srcLabel = "template:" + template
|
||||
case len(args) == 2:
|
||||
xlsxPath = args[1]
|
||||
if docpipe.Ext(xlsxPath) != ".xlsx" {
|
||||
return fmt.Errorf("expected .xlsx, got %s", docpipe.Ext(xlsxPath))
|
||||
}
|
||||
srcLabel = xlsxPath
|
||||
default:
|
||||
return fmt.Errorf("pass LOCAL_XLSX or --template")
|
||||
}
|
||||
|
||||
if keepLocal != "" {
|
||||
b, err := os.ReadFile(xlsxPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := docpipe.EnsureDir(keepLocal); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(keepLocal, b, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
ctx := cmd.Context()
|
||||
ent, deleted, err := uploadProjectDoc(ctx, c, pid, xlsxPath, folderID, replace)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
obj := map[string]any{
|
||||
"project_id": pid,
|
||||
"source": srcLabel,
|
||||
"uploaded": fileEntryToMap(ent),
|
||||
}
|
||||
if folderID != "" {
|
||||
obj["folder_id"] = folderID
|
||||
}
|
||||
if len(deleted) > 0 {
|
||||
obj["replaced_file_ids"] = deleted
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&folderID, "folder", "", "Documents folder id (default: project root)")
|
||||
cmd.Flags().StringVar(&template, "template", "", "built-in workbook template (cutover-portugal)")
|
||||
cmd.Flags().StringVar(&title, "title", "", "upload file name when using --template")
|
||||
cmd.Flags().StringVar(&keepLocal, "keep-xlsx", "", "also write generated/uploaded bytes to this local path")
|
||||
cmd.Flags().BoolVar(&replace, "replace", true, "replace same stem|ext before upload (default); false = fail if name taken")
|
||||
return cmd
|
||||
}
|
||||
|
||||
// uploadProjectDoc upserts (--replace, default) or no-clobbers into project/folder Documents.
|
||||
func uploadProjectDoc(ctx context.Context, c *onlyoffice.Client, pid, localPath, folderID string, replace bool) (*onlyoffice.FileEntry, []int, error) {
|
||||
if folderID != "" {
|
||||
if replace {
|
||||
return c.UploadToFolderReplacing(ctx, folderID, localPath)
|
||||
}
|
||||
if err := c.AssertNoFileConflict(ctx, folderID, localPath); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
ent, err := c.UploadToFolder(ctx, folderID, localPath)
|
||||
return ent, nil, err
|
||||
}
|
||||
if replace {
|
||||
return c.UploadProjectFileReplacing(ctx, pid, localPath)
|
||||
}
|
||||
ent, err := c.UploadProjectFileNoClobber(ctx, pid, localPath)
|
||||
return ent, nil, err
|
||||
}
|
||||
+158
@@ -0,0 +1,158 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"strings"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
func init() {
|
||||
rootCmd.AddCommand(indexCmd())
|
||||
}
|
||||
|
||||
// indexFlags are shared by the `oo index folder` and `oo index files` verbs.
|
||||
type indexFlags struct {
|
||||
recursive bool
|
||||
exts string
|
||||
limit int
|
||||
workers int
|
||||
lang string
|
||||
minChars int
|
||||
workDir string
|
||||
backend string
|
||||
dryRun bool
|
||||
asJSON bool
|
||||
}
|
||||
|
||||
// indexCmd populates the own full-text index (oo_docs_text) that makes PDF and
|
||||
// scanned content searchable. The OnlyOffice index is left untouched.
|
||||
func indexCmd() *cobra.Command {
|
||||
f := &indexFlags{}
|
||||
cmd := &cobra.Command{
|
||||
Use: "index",
|
||||
Short: "Populate the own full-text index for PDF/scan content",
|
||||
Long: "Index document text that the OnlyOffice Elasticsearch index does not\n" +
|
||||
"cover (PDFs and scans) into a separate index (ONLYOFFICE_ES_TEXT_INDEX,\n" +
|
||||
"default oo_docs_text). Text is extracted with docpipe (pdftotext, OCR)\n" +
|
||||
"and the OnlyOffice server is never modified.\n\n" +
|
||||
"Requires ONLYOFFICE_URL/USER/PASS (to download files) and ONLYOFFICE_ES_URL\n" +
|
||||
"(to write the index). See docs/elasticsearch.md.",
|
||||
}
|
||||
cmd.PersistentFlags().BoolVar(&f.recursive, "recursive", false, "folder: descend into subfolders")
|
||||
cmd.PersistentFlags().StringVar(&f.exts, "exts", "pdf", "comma-separated extensions to index")
|
||||
cmd.PersistentFlags().IntVar(&f.limit, "limit", 0, "maximum number of files to index (0 = all)")
|
||||
cmd.PersistentFlags().IntVar(&f.workers, "workers", 3, "parallel downloads/extractions")
|
||||
cmd.PersistentFlags().StringVar(&f.lang, "lang", "deu+eng", "OCR language(s)")
|
||||
cmd.PersistentFlags().IntVar(&f.minChars, "min-chars", 0, "text-layer threshold below which OCR runs")
|
||||
cmd.PersistentFlags().StringVar(&f.workDir, "work-dir", "", "temp dir for downloads (default: system temp)")
|
||||
cmd.PersistentFlags().StringVar(&f.backend, "backend", "rest", "file backend: rest|dav")
|
||||
cmd.PersistentFlags().BoolVar(&f.dryRun, "dry-run", false, "list what would be indexed, without changes")
|
||||
cmd.PersistentFlags().BoolVar(&f.asJSON, "json", false, "shorthand for --output json")
|
||||
|
||||
cmd.AddCommand(indexFolderCmd(f), indexFilesCmd(f))
|
||||
return cmd
|
||||
}
|
||||
|
||||
func indexFolderCmd(f *indexFlags) *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "folder FOLDER_ID",
|
||||
Short: "Index every matching file in a Documents folder",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
return runIndex(cmd, f, args[0], nil)
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func indexFilesCmd(f *indexFlags) *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "files FILE_ID...",
|
||||
Short: "Index specific Documents files",
|
||||
Args: cobra.MinimumNArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
return runIndex(cmd, f, "", args)
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func runIndex(cmd *cobra.Command, f *indexFlags, folderID string, ids []string) error {
|
||||
if f.asJSON {
|
||||
outputFormat = "json"
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
idx, err := onlyoffice.NewESTextIndex(onlyoffice.ESTextConfigFromEnv())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ti := onlyoffice.NewTextIndexer(c.FileStore(f.backend), idx)
|
||||
ti.WorkDir = f.workDir
|
||||
opts := onlyoffice.IndexOptions{
|
||||
Recursive: f.recursive,
|
||||
Extensions: splitList(f.exts),
|
||||
Limit: f.limit,
|
||||
Lang: f.lang,
|
||||
MinChars: f.minChars,
|
||||
Workers: f.workers,
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
|
||||
if f.dryRun {
|
||||
var entries []onlyoffice.Entry
|
||||
if folderID != "" {
|
||||
entries, err = ti.PlanFolder(ctx, folderID, opts)
|
||||
} else {
|
||||
entries, err = ti.PlanFiles(ctx, ids, opts)
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(entries))
|
||||
for _, e := range entries {
|
||||
rows = append(rows, map[string]any{
|
||||
"id": e.ID,
|
||||
"title": e.Title,
|
||||
"folder": e.ParentID,
|
||||
})
|
||||
}
|
||||
printTable([]string{"id", "title", "folder"}, rows)
|
||||
return nil
|
||||
}
|
||||
|
||||
if err := ti.Ensure(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
var res onlyoffice.IndexResult
|
||||
if folderID != "" {
|
||||
res, err = ti.IndexFolder(ctx, folderID, opts)
|
||||
} else {
|
||||
res, err = ti.IndexFiles(ctx, ids, opts)
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{
|
||||
"index": idx.Index(),
|
||||
"scanned": res.Scanned,
|
||||
"indexed": res.Indexed,
|
||||
"skipped": res.Skipped,
|
||||
"failed": res.Failed,
|
||||
"errors": res.Errors,
|
||||
})
|
||||
return nil
|
||||
}
|
||||
|
||||
// splitList parses a comma-separated flag value, dropping blanks.
|
||||
func splitList(s string) []string {
|
||||
parts := strings.Split(s, ",")
|
||||
out := make([]string, 0, len(parts))
|
||||
for _, p := range parts {
|
||||
if p = strings.TrimSpace(p); p != "" {
|
||||
out = append(out, p)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestIndexCommandRegistered(t *testing.T) {
|
||||
cmd, _, err := rootCmd.Find([]string{"index"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if cmd.Name() != "index" {
|
||||
t.Fatalf("index resolved to %q", cmd.Name())
|
||||
}
|
||||
for _, name := range []string{"exts", "limit", "workers", "lang", "min-chars", "work-dir", "backend", "dry-run", "recursive", "json"} {
|
||||
if cmd.PersistentFlags().Lookup(name) == nil {
|
||||
t.Errorf("index: missing --%s flag", name)
|
||||
}
|
||||
}
|
||||
if cmd.PersistentFlags().Lookup("exts").DefValue != "pdf" {
|
||||
t.Errorf("--exts default = %q, want pdf", cmd.PersistentFlags().Lookup("exts").DefValue)
|
||||
}
|
||||
for _, verb := range []string{"index folder", "index files"} {
|
||||
sub, _, err := rootCmd.Find(strings.Fields(verb))
|
||||
if err != nil {
|
||||
t.Fatalf("%s: %v", verb, err)
|
||||
}
|
||||
if sub.Name() != strings.Fields(verb)[1] {
|
||||
t.Errorf("%s resolved to %q", verb, sub.Name())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestIndexSearchBackendFlag(t *testing.T) {
|
||||
cmd, _, err := rootCmd.Find([]string{"search"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if cmd.Flags().Lookup("backend") == nil {
|
||||
t.Fatal("search: missing --backend flag")
|
||||
}
|
||||
if cmd.Flags().Lookup("backend").DefValue != "oo" {
|
||||
t.Errorf("--backend default = %q, want oo", cmd.Flags().Lookup("backend").DefValue)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSplitList(t *testing.T) {
|
||||
got := splitList(" pdf , .PDF, docx ,, ")
|
||||
want := []string{"pdf", ".PDF", "docx"}
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("splitList = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,441 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
var invoicesCmd = &cobra.Command{
|
||||
Use: "invoices",
|
||||
Aliases: []string{"invoice"},
|
||||
Short: "CRM invoices (billing)",
|
||||
}
|
||||
|
||||
func init() {
|
||||
rootCmd.AddCommand(invoicesCmd)
|
||||
invoicesCmd.AddCommand(invoiceListCmd())
|
||||
invoicesCmd.AddCommand(invoiceGetCmd())
|
||||
invoicesCmd.AddCommand(invoiceCreateCmd())
|
||||
invoicesCmd.AddCommand(invoiceUpdateCmd())
|
||||
invoicesCmd.AddCommand(invoicePDFCmd())
|
||||
invoicesCmd.AddCommand(invoicePDFCleanupCmd())
|
||||
invoicesCmd.AddCommand(invoiceStatusCmd())
|
||||
invoicesCmd.AddCommand(invoiceDeleteCmd())
|
||||
invoicesCmd.AddCommand(invoiceItemsCmd())
|
||||
}
|
||||
|
||||
func invoiceListCmd() *cobra.Command {
|
||||
var count, offset int
|
||||
cmd := &cobra.Command{
|
||||
Use: "list",
|
||||
Short: "List CRM invoices",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
list, total, err := c.ListInvoices(cmd.Context(), count, offset)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(list))
|
||||
for _, inv := range list {
|
||||
rows = append(rows, onlyoffice.FlattenInvoiceRow(inv))
|
||||
}
|
||||
if outputFormat == "table" {
|
||||
fmt.Printf("total: %d (shown: %d)\n", total, len(rows))
|
||||
}
|
||||
printTable([]string{"id", "number", "cost", "statusTitle", "contactName", "issueDate", "dueDate"}, rows)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().IntVar(&count, "count", 50, "")
|
||||
cmd.Flags().IntVar(&offset, "offset", 0, "")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func invoiceGetCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "get INVOICE_ID",
|
||||
Short: "Show an invoice by id (incl. lines)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out, err := c.GetInvoice(cmd.Context(), args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func invoiceCreateCmd() *cobra.Command {
|
||||
var (
|
||||
number, issueDate, dueDate, language, currency, terms, description, po string
|
||||
contactID, consigneeID, itemID, opportunityID int64
|
||||
price, qty float64
|
||||
lineDesc string
|
||||
)
|
||||
cmd := &cobra.Command{
|
||||
Use: "create",
|
||||
Short: "Create a draft invoice with one line",
|
||||
Long: `Create a CRM invoice (Draft) with a single line.
|
||||
|
||||
Always pass --opportunity when a deal exists (entity link at create). Updating
|
||||
--opportunity later often fails with HTTP 400 — see docs/crm-associations.md.
|
||||
|
||||
Example:
|
||||
oo invoices create --number INV-2026-01 --contact CONTACT_ID --item ITEM_ID \
|
||||
--price 300 --opportunity OPPORTUNITY_ID --line-description "Service package" \
|
||||
--terms "…payment terms…"
|
||||
`,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if number == "" || contactID == 0 || itemID == 0 {
|
||||
return fmt.Errorf("--number, --contact and --item are required")
|
||||
}
|
||||
if issueDate == "" {
|
||||
issueDate = time.Now().Format("2006-01-02") + "T00:00:00.0000000+01:00"
|
||||
} else if !strings.Contains(issueDate, "T") {
|
||||
issueDate = issueDate + "T00:00:00.0000000+01:00"
|
||||
}
|
||||
if dueDate == "" {
|
||||
dueDate = time.Now().Add(14 * 24 * time.Hour).Format("2006-01-02") + "T00:00:00.0000000+01:00"
|
||||
} else if !strings.Contains(dueDate, "T") {
|
||||
dueDate = dueDate + "T00:00:00.0000000+01:00"
|
||||
}
|
||||
if qty == 0 {
|
||||
qty = 1
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out, err := c.CreateInvoice(cmd.Context(), onlyoffice.CreateInvoiceParams{
|
||||
Number: number,
|
||||
IssueDate: issueDate,
|
||||
DueDate: dueDate,
|
||||
ContactID: contactID,
|
||||
ConsigneeID: consigneeID,
|
||||
EntityID: opportunityID,
|
||||
EntityType: 0, // Opportunity
|
||||
Language: language,
|
||||
Currency: currency,
|
||||
ExchangeRate: 1,
|
||||
PurchaseOrderNumber: po,
|
||||
Terms: terms,
|
||||
Description: description,
|
||||
Lines: []onlyoffice.InvoiceLine{{
|
||||
InvoiceItemID: itemID,
|
||||
Description: lineDesc,
|
||||
Quantity: qty,
|
||||
Price: price,
|
||||
SortOrder: 0,
|
||||
}},
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&number, "number", "", "invoice number (e.g. INV-2026-01)")
|
||||
cmd.Flags().Int64Var(&contactID, "contact", 0, "bill-to company contact id (canonical company, not a PDF-only clone)")
|
||||
cmd.Flags().Int64Var(&consigneeID, "consignee", 0, "optional Empfänger person/contact id")
|
||||
cmd.Flags().Int64Var(&opportunityID, "opportunity", 0, "link to CRM opportunity/deal id (set at create)")
|
||||
cmd.Flags().Int64Var(&itemID, "item", 0, "catalog invoice item id")
|
||||
cmd.Flags().Float64Var(&price, "price", 0, "line price")
|
||||
cmd.Flags().Float64Var(&qty, "qty", 1, "line quantity")
|
||||
cmd.Flags().StringVar(&lineDesc, "line-description", "", "line description override")
|
||||
cmd.Flags().StringVar(&issueDate, "issue-date", "", "YYYY-MM-DD (default today)")
|
||||
cmd.Flags().StringVar(&dueDate, "due-date", "", "YYYY-MM-DD (default +14d)")
|
||||
cmd.Flags().StringVar(&language, "language", "de-DE", "invoice language")
|
||||
cmd.Flags().StringVar(¤cy, "currency", "EUR", "currency abbreviation")
|
||||
cmd.Flags().StringVar(&terms, "terms", "", "payment terms / footer")
|
||||
cmd.Flags().StringVar(&description, "description", "", "invoice description")
|
||||
cmd.Flags().StringVar(&po, "po", "", "purchase order number")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func invoiceUpdateCmd() *cobra.Command {
|
||||
var opportunityID int64
|
||||
var description, po, terms string
|
||||
cmd := &cobra.Command{
|
||||
Use: "update INVOICE_ID",
|
||||
Short: "Update invoice (notes, PO, terms; opportunity link unreliable)",
|
||||
Long: `Update Draft invoice fields.
|
||||
|
||||
--opportunity often returns HTTP 400 on existing invoices. Prefer
|
||||
oo invoices create … --opportunity, or delete+recreate. See docs/crm-associations.md.
|
||||
`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if opportunityID == 0 && !cmd.Flags().Changed("description") && !cmd.Flags().Changed("po") && !cmd.Flags().Changed("terms") {
|
||||
return fmt.Errorf("set --opportunity and/or --description and/or --po and/or --terms")
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out, err := c.UpdateInvoice(cmd.Context(), args[0], onlyoffice.UpdateInvoiceParams{
|
||||
EntityID: opportunityID,
|
||||
EntityType: 0,
|
||||
Description: description,
|
||||
DescriptionSet: cmd.Flags().Changed("description"),
|
||||
PurchaseOrder: po,
|
||||
PurchaseOrderSet: cmd.Flags().Changed("po"),
|
||||
Terms: terms,
|
||||
TermsSet: cmd.Flags().Changed("terms"),
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().Int64Var(&opportunityID, "opportunity", 0, "CRM opportunity/deal id (may 400 — prefer create)")
|
||||
cmd.Flags().StringVar(&description, "description", "", "invoice notes (Notizen); use \\n for line breaks")
|
||||
cmd.Flags().StringVar(&po, "po", "", "purchase order number")
|
||||
cmd.Flags().StringVar(&terms, "terms", "", "payment terms / footer (Bedingungen)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func invoicePDFCmd() *cobra.Command {
|
||||
var force bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "pdf INVOICE_ID",
|
||||
Short: "Return / regenerate invoice PDF file metadata",
|
||||
Long: `GET /api/2.0/crm/invoice/{id}/pdf.
|
||||
|
||||
With --force, touches the Draft to clear cached fileID first (layout changes).
|
||||
`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
var out map[string]any
|
||||
if force {
|
||||
out, err = c.ForceRegenerateInvoicePDF(cmd.Context(), args[0])
|
||||
} else {
|
||||
out, err = c.InvoicePDFFile(cmd.Context(), args[0])
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().BoolVar(&force, "force", false, "clear PDF cache then regenerate")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func invoicePDFCleanupCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "pdf-cleanup INVOICE_ID",
|
||||
Short: "Delete older P-*.pdf copies on company/deal; keep invoice.fileID",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
deleted, err := c.PurgeStaleInvoicePDFs(cmd.Context(), args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if outputFormat == "json" {
|
||||
printObject(map[string]any{"deleted": deleted})
|
||||
return nil
|
||||
}
|
||||
fmt.Printf("deleted %d stale PDF file(s): %v\n", len(deleted), deleted)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func invoiceStatusCmd() *cobra.Command {
|
||||
var status string
|
||||
cmd := &cobra.Command{
|
||||
Use: "status INVOICE_ID [INVOICE_ID...]",
|
||||
Short: "Set invoice status (draft|billed|rejected|paid)",
|
||||
Long: `PUT /api/2.0/crm/invoice/status/{id}.
|
||||
|
||||
Billed invoices are not content-editable. Billed→Draft often does not work —
|
||||
recreate as Draft instead (docs/crm-associations.md).
|
||||
`,
|
||||
Args: cobra.MinimumNArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
statusID, err := parseInvoiceStatus(status)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ids := make([]int64, 0, len(args))
|
||||
for _, a := range args {
|
||||
var id int64
|
||||
if _, err := fmt.Sscan(a, &id); err != nil || id <= 0 {
|
||||
return fmt.Errorf("invalid invoice id %q", a)
|
||||
}
|
||||
ids = append(ids, id)
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out, err := c.SetInvoiceStatus(cmd.Context(), statusID, ids...)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&status, "status", "draft", "draft|billed|rejected|paid or numeric id")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func parseInvoiceStatus(s string) (int, error) {
|
||||
s = strings.TrimSpace(strings.ToLower(s))
|
||||
switch s {
|
||||
case "", "draft", "1":
|
||||
return onlyoffice.InvoiceStatusDraft, nil
|
||||
case "billed", "2":
|
||||
return onlyoffice.InvoiceStatusBilled, nil
|
||||
case "rejected", "3":
|
||||
return onlyoffice.InvoiceStatusRejected, nil
|
||||
case "paid", "4":
|
||||
return onlyoffice.InvoiceStatusPaid, nil
|
||||
default:
|
||||
var n int
|
||||
if _, err := fmt.Sscan(s, &n); err != nil || n <= 0 {
|
||||
return 0, fmt.Errorf("unknown status %q (draft|billed|rejected|paid)", s)
|
||||
}
|
||||
return n, nil
|
||||
}
|
||||
}
|
||||
|
||||
func invoiceDeleteCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "delete INVOICE_ID [INVOICE_ID...]",
|
||||
Short: "Delete one or more invoices",
|
||||
Args: cobra.MinimumNArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for _, id := range args {
|
||||
if _, err := c.DeleteInvoice(cmd.Context(), id); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Println("deleted", id)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func invoiceItemsCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "items",
|
||||
Aliases: []string{"item"},
|
||||
Short: "Invoice catalog items",
|
||||
}
|
||||
cmd.AddCommand(invoiceItemsListCmd())
|
||||
cmd.AddCommand(invoiceItemsCreateCmd())
|
||||
cmd.AddCommand(invoiceItemsDeleteCmd())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func invoiceItemsListCmd() *cobra.Command {
|
||||
var count, offset int
|
||||
cmd := &cobra.Command{
|
||||
Use: "list",
|
||||
Short: "List catalog invoice items",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
list, total, err := c.ListInvoiceItems(cmd.Context(), count, offset)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if outputFormat == "table" {
|
||||
fmt.Printf("total: %d (shown: %d)\n", total, len(list))
|
||||
for _, row := range list {
|
||||
if cur, ok := row["currency"].(map[string]any); ok {
|
||||
row["currency"] = cur["abbreviation"]
|
||||
}
|
||||
}
|
||||
}
|
||||
printTable([]string{"id", "title", "price", "currency", "description"}, list)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().IntVar(&count, "count", 50, "")
|
||||
cmd.Flags().IntVar(&offset, "offset", 0, "")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func invoiceItemsCreateCmd() *cobra.Command {
|
||||
var title, description, currency string
|
||||
var price float64
|
||||
cmd := &cobra.Command{
|
||||
Use: "create",
|
||||
Short: "Create a catalog invoice item",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if title == "" {
|
||||
return fmt.Errorf("--title is required")
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out, err := c.CreateInvoiceItem(cmd.Context(), title, description, price, currency)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&title, "title", "", "item title")
|
||||
cmd.Flags().StringVar(&description, "description", "", "item description")
|
||||
cmd.Flags().Float64Var(&price, "price", 0, "unit price")
|
||||
cmd.Flags().StringVar(¤cy, "currency", "EUR", "currency")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func invoiceItemsDeleteCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "delete ITEM_ID [ITEM_ID...]",
|
||||
Short: "Delete catalog invoice items",
|
||||
Args: cobra.MinimumNArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for _, id := range args {
|
||||
if _, err := c.DeleteInvoiceItem(cmd.Context(), id); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Println("deleted", id)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
+288
-1
@@ -1,6 +1,8 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
@@ -11,7 +13,7 @@ import (
|
||||
var mailsCmd = &cobra.Command{
|
||||
Use: "mails",
|
||||
Aliases: []string{"mail"},
|
||||
Short: "OnlyOffice Workspace mail — list, read, delete",
|
||||
Short: "OnlyOffice Workspace mail — list, read, draft, delete",
|
||||
}
|
||||
|
||||
func init() {
|
||||
@@ -20,6 +22,11 @@ func init() {
|
||||
mailsCmd.AddCommand(mailsFoldersCmd())
|
||||
mailsCmd.AddCommand(mailsListCmd())
|
||||
mailsCmd.AddCommand(mailsGetCmd())
|
||||
mailsCmd.AddCommand(mailsDownloadAttachmentCmd())
|
||||
mailsCmd.AddCommand(mailsDraftCmd())
|
||||
mailsCmd.AddCommand(mailsAttachCmd())
|
||||
mailsCmd.AddCommand(mailsDraftInvoiceCmd())
|
||||
mailsCmd.AddCommand(mailsSendCmd())
|
||||
mailsCmd.AddCommand(mailsDeleteCmd())
|
||||
}
|
||||
|
||||
@@ -126,6 +133,286 @@ func mailsGetCmd() *cobra.Command {
|
||||
}
|
||||
}
|
||||
|
||||
func mailsDownloadAttachmentCmd() *cobra.Command {
|
||||
var outPath string
|
||||
cmd := &cobra.Command{
|
||||
Use: "download-attachment ATTACHMENT_ID",
|
||||
Short: "Download a mail attachment by attachment id",
|
||||
Long: `Download a raw attachment from OnlyOffice Mail's download.ashx handler.
|
||||
|
||||
Example:
|
||||
oo mails download-attachment 12345 --out /tmp/attach.bin
|
||||
`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if strings.TrimSpace(outPath) == "" {
|
||||
return fmt.Errorf("--out is required")
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
body, err := c.DownloadMailAttachment(cmd.Context(), args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := writeMailAttachment(outPath, body); err != nil {
|
||||
return err
|
||||
}
|
||||
if outputFormat == "json" {
|
||||
printObject(map[string]any{
|
||||
"attachmentId": args[0],
|
||||
"bytes": len(body),
|
||||
"path": outPath,
|
||||
})
|
||||
return nil
|
||||
}
|
||||
fmt.Printf("saved %d bytes to %s\n", len(body), outPath)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&outPath, "out", "", "output file path")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func mailsDraftCmd() *cobra.Command {
|
||||
var from, to, cc, bcc, subject, body, html string
|
||||
var id int64
|
||||
cmd := &cobra.Command{
|
||||
Use: "draft",
|
||||
Short: "Create or update a mail draft (OnlyOffice Mail)",
|
||||
Long: `Save a draft in /addons/mail (PUT /api/2.0/mail/drafts/save).
|
||||
|
||||
oo mails draft --to a@b.com --subject "…" --body "<p>…</p>"
|
||||
oo mails draft --id 123 --to a@b.com --subject "…" --html "<p>…</p>"
|
||||
`,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if to == "" {
|
||||
return fmt.Errorf("--to is required")
|
||||
}
|
||||
htmlBody := body
|
||||
if html != "" {
|
||||
htmlBody = html
|
||||
}
|
||||
if htmlBody == "" {
|
||||
return fmt.Errorf("--body or --html is required")
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out, err := c.SaveMailDraft(cmd.Context(), onlyoffice.SaveMailDraftParams{
|
||||
ID: id,
|
||||
From: from,
|
||||
To: to,
|
||||
Cc: cc,
|
||||
Bcc: bcc,
|
||||
Subject: subject,
|
||||
Body: onlyoffice.PlainTextToMailHTML(htmlBody),
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().Int64Var(&id, "id", 0, "existing draft id (0 = create)")
|
||||
cmd.Flags().StringVar(&from, "from", "", "from address (default: first enabled mailbox)")
|
||||
cmd.Flags().StringVar(&to, "to", "", "recipient (required)")
|
||||
cmd.Flags().StringVar(&cc, "cc", "", "cc")
|
||||
cmd.Flags().StringVar(&bcc, "bcc", "", "bcc")
|
||||
cmd.Flags().StringVar(&subject, "subject", "", "subject")
|
||||
cmd.Flags().StringVar(&body, "body", "", "plain text or HTML body")
|
||||
cmd.Flags().StringVar(&html, "html", "", "HTML body (alias of --body when set)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func mailsAttachCmd() *cobra.Command {
|
||||
var fileID int64
|
||||
cmd := &cobra.Command{
|
||||
Use: "attach MESSAGE_ID",
|
||||
Short: "Attach an OnlyOffice Files document to a draft/message",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if fileID <= 0 {
|
||||
return fmt.Errorf("--file-id is required")
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out, err := c.AttachMailDocument(cmd.Context(), args[0], fileID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().Int64Var(&fileID, "file-id", 0, "OnlyOffice file id (e.g. invoice PDF)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func mailsDraftInvoiceCmd() *cobra.Command {
|
||||
var to, from, subject, body string
|
||||
var invoiceID int
|
||||
cmd := &cobra.Command{
|
||||
Use: "draft-invoice",
|
||||
Short: "Create a mail draft with regenerated invoice PDF attached",
|
||||
Long: `Regenerate the CRM invoice PDF, save an OnlyOffice Mail draft, and attach the PDF.
|
||||
|
||||
Does not send. Open /addons/mail/#drafts to review.
|
||||
|
||||
oo mails draft-invoice --invoice 16 --to info@example.com
|
||||
`,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if invoiceID <= 0 || to == "" {
|
||||
return fmt.Errorf("--invoice and --to are required")
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
pdf, err := c.ForceRegenerateInvoicePDF(cmd.Context(), strconv.Itoa(invoiceID))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fileID := onlyoffice.Int64FromMap(pdf, "id")
|
||||
if fileID <= 0 {
|
||||
return fmt.Errorf("invoice PDF has no file id: %+v", pdf)
|
||||
}
|
||||
inv, err := c.GetInvoice(cmd.Context(), strconv.Itoa(invoiceID))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
number := strings.TrimSpace(fmt.Sprint(inv["number"]))
|
||||
cost := formatInvoiceCostEUR(inv["cost"])
|
||||
if subject == "" {
|
||||
subject = fmt.Sprintf("Invoice %s (%s EUR)", number, cost)
|
||||
}
|
||||
if body == "" {
|
||||
body = fmt.Sprintf(`Hello,
|
||||
|
||||
please find attached invoice %s for %s EUR.
|
||||
|
||||
Payment terms: see invoice notes.
|
||||
|
||||
Best regards`, number, cost)
|
||||
}
|
||||
draft, err := c.SaveMailDraft(cmd.Context(), onlyoffice.SaveMailDraftParams{
|
||||
From: from,
|
||||
To: to,
|
||||
Subject: subject,
|
||||
Body: onlyoffice.MailHTMLWithBlankParagraphs(onlyoffice.PlainTextToMailHTML(body)),
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
mid := fmt.Sprint(draft["id"])
|
||||
att, err := c.AttachMailDocument(cmd.Context(), mid, fileID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("draft %s created but attach failed: %w", mid, err)
|
||||
}
|
||||
if outputFormat == "json" {
|
||||
printObject(map[string]any{"draft": draft, "attachment": att, "pdfFileId": fileID})
|
||||
return nil
|
||||
}
|
||||
fmt.Printf("draft %s to=%s subject=%q pdfFileId=%d\n", mid, to, subject, fileID)
|
||||
fmt.Printf("open: %s/addons/mail/#drafts\n", strings.TrimRight(os.Getenv("ONLYOFFICE_URL"), "/"))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().IntVar(&invoiceID, "invoice", 0, "CRM invoice id")
|
||||
cmd.Flags().StringVar(&to, "to", "", "recipient")
|
||||
cmd.Flags().StringVar(&from, "from", "", "from address (default mailbox)")
|
||||
cmd.Flags().StringVar(&subject, "subject", "", "override subject")
|
||||
cmd.Flags().StringVar(&body, "body", "", "override plain-text body")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func formatInvoiceCostEUR(v any) string {
|
||||
s := strings.TrimSpace(fmt.Sprint(v))
|
||||
s = strings.TrimSuffix(s, ".00")
|
||||
s = strings.TrimSuffix(s, ".0")
|
||||
if s == "" || s == "<nil>" {
|
||||
return "?"
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func writeMailAttachment(path string, body []byte) error {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return fmt.Errorf("attachment output path is required")
|
||||
}
|
||||
return os.WriteFile(path, body, 0o644)
|
||||
}
|
||||
|
||||
func mailsSendCmd() *cobra.Command {
|
||||
var from, to, cc, bcc, subject, body, html string
|
||||
var id int64
|
||||
cmd := &cobra.Command{
|
||||
Use: "send",
|
||||
Short: "Send a mail message (OnlyOffice Mail)",
|
||||
Long: `Send via PUT /api/2.0/mail/messages/send.json.
|
||||
|
||||
oo mails send --id 7803 --body "…" # send referencing a draft id
|
||||
oo mails send --to a@b.com --subject "…" --body "…"
|
||||
oo mails send --id 7803 --to a@b.com --subject "…" --body "…" --cc x@y.com
|
||||
|
||||
IMPORTANT: send.json does NOT copy subject/body from the referenced draft — the
|
||||
content must be in this request (--subject/--body). Cc/Bcc are omitted when empty
|
||||
(the API 400s on empty strings). The API send does not append the UI signature —
|
||||
put the chat line in --body if needed.
|
||||
`,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if to == "" && id == 0 {
|
||||
return fmt.Errorf("--to is required (or --id of an existing draft)")
|
||||
}
|
||||
htmlBody := body
|
||||
if html != "" {
|
||||
htmlBody = html
|
||||
}
|
||||
if htmlBody == "" && id != 0 {
|
||||
// The send.json endpoint does NOT copy subject/body from the
|
||||
// referenced draft — an empty body here sends an empty message.
|
||||
// Warn instead of silently mailing an empty email.
|
||||
return fmt.Errorf("--body/--html is required when sending by --id (send.json needs the content in the request)")
|
||||
}
|
||||
if htmlBody == "" && to == "" {
|
||||
return fmt.Errorf("--body is required for a fresh message")
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
raw, err := c.SendMail(cmd.Context(), onlyoffice.SendMailParams{
|
||||
ID: id,
|
||||
From: from,
|
||||
To: to,
|
||||
Cc: cc,
|
||||
Bcc: bcc,
|
||||
Subject: subject,
|
||||
Body: htmlBody,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Println(string(raw))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().Int64Var(&id, "id", 0, "existing draft id to send (0 = fresh message)")
|
||||
cmd.Flags().StringVar(&from, "from", "", "from address (default: first enabled mailbox)")
|
||||
cmd.Flags().StringVar(&to, "to", "", "recipient (required unless --id)")
|
||||
cmd.Flags().StringVar(&cc, "cc", "", "cc")
|
||||
cmd.Flags().StringVar(&bcc, "bcc", "", "bcc")
|
||||
cmd.Flags().StringVar(&subject, "subject", "", "subject")
|
||||
cmd.Flags().StringVar(&body, "body", "", "plain text or HTML body")
|
||||
cmd.Flags().StringVar(&html, "html", "", "HTML body (alias of --body when set)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func mailsDeleteCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "delete ID [ID...]",
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestWriteMailAttachment(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "attach.bin")
|
||||
body := []byte("payload")
|
||||
if err := writeMailAttachment(path, body); err != nil {
|
||||
t.Fatalf("writeMailAttachment: %v", err)
|
||||
}
|
||||
got, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadFile: %v", err)
|
||||
}
|
||||
if string(got) != string(body) {
|
||||
t.Fatalf("body = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteMailAttachmentRequiresPath(t *testing.T) {
|
||||
if err := writeMailAttachment("", []byte("x")); err == nil {
|
||||
t.Fatal("expected error for empty path")
|
||||
}
|
||||
}
|
||||
+13
-7
@@ -3,19 +3,25 @@
|
||||
// Command tree is subject-based (mirrors the library split and the `tea` CLI):
|
||||
//
|
||||
// oo calendar list | events | add | delete
|
||||
// oo projects list | get | milestones | create | update | delete | files (list|upload|download|rename|delete)
|
||||
// oo projects list | get | milestones | milestone-create | create | update | delete | contacts (add|remove) | link-authors | link-git | files (list|upload|download|rename|delete|dedupe|as-md|put-md|put-txt|put-xlsx)
|
||||
// oo tasks list | get | create | update | delete | subtask add | files (list|upload|detach)
|
||||
// oo users list | self (alias: oo whoami)
|
||||
// oo contacts list | get | delete | info-add | dedupe-info
|
||||
// oo contacts list | get | delete | info-add | merge | dedupe-info | tags | tag-add | tag-create | tag-remove
|
||||
// oo persons list | create | delete | dedupe
|
||||
// oo companies list | create | delete | dedupe | dedupe-persons
|
||||
// oo opportunities list | get | create | delete | stages | member-add | dedupe | dedupe-members | fix-titles
|
||||
// oo opportunities list | get | create | update | delete | stages | member-add | dedupe | dedupe-members | fix-titles
|
||||
// oo cases list | create | delete | member-add
|
||||
// oo crm-tasks list | create | delete | categories
|
||||
// oo crm-tasks list | create | delete | categories | reassign-self
|
||||
// oo crm cleanup
|
||||
// oo mails accounts | folders | list | get | delete
|
||||
// oo applications sync
|
||||
// oo catalog scan-contacts | scan-projects | scan-thunderbird | merge | match | apply
|
||||
// oo mails accounts | folders | list | get | download-attachment | draft | attach | draft-invoice | send | delete
|
||||
// oo invoices list | get | create | update | pdf | pdf-cleanup | status | delete | items …
|
||||
// oo docs tools | convert | optimize | ocr | hocr | as-md | put-md | put-txt | put-xlsx
|
||||
// oo catalog match | merge | apply | scan-contacts | scan-projects | scan-thunderbird
|
||||
// oo dav ls | move | copy | mkdir | rename-file | rename-folder | download | fileops
|
||||
// oo search QUERY [--content] [--folder ID] [--limit N] [--backend oo|own] [--json]
|
||||
// oo index folder FOLDER_ID | files FILE_ID... [--recursive] [--exts pdf] [--dry-run]
|
||||
//
|
||||
// CRM association rules: docs/crm-associations.md
|
||||
//
|
||||
// Every list supports `--output/-o json|table` (table is the default).
|
||||
//
|
||||
|
||||
@@ -19,6 +19,7 @@ func init() {
|
||||
opportunitiesCmd.AddCommand(oppListCmd())
|
||||
opportunitiesCmd.AddCommand(oppGetCmd())
|
||||
opportunitiesCmd.AddCommand(oppCreateCmd())
|
||||
opportunitiesCmd.AddCommand(oppUpdateCmd())
|
||||
opportunitiesCmd.AddCommand(oppDeleteCmd())
|
||||
opportunitiesCmd.AddCommand(oppStagesCmd())
|
||||
opportunitiesCmd.AddCommand(oppMemberAddCmd())
|
||||
@@ -118,6 +119,40 @@ func oppCreateCmd() *cobra.Command {
|
||||
return cmd
|
||||
}
|
||||
|
||||
func oppUpdateCmd() *cobra.Command {
|
||||
var title, desc string
|
||||
var stage int64
|
||||
var bid float64
|
||||
cmd := &cobra.Command{
|
||||
Use: "update OPPORTUNITY_ID",
|
||||
Short: "Update opportunity title, description, stage, or bid",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out, err := c.UpdateOpportunity(cmd.Context(), args[0], onlyoffice.UpdateOpportunityParams{
|
||||
Title: title,
|
||||
Description: desc,
|
||||
StageID: stage,
|
||||
BidValue: bid,
|
||||
BidValueSet: cmd.Flags().Changed("bid"),
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&title, "title", "", "new title")
|
||||
cmd.Flags().StringVar(&desc, "description", "", "new description")
|
||||
cmd.Flags().Int64Var(&stage, "stage", 0, "pipeline stage id")
|
||||
cmd.Flags().Float64Var(&bid, "bid", 0, "bid value")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func oppDeleteCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "delete OPPORTUNITY_ID [OPPORTUNITY_ID...]",
|
||||
|
||||
+52
-7
@@ -6,9 +6,9 @@ import (
|
||||
"os/exec"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/eslider/go-onlyoffice/catalog"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
@@ -23,6 +23,7 @@ func init() {
|
||||
projectsCmd.AddCommand(prjListCmd())
|
||||
projectsCmd.AddCommand(prjGetCmd())
|
||||
projectsCmd.AddCommand(prjMilestonesCmd())
|
||||
projectsCmd.AddCommand(prjMilestoneCreateCmd())
|
||||
projectsCmd.AddCommand(prjCreateCmd())
|
||||
projectsCmd.AddCommand(prjUpdateCmd())
|
||||
projectsCmd.AddCommand(prjDeleteCmd())
|
||||
@@ -119,6 +120,52 @@ func prjMilestonesCmd() *cobra.Command {
|
||||
}
|
||||
}
|
||||
|
||||
func prjMilestoneCreateCmd() *cobra.Command {
|
||||
var deadline, desc string
|
||||
var key bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "milestone-create PROJECT_ID TITLE",
|
||||
Short: "Create a project milestone (Gantt row)",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
pid, err := strconv.Atoi(args[0])
|
||||
if err != nil {
|
||||
return fmt.Errorf("project id must be integer: %w", err)
|
||||
}
|
||||
if deadline == "" {
|
||||
return fmt.Errorf("--deadline YYYY-MM-DD is required")
|
||||
}
|
||||
day, err := time.Parse("2006-01-02", deadline)
|
||||
if err != nil {
|
||||
return fmt.Errorf("deadline: %w", err)
|
||||
}
|
||||
ms, err := c.CreateMilestone(onlyoffice.NewMilestoneRequest{
|
||||
ProjectID: pid,
|
||||
Title: args[1],
|
||||
Deadline: onlyoffice.Time(day),
|
||||
Description: desc,
|
||||
IsKey: key,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{
|
||||
"id": derefInt64(ms.ID),
|
||||
"title": derefString(ms.Title),
|
||||
})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&deadline, "deadline", "", "deadline YYYY-MM-DD")
|
||||
cmd.Flags().StringVar(&desc, "description", "", "description")
|
||||
cmd.Flags().BoolVar(&key, "key", false, "mark as key milestone")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func prjCreateCmd() *cobra.Command {
|
||||
var desc, resp string
|
||||
var country, company string
|
||||
@@ -143,7 +190,7 @@ full title is composed as "CC | Company | Title".`,
|
||||
}
|
||||
title := args[0]
|
||||
if country != "" || company != "" {
|
||||
title = catalog.FormatProjectTitle(country, company, args[0])
|
||||
title = onlyoffice.FormatProjectTitle(country, company, args[0])
|
||||
}
|
||||
p, err := c.CreateProject(onlyoffice.NewProjectRequest{
|
||||
Title: title,
|
||||
@@ -381,9 +428,9 @@ func prjContactsLinkGitCmd() *cobra.Command {
|
||||
Short: "Link CRM persons matched from git shortlog (+ optional companies)",
|
||||
Long: `Runs git shortlog -sne --all in --git-root, matches author emails to CRM
|
||||
persons (FindPersonByEmail), and links found contacts to the project.
|
||||
Does not create new persons — approve/apply them via catalog first.
|
||||
Does not create new persons — create/link them in CRM first.
|
||||
|
||||
Also links --company-id contacts (e.g. Acme + end-client company).`,
|
||||
Also links --company-id contacts (employer and/or client company).`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if gitRoot == "" && len(companyIDs) == 0 {
|
||||
@@ -425,10 +472,8 @@ func prjContactsLinkAuthorsCmd() *cobra.Command {
|
||||
Long: `Same as link-git, but reads authors from a file produced by:
|
||||
|
||||
git shortlog -sne --all > authors.txt
|
||||
# or via ssh:
|
||||
ssh ops-host 'git -C /path shortlog -sne --all' > authors.txt
|
||||
|
||||
Then: oo projects contacts link-authors 59 --from authors.txt --company-id 9`,
|
||||
Then: oo projects contacts link-authors 59 --from authors.txt --company-id COMPANY_ID`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if authorsFile == "" && len(companyIDs) == 0 {
|
||||
|
||||
+150
-10
@@ -3,6 +3,7 @@ package main
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
@@ -24,9 +25,43 @@ func projectFilesCmd() *cobra.Command {
|
||||
cmd.AddCommand(prjFilesDownloadCmd())
|
||||
cmd.AddCommand(prjFilesRenameCmd())
|
||||
cmd.AddCommand(prjFilesDeleteCmd())
|
||||
cmd.AddCommand(prjFilesDedupeCmd())
|
||||
// Convenience aliases into oo docs (md↔docx / OCR pipeline).
|
||||
cmd.AddCommand(aliasDocsAsMD())
|
||||
cmd.AddCommand(aliasDocsPutMD())
|
||||
cmd.AddCommand(aliasDocsPutTxt())
|
||||
cmd.AddCommand(aliasDocsPutXlsx())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func aliasDocsAsMD() *cobra.Command {
|
||||
c := docsAsMDCmd()
|
||||
c.Use = "as-md FILE_ID"
|
||||
c.Short = "Alias of `oo docs as-md` — download OO file as Markdown (OCR if needed)"
|
||||
return c
|
||||
}
|
||||
|
||||
func aliasDocsPutMD() *cobra.Command {
|
||||
c := docsPutMDCmd()
|
||||
c.Use = "put-md PROJECT_ID MARKDOWN_PATH"
|
||||
c.Short = "Alias of `oo docs put-md` — Markdown→DOCX upload into project"
|
||||
return c
|
||||
}
|
||||
|
||||
func aliasDocsPutTxt() *cobra.Command {
|
||||
c := docsPutTxtCmd()
|
||||
c.Use = "put-txt PROJECT_ID TEXT_PATH"
|
||||
c.Short = "Alias of `oo docs put-txt` — plain text→DOCX upload into project"
|
||||
return c
|
||||
}
|
||||
|
||||
func aliasDocsPutXlsx() *cobra.Command {
|
||||
c := docsPutXlsxCmd()
|
||||
c.Use = "put-xlsx PROJECT_ID [LOCAL_XLSX]"
|
||||
c.Short = "Alias of `oo docs put-xlsx` — generate/upload XLSX with formulas"
|
||||
return c
|
||||
}
|
||||
|
||||
func prjFilesListCmd() *cobra.Command {
|
||||
var showFolders bool
|
||||
cmd := &cobra.Command{
|
||||
@@ -73,9 +108,12 @@ func prjFilesListCmd() *cobra.Command {
|
||||
}
|
||||
|
||||
func prjFilesUploadCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
var replace, allowDuplicate bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "upload PROJECT_ID LOCAL_PATH [LOCAL_PATH...]",
|
||||
Short: "Upload file(s) into the project's Documents folder",
|
||||
Short: "Upload file(s) into the project's Documents folder (upsert by stem|ext)",
|
||||
Long: `Default: replace an existing file with the same logical name (stem|ext), like cp overwrite.
|
||||
Pass --no-replace to fail when the name is taken; --allow-duplicate to always create a new file id.`,
|
||||
Args: cobra.MinimumNArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
@@ -84,15 +122,31 @@ func prjFilesUploadCmd() *cobra.Command {
|
||||
}
|
||||
pid := args[0]
|
||||
for _, p := range args[1:] {
|
||||
entry, err := c.UploadProjectFile(cmd.Context(), pid, p)
|
||||
var entry *onlyoffice.FileEntry
|
||||
var deleted []int
|
||||
switch {
|
||||
case allowDuplicate:
|
||||
entry, err = c.UploadProjectFile(cmd.Context(), pid, p)
|
||||
case replace:
|
||||
entry, deleted, err = c.UploadProjectFileReplacing(cmd.Context(), pid, p)
|
||||
default:
|
||||
entry, err = c.UploadProjectFileNoClobber(cmd.Context(), pid, p)
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(fileEntryToMap(entry))
|
||||
obj := fileEntryToMap(entry)
|
||||
if len(deleted) > 0 {
|
||||
obj["replaced_file_ids"] = deleted
|
||||
}
|
||||
printObject(obj)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().BoolVar(&replace, "replace", true, "replace same stem|ext in project folder (default)")
|
||||
cmd.Flags().BoolVar(&allowDuplicate, "allow-duplicate", false, "always create a new file even when the name exists")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func prjFilesDownloadCmd() *cobra.Command {
|
||||
@@ -106,20 +160,22 @@ func prjFilesDownloadCmd() *cobra.Command {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
f, err := c.GetFile(cmd.Context(), args[0])
|
||||
ctx := cmd.Context()
|
||||
store := c.Files()
|
||||
e, err := store.Stat(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
path := to
|
||||
if path == "" {
|
||||
path = onlyoffice.SafeLocalFileName(onlyoffice.FileEntryTitle(f))
|
||||
path = onlyoffice.SafeLocalFileName(e.Title)
|
||||
}
|
||||
out, err := os.Create(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer out.Close()
|
||||
n, err := c.DownloadFile(cmd.Context(), args[0], out)
|
||||
n, err := store.Download(ctx, args[0], out)
|
||||
if err != nil {
|
||||
_ = os.Remove(path)
|
||||
return err
|
||||
@@ -146,11 +202,16 @@ func prjFilesRenameCmd() *cobra.Command {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entry, err := c.RenameFile(cmd.Context(), args[0], args[1])
|
||||
store := c.Files()
|
||||
if err := store.Rename(cmd.Context(), args[0], args[1]); err != nil {
|
||||
return err
|
||||
}
|
||||
entry, err := store.Stat(cmd.Context(), args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(fileEntryToMap(entry))
|
||||
entry.Title = args[1]
|
||||
printObject(entryToMap(entry))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
@@ -175,7 +236,7 @@ func prjFilesDeleteCmd() *cobra.Command {
|
||||
}
|
||||
ids = append(ids, id)
|
||||
}
|
||||
if err := c.DeleteFiles(cmd.Context(), ids); err != nil {
|
||||
if err := c.Files().Delete(cmd.Context(), args); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"deleted": ids})
|
||||
@@ -184,6 +245,62 @@ func prjFilesDeleteCmd() *cobra.Command {
|
||||
}
|
||||
}
|
||||
|
||||
func prjFilesDedupeCmd() *cobra.Command {
|
||||
var apply, cross bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "dedupe PROJECT_ID",
|
||||
Short: "Find (and optionally remove) duplicate files in project Documents folders",
|
||||
Long: `Duplicates share the same logical name: stem|ext (OO title+fileExst).
|
||||
|
||||
Default: dry-run report. Pass --apply to delete older copies (keeps newest; --cross prefers non-trash folders).`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
groups, deleted, err := c.DedupeProject(cmd.Context(), args[0], onlyoffice.DedupOptions{
|
||||
CrossFolder: cross,
|
||||
}, apply)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(groups))
|
||||
for _, g := range groups {
|
||||
row := map[string]any{
|
||||
"key": g.Key,
|
||||
"folder_id": g.FolderID,
|
||||
"folder_title": g.FolderTitle,
|
||||
"keep_id": fileIDStr(g.Keep),
|
||||
"keep_title": onlyoffice.FileEntryTitle(g.Keep),
|
||||
"remove_count": len(g.Remove),
|
||||
}
|
||||
removeIDs := make([]string, 0, len(g.Remove))
|
||||
for _, f := range g.Remove {
|
||||
removeIDs = append(removeIDs, fileIDStr(f))
|
||||
}
|
||||
row["remove_ids"] = removeIDs
|
||||
rows = append(rows, row)
|
||||
}
|
||||
out := map[string]any{
|
||||
"project_id": args[0],
|
||||
"dry_run": !apply,
|
||||
"cross": cross,
|
||||
"groups": len(groups),
|
||||
"duplicates": rows,
|
||||
}
|
||||
if apply {
|
||||
out["deleted_ids"] = deleted
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().BoolVar(&apply, "apply", false, "delete duplicate files (default: report only)")
|
||||
cmd.Flags().BoolVar(&cross, "cross", false, "also dedupe same stem|ext across folders (prefers non-_trash)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func fileEntryRows(files []*onlyoffice.FileEntry) []map[string]any {
|
||||
rows := make([]map[string]any, 0, len(files))
|
||||
for _, f := range files {
|
||||
@@ -208,6 +325,29 @@ func fileEntryToMap(f *onlyoffice.FileEntry) map[string]any {
|
||||
return m
|
||||
}
|
||||
|
||||
// entryToMap renders a canonical Entry with the same keys as fileEntryToMap.
|
||||
func entryToMap(e onlyoffice.Entry) map[string]any {
|
||||
m := map[string]any{
|
||||
"id": e.ID,
|
||||
"title": e.Title,
|
||||
"fileExst": filepath.Ext(e.Title),
|
||||
"contentLength": contentLengthString(e.Size),
|
||||
}
|
||||
if e.Updated != "" {
|
||||
m["updated"] = e.Updated
|
||||
} else if !e.Modified.IsZero() {
|
||||
m["updated"] = e.Modified.Format(time.RFC3339)
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
func contentLengthString(n int64) string {
|
||||
if n <= 0 {
|
||||
return ""
|
||||
}
|
||||
return strconv.FormatInt(n, 10)
|
||||
}
|
||||
|
||||
func fileIDStr(f *onlyoffice.FileEntry) string {
|
||||
if f == nil || f.ID == nil {
|
||||
return ""
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/eslider/go-onlyoffice/cmd/internal/bootstrap"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
func init() {
|
||||
rootCmd.AddCommand(searchCmd())
|
||||
}
|
||||
|
||||
// searchCmd queries the OnlyOffice document index through the file facade. The
|
||||
// REST /api/2.0/files/@search endpoint only searches file names in the database;
|
||||
// content search needs Elasticsearch (see docs/elasticsearch.md).
|
||||
func searchCmd() *cobra.Command {
|
||||
var (
|
||||
content bool
|
||||
folder string
|
||||
limit int
|
||||
backend string
|
||||
asJSON bool
|
||||
substring bool
|
||||
)
|
||||
cmd := &cobra.Command{
|
||||
Use: "search QUERY...",
|
||||
Short: "Full-text search over documents by name, optionally by content (Elasticsearch)",
|
||||
Long: "Search the OnlyOffice Documents index.\n\n" +
|
||||
"By default only file names are matched. With --content the query also\n" +
|
||||
"matches extracted document text (document.attachment.content); this covers\n" +
|
||||
"Office formats (docx/xlsx/pptx) and is slower.\n\n" +
|
||||
"--backend own queries the separate index populated by `oo index`\n" +
|
||||
"(ONLYOFFICE_ES_TEXT_INDEX, default oo_docs_text) instead, which also holds\n" +
|
||||
"PDFs and scans (see docs/elasticsearch.md).\n\n" +
|
||||
"Requires ONLYOFFICE_ES_URL (and optionally ONLYOFFICE_ES_INDEX,\n" +
|
||||
"ONLYOFFICE_TENANT). See docs/elasticsearch.md for the tunnel setup.",
|
||||
Args: cobra.MinimumNArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if asJSON {
|
||||
outputFormat = "json"
|
||||
}
|
||||
bootstrap.LoadEnv()
|
||||
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
|
||||
var (
|
||||
searcher onlyoffice.Searcher
|
||||
err error
|
||||
)
|
||||
switch strings.ToLower(strings.TrimSpace(backend)) {
|
||||
case "", "oo", "elasticsearch":
|
||||
searcher, err = c.Files().Search()
|
||||
case "own", "es-text":
|
||||
searcher, err = onlyoffice.NewESTextIndex(onlyoffice.ESTextConfigFromEnv())
|
||||
default:
|
||||
return fmt.Errorf("unknown search backend %q (want oo|own)", backend)
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
hits, err := searcher.Search(cmd.Context(), onlyoffice.SearchQuery{
|
||||
Text: strings.Join(args, " "),
|
||||
InContent: content,
|
||||
FolderID: folder,
|
||||
Limit: limit,
|
||||
Substring: substring,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(hits))
|
||||
for _, h := range hits {
|
||||
folderPath := h.Path
|
||||
if len(folderPath) == 0 && h.ParentID != "" {
|
||||
folderPath = []string{h.ParentID}
|
||||
}
|
||||
rows = append(rows, map[string]any{
|
||||
"path": c.UniquePath(cmd.Context(), folderPath, h.Title),
|
||||
"id": h.ID,
|
||||
"title": h.Title,
|
||||
"folder": h.ParentID,
|
||||
"score": h.Score,
|
||||
"highlight": h.Highlight,
|
||||
})
|
||||
}
|
||||
if outputFormat == "json" {
|
||||
printJSON(rows)
|
||||
return nil
|
||||
}
|
||||
printTable([]string{"path", "id", "title", "folder", "score", "highlight"}, rows)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().BoolVar(&content, "content", false, "also match extracted document content")
|
||||
cmd.Flags().BoolVar(&substring, "substring", false, "case-insensitive *term* title match; multiple QUERY args are ANDed")
|
||||
cmd.Flags().StringVar(&folder, "folder", "", "limit to a Documents folder id (matches the folder subtree)")
|
||||
cmd.Flags().IntVar(&limit, "limit", 20, "maximum number of results")
|
||||
cmd.Flags().StringVar(&backend, "backend", "oo", "index to query: oo (OnlyOffice) | own (oo index)")
|
||||
cmd.Flags().BoolVar(&asJSON, "json", false, "shorthand for --output json")
|
||||
return cmd
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestSearchCommandRegisteredWithFlags(t *testing.T) {
|
||||
cmd, _, err := rootCmd.Find([]string{"search"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if cmd.Name() != "search" {
|
||||
t.Fatalf("search resolved to %q", cmd.Name())
|
||||
}
|
||||
for _, name := range []string{"content", "folder", "limit", "json"} {
|
||||
if cmd.Flags().Lookup(name) == nil {
|
||||
t.Errorf("search: missing --%s flag", name)
|
||||
}
|
||||
}
|
||||
if got := cmd.Flags().Lookup("limit").DefValue; got != "20" {
|
||||
t.Errorf("--limit default = %q, want 20", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSearchWithoutESURLIsClearError(t *testing.T) {
|
||||
clearEnv(t, "ONLYOFFICE_ES_URL", "ONLYOFFICE_ES_INDEX", "ONLYOFFICE_TENANT")
|
||||
errBuf := &bytes.Buffer{}
|
||||
rootCmd.SetErr(errBuf)
|
||||
rootCmd.SetOut(&bytes.Buffer{})
|
||||
rootCmd.SetArgs([]string{"search", "Rechnung"})
|
||||
t.Cleanup(func() {
|
||||
rootCmd.SetArgs(nil)
|
||||
rootCmd.SetOut(nil)
|
||||
rootCmd.SetErr(nil)
|
||||
})
|
||||
|
||||
err := rootCmd.Execute()
|
||||
if err == nil {
|
||||
t.Fatal("expected error without ONLYOFFICE_ES_URL")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "ONLYOFFICE_ES_URL") {
|
||||
t.Fatalf("error %q missing ONLYOFFICE_ES_URL", err.Error())
|
||||
}
|
||||
}
|
||||
+55
-3
@@ -2,7 +2,11 @@ package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
@@ -82,11 +86,12 @@ func taskGetCmd() *cobra.Command {
|
||||
}
|
||||
|
||||
func taskCreateCmd() *cobra.Command {
|
||||
var project, desc, deadline, prio string
|
||||
var project, desc, deadline, start, prio string
|
||||
var milestone int64
|
||||
cmd := &cobra.Command{
|
||||
Use: "create TITLE",
|
||||
Aliases: []string{"add"},
|
||||
Short: "Create a project task",
|
||||
Short: "Create a project task (Gantt bar when --start/--deadline/--milestone set)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
@@ -100,6 +105,51 @@ func taskCreateCmd() *cobra.Command {
|
||||
case "low":
|
||||
p = -1
|
||||
}
|
||||
if start != "" || milestone != 0 {
|
||||
pid, err := strconv.Atoi(project)
|
||||
if err != nil && project != "" {
|
||||
return fmt.Errorf("project id: %w", err)
|
||||
}
|
||||
if project == "" {
|
||||
pid, err = strconv.Atoi(os.Getenv("OO_PROJECT_ID"))
|
||||
if err != nil {
|
||||
return fmt.Errorf("--project or OO_PROJECT_ID required")
|
||||
}
|
||||
}
|
||||
req := onlyoffice.NewProjectTaskRequest{
|
||||
ProjectId: pid,
|
||||
Title: args[0],
|
||||
Description: desc,
|
||||
Priority: p,
|
||||
MilestoneId: int(milestone),
|
||||
}
|
||||
if deadline == "" {
|
||||
deadline = time.Now().Add(14 * 24 * time.Hour).Format("2006-01-02")
|
||||
}
|
||||
d, err := time.Parse("2006-01-02", deadline)
|
||||
if err != nil {
|
||||
return fmt.Errorf("deadline: %w", err)
|
||||
}
|
||||
req.Deadline = onlyoffice.Time(d)
|
||||
if start != "" {
|
||||
s, err := time.Parse("2006-01-02", start)
|
||||
if err != nil {
|
||||
return fmt.Errorf("start: %w", err)
|
||||
}
|
||||
req.StartDate = onlyoffice.Time(s)
|
||||
} else {
|
||||
req.StartDate = onlyoffice.Time(time.Now())
|
||||
}
|
||||
task, err := c.CreateProjectTask(req)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{
|
||||
"id": derefInt(task.ID),
|
||||
"title": derefString(task.Title),
|
||||
})
|
||||
return nil
|
||||
}
|
||||
out, err := c.AddTask(cmd.Context(), project, args[0], desc, p, deadline)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -110,7 +160,9 @@ func taskCreateCmd() *cobra.Command {
|
||||
}
|
||||
cmd.Flags().StringVarP(&project, "project", "p", "", "project id (default $OO_PROJECT_ID)")
|
||||
cmd.Flags().StringVar(&desc, "description", "", "description")
|
||||
cmd.Flags().StringVar(&deadline, "deadline", "", "deadline YYYY-MM-DD (default now+14d; always assigned to you)")
|
||||
cmd.Flags().StringVar(&deadline, "deadline", "", "deadline YYYY-MM-DD (default now+14d)")
|
||||
cmd.Flags().StringVar(&start, "start", "", "start YYYY-MM-DD (uses CreateProjectTask / Gantt)")
|
||||
cmd.Flags().Int64Var(&milestone, "milestone", 0, "milestone id (Gantt row)")
|
||||
cmd.Flags().StringVar(&prio, "priority", "normal", "high|normal|low")
|
||||
return cmd
|
||||
}
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
// Command ooscan recursively lists OnlyOffice Documents folders into a TSV
|
||||
// index: file_id, folder_id, path, title.
|
||||
//
|
||||
// Usage: ooscan <FOLDER_ID> [<FOLDER_ID>...]
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
)
|
||||
|
||||
func main() {
|
||||
ctx := context.Background()
|
||||
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
|
||||
seen := map[string]bool{}
|
||||
for _, root := range os.Args[1:] {
|
||||
walk(ctx, c, root, "", 0, seen)
|
||||
}
|
||||
}
|
||||
|
||||
func walk(ctx context.Context, c *onlyoffice.Client, folderID, path string, depth int, seen map[string]bool) {
|
||||
if depth > 8 || seen[folderID] {
|
||||
return
|
||||
}
|
||||
seen[folderID] = true
|
||||
// Throttle: OnlyOffice rate-limits (429) and the host must not be flooded.
|
||||
time.Sleep(350 * time.Millisecond)
|
||||
ctx, cancel := context.WithTimeout(ctx, 60*time.Second)
|
||||
defer cancel()
|
||||
var l *onlyoffice.DavListing
|
||||
derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
var err error
|
||||
l, err = c.ListDavFolder(ctx, folderID)
|
||||
return err
|
||||
})
|
||||
if derr != nil {
|
||||
fmt.Fprintf(os.Stderr, "list %s (%s): %v\n", path, folderID, derr)
|
||||
return
|
||||
}
|
||||
for _, f := range l.Files {
|
||||
fmt.Printf("%s\t%s\t%s\t%s\n", f.ID, folderID, path, f.Title)
|
||||
}
|
||||
for _, sub := range l.Folders {
|
||||
walk(ctx, c, sub.ID, path+"/"+sub.Title, depth+1, seen)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,308 @@
|
||||
// Command pdfamount walks a Documents folder, downloads matching PDFs and
|
||||
// extracts the payable amount, printing "file_id\ttitle\tamount".
|
||||
//
|
||||
// Usage: pdfamount <FOLDER_ID> [TITLE_FILTER_REGEX]
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
)
|
||||
|
||||
// amountPat is the amount capture shared by every amount regex.
|
||||
const amountPat = `([0-9]+(?:[.,][0-9]+)*)`
|
||||
|
||||
// amountRE builds "<label> [optional (comment)] [: -] <number>".
|
||||
func amountRE(label string) *regexp.Regexp {
|
||||
return regexp.MustCompile(
|
||||
`(?i)\b` + regexp.QuoteMeta(label) + `\b\s*(?:\([^)]*\))?\s*[:\-]?\s*` + amountPat)
|
||||
}
|
||||
|
||||
// amountRes lists the payable-amount patterns in strict priority order: the
|
||||
// first pattern with a usable amount wins, and a lower-priority label can never
|
||||
// override a higher-priority one ("zu zahlender betrag" > "rechnungsbetrag" >
|
||||
// "rechnungsendbetrag" > "gesamtbetrag" > "gesamtsumme (inkl. steuern)").
|
||||
//
|
||||
// "gesamtbetrag" and "gesamtsumme" are not in the original set but are the real
|
||||
// labels on Diashop invoices ("Gesamtsumme (inkl. Steuern)"). The inclusive
|
||||
// variant is matched before a plain "gesamtsumme". Everything after those
|
||||
// primary labels is the broader fallback set, consulted only when no primary
|
||||
// label yields an amount. Within one pattern the last usable amount is taken,
|
||||
// because totals usually come last.
|
||||
var amountRes = []*regexp.Regexp{
|
||||
amountRE("zu zahlender betrag"),
|
||||
amountRE("rechnungsbetrag"),
|
||||
amountRE("rechnungsendbetrag"),
|
||||
amountRE("gesamtbetrag"),
|
||||
regexp.MustCompile(`(?i)\bgesamtsumme\b\s*\(\s*inkl\.?\s*steuern\s*\)\s*[:\-]?\s*` + amountPat),
|
||||
amountRE("gesamtsumme"),
|
||||
amountRE("endbetrag"),
|
||||
amountRE("zahlbetrag"),
|
||||
amountRE("bruttobetrag"),
|
||||
amountRE("betrag"),
|
||||
amountRE("total"),
|
||||
amountRE("summe"),
|
||||
}
|
||||
|
||||
// taxLineRe marks a line whose number is a tax rate/percentage: an explicit
|
||||
// percent sign or a VAT/tax keyword. "Steuern" (plural, as in "inkl. Steuern")
|
||||
// is handled separately so the inclusive total stays usable.
|
||||
var taxLineRe = regexp.MustCompile(`(?i)%|\bMwSt\b|\bUSt\b|\bProzent\b`)
|
||||
|
||||
// steuerRe finds "Steuer"/"Umsatzsteuer" etc. RE2 has no lookahead, so the
|
||||
// plural "Steuern" is excluded in isTaxLine.
|
||||
var steuerRe = regexp.MustCompile(`(?i)steuer`)
|
||||
|
||||
// percentAfterRe detects a percent sign directly after a number (spaces ok).
|
||||
var percentAfterRe = regexp.MustCompile(`^\s*%`)
|
||||
|
||||
func main() {
|
||||
if len(os.Args) < 2 {
|
||||
fmt.Fprintln(os.Stderr, "usage: pdfamount <FOLDER_ID> [TITLE_FILTER_REGEX]")
|
||||
os.Exit(2)
|
||||
}
|
||||
folder := os.Args[1]
|
||||
filter := regexp.MustCompile(`(?i)rechnung`)
|
||||
if len(os.Args) >= 3 {
|
||||
filter = regexp.MustCompile(os.Args[2])
|
||||
}
|
||||
ctx := context.Background()
|
||||
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
|
||||
|
||||
files := listAll(ctx, c, folder)
|
||||
for _, f := range files {
|
||||
if !filter.MatchString(f.title) {
|
||||
continue
|
||||
}
|
||||
if !strings.HasSuffix(strings.ToLower(f.title), ".pdf") {
|
||||
continue
|
||||
}
|
||||
amount, err := pdfAmount(ctx, c, f.id)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", f.title, err)
|
||||
continue
|
||||
}
|
||||
if amount == "" {
|
||||
continue
|
||||
}
|
||||
fmt.Printf("%s\t%s\t%s\n", f.id, f.title, amount)
|
||||
}
|
||||
}
|
||||
|
||||
type file struct{ id, title string }
|
||||
|
||||
func listAll(ctx context.Context, c *onlyoffice.Client, folder string) []file {
|
||||
seen := map[string]bool{}
|
||||
var out []file
|
||||
var walk func(string)
|
||||
walk = func(id string) {
|
||||
if seen[id] {
|
||||
return
|
||||
}
|
||||
seen[id] = true
|
||||
time.Sleep(300 * time.Millisecond)
|
||||
l, err := c.ListDavFolder(ctx, id)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "list %s: %v\n", id, err)
|
||||
return
|
||||
}
|
||||
for _, f := range l.Files {
|
||||
out = append(out, file{f.ID, f.Title})
|
||||
}
|
||||
for _, sub := range l.Folders {
|
||||
walk(sub.ID)
|
||||
}
|
||||
}
|
||||
walk(folder)
|
||||
return out
|
||||
}
|
||||
|
||||
func pdfAmount(ctx context.Context, c *onlyoffice.Client, id string) (string, error) {
|
||||
time.Sleep(time.Second)
|
||||
tmp, err := os.CreateTemp("", "pdf-*.pdf")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer os.Remove(tmp.Name())
|
||||
derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
_ = tmp.Truncate(0)
|
||||
_, _ = tmp.Seek(0, 0)
|
||||
_, err := c.DownloadFile(ctx, id, tmp)
|
||||
return err
|
||||
})
|
||||
if derr != nil {
|
||||
tmp.Close()
|
||||
return "", derr
|
||||
}
|
||||
tmp.Close()
|
||||
var buf bytes.Buffer
|
||||
cmd := exec.CommandContext(ctx, "pdftotext", "-layout", tmp.Name(), "-")
|
||||
cmd.Stdout = &buf
|
||||
if err := cmd.Run(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return extractAmount(buf.String()), nil
|
||||
}
|
||||
|
||||
// extractAmount returns the normalised ("1234.56") payable amount found in
|
||||
// text, or "" if no usable amount matches.
|
||||
//
|
||||
// DKV invoices are special-cased first: they repeat a per-vehicle "TOTAL:" line
|
||||
// and carry the real total only in the "Gesamtsummenaufstellung" section.
|
||||
func extractAmount(text string) string {
|
||||
if v, ok := dkvGrandTotal(text); ok {
|
||||
return v
|
||||
}
|
||||
for _, re := range amountRes {
|
||||
if v, ok := lastUsableAmount(text, re); ok {
|
||||
return v
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// dkvGrandTotal extracts the total of a DKV "Gesamtsummenaufstellung" section.
|
||||
//
|
||||
// Rule: DKV invoices repeat a per-vehicle "TOTAL:" line, so the last TOTAL is
|
||||
// not the invoice total. When a "Gesamtsummenaufstellung" section exists, its
|
||||
// total wins over every "TOTAL:" line: the first amount after the "»" marker,
|
||||
// or, if there is none, the last amount in the section. The section ends at the
|
||||
// page break (form feed) or end of text.
|
||||
func dkvGrandTotal(text string) (string, bool) {
|
||||
idx := strings.Index(strings.ToLower(text), "gesamtsummenaufstellung")
|
||||
if idx < 0 {
|
||||
return "", false
|
||||
}
|
||||
section := text[idx:]
|
||||
if ff := strings.IndexByte(section, '\f'); ff >= 0 {
|
||||
section = section[:ff]
|
||||
}
|
||||
if m := strings.Index(section, "»"); m >= 0 {
|
||||
if v, ok := firstAmount(section[m:]); ok {
|
||||
return v, true
|
||||
}
|
||||
}
|
||||
return lastAmount(section)
|
||||
}
|
||||
|
||||
// lastUsableAmount returns the last amount matched by re that is not a tax rate
|
||||
// or percentage. Within one label the last usable amount wins.
|
||||
func lastUsableAmount(text string, re *regexp.Regexp) (string, bool) {
|
||||
ms := re.FindAllStringSubmatchIndex(text, -1)
|
||||
for i := len(ms) - 1; i >= 0; i-- {
|
||||
m := ms[i]
|
||||
if isTaxRate(text, m[2], m[3]) {
|
||||
continue
|
||||
}
|
||||
if v, ok := normalizeAmount(text[m[2]:m[3]]); ok {
|
||||
return v, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// isTaxRate reports whether the number at text[start:end] is a tax rate or a
|
||||
// percentage instead of a payable amount. A candidate is rejected when the
|
||||
// token right after the number is "%" or the number's line carries a percent
|
||||
// sign or a tax keyword. Rejecting is deliberate: office matching treats a
|
||||
// known-but-different amount as a hard disqualifier, so an empty result is
|
||||
// safer than the VAT rate.
|
||||
func isTaxRate(text string, start, end int) bool {
|
||||
if percentAfterRe.MatchString(text[end:]) {
|
||||
return true
|
||||
}
|
||||
lineStart := strings.LastIndexByte(text[:start], '\n') + 1
|
||||
line := text[lineStart:]
|
||||
if n := strings.IndexByte(text[end:], '\n'); n >= 0 {
|
||||
line = text[lineStart : end+n]
|
||||
}
|
||||
return isTaxLine(line)
|
||||
}
|
||||
|
||||
// isTaxLine reports whether a line looks like a tax rate rather than a payable
|
||||
// amount. "Steuern" is treated as a qualifier ("inkl. Steuern"), not a rate.
|
||||
func isTaxLine(line string) bool {
|
||||
if taxLineRe.MatchString(line) {
|
||||
return true
|
||||
}
|
||||
for _, loc := range steuerRe.FindAllStringIndex(line, -1) {
|
||||
if loc[1] >= len(line) || (line[loc[1]] != 'n' && line[loc[1]] != 'N') {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// numberRe finds bare numbers (with optional thousands/decimal separators).
|
||||
var numberRe = regexp.MustCompile(`[0-9]+(?:[.,][0-9]+)*`)
|
||||
|
||||
func firstAmount(s string) (string, bool) {
|
||||
for _, m := range numberRe.FindAllString(s, -1) {
|
||||
if v, ok := normalizeAmount(m); ok {
|
||||
return v, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
func lastAmount(s string) (string, bool) {
|
||||
ms := numberRe.FindAllString(s, -1)
|
||||
for i := len(ms) - 1; i >= 0; i-- {
|
||||
if v, ok := normalizeAmount(ms[i]); ok {
|
||||
return v, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// normalizeAmount turns "1.234,56" (DE), "1,234.56" (EN) or "1234.56" into
|
||||
// "1234.56". The rightmost separator is decimal only when followed by one or
|
||||
// two digits; otherwise every separator is a thousands separator.
|
||||
func normalizeAmount(s string) (string, bool) {
|
||||
last := -1
|
||||
for i := 0; i < len(s); i++ {
|
||||
if s[i] == '.' || s[i] == ',' {
|
||||
last = i
|
||||
}
|
||||
}
|
||||
var dec byte
|
||||
if last >= 0 {
|
||||
digits := 0
|
||||
for i := last + 1; i < len(s); i++ {
|
||||
if s[i] < '0' || s[i] > '9' {
|
||||
return "", false
|
||||
}
|
||||
digits++
|
||||
}
|
||||
if digits == 1 || digits == 2 {
|
||||
dec = s[last]
|
||||
}
|
||||
}
|
||||
var b strings.Builder
|
||||
for i := 0; i < len(s); i++ {
|
||||
switch c := s[i]; {
|
||||
case c >= '0' && c <= '9':
|
||||
b.WriteByte(c)
|
||||
case (c == '.' || c == ',') && c == dec:
|
||||
b.WriteByte('.')
|
||||
case c == '.' || c == ',':
|
||||
// thousands separator
|
||||
default:
|
||||
return "", false
|
||||
}
|
||||
}
|
||||
v, err := strconv.ParseFloat(b.String(), 64)
|
||||
if err != nil {
|
||||
return "", false
|
||||
}
|
||||
return strconv.FormatFloat(v, 'f', 2, 64), true
|
||||
}
|
||||
@@ -0,0 +1,175 @@
|
||||
package main
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestExtractAmount(t *testing.T) {
|
||||
tests := []struct {
|
||||
name, text, want string
|
||||
}{
|
||||
{
|
||||
name: "rechnungsbetrag de format",
|
||||
text: "Rechnungsbetrag: 1.234,56 €",
|
||||
want: "1234.56",
|
||||
},
|
||||
{
|
||||
name: "rechnungsbetrag en thousands and dot",
|
||||
text: "Rechnungsbetrag: 1,234.56",
|
||||
want: "1234.56",
|
||||
},
|
||||
{
|
||||
name: "rechnungsbetrag plain dot",
|
||||
text: "Rechnungsbetrag: 1234.56",
|
||||
want: "1234.56",
|
||||
},
|
||||
{
|
||||
name: "rechnungsbetrag de comma only",
|
||||
text: "Rechnungsbetrag: 1234,56",
|
||||
want: "1234.56",
|
||||
},
|
||||
{
|
||||
name: "currency suffix eur",
|
||||
text: "Rechnungsbetrag: 1.234,56 EUR",
|
||||
want: "1234.56",
|
||||
},
|
||||
{
|
||||
name: "zu zahlender betrag wins over rechnungsbetrag",
|
||||
text: "Zu zahlender Betrag: 10,00\nRechnungsbetrag: 99,00",
|
||||
want: "10.00",
|
||||
},
|
||||
{
|
||||
name: "rechnungsbetrag wins over endbetrag",
|
||||
text: "Endbetrag: 20,00\nRechnungsbetrag: 30,00",
|
||||
want: "30.00",
|
||||
},
|
||||
{
|
||||
name: "bruttobetrag wins over bare betrag",
|
||||
text: "Bruttobetrag: 50,00\nBetrag: 10,00",
|
||||
want: "50.00",
|
||||
},
|
||||
{
|
||||
name: "gesamtbetrag wins over bare betrag",
|
||||
text: "Gesamtbetrag: 80,00\nBetrag: 10,00",
|
||||
want: "80.00",
|
||||
},
|
||||
{
|
||||
name: "last occurrence of same label wins",
|
||||
text: "Rechnungsbetrag: 10,00\nRechnungsbetrag: 20,00",
|
||||
want: "20.00",
|
||||
},
|
||||
{
|
||||
name: "endbetrag fallback",
|
||||
text: "Endbetrag: 42,00",
|
||||
want: "42.00",
|
||||
},
|
||||
{
|
||||
name: "zahlbetrag fallback without colon",
|
||||
text: "Zahlbetrag 7,50 €",
|
||||
want: "7.50",
|
||||
},
|
||||
{
|
||||
name: "rechnungsendbetrag beats endbetrag",
|
||||
text: "Rechnungsendbetrag: 12,00\nEndbetrag: 13,00",
|
||||
want: "12.00",
|
||||
},
|
||||
{
|
||||
name: "dkv style total line",
|
||||
text: "Kundenbezogene Daten\n» TOTAL: 123,45 100,00 23,45 123,45\n",
|
||||
want: "123.45",
|
||||
},
|
||||
{
|
||||
name: "dkv gesamtsummenaufstellung grand total after marker",
|
||||
text: "» TOTAL: 111,11 100,00 11,11 111,11\n" +
|
||||
"» TOTAL: 222,22 200,00 22,22 222,22\n" +
|
||||
"Gesamtsummenaufstellung\n" +
|
||||
"Netto 240,00\n" +
|
||||
"MwSt 47,25\n" +
|
||||
"» 287,25\n",
|
||||
want: "287.25",
|
||||
},
|
||||
{
|
||||
name: "dkv gesamtsummenaufstellung total on next line",
|
||||
text: "» TOTAL: 111,11\nGesamtsummenaufstellung\n»\n287,25\n",
|
||||
want: "287.25",
|
||||
},
|
||||
{
|
||||
name: "tax rate with percent sign is not an amount",
|
||||
text: "Betrag: 19,00 % MwSt",
|
||||
want: "",
|
||||
},
|
||||
{
|
||||
name: "mehrwertsteuer rate is not an amount",
|
||||
text: "Gesamtsumme: 19,00% MwSt",
|
||||
want: "",
|
||||
},
|
||||
{
|
||||
name: "steuer word on the number line rejects it",
|
||||
text: "Betrag: 2,83 Steuer",
|
||||
want: "",
|
||||
},
|
||||
{
|
||||
name: "rejected primary falls back to a usable label",
|
||||
text: "Gesamtsumme: 19,00 % MwSt\nEndbetrag: 42,00",
|
||||
want: "42.00",
|
||||
},
|
||||
{
|
||||
name: "labeled zu zahlender betrag beats unlabeled larger number",
|
||||
text: "unlabeled 999,99\nZu zahlender Betrag: 10,00",
|
||||
want: "10.00",
|
||||
},
|
||||
{
|
||||
name: "labeled zu zahlender betrag beats lower label larger number",
|
||||
text: "Endbetrag: 999,99\nZu zahlender Betrag: 10,00",
|
||||
want: "10.00",
|
||||
},
|
||||
{
|
||||
name: "diashop style gesamtsumme with comment",
|
||||
text: "Zwischensumme\n12,34 €\nZwischensumme\n12,34 €\nVersand & Bearbeitung\n4,95 €\nGesamtsumme (inkl. Steuern)\n17,29 €\n",
|
||||
want: "17.29",
|
||||
},
|
||||
{
|
||||
name: "diashop picks inclusive total last",
|
||||
text: "Gesamtsumme (exkl. Steuern)\n12,34 €\nGesamtsumme (inkl. Steuern)\n17,29 €",
|
||||
want: "17.29",
|
||||
},
|
||||
{
|
||||
name: "no label",
|
||||
text: "some text without any amount label 12,34",
|
||||
want: "",
|
||||
},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if got := extractAmount(tc.text); got != tc.want {
|
||||
t.Fatalf("extractAmount()=%q want %q", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeAmount(t *testing.T) {
|
||||
tests := []struct {
|
||||
in string
|
||||
want string
|
||||
ok bool
|
||||
}{
|
||||
{"1.234,56", "1234.56", true},
|
||||
{"1,234.56", "1234.56", true},
|
||||
{"1234.56", "1234.56", true},
|
||||
{"1234,56", "1234.56", true},
|
||||
{"1.234.567,89", "1234567.89", true},
|
||||
{"1,234,567.89", "1234567.89", true},
|
||||
{"1.234", "1234.00", true},
|
||||
{"12,5", "12.50", true},
|
||||
{"12", "12.00", true},
|
||||
{"", "0.00", false},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
got, ok := normalizeAmount(tc.in)
|
||||
if ok != tc.ok {
|
||||
t.Fatalf("normalizeAmount(%q) ok=%v want %v", tc.in, ok, tc.ok)
|
||||
}
|
||||
if ok && got != tc.want {
|
||||
t.Fatalf("normalizeAmount(%q)=%q want %q", tc.in, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// History entities that OnlyOffice CRM actually accepts for history notes.
|
||||
// There is NO person/contact history in this API version: POST /api/2.0/crm/history.json
|
||||
// returns 400 "Value does not fall within the expected range." for entityType
|
||||
// contact/person/people/client/member. Verified against a live instance (#74).
|
||||
const (
|
||||
HistoryEntityOpportunity = "opportunity"
|
||||
HistoryEntityCase = "case"
|
||||
)
|
||||
|
||||
// IsCompany reports whether a CRM contact row is a company (vs a person).
|
||||
// The field arrives as JSON bool; be liberal about what we accept.
|
||||
func IsCompany(person map[string]any) bool {
|
||||
b, _ := person["isCompany"].(bool)
|
||||
return b
|
||||
}
|
||||
|
||||
// ContactID returns the CRM id of a contact row as a plain string.
|
||||
func ContactID(row map[string]any) string {
|
||||
return fmt.Sprint(row["id"])
|
||||
}
|
||||
|
||||
// BuildContactEmailIndex scans all persons once and maps lowercase email →
|
||||
// contact id. Use this instead of calling FindPersonByEmail per address:
|
||||
// the index is O(N) over the whole CRM, the per-address lookup is O(N×M).
|
||||
func (c *Client) BuildContactEmailIndex(ctx context.Context) (map[string]string, error) {
|
||||
all, err := c.ListAllContacts(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
index := make(map[string]string, len(all)*2)
|
||||
for _, person := range all {
|
||||
if IsCompany(person) {
|
||||
continue
|
||||
}
|
||||
id := ContactID(person)
|
||||
for _, row := range ContactInfoRows(person) {
|
||||
if NormalizeContactInfoType(fmt.Sprint(row["infoType"])) != "email" {
|
||||
continue
|
||||
}
|
||||
email := strings.ToLower(strings.TrimSpace(fmt.Sprint(row["data"])))
|
||||
if email != "" && email != "<nil>" {
|
||||
index[email] = id
|
||||
}
|
||||
}
|
||||
}
|
||||
return index, nil
|
||||
}
|
||||
|
||||
// BuildPersonOpportunityIndex maps every opportunity member's contact id to a
|
||||
// deterministic representative opportunity: the one with the lowest numeric id.
|
||||
// OnlyOffice has no person-level history, so notes for a person go on their
|
||||
// deal — this index answers "which deal" in one pass.
|
||||
func (c *Client) BuildPersonOpportunityIndex(ctx context.Context) (map[string]string, error) {
|
||||
opps, err := c.ListAllOpportunities(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
index := map[string]string{}
|
||||
for _, opp := range opps {
|
||||
oppID := ContactID(opp)
|
||||
for _, member := range OpportunityMembers(opp) {
|
||||
pid := ContactID(member)
|
||||
if cur, ok := index[pid]; !ok || NumericIDLess(oppID, cur) {
|
||||
index[pid] = oppID
|
||||
}
|
||||
}
|
||||
}
|
||||
return index, nil
|
||||
}
|
||||
|
||||
// NumericIDLess compares two string ids numerically when possible, falling
|
||||
// back to lexicographic order so results stay deterministic either way.
|
||||
func NumericIDLess(a, b string) bool {
|
||||
na, errA := strconv.Atoi(strings.TrimSpace(a))
|
||||
nb, errB := strconv.Atoi(strings.TrimSpace(b))
|
||||
if errA == nil && errB == nil && na != nb {
|
||||
return na < nb
|
||||
}
|
||||
return a < b
|
||||
}
|
||||
|
||||
// SortIDs orders id strings deterministically (numeric first, then lexical).
|
||||
func SortIDs(ids []string) {
|
||||
sort.Strings(ids)
|
||||
sort.SliceStable(ids, func(i, j int) bool { return NumericIDLess(ids[i], ids[j]) })
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
package onlyoffice
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestIsCompany(t *testing.T) {
|
||||
if IsCompany(map[string]any{"isCompany": true}) != true {
|
||||
t.Fatal("true row not detected")
|
||||
}
|
||||
if IsCompany(map[string]any{"isCompany": false}) {
|
||||
t.Fatal("false row detected as company")
|
||||
}
|
||||
if IsCompany(map[string]any{}) {
|
||||
t.Fatal("missing field detected as company")
|
||||
}
|
||||
if IsCompany(nil) {
|
||||
t.Fatal("nil row detected as company")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNumericIDLess(t *testing.T) {
|
||||
cases := []struct {
|
||||
a, b string
|
||||
want bool
|
||||
}{
|
||||
{"9", "10", true},
|
||||
{"1747", "1748", true},
|
||||
{"abc", "abd", true},
|
||||
{"10", "9", false},
|
||||
{" 12 ", "13", true},
|
||||
{"x1", "2", false}, // non-numeric falls back lexical: "x1" > "2"
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := NumericIDLess(c.a, c.b); got != c.want {
|
||||
t.Errorf("NumericIDLess(%q,%q)=%v want %v", c.a, c.b, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSortIDs(t *testing.T) {
|
||||
ids := []string{"20", "3", "100", "1"}
|
||||
SortIDs(ids)
|
||||
want := "1 3 20 100"
|
||||
got := ""
|
||||
for i, id := range ids {
|
||||
if i > 0 {
|
||||
got += " "
|
||||
}
|
||||
got += id
|
||||
}
|
||||
if got != want {
|
||||
t.Fatalf("SortIDs=%q want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestContactID(t *testing.T) {
|
||||
if ContactID(map[string]any{"id": float64(42)}) != "42" {
|
||||
t.Fatal("numeric id formatting broken")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHistoryEntityConstants(t *testing.T) {
|
||||
if HistoryEntityOpportunity != "opportunity" || HistoryEntityCase != "case" {
|
||||
t.Fatal("history entity whitelist drifted from live-verified values")
|
||||
}
|
||||
}
|
||||
@@ -2,7 +2,7 @@ package onlyoffice
|
||||
|
||||
// Minimal CRM helpers: contacts, opportunities, cases, tasks, and history notes.
|
||||
// These expose untyped maps for flexibility — they are primarily consumed by
|
||||
// cmd/oo and the applications-sync workflow.
|
||||
// cmd/oo and CRM sync tooling.
|
||||
|
||||
import (
|
||||
"context"
|
||||
@@ -15,10 +15,15 @@ import (
|
||||
)
|
||||
|
||||
// ListContacts returns a page of CRM contacts and the total count.
|
||||
// sortBy=id is always set: OnlyOffice filter.json without an explicit sort
|
||||
// order is non-deterministic on large contact sets, so a paged walk
|
||||
// (ListAllContacts, ListContactsByTag, FindCompany, FindPerson) can skip or
|
||||
// duplicate contacts across page boundaries.
|
||||
func (c *Client) ListContacts(ctx context.Context, count, startIndex int, search string) ([]map[string]any, int, error) {
|
||||
q := url.Values{}
|
||||
q.Set("count", strconv.Itoa(count))
|
||||
q.Set("startIndex", strconv.Itoa(startIndex))
|
||||
q.Set("sortBy", "id")
|
||||
if search != "" {
|
||||
q.Set("filterValue", search)
|
||||
}
|
||||
@@ -188,21 +193,41 @@ func (c *Client) CreatePerson(ctx context.Context, first, last string, companyID
|
||||
}
|
||||
|
||||
// UpdatePerson updates first/last name and optional company link on a person.
|
||||
// companyID == 0 leaves the company association unchanged.
|
||||
// companyID == 0 leaves the company association unchanged (re-sends current
|
||||
// company id when present). OnlyOffice ignores companyId/about on form-encoded
|
||||
// PUT and may unlink the company when companyId is omitted — use JSON body.
|
||||
func (c *Client) UpdatePerson(ctx context.Context, personID, first, last string, companyID int, jobTitle, about string) (map[string]any, error) {
|
||||
fields := url.Values{}
|
||||
fields.Set("firstName", first)
|
||||
fields.Set("lastName", last)
|
||||
body := map[string]any{
|
||||
"firstName": first,
|
||||
"lastName": last,
|
||||
}
|
||||
if companyID != 0 {
|
||||
fields.Set("companyId", strconv.Itoa(companyID))
|
||||
body["companyId"] = companyID
|
||||
} else {
|
||||
// Preserve existing employer: omitted companyId unlinks on this API.
|
||||
if cur, err := c.GetContact(ctx, personID); err == nil {
|
||||
if co, ok := cur["company"].(map[string]any); ok {
|
||||
if id := flexInt(co["id"]); id != 0 {
|
||||
body["companyId"] = id
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if jobTitle != "" {
|
||||
fields.Set("jobTitle", jobTitle)
|
||||
body["jobTitle"] = jobTitle
|
||||
}
|
||||
if about != "" {
|
||||
fields.Set("about", about)
|
||||
body["about"] = about
|
||||
}
|
||||
return c.putFormObject(ctx, fmt.Sprintf("/api/2.0/crm/contact/person/%s.json", url.PathEscape(personID)), fields)
|
||||
out, err := c.putJSONObject(ctx, fmt.Sprintf("/api/2.0/crm/contact/person/%s.json", url.PathEscape(personID)), body)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
// PUT response body is often stale; re-fetch for authoritative fields.
|
||||
if fresh, gerr := c.GetContact(ctx, personID); gerr == nil && fresh != nil {
|
||||
out = fresh
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// AddContactInfo attaches an email/website/phone/etc. to a contact.
|
||||
@@ -223,6 +248,88 @@ func (c *Client) DeleteContact(ctx context.Context, contactID string) (map[strin
|
||||
return c.deleteObject(ctx, fmt.Sprintf("/api/2.0/crm/contact/%s.json", url.PathEscape(contactID)))
|
||||
}
|
||||
|
||||
// UpdateContactName renames a CRM company contact displayName.
|
||||
// Uses the company endpoint (person names go through /crm/contact/person/{id}).
|
||||
func (c *Client) UpdateContactName(ctx context.Context, contactID, newName string) (map[string]any, error) {
|
||||
body := map[string]any{
|
||||
"displayName": newName,
|
||||
"companyName": newName,
|
||||
"isCompany": true,
|
||||
}
|
||||
out, err := c.putJSONObject(ctx, fmt.Sprintf("/api/2.0/crm/contact/company/%s.json", url.PathEscape(contactID)), body)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if fresh, gerr := c.GetContact(ctx, contactID); gerr == nil && fresh != nil {
|
||||
out = fresh
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// ListContactTags returns all CRM contact tags (title + relativeItemsCount).
|
||||
func (c *Client) ListContactTags(ctx context.Context) ([]map[string]any, error) {
|
||||
return c.ResponseArray(ctx, "/api/2.0/crm/contact/tag.json")
|
||||
}
|
||||
|
||||
// CreateContactTag creates a contact tag by name. Idempotent: "already exists" is OK.
|
||||
func (c *Client) CreateContactTag(ctx context.Context, tagName string) error {
|
||||
fields := url.Values{}
|
||||
fields.Set("tagName", tagName)
|
||||
_, err := c.postFormObject(ctx, "/api/2.0/crm/contact/tag.json", fields)
|
||||
if err != nil && strings.Contains(err.Error(), "already exists") {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// AddContactTag attaches tagName to a contact.
|
||||
func (c *Client) AddContactTag(ctx context.Context, contactID, tagName string) error {
|
||||
fields := url.Values{}
|
||||
fields.Set("tagName", tagName)
|
||||
_, err := c.postFormObject(ctx, fmt.Sprintf("/api/2.0/crm/contact/%s/tag.json", url.PathEscape(contactID)), fields)
|
||||
return err
|
||||
}
|
||||
|
||||
// RemoveContactTag removes tagName from a contact.
|
||||
func (c *Client) RemoveContactTag(ctx context.Context, contactID, tagName string) error {
|
||||
fields := url.Values{}
|
||||
fields.Set("tagName", tagName)
|
||||
raw, err := c.deleteForm(ctx, fmt.Sprintf("/api/2.0/crm/contact/%s/tag.json", url.PathEscape(contactID)), fields)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
_, err = unmarshalResponseObject(raw)
|
||||
return err
|
||||
}
|
||||
|
||||
// ListContactsByTag returns contacts filtered by a single CRM tag name.
|
||||
func (c *Client) ListContactsByTag(ctx context.Context, tagName string, count, startIndex int) ([]map[string]any, int, error) {
|
||||
if count <= 0 {
|
||||
count = 50
|
||||
}
|
||||
q := url.Values{}
|
||||
q.Set("count", strconv.Itoa(count))
|
||||
q.Set("startIndex", strconv.Itoa(startIndex))
|
||||
q.Set("tags", tagName)
|
||||
q.Set("sortBy", "id")
|
||||
raw, err := c.getJSON(ctx, "/api/2.0/crm/contact/filter.json?"+q.Encode())
|
||||
if err != nil {
|
||||
return nil, 0, err
|
||||
}
|
||||
var env struct {
|
||||
Response []map[string]any `json:"response"`
|
||||
Total int `json:"total"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &env); err != nil {
|
||||
return nil, 0, err
|
||||
}
|
||||
total := env.Total
|
||||
if total == 0 && len(env.Response) > 0 {
|
||||
total = len(env.Response)
|
||||
}
|
||||
return env.Response, total, nil
|
||||
}
|
||||
|
||||
// ListAllContacts paginates through every CRM contact.
|
||||
func (c *Client) ListAllContacts(ctx context.Context) ([]map[string]any, error) {
|
||||
const page = 100
|
||||
@@ -580,7 +687,7 @@ func (c *Client) CreateCRMTask(ctx context.Context, title, deadline string, cate
|
||||
deadline = time.Now().Add(14 * 24 * time.Hour).Format("2006-01-02T15:04:05")
|
||||
}
|
||||
if categoryID == 0 {
|
||||
categoryID = 2 // Opportunity — matches applications sync
|
||||
categoryID = 2 // Opportunity (default CRM task category)
|
||||
}
|
||||
uid, err := c.SelfUserID(ctx)
|
||||
if err != nil {
|
||||
@@ -634,6 +741,11 @@ func (c *Client) DeleteCRMTask(ctx context.Context, id string) (map[string]any,
|
||||
return c.deleteObject(ctx, fmt.Sprintf("/api/2.0/crm/task/%s.json", url.PathEscape(id)))
|
||||
}
|
||||
|
||||
// CloseCRMTask closes (completes) a CRM task via the task close endpoint.
|
||||
func (c *Client) CloseCRMTask(ctx context.Context, id string) (map[string]any, error) {
|
||||
return c.putFormObject(ctx, fmt.Sprintf("/api/2.0/crm/task/%s/close.json", url.PathEscape(id)), url.Values{})
|
||||
}
|
||||
|
||||
// ListTaskCategories returns CRM task categories.
|
||||
func (c *Client) ListTaskCategories(ctx context.Context) ([]map[string]any, error) {
|
||||
return c.ResponseArray(ctx, "/api/2.0/crm/task/category.json")
|
||||
|
||||
+172
@@ -0,0 +1,172 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"strings"
|
||||
"unicode"
|
||||
)
|
||||
|
||||
var (
|
||||
parenSuffixRE = regexp.MustCompile(`(?i)\s*[\(\[\{][^)\]\}]*[\)\]\}]\s*$`)
|
||||
dashCompanyRE = regexp.MustCompile(`(?i)\s+[-–—]\s+[A-Za-z0-9].*$`)
|
||||
emailLocalRE = regexp.MustCompile(`(?i)^[a-z0-9._%+\-]+@[a-z0-9.\-]+\.[a-z]{2,}$`)
|
||||
nonNameTokenRE = regexp.MustCompile(`[^a-zA-ZÀ-öø-ÿĀ-ž0-9'’.\-]+`)
|
||||
)
|
||||
|
||||
// CleanPersonNames strips company annotations from display names and fills
|
||||
// first/last from email local-part when the source used an address as the name.
|
||||
// Company affiliation belongs on CRM companyId — never in LastName.
|
||||
func CleanPersonNames(first, last, display, org string, emails []string) (cleanFirst, cleanLast string) {
|
||||
first = strings.TrimSpace(first)
|
||||
last = strings.TrimSpace(last)
|
||||
display = strings.TrimSpace(display)
|
||||
org = strings.TrimSpace(org)
|
||||
|
||||
if looksLikeEmail(first) {
|
||||
ef, el := GuessNameFromEmail(first)
|
||||
first, last = ef, el
|
||||
}
|
||||
if looksLikeEmail(display) && first == "" && last == "" {
|
||||
display = ""
|
||||
}
|
||||
|
||||
if first == "" && last == "" && display != "" {
|
||||
first, last = SplitDisplayName(display)
|
||||
}
|
||||
|
||||
first = stripCompanyAnnotation(first, org)
|
||||
last = stripCompanyAnnotation(last, org)
|
||||
|
||||
last = stripCompanyAnnotation(last, org)
|
||||
if i := strings.IndexAny(first, "(["); i > 0 {
|
||||
first = strings.TrimSpace(first[:i])
|
||||
}
|
||||
if org != "" && personLastIsOrg(last, org) {
|
||||
last = ""
|
||||
}
|
||||
|
||||
if (first == "" || looksLikeEmail(first)) && len(emails) > 0 {
|
||||
ef, el := GuessNameFromEmail(emails[0])
|
||||
if first == "" || looksLikeEmail(first) {
|
||||
first = ef
|
||||
}
|
||||
if last == "" || last == "-" {
|
||||
last = el
|
||||
}
|
||||
}
|
||||
|
||||
first = strings.TrimSpace(first)
|
||||
last = strings.TrimSpace(last)
|
||||
if last == "" {
|
||||
last = "-"
|
||||
}
|
||||
return first, last
|
||||
}
|
||||
|
||||
func stripCompanyAnnotation(s, org string) string {
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return ""
|
||||
}
|
||||
s = parenSuffixRE.ReplaceAllString(s, "")
|
||||
s = strings.TrimSpace(s)
|
||||
s = dashCompanyRE.ReplaceAllString(s, "")
|
||||
s = strings.TrimSpace(s)
|
||||
if org != "" {
|
||||
for _, sep := range []string{" - ", " – ", " — ", " / "} {
|
||||
if i := strings.LastIndex(strings.ToLower(s), strings.ToLower(sep+org)); i >= 0 {
|
||||
s = strings.TrimSpace(s[:i])
|
||||
}
|
||||
}
|
||||
suf := " (" + org + ")"
|
||||
if strings.HasSuffix(strings.ToLower(s), strings.ToLower(suf)) {
|
||||
s = strings.TrimSpace(s[:len(s)-len(suf)])
|
||||
}
|
||||
}
|
||||
return strings.TrimSpace(s)
|
||||
}
|
||||
|
||||
func personLastIsOrg(last, org string) bool {
|
||||
last = NormalizePersonNameKey(last)
|
||||
org = NormalizePersonNameKey(org)
|
||||
if last == "" || org == "" {
|
||||
return false
|
||||
}
|
||||
if last == org {
|
||||
return true
|
||||
}
|
||||
return strings.HasPrefix(org, last+" ") || strings.HasPrefix(org, last+",")
|
||||
}
|
||||
|
||||
func looksLikeEmail(s string) bool {
|
||||
return emailLocalRE.MatchString(strings.TrimSpace(s))
|
||||
}
|
||||
|
||||
// GuessNameFromEmail turns local@domain into Title-Case first/last when the
|
||||
// local part looks like first.last / first_last / first-last.
|
||||
func GuessNameFromEmail(email string) (first, last string) {
|
||||
email = NormalizeContactEmail(email)
|
||||
local, _, ok := strings.Cut(email, "@")
|
||||
if !ok || local == "" {
|
||||
return "", ""
|
||||
}
|
||||
local = strings.Split(local, "+")[0]
|
||||
parts := strings.FieldsFunc(local, func(r rune) bool {
|
||||
return r == '.' || r == '_' || r == '-'
|
||||
})
|
||||
if len(parts) == 0 {
|
||||
return titleNameToken(local), ""
|
||||
}
|
||||
if len(parts) == 1 {
|
||||
return titleNameToken(parts[0]), ""
|
||||
}
|
||||
return titleNameToken(parts[0]), titleNameToken(strings.Join(parts[1:], " "))
|
||||
}
|
||||
|
||||
func titleNameToken(s string) string {
|
||||
s = nonNameTokenRE.ReplaceAllString(s, " ")
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return ""
|
||||
}
|
||||
runes := []rune(strings.ToLower(s))
|
||||
runes[0] = unicode.ToTitle(runes[0])
|
||||
return string(runes)
|
||||
}
|
||||
|
||||
// FormatProjectTitle builds "CC | Company | Title" (spaces around |).
|
||||
// Country should be a short code (DE, TF, UA, …). Empty segments are dropped.
|
||||
func FormatProjectTitle(country, company, title string) string {
|
||||
parts := make([]string, 0, 3)
|
||||
for _, p := range []string{country, company, title} {
|
||||
p = strings.TrimSpace(p)
|
||||
p = strings.ReplaceAll(p, "|", "/")
|
||||
if p != "" {
|
||||
parts = append(parts, p)
|
||||
}
|
||||
}
|
||||
return strings.Join(parts, " | ")
|
||||
}
|
||||
|
||||
// NormalizeContactEmail lowercases and trims.
|
||||
func NormalizeContactEmail(s string) string {
|
||||
return strings.ToLower(strings.TrimSpace(s))
|
||||
}
|
||||
|
||||
// NormalizePersonNameKey collapses whitespace and lowercases for matching.
|
||||
func NormalizePersonNameKey(s string) string {
|
||||
fields := strings.Fields(strings.ToLower(strings.TrimSpace(s)))
|
||||
return strings.Join(fields, " ")
|
||||
}
|
||||
|
||||
// SplitDisplayName splits "First Last …" into first/last (last = remainder).
|
||||
func SplitDisplayName(name string) (first, last string) {
|
||||
parts := strings.Fields(strings.TrimSpace(name))
|
||||
if len(parts) == 0 {
|
||||
return "", ""
|
||||
}
|
||||
if len(parts) == 1 {
|
||||
return parts[0], ""
|
||||
}
|
||||
return parts[0], strings.Join(parts[1:], " ")
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
package onlyoffice
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestCleanPersonNames_stripsCompanyParen(t *testing.T) {
|
||||
f, l := CleanPersonNames("Jane", "Doe (Acme)", "Jane Doe (Acme)", "Acme", nil)
|
||||
if f != "Jane" || l != "Doe" {
|
||||
t.Fatalf("got %q %q", f, l)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCleanPersonNames_stripsDashCompany(t *testing.T) {
|
||||
f, l := CleanPersonNames("Jane", "Doe - Acme", "", "Acme", nil)
|
||||
if f != "Jane" || l != "Doe" {
|
||||
t.Fatalf("got %q %q", f, l)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCleanPersonNames_fromEmail(t *testing.T) {
|
||||
f, l := CleanPersonNames("jane.doe@example.com", "-", "", "Acme",
|
||||
[]string{"jane.doe@example.com"})
|
||||
if f != "Jane" || l != "Doe" {
|
||||
t.Fatalf("got %q %q", f, l)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCleanPersonNames_lastIsCompany(t *testing.T) {
|
||||
f, l := CleanPersonNames("Jane", "Acme", "", "Acme GmbH & Co. KG", nil)
|
||||
if f != "Jane" || l != "-" {
|
||||
t.Fatalf("got %q %q", f, l)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFormatProjectTitle(t *testing.T) {
|
||||
got := FormatProjectTitle("DE", "Acme", "Map App")
|
||||
if got != "DE | Acme | Map App" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
}
|
||||
+1
-1
@@ -69,7 +69,7 @@ func StripCompanySuffix(title string) string {
|
||||
}
|
||||
|
||||
// FixDealTitle strips a leading @, normalizes separator spacing, and collapses
|
||||
// empty-position titles like " @ contoso" to "contoso".
|
||||
// empty-position titles like " @ Acme" to "Acme".
|
||||
func FixDealTitle(s string) string {
|
||||
s = strings.TrimSpace(s)
|
||||
for strings.HasPrefix(s, "@") {
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
# rclone WebDAV mount of the oo-webdav sidecar (OnlyOffice Documents).
|
||||
#
|
||||
# The sidecar publishes the portal's WebDAV tree on the docker bridge
|
||||
# (http://172.17.0.1:8098, prefix /webdav). rclone mounts it inside this
|
||||
# container as a normal filesystem.
|
||||
#
|
||||
# Credentials are OnlyOffice portal credentials (the same the `oo` CLI uses).
|
||||
# Export them before `docker compose up`, e.g.
|
||||
# set -a; . .secrets/oo.env; set +a
|
||||
# or copy .env.example to .env and fill it. Only names live in git.
|
||||
#
|
||||
# docker compose -f deploy/docker-compose.rclone-webdav.yml up -d
|
||||
# docker exec rclone-webdav ls /mnt/onlyoffice
|
||||
# docker compose -f deploy/docker-compose.rclone-webdav.yml down
|
||||
#
|
||||
# The mount point is the named volume `onlyoffice-mnt`; join it from another
|
||||
# container with `external_volumes` to read the tree without the API.
|
||||
name: rclone-webdav
|
||||
|
||||
services:
|
||||
rclone-webdav:
|
||||
image: rclone/rclone:latest
|
||||
container_name: rclone-webdav
|
||||
restart: unless-stopped
|
||||
# rclone mount needs FUSE. The remote is defined on the fly via flags for
|
||||
# the mount; the same credentials are written to an rclone config so that
|
||||
# `docker exec rclone-webdav rclone ls webdav:` works without flags.
|
||||
entrypoint: ["/bin/sh", "-c"]
|
||||
command:
|
||||
- |
|
||||
set -eu
|
||||
: "$${ONLYOFFICE_USER:?ONLYOFFICE_USER is required}"
|
||||
: "$${ONLYOFFICE_PASSWORD:?ONLYOFFICE_PASSWORD is required}"
|
||||
obscured=$$(rclone obscure "$${ONLYOFFICE_PASSWORD}")
|
||||
mkdir -p /config/rclone
|
||||
cat > /config/rclone/rclone.conf <<EOF
|
||||
[webdav]
|
||||
type = webdav
|
||||
url = $${ONLYOFFICE_WEBDAV_URL}
|
||||
vendor = other
|
||||
user = $${ONLYOFFICE_USER}
|
||||
pass = $${obscured}
|
||||
EOF
|
||||
exec rclone mount :webdav: /mnt/onlyoffice \
|
||||
--webdav-url "$${ONLYOFFICE_WEBDAV_URL}" \
|
||||
--webdav-user "$${ONLYOFFICE_USER}" \
|
||||
--webdav-pass "$${obscured}" \
|
||||
--vfs-cache-mode writes \
|
||||
--cache-dir /cache \
|
||||
--dir-cache-time 1m \
|
||||
--allow-other \
|
||||
--allow-non-empty
|
||||
environment:
|
||||
ONLYOFFICE_USER: ${ONLYOFFICE_USER:?ONLYOFFICE_USER is required}
|
||||
ONLYOFFICE_PASSWORD: ${ONLYOFFICE_PASSWORD:-${ONLYOFFICE_PASS:?ONLYOFFICE_PASSWORD is required}}
|
||||
ONLYOFFICE_WEBDAV_URL: ${ONLYOFFICE_WEBDAV_URL:-http://172.17.0.1:8098/webdav}
|
||||
cap_add:
|
||||
- SYS_ADMIN
|
||||
devices:
|
||||
- /dev/fuse
|
||||
# Ubuntu hosts run AppArmor; FUSE mounts need it unconfined.
|
||||
security_opt:
|
||||
- apparmor:unconfined
|
||||
volumes:
|
||||
- onlyoffice-mnt:/mnt/onlyoffice
|
||||
- rclone-cache:/cache
|
||||
|
||||
volumes:
|
||||
onlyoffice-mnt:
|
||||
rclone-cache:
|
||||
@@ -0,0 +1,25 @@
|
||||
---
|
||||
type: reference
|
||||
status: current
|
||||
related:
|
||||
- README.md
|
||||
---
|
||||
|
||||
# go-onlyoffice — docs
|
||||
|
||||
Индекс справочников. Общее — [README.md](../README.md), правила — [AGENTS.md](../AGENTS.md).
|
||||
|
||||
## Файлы
|
||||
- [unified-file-client.md](unified-file-client.md) — единый файловый клиент:
|
||||
`Entry`/`FileStore`/`FileClient`, бэкенды REST/DAV/SQL/ES, env, как добавить
|
||||
бэкенд.
|
||||
- [community-server-db.md](community-server-db.md) — read-only SQL-бэкенд
|
||||
(MySQL/PostgreSQL): схема, SSH-туннель, DSN, MinIO download.
|
||||
- [elasticsearch.md](elasticsearch.md) — поиск: индекс OnlyOffice `files_file`
|
||||
и свой `oo_docs_text` (PDF/сканы), туннель.
|
||||
- [rclone-webdav.md](rclone-webdav.md) — rclone-монтирование Documents
|
||||
(`deploy/docker-compose.rclone-webdav.yml`), smoke, ограничения.
|
||||
- [crm-associations.md](crm-associations.md) — правила ассоциаций CRM.
|
||||
|
||||
## Тесты
|
||||
Команды и туннели — раздел Testing в [README.md](../README.md#testing).
|
||||
@@ -0,0 +1,179 @@
|
||||
---
|
||||
type: reference
|
||||
status: current
|
||||
related:
|
||||
- README.md
|
||||
- file_pg.go
|
||||
- docs/elasticsearch.md
|
||||
---
|
||||
|
||||
# Community Server DB — прямой SQL-доступ (read-only)
|
||||
|
||||
## Что это
|
||||
|
||||
Бэкенд `pgStore` (`file_pg.go`) читает файлы и папки **напрямую из БД
|
||||
Community Server**, без HTTP-слоя. Реализует `FileStore` (`List`/`Stat`/
|
||||
`Download`) и `Searcher` по имени. Запись запрещена: все write-методы
|
||||
возвращают `ErrReadOnly`.
|
||||
|
||||
## Что за БД (research, live)
|
||||
|
||||
Проверено на VM `onlyoffice-v2` (SSH `127.0.0.1:32`):
|
||||
|
||||
- Community Server работает на **MySQL 8.0**, не на PostgreSQL.
|
||||
- Хост: `127.0.0.1:3306` внутри VM, база `onlyoffice`.
|
||||
- Конфиг: `/etc/onlyoffice/communityserver/appsettings.production.json`,
|
||||
`providerName: MySql.Data.MySqlClient`.
|
||||
- Таблицы: `files_file`, `files_folder`, `files_folder_tree`,
|
||||
`files_security`, тенанты — `tenants_tenants` (не `tenants`).
|
||||
- PostgreSQL 16 в той же VM — **наш** контур (`edw_docs`, роли `edw`/`edw_ro`,
|
||||
office-assistant), к OnlyOffice отношения не имеет. `files_file` в PG нет.
|
||||
- Портал хранит файлы в **S3/MinIO** (DiscStorage только для мелочи).
|
||||
Бакет `office`, объект — по ключу (см. ниже).
|
||||
|
||||
Вывод: бэкенд назван по issue «PostgreSQL», но живой источник — MySQL.
|
||||
`database/sql` + драйвер по DSN: `mysql` для MySQL, `pgx` для PostgreSQL.
|
||||
`Name()` возвращает фактический движок (`mysql` или `postgres`).
|
||||
|
||||
## Схема
|
||||
|
||||
`files_file` — одна строка **на версию** (PK `tenant_id, id, version`):
|
||||
|
||||
| поле | смысл |
|
||||
|------|-------|
|
||||
| `id` | id файла (тот же, что в REST/ES) |
|
||||
| `version` | номер версии этой строки |
|
||||
| `version_group` | номер версии |
|
||||
| `current_version` | `1` = текущая версия, `0` = старая |
|
||||
| `folder_id` | id родительской папки |
|
||||
| `title` | имя файла с расширением |
|
||||
| `content_length` | размер в байтах |
|
||||
| `create_on`, `modified_on` | даты (UTC, без зоны) |
|
||||
| `tenant_id` | тенант (портал) |
|
||||
|
||||
`files_folder`: `id`, `parent_id`, `title`, `create_on`, `modified_on`,
|
||||
`tenant_id`. `files_folder_tree`: `folder_id`, `parent_id`, `level` — готовое
|
||||
дерево, пока не используется.
|
||||
|
||||
Текущую строку файла берём по `current_version = 1`.
|
||||
|
||||
## Доступ (SSH-туннель)
|
||||
|
||||
MySQL слушает только `127.0.0.1:3306` внутри VM. Снаружи — SSH-туннель
|
||||
(SSH в VM открыт как `127.0.0.1:32`):
|
||||
|
||||
```bash
|
||||
ssh -f -N -o ControlMaster=no -o ControlPath=none \
|
||||
-p 32 -i ~/.ssh/id_ed25519 \
|
||||
-L 3306:127.0.0.1:3306 root@127.0.0.1
|
||||
|
||||
# MySQL DSN затем:
|
||||
# root:<pw>@tcp(127.0.0.1:3306)/onlyoffice?parseTime=true
|
||||
```
|
||||
|
||||
Любой свободный локальный порт подойдёт (напр. `13306`); тогда тот же порт —
|
||||
в DSN. `-o ControlMaster=no -o ControlPath=none` обязательны: иначе forward
|
||||
уходит в persistent master из `~/.ssh/config`.
|
||||
|
||||
Креды MySQL — в конфиге Community Server внутри VM:
|
||||
`/etc/onlyoffice/communityserver/appsettings.production.json` →
|
||||
`ConnectionStrings.connectionString` (поля `User ID`, `Password`), база
|
||||
`onlyoffice`. В самом MySQL-контейнере (`onlyoffice-mysql-server`) база пустая;
|
||||
рабочий сервер — host-mysqld на `127.0.0.1:3306` (207 таблиц). Не печатать
|
||||
пароль.
|
||||
|
||||
## Переменные
|
||||
|
||||
| env | default | смысл |
|
||||
|-----|---------|-------|
|
||||
| `ONLYOFFICE_DSN` | — | DSN драйвера (MySQL `...@tcp(...)/...` или `postgres://...`) |
|
||||
| `ONLYOFFICE_PG_DRIVER` | авто | `postgres` или `mysql`; иначе по форме DSN |
|
||||
| `ONLYOFFICE_PG_TENANT` | `ONLYOFFICE_TENANT` | фильтр `tenant_id` (пусто = все) |
|
||||
| `ONLYOFFICE_PG_HOST/PORT/USER/PASSWORD/DBNAME/SSLMODE` | — | собрать PG DSN, если `ONLYOFFICE_DSN` пуст |
|
||||
|
||||
Имена — в [`.env.example`](../.env.example). Секретов нет.
|
||||
|
||||
## Использование
|
||||
|
||||
Напрямую: `NewPGStore(PGConfigFromEnv())`.
|
||||
|
||||
Через фасад (эпик #34): SQL-стор регистрируется на `FileClient`. После этого
|
||||
`Read()` и все чтения (`Stat`/`List`) идут в БД, `Write()` остаётся REST/DAV.
|
||||
|
||||
```go
|
||||
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
|
||||
|
||||
sql, err := c.SQLFileStore() // открыть из env; caller закрывает
|
||||
if err != nil { /* нет DSN / нет связи */ }
|
||||
if closer, ok := sql.(interface{ Close() error }); ok { defer closer.Close() }
|
||||
|
||||
f := c.Files()
|
||||
f.RegisterStore(onlyoffice.ProviderPG, sql)
|
||||
e, _ := f.Stat(ctx, "19423") // e.Provider == "mysql" — ответил SQL
|
||||
```
|
||||
|
||||
`Client.FileStore("pg"|"sql"|"postgres"|"mysql")` тоже отдаёт SQL-стор
|
||||
(открывает из env). Если DSN нет/битый — возвращается не `nil`, а заглушка,
|
||||
чей метод отдаёт ошибку открытия; ошибку как таковую даёт `SQLFileStore()`.
|
||||
Различить бэкенд в ответе можно по `Entry.Provider` (`mysql` у SQL, `rest` у
|
||||
REST).
|
||||
|
||||
## Download (MinIO)
|
||||
|
||||
`Download` не ходит в REST. Ключ объекта собирается из строки `files_file`:
|
||||
|
||||
```
|
||||
00/00/<tenant>/files/folder_<shard>/file_<id>/v<version>/content.<ext>
|
||||
shard = (id/1000 + 1) * 1000
|
||||
```
|
||||
|
||||
`shard` — не `folder_id`, а следующая тысяча над `id` (файл 3727 →
|
||||
`folder_4000`). Проверено live по бакету `office`.
|
||||
|
||||
Стриминг переиспользует `downloadMinioObject` из `storage_fallback.go`
|
||||
(та же подпись SigV4 и `MINIO_*`), без дублирования.
|
||||
|
||||
Ограничение: схема валидна только для файлов, лежащих в **MinIO/S3** (старые
|
||||
папки). Файлы в **Disc**-хранилище портала (`Data/Products/Files/...`, новые
|
||||
папки) по этому ключу недоступны — `Download` вернёт `404`. Если
|
||||
`MINIO_ACCESS_KEY`/`MINIO_SECRET_KEY` не заданы, `Download` вернёт явную
|
||||
ошибку; `Stat`/`List`/`Search` работают и без них.
|
||||
|
||||
## Тесты
|
||||
|
||||
```bash
|
||||
go test ./... # unit: rebind, csObjectKey, маппинг
|
||||
go test -race ./...
|
||||
|
||||
# integration (нужен DSN; skip без него)
|
||||
ONLYOFFICE_DSN='root:<pw>@tcp(127.0.0.1:3306)/onlyoffice?parseTime=true' \
|
||||
ONLYOFFICE_PG_TENANT=1 \
|
||||
ONLYOFFICE_PG_TEST_FILE_ID=19423 \
|
||||
ONLYOFFICE_PG_TEST_FOLDER_ID=676 \
|
||||
go test -tags=integration -run 'TestIntegrationPGStore|TestIntegrationSQLFacade' -v ./...
|
||||
|
||||
# плюс MINIO_* для сверки Download с REST (иначе этот шаг skip)
|
||||
MINIO_ENDPOINT=http://127.0.0.1:9000 MINIO_BUCKET=office \
|
||||
MINIO_ACCESS_KEY=... MINIO_SECRET_KEY=... \
|
||||
go test -tags=integration -run TestIntegrationPGStore -v ./...
|
||||
```
|
||||
|
||||
- `TestIntegrationPGStore` — `Stat`/`List`/`Download` SQL против REST и
|
||||
`ErrReadOnly` у write-методов.
|
||||
- `TestIntegrationSQLFacade` — SQL-стор, зарегистрированный на фасаде, реально
|
||||
обслуживает чтения: `Read().Name()` = SQL-бэкенд, `Entry.Provider == "mysql"`
|
||||
(у REST — `"rest"`), сверка `Stat`/`List` с REST, и прямой
|
||||
`Client.FileStore("pg")`.
|
||||
|
||||
Без `ONLYOFFICE_DSN` оба теста делают чистый `skip`.
|
||||
|
||||
## Грабли
|
||||
|
||||
- MySQL хранит `datetime` без зоны; `parseTime=true` (ставится автоматически)
|
||||
читает их как UTC. REST отдаёт `+02:00` — сравнивать моменты, не строки.
|
||||
- `GetFile` (REST) не отдаёт `contentLength` — размер сверять с `Stat` SQL.
|
||||
- Один файл = много строк `files_file` (по версиям). Без `current_version = 1`
|
||||
получите дубликаты.
|
||||
- `folder_id` не входит в ключ MinIO; ключ считает `shard` от `id`.
|
||||
- Searcher SQL ищет только по имени (`LIKE`). Контент — Elasticsearch
|
||||
([elasticsearch.md](elasticsearch.md)).
|
||||
@@ -0,0 +1,124 @@
|
||||
# CRM associations (company ↔ person ↔ deal ↔ project ↔ invoice ↔ mail)
|
||||
|
||||
Operational rules for the `oo` CLI and this library. Business SSOT remains
|
||||
OnlyOffice Workspace CRM + Projects.
|
||||
|
||||
## Canonical graph
|
||||
|
||||
One **legal company** owns the relationship. Do not invent a second “bill-to”
|
||||
company just for PDF layout.
|
||||
|
||||
```text
|
||||
Company
|
||||
├── Person (buyer contact) oo persons create --company-id
|
||||
├── Opportunity / Deal oo opportunities … ; member-add company + person
|
||||
├── Project (hub) oo projects … ; contacts add company + person
|
||||
│ └── Epic + subtasks
|
||||
└── Invoice (Draft → …) oo invoices create --contact COMPANY --opportunity DEAL
|
||||
└── PDF file oo invoices pdf ID
|
||||
└── Mail draft oo mails draft-invoice --invoice ID --to …
|
||||
```
|
||||
|
||||
| Layer | CLI | Must link |
|
||||
|-------|-----|-----------|
|
||||
| Company | `oo companies create` | website, email, phone, **one** Billing address |
|
||||
| Person | `oo persons create --company-id` / `oo persons update ID` | job title; never encode employer in `lastName`; **update uses JSON** (form PUT ignores `companyId`/`about`) |
|
||||
| Deal | `oo opportunities create` + `member-add` | company **and** person as members |
|
||||
| Project | `oo projects create` + `contacts add` | same company + person |
|
||||
| Invoice | `oo invoices create --contact COMPANY --opportunity DEAL` | `entityId` at **create** |
|
||||
| Mail | `oo mails draft-invoice` | attach current PDF; **do not send** until confirmed |
|
||||
|
||||
UI checks (same company card):
|
||||
|
||||
- `#contacts` → person
|
||||
- `#deals` → opportunity
|
||||
- `#projects` → hub project
|
||||
- `#invoices` on the **deal** → invoice (needs `entity`)
|
||||
- `#files` → preferably **one** current invoice PDF
|
||||
|
||||
**Project Team ≠ Project Contacts.** Team = portal users. CRM people/companies
|
||||
show under the project **Contacts** tab (`oo projects contacts list`).
|
||||
|
||||
## Hard rules
|
||||
|
||||
1. **One company per legal entity.** Duplicate “bill-to” contacts empty Deals /
|
||||
Projects / Contacts tabs and break merge. Prefer
|
||||
`oo contacts merge FROM INTO` (keeps `INTO`) or `oo companies dedupe`.
|
||||
2. **Link invoice → deal at create.**
|
||||
`POST /crm/invoice` with `entityId` + `entityType: 0` (Opportunity).
|
||||
`oo invoices update … --opportunity` often returns **400**
|
||||
(“Value does not fall within the expected range”). If the link is missing,
|
||||
delete the Draft and recreate with `--opportunity`.
|
||||
3. **Bill To = company id**, not a throwaway contact. Person stays under the
|
||||
company (`companyId`). Optional `consigneeId` for Empfänger when the portal
|
||||
template prints it.
|
||||
4. **Stay Draft until mail is ready.** Billed (`status id=2`) is **not editable**
|
||||
via content PUT. Going Billed → Draft via `…/crm/invoice/status/1` usually
|
||||
**does not work** — delete + recreate Draft instead.
|
||||
5. **Do not regenerate PDF in a loop** without cleanup. Each
|
||||
`GET …/crm/invoice/{id}/pdf` attaches a new file to the company (and often
|
||||
the deal). Keep `invoice.fileID`; delete older PDFs with
|
||||
`oo invoices pdf-cleanup ID` / Documents `fileops/delete`.
|
||||
|
||||
## Invoice PDF quirks
|
||||
|
||||
| Symptom | Workaround |
|
||||
|---------|------------|
|
||||
| Cached / stale PDF | Touch invoice (Draft PUT that clears `fileID`), then `GET …/pdf` — `oo invoices pdf ID --force` |
|
||||
| Billing address missing on **new** PDFs | Temporary multiline `companyName` (`Line1\nLine2\n…`) on the **canonical** company → force PDF → restore clean name. Cached `fileID` keeps the multiline Bill To. |
|
||||
| Separate bill-to company for newlines | **Forbidden** — merge back to the real company |
|
||||
| Invoice **number** won’t change on PUT | Delete Draft and recreate with the desired number |
|
||||
| Notizen / Bedingungen spacing | Leading `\n` and blank lines only — no HTML (tags print literally) |
|
||||
| Issuer street lines | Organisation profile address (`street` with `\n`), not only terms |
|
||||
|
||||
Status ids commonly used: `1` Draft, `2` Billed, `3` Rejected, `4` Paid.
|
||||
|
||||
## Mail quirks
|
||||
|
||||
| Symptom | Workaround |
|
||||
|---------|------------|
|
||||
| Signature / body doubles chat URL | Put chat in **one** place only. UI drafts: signature. API send: body (API **does not** append signature). |
|
||||
| Signature / body cuts URL at `#` | Plain text URLs — avoid `<a href="…#…">` (or encode `#` as `%23` in href) |
|
||||
| German letter spacing | Blank `<p> </p>` between blocks (`MailHTMLWithBlankParagraphs`) |
|
||||
| Send | `PUT /api/2.0/mail/messages/send.json` with `id/from/to/subject/body`; omit empty `cc`/`bcc`. Never auto-send; draft only until the human confirms |
|
||||
|
||||
Prefer OnlyOffice Mail (`/addons/mail/#drafts`) for invoice delivery until confirmed.
|
||||
|
||||
## Project / task quirks
|
||||
|
||||
- Hub title: `CC | Company` (e.g. `DE | Acme GmbH`).
|
||||
- Streams = epics/tasks under the hub, not a third title segment (unless the
|
||||
project itself is a named delivery stream).
|
||||
- Closing a **subtask**:
|
||||
`PUT /api/2.0/project/task/{epicId}/{subtaskId}/status` with `status=2`.
|
||||
`oo tasks update SUBTASK -s closed` returns **404** for subtasks.
|
||||
- After deleting a CRM contact, `GET /project/contact/{deletedId}` may still
|
||||
return projects (ghost). Official project contact list should only show live
|
||||
ids; unlink may 400 if the contact is gone.
|
||||
|
||||
## Merge / cleanup cheat sheet
|
||||
|
||||
```bash
|
||||
# Keep the preferred company (INTO), drop the duplicate (FROM)
|
||||
oo contacts merge FROM_ID INTO_ID
|
||||
|
||||
# Or by normalized name (careful — whole CRM)
|
||||
oo companies dedupe
|
||||
|
||||
# Invoice ↔ deal must exist at create
|
||||
oo invoices create --number P-YYYY-NN --contact COMPANY_ID --item ITEM_ID \
|
||||
--price 300 --opportunity DEAL_ID --language de-DE …
|
||||
|
||||
# Fresh PDF + prune older PDFs on company/deal
|
||||
oo invoices pdf INVOICE_ID --force
|
||||
oo invoices pdf-cleanup INVOICE_ID
|
||||
|
||||
# Mail draft (no send)
|
||||
oo mails draft-invoice --invoice INVOICE_ID --to billing@example.com
|
||||
```
|
||||
|
||||
## Related
|
||||
|
||||
- README § invoices / mail / CRM cleanup
|
||||
- Personal workspace tooling (disk inventory, dossier sync): private
|
||||
`git.produktor.io/eSlider/oo-workspace` (`oow` CLI)
|
||||
@@ -0,0 +1,255 @@
|
||||
---
|
||||
type: reference
|
||||
status: current
|
||||
related:
|
||||
- README.md
|
||||
- file_es.go
|
||||
---
|
||||
|
||||
# Elasticsearch — полнотекстовый поиск OnlyOffice
|
||||
|
||||
## Что это
|
||||
|
||||
Полнотекстовый поиск OnlyOffice Workspace работает на **Elasticsearch**.
|
||||
Клиент на сервере — NEST. Индекс — имя таблицы.
|
||||
|
||||
Для файлов индекс `files_file`:
|
||||
|
||||
| поле | тип | смысл |
|
||||
|------|-----|-------|
|
||||
| `id` | integer | id файла (тот же, что в REST/Documents) |
|
||||
| `title` | text (`whitespacecustom`) | имя файла |
|
||||
| `tenantId` | integer | тенант (портал) |
|
||||
| `folders` | nested | список папок: `folderId` (строка), `id`, `tenantId` |
|
||||
| `document.attachment.content` | text (`document`) | извлеченный текст (ingest-attachment) |
|
||||
| `document.attachment.content_type` | text | MIME |
|
||||
|
||||
Важно:
|
||||
- Живой сервер — **Elasticsearch 7.16.3**, кластер `elasticsearch`.
|
||||
- REST `GET /api/2.0/files/@search/{query}` ищет **только по имени в БД**
|
||||
(`fileDao.Search`), ES не задействует. Для поиска по содержимому нужен
|
||||
прямой ES — это и делает `oo search`.
|
||||
- `title` analyzer `whitespacecustom` режет по пробелам и lower-case. Полное
|
||||
имя файла — один токен (`Rechnung-4711.pdf`), поэтому поиск по имени ищет
|
||||
слово целиком, а не подстроку.
|
||||
- `document.attachment.content` заполняется **только для Office-форматов**
|
||||
(docx / xlsx / pptx). У PDF/txt, залитых через API, контент не извлекается.
|
||||
- Индексация асинхронная (TeamLabSvc) — файл появляется в ES не мгновенно.
|
||||
|
||||
## Доступ
|
||||
|
||||
ES слушает `127.0.0.1:9200` **внутри** VM OnlyOffice. Снаружи порт закрыт,
|
||||
SSH в VM открыт на хосте как `127.0.0.1:32` (контейнер `onlyoffice-v2`,
|
||||
QEMU). Схема — SSH-туннель.
|
||||
|
||||
```bash
|
||||
# из корня go-onlyoffice (ключ и хост — как в infra-доках)
|
||||
ssh -f -N -o ControlMaster=no -o ControlPath=none \
|
||||
-p 32 -i ~/.ssh/id_ed25519 \
|
||||
-L 9200:127.0.0.1:9200 root@127.0.0.1
|
||||
|
||||
curl -s http://127.0.0.1:9200/ | head # tagline + version
|
||||
curl -s 'http://127.0.0.1:9200/_cat/indices?h=index,docs.count'
|
||||
```
|
||||
|
||||
`-o ControlMaster=no -o ControlPath=none` обязательны: иначе forward уходит
|
||||
в persistent master-соединение из `~/.ssh/config` и порт остаётся занят.
|
||||
|
||||
Проверить, что туннель жив:
|
||||
|
||||
```bash
|
||||
curl -s http://127.0.0.1:9200/files_file/_count
|
||||
```
|
||||
|
||||
## Переменные
|
||||
|
||||
| env | default | смысл |
|
||||
|-----|---------|-------|
|
||||
| `ONLYOFFICE_ES_URL` | — (обязателен) | `scheme://host:port` ES |
|
||||
| `ONLYOFFICE_ES_INDEX` | `files_file` | индекс |
|
||||
| `ONLYOFFICE_TENANT` | пусто (все) | фильтр `tenantId` |
|
||||
|
||||
Имена — в [`.env.example`](../.env.example). Секретов нет: ES без пароля.
|
||||
|
||||
## CLI
|
||||
|
||||
```bash
|
||||
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 oo search "Rechnung"
|
||||
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 oo search "Mahngebühr" --content
|
||||
oo search "Rechnung" --folder 649 --limit 50 --json
|
||||
```
|
||||
|
||||
Флаги: `--content` (искать и по тексту), `--folder ID` (папка
|
||||
`folders.folderId`), `--limit N` (по умолчанию 20, максимум 200),
|
||||
`--json` = `-o json`.
|
||||
|
||||
## Библиотека
|
||||
|
||||
`file_es.go` — `ESSearcher` (`Name() = "elasticsearch"`), прямой ES REST на
|
||||
stdlib `net/http`:
|
||||
|
||||
```go
|
||||
es, _ := onlyoffice.NewESSearcher(onlyoffice.ESConfigFromEnv())
|
||||
hits, _ := es.Search(ctx, onlyoffice.SearchQuery{
|
||||
Text: "Rechnung", InContent: true, Limit: 20,
|
||||
})
|
||||
```
|
||||
|
||||
Запрос: `multi_match` по `title^2` (+ `document.attachment.content` при
|
||||
`InContent`), фильтры `tenantId` и `folders.folderId`, `_source`
|
||||
id/title/folders, `highlight` для фрагмента. Ответ → `[]SearchHit` (модель из
|
||||
эпика #34; пока объявлена в `file_es.go`, переедет в `file_core.go` с F1 #35).
|
||||
|
||||
## Тесты
|
||||
|
||||
```bash
|
||||
# unit — чистые builders/парсеры, без сети
|
||||
go test ./ -run ES
|
||||
|
||||
# integration — нужен ONLYOFFICE_ES_URL (+ креды REST для залива)
|
||||
set -a; . .env; set +a
|
||||
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 ONLYOFFICE_TENANT=1 \
|
||||
go test -tags=integration -run TestIntegrationESSearch -v .
|
||||
```
|
||||
|
||||
Интеграционный тест заливает временный xlsx (в имени и в ячейке — уникальные
|
||||
токены), ждёт индексации, проверяет поиск по имени и по содержимому, затем
|
||||
удаляет проект.
|
||||
|
||||
## Грабли
|
||||
|
||||
- `locale`/версия ES: 7.16.3, `_search` совместим с REST 7.x.
|
||||
- ES без auth и слушает только localhost — туннель обязателен.
|
||||
- Фильтр `tenantId` сузит выдачу; без него видны документы всех тенантов.
|
||||
- Поиск по содержимому PDF в индексе OnlyOffice не работает (для PDF нет
|
||||
`attachment.content`) — только Office-форматы. Решение для PDF — свой индекс
|
||||
`oo_docs_text` (F6 #42), см. ниже.
|
||||
|
||||
# PDF и сканы — свой индекс (F6 #42)
|
||||
|
||||
## Проблема
|
||||
|
||||
`oo search --content "S1019"` не находил номер внутри PDF-счёта: в индексе
|
||||
OnlyOffice PDF лежит только по имени.
|
||||
|
||||
## Почему PDF исключён (исходники CommunityServer)
|
||||
|
||||
Разобрано в `ONLYOFFICE/CommunityServer`:
|
||||
|
||||
- `web/core/ASC.Web.Core/Files/FileUtility.cs` — `CanIndex(fileName)` читает
|
||||
серверную настройку `files.index.formats` (в `web/studio/ASC.Web.Studio/web.appsettings.config`
|
||||
значение по умолчанию `".pptx|.xlsx|.docx"`).
|
||||
- `web/studio/ASC.Web.Studio/Products/Files/Core/Search/FilesWrapper.cs` —
|
||||
`GetDocumentStream*` возвращает `null`, если `!FileUtility.CanIndex(Title)`,
|
||||
файл зашифрован или больше `MaxFileSize`.
|
||||
- `module/ASC.ElasticSearch/Core/WrapperWithDoc.cs` + mapping в `Wrapper.cs` —
|
||||
маппинг `document.attachment.content` и ingest-pipeline `attachments`
|
||||
формат-агностичны: они распарсят любой поток.
|
||||
|
||||
Вывод: PDF исключён **только настройкой** `files.index.formats`; жёсткого
|
||||
ограничения на формат в коде нет.
|
||||
|
||||
## Варианты и решение
|
||||
|
||||
| # | Вариант | Оценка |
|
||||
|---|---------|--------|
|
||||
| a | Включить `.pdf` в `files.index.formats` + reindex | Правка сервера OO; настройка может потеряться при обновлении; полный reindex 39k док-в; Tika **не OCR** — сканы без текстового слоя дадут пустой контент. Отклонён без решения PO. |
|
||||
| b | Server-side ingest/attachment для PDF | По факту то же, что (a): сервер кормит поток только для `CanIndex`. |
|
||||
| c | **Свой индекс** `oo_docs_text`, наполняемый `internal/docpipe` | **Выбран.** Сервер OO не трогаем; детерминированно; работает OCR для сканов; независимо от обновлений OO; любые форматы; фильтры папка/тип. |
|
||||
| d | Локальный поиск без индекса | Отклонён как основной: качаем и извлекаем на каждый запрос, нет выдачи/ранжирования/highlight. |
|
||||
|
||||
Итог: **вариант c**. Индекс OnlyOffice (`files_file`) не изменяется; наш
|
||||
индекс живёт рядом.
|
||||
|
||||
## Устройство
|
||||
|
||||
- `file_es_text.go` — `ESTextIndex` (`Name() = "es-text"`):
|
||||
`Ensure` (создаёт индекс с явным маппингом), `Put` (bulk, `refresh`),
|
||||
`Delete` (по `id`), `Search` (`multi_match` по `title^2` + `content`,
|
||||
фильтры `folder`/`ext`, highlight).
|
||||
- `file_text_index.go` — `TextIndexer`: листает папки (`FileStore.List`),
|
||||
качает файлы (`FileStore.Download`), извлекает текст через
|
||||
`internal/docpipe` (`pdftotext`, для сканов — `ocrmypdf`/`tesseract`),
|
||||
пишет в `TextIndex`. Пул воркеров (по умолчанию 3).
|
||||
- CLI: `oo index folder|files` наполняет индекс; `oo search --backend own`
|
||||
ищет по нему.
|
||||
|
||||
### Встроенные вложения PDF
|
||||
|
||||
Оцифрованные PDF несут вложения (`<doc>.md` — текст/таблицы скана,
|
||||
`<doc>.yaml`/`.json` — метаданные, `.xml` — EN 16931 CII eRechnung,
|
||||
`factur-x.xml` у ZUGFeRD; см. `office-assistant/docs/reference/document-metadata.md`).
|
||||
`TextIndexer` обходит их: `pdfdetach -list` перечисляет, `-save` сохраняет,
|
||||
каждое вложение проходит штатный `docpipe.ToMarkdown` (PDF/картинки → OCR,
|
||||
`.md`/`.txt` — как есть). Форматы, которые docpipe не конвертирует
|
||||
(`.xml`/`.html` — снимаются теги; `.json`/`.csv` — как текст), извлекаются
|
||||
текстом; нечитаемые — пропускаются.
|
||||
|
||||
Текст склеивается: тело, затем по секции на вложение с маркером
|
||||
`[attachment: <имя>]` (функция `docpipe.JoinWithAttachments`). Индекс — тот же
|
||||
`file_id`, upsert идемпотентен. Нет вложений или pdfdetach/формат нечитаем —
|
||||
индексируется тело (без падения).
|
||||
|
||||
Поля `oo_docs_text`:
|
||||
|
||||
| поле | тип | смысл |
|
||||
|------|-----|-------|
|
||||
| `id` | keyword | id файла Documents |
|
||||
| `title` | text (+`.keyword`) | имя файла |
|
||||
| `folder` | keyword | id папки |
|
||||
| `ext` | keyword | расширение |
|
||||
| `content` | text | извлечённый текст (pdftotext/OCR) |
|
||||
|
||||
## CLI
|
||||
|
||||
```bash
|
||||
set -a; . .env; set +a # ONLYOFFICE_URL/USER/PASS + ONLYOFFICE_ES_URL
|
||||
oo index folder 634 # PDF в папке 634
|
||||
oo index folder 634 --recursive --exts pdf,png --limit 100
|
||||
oo index files 3576 3578 # точечно
|
||||
oo index folder 634 --dry-run # показать план, ничего не менять
|
||||
|
||||
oo search "S1021" --content --backend own
|
||||
oo search "S1021" --backend own --folder 634 --json
|
||||
```
|
||||
|
||||
`--backend` у `oo search`: `oo` (по умолчанию, индекс OnlyOffice) или `own`
|
||||
(наш `ONLYOFFICE_ES_TEXT_INDEX`).
|
||||
|
||||
## Переменные (дополнение)
|
||||
|
||||
| env | default | смысл |
|
||||
|-----|---------|-------|
|
||||
| `ONLYOFFICE_ES_TEXT_INDEX` | `oo_docs_text` | индекс своего конвейера |
|
||||
|
||||
`ONLYOFFICE_ES_URL` — общий для обоих индексов.
|
||||
|
||||
## Тесты
|
||||
|
||||
```bash
|
||||
go test -run 'ESText|TextIndexer|Index' ./ ./cmd/oo/ # unit, без сети
|
||||
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 \
|
||||
go test -tags=integration -run TestIntegrationESTextIndex -v .
|
||||
```
|
||||
|
||||
Интеграционный тест создаёт временный индекс, наполняет, ищет по контенту,
|
||||
проверяет фильтры и удаление, затем удаляет индекс;
|
||||
`TestIntegrationESTextIndexPDFAttachment` индексирует
|
||||
`testdata/pdf-with-attachment.pdf` реальным конвейером (pdfdetach + pdftotext)
|
||||
и ищет токен, лежащий только во вложении. Unit-тесты используют
|
||||
fake-store/fake-extractor и не требуют pdftotext/OCR (парсер списка, склейка
|
||||
`JoinWithAttachments`, снятие тегов `xmlToText` — чистые).
|
||||
|
||||
## Грабли
|
||||
|
||||
- Наполнение — ручное (`oo index`); после изменения/добавления PDF повтори.
|
||||
Повтор идемпотентен (upsert по id файла).
|
||||
- В индексе ищется только то, что проиндексировано; `oo index` качает каждый
|
||||
файл и (для сканов) гоняет OCR — это медленно, отсюда `--limit`/`--exts`.
|
||||
- `folder` фильтруется как id папки, а не как путь.
|
||||
- Дубликаты (напр. `S1055.pdf` и `2026-08-20-S1055-…`) дадут несколько строк —
|
||||
это ожидаемо, дедуп — на стороне потребителя.
|
||||
- Вложения: нужен `pdfdetach` (poppler); если его нет — индексируется только
|
||||
тело. Вложенный PDF/картинка с плохим текстовым слоем проходит OCR, это
|
||||
медленно. `.json`-метаданные (CuraSoft) индексируются как текст и могут
|
||||
добавить шумовых токенов.
|
||||
@@ -0,0 +1,109 @@
|
||||
---
|
||||
type: reference
|
||||
status: current
|
||||
related:
|
||||
- deploy/docker-compose.rclone-webdav.yml
|
||||
- docs/unified-file-client.md
|
||||
---
|
||||
|
||||
# rclone WebDAV mount — OnlyOffice Documents как файловая ФС
|
||||
|
||||
`deploy/docker-compose.rclone-webdav.yml` монтирует WebDAV-дерево OnlyOffice
|
||||
через `rclone`. Не systemd, только compose. Контейнер — `rclone-webdav`.
|
||||
|
||||
## Что монтируется
|
||||
|
||||
- Источник — сайдкар `oo-webdav` (`ghcr.io/eslider/oo-webdav`).
|
||||
- Адрес — `http://172.17.0.1:8098`, префикс `/webdav`.
|
||||
- Basic-auth — креды портала OnlyOffice (те же, что у `oo`).
|
||||
- Корень — `Dokumente der Projekte/…` (папка `Fibu EDL` внутри).
|
||||
|
||||
## Конфиг
|
||||
|
||||
- Образ — `rclone/rclone`.
|
||||
- Remote — on-the-fly `:webdav:` плюс флаги `--webdav-url`, `--webdav-user`,
|
||||
`--webdav-pass`.
|
||||
- На старте контейнер пишет `/config/rclone/rclone.conf` (`[webdav]`), чтобы
|
||||
`rclone ls webdav:` внутри контейнера работал без флагов.
|
||||
- `ONLYOFFICE_PASSWORD` — plain; rclone сам зовёт `rclone obscure`.
|
||||
- Флаги mount: `--vfs-cache-mode writes`, `--cache-dir /cache`,
|
||||
`--dir-cache-time 1m`, `--allow-other`, `--allow-non-empty`.
|
||||
- FUSE: `cap_add: [SYS_ADMIN]`, `devices: [/dev/fuse]`,
|
||||
`security_opt: apparmor:unconfined`.
|
||||
- Тома: `onlyoffice-mnt` → `/mnt/onlyoffice`, `rclone-cache` → `/cache`.
|
||||
|
||||
### env (только имена)
|
||||
|
||||
| Переменная | Значение |
|
||||
|------------|----------|
|
||||
| `ONLYOFFICE_USER` | пользователь портала |
|
||||
| `ONLYOFFICE_PASSWORD` | пароль портала (алиас `ONLYOFFICE_PASS`) |
|
||||
| `ONLYOFFICE_WEBDAV_URL` | по умолчанию `http://172.17.0.1:8098/webdav` |
|
||||
|
||||
## Команды
|
||||
|
||||
```bash
|
||||
# креды (или .env рядом с compose)
|
||||
set -a; . .secrets/oo.env; set +a
|
||||
|
||||
docker compose -f deploy/docker-compose.rclone-webdav.yml up -d
|
||||
docker compose -f deploy/docker-compose.rclone-webdav.yml ps
|
||||
docker compose -f deploy/docker-compose.rclone-webdav.yml down
|
||||
|
||||
# дерево
|
||||
docker exec rclone-webdav rclone ls webdav:
|
||||
docker exec rclone-webdav rclone lsf webdav:
|
||||
|
||||
# содержимое точки монтирования
|
||||
docker exec rclone-webdav ls /mnt/onlyoffice
|
||||
|
||||
# чтение/запись как обычная ФС
|
||||
docker exec rclone-webdav cat "/mnt/onlyoffice/Meine Dokumente/x.txt"
|
||||
```
|
||||
|
||||
## Smoke (проверено 2026-09-16)
|
||||
|
||||
```text
|
||||
$ docker exec rclone-webdav rclone ls webdav:
|
||||
5357109 Meine Dokumente/ONLYOFFICE-Audiobeispiel.mp3
|
||||
44805 Meine Dokumente/ONLYOFFICE-Beispiel-Tabellenblatt.xlsx
|
||||
58988 Meine Dokumente/ONLYOFFICE-Beispieldokument.docx
|
||||
...
|
||||
|
||||
$ TS=20260916-212459
|
||||
$ docker exec rclone-webdav sh -c "printf 'rclone-webdav round-trip $TS\n' \
|
||||
> '/mnt/onlyoffice/Meine Dokumente/rclone-smoke-$TS/hello.txt'"
|
||||
# ждём появления на сервере (host HTTP, мимо монтирования):
|
||||
GET /webdav/Meine%20Dokumente/rclone-smoke-$TS/hello.txt -> 200 (через 7s)
|
||||
server bytes: rclone-webdav round-trip 20260916-212459
|
||||
# читаем обратно через монтирование:
|
||||
mount read: rclone-webdav round-trip 20260916-212459
|
||||
MATCH: yes
|
||||
|
||||
$ docker exec rclone-webdav rm -f \
|
||||
"/mnt/onlyoffice/Meine Dokumente/rclone-smoke-$TS/hello.txt"
|
||||
GET /webdav/Meine%20Dokumente/rclone-smoke-$TS/hello.txt -> 404 (через 1s)
|
||||
|
||||
# пустой каталог: rmdir через монтирование на сервер не доходит,
|
||||
# удаляем напрямую:
|
||||
$ docker exec rclone-webdav rclone rmdir "webdav:Meine Dokumente/rclone-smoke-$TS"
|
||||
PROPFIND rclone-smoke-20260916-212459/ -> 404
|
||||
no rclone-smoke leftovers
|
||||
```
|
||||
|
||||
Вывод: запись → чтение → удаление файла подтверждены. Файл удалён.
|
||||
|
||||
## Ограничения
|
||||
|
||||
- Нужен FUSE: `SYS_ADMIN` + `/dev/fuse`. На хосте без FUSE не поедет.
|
||||
- Сайдкар слушает docker-bridge `172.17.0.1:8098`, публично не выставлен.
|
||||
Только localhost/локальные контейнеры.
|
||||
- Запись идёт через VFS write-back (по умолчанию ~5s). Перед проверкой
|
||||
«файл на сервере» опрашивать сервер, не доверять сразу после `write`.
|
||||
- `rmdir` через монтирование НЕ доходит до `oo-webdav` (пустой каталог
|
||||
остаётся на сервере). Удалять каталоги напрямую:
|
||||
`rclone rmdir webdav:<path>` или `rclone purge webdav:<path>`.
|
||||
- VFS-кэш растёт в томе `rclone-cache`; ограничить `--vfs-cache-max-size`.
|
||||
- Блокировок между редактором OnlyOffice и монтированием нет. Не редактировать
|
||||
один и тот же файл одновременно.
|
||||
- Данные не шифруются на диске хоста в `rclone-cache` (том Docker).
|
||||
@@ -0,0 +1,205 @@
|
||||
---
|
||||
type: reference
|
||||
status: current
|
||||
related:
|
||||
- README.md
|
||||
- file_core.go
|
||||
- file_facade.go
|
||||
- docs/elasticsearch.md
|
||||
- docs/community-server-db.md
|
||||
---
|
||||
|
||||
# Unified file client — контракт файловых бэкендов
|
||||
|
||||
## Что это
|
||||
|
||||
Один файловый клиент на все бэкенды (эпик #34). Модель и интерфейсы —
|
||||
`file_core.go`. Фасад `FileClient` — `file_facade.go`. Бэкенды:
|
||||
REST, WebDAV, SQL (PostgreSQL/MySQL), Elasticsearch. Правило одно:
|
||||
код зовёт `c.Files()` и не знает про транспорт.
|
||||
|
||||
## Модель
|
||||
|
||||
- `Kind` — `File` (0) или `Folder` (1).
|
||||
- `Entry` — бэкенд-независимая строка: `ID`, `ParentID`, `Title`, `Kind`,
|
||||
`Size`, `MIME`, `Created`, `Modified`, `Updated` (сырая строка API),
|
||||
`Version`, `Provider`, `FilesCount`/`FoldersCount` (папки).
|
||||
Чего бэкенд не даёт — остаётся в нуле.
|
||||
- `SearchQuery` — `Text`, `InContent`, `FolderID`, `Extensions`, `Limit`.
|
||||
- `SearchHit` — `Entry` + `Score`, `Highlight`, `Path`.
|
||||
|
||||
## Интерфейсы
|
||||
|
||||
`FileStore` — операции с файлами:
|
||||
|
||||
```go
|
||||
type FileStore interface {
|
||||
Name() string
|
||||
List(ctx, parentID) ([]Entry, error)
|
||||
Stat(ctx, id) (Entry, error)
|
||||
CreateFolder(ctx, parentID, title) (Entry, error)
|
||||
Upload(ctx, parentID, title, r) (Entry, error)
|
||||
Download(ctx, id, w) (int64, error)
|
||||
Move(ctx, ids, parentID) error
|
||||
Copy(ctx, ids, parentID) error
|
||||
Rename(ctx, id, title) error
|
||||
Delete(ctx, ids) error
|
||||
}
|
||||
```
|
||||
|
||||
`Searcher` — поиск (необязательный):
|
||||
|
||||
```go
|
||||
type Searcher interface {
|
||||
Search(ctx, q SearchQuery) ([]SearchHit, error)
|
||||
Name() string
|
||||
}
|
||||
```
|
||||
|
||||
`TextIndex` (`file_es_text.go`) — свой индекс: `Put`, `Delete`, `Search`,
|
||||
`Name`. `ESTextIndex` реализует и `Searcher`, и `TextIndex`.
|
||||
|
||||
## Бэкенды
|
||||
|
||||
| бэкенд | провайдер | файл | что умеет |
|
||||
|--------|-----------|------|-----------|
|
||||
| REST | `rest` | `file_rest.go` | read + write, Documents API |
|
||||
| WebDAV | `dav` | `file_dav.go` | read + write, Documents fileops |
|
||||
| SQL | `postgres` / `mysql` | `file_pg.go` | **read-only** |
|
||||
| OnlyOffice ES | `elasticsearch` | `file_es.go` | поиск (имя + контент Office) |
|
||||
| свой ES-индекс | `es-text` | `file_es_text.go` | поиск + запись (PDF/сканы) |
|
||||
|
||||
- REST: `Stat` знает только файлы; папки — через `List`.
|
||||
- WebDAV: `Move`/`Copy`/`Delete` сперва `Stat`-ят id (папка/файл), потом зовут
|
||||
fileops.
|
||||
- SQL: `List`/`Stat`/`Download`/`Search` (по имени). Все write-методы →
|
||||
`ErrReadOnly`. `Download` идёт в S3/MinIO по layout портала.
|
||||
- OnlyOffice ES: индекс `files_file`, контент только для docx/xlsx/pptx.
|
||||
- Свой ES: индекс `oo_docs_text`, контент из `internal/docpipe`, в т.ч.
|
||||
встроенные PDF-вложения.
|
||||
|
||||
## Фасад `FileClient`
|
||||
|
||||
`c.Files()` → `*FileClient`. Он же реализует `FileStore`, старый код
|
||||
компилируется.
|
||||
|
||||
- `Read()` — первый зарегистрированный из `readOrder`:
|
||||
`postgres` → `mysql` → `rest` → `dav`.
|
||||
- `Write()` — первый из `writeOrder`: `rest` → `dav`. SQL не пишет.
|
||||
- `Search()` — первый из `searchOrder`: `elasticsearch`. Нет бэкенда →
|
||||
ошибка (`ONLYOFFICE_ES_URL`).
|
||||
- `RegisterStore(name, s)` / `RegisterSearcher(name, s)` — добавить бэкенд.
|
||||
|
||||
Fallback:
|
||||
|
||||
- `List`/`Stat` идут по `readOrder`; переходят к следующему только на
|
||||
transient-ошибке (429/502/503/504). Иначе ошибка финальная.
|
||||
- `Download` **без** fallback: часть байтов уже в `w`, второй бэкенд допишет.
|
||||
- Запись (`CreateFolder`/`Upload`/`Move`/`Copy`/`Rename`/`Delete`) — только
|
||||
`Write()`, без fallback.
|
||||
|
||||
`newFileClient` сам кладёт `rest` и `dav`; ES-поиск — если задан
|
||||
`ONLYOFFICE_ES_URL`. SQL-стор регистрирует вызывающий: фасад создаётся на
|
||||
каждый `c.Files()`, регистрируй на том же экземпляре.
|
||||
|
||||
```go
|
||||
sql, err := c.SQLFileStore() // открыть из env (ONLYOFFICE_DSN)
|
||||
if err != nil { /* нет DSN */ }
|
||||
if closer, ok := sql.(interface{ Close() error }); ok { defer closer.Close() }
|
||||
|
||||
f := c.Files()
|
||||
f.RegisterStore(onlyoffice.ProviderPG, sql) // или sql.Name() == "mysql"
|
||||
e, _ := f.Stat(ctx, "19423") // e.Provider == "mysql"
|
||||
entries, _ := f.List(ctx, "676") // пойдёт в SQL
|
||||
```
|
||||
|
||||
`Client.FileStore("pg"|"sql"|"postgres"|"mysql")` — одноразовый доступ к
|
||||
SQL-стору без фасада: открывает из env; при ошибке возвращает заглушку,
|
||||
которая отдаёт ошибку открытия на каждом вызове (не `nil`). `SQLFileStore()`
|
||||
— тот же открыватель, но с ошибкой. Отвечавший бэкенд видно по
|
||||
`Entry.Provider` (`mysql` / `postgres` у SQL, `rest` у REST).
|
||||
|
||||
## CLI
|
||||
|
||||
```bash
|
||||
# поиск: --backend oo (индекс OnlyOffice) | own (свой oo_docs_text)
|
||||
oo search "Rechnung"
|
||||
oo search "Mahngebühr" --content
|
||||
oo search "S1021" --content --backend own --folder 634 --limit 50 --json
|
||||
|
||||
# наполнение своего индекса (PDF/сканы, idempotent upsert по file id)
|
||||
oo index folder 634
|
||||
oo index folder 634 --recursive --exts pdf,png --limit 100
|
||||
oo index files 3576 3578
|
||||
oo index folder 634 --dry-run
|
||||
```
|
||||
|
||||
`oo index` флаги: `--recursive`, `--exts` (default `pdf`), `--limit`,
|
||||
`--workers` (3), `--lang` (`deu+eng`), `--min-chars`, `--work-dir`,
|
||||
`--backend rest|dav`, `--dry-run`, `--json`.
|
||||
|
||||
Библиотека:
|
||||
|
||||
```go
|
||||
idx, _ := onlyoffice.NewESTextIndex(onlyoffice.ESTextConfigFromEnv())
|
||||
ti := onlyoffice.NewTextIndexer(store, idx) // store = FileStore
|
||||
res, _ := ti.IndexFolder(ctx, "634", onlyoffice.IndexOptions{Recursive: true})
|
||||
```
|
||||
|
||||
## Env (только имена)
|
||||
|
||||
| env | default | кто читает |
|
||||
|-----|---------|------------|
|
||||
| `ONLYOFFICE_ES_URL` | — | ES (оба индекса), обязателен |
|
||||
| `ONLYOFFICE_ES_INDEX` | `files_file` | индекс OnlyOffice |
|
||||
| `ONLYOFFICE_ES_TEXT_INDEX` | `oo_docs_text` | свой индекс |
|
||||
| `ONLYOFFICE_TENANT` | пусто | фильтр `tenantId` |
|
||||
| `ONLYOFFICE_DSN` | — | SQL DSN (MySQL/PostgreSQL) |
|
||||
| `ONLYOFFICE_PG_DRIVER` | auto | `postgres` / `mysql` |
|
||||
| `ONLYOFFICE_PG_TENANT` | `ONLYOFFICE_TENANT` | SQL tenant |
|
||||
| `ONLYOFFICE_PG_HOST` `_PORT` `_USER` `_PASSWORD` `_DBNAME` `_SSLMODE` | — | DSN по частям |
|
||||
| `MINIO_ENDPOINT` `MINIO_BUCKET` `MINIO_ACCESS_KEY` `MINIO_SECRET_KEY` | — | download SQL-стора |
|
||||
| `OO_URL` `OO_USER` `OO_PASS` | — | CLI-алиасы |
|
||||
|
||||
Имена — в [`.env.example`](../.env.example). Секретов в репо нет.
|
||||
|
||||
## Ограничения
|
||||
|
||||
- OnlyOffice ES: контент только Office-форматов. PDF — только по имени.
|
||||
Встроенные вложения PDF сервер не индексирует.
|
||||
- Свой индекс `oo_docs_text`: покрывает PDF/сканы и вложения (pdfdetach), но
|
||||
наполняется вручную (`oo index`) и идемпотентен. Фильтр `folder` — id папки,
|
||||
не путь. Дубли дают несколько строк — дедуп на потребителе.
|
||||
- SQL: read-only. `InContent` игнорируется (только имя). Download — через
|
||||
MinIO-схему, не HTTP.
|
||||
- ES: без auth, слушает localhost внутри VM — нужен SSH-туннель
|
||||
(см. [elasticsearch.md](elasticsearch.md)).
|
||||
- `oo index` качает каждый файл и для сканов гоняет OCR — медленно; отсюда
|
||||
`--limit` и `--exts`. Нужен `pdfdetach` (poppler); без него — только тело PDF.
|
||||
|
||||
## Как добавить бэкенд
|
||||
|
||||
1. Файл `file_<name>.go`. Реализуй `FileStore` (`Name` + 9 методов). Нужен
|
||||
поиск — добавь `Searcher`; нужна запись своего индекса — `TextIndex`.
|
||||
2. Добавь const провайдера рядом с `ProviderREST`/`ProviderDAV`.
|
||||
3. Зарегистрируй: в `newFileClient` или снаружи через
|
||||
`RegisterStore`/`RegisterSearcher`.
|
||||
4. Внеси имя в `readOrder` / `writeOrder` / `searchOrder`.
|
||||
5. Есть CLI-команда — добавь значение в `--backend`.
|
||||
6. Тесты: unit (чистые builders/парсеры, без сети) + интеграционный
|
||||
(`//go:build integration`, skip без кред).
|
||||
|
||||
## Тесты
|
||||
|
||||
```bash
|
||||
go test ./... # unit, без сети
|
||||
go test -tags=integration ./... # live (креды в .env)
|
||||
go test ./ -run 'FileStore|Facade|ESText|PG'
|
||||
```
|
||||
|
||||
## См. также
|
||||
|
||||
- [README.md](README.md) — индекс справочников.
|
||||
- [elasticsearch.md](elasticsearch.md) — индекс OnlyOffice и свой `oo_docs_text`.
|
||||
- [community-server-db.md](community-server-db.md) — SQL-стор и схема БД.
|
||||
- [rclone-webdav.md](rclone-webdav.md) — монтирование Documents как ФС.
|
||||
+262
@@ -0,0 +1,262 @@
|
||||
package onlyoffice
|
||||
|
||||
// Canonical file model and the backend-agnostic store interface. REST
|
||||
// (files.go), WebDAV (files_webdav.go) and future backends (PostgreSQL,
|
||||
// Elasticsearch) implement FileStore/Searcher so callers stop depending on a
|
||||
// concrete transport. This file holds only types and pure conversions — no IO.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"mime"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Kind distinguishes files from folders in the canonical model.
|
||||
type Kind int
|
||||
|
||||
const (
|
||||
File Kind = iota
|
||||
Folder
|
||||
)
|
||||
|
||||
// String renders the kind for logs and table output.
|
||||
func (k Kind) String() string {
|
||||
switch k {
|
||||
case File:
|
||||
return "file"
|
||||
case Folder:
|
||||
return "folder"
|
||||
default:
|
||||
return "unknown"
|
||||
}
|
||||
}
|
||||
|
||||
// Provider names for the FileStore adapters.
|
||||
const (
|
||||
ProviderREST = "rest"
|
||||
ProviderDAV = "dav"
|
||||
)
|
||||
|
||||
// Entry is the backend-independent representation of a document or folder.
|
||||
// Fields that a backend cannot supply stay at their zero value.
|
||||
type Entry struct {
|
||||
ID string
|
||||
ParentID string
|
||||
Title string
|
||||
Kind Kind
|
||||
Size int64
|
||||
MIME string
|
||||
Created time.Time
|
||||
Modified time.Time
|
||||
// Updated is the backend-native timestamp string, when the backend exposes
|
||||
// one. It lets list output round-trip the API value; Modified is the
|
||||
// parsed form for logic.
|
||||
Updated string
|
||||
Version int
|
||||
Provider string
|
||||
|
||||
// Folder-only counters. Zero for files and for backends that do not
|
||||
// report them.
|
||||
FilesCount int
|
||||
FoldersCount int
|
||||
}
|
||||
|
||||
// FileStore is the operation surface every file backend implements.
|
||||
type FileStore interface {
|
||||
Name() string
|
||||
List(ctx context.Context, parentID string) ([]Entry, error)
|
||||
Stat(ctx context.Context, id string) (Entry, error)
|
||||
CreateFolder(ctx context.Context, parentID, title string) (Entry, error)
|
||||
Upload(ctx context.Context, parentID, title string, r io.Reader) (Entry, error)
|
||||
Download(ctx context.Context, id string, w io.Writer) (int64, error)
|
||||
Move(ctx context.Context, ids []string, parentID string) error
|
||||
Copy(ctx context.Context, ids []string, parentID string) error
|
||||
Rename(ctx context.Context, id, title string) error
|
||||
Delete(ctx context.Context, ids []string) error
|
||||
}
|
||||
|
||||
// SearchQuery narrows a Searcher request. InContent asks the backend to match
|
||||
// document bodies, not just titles. Substring switches title matching from the
|
||||
// analyzer's whole-token match to a case-insensitive "*term*" wildcard and ANDs
|
||||
// every whitespace-separated term (e.g. "rechnung 2025").
|
||||
type SearchQuery struct {
|
||||
Text string
|
||||
InContent bool
|
||||
FolderID string
|
||||
Extensions []string
|
||||
Limit int
|
||||
Substring bool
|
||||
}
|
||||
|
||||
// SearchHit is one Searcher result: the matching entry plus backend-specific
|
||||
// ranking metadata.
|
||||
type SearchHit struct {
|
||||
Entry
|
||||
Score float64
|
||||
Highlight string
|
||||
Path []string
|
||||
}
|
||||
|
||||
// Searcher is the optional content/name search surface. Only some backends
|
||||
// (for example Elasticsearch) provide it.
|
||||
type Searcher interface {
|
||||
Search(ctx context.Context, q SearchQuery) ([]SearchHit, error)
|
||||
Name() string
|
||||
}
|
||||
|
||||
// FileStore returns the adapter for a backend name: ProviderREST (default),
|
||||
// ProviderDAV (alias "webdav") or the read-only SQL store (ProviderPG,
|
||||
// ProviderMySQL and the aliases "pg"/"sql"). The SQL store is opened from the
|
||||
// environment (ONLYOFFICE_DSN / ONLYOFFICE_PG_*); when it cannot be opened the
|
||||
// returned store surfaces that error on every operation instead of returning
|
||||
// nil. Use SQLFileStore when the open error itself is needed. Unknown or empty
|
||||
// names select the REST backend. The composed facade (backend
|
||||
// selection/fallback) lives on FileClient in file_facade.go.
|
||||
func (c *Client) FileStore(backend string) FileStore {
|
||||
switch strings.ToLower(strings.TrimSpace(backend)) {
|
||||
case ProviderDAV, "webdav":
|
||||
return &davStore{c: c}
|
||||
case ProviderPG, ProviderMySQL, "pg", "sql":
|
||||
s, err := c.SQLFileStore()
|
||||
if err != nil {
|
||||
return &errStore{name: strings.ToLower(strings.TrimSpace(backend)), err: err}
|
||||
}
|
||||
return s
|
||||
default:
|
||||
return &restStore{c: c}
|
||||
}
|
||||
}
|
||||
|
||||
// errStore is the FileStore placeholder returned when a backend cannot be
|
||||
// opened (for example SQL without a DSN). Every operation returns the recorded
|
||||
// error instead of panicking on a nil interface.
|
||||
type errStore struct {
|
||||
name string
|
||||
err error
|
||||
}
|
||||
|
||||
func (s *errStore) Name() string { return s.name }
|
||||
|
||||
func (s *errStore) List(context.Context, string) ([]Entry, error) { return nil, s.err }
|
||||
|
||||
func (s *errStore) Stat(context.Context, string) (Entry, error) { return Entry{}, s.err }
|
||||
|
||||
func (s *errStore) CreateFolder(context.Context, string, string) (Entry, error) {
|
||||
return Entry{}, s.err
|
||||
}
|
||||
|
||||
func (s *errStore) Upload(context.Context, string, string, io.Reader) (Entry, error) {
|
||||
return Entry{}, s.err
|
||||
}
|
||||
|
||||
func (s *errStore) Download(context.Context, string, io.Writer) (int64, error) {
|
||||
return 0, s.err
|
||||
}
|
||||
|
||||
func (s *errStore) Move(context.Context, []string, string) error { return s.err }
|
||||
|
||||
func (s *errStore) Copy(context.Context, []string, string) error { return s.err }
|
||||
|
||||
func (s *errStore) Rename(context.Context, string, string) error { return s.err }
|
||||
|
||||
func (s *errStore) Delete(context.Context, []string) error { return s.err }
|
||||
|
||||
// Files returns the composed file facade. The returned *FileClient implements
|
||||
// FileStore, so callers that used Files() as the plain REST store keep working.
|
||||
func (c *Client) Files() *FileClient { return c.newFileClient() }
|
||||
|
||||
// retryStoreOp runs one store operation under the shared deterministic
|
||||
// transient-error policy (429/502/503/504).
|
||||
func retryStoreOp(ctx context.Context, fn func() error) error {
|
||||
return DoRetry(ctx, DefaultRetryPolicy(), fn)
|
||||
}
|
||||
|
||||
// FileEntryToEntry converts a Files-module file row to the canonical model.
|
||||
func FileEntryToEntry(f *FileEntry, provider string) Entry {
|
||||
e := Entry{Kind: File, Provider: provider}
|
||||
if f == nil {
|
||||
return e
|
||||
}
|
||||
if f.ID != nil {
|
||||
e.ID = f.ID.String()
|
||||
}
|
||||
e.ParentID = FileFolderID(f)
|
||||
if f.Title != nil {
|
||||
e.Title = *f.Title
|
||||
}
|
||||
if f.ContentLength != nil {
|
||||
e.Size = parseContentLength(*f.ContentLength)
|
||||
}
|
||||
exst := ""
|
||||
if f.FileExst != nil {
|
||||
exst = *f.FileExst
|
||||
}
|
||||
e.MIME = mimeForTitle(e.Title, exst)
|
||||
if f.Updated != nil {
|
||||
e.Modified = *f.Updated
|
||||
e.Updated = f.Updated.Format(time.RFC3339)
|
||||
}
|
||||
return e
|
||||
}
|
||||
|
||||
// DavFileToEntry converts a WebDAV file row to the canonical model.
|
||||
func DavFileToEntry(f DavFile, provider string) Entry {
|
||||
return Entry{
|
||||
ID: f.ID,
|
||||
Title: f.Title,
|
||||
Kind: File,
|
||||
Size: f.Size,
|
||||
MIME: mimeForTitle(f.Title, ""),
|
||||
Modified: f.ModTime(),
|
||||
Updated: f.Updated,
|
||||
Provider: provider,
|
||||
}
|
||||
}
|
||||
|
||||
// DavFolderToEntry converts a WebDAV folder row to the canonical model.
|
||||
func DavFolderToEntry(f DavFolder, provider string) Entry {
|
||||
return Entry{
|
||||
ID: f.ID,
|
||||
ParentID: f.ParentID,
|
||||
Title: f.Title,
|
||||
Kind: Folder,
|
||||
Modified: f.ModTime(),
|
||||
Updated: f.Updated,
|
||||
Provider: provider,
|
||||
FilesCount: f.FilesCount,
|
||||
FoldersCount: f.FoldersCount,
|
||||
}
|
||||
}
|
||||
|
||||
// parseContentLength reads the leading integer of an OnlyOffice contentLength
|
||||
// string (the API sometimes appends a unit, e.g. "12345 b").
|
||||
func parseContentLength(s string) int64 {
|
||||
fields := strings.Fields(s)
|
||||
if len(fields) == 0 {
|
||||
return 0
|
||||
}
|
||||
n, err := strconv.ParseInt(fields[0], 10, 64)
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// mimeForTitle derives a MIME type from an explicit extension or the title.
|
||||
func mimeForTitle(title, exst string) string {
|
||||
ext := strings.TrimSpace(exst)
|
||||
if ext == "" {
|
||||
ext = filepath.Ext(title)
|
||||
}
|
||||
if ext == "" {
|
||||
return ""
|
||||
}
|
||||
if !strings.HasPrefix(ext, ".") {
|
||||
ext = "." + ext
|
||||
}
|
||||
return mime.TypeByExtension(strings.ToLower(ext))
|
||||
}
|
||||
@@ -0,0 +1,193 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestFileEntryToEntry(t *testing.T) {
|
||||
id := json.Number("42")
|
||||
title := "invoice.pdf"
|
||||
exst := ".pdf"
|
||||
size := "12345"
|
||||
parent := json.Number("7")
|
||||
updated := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
|
||||
f := &FileEntry{
|
||||
ID: &id,
|
||||
Title: &title,
|
||||
FileExst: &exst,
|
||||
ContentLength: &size,
|
||||
FolderID: &parent,
|
||||
Updated: &updated,
|
||||
}
|
||||
|
||||
e := FileEntryToEntry(f, ProviderREST)
|
||||
if e.ID != "42" {
|
||||
t.Errorf("ID = %q, want 42", e.ID)
|
||||
}
|
||||
if e.ParentID != "7" {
|
||||
t.Errorf("ParentID = %q, want 7", e.ParentID)
|
||||
}
|
||||
if e.Title != title {
|
||||
t.Errorf("Title = %q, want %q", e.Title, title)
|
||||
}
|
||||
if e.Kind != File {
|
||||
t.Errorf("Kind = %v, want file", e.Kind)
|
||||
}
|
||||
if e.Size != 12345 {
|
||||
t.Errorf("Size = %d, want 12345", e.Size)
|
||||
}
|
||||
if e.MIME != "application/pdf" {
|
||||
t.Errorf("MIME = %q, want application/pdf", e.MIME)
|
||||
}
|
||||
if !e.Modified.Equal(updated) {
|
||||
t.Errorf("Modified = %v, want %v", e.Modified, updated)
|
||||
}
|
||||
if e.Provider != ProviderREST {
|
||||
t.Errorf("Provider = %q, want %q", e.Provider, ProviderREST)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileEntryToEntryNil(t *testing.T) {
|
||||
e := FileEntryToEntry(nil, ProviderDAV)
|
||||
if e.Kind != File {
|
||||
t.Errorf("Kind = %v, want file", e.Kind)
|
||||
}
|
||||
if e.ID != "" || e.Title != "" {
|
||||
t.Errorf("nil entry should be empty: %+v", e)
|
||||
}
|
||||
if e.Provider != ProviderDAV {
|
||||
t.Errorf("Provider = %q, want %q", e.Provider, ProviderDAV)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileEntryToEntrySizeFormats(t *testing.T) {
|
||||
cases := map[string]int64{
|
||||
"12345": 12345,
|
||||
"12345 b": 12345,
|
||||
"0": 0,
|
||||
"": 0,
|
||||
"notanum": 0,
|
||||
}
|
||||
for in, want := range cases {
|
||||
got := parseContentLength(in)
|
||||
if got != want {
|
||||
t.Errorf("parseContentLength(%q) = %d, want %d", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDavFileToEntry(t *testing.T) {
|
||||
f := DavFile{
|
||||
ID: "9",
|
||||
Title: "note.txt",
|
||||
Size: 10,
|
||||
Updated: "2026-01-02T03:04:05.0000000+01:00",
|
||||
}
|
||||
e := DavFileToEntry(f, ProviderDAV)
|
||||
if e.ID != "9" || e.Title != "note.txt" {
|
||||
t.Errorf("identity mismatch: %+v", e)
|
||||
}
|
||||
if e.Kind != File {
|
||||
t.Errorf("Kind = %v, want file", e.Kind)
|
||||
}
|
||||
if e.Size != 10 {
|
||||
t.Errorf("Size = %d, want 10", e.Size)
|
||||
}
|
||||
if !strings.HasPrefix(e.MIME, "text/plain") {
|
||||
t.Errorf("MIME = %q, want text/plain*", e.MIME)
|
||||
}
|
||||
if e.Modified.IsZero() {
|
||||
t.Error("Modified not parsed")
|
||||
}
|
||||
if e.Provider != ProviderDAV {
|
||||
t.Errorf("Provider = %q, want %q", e.Provider, ProviderDAV)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDavFolderToEntry(t *testing.T) {
|
||||
f := DavFolder{
|
||||
ID: "5",
|
||||
Title: "inbox",
|
||||
ParentID: "1",
|
||||
Updated: "2026-01-02T03:04:05.0000000+01:00",
|
||||
}
|
||||
e := DavFolderToEntry(f, ProviderDAV)
|
||||
if e.ID != "5" || e.Title != "inbox" || e.ParentID != "1" {
|
||||
t.Errorf("identity mismatch: %+v", e)
|
||||
}
|
||||
if e.Kind != Folder {
|
||||
t.Errorf("Kind = %v, want folder", e.Kind)
|
||||
}
|
||||
if e.MIME != "" {
|
||||
t.Errorf("folder MIME = %q, want empty", e.MIME)
|
||||
}
|
||||
if e.Modified.IsZero() {
|
||||
t.Error("Modified not parsed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestEntriesFromFolderMap(t *testing.T) {
|
||||
m := map[string]any{
|
||||
"files": []any{
|
||||
map[string]any{"id": float64(42), "title": "a.pdf", "pureContentLength": float64(7)},
|
||||
},
|
||||
"folders": []any{
|
||||
map[string]any{"id": float64(7), "title": "sub", "parentId": float64(1)},
|
||||
},
|
||||
}
|
||||
entries, err := entriesFromFolderMap(m, ProviderREST)
|
||||
if err != nil {
|
||||
t.Fatalf("entriesFromFolderMap: %v", err)
|
||||
}
|
||||
if len(entries) != 2 {
|
||||
t.Fatalf("got %d entries, want 2: %+v", len(entries), entries)
|
||||
}
|
||||
byID := map[string]Entry{}
|
||||
for _, e := range entries {
|
||||
byID[e.ID] = e
|
||||
}
|
||||
if got := byID["42"]; got.Kind != File || got.Size != 7 || got.Title != "a.pdf" {
|
||||
t.Errorf("file entry = %+v", got)
|
||||
}
|
||||
if got := byID["7"]; got.Kind != Folder || got.ParentID != "1" || got.Title != "sub" {
|
||||
t.Errorf("folder entry = %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEntriesFromFolderMapNil(t *testing.T) {
|
||||
entries, err := entriesFromFolderMap(nil, ProviderREST)
|
||||
if err != nil || entries != nil {
|
||||
t.Fatalf("got %v, %v; want nil, nil", entries, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestKindString(t *testing.T) {
|
||||
if File.String() != "file" || Folder.String() != "folder" {
|
||||
t.Errorf("kind strings: %q %q", File.String(), Folder.String())
|
||||
}
|
||||
if Kind(9).String() != "unknown" {
|
||||
t.Errorf("unknown kind = %q", Kind(9).String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestClientFileStoreSelection(t *testing.T) {
|
||||
c := NewClient(Credentials{})
|
||||
if got := c.FileStore(ProviderDAV).Name(); got != ProviderDAV {
|
||||
t.Errorf("FileStore(dav).Name() = %q", got)
|
||||
}
|
||||
if got := c.FileStore("webdav").Name(); got != ProviderDAV {
|
||||
t.Errorf("FileStore(webdav).Name() = %q", got)
|
||||
}
|
||||
if got := c.FileStore(ProviderREST).Name(); got != ProviderREST {
|
||||
t.Errorf("FileStore(rest).Name() = %q", got)
|
||||
}
|
||||
if got := c.FileStore("").Name(); got != ProviderREST {
|
||||
t.Errorf("FileStore(\"\").Name() = %q", got)
|
||||
}
|
||||
if got := c.Files().Name(); got != ProviderREST {
|
||||
t.Errorf("Files().Name() = %q", got)
|
||||
}
|
||||
}
|
||||
+189
@@ -0,0 +1,189 @@
|
||||
package onlyoffice
|
||||
|
||||
// davStore implements FileStore on top of the Documents/WebDAV methods in
|
||||
// files_webdav.go. The Documents fileops calls need folder and file ids
|
||||
// separated, so ids are classified through Stat before move/copy/rename/delete.
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
)
|
||||
|
||||
// davStore is a FileStore over the WebDAV-oriented Documents API.
|
||||
type davStore struct{ c *Client }
|
||||
|
||||
// Name reports the backend name.
|
||||
func (s *davStore) Name() string { return ProviderDAV }
|
||||
|
||||
// List returns the files and folders directly below parentID.
|
||||
func (s *davStore) List(ctx context.Context, parentID string) ([]Entry, error) {
|
||||
var out []Entry
|
||||
err := retryStoreOp(ctx, func() error {
|
||||
l, err := s.c.ListDavFolder(ctx, parentID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entries := make([]Entry, 0, len(l.Folders)+len(l.Files))
|
||||
for _, f := range l.Folders {
|
||||
entries = append(entries, DavFolderToEntry(f, ProviderDAV))
|
||||
}
|
||||
for _, f := range l.Files {
|
||||
entries = append(entries, DavFileToEntry(f, ProviderDAV))
|
||||
}
|
||||
out = entries
|
||||
return nil
|
||||
})
|
||||
return out, err
|
||||
}
|
||||
|
||||
// Stat resolves a folder or file entry by id. A folder answers ListDavFolder
|
||||
// with its own metadata in Current; otherwise the file metadata API is used.
|
||||
func (s *davStore) Stat(ctx context.Context, id string) (Entry, error) {
|
||||
return s.stat(ctx, id)
|
||||
}
|
||||
|
||||
// CreateFolder creates a subfolder under parentID.
|
||||
func (s *davStore) CreateFolder(ctx context.Context, parentID, title string) (Entry, error) {
|
||||
var out Entry
|
||||
err := retryStoreOp(ctx, func() error {
|
||||
f, err := s.c.CreateDavFolder(ctx, parentID, title)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if f == nil {
|
||||
return fmt.Errorf("onlyoffice: dav store: empty create-folder response")
|
||||
}
|
||||
out = DavFolderToEntry(*f, ProviderDAV)
|
||||
return nil
|
||||
})
|
||||
return out, err
|
||||
}
|
||||
|
||||
// Upload streams r into parentID as title. The reader is buffered once so a
|
||||
// retry re-sends the same bytes instead of an exhausted stream.
|
||||
func (s *davStore) Upload(ctx context.Context, parentID, title string, r io.Reader) (Entry, error) {
|
||||
data, err := io.ReadAll(r)
|
||||
if err != nil {
|
||||
return Entry{}, err
|
||||
}
|
||||
var out Entry
|
||||
err = retryStoreOp(ctx, func() error {
|
||||
f, err := s.c.UploadDavFile(ctx, parentID, title, bytes.NewReader(data))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if f == nil {
|
||||
return fmt.Errorf("onlyoffice: dav store: empty upload response")
|
||||
}
|
||||
out = DavFileToEntry(*f, ProviderDAV)
|
||||
return nil
|
||||
})
|
||||
return out, err
|
||||
}
|
||||
|
||||
// Download streams the file bytes into w.
|
||||
func (s *davStore) Download(ctx context.Context, id string, w io.Writer) (int64, error) {
|
||||
var n int64
|
||||
err := retryStoreOp(ctx, func() error {
|
||||
var e error
|
||||
n, e = s.c.DownloadDavFile(ctx, id, w)
|
||||
return e
|
||||
})
|
||||
return n, err
|
||||
}
|
||||
|
||||
// Move moves ids into parentID, splitting folders from files.
|
||||
func (s *davStore) Move(ctx context.Context, ids []string, parentID string) error {
|
||||
folders, files, err := s.split(ctx, ids)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(folders) == 0 && len(files) == 0 {
|
||||
return nil
|
||||
}
|
||||
return retryStoreOp(ctx, func() error {
|
||||
return s.c.MoveDavItems(ctx, folders, files, parentID)
|
||||
})
|
||||
}
|
||||
|
||||
// Copy copies ids into parentID, splitting folders from files.
|
||||
func (s *davStore) Copy(ctx context.Context, ids []string, parentID string) error {
|
||||
folders, files, err := s.split(ctx, ids)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(folders) == 0 && len(files) == 0 {
|
||||
return nil
|
||||
}
|
||||
return retryStoreOp(ctx, func() error {
|
||||
return s.c.CopyDavItems(ctx, folders, files, parentID)
|
||||
})
|
||||
}
|
||||
|
||||
// Rename renames a folder or file.
|
||||
func (s *davStore) Rename(ctx context.Context, id, title string) error {
|
||||
e, err := s.stat(ctx, id)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return retryStoreOp(ctx, func() error {
|
||||
if e.Kind == Folder {
|
||||
return s.c.RenameDavFolder(ctx, id, title)
|
||||
}
|
||||
return s.c.RenameDavFile(ctx, id, title)
|
||||
})
|
||||
}
|
||||
|
||||
// Delete removes ids, splitting folders from files.
|
||||
func (s *davStore) Delete(ctx context.Context, ids []string) error {
|
||||
folders, files, err := s.split(ctx, ids)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(folders) == 0 && len(files) == 0 {
|
||||
return nil
|
||||
}
|
||||
return retryStoreOp(ctx, func() error {
|
||||
return s.c.DeleteDavItems(ctx, folders, files)
|
||||
})
|
||||
}
|
||||
|
||||
// stat resolves a single id to a folder or file Entry.
|
||||
func (s *davStore) stat(ctx context.Context, id string) (Entry, error) {
|
||||
var out Entry
|
||||
err := retryStoreOp(ctx, func() error {
|
||||
if l, err := s.c.ListDavFolder(ctx, id); err == nil {
|
||||
if l != nil && l.Current.ID != "" && l.Current.ID == id {
|
||||
out = DavFolderToEntry(l.Current, ProviderDAV)
|
||||
return nil
|
||||
}
|
||||
} else if Transient(err) {
|
||||
return err
|
||||
}
|
||||
f, err := s.c.GetFile(ctx, id)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out = FileEntryToEntry(f, ProviderDAV)
|
||||
return nil
|
||||
})
|
||||
return out, err
|
||||
}
|
||||
|
||||
// split classifies ids into folder and file id lists.
|
||||
func (s *davStore) split(ctx context.Context, ids []string) (folders, files []string, err error) {
|
||||
for _, id := range ids {
|
||||
e, err := s.stat(ctx, id)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if e.Kind == Folder {
|
||||
folders = append(folders, id)
|
||||
} else {
|
||||
files = append(files, id)
|
||||
}
|
||||
}
|
||||
return folders, files, nil
|
||||
}
|
||||
+325
@@ -0,0 +1,325 @@
|
||||
package onlyoffice
|
||||
|
||||
// Elasticsearch backend of the unified file client (epic #34, F3 #37).
|
||||
//
|
||||
// OnlyOffice full-text search runs on Elasticsearch (index `files_file`, NEST
|
||||
// client on the server). The REST endpoint GET /api/2.0/files/@search/{query}
|
||||
// only searches file names in the database, so content search needs a direct
|
||||
// ES query. The live server is Elasticsearch 7.16.3; the request shape below
|
||||
// is plain REST and stays stdlib-only, matching the repo's no-extra-deps rule.
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// The canonical model (Kind, Entry, SearchQuery, SearchHit, Searcher) lives in
|
||||
// file_core.go (F1 #35).
|
||||
|
||||
const (
|
||||
defaultESIndex = "files_file"
|
||||
defaultESLimit = 20
|
||||
maxESLimit = 1000
|
||||
maxESResponseSize = 8 << 20
|
||||
)
|
||||
|
||||
// ESConfig configures the direct Elasticsearch searcher.
|
||||
type ESConfig struct {
|
||||
URL string // scheme://host:port of the ES HTTP endpoint
|
||||
Index string // index name, default files_file
|
||||
Tenant string // tenantId filter, empty means all tenants
|
||||
}
|
||||
|
||||
// ESConfigFromEnv reads ONLYOFFICE_ES_URL, ONLYOFFICE_ES_INDEX (default
|
||||
// files_file) and ONLYOFFICE_TENANT. The library never loads dotfiles — the
|
||||
// CLI does that.
|
||||
func ESConfigFromEnv() ESConfig {
|
||||
return ESConfig{
|
||||
URL: strings.TrimRight(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")), "/"),
|
||||
Index: firstNonEmpty(os.Getenv("ONLYOFFICE_ES_INDEX"), defaultESIndex),
|
||||
Tenant: strings.TrimSpace(os.Getenv("ONLYOFFICE_TENANT")),
|
||||
}
|
||||
}
|
||||
|
||||
// ESSearcher queries OnlyOffice's Elasticsearch index directly for file name
|
||||
// and document content.
|
||||
type ESSearcher struct {
|
||||
cfg ESConfig
|
||||
http *http.Client
|
||||
}
|
||||
|
||||
// NewESSearcher returns a searcher for the OnlyOffice Elasticsearch index.
|
||||
// The URL is required; an empty index falls back to files_file.
|
||||
func NewESSearcher(cfg ESConfig) (*ESSearcher, error) {
|
||||
if strings.TrimSpace(cfg.URL) == "" {
|
||||
return nil, fmt.Errorf("onlyoffice: elasticsearch URL is empty (set ONLYOFFICE_ES_URL)")
|
||||
}
|
||||
cfg.URL = strings.TrimRight(cfg.URL, "/")
|
||||
if cfg.Index == "" {
|
||||
cfg.Index = defaultESIndex
|
||||
}
|
||||
return &ESSearcher{cfg: cfg, http: &http.Client{Timeout: 30 * time.Second}}, nil
|
||||
}
|
||||
|
||||
// Name implements Searcher.
|
||||
func (s *ESSearcher) Name() string { return "elasticsearch" }
|
||||
|
||||
// Search runs a multi_match over title (and, when q.InContent is set,
|
||||
// document.attachment.content), filtered by tenant and optional folder.
|
||||
func (s *ESSearcher) Search(ctx context.Context, q SearchQuery) ([]SearchHit, error) {
|
||||
q.Text = strings.TrimSpace(q.Text)
|
||||
if q.Text == "" {
|
||||
return nil, fmt.Errorf("onlyoffice: empty search query")
|
||||
}
|
||||
body, err := json.Marshal(esSearchRequest(q, s.cfg.Tenant))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: build elasticsearch query: %w", err)
|
||||
}
|
||||
endpoint := s.cfg.URL + "/" + s.cfg.Index + "/_search"
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, endpoint, bytes.NewReader(body))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
req.Header.Set("Accept", "application/json")
|
||||
resp, err := s.http.Do(req)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: elasticsearch search: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
raw, err := io.ReadAll(io.LimitReader(resp.Body, maxESResponseSize))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("onlyoffice: elasticsearch search: %d %s", resp.StatusCode, truncate(string(raw), 400))
|
||||
}
|
||||
return parseESSearchResponse(raw)
|
||||
}
|
||||
|
||||
// esSearchRequest builds the ES query body. Pure, so it is unit-tested.
|
||||
func esSearchRequest(q SearchQuery, tenant string) esRequest {
|
||||
limit := q.Limit
|
||||
if limit <= 0 {
|
||||
limit = defaultESLimit
|
||||
}
|
||||
if limit > maxESLimit {
|
||||
limit = maxESLimit
|
||||
}
|
||||
fields := []string{"title^2"}
|
||||
if q.InContent {
|
||||
fields = append(fields, "document.attachment.content")
|
||||
}
|
||||
var must []esClause
|
||||
if q.Substring {
|
||||
for _, term := range strings.Fields(strings.ToLower(q.Text)) {
|
||||
if term = escapeWildcard(term); term != "" {
|
||||
must = append(must, esClause{Wildcard: map[string]any{"title": "*" + term + "*"}})
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(must) == 0 {
|
||||
must = []esClause{{MultiMatch: &esMultiMatch{Query: q.Text, Fields: fields}}}
|
||||
}
|
||||
|
||||
var filter []esClause
|
||||
if t := strings.TrimSpace(tenant); t != "" {
|
||||
filter = append(filter, esClause{Term: map[string]any{"tenantId": numericOrString(t)}})
|
||||
}
|
||||
if f := strings.TrimSpace(q.FolderID); f != "" {
|
||||
// folders is an ES nested field; a plain term on folders.folderId would
|
||||
// not match. The stored Folders list holds every ancestor id, so
|
||||
// filtering by a project root id scopes to its whole subtree.
|
||||
filter = append(filter, esClause{Nested: &esNested{
|
||||
Path: "folders",
|
||||
Query: esNestedTerm{Term: map[string]any{"folders.folderId": f}},
|
||||
}})
|
||||
}
|
||||
for _, ext := range normalizeExtensions(q.Extensions) {
|
||||
filter = append(filter, esClause{Wildcard: map[string]any{"title": "*." + ext}})
|
||||
}
|
||||
|
||||
highlightFields := map[string]struct{}{"title": {}}
|
||||
if q.InContent {
|
||||
highlightFields["document.attachment.content"] = struct{}{}
|
||||
}
|
||||
return esRequest{
|
||||
Size: limit,
|
||||
Source: []string{"id", "title", "folders"},
|
||||
Query: esQuery{Bool: esBool{Must: must, Filter: filter}},
|
||||
Highlight: esHighlight{PreTags: []string{"<em>"}, PostTags: []string{"</em>"}, Fields: highlightFields},
|
||||
}
|
||||
}
|
||||
|
||||
// normalizeExtensions lowercases, trims leading dots and drops empties.
|
||||
func normalizeExtensions(exts []string) []string {
|
||||
out := make([]string, 0, len(exts))
|
||||
seen := map[string]bool{}
|
||||
for _, e := range exts {
|
||||
e = strings.ToLower(strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(e), ".")))
|
||||
if e == "" || seen[e] {
|
||||
continue
|
||||
}
|
||||
seen[e] = true
|
||||
out = append(out, e)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// numericOrString keeps an integer-looking filter value numeric (tenantId is
|
||||
// a long) and leaves anything else as a string (folderId is a text token).
|
||||
func numericOrString(s string) any {
|
||||
if n, err := strconv.ParseInt(s, 10, 64); err == nil {
|
||||
return n
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// esRequest is the subset of the ES query DSL this client emits.
|
||||
type esRequest struct {
|
||||
Size int `json:"size"`
|
||||
Source []string `json:"_source"`
|
||||
Query esQuery `json:"query"`
|
||||
Highlight esHighlight `json:"highlight"`
|
||||
}
|
||||
|
||||
type esQuery struct {
|
||||
Bool esBool `json:"bool"`
|
||||
}
|
||||
|
||||
type esBool struct {
|
||||
Must []esClause `json:"must,omitempty"`
|
||||
Filter []esClause `json:"filter,omitempty"`
|
||||
}
|
||||
|
||||
type esClause struct {
|
||||
MultiMatch *esMultiMatch `json:"multi_match,omitempty"`
|
||||
Term map[string]any `json:"term,omitempty"`
|
||||
Terms map[string]any `json:"terms,omitempty"`
|
||||
Wildcard map[string]any `json:"wildcard,omitempty"`
|
||||
Nested *esNested `json:"nested,omitempty"`
|
||||
}
|
||||
|
||||
type esNested struct {
|
||||
Path string `json:"path"`
|
||||
Query esNestedTerm `json:"query"`
|
||||
}
|
||||
|
||||
type esNestedTerm struct {
|
||||
Term map[string]any `json:"term,omitempty"`
|
||||
}
|
||||
|
||||
// escapeWildcard strips ES wildcard metacharacters from a user term so a query
|
||||
// cannot inject wildcard syntax. Pure, so it is unit-tested.
|
||||
func escapeWildcard(s string) string {
|
||||
return strings.NewReplacer("*", "", "?", "", `\`, "").Replace(s)
|
||||
}
|
||||
|
||||
type esMultiMatch struct {
|
||||
Query string `json:"query"`
|
||||
Fields []string `json:"fields"`
|
||||
}
|
||||
|
||||
type esHighlight struct {
|
||||
PreTags []string `json:"pre_tags,omitempty"`
|
||||
PostTags []string `json:"post_tags,omitempty"`
|
||||
Fields map[string]struct{} `json:"fields"`
|
||||
}
|
||||
|
||||
// esResponse is the subset of an ES search response we consume.
|
||||
type esResponse struct {
|
||||
Took int `json:"took"`
|
||||
Hits struct {
|
||||
Total struct {
|
||||
Value int `json:"value"`
|
||||
Relation string `json:"relation"`
|
||||
} `json:"total"`
|
||||
Hits []esResponseHit `json:"hits"`
|
||||
} `json:"hits"`
|
||||
}
|
||||
|
||||
type esResponseHit struct {
|
||||
ID string `json:"_id"`
|
||||
Score float64 `json:"_score"`
|
||||
Source struct {
|
||||
ID int `json:"id"`
|
||||
Title string `json:"title"`
|
||||
Folders []struct {
|
||||
FolderID string `json:"folderId"`
|
||||
ID int `json:"id"`
|
||||
} `json:"folders"`
|
||||
} `json:"_source"`
|
||||
Highlight map[string][]string `json:"highlight"`
|
||||
}
|
||||
|
||||
// parseESSearchResponse converts an ES search response into SearchHit values.
|
||||
// Pure, so it is unit-tested.
|
||||
func parseESSearchResponse(raw []byte) ([]SearchHit, error) {
|
||||
var r esResponse
|
||||
if err := json.Unmarshal(raw, &r); err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: decode elasticsearch response: %w", err)
|
||||
}
|
||||
hits := make([]SearchHit, 0, len(r.Hits.Hits))
|
||||
for _, h := range r.Hits.Hits {
|
||||
id := strconv.Itoa(h.Source.ID)
|
||||
if h.Source.ID == 0 {
|
||||
id = h.ID
|
||||
}
|
||||
// Folders is the ancestor breadcrumb in root → leaf order, so the last
|
||||
// entry is the immediate parent (the previous "first" value was the
|
||||
// project root, which made every result look like it lived in #522).
|
||||
var parent string
|
||||
path := make([]string, 0, len(h.Source.Folders))
|
||||
for _, f := range h.Source.Folders {
|
||||
if strings.TrimSpace(f.FolderID) == "" {
|
||||
continue
|
||||
}
|
||||
path = append(path, f.FolderID)
|
||||
}
|
||||
if len(path) > 0 {
|
||||
parent = path[len(path)-1]
|
||||
}
|
||||
hits = append(hits, SearchHit{
|
||||
Entry: Entry{
|
||||
ID: id,
|
||||
ParentID: parent,
|
||||
Title: h.Source.Title,
|
||||
Kind: File,
|
||||
Provider: "elasticsearch",
|
||||
},
|
||||
Score: h.Score,
|
||||
Highlight: esHighlightText(h.Highlight),
|
||||
Path: path,
|
||||
})
|
||||
}
|
||||
return hits, nil
|
||||
}
|
||||
|
||||
var esHighlightTag = regexp.MustCompile(`</?em[^>]*>`)
|
||||
|
||||
// esHighlightText flattens a highlight map into one plain-text snippet,
|
||||
// preferring the content fragment over the title. It covers both the
|
||||
// OnlyOffice content field and the own-index "content" field.
|
||||
func esHighlightText(hl map[string][]string) string {
|
||||
for _, key := range []string{"document.attachment.content", "content", "title"} {
|
||||
frags := hl[key]
|
||||
if len(frags) == 0 {
|
||||
continue
|
||||
}
|
||||
clean := make([]string, 0, len(frags))
|
||||
for _, f := range frags {
|
||||
clean = append(clean, esHighlightTag.ReplaceAllString(f, ""))
|
||||
}
|
||||
return strings.Join(clean, " … ")
|
||||
}
|
||||
return ""
|
||||
}
|
||||
@@ -0,0 +1,226 @@
|
||||
//go:build integration
|
||||
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
// Default known fixtures for TestIntegrationESFacadeUsesES on the live index.
|
||||
const (
|
||||
defaultESTestQuery = "Rechnung_986-2025.pdf"
|
||||
defaultESTestSubstring = "rechnung 2025"
|
||||
defaultESTestFolder = "522"
|
||||
)
|
||||
|
||||
// TestIntegrationESFacadeUsesES proves that the public search path — `oo search`
|
||||
// and Client.Files().Search() — really runs against the OnlyOffice
|
||||
// Elasticsearch backend and not the REST @search endpoint, which only looks at
|
||||
// file names in the database and is not a Searcher at all (see
|
||||
// docs/elasticsearch.md). It pins the concrete backend and checks that a known
|
||||
// document comes back with a non-empty id and folder path.
|
||||
//
|
||||
// Requires ONLYOFFICE_ES_URL only — the query never touches the REST API, so no
|
||||
// OnlyOffice credentials are needed. Skips when it is missing. The fixture is
|
||||
// overridable with ONLYOFFICE_ES_TEST_QUERY, ONLYOFFICE_ES_TEST_TITLE,
|
||||
// ONLYOFFICE_ES_TEST_SUBSTRING and ONLYOFFICE_ES_TEST_FOLDER.
|
||||
func TestIntegrationESFacadeUsesES(t *testing.T) {
|
||||
esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL"))
|
||||
if esURL == "" {
|
||||
t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test")
|
||||
}
|
||||
query := firstNonEmpty(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_TEST_QUERY")), defaultESTestQuery)
|
||||
wantTitle := firstNonEmpty(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_TEST_TITLE")), query)
|
||||
substring := firstNonEmpty(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_TEST_SUBSTRING")), defaultESTestSubstring)
|
||||
folder := firstNonEmpty(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_TEST_FOLDER")), defaultESTestFolder)
|
||||
|
||||
c := NewClient(Credentials{})
|
||||
searcher, err := c.Files().Search()
|
||||
if err != nil {
|
||||
t.Fatalf("Files().Search(): %v", err)
|
||||
}
|
||||
if got := searcher.Name(); got != ProviderES {
|
||||
t.Fatalf("searcher.Name() = %q, want %q (REST @search is not a Searcher)", got, ProviderES)
|
||||
}
|
||||
if _, ok := searcher.(*ESSearcher); !ok {
|
||||
t.Fatalf("searcher = %T, want *ESSearcher (ES backend, not REST)", searcher)
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
||||
defer cancel()
|
||||
|
||||
// Known file name: multi_match over title, as `oo search <file>` does.
|
||||
start := time.Now()
|
||||
hits, err := searcher.Search(ctx, SearchQuery{Text: query, Limit: 20})
|
||||
if err != nil {
|
||||
t.Fatalf("Search(%q): %v", query, err)
|
||||
}
|
||||
t.Logf("ES facade query %q: %d hits in %s", query, len(hits), time.Since(start))
|
||||
|
||||
known := findHitByTitle(hits, wantTitle)
|
||||
if known == nil {
|
||||
t.Fatalf("query %q returned %d hits, none titled %q", query, len(hits), wantTitle)
|
||||
}
|
||||
if strings.TrimSpace(known.ID) == "" {
|
||||
t.Errorf("hit %q has empty id", known.Title)
|
||||
}
|
||||
if len(known.Path) == 0 {
|
||||
t.Errorf("hit %q has empty path", known.Title)
|
||||
}
|
||||
if known.Provider != ProviderES {
|
||||
t.Errorf("hit provider = %q, want %q", known.Provider, ProviderES)
|
||||
}
|
||||
|
||||
// Substring + folder subtree, as `oo search <terms> --substring --folder N`
|
||||
// does: wildcard terms ANDed together, scoped to the folder's subtree.
|
||||
start = time.Now()
|
||||
subHits, err := searcher.Search(ctx, SearchQuery{Text: substring, Substring: true, FolderID: folder, Limit: 200})
|
||||
if err != nil {
|
||||
t.Fatalf("substring Search(%q, folder %s): %v", substring, folder, err)
|
||||
}
|
||||
t.Logf("ES facade substring %q folder %s: %d hits in %s", substring, folder, len(subHits), time.Since(start))
|
||||
if len(subHits) == 0 {
|
||||
t.Fatalf("substring query %q in folder %s returned no hits", substring, folder)
|
||||
}
|
||||
if findHitByTitle(subHits, wantTitle) == nil {
|
||||
t.Errorf("substring query %q in folder %s did not return %q", substring, folder, wantTitle)
|
||||
}
|
||||
}
|
||||
|
||||
// findHitByTitle returns the first hit whose title matches, case-insensitively.
|
||||
func findHitByTitle(hits []SearchHit, title string) *SearchHit {
|
||||
for i := range hits {
|
||||
if strings.EqualFold(strings.TrimSpace(hits[i].Title), title) {
|
||||
return &hits[i]
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// TestIntegrationESSearch uploads a throwaway workbook and verifies that the
|
||||
// direct Elasticsearch search finds it by file name and by content.
|
||||
//
|
||||
// The content index (document.attachment.content) is only populated for Office
|
||||
// formats (docx/xlsx/pptx), so the fixture is an xlsx whose cell carries a
|
||||
// unique token. Requires ONLYOFFICE_ES_URL (a reachable ES endpoint — in the
|
||||
// current setup a tunnel to the ES inside the OnlyOffice VM, see
|
||||
// docs/elasticsearch.md) plus the regular REST credentials for the upload.
|
||||
// Skips when either is missing.
|
||||
func TestIntegrationESSearch(t *testing.T) {
|
||||
esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL"))
|
||||
if esURL == "" {
|
||||
t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test")
|
||||
}
|
||||
c := liveClient(t)
|
||||
t.Cleanup(func() { cleanupTestProjects(t, c) })
|
||||
|
||||
stamp := time.Now().UTC().Format("20060102-150405")
|
||||
nameToken := "goesname" + stamp
|
||||
contentToken := "goescontent" + stamp
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
|
||||
defer cancel()
|
||||
|
||||
project, err := c.CreateProject(NewProjectRequest{
|
||||
Title: testProjectPrefix + "es-" + stamp,
|
||||
Description: "go-onlyoffice elasticsearch integration",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("CreateProject: %v", err)
|
||||
}
|
||||
if project.ID == nil {
|
||||
t.Fatal("created project without id")
|
||||
}
|
||||
pid := strconv.Itoa(*project.ID)
|
||||
|
||||
title := nameToken + ".xlsx"
|
||||
localPath := filepath.Join(t.TempDir(), title)
|
||||
book := excelize.NewFile()
|
||||
if err := book.SetCellValue("Sheet1", "A1", "OnlyOffice Elasticsearch content fixture "+contentToken); err != nil {
|
||||
t.Fatalf("SetCellValue: %v", err)
|
||||
}
|
||||
if err := book.SaveAs(localPath); err != nil {
|
||||
t.Fatalf("SaveAs: %v", err)
|
||||
}
|
||||
|
||||
entry, err := c.UploadProjectFile(ctx, pid, localPath)
|
||||
if err != nil {
|
||||
t.Fatalf("UploadProjectFile: %v", err)
|
||||
}
|
||||
fileID := strconv.Itoa(int(FileEntryNumericID(entry)))
|
||||
if fileID == "0" {
|
||||
t.Fatalf("upload returned no file id: %+v", entry)
|
||||
}
|
||||
|
||||
es, err := NewESSearcher(ESConfig{
|
||||
URL: esURL,
|
||||
Index: os.Getenv("ONLYOFFICE_ES_INDEX"),
|
||||
Tenant: os.Getenv("ONLYOFFICE_TENANT"),
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("NewESSearcher: %v", err)
|
||||
}
|
||||
|
||||
// Indexing is asynchronous on the server; poll until the file shows up.
|
||||
// The server's title analyzer splits on whitespace, so the name query is
|
||||
// the full file name token (including extension), as a user would type it.
|
||||
nameHit := waitForHit(t, ctx, es, SearchQuery{Text: title}, fileID)
|
||||
if nameHit.Title != title {
|
||||
t.Errorf("name hit title = %q, want %q", nameHit.Title, title)
|
||||
}
|
||||
contentHit := waitForHit(t, ctx, es, SearchQuery{Text: contentToken, InContent: true}, fileID)
|
||||
if contentHit.Highlight == "" {
|
||||
t.Error("content hit has no highlight fragment")
|
||||
}
|
||||
if !strings.Contains(contentHit.Title, nameToken) {
|
||||
t.Errorf("content hit title = %q, want the uploaded workbook", contentHit.Title)
|
||||
}
|
||||
|
||||
// The content token is absent from the title, so a name-only search must
|
||||
// not return the file — this proves the content field is really queried.
|
||||
if hits := searchQuiet(t, es, SearchQuery{Text: contentToken}); len(hits) != 0 {
|
||||
t.Errorf("name-only search for content token returned %d hits, want 0", len(hits))
|
||||
}
|
||||
}
|
||||
|
||||
// waitForHit polls ES until the file with fileID appears and returns that hit.
|
||||
func waitForHit(t *testing.T, ctx context.Context, s *ESSearcher, q SearchQuery, fileID string) SearchHit {
|
||||
t.Helper()
|
||||
var lastErr error
|
||||
for {
|
||||
hits, err := s.Search(ctx, q)
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
} else {
|
||||
for _, h := range hits {
|
||||
if h.ID == fileID {
|
||||
return h
|
||||
}
|
||||
}
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
t.Fatalf("search %q: file %s not indexed in time (last err: %v)", q.Text, fileID, lastErr)
|
||||
case <-time.After(3 * time.Second):
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func searchQuiet(t *testing.T, s *ESSearcher, q SearchQuery) []SearchHit {
|
||||
t.Helper()
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
hits, err := s.Search(ctx, q)
|
||||
if err != nil {
|
||||
t.Fatalf("Search(%q): %v", q.Text, err)
|
||||
}
|
||||
return hits
|
||||
}
|
||||
+217
@@ -0,0 +1,217 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"reflect"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestESSearchRequestNameOnly(t *testing.T) {
|
||||
got := esSearchRequest(SearchQuery{Text: "Rechnung"}, "1")
|
||||
if got.Size != defaultESLimit {
|
||||
t.Errorf("size = %d, want %d", got.Size, defaultESLimit)
|
||||
}
|
||||
if !reflect.DeepEqual(got.Source, []string{"id", "title", "folders"}) {
|
||||
t.Errorf("_source = %v", got.Source)
|
||||
}
|
||||
if len(got.Query.Bool.Must) != 1 || got.Query.Bool.Must[0].MultiMatch == nil {
|
||||
t.Fatalf("must = %+v, want one multi_match", got.Query.Bool.Must)
|
||||
}
|
||||
mm := got.Query.Bool.Must[0].MultiMatch
|
||||
if mm.Query != "Rechnung" {
|
||||
t.Errorf("query = %q", mm.Query)
|
||||
}
|
||||
if !reflect.DeepEqual(mm.Fields, []string{"title^2"}) {
|
||||
t.Errorf("fields = %v, want title only", mm.Fields)
|
||||
}
|
||||
if _, ok := got.Highlight.Fields["document.attachment.content"]; ok {
|
||||
t.Error("content highlight present without InContent")
|
||||
}
|
||||
if _, ok := got.Highlight.Fields["title"]; !ok {
|
||||
t.Error("title highlight missing")
|
||||
}
|
||||
if len(got.Query.Bool.Filter) != 1 || got.Query.Bool.Filter[0].Term["tenantId"] != int64(1) {
|
||||
t.Errorf("tenant filter = %+v, want numeric tenantId=1", got.Query.Bool.Filter)
|
||||
}
|
||||
}
|
||||
|
||||
func TestESSearchRequestContentFields(t *testing.T) {
|
||||
got := esSearchRequest(SearchQuery{Text: "Mahnung", InContent: true}, "")
|
||||
mm := got.Query.Bool.Must[0].MultiMatch
|
||||
want := []string{"title^2", "document.attachment.content"}
|
||||
if !reflect.DeepEqual(mm.Fields, want) {
|
||||
t.Errorf("fields = %v, want %v", mm.Fields, want)
|
||||
}
|
||||
if _, ok := got.Highlight.Fields["document.attachment.content"]; !ok {
|
||||
t.Error("content highlight missing with InContent")
|
||||
}
|
||||
if len(got.Query.Bool.Filter) != 0 {
|
||||
t.Errorf("filter = %+v, want none without tenant/folder", got.Query.Bool.Filter)
|
||||
}
|
||||
}
|
||||
|
||||
func TestESSearchRequestFiltersAndLimit(t *testing.T) {
|
||||
got := esSearchRequest(SearchQuery{
|
||||
Text: "Storchen",
|
||||
FolderID: "649",
|
||||
Extensions: []string{".PDF", "pdf", "docx"},
|
||||
Limit: 5000,
|
||||
}, "42")
|
||||
if got.Size != maxESLimit {
|
||||
t.Errorf("size = %d, want cap %d", got.Size, maxESLimit)
|
||||
}
|
||||
var tenant, folder, wildcards int
|
||||
for _, f := range got.Query.Bool.Filter {
|
||||
switch {
|
||||
case f.Term != nil && f.Term["tenantId"] != nil:
|
||||
tenant++
|
||||
case f.Nested != nil:
|
||||
folder++
|
||||
if f.Nested.Path != "folders" || f.Nested.Query.Term["folders.folderId"] != "649" {
|
||||
t.Errorf("folder filter = %+v, want nested folders term 649", f.Nested)
|
||||
}
|
||||
case f.Wildcard != nil:
|
||||
wildcards++
|
||||
}
|
||||
}
|
||||
if tenant != 1 || folder != 1 {
|
||||
t.Errorf("term filters tenant=%d folder=%d, want 1 each", tenant, folder)
|
||||
}
|
||||
if wildcards != 2 {
|
||||
t.Errorf("wildcard filters = %d, want deduped PDF+docx", wildcards)
|
||||
}
|
||||
}
|
||||
|
||||
func TestESSearchRequestSubstringAndsTerms(t *testing.T) {
|
||||
got := esSearchRequest(SearchQuery{Text: "Rechnung 2025", Substring: true, FolderID: "522"}, "")
|
||||
if got.Query.Bool.Must[0].MultiMatch != nil {
|
||||
t.Fatalf("substring must not use multi_match: %+v", got.Query.Bool.Must)
|
||||
}
|
||||
if len(got.Query.Bool.Must) != 2 {
|
||||
t.Fatalf("must = %+v, want two ANDed wildcard terms", got.Query.Bool.Must)
|
||||
}
|
||||
want := []string{"*rechnung*", "*2025*"}
|
||||
for i, m := range got.Query.Bool.Must {
|
||||
if m.Wildcard == nil || m.Wildcard["title"] != want[i] {
|
||||
t.Errorf("must[%d] = %+v, want title wildcard %q", i, m, want[i])
|
||||
}
|
||||
}
|
||||
if len(got.Query.Bool.Filter) != 1 || got.Query.Bool.Filter[0].Nested == nil {
|
||||
t.Errorf("folder filter = %+v, want nested", got.Query.Bool.Filter)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEscapeWildcard(t *testing.T) {
|
||||
cases := map[string]string{"*rechnung*": "rechnung", "a?b\\c": "abc", "plain": "plain"}
|
||||
for in, want := range cases {
|
||||
if got := escapeWildcard(in); got != want {
|
||||
t.Errorf("escapeWildcard(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestESSearchRequestRejectsEmptyTextAtSearch(t *testing.T) {
|
||||
s, err := NewESSearcher(ESConfig{URL: "http://localhost:9200"})
|
||||
if err != nil {
|
||||
t.Fatalf("NewESSearcher: %v", err)
|
||||
}
|
||||
if _, err := s.Search(t.Context(), SearchQuery{Text: " "}); err == nil {
|
||||
t.Error("empty query: want error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewESSearcherRequiresURL(t *testing.T) {
|
||||
if _, err := NewESSearcher(ESConfig{}); err == nil {
|
||||
t.Error("empty URL: want error")
|
||||
}
|
||||
s, err := NewESSearcher(ESConfig{URL: "http://es:9200/"})
|
||||
if err != nil {
|
||||
t.Fatalf("NewESSearcher: %v", err)
|
||||
}
|
||||
if s.cfg.Index != defaultESIndex {
|
||||
t.Errorf("index = %q, want %q", s.cfg.Index, defaultESIndex)
|
||||
}
|
||||
if s.cfg.URL != "http://es:9200" {
|
||||
t.Errorf("url = %q, want trimmed", s.cfg.URL)
|
||||
}
|
||||
if s.Name() != "elasticsearch" {
|
||||
t.Errorf("Name() = %q", s.Name())
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeExtensions(t *testing.T) {
|
||||
got := normalizeExtensions([]string{" .PDF ", "pdf", "", "xlsx"})
|
||||
want := []string{"pdf", "xlsx"}
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("normalizeExtensions = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseESSearchResponse(t *testing.T) {
|
||||
raw := []byte(`{
|
||||
"took": 12,
|
||||
"hits": {
|
||||
"total": {"value": 2, "relation": "eq"},
|
||||
"hits": [
|
||||
{
|
||||
"_id": "2395",
|
||||
"_score": 7.31,
|
||||
"_source": {"id": 2395, "title": "Rechnung-4711.pdf",
|
||||
"folders": [{"folderId": "438", "id": 0}, {"folderId": "11", "id": 0}]},
|
||||
"highlight": {
|
||||
"title": ["<em>Rechnung</em>-4711.pdf"],
|
||||
"document.attachment.content": ["… Zahlung der <em>Rechnung</em> …"]
|
||||
}
|
||||
},
|
||||
{
|
||||
"_id": "2318",
|
||||
"_score": 6.02,
|
||||
"_source": {"id": 2318, "title": "Mahnung.pdf", "folders": []},
|
||||
"highlight": {"title": ["<em>Mahnung</em>.pdf"]}
|
||||
}
|
||||
]
|
||||
}
|
||||
}`)
|
||||
hits, err := parseESSearchResponse(raw)
|
||||
if err != nil {
|
||||
t.Fatalf("parseESSearchResponse: %v", err)
|
||||
}
|
||||
if len(hits) != 2 {
|
||||
t.Fatalf("hits = %d, want 2", len(hits))
|
||||
}
|
||||
h0 := hits[0]
|
||||
if h0.ID != "2395" || h0.Title != "Rechnung-4711.pdf" || h0.Kind != File {
|
||||
t.Errorf("hit0 entry = %+v", h0.Entry)
|
||||
}
|
||||
// folders is root → leaf; the immediate parent is the last entry.
|
||||
if h0.ParentID != "11" || !reflect.DeepEqual(h0.Path, []string{"438", "11"}) {
|
||||
t.Errorf("hit0 path = %v parent = %q, want parent 11", h0.Path, h0.ParentID)
|
||||
}
|
||||
if h0.Score != 7.31 {
|
||||
t.Errorf("hit0 score = %v", h0.Score)
|
||||
}
|
||||
if h0.Highlight != "… Zahlung der Rechnung …" {
|
||||
t.Errorf("hit0 highlight = %q, want content fragment", h0.Highlight)
|
||||
}
|
||||
if hits[1].Highlight != "Mahnung.pdf" {
|
||||
t.Errorf("hit1 highlight = %q, want title without tags", hits[1].Highlight)
|
||||
}
|
||||
if hits[1].ParentID != "" || len(hits[1].Path) != 0 {
|
||||
t.Errorf("hit1 path = %v", hits[1].Path)
|
||||
}
|
||||
}
|
||||
|
||||
func TestESSearchRequestJSONShape(t *testing.T) {
|
||||
got := esSearchRequest(SearchQuery{Text: "Rechnung", InContent: true}, "1")
|
||||
b, err := json.Marshal(got)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal: %v", err)
|
||||
}
|
||||
var back map[string]any
|
||||
if err := json.Unmarshal(b, &back); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if _, ok := back["query"].(map[string]any)["bool"]; !ok {
|
||||
t.Errorf("query.bool missing: %s", b)
|
||||
}
|
||||
}
|
||||
+336
@@ -0,0 +1,336 @@
|
||||
package onlyoffice
|
||||
|
||||
// Own full-text index (epic #34, F6 #42).
|
||||
//
|
||||
// The OnlyOffice Elasticsearch index (files_file) only holds extracted content
|
||||
// for Office formats. FileUtility.CanIndex gates extraction by the server
|
||||
// setting files.index.formats, whose default is ".pptx|.xlsx|.docx", so PDFs
|
||||
// are indexed by name only. Instead of patching the server (risky: lost on
|
||||
// upgrade, forces a full reindex) this file implements a second, independent
|
||||
// index (default oo_docs_text) that our own pipeline fills from
|
||||
// internal/docpipe (pdftotext + OCR). The OnlyOffice index is never touched.
|
||||
//
|
||||
// See docs/elasticsearch.md for the decision and the trade-offs.
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
const defaultESTextIndex = "oo_docs_text"
|
||||
|
||||
// ESTextConfig configures the own full-text index.
|
||||
type ESTextConfig struct {
|
||||
URL string // scheme://host:port of the ES HTTP endpoint
|
||||
Index string // index name, default oo_docs_text
|
||||
Tenant string // reserved for future multi-tenant data; unused for now
|
||||
}
|
||||
|
||||
// ESTextConfigFromEnv reads ONLYOFFICE_ES_URL and ONLYOFFICE_ES_TEXT_INDEX
|
||||
// (default oo_docs_text). The library never loads dotfiles — the CLI does that.
|
||||
func ESTextConfigFromEnv() ESTextConfig {
|
||||
return ESTextConfig{
|
||||
URL: strings.TrimRight(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")), "/"),
|
||||
Index: firstNonEmpty(os.Getenv("ONLYOFFICE_ES_TEXT_INDEX"), defaultESTextIndex),
|
||||
Tenant: strings.TrimSpace(os.Getenv("ONLYOFFICE_TENANT")),
|
||||
}
|
||||
}
|
||||
|
||||
// TextDoc is one document in the own full-text index. It is keyed by the
|
||||
// OnlyOffice file id so hits map straight back to Documents entries.
|
||||
type TextDoc struct {
|
||||
ID string `json:"id"`
|
||||
Title string `json:"title"`
|
||||
FolderID string `json:"folder,omitempty"`
|
||||
Ext string `json:"ext,omitempty"`
|
||||
Content string `json:"content"`
|
||||
}
|
||||
|
||||
// TextIndex is the storage/search surface for locally extracted document text.
|
||||
// It complements Searcher: ESSearcher reads OnlyOffice's index, ESTextIndex
|
||||
// reads ours.
|
||||
type TextIndex interface {
|
||||
Put(ctx context.Context, docs []TextDoc) error
|
||||
Delete(ctx context.Context, ids []string) error
|
||||
Search(ctx context.Context, q SearchQuery) ([]SearchHit, error)
|
||||
Name() string
|
||||
}
|
||||
|
||||
// ESTextIndex is a TextIndex (and Searcher) over a dedicated Elasticsearch
|
||||
// index filled by TextIndexer.
|
||||
type ESTextIndex struct {
|
||||
cfg ESTextConfig
|
||||
http *http.Client
|
||||
}
|
||||
|
||||
var (
|
||||
_ TextIndex = (*ESTextIndex)(nil)
|
||||
_ Searcher = (*ESTextIndex)(nil)
|
||||
)
|
||||
|
||||
// NewESTextIndex returns a searcher/writer for the own full-text index. The URL
|
||||
// is required; an empty index falls back to oo_docs_text.
|
||||
func NewESTextIndex(cfg ESTextConfig) (*ESTextIndex, error) {
|
||||
if strings.TrimSpace(cfg.URL) == "" {
|
||||
return nil, fmt.Errorf("onlyoffice: elasticsearch URL is empty (set ONLYOFFICE_ES_URL)")
|
||||
}
|
||||
cfg.URL = strings.TrimRight(cfg.URL, "/")
|
||||
if cfg.Index == "" {
|
||||
cfg.Index = defaultESTextIndex
|
||||
}
|
||||
return &ESTextIndex{cfg: cfg, http: &http.Client{Timeout: 120 * time.Second}}, nil
|
||||
}
|
||||
|
||||
// Name implements Searcher and TextIndex.
|
||||
func (x *ESTextIndex) Name() string { return "es-text" }
|
||||
|
||||
// Index returns the configured index name.
|
||||
func (x *ESTextIndex) Index() string { return x.cfg.Index }
|
||||
|
||||
// esTextMapping pins explicit types: content must stay a plain text field (no
|
||||
// keyword subfield) and folder/ext stay exact keywords for filters.
|
||||
const esTextMapping = `{
|
||||
"mappings": {
|
||||
"properties": {
|
||||
"id": {"type": "keyword"},
|
||||
"title": {"type": "text", "fields": {"keyword": {"type": "keyword", "ignore_above": 512}}},
|
||||
"folder": {"type": "keyword"},
|
||||
"ext": {"type": "keyword"},
|
||||
"content": {"type": "text"}
|
||||
}
|
||||
}
|
||||
}`
|
||||
|
||||
// Ensure creates the index with the explicit mapping. A missing index is
|
||||
// created; an already existing one is left untouched.
|
||||
func (x *ESTextIndex) Ensure(ctx context.Context) error {
|
||||
status, raw, err := x.do(ctx, http.MethodPut, "/"+x.cfg.Index, []byte(esTextMapping), "application/json")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if status == http.StatusOK {
|
||||
return nil
|
||||
}
|
||||
if status == http.StatusBadRequest && bytes.Contains(raw, []byte("resource_already_exists_exception")) {
|
||||
return nil
|
||||
}
|
||||
return fmt.Errorf("onlyoffice: create text index %s: %d %s", x.cfg.Index, status, truncate(string(raw), 300))
|
||||
}
|
||||
|
||||
// Put upserts documents via the bulk API and refreshes so they are immediately
|
||||
// searchable.
|
||||
func (x *ESTextIndex) Put(ctx context.Context, docs []TextDoc) error {
|
||||
if len(docs) == 0 {
|
||||
return nil
|
||||
}
|
||||
status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_bulk?refresh=true", esTextBulkBody(docs), "application/x-ndjson")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if status >= 400 {
|
||||
return fmt.Errorf("onlyoffice: bulk index %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400))
|
||||
}
|
||||
var res esBulkResponse
|
||||
if err := json.Unmarshal(raw, &res); err != nil {
|
||||
return fmt.Errorf("onlyoffice: decode bulk response: %w", err)
|
||||
}
|
||||
if !res.Errors {
|
||||
return nil
|
||||
}
|
||||
return fmt.Errorf("onlyoffice: bulk index %s: %s", x.cfg.Index, res.firstError())
|
||||
}
|
||||
|
||||
// Delete removes documents by OnlyOffice file id. A missing index means there
|
||||
// is nothing to delete.
|
||||
func (x *ESTextIndex) Delete(ctx context.Context, ids []string) error {
|
||||
if len(ids) == 0 {
|
||||
return nil
|
||||
}
|
||||
body, err := json.Marshal(map[string]any{"query": map[string]any{"terms": map[string]any{"id": ids}}})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_delete_by_query?refresh=true", body, "application/json")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if status == http.StatusNotFound {
|
||||
return nil
|
||||
}
|
||||
if status >= 400 {
|
||||
return fmt.Errorf("onlyoffice: delete from %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Search runs a multi_match over title (boosted) and content, with optional
|
||||
// folder and extension filters. A missing index yields no hits, not an error.
|
||||
func (x *ESTextIndex) Search(ctx context.Context, q SearchQuery) ([]SearchHit, error) {
|
||||
q.Text = strings.TrimSpace(q.Text)
|
||||
if q.Text == "" {
|
||||
return nil, fmt.Errorf("onlyoffice: empty search query")
|
||||
}
|
||||
body, err := json.Marshal(esTextSearchRequest(q))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: build elasticsearch query: %w", err)
|
||||
}
|
||||
status, raw, err := x.do(ctx, http.MethodPost, "/"+x.cfg.Index+"/_search", body, "application/json")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if status == http.StatusNotFound {
|
||||
return nil, nil
|
||||
}
|
||||
if status >= 400 {
|
||||
return nil, fmt.Errorf("onlyoffice: search %s: %d %s", x.cfg.Index, status, truncate(string(raw), 400))
|
||||
}
|
||||
return parseESTextResponse(raw)
|
||||
}
|
||||
|
||||
// do sends one request and returns the status and body (bounded). The caller
|
||||
// decides which statuses are errors.
|
||||
func (x *ESTextIndex) do(ctx context.Context, method, path string, body []byte, contentType string) (int, []byte, error) {
|
||||
var r io.Reader
|
||||
if body != nil {
|
||||
r = bytes.NewReader(body)
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, method, x.cfg.URL+path, r)
|
||||
if err != nil {
|
||||
return 0, nil, err
|
||||
}
|
||||
req.Header.Set("Accept", "application/json")
|
||||
if contentType != "" {
|
||||
req.Header.Set("Content-Type", contentType)
|
||||
}
|
||||
resp, err := x.http.Do(req)
|
||||
if err != nil {
|
||||
return 0, nil, fmt.Errorf("onlyoffice: elasticsearch %s: %w", method, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
raw, err := io.ReadAll(io.LimitReader(resp.Body, maxESResponseSize))
|
||||
if err != nil {
|
||||
return resp.StatusCode, nil, err
|
||||
}
|
||||
return resp.StatusCode, raw, nil
|
||||
}
|
||||
|
||||
// esTextBulkBody renders the NDJSON bulk payload. Pure, so it is unit-tested.
|
||||
func esTextBulkBody(docs []TextDoc) []byte {
|
||||
var b bytes.Buffer
|
||||
enc := json.NewEncoder(&b)
|
||||
enc.SetEscapeHTML(false)
|
||||
for _, d := range docs {
|
||||
_ = enc.Encode(map[string]any{"index": map[string]any{"_id": d.ID}})
|
||||
_ = enc.Encode(d)
|
||||
}
|
||||
return b.Bytes()
|
||||
}
|
||||
|
||||
// esTextSearchRequest builds the own-index query. Pure, so it is unit-tested.
|
||||
func esTextSearchRequest(q SearchQuery) esRequest {
|
||||
limit := q.Limit
|
||||
if limit <= 0 {
|
||||
limit = defaultESLimit
|
||||
}
|
||||
if limit > maxESLimit {
|
||||
limit = maxESLimit
|
||||
}
|
||||
fields := []string{"title^2", "content"}
|
||||
must := []esClause{{MultiMatch: &esMultiMatch{Query: q.Text, Fields: fields}}}
|
||||
|
||||
var filter []esClause
|
||||
if f := strings.TrimSpace(q.FolderID); f != "" {
|
||||
filter = append(filter, esClause{Term: map[string]any{"folder": f}})
|
||||
}
|
||||
if exts := normalizeExtensions(q.Extensions); len(exts) > 0 {
|
||||
filter = append(filter, esClause{Terms: map[string]any{"ext": exts}})
|
||||
}
|
||||
|
||||
return esRequest{
|
||||
Size: limit,
|
||||
Source: []string{"id", "title", "folder", "ext"},
|
||||
Query: esQuery{Bool: esBool{Must: must, Filter: filter}},
|
||||
Highlight: esHighlight{PreTags: []string{"<em>"}, PostTags: []string{"</em>"}, Fields: map[string]struct{}{"title": {}, "content": {}}},
|
||||
}
|
||||
}
|
||||
|
||||
// esBulkResponse is the subset of an ES bulk response we consume.
|
||||
type esBulkResponse struct {
|
||||
Errors bool `json:"errors"`
|
||||
Items []map[string]struct {
|
||||
ID string `json:"_id"`
|
||||
Status int `json:"status"`
|
||||
Error *struct {
|
||||
Type string `json:"type"`
|
||||
Reason string `json:"reason"`
|
||||
} `json:"error"`
|
||||
} `json:"items"`
|
||||
}
|
||||
|
||||
// firstError returns a compact description of the first failed bulk item.
|
||||
func (r esBulkResponse) firstError() string {
|
||||
for _, item := range r.Items {
|
||||
for op, res := range item {
|
||||
if res.Error != nil {
|
||||
return fmt.Sprintf("%s %s: %s %s", op, res.ID, res.Error.Type, res.Error.Reason)
|
||||
}
|
||||
}
|
||||
}
|
||||
return "unknown bulk error"
|
||||
}
|
||||
|
||||
// esTextResponse is the subset of an own-index search response we consume.
|
||||
type esTextResponse struct {
|
||||
Hits struct {
|
||||
Total struct {
|
||||
Value int `json:"value"`
|
||||
} `json:"total"`
|
||||
Hits []struct {
|
||||
ID string `json:"_id"`
|
||||
Score float64 `json:"_score"`
|
||||
Source TextDoc `json:"_source"`
|
||||
HL map[string][]string `json:"highlight"`
|
||||
} `json:"hits"`
|
||||
} `json:"hits"`
|
||||
}
|
||||
|
||||
// parseESTextResponse converts an own-index search response into SearchHit
|
||||
// values. Pure, so it is unit-tested.
|
||||
func parseESTextResponse(raw []byte) ([]SearchHit, error) {
|
||||
var r esTextResponse
|
||||
if err := json.Unmarshal(raw, &r); err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: decode elasticsearch response: %w", err)
|
||||
}
|
||||
hits := make([]SearchHit, 0, len(r.Hits.Hits))
|
||||
for _, h := range r.Hits.Hits {
|
||||
id := h.Source.ID
|
||||
if id == "" {
|
||||
id = h.ID
|
||||
}
|
||||
parent := h.Source.FolderID
|
||||
var path []string
|
||||
if parent != "" {
|
||||
path = []string{parent}
|
||||
}
|
||||
hits = append(hits, SearchHit{
|
||||
Entry: Entry{
|
||||
ID: id,
|
||||
ParentID: parent,
|
||||
Title: h.Source.Title,
|
||||
Kind: File,
|
||||
Provider: "es-text",
|
||||
},
|
||||
Score: h.Score,
|
||||
Highlight: esHighlightText(h.HL),
|
||||
Path: path,
|
||||
})
|
||||
}
|
||||
return hits, nil
|
||||
}
|
||||
@@ -0,0 +1,158 @@
|
||||
//go:build integration
|
||||
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/eslider/go-onlyoffice/internal/docpipe"
|
||||
)
|
||||
|
||||
// TestIntegrationESTextIndex verifies the own full-text index end to end
|
||||
// against a live Elasticsearch: create the index with its mapping, index a
|
||||
// document, find it by content (and reject it via a folder filter and after
|
||||
// deletion), then drop the throwaway index.
|
||||
//
|
||||
// Requires ONLYOFFICE_ES_URL (a reachable ES endpoint — in the current setup a
|
||||
// tunnel to the ES inside the OnlyOffice VM, see docs/elasticsearch.md). It
|
||||
// does not need OnlyOffice credentials because no file is downloaded: the
|
||||
// TextIndexer write path is covered by unit tests with a fake extractor.
|
||||
func TestIntegrationESTextIndex(t *testing.T) {
|
||||
esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL"))
|
||||
if esURL == "" {
|
||||
t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test")
|
||||
}
|
||||
stamp := time.Now().UTC().Format("20060102150405")
|
||||
idx, err := NewESTextIndex(ESTextConfig{URL: esURL, Index: "oo_docs_text_it_" + stamp})
|
||||
if err != nil {
|
||||
t.Fatalf("NewESTextIndex: %v", err)
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||
defer cancel()
|
||||
|
||||
t.Cleanup(func() {
|
||||
cleanupCtx, done := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer done()
|
||||
_, _, _ = idx.do(cleanupCtx, http.MethodDelete, "/"+idx.Index(), nil, "")
|
||||
})
|
||||
|
||||
if err := idx.Ensure(ctx); err != nil {
|
||||
t.Fatalf("Ensure: %v", err)
|
||||
}
|
||||
// Ensure is idempotent.
|
||||
if err := idx.Ensure(ctx); err != nil {
|
||||
t.Fatalf("Ensure (second): %v", err)
|
||||
}
|
||||
|
||||
token := "gotes" + stamp
|
||||
doc := TextDoc{
|
||||
ID: "3578",
|
||||
Title: "2026-07-28-S1021-Edelweiss-rechnung.pdf",
|
||||
FolderID: "634",
|
||||
Ext: "pdf",
|
||||
Content: "Begleitzettel SGB XI — Rechnung " + token,
|
||||
}
|
||||
if err := idx.Put(ctx, []TextDoc{doc}); err != nil {
|
||||
t.Fatalf("Put: %v", err)
|
||||
}
|
||||
|
||||
hits, err := idx.Search(ctx, SearchQuery{Text: token})
|
||||
if err != nil {
|
||||
t.Fatalf("Search: %v", err)
|
||||
}
|
||||
if len(hits) != 1 || hits[0].ID != "3578" {
|
||||
t.Fatalf("content search hits = %+v, want doc 3578", hits)
|
||||
}
|
||||
if !strings.Contains(hits[0].Highlight, token) {
|
||||
t.Errorf("highlight = %q, want token", hits[0].Highlight)
|
||||
}
|
||||
|
||||
if hits, err := idx.Search(ctx, SearchQuery{Text: token, FolderID: "999"}); err != nil {
|
||||
t.Fatalf("Search with folder filter: %v", err)
|
||||
} else if len(hits) != 0 {
|
||||
t.Errorf("folder filter returned %d hits, want 0", len(hits))
|
||||
}
|
||||
if hits, err := idx.Search(ctx, SearchQuery{Text: token, Extensions: []string{"docx"}}); err != nil {
|
||||
t.Fatalf("Search with ext filter: %v", err)
|
||||
} else if len(hits) != 0 {
|
||||
t.Errorf("ext filter returned %d hits, want 0", len(hits))
|
||||
}
|
||||
|
||||
if err := idx.Delete(ctx, []string{"3578"}); err != nil {
|
||||
t.Fatalf("Delete: %v", err)
|
||||
}
|
||||
if hits, err := idx.Search(ctx, SearchQuery{Text: token}); err != nil {
|
||||
t.Fatalf("Search after delete: %v", err)
|
||||
} else if len(hits) != 0 {
|
||||
t.Errorf("after delete search returned %d hits, want 0", len(hits))
|
||||
}
|
||||
}
|
||||
|
||||
// TestIntegrationESTextIndexPDFAttachment indexes testdata/pdf-with-attachment.pdf
|
||||
// through the real pipeline (TextIndexer + docpipe: pdfdetach + pdftotext) and
|
||||
// verifies that text living only in the embedded attachment is searchable.
|
||||
//
|
||||
// Requires ONLYOFFICE_ES_URL plus poppler (pdfdetach/pdftotext). No OnlyOffice
|
||||
// credentials are needed: a fixture FileStore serves the PDF bytes.
|
||||
func TestIntegrationESTextIndexPDFAttachment(t *testing.T) {
|
||||
esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL"))
|
||||
if esURL == "" {
|
||||
t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test")
|
||||
}
|
||||
if docpipe.LookPath().PDFDetach == "" {
|
||||
t.Skip("pdfdetach not on PATH — skipping PDF attachment integration test")
|
||||
}
|
||||
pdf, err := os.ReadFile("testdata/pdf-with-attachment.pdf")
|
||||
if err != nil {
|
||||
t.Fatalf("read fixture: %v", err)
|
||||
}
|
||||
|
||||
stamp := time.Now().UTC().Format("20060102150405")
|
||||
idx, err := NewESTextIndex(ESTextConfig{URL: esURL, Index: "oo_docs_text_it_att_" + stamp})
|
||||
if err != nil {
|
||||
t.Fatalf("NewESTextIndex: %v", err)
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||
defer cancel()
|
||||
t.Cleanup(func() {
|
||||
cleanupCtx, done := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer done()
|
||||
_, _, _ = idx.do(cleanupCtx, http.MethodDelete, "/"+idx.Index(), nil, "")
|
||||
})
|
||||
if err := idx.Ensure(ctx); err != nil {
|
||||
t.Fatalf("Ensure: %v", err)
|
||||
}
|
||||
|
||||
store := &textFakeStore{files: map[string][]byte{"9001": pdf}}
|
||||
ix := NewTextIndexer(store, idx)
|
||||
res, err := ix.IndexEntries(ctx, []Entry{{ID: "9001", Title: "scan.pdf", ParentID: "777", Kind: File}}, IndexOptions{MinChars: 1})
|
||||
if err != nil {
|
||||
t.Fatalf("IndexEntries: %v", err)
|
||||
}
|
||||
if res.Indexed != 1 || res.Failed != 0 {
|
||||
t.Fatalf("result = %+v, want one indexed doc", res)
|
||||
}
|
||||
|
||||
// Token appears only inside the embedded goo-note.txt attachment.
|
||||
hits, err := idx.Search(ctx, SearchQuery{Text: "gooattachmenttoken"})
|
||||
if err != nil {
|
||||
t.Fatalf("Search attachment token: %v", err)
|
||||
}
|
||||
if len(hits) != 1 || hits[0].ID != "9001" {
|
||||
t.Fatalf("attachment-token hits = %+v, want doc 9001", hits)
|
||||
}
|
||||
if !strings.Contains(hits[0].Highlight, "gooattachmenttoken") {
|
||||
t.Errorf("highlight = %q, want attachment token", hits[0].Highlight)
|
||||
}
|
||||
// Body text is indexed as before.
|
||||
if hits, err := idx.Search(ctx, SearchQuery{Text: "goobodytoken"}); err != nil {
|
||||
t.Fatalf("Search body token: %v", err)
|
||||
} else if len(hits) != 1 {
|
||||
t.Errorf("body-token hits = %d, want 1", len(hits))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,275 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestESTextSearchRequestShape(t *testing.T) {
|
||||
got := esTextSearchRequest(SearchQuery{
|
||||
Text: "S1021",
|
||||
FolderID: "634",
|
||||
Extensions: []string{".PDF", "pdf"},
|
||||
Limit: 5,
|
||||
})
|
||||
if got.Size != 5 {
|
||||
t.Errorf("size = %d, want 5", got.Size)
|
||||
}
|
||||
mm := got.Query.Bool.Must[0].MultiMatch
|
||||
if mm == nil || !reflect.DeepEqual(mm.Fields, []string{"title^2", "content"}) {
|
||||
t.Fatalf("multi_match = %+v, want title^2 + content", mm)
|
||||
}
|
||||
if _, ok := got.Highlight.Fields["content"]; !ok {
|
||||
t.Error("content highlight missing")
|
||||
}
|
||||
if _, ok := got.Highlight.Fields["title"]; !ok {
|
||||
t.Error("title highlight missing")
|
||||
}
|
||||
var folder, exts int
|
||||
for _, f := range got.Query.Bool.Filter {
|
||||
switch {
|
||||
case f.Term != nil && f.Term["folder"] != nil:
|
||||
folder++
|
||||
if f.Term["folder"] != "634" {
|
||||
t.Errorf("folder term = %+v", f.Term)
|
||||
}
|
||||
case f.Terms != nil:
|
||||
exts++
|
||||
if !reflect.DeepEqual(f.Terms["ext"], []string{"pdf"}) {
|
||||
t.Errorf("ext terms = %+v, want deduped pdf", f.Terms)
|
||||
}
|
||||
}
|
||||
}
|
||||
if folder != 1 || exts != 1 {
|
||||
t.Errorf("filters folder=%d exts=%d, want 1 each", folder, exts)
|
||||
}
|
||||
}
|
||||
|
||||
func TestESTextBulkBody(t *testing.T) {
|
||||
body := esTextBulkBody([]TextDoc{
|
||||
{ID: "3578", Title: "S1021.pdf", FolderID: "634", Ext: "pdf", Content: "Begleitzettel <S1021> & mehr"},
|
||||
{ID: "3579", Title: "S1023.pdf", Ext: "pdf", Content: "x"},
|
||||
})
|
||||
lines := strings.Split(strings.TrimRight(string(body), "\n"), "\n")
|
||||
if len(lines) != 4 {
|
||||
t.Fatalf("bulk body has %d lines, want 4:\n%s", len(lines), body)
|
||||
}
|
||||
if !strings.Contains(lines[0], `"index"`) || !strings.Contains(lines[0], `"_id":"3578"`) {
|
||||
t.Errorf("action line = %q", lines[0])
|
||||
}
|
||||
if !strings.Contains(lines[1], `"content":"Begleitzettel <S1021> & mehr"`) {
|
||||
t.Errorf("source line should keep HTML unescaped, got %q", lines[1])
|
||||
}
|
||||
if !strings.Contains(lines[2], `"_id":"3579"`) {
|
||||
t.Errorf("second action line = %q", lines[2])
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseESTextResponse(t *testing.T) {
|
||||
raw := []byte(`{
|
||||
"hits": {
|
||||
"total": {"value": 1, "relation": "eq"},
|
||||
"hits": [
|
||||
{
|
||||
"_id": "3578",
|
||||
"_score": 3.21,
|
||||
"_source": {"id": "3578", "title": "2026-07-28-S1021-Edelweiss-rechnung.pdf", "folder": "634", "ext": "pdf"},
|
||||
"highlight": {"content": ["Begleitzettel … <em>S1021</em> …"]}
|
||||
}
|
||||
]
|
||||
}
|
||||
}`)
|
||||
hits, err := parseESTextResponse(raw)
|
||||
if err != nil {
|
||||
t.Fatalf("parseESTextResponse: %v", err)
|
||||
}
|
||||
if len(hits) != 1 {
|
||||
t.Fatalf("hits = %d, want 1", len(hits))
|
||||
}
|
||||
h := hits[0]
|
||||
if h.ID != "3578" || h.Title != "2026-07-28-S1021-Edelweiss-rechnung.pdf" || h.Kind != File {
|
||||
t.Errorf("entry = %+v", h.Entry)
|
||||
}
|
||||
if h.ParentID != "634" || !reflect.DeepEqual(h.Path, []string{"634"}) {
|
||||
t.Errorf("path = %v parent = %q", h.Path, h.ParentID)
|
||||
}
|
||||
if h.Provider != "es-text" {
|
||||
t.Errorf("provider = %q", h.Provider)
|
||||
}
|
||||
if h.Highlight != "Begleitzettel … S1021 …" {
|
||||
t.Errorf("highlight = %q, want tags stripped", h.Highlight)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewESTextIndexDefaults(t *testing.T) {
|
||||
if _, err := NewESTextIndex(ESTextConfig{}); err == nil {
|
||||
t.Error("empty URL: want error")
|
||||
}
|
||||
x, err := NewESTextIndex(ESTextConfig{URL: "http://es:9200/"})
|
||||
if err != nil {
|
||||
t.Fatalf("NewESTextIndex: %v", err)
|
||||
}
|
||||
if x.Index() != defaultESTextIndex {
|
||||
t.Errorf("index = %q, want %q", x.Index(), defaultESTextIndex)
|
||||
}
|
||||
if x.cfg.URL != "http://es:9200" {
|
||||
t.Errorf("url = %q, want trimmed", x.cfg.URL)
|
||||
}
|
||||
if x.Name() != "es-text" {
|
||||
t.Errorf("Name() = %q", x.Name())
|
||||
}
|
||||
}
|
||||
|
||||
func TestESTextConfigFromEnvIndexDefault(t *testing.T) {
|
||||
t.Setenv("ONLYOFFICE_ES_URL", "http://es:9200/")
|
||||
t.Setenv("ONLYOFFICE_ES_TEXT_INDEX", "")
|
||||
cfg := ESTextConfigFromEnv()
|
||||
if cfg.Index != defaultESTextIndex {
|
||||
t.Errorf("index = %q, want %q", cfg.Index, defaultESTextIndex)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextIndexerIndexEntries(t *testing.T) {
|
||||
store := &textFakeStore{
|
||||
files: map[string][]byte{"1": []byte("PDFBYTES")},
|
||||
}
|
||||
idx := &textFakeIndex{}
|
||||
ix := NewTextIndexer(store, idx)
|
||||
ix.Extractor = textFakeExtractor{prefix: "TEXT "}
|
||||
|
||||
res, err := ix.IndexEntries(context.Background(), []Entry{
|
||||
{ID: "1", Title: "Rechnung.PDF", ParentID: "649", Kind: File},
|
||||
{ID: "2", Title: "Tabelle.xlsx", ParentID: "649", Kind: File},
|
||||
{ID: "3", Title: "Unterordner", Kind: Folder},
|
||||
}, IndexOptions{})
|
||||
if err != nil {
|
||||
t.Fatalf("IndexEntries: %v", err)
|
||||
}
|
||||
if res.Scanned != 3 || res.Indexed != 1 || res.Skipped != 2 || res.Failed != 0 {
|
||||
t.Errorf("result = %+v, want scanned=3 indexed=1 skipped=2 failed=0", res)
|
||||
}
|
||||
if len(idx.docs) != 1 {
|
||||
t.Fatalf("indexed docs = %d, want 1", len(idx.docs))
|
||||
}
|
||||
got := idx.docs[0]
|
||||
want := TextDoc{ID: "1", Title: "Rechnung.PDF", FolderID: "649", Ext: "pdf", Content: "TEXT PDFBYTES"}
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("doc = %+v, want %+v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextIndexerRecordsExtractionFailure(t *testing.T) {
|
||||
store := &textFakeStore{files: map[string][]byte{"1": []byte("x")}}
|
||||
idx := &textFakeIndex{}
|
||||
ix := NewTextIndexer(store, idx)
|
||||
ix.Extractor = textFailingExtractor{}
|
||||
|
||||
res, err := ix.IndexEntries(context.Background(), []Entry{{ID: "1", Title: "a.pdf", Kind: File}}, IndexOptions{})
|
||||
if err != nil {
|
||||
t.Fatalf("IndexEntries: %v", err)
|
||||
}
|
||||
if res.Indexed != 0 || res.Failed != 1 || len(res.Errors) != 1 {
|
||||
t.Errorf("result = %+v, want one failure recorded", res)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextIndexerPlanFolder(t *testing.T) {
|
||||
store := &textFakeStore{dirs: map[string][]Entry{
|
||||
"root": {
|
||||
{ID: "10", Title: "a.pdf", Kind: File},
|
||||
{ID: "11", Title: "sub", Kind: Folder},
|
||||
},
|
||||
"11": {
|
||||
{ID: "12", Title: "b.PDF", Kind: File},
|
||||
{ID: "13", Title: "c.xlsx", Kind: File},
|
||||
},
|
||||
}}
|
||||
ix := NewTextIndexer(store, &textFakeIndex{})
|
||||
|
||||
flat, err := ix.PlanFolder(context.Background(), "root", IndexOptions{})
|
||||
if err != nil {
|
||||
t.Fatalf("PlanFolder: %v", err)
|
||||
}
|
||||
if len(flat) != 1 || flat[0].ID != "10" {
|
||||
t.Errorf("flat plan = %+v, want only a.pdf", flat)
|
||||
}
|
||||
deep, err := ix.PlanFolder(context.Background(), "root", IndexOptions{Recursive: true, Limit: 10})
|
||||
if err != nil {
|
||||
t.Fatalf("PlanFolder recursive: %v", err)
|
||||
}
|
||||
if len(deep) != 2 {
|
||||
t.Errorf("recursive plan = %d entries, want 2", len(deep))
|
||||
}
|
||||
}
|
||||
|
||||
// --- fakes -----------------------------------------------------------------
|
||||
|
||||
type textFakeStore struct {
|
||||
dirs map[string][]Entry
|
||||
files map[string][]byte
|
||||
stat map[string]Entry
|
||||
}
|
||||
|
||||
func (f *textFakeStore) Name() string { return "fake" }
|
||||
|
||||
func (f *textFakeStore) List(_ context.Context, parentID string) ([]Entry, error) {
|
||||
return f.dirs[parentID], nil
|
||||
}
|
||||
|
||||
func (f *textFakeStore) Stat(_ context.Context, id string) (Entry, error) {
|
||||
if e, ok := f.stat[id]; ok {
|
||||
return e, nil
|
||||
}
|
||||
return Entry{}, fmt.Errorf("not found: %s", id)
|
||||
}
|
||||
|
||||
func (f *textFakeStore) Download(_ context.Context, id string, w io.Writer) (int64, error) {
|
||||
b, ok := f.files[id]
|
||||
if !ok {
|
||||
return 0, fmt.Errorf("no bytes for %s", id)
|
||||
}
|
||||
n, err := w.Write(b)
|
||||
return int64(n), err
|
||||
}
|
||||
|
||||
func (f *textFakeStore) CreateFolder(context.Context, string, string) (Entry, error) {
|
||||
return Entry{}, nil
|
||||
}
|
||||
func (f *textFakeStore) Upload(context.Context, string, string, io.Reader) (Entry, error) {
|
||||
return Entry{}, nil
|
||||
}
|
||||
func (f *textFakeStore) Move(context.Context, []string, string) error { return nil }
|
||||
func (f *textFakeStore) Copy(context.Context, []string, string) error { return nil }
|
||||
func (f *textFakeStore) Rename(context.Context, string, string) error { return nil }
|
||||
func (f *textFakeStore) Delete(context.Context, []string) error { return nil }
|
||||
|
||||
type textFakeIndex struct{ docs []TextDoc }
|
||||
|
||||
func (f *textFakeIndex) Put(_ context.Context, docs []TextDoc) error {
|
||||
f.docs = append(f.docs, docs...)
|
||||
return nil
|
||||
}
|
||||
func (f *textFakeIndex) Delete(context.Context, []string) error { return nil }
|
||||
func (f *textFakeIndex) Search(context.Context, SearchQuery) ([]SearchHit, error) { return nil, nil }
|
||||
func (f *textFakeIndex) Name() string { return "fake" }
|
||||
|
||||
type textFakeExtractor struct{ prefix string }
|
||||
|
||||
func (f textFakeExtractor) Extract(path, _, _ string, _ int) (string, error) {
|
||||
b, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return f.prefix + string(b), nil
|
||||
}
|
||||
|
||||
type textFailingExtractor struct{}
|
||||
|
||||
func (textFailingExtractor) Extract(string, string, string, int) (string, error) {
|
||||
return "", fmt.Errorf("boom")
|
||||
}
|
||||
+249
@@ -0,0 +1,249 @@
|
||||
package onlyoffice
|
||||
|
||||
// Single file client (epic #34, F4 #38). FileClient composes the registered
|
||||
// FileStore and Searcher backends and picks one per operation: REST/DAV for
|
||||
// writes, PostgreSQL (when registered) for fast reads, Elasticsearch for name
|
||||
// and content search. Client.Files returns the facade; it also implements
|
||||
// FileStore, so existing callers keep compiling.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ProviderES is the composed Elasticsearch searcher. The SQL store owns
|
||||
// ProviderPG/ProviderMySQL (file_pg.go); the facade references ProviderPG in
|
||||
// readOrder.
|
||||
const ProviderES = "elasticsearch"
|
||||
|
||||
var (
|
||||
errNoReadBackend = errors.New("onlyoffice: no file backend registered for reads")
|
||||
errNoWriteBackend = errors.New("onlyoffice: no file backend registered for writes")
|
||||
errNoSearcher = errors.New("onlyoffice: no search backend registered (set ONLYOFFICE_ES_URL)")
|
||||
)
|
||||
|
||||
// FileClient is the single entry point for file operations. It holds the
|
||||
// registered backends and the order in which each operation tries them.
|
||||
type FileClient struct {
|
||||
stores map[string]FileStore
|
||||
searchers map[string]Searcher
|
||||
|
||||
readOrder []string
|
||||
writeOrder []string
|
||||
searchOrder []string
|
||||
}
|
||||
|
||||
// newFileClient builds the facade over the built-in REST and DAV stores. The
|
||||
// Elasticsearch searcher is registered when ONLYOFFICE_ES_URL is set; the
|
||||
// missing-credential case is left to Search so read-only commands still work.
|
||||
func (c *Client) newFileClient() *FileClient {
|
||||
f := &FileClient{
|
||||
stores: map[string]FileStore{
|
||||
ProviderREST: &restStore{c: c},
|
||||
ProviderDAV: &davStore{c: c},
|
||||
},
|
||||
searchers: map[string]Searcher{},
|
||||
readOrder: []string{ProviderPG, ProviderMySQL, ProviderREST, ProviderDAV},
|
||||
writeOrder: []string{ProviderREST, ProviderDAV},
|
||||
searchOrder: []string{ProviderES},
|
||||
}
|
||||
if cfg := ESConfigFromEnv(); cfg.URL != "" {
|
||||
if es, err := NewESSearcher(cfg); err == nil {
|
||||
f.searchers[ProviderES] = es
|
||||
}
|
||||
}
|
||||
return f
|
||||
}
|
||||
|
||||
// RegisterStore adds or replaces a named backend (for example the PostgreSQL
|
||||
// read store). The name is matched case-insensitively.
|
||||
func (f *FileClient) RegisterStore(name string, s FileStore) {
|
||||
if f == nil || s == nil {
|
||||
return
|
||||
}
|
||||
name = normalizeProvider(name)
|
||||
if name == "" {
|
||||
return
|
||||
}
|
||||
if f.stores == nil {
|
||||
f.stores = map[string]FileStore{}
|
||||
}
|
||||
f.stores[name] = s
|
||||
}
|
||||
|
||||
// RegisterSearcher adds or replaces a named search backend.
|
||||
func (f *FileClient) RegisterSearcher(name string, s Searcher) {
|
||||
if f == nil || s == nil {
|
||||
return
|
||||
}
|
||||
name = normalizeProvider(name)
|
||||
if name == "" {
|
||||
return
|
||||
}
|
||||
if f.searchers == nil {
|
||||
f.searchers = map[string]Searcher{}
|
||||
}
|
||||
f.searchers[name] = s
|
||||
}
|
||||
|
||||
// Read returns the preferred backend for reads: the SQL store (PostgreSQL or
|
||||
// MySQL) when registered, then REST, then WebDAV.
|
||||
func (f *FileClient) Read() FileStore { return f.firstStore(f.readOrder) }
|
||||
|
||||
// Write returns the preferred backend for writes: REST, then WebDAV.
|
||||
func (f *FileClient) Write() FileStore { return f.firstStore(f.writeOrder) }
|
||||
|
||||
// Search returns the preferred name/content searcher (Elasticsearch), or an
|
||||
// error when no search backend is configured.
|
||||
func (f *FileClient) Search() (Searcher, error) {
|
||||
if f == nil {
|
||||
return nil, errNoSearcher
|
||||
}
|
||||
for _, name := range f.searchOrder {
|
||||
if s := f.searchers[normalizeProvider(name)]; s != nil {
|
||||
return s, nil
|
||||
}
|
||||
}
|
||||
return nil, errNoSearcher
|
||||
}
|
||||
|
||||
// firstStore returns the first registered store in the order.
|
||||
func (f *FileClient) firstStore(order []string) FileStore {
|
||||
if f == nil {
|
||||
return nil
|
||||
}
|
||||
for _, name := range order {
|
||||
if s := f.stores[normalizeProvider(name)]; s != nil {
|
||||
return s
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// orderedStores returns the registered stores in the order.
|
||||
func (f *FileClient) orderedStores(order []string) []FileStore {
|
||||
if f == nil {
|
||||
return nil
|
||||
}
|
||||
out := make([]FileStore, 0, len(order))
|
||||
for _, name := range order {
|
||||
if s := f.stores[normalizeProvider(name)]; s != nil {
|
||||
out = append(out, s)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func normalizeProvider(name string) string {
|
||||
return strings.ToLower(strings.TrimSpace(name))
|
||||
}
|
||||
|
||||
// Name implements FileStore and reports the preferred read backend.
|
||||
func (f *FileClient) Name() string {
|
||||
if s := f.Read(); s != nil {
|
||||
return s.Name()
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// List reads from the preferred backend, falling back to the next read backend
|
||||
// only on a transient error (429/502/503/504).
|
||||
func (f *FileClient) List(ctx context.Context, parentID string) ([]Entry, error) {
|
||||
return fallbackRead(ctx, f.orderedStores(f.readOrder), func(s FileStore) ([]Entry, error) {
|
||||
return s.List(ctx, parentID)
|
||||
})
|
||||
}
|
||||
|
||||
// Stat reads from the preferred backend, with the same transient fallback.
|
||||
func (f *FileClient) Stat(ctx context.Context, id string) (Entry, error) {
|
||||
return fallbackRead(ctx, f.orderedStores(f.readOrder), func(s FileStore) (Entry, error) {
|
||||
return s.Stat(ctx, id)
|
||||
})
|
||||
}
|
||||
|
||||
// Download streams file bytes. It does not fall back: a failed attempt may have
|
||||
// already written partial bytes into w, so a second backend would append.
|
||||
func (f *FileClient) Download(ctx context.Context, id string, w io.Writer) (int64, error) {
|
||||
s := f.Read()
|
||||
if s == nil {
|
||||
return 0, errNoReadBackend
|
||||
}
|
||||
return s.Download(ctx, id, w)
|
||||
}
|
||||
|
||||
// CreateFolder writes to the preferred write backend.
|
||||
func (f *FileClient) CreateFolder(ctx context.Context, parentID, title string) (Entry, error) {
|
||||
s := f.Write()
|
||||
if s == nil {
|
||||
return Entry{}, errNoWriteBackend
|
||||
}
|
||||
return s.CreateFolder(ctx, parentID, title)
|
||||
}
|
||||
|
||||
// Upload writes to the preferred write backend.
|
||||
func (f *FileClient) Upload(ctx context.Context, parentID, title string, r io.Reader) (Entry, error) {
|
||||
s := f.Write()
|
||||
if s == nil {
|
||||
return Entry{}, errNoWriteBackend
|
||||
}
|
||||
return s.Upload(ctx, parentID, title, r)
|
||||
}
|
||||
|
||||
// Move writes to the preferred write backend.
|
||||
func (f *FileClient) Move(ctx context.Context, ids []string, parentID string) error {
|
||||
s := f.Write()
|
||||
if s == nil {
|
||||
return errNoWriteBackend
|
||||
}
|
||||
return s.Move(ctx, ids, parentID)
|
||||
}
|
||||
|
||||
// Copy writes to the preferred write backend.
|
||||
func (f *FileClient) Copy(ctx context.Context, ids []string, parentID string) error {
|
||||
s := f.Write()
|
||||
if s == nil {
|
||||
return errNoWriteBackend
|
||||
}
|
||||
return s.Copy(ctx, ids, parentID)
|
||||
}
|
||||
|
||||
// Rename writes to the preferred write backend.
|
||||
func (f *FileClient) Rename(ctx context.Context, id, title string) error {
|
||||
s := f.Write()
|
||||
if s == nil {
|
||||
return errNoWriteBackend
|
||||
}
|
||||
return s.Rename(ctx, id, title)
|
||||
}
|
||||
|
||||
// Delete writes to the preferred write backend.
|
||||
func (f *FileClient) Delete(ctx context.Context, ids []string) error {
|
||||
s := f.Write()
|
||||
if s == nil {
|
||||
return errNoWriteBackend
|
||||
}
|
||||
return s.Delete(ctx, ids)
|
||||
}
|
||||
|
||||
// fallbackRead runs op against each store in order, moving on only when the
|
||||
// error is transient. Non-transient errors (not found, forbidden) are final.
|
||||
func fallbackRead[T any](ctx context.Context, stores []FileStore, op func(FileStore) (T, error)) (T, error) {
|
||||
var zero T
|
||||
if len(stores) == 0 {
|
||||
return zero, errNoReadBackend
|
||||
}
|
||||
var err error
|
||||
for i, s := range stores {
|
||||
var v T
|
||||
v, err = op(s)
|
||||
if err == nil {
|
||||
return v, nil
|
||||
}
|
||||
if i == len(stores)-1 || !Transient(err) {
|
||||
return zero, err
|
||||
}
|
||||
}
|
||||
return zero, err
|
||||
}
|
||||
@@ -0,0 +1,268 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// fakeStore is a FileStore test double; it records which backend served a call
|
||||
// and returns a canned result or error.
|
||||
type fakeStore struct {
|
||||
name string
|
||||
entries []Entry
|
||||
err error
|
||||
calls *[]string
|
||||
}
|
||||
|
||||
func (f *fakeStore) record(op string) {
|
||||
if f.calls != nil {
|
||||
*f.calls = append(*f.calls, op+":"+f.name)
|
||||
}
|
||||
}
|
||||
|
||||
func (f *fakeStore) Name() string { return f.name }
|
||||
|
||||
func (f *fakeStore) List(_ context.Context, _ string) ([]Entry, error) {
|
||||
f.record("list")
|
||||
if f.err != nil {
|
||||
return nil, f.err
|
||||
}
|
||||
return f.entries, nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) Stat(_ context.Context, id string) (Entry, error) {
|
||||
f.record("stat")
|
||||
if f.err != nil {
|
||||
return Entry{}, f.err
|
||||
}
|
||||
return Entry{ID: id, Title: "t-" + f.name, Provider: f.name}, nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) CreateFolder(_ context.Context, _, title string) (Entry, error) {
|
||||
f.record("mkdir")
|
||||
if f.err != nil {
|
||||
return Entry{}, f.err
|
||||
}
|
||||
return Entry{ID: "new", Title: title, Provider: f.name}, nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) Upload(_ context.Context, _, title string, _ io.Reader) (Entry, error) {
|
||||
f.record("upload")
|
||||
return Entry{ID: "up", Title: title, Provider: f.name}, f.err
|
||||
}
|
||||
|
||||
func (f *fakeStore) Download(_ context.Context, _ string, _ io.Writer) (int64, error) {
|
||||
f.record("download")
|
||||
return 0, f.err
|
||||
}
|
||||
|
||||
func (f *fakeStore) Move(_ context.Context, _ []string, _ string) error {
|
||||
f.record("move")
|
||||
return f.err
|
||||
}
|
||||
|
||||
func (f *fakeStore) Copy(_ context.Context, _ []string, _ string) error {
|
||||
f.record("copy")
|
||||
return f.err
|
||||
}
|
||||
|
||||
func (f *fakeStore) Rename(_ context.Context, _, _ string) error {
|
||||
f.record("rename")
|
||||
return f.err
|
||||
}
|
||||
|
||||
func (f *fakeStore) Delete(_ context.Context, _ []string) error {
|
||||
f.record("delete")
|
||||
return f.err
|
||||
}
|
||||
|
||||
type fakeSearcher struct{ name string }
|
||||
|
||||
func (s *fakeSearcher) Name() string { return s.name }
|
||||
|
||||
func (s *fakeSearcher) Search(_ context.Context, _ SearchQuery) ([]SearchHit, error) {
|
||||
return []SearchHit{{Entry: Entry{Title: s.name}}}, nil
|
||||
}
|
||||
|
||||
func newFacadeTestClient(stores map[string]FileStore, read, write []string) *FileClient {
|
||||
return &FileClient{
|
||||
stores: stores,
|
||||
searchers: map[string]Searcher{},
|
||||
readOrder: read,
|
||||
writeOrder: write,
|
||||
}
|
||||
}
|
||||
|
||||
// TestFileClientIsFileStore guarantees the facade can stand in for the
|
||||
// interface anywhere a plain FileStore is expected.
|
||||
func TestFileClientIsFileStore(t *testing.T) {
|
||||
var _ FileStore = (*FileClient)(nil)
|
||||
}
|
||||
|
||||
func TestClientFilesPrefersRESTForReadsAndWrites(t *testing.T) {
|
||||
c := NewClient(Credentials{})
|
||||
f := c.Files()
|
||||
if got := f.Read().Name(); got != ProviderREST {
|
||||
t.Errorf("Read().Name() = %q, want %q", got, ProviderREST)
|
||||
}
|
||||
if got := f.Write().Name(); got != ProviderREST {
|
||||
t.Errorf("Write().Name() = %q, want %q", got, ProviderREST)
|
||||
}
|
||||
if got := f.Name(); got != ProviderREST {
|
||||
t.Errorf("Name() = %q, want %q", got, ProviderREST)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileClientPostgresTakesReadPriority(t *testing.T) {
|
||||
pg := &fakeStore{name: ProviderPG}
|
||||
f := newFacadeTestClient(
|
||||
map[string]FileStore{ProviderREST: &fakeStore{name: ProviderREST}, ProviderPG: pg},
|
||||
[]string{ProviderPG, ProviderREST},
|
||||
[]string{ProviderREST},
|
||||
)
|
||||
if got := f.Read().Name(); got != ProviderPG {
|
||||
t.Errorf("Read().Name() = %q, want %q", got, ProviderPG)
|
||||
}
|
||||
if got := f.Write().Name(); got != ProviderREST {
|
||||
t.Errorf("Write().Name() = %q, want %q (PG is read-only)", got, ProviderREST)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileClientRegisterStoreNormalizesName(t *testing.T) {
|
||||
pg := &fakeStore{name: "pg"}
|
||||
f := newFacadeTestClient(map[string]FileStore{}, []string{ProviderPG}, nil)
|
||||
f.RegisterStore(" POSTGRES ", pg)
|
||||
if got := f.Read(); got != pg {
|
||||
t.Fatalf("Read() = %v, want registered postgres store", got)
|
||||
}
|
||||
f.RegisterStore("", pg)
|
||||
f.RegisterStore("pg", nil)
|
||||
}
|
||||
|
||||
func TestFileClientReadFallsBackOnlyOnTransient(t *testing.T) {
|
||||
var calls []string
|
||||
primary := &fakeStore{name: "primary", err: fmt.Errorf("onlyoffice: list: 503 unavailable"), calls: &calls}
|
||||
secondary := &fakeStore{name: "secondary", entries: []Entry{{ID: "1"}}, calls: &calls}
|
||||
f := newFacadeTestClient(
|
||||
map[string]FileStore{"primary": primary, "secondary": secondary},
|
||||
[]string{"primary", "secondary"},
|
||||
nil,
|
||||
)
|
||||
got, err := f.List(context.Background(), "root")
|
||||
if err != nil {
|
||||
t.Fatalf("List: %v", err)
|
||||
}
|
||||
if len(got) != 1 || got[0].ID != "1" {
|
||||
t.Fatalf("List() = %+v, want secondary entry", got)
|
||||
}
|
||||
want := []string{"list:primary", "list:secondary"}
|
||||
if fmt.Sprint(calls) != fmt.Sprint(want) {
|
||||
t.Fatalf("call order = %v, want %v", calls, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileClientReadStopsOnPermanentError(t *testing.T) {
|
||||
var calls []string
|
||||
primary := &fakeStore{name: "primary", err: errors.New("onlyoffice: not found"), calls: &calls}
|
||||
secondary := &fakeStore{name: "secondary", entries: []Entry{{ID: "1"}}, calls: &calls}
|
||||
f := newFacadeTestClient(
|
||||
map[string]FileStore{"primary": primary, "secondary": secondary},
|
||||
[]string{"primary", "secondary"},
|
||||
nil,
|
||||
)
|
||||
if _, err := f.List(context.Background(), "root"); err == nil {
|
||||
t.Fatal("expected permanent error to be returned")
|
||||
}
|
||||
if len(calls) != 1 || calls[0] != "list:primary" {
|
||||
t.Fatalf("secondary backend must not run on a permanent error: %v", calls)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileClientWriteUsesWriteBackend(t *testing.T) {
|
||||
var calls []string
|
||||
rest := &fakeStore{name: ProviderREST, calls: &calls}
|
||||
dav := &fakeStore{name: ProviderDAV, calls: &calls}
|
||||
f := newFacadeTestClient(
|
||||
map[string]FileStore{ProviderREST: rest, ProviderDAV: dav},
|
||||
[]string{ProviderREST},
|
||||
[]string{ProviderREST, ProviderDAV},
|
||||
)
|
||||
if _, err := f.CreateFolder(context.Background(), "p", "t"); err != nil {
|
||||
t.Fatalf("CreateFolder: %v", err)
|
||||
}
|
||||
if _, err := f.Upload(context.Background(), "p", "t", nil); err != nil {
|
||||
t.Fatalf("Upload: %v", err)
|
||||
}
|
||||
if len(calls) != 2 || calls[0] != "mkdir:rest" || calls[1] != "upload:rest" {
|
||||
t.Fatalf("write calls = %v, want REST", calls)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileClientWriteWithoutBackend(t *testing.T) {
|
||||
f := newFacadeTestClient(map[string]FileStore{}, nil, nil)
|
||||
if err := f.Delete(context.Background(), []string{"1"}); !errors.Is(err, errNoWriteBackend) {
|
||||
t.Fatalf("Delete err = %v, want errNoWriteBackend", err)
|
||||
}
|
||||
if _, err := f.List(context.Background(), "root"); !errors.Is(err, errNoReadBackend) {
|
||||
t.Fatalf("List err = %v, want errNoReadBackend", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewFileClientReadOrderIncludesSQL(t *testing.T) {
|
||||
f := NewClient(Credentials{}).newFileClient()
|
||||
want := []string{ProviderPG, ProviderMySQL, ProviderREST, ProviderDAV}
|
||||
if fmt.Sprint(f.readOrder) != fmt.Sprint(want) {
|
||||
t.Fatalf("readOrder = %v, want %v", f.readOrder, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestClientFileStoreSQLRoutingWithoutDSN checks that the SQL backend names are
|
||||
// recognised and never yield nil: without a DSN the returned store surfaces the
|
||||
// open error on use.
|
||||
func TestClientFileStoreSQLRoutingWithoutDSN(t *testing.T) {
|
||||
t.Setenv("ONLYOFFICE_DSN", "")
|
||||
t.Setenv("ONLYOFFICE_PG_HOST", "")
|
||||
c := NewClient(Credentials{})
|
||||
for _, name := range []string{"pg", "sql", ProviderPG, ProviderMySQL} {
|
||||
s := c.FileStore(name)
|
||||
if s == nil {
|
||||
t.Fatalf("FileStore(%q) = nil", name)
|
||||
}
|
||||
if _, err := s.Stat(context.Background(), "1"); err == nil {
|
||||
t.Errorf("FileStore(%q).Stat without DSN: want error", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileClientMySQLStoreIsPreferredForReads(t *testing.T) {
|
||||
mysql := &fakeStore{name: ProviderMySQL}
|
||||
f := newFacadeTestClient(
|
||||
map[string]FileStore{ProviderREST: &fakeStore{name: ProviderREST}, ProviderMySQL: mysql},
|
||||
[]string{ProviderPG, ProviderMySQL, ProviderREST},
|
||||
[]string{ProviderREST},
|
||||
)
|
||||
if got := f.Read().Name(); got != ProviderMySQL {
|
||||
t.Errorf("Read().Name() = %q, want %q", got, ProviderMySQL)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileClientSearchSelection(t *testing.T) {
|
||||
f := &FileClient{searchers: map[string]Searcher{}, searchOrder: []string{ProviderES}}
|
||||
_, err := f.Search()
|
||||
if err == nil || !strings.Contains(err.Error(), "ONLYOFFICE_ES_URL") {
|
||||
t.Fatalf("Search without backend = %v, want ONLYOFFICE_ES_URL hint", err)
|
||||
}
|
||||
es := &fakeSearcher{name: "fake-es"}
|
||||
f.RegisterSearcher(ProviderES, es)
|
||||
got, err := f.Search()
|
||||
if err != nil {
|
||||
t.Fatalf("Search: %v", err)
|
||||
}
|
||||
if got.Name() != "fake-es" {
|
||||
t.Fatalf("searcher = %q, want fake-es", got.Name())
|
||||
}
|
||||
}
|
||||
+525
@@ -0,0 +1,525 @@
|
||||
package onlyoffice
|
||||
|
||||
// Read-only SQL backend of the unified file client (epic #34, F2 #36).
|
||||
//
|
||||
// The goal is to read files and folders straight from the Community Server
|
||||
// database, without the REST layer. Research on the live portal (VM
|
||||
// `onlyoffice-v2`) showed the server runs on **MySQL 8.0** (`files_file`,
|
||||
// `files_folder`, `files_folder_tree`, tenant `tenants_tenants`), not
|
||||
// PostgreSQL — see docs/community-server-db.md. The store below therefore
|
||||
// speaks `database/sql` and selects its driver from the DSN, so it works
|
||||
// against the live MySQL today and against PostgreSQL if the portal is ever
|
||||
// migrated. Every query is a SELECT; the write methods of FileStore return
|
||||
// ErrReadOnly.
|
||||
//
|
||||
// Downloads follow the portal's S3/MinIO object layout through the shared
|
||||
// MinIO helper in storage_fallback.go — no HTTP file endpoint is used.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/go-sql-driver/mysql"
|
||||
_ "github.com/jackc/pgx/v5/stdlib"
|
||||
)
|
||||
|
||||
// Provider names for the SQL backend. ProviderPG is the value Name reports for
|
||||
// a PostgreSQL connection and ProviderMySQL for MySQL.
|
||||
const (
|
||||
ProviderPG = "postgres"
|
||||
ProviderMySQL = "mysql"
|
||||
)
|
||||
|
||||
// ErrReadOnly is returned by every FileStore write method of the SQL backend.
|
||||
var ErrReadOnly = errors.New("onlyoffice: sql file store is read-only")
|
||||
|
||||
const (
|
||||
pgConnectTimeout = 10 * time.Second
|
||||
pgSearchLimit = 50
|
||||
pgSearchMaxLimit = 500
|
||||
)
|
||||
|
||||
// PGConfig configures the read-only SQL store. DSN is a driver DSN:
|
||||
// `user:pass@tcp(host:port)/onlyoffice?parseTime=true` for MySQL or a
|
||||
// `postgres://` / libpq keyword string for PostgreSQL. Driver, when set,
|
||||
// forces the engine ("postgres" or "mysql"); otherwise it is detected from the
|
||||
// DSN. Tenant filters rows (empty means all tenants).
|
||||
type PGConfig struct {
|
||||
DSN string
|
||||
Driver string
|
||||
Tenant string
|
||||
}
|
||||
|
||||
// PGConfigFromEnv reads ONLYOFFICE_DSN (or the ONLYOFFICE_PG_* parts),
|
||||
// ONLYOFFICE_PG_DRIVER and the tenant from ONLYOFFICE_PG_TENANT /
|
||||
// ONLYOFFICE_TENANT. The library never loads dotfiles — the CLI does that.
|
||||
func PGConfigFromEnv() PGConfig {
|
||||
dsn := strings.TrimSpace(os.Getenv("ONLYOFFICE_DSN"))
|
||||
if dsn == "" {
|
||||
dsn = pgDSNFromParts()
|
||||
}
|
||||
return PGConfig{
|
||||
DSN: dsn,
|
||||
Driver: strings.TrimSpace(os.Getenv("ONLYOFFICE_PG_DRIVER")),
|
||||
Tenant: firstNonEmpty(os.Getenv("ONLYOFFICE_PG_TENANT"), os.Getenv("ONLYOFFICE_TENANT")),
|
||||
}
|
||||
}
|
||||
|
||||
// pgDSNFromParts builds a libpq keyword DSN from ONLYOFFICE_PG_* variables.
|
||||
// It returns "" unless a host is set, which keeps the MySQL path (ONLYOFFICE_DSN)
|
||||
// the default.
|
||||
func pgDSNFromParts() string {
|
||||
host := strings.TrimSpace(os.Getenv("ONLYOFFICE_PG_HOST"))
|
||||
if host == "" {
|
||||
return ""
|
||||
}
|
||||
port := firstNonEmpty(os.Getenv("ONLYOFFICE_PG_PORT"), "5432")
|
||||
dbname := firstNonEmpty(os.Getenv("ONLYOFFICE_PG_DBNAME"), "onlyoffice")
|
||||
sslmode := firstNonEmpty(os.Getenv("ONLYOFFICE_PG_SSLMODE"), "disable")
|
||||
return fmt.Sprintf("host=%s port=%s user=%s password=%s dbname=%s sslmode=%s",
|
||||
host, port, os.Getenv("ONLYOFFICE_PG_USER"), os.Getenv("ONLYOFFICE_PG_PASSWORD"), dbname, sslmode)
|
||||
}
|
||||
|
||||
// SQLFileStore opens the read-only SQL store from the environment
|
||||
// (PGConfigFromEnv: ONLYOFFICE_DSN or the ONLYOFFICE_PG_* parts). It is the
|
||||
// error-aware counterpart of Client.FileStore("pg"/"sql"), which returns an
|
||||
// errStore when the open fails. The caller owns the returned store and should
|
||||
// close it (the concrete type has a Close method).
|
||||
func (c *Client) SQLFileStore() (FileStore, error) {
|
||||
return NewPGStore(PGConfigFromEnv())
|
||||
}
|
||||
|
||||
// pgStore is a read-only FileStore/Searcher over the Community Server database.
|
||||
type pgStore struct {
|
||||
db *sql.DB
|
||||
driver string
|
||||
tenantID int64
|
||||
hasTenant bool
|
||||
http *http.Client
|
||||
}
|
||||
|
||||
var (
|
||||
_ FileStore = (*pgStore)(nil)
|
||||
_ Searcher = (*pgStore)(nil)
|
||||
)
|
||||
|
||||
// NewPGStore opens the database and verifies connectivity. It never writes.
|
||||
func NewPGStore(cfg PGConfig) (*pgStore, error) {
|
||||
dsn := strings.TrimSpace(cfg.DSN)
|
||||
if dsn == "" {
|
||||
return nil, fmt.Errorf("onlyoffice: sql file store: empty DSN (set ONLYOFFICE_DSN)")
|
||||
}
|
||||
driver := pgDriver(dsn, cfg.Driver)
|
||||
dsn, err := normalizeSQLDSN(driver, dsn)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
db, err := sql.Open(sqlDriverName(driver), dsn)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: sql file store: open %s: %w", driver, err)
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), pgConnectTimeout)
|
||||
defer cancel()
|
||||
if err := db.PingContext(ctx); err != nil {
|
||||
db.Close()
|
||||
return nil, fmt.Errorf("onlyoffice: sql file store: ping %s: %w", driver, err)
|
||||
}
|
||||
s := &pgStore{db: db, driver: driver, http: &http.Client{}}
|
||||
if t := strings.TrimSpace(cfg.Tenant); t != "" {
|
||||
n, err := strconv.ParseInt(t, 10, 64)
|
||||
if err != nil {
|
||||
db.Close()
|
||||
return nil, fmt.Errorf("onlyoffice: sql file store: non-numeric tenant %q", t)
|
||||
}
|
||||
s.tenantID, s.hasTenant = n, true
|
||||
}
|
||||
return s, nil
|
||||
}
|
||||
|
||||
// Close releases the database handle.
|
||||
func (s *pgStore) Close() error { return s.db.Close() }
|
||||
|
||||
// Name implements FileStore and Searcher.
|
||||
func (s *pgStore) Name() string { return s.driver }
|
||||
|
||||
// pgDriver resolves the engine: the explicit value wins, otherwise the DSN
|
||||
// shape decides. A leading postgres:// scheme or a libpq keyword DSN (which
|
||||
// always carries '=') selects PostgreSQL; anything else is MySQL.
|
||||
func pgDriver(dsn, explicit string) string {
|
||||
switch strings.ToLower(strings.TrimSpace(explicit)) {
|
||||
case ProviderPG, "pg", "postgresql", "pgx":
|
||||
return ProviderPG
|
||||
case ProviderMySQL, "mariadb":
|
||||
return ProviderMySQL
|
||||
}
|
||||
l := strings.ToLower(strings.TrimSpace(dsn))
|
||||
switch {
|
||||
case strings.HasPrefix(l, "postgres://"), strings.HasPrefix(l, "postgresql://"):
|
||||
return ProviderPG
|
||||
case strings.HasPrefix(l, "mysql://"), strings.Contains(l, "@tcp("), strings.Contains(l, "@unix("):
|
||||
return ProviderMySQL
|
||||
case strings.Contains(l, "="):
|
||||
return ProviderPG
|
||||
default:
|
||||
return ProviderMySQL
|
||||
}
|
||||
}
|
||||
|
||||
// sqlDriverName maps the engine to its registered database/sql driver.
|
||||
func sqlDriverName(driver string) string {
|
||||
if driver == ProviderPG {
|
||||
return "pgx"
|
||||
}
|
||||
return "mysql"
|
||||
}
|
||||
|
||||
// normalizeSQLDSN converts a mysql:// URL to the go-sql-driver form and forces
|
||||
// parseTime so datetime columns scan into time.Time. PostgreSQL DSNs pass
|
||||
// through untouched.
|
||||
func normalizeSQLDSN(driver, dsn string) (string, error) {
|
||||
if driver != ProviderMySQL {
|
||||
return dsn, nil
|
||||
}
|
||||
if strings.HasPrefix(strings.ToLower(dsn), "mysql://") {
|
||||
converted, err := mysqlDSNFromURL(dsn)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
dsn = converted
|
||||
}
|
||||
cfg, err := mysql.ParseDSN(dsn)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("onlyoffice: sql file store: parse mysql DSN: %w", err)
|
||||
}
|
||||
cfg.ParseTime = true
|
||||
return cfg.FormatDSN(), nil
|
||||
}
|
||||
|
||||
// mysqlDSNFromURL turns mysql://user:pass@host:port/db into the driver DSN.
|
||||
func mysqlDSNFromURL(raw string) (string, error) {
|
||||
u, err := url.Parse(raw)
|
||||
if err != nil || u.Host == "" {
|
||||
return "", fmt.Errorf("onlyoffice: sql file store: bad mysql URL %q", raw)
|
||||
}
|
||||
user := ""
|
||||
if u.User != nil {
|
||||
user = u.User.Username()
|
||||
if p, ok := u.User.Password(); ok {
|
||||
user += ":" + p
|
||||
}
|
||||
}
|
||||
q := u.Query()
|
||||
q.Set("parseTime", "true")
|
||||
return fmt.Sprintf("%s@tcp(%s)/%s?%s", user, u.Host, strings.TrimPrefix(u.Path, "/"), q.Encode()), nil
|
||||
}
|
||||
|
||||
// rebind rewrites '?' placeholders to PostgreSQL's $1..$n. MySQL keeps them.
|
||||
func rebind(query, driver string) string {
|
||||
if driver != ProviderPG {
|
||||
return query
|
||||
}
|
||||
var b strings.Builder
|
||||
b.Grow(len(query) + 8)
|
||||
n := 0
|
||||
for _, r := range query {
|
||||
if r == '?' {
|
||||
n++
|
||||
b.WriteByte('$')
|
||||
b.WriteString(strconv.Itoa(n))
|
||||
continue
|
||||
}
|
||||
b.WriteRune(r)
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// List returns the folders and files directly below parentID, folders first.
|
||||
func (s *pgStore) List(ctx context.Context, parentID string) ([]Entry, error) {
|
||||
pid, err := parseEntryID(parentID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
folders, err := s.queryFolders(ctx, "parent_id = ?", pid)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
files, err := s.queryFiles(ctx, "folder_id = ? AND current_version = 1", pid)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]Entry, 0, len(folders)+len(files))
|
||||
for _, f := range folders {
|
||||
out = append(out, folderRowToEntry(f, s.Name()))
|
||||
}
|
||||
for _, f := range files {
|
||||
out = append(out, fileRowToEntry(f, s.Name()))
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Stat resolves a folder or file id to an Entry. Folders win when both id
|
||||
// spaces overlap (they never do on a real portal, but the lookup is cheap).
|
||||
func (s *pgStore) Stat(ctx context.Context, id string) (Entry, error) {
|
||||
n, err := parseEntryID(id)
|
||||
if err != nil {
|
||||
return Entry{}, err
|
||||
}
|
||||
folders, err := s.queryFolders(ctx, "id = ?", n)
|
||||
if err != nil {
|
||||
return Entry{}, err
|
||||
}
|
||||
if len(folders) > 0 {
|
||||
return folderRowToEntry(folders[0], s.Name()), nil
|
||||
}
|
||||
files, err := s.queryFiles(ctx, "id = ? AND current_version = 1", n)
|
||||
if err != nil {
|
||||
return Entry{}, err
|
||||
}
|
||||
if len(files) == 0 {
|
||||
return Entry{}, fmt.Errorf("onlyoffice: sql file store: id %s not found", id)
|
||||
}
|
||||
return fileRowToEntry(files[0], s.Name()), nil
|
||||
}
|
||||
|
||||
// Download streams the file's current version from the portal's S3/MinIO store.
|
||||
// The object key is reconstructed from the file id and version; the parent
|
||||
// folder id is not part of the key.
|
||||
func (s *pgStore) Download(ctx context.Context, id string, w io.Writer) (int64, error) {
|
||||
n, err := parseEntryID(id)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
files, err := s.queryFiles(ctx, "id = ? AND current_version = 1", n)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if len(files) == 0 {
|
||||
return 0, fmt.Errorf("onlyoffice: sql file store: file %s not found", id)
|
||||
}
|
||||
f := files[0]
|
||||
key := csObjectKey(s.tenantID, f.id, f.version, filepath.Ext(f.title))
|
||||
return downloadMinioObject(ctx, s.http, key, w)
|
||||
}
|
||||
|
||||
// CreateFolder is unavailable: the SQL backend is read-only.
|
||||
func (s *pgStore) CreateFolder(context.Context, string, string) (Entry, error) {
|
||||
return Entry{}, ErrReadOnly
|
||||
}
|
||||
|
||||
// Upload is unavailable: the SQL backend is read-only.
|
||||
func (s *pgStore) Upload(context.Context, string, string, io.Reader) (Entry, error) {
|
||||
return Entry{}, ErrReadOnly
|
||||
}
|
||||
|
||||
// Move is unavailable: the SQL backend is read-only.
|
||||
func (s *pgStore) Move(context.Context, []string, string) error { return ErrReadOnly }
|
||||
|
||||
// Copy is unavailable: the SQL backend is read-only.
|
||||
func (s *pgStore) Copy(context.Context, []string, string) error { return ErrReadOnly }
|
||||
|
||||
// Rename is unavailable: the SQL backend is read-only.
|
||||
func (s *pgStore) Rename(context.Context, string, string) error { return ErrReadOnly }
|
||||
|
||||
// Delete is unavailable: the SQL backend is read-only.
|
||||
func (s *pgStore) Delete(context.Context, []string) error { return ErrReadOnly }
|
||||
|
||||
// Search matches file titles by substring. Content search lives in the
|
||||
// Elasticsearch backend; q.InContent is ignored here.
|
||||
func (s *pgStore) Search(ctx context.Context, q SearchQuery) ([]SearchHit, error) {
|
||||
text := strings.TrimSpace(q.Text)
|
||||
if text == "" {
|
||||
return nil, fmt.Errorf("onlyoffice: empty search query")
|
||||
}
|
||||
limit := q.Limit
|
||||
if limit <= 0 {
|
||||
limit = pgSearchLimit
|
||||
}
|
||||
if limit > pgSearchMaxLimit {
|
||||
limit = pgSearchMaxLimit
|
||||
}
|
||||
|
||||
where := "title LIKE ? AND current_version = 1"
|
||||
args := []any{"%" + text + "%"}
|
||||
if s.hasTenant {
|
||||
where += " AND tenant_id = ?"
|
||||
args = append(args, s.tenantID)
|
||||
}
|
||||
if fid := strings.TrimSpace(q.FolderID); fid != "" {
|
||||
n, err := parseEntryID(fid)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
where += " AND folder_id = ?"
|
||||
args = append(args, n)
|
||||
}
|
||||
for _, ext := range normalizeExtensions(q.Extensions) {
|
||||
where += " AND LOWER(title) LIKE ?"
|
||||
args = append(args, "%."+ext)
|
||||
}
|
||||
query := rebind(`SELECT id, folder_id, title, content_length, version, create_on, modified_on
|
||||
FROM files_file WHERE `+where+` ORDER BY modified_on DESC, id DESC LIMIT ?`, s.driver)
|
||||
args = append(args, limit)
|
||||
|
||||
rows, err := s.db.QueryContext(ctx, query, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: sql search: %w", err)
|
||||
}
|
||||
defer rows.Close()
|
||||
var hits []SearchHit
|
||||
for rows.Next() {
|
||||
f, err := scanFileRow(rows)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
hits = append(hits, SearchHit{Entry: fileRowToEntry(f, s.Name())})
|
||||
}
|
||||
return hits, rows.Err()
|
||||
}
|
||||
|
||||
// queryFolders runs a folder SELECT with the tenant filter applied.
|
||||
func (s *pgStore) queryFolders(ctx context.Context, where string, arg any) ([]pgFolderRow, error) {
|
||||
args := []any{arg}
|
||||
if s.hasTenant {
|
||||
where += " AND tenant_id = ?"
|
||||
args = append(args, s.tenantID)
|
||||
}
|
||||
query := rebind(`SELECT id, parent_id, title, create_on, modified_on
|
||||
FROM files_folder WHERE `+where+` ORDER BY title, id`, s.driver)
|
||||
rows, err := s.db.QueryContext(ctx, query, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: sql list folders: %w", err)
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []pgFolderRow
|
||||
for rows.Next() {
|
||||
var r pgFolderRow
|
||||
if err := rows.Scan(&r.id, &r.parentID, &r.title, &r.created, &r.modified); err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: sql folder row: %w", err)
|
||||
}
|
||||
out = append(out, r)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// queryFiles runs a file SELECT for the current version with the tenant filter.
|
||||
func (s *pgStore) queryFiles(ctx context.Context, where string, arg any) ([]pgFileRow, error) {
|
||||
args := []any{arg}
|
||||
if s.hasTenant {
|
||||
where += " AND tenant_id = ?"
|
||||
args = append(args, s.tenantID)
|
||||
}
|
||||
query := rebind(`SELECT id, folder_id, title, content_length, version, create_on, modified_on
|
||||
FROM files_file WHERE `+where+` ORDER BY title, id`, s.driver)
|
||||
rows, err := s.db.QueryContext(ctx, query, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: sql list files: %w", err)
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []pgFileRow
|
||||
for rows.Next() {
|
||||
f, err := scanFileRow(rows)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, f)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// pgFileRow is one current files_file row.
|
||||
type pgFileRow struct {
|
||||
id int64
|
||||
folderID int64
|
||||
title string
|
||||
size int64
|
||||
version int
|
||||
created time.Time
|
||||
modified time.Time
|
||||
}
|
||||
|
||||
// pgFolderRow is one files_folder row.
|
||||
type pgFolderRow struct {
|
||||
id int64
|
||||
parentID int64
|
||||
title string
|
||||
created time.Time
|
||||
modified time.Time
|
||||
}
|
||||
|
||||
// scanFileRow reads the canonical file column order.
|
||||
func scanFileRow(rows *sql.Rows) (pgFileRow, error) {
|
||||
var f pgFileRow
|
||||
if err := rows.Scan(&f.id, &f.folderID, &f.title, &f.size, &f.version, &f.created, &f.modified); err != nil {
|
||||
return f, fmt.Errorf("onlyoffice: sql file row: %w", err)
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
|
||||
// fileRowToEntry maps a files_file row to the canonical model.
|
||||
func fileRowToEntry(f pgFileRow, provider string) Entry {
|
||||
return Entry{
|
||||
ID: strconv.FormatInt(f.id, 10),
|
||||
ParentID: strconv.FormatInt(f.folderID, 10),
|
||||
Title: f.title,
|
||||
Kind: File,
|
||||
Size: f.size,
|
||||
MIME: mimeForTitle(f.title, ""),
|
||||
Created: f.created.UTC(),
|
||||
Modified: f.modified.UTC(),
|
||||
Version: f.version,
|
||||
Provider: provider,
|
||||
}
|
||||
}
|
||||
|
||||
// folderRowToEntry maps a files_folder row to the canonical model.
|
||||
func folderRowToEntry(f pgFolderRow, provider string) Entry {
|
||||
return Entry{
|
||||
ID: strconv.FormatInt(f.id, 10),
|
||||
ParentID: strconv.FormatInt(f.parentID, 10),
|
||||
Title: f.title,
|
||||
Kind: Folder,
|
||||
Created: f.created.UTC(),
|
||||
Modified: f.modified.UTC(),
|
||||
Provider: provider,
|
||||
}
|
||||
}
|
||||
|
||||
// csObjectKey reconstructs the object key the portal's S3 consumer uses:
|
||||
//
|
||||
// 00/00/<tenant>/files/folder_<shard>/file_<id>/v<version>/content.<ext>
|
||||
//
|
||||
// The shard is the next thousand above the file id (file 3727 -> folder_4000),
|
||||
// NOT the parent folder id — verified live against the MinIO bucket.
|
||||
func csObjectKey(tenant int64, fileID int64, version int, ext string) string {
|
||||
shard := (fileID/1000 + 1) * 1000
|
||||
ext = strings.TrimPrefix(strings.ToLower(strings.TrimSpace(ext)), ".")
|
||||
if ext == "" {
|
||||
ext = "bin"
|
||||
}
|
||||
if version < 1 {
|
||||
version = 1
|
||||
}
|
||||
if tenant <= 0 {
|
||||
tenant = 1
|
||||
}
|
||||
return fmt.Sprintf("00/00/%02d/files/folder_%d/file_%d/v%d/content.%s", tenant, shard, fileID, version, ext)
|
||||
}
|
||||
|
||||
// parseEntryID parses a numeric OnlyOffice id or returns a store error.
|
||||
func parseEntryID(id string) (int64, error) {
|
||||
n, err := strconv.ParseInt(strings.TrimSpace(id), 10, 64)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("onlyoffice: sql file store: non-numeric id %q", id)
|
||||
}
|
||||
return n, nil
|
||||
}
|
||||
@@ -0,0 +1,224 @@
|
||||
//go:build integration
|
||||
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestIntegrationPGStore exercises the read-only SQL backend against the live
|
||||
// Community Server database and cross-checks list/stat/download with the REST
|
||||
// FileStore. It needs ONLYOFFICE_DSN plus the usual ONLYOFFICE_URL/USER/PASS;
|
||||
// ONLYOFFICE_PG_TEST_FILE_ID / ONLYOFFICE_PG_TEST_FOLDER_ID pick a real file
|
||||
// (a file reachable over REST too). Download streams from MinIO, so it also
|
||||
// needs MINIO_ACCESS_KEY/MINIO_SECRET_KEY.
|
||||
//
|
||||
// The live Community Server runs on MySQL; PostgreSQL is supported by the same
|
||||
// code path when the DSN says so.
|
||||
func TestIntegrationPGStore(t *testing.T) {
|
||||
cfg := PGConfigFromEnv()
|
||||
if strings.TrimSpace(cfg.DSN) == "" {
|
||||
t.Skip("ONLYOFFICE_DSN not set — skipping SQL store integration test")
|
||||
}
|
||||
store, err := NewPGStore(cfg)
|
||||
if err != nil {
|
||||
t.Fatalf("NewPGStore: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { _ = store.Close() })
|
||||
t.Logf("sql store backend: %s", store.Name())
|
||||
ctx := context.Background()
|
||||
|
||||
if err := testPGStoreReadOnly(ctx, store); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
fileID := strings.TrimSpace(os.Getenv("ONLYOFFICE_PG_TEST_FILE_ID"))
|
||||
folderID := strings.TrimSpace(os.Getenv("ONLYOFFICE_PG_TEST_FOLDER_ID"))
|
||||
if fileID == "" || folderID == "" {
|
||||
t.Skip("ONLYOFFICE_PG_TEST_FILE_ID / ONLYOFFICE_PG_TEST_FOLDER_ID not set — skipping live comparison")
|
||||
}
|
||||
|
||||
c := liveClient(t)
|
||||
rest := c.Files()
|
||||
|
||||
dbFile, err := store.Stat(ctx, fileID)
|
||||
if err != nil {
|
||||
t.Fatalf("sql Stat(%s): %v", fileID, err)
|
||||
}
|
||||
restFile, err := rest.Stat(ctx, fileID)
|
||||
if err != nil {
|
||||
t.Fatalf("rest Stat(%s): %v", fileID, err)
|
||||
}
|
||||
if dbFile.Kind != File {
|
||||
t.Errorf("sql kind = %v, want file", dbFile.Kind)
|
||||
}
|
||||
if dbFile.ID != restFile.ID || dbFile.Title != restFile.Title || dbFile.ParentID != restFile.ParentID {
|
||||
t.Errorf("stat mismatch sql=%+v rest=%+v", dbFile, restFile)
|
||||
}
|
||||
// GetFile omits contentLength, so size is only comparable when REST has it.
|
||||
if restFile.Size > 0 && dbFile.Size != restFile.Size {
|
||||
t.Errorf("size sql=%d rest=%d", dbFile.Size, restFile.Size)
|
||||
}
|
||||
if d := dbFile.Modified.Sub(restFile.Modified); d > 2*time.Minute || d < -2*time.Minute {
|
||||
t.Errorf("modified sql=%v rest=%v", dbFile.Modified, restFile.Modified)
|
||||
}
|
||||
|
||||
list, err := store.List(ctx, folderID)
|
||||
if err != nil {
|
||||
t.Fatalf("sql List(%s): %v", folderID, err)
|
||||
}
|
||||
if entryByID(list, fileID) == nil {
|
||||
t.Errorf("file %s not in sql List(%s)", fileID, folderID)
|
||||
}
|
||||
|
||||
// Every file the REST layer can see in the folder must be in the SQL list
|
||||
// (the SQL store sees more, so only assert this direction).
|
||||
restList, err := rest.List(ctx, folderID)
|
||||
if err != nil {
|
||||
t.Fatalf("rest List(%s): %v", folderID, err)
|
||||
}
|
||||
dbIDs := make(map[string]bool, len(list))
|
||||
for _, e := range list {
|
||||
dbIDs[e.ID] = true
|
||||
}
|
||||
for _, e := range restList {
|
||||
if e.Kind == File && !dbIDs[e.ID] {
|
||||
t.Errorf("rest file %s (%q) missing from sql list", e.ID, e.Title)
|
||||
}
|
||||
}
|
||||
|
||||
// Download reads the object store, not the database, so it only runs with
|
||||
// the MinIO credentials configured (the portal's S3 layout). Without them
|
||||
// the DSN-only assertions above still prove the SQL reads.
|
||||
if os.Getenv("MINIO_ACCESS_KEY") == "" || os.Getenv("MINIO_SECRET_KEY") == "" {
|
||||
t.Log("MINIO_ACCESS_KEY/MINIO_SECRET_KEY not set — SQL download cross-check skipped")
|
||||
return
|
||||
}
|
||||
|
||||
var buf bytes.Buffer
|
||||
n, err := store.Download(ctx, fileID, &buf)
|
||||
if err != nil {
|
||||
t.Fatalf("sql Download(%s): %v", fileID, err)
|
||||
}
|
||||
if n == 0 || n != dbFile.Size {
|
||||
t.Errorf("sql Download = %d bytes, stat says %d", n, dbFile.Size)
|
||||
}
|
||||
var restBuf bytes.Buffer
|
||||
rn, err := rest.Download(ctx, fileID, &restBuf)
|
||||
if err != nil {
|
||||
t.Fatalf("rest Download(%s): %v", fileID, err)
|
||||
}
|
||||
if rn != n || !bytes.Equal(restBuf.Bytes(), buf.Bytes()) {
|
||||
t.Errorf("download mismatch sql=%d rest=%d bytes", n, rn)
|
||||
}
|
||||
}
|
||||
|
||||
// TestIntegrationSQLFacade proves that reads are served by the SQL store when
|
||||
// it is part of the file client, not by REST. It needs ONLYOFFICE_DSN plus
|
||||
// ONLYOFFICE_PG_TEST_FILE_ID / ONLYOFFICE_PG_TEST_FOLDER_ID and the usual REST
|
||||
// credentials (for the cross-check). Every Entry served by SQL carries
|
||||
// Provider "mysql"; REST entries carry "rest", so the provider is the proof of
|
||||
// which backend answered.
|
||||
func TestIntegrationSQLFacade(t *testing.T) {
|
||||
cfg := PGConfigFromEnv()
|
||||
if strings.TrimSpace(cfg.DSN) == "" {
|
||||
t.Skip("ONLYOFFICE_DSN not set — skipping SQL facade integration test")
|
||||
}
|
||||
fileID := strings.TrimSpace(os.Getenv("ONLYOFFICE_PG_TEST_FILE_ID"))
|
||||
folderID := strings.TrimSpace(os.Getenv("ONLYOFFICE_PG_TEST_FOLDER_ID"))
|
||||
if fileID == "" || folderID == "" {
|
||||
t.Skip("ONLYOFFICE_PG_TEST_FILE_ID / ONLYOFFICE_PG_TEST_FOLDER_ID not set — skipping SQL facade integration test")
|
||||
}
|
||||
c := liveClient(t)
|
||||
ctx := context.Background()
|
||||
|
||||
sqlStore, err := c.SQLFileStore()
|
||||
if err != nil {
|
||||
t.Fatalf("SQLFileStore: %v", err)
|
||||
}
|
||||
if closer, ok := sqlStore.(interface{ Close() error }); ok {
|
||||
t.Cleanup(func() { _ = closer.Close() })
|
||||
}
|
||||
if sqlStore.Name() == ProviderREST {
|
||||
t.Fatalf("SQLFileStore returned REST")
|
||||
}
|
||||
|
||||
// Direct constructor: Client.FileStore("pg") must not be REST either.
|
||||
direct := c.FileStore("pg")
|
||||
if direct == nil || direct.Name() == ProviderREST {
|
||||
t.Fatalf("FileStore(\"pg\") = %v, want SQL backend", direct)
|
||||
}
|
||||
if closer, ok := direct.(interface{ Close() error }); ok {
|
||||
t.Cleanup(func() { _ = closer.Close() })
|
||||
}
|
||||
|
||||
f := c.Files()
|
||||
f.RegisterStore(ProviderPG, sqlStore)
|
||||
if got := f.Read().Name(); got != sqlStore.Name() {
|
||||
t.Fatalf("facade read backend = %q, want %q (SQL)", got, sqlStore.Name())
|
||||
}
|
||||
|
||||
got, err := f.Stat(ctx, fileID)
|
||||
if err != nil {
|
||||
t.Fatalf("facade Stat(%s): %v", fileID, err)
|
||||
}
|
||||
if got.Provider != ProviderMySQL {
|
||||
t.Errorf("facade Stat provider = %q, want %q (SQL, not REST)", got.Provider, ProviderMySQL)
|
||||
}
|
||||
want, err := c.FileStore(ProviderREST).Stat(ctx, fileID)
|
||||
if err != nil {
|
||||
t.Fatalf("rest Stat(%s): %v", fileID, err)
|
||||
}
|
||||
if got.ID != want.ID || got.Title != want.Title || got.ParentID != want.ParentID {
|
||||
t.Errorf("facade SQL stat %+v != REST %+v", got, want)
|
||||
}
|
||||
|
||||
list, err := f.List(ctx, folderID)
|
||||
if err != nil {
|
||||
t.Fatalf("facade List(%s): %v", folderID, err)
|
||||
}
|
||||
entry := entryByID(list, fileID)
|
||||
if entry == nil {
|
||||
t.Fatalf("file %s not in facade List(%s)", fileID, folderID)
|
||||
}
|
||||
if entry.Provider != ProviderMySQL {
|
||||
t.Errorf("facade List provider = %q, want %q", entry.Provider, ProviderMySQL)
|
||||
}
|
||||
|
||||
d, err := direct.Stat(ctx, fileID)
|
||||
if err != nil {
|
||||
t.Fatalf("FileStore(\"pg\").Stat(%s): %v", fileID, err)
|
||||
}
|
||||
if d.Provider != ProviderMySQL {
|
||||
t.Errorf("FileStore(\"pg\") provider = %q, want %q", d.Provider, ProviderMySQL)
|
||||
}
|
||||
}
|
||||
|
||||
// testPGStoreReadOnly asserts that every write method returns ErrReadOnly.
|
||||
func testPGStoreReadOnly(ctx context.Context, s *pgStore) error {
|
||||
if _, err := s.CreateFolder(ctx, "1", "x"); !errors.Is(err, ErrReadOnly) {
|
||||
return errors.New("CreateFolder did not return ErrReadOnly")
|
||||
}
|
||||
if _, err := s.Upload(ctx, "1", "x", strings.NewReader("x")); !errors.Is(err, ErrReadOnly) {
|
||||
return errors.New("Upload did not return ErrReadOnly")
|
||||
}
|
||||
if err := s.Move(ctx, []string{"1"}, "2"); !errors.Is(err, ErrReadOnly) {
|
||||
return errors.New("Move did not return ErrReadOnly")
|
||||
}
|
||||
if err := s.Copy(ctx, []string{"1"}, "2"); !errors.Is(err, ErrReadOnly) {
|
||||
return errors.New("Copy did not return ErrReadOnly")
|
||||
}
|
||||
if err := s.Rename(ctx, "1", "x"); !errors.Is(err, ErrReadOnly) {
|
||||
return errors.New("Rename did not return ErrReadOnly")
|
||||
}
|
||||
if err := s.Delete(ctx, []string{"1"}); !errors.Is(err, ErrReadOnly) {
|
||||
return errors.New("Delete did not return ErrReadOnly")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
+165
@@ -0,0 +1,165 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestRebind(t *testing.T) {
|
||||
mysqlQuery := "SELECT id FROM files_file WHERE folder_id = ? AND title = ? LIMIT ?"
|
||||
if got := rebind(mysqlQuery, ProviderMySQL); got != mysqlQuery {
|
||||
t.Errorf("mysql query changed: %q", got)
|
||||
}
|
||||
want := "SELECT id FROM files_file WHERE folder_id = $1 AND title = $2 LIMIT $3"
|
||||
if got := rebind(mysqlQuery, ProviderPG); got != want {
|
||||
t.Errorf("rebind = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCSPObjectKey(t *testing.T) {
|
||||
cases := []struct {
|
||||
tenant int64
|
||||
fileID int64
|
||||
version int
|
||||
ext string
|
||||
want string
|
||||
}{
|
||||
{1, 2, 1, ".docx", "00/00/01/files/folder_1000/file_2/v1/content.docx"},
|
||||
{1, 999, 1, ".pdf", "00/00/01/files/folder_1000/file_999/v1/content.pdf"},
|
||||
{1, 1000, 1, ".xlsx", "00/00/01/files/folder_2000/file_1000/v1/content.xlsx"},
|
||||
{1, 3727, 1, ".pdf", "00/00/01/files/folder_4000/file_3727/v1/content.pdf"},
|
||||
{1, 22484, 1, ".PDF", "00/00/01/files/folder_23000/file_22484/v1/content.pdf"},
|
||||
{1, 4, 6, "xlsx", "00/00/01/files/folder_1000/file_4/v6/content.xlsx"},
|
||||
{0, 7, 0, "", "00/00/01/files/folder_1000/file_7/v1/content.bin"},
|
||||
{2, 11, 3, ".doc", "00/00/02/files/folder_1000/file_11/v3/content.doc"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
if got := csObjectKey(tc.tenant, tc.fileID, tc.version, tc.ext); got != tc.want {
|
||||
t.Errorf("csObjectKey(%d,%d,%d,%q) = %q, want %q", tc.tenant, tc.fileID, tc.version, tc.ext, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestPGDriverDetection(t *testing.T) {
|
||||
cases := []struct {
|
||||
dsn, explicit, want string
|
||||
}{
|
||||
{"postgres://u:p@h:5432/onlyoffice", "", ProviderPG},
|
||||
{"postgresql://u:p@h/db", "", ProviderPG},
|
||||
{"host=h user=u password=p dbname=onlyoffice sslmode=disable", "", ProviderPG},
|
||||
{"root:secret@tcp(127.0.0.1:3306)/onlyoffice?parseTime=true", "", ProviderMySQL},
|
||||
{"mysql://root:secret@127.0.0.1:3306/onlyoffice", "", ProviderMySQL},
|
||||
{"root:secret@tcp(h:3306)/db", "postgres", ProviderPG},
|
||||
{"postgres://u:p@h/db", "mysql", ProviderMySQL},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
if got := pgDriver(tc.dsn, tc.explicit); got != tc.want {
|
||||
t.Errorf("pgDriver(%q, %q) = %q, want %q", tc.dsn, tc.explicit, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeSQLDSNMySQL(t *testing.T) {
|
||||
got, err := normalizeSQLDSN(ProviderMySQL, "mysql://root:secret@127.0.0.1:3306/onlyoffice")
|
||||
if err != nil {
|
||||
t.Fatalf("normalizeSQLDSN: %v", err)
|
||||
}
|
||||
want := "root:secret@tcp(127.0.0.1:3306)/onlyoffice?parseTime=true"
|
||||
if got != want {
|
||||
t.Errorf("normalize = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
// A driver DSN keeps parseTime and gains it when missing.
|
||||
got, err = normalizeSQLDSN(ProviderMySQL, "root:secret@tcp(127.0.0.1:3306)/onlyoffice")
|
||||
if err != nil {
|
||||
t.Fatalf("normalizeSQLDSN: %v", err)
|
||||
}
|
||||
if got != want {
|
||||
t.Errorf("normalize = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileRowToEntry(t *testing.T) {
|
||||
created := time.Date(2026, 9, 12, 18, 0, 37, 0, time.UTC)
|
||||
modified := time.Date(2026, 9, 13, 13, 50, 36, 0, time.UTC)
|
||||
e := fileRowToEntry(pgFileRow{
|
||||
id: 22484, folderID: 649, title: "Rechnung.pdf",
|
||||
size: 123433, version: 2, created: created, modified: modified,
|
||||
}, ProviderMySQL)
|
||||
if e.ID != "22484" || e.ParentID != "649" {
|
||||
t.Errorf("ids = %q/%q", e.ID, e.ParentID)
|
||||
}
|
||||
if e.Title != "Rechnung.pdf" || e.Kind != File {
|
||||
t.Errorf("title/kind = %q/%v", e.Title, e.Kind)
|
||||
}
|
||||
if e.Size != 123433 || e.Version != 2 {
|
||||
t.Errorf("size/version = %d/%d", e.Size, e.Version)
|
||||
}
|
||||
if e.MIME != "application/pdf" {
|
||||
t.Errorf("mime = %q", e.MIME)
|
||||
}
|
||||
if !e.Created.Equal(created) || !e.Modified.Equal(modified) {
|
||||
t.Errorf("times = %v/%v", e.Created, e.Modified)
|
||||
}
|
||||
if e.Provider != ProviderMySQL {
|
||||
t.Errorf("provider = %q", e.Provider)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFolderRowToEntry(t *testing.T) {
|
||||
modified := time.Date(2026, 8, 1, 10, 30, 0, 0, time.UTC)
|
||||
e := folderRowToEntry(pgFolderRow{id: 649, parentID: 647, title: "2025", modified: modified}, ProviderMySQL)
|
||||
if e.ID != "649" || e.ParentID != "647" || e.Title != "2025" {
|
||||
t.Errorf("folder = %+v", e)
|
||||
}
|
||||
if e.Kind != Folder {
|
||||
t.Errorf("kind = %v, want folder", e.Kind)
|
||||
}
|
||||
if e.Size != 0 || e.MIME != "" {
|
||||
t.Errorf("folder size/mime = %d/%q", e.Size, e.MIME)
|
||||
}
|
||||
if !e.Modified.Equal(modified) {
|
||||
t.Errorf("modified = %v", e.Modified)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPGStoreWriteMethodsReadOnly(t *testing.T) {
|
||||
s := &pgStore{driver: ProviderPG}
|
||||
ctx := context.Background()
|
||||
if _, err := s.CreateFolder(ctx, "1", "x"); !errors.Is(err, ErrReadOnly) {
|
||||
t.Errorf("CreateFolder err = %v", err)
|
||||
}
|
||||
if _, err := s.Upload(ctx, "1", "x", nil); !errors.Is(err, ErrReadOnly) {
|
||||
t.Errorf("Upload err = %v", err)
|
||||
}
|
||||
if err := s.Move(ctx, nil, "1"); !errors.Is(err, ErrReadOnly) {
|
||||
t.Errorf("Move err = %v", err)
|
||||
}
|
||||
if err := s.Copy(ctx, nil, "1"); !errors.Is(err, ErrReadOnly) {
|
||||
t.Errorf("Copy err = %v", err)
|
||||
}
|
||||
if err := s.Rename(ctx, "1", "x"); !errors.Is(err, ErrReadOnly) {
|
||||
t.Errorf("Rename err = %v", err)
|
||||
}
|
||||
if err := s.Delete(ctx, nil); !errors.Is(err, ErrReadOnly) {
|
||||
t.Errorf("Delete err = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPGStoreName(t *testing.T) {
|
||||
if got := (&pgStore{driver: ProviderPG}).Name(); got != ProviderPG {
|
||||
t.Errorf("Name = %q, want %q", got, ProviderPG)
|
||||
}
|
||||
if got := (&pgStore{driver: ProviderMySQL}).Name(); got != ProviderMySQL {
|
||||
t.Errorf("Name = %q, want %q", got, ProviderMySQL)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPGStoreStatRejectsNonNumeric(t *testing.T) {
|
||||
s := &pgStore{driver: ProviderPG}
|
||||
if _, err := s.Stat(context.Background(), "not-a-number"); err == nil {
|
||||
t.Error("Stat accepted a non-numeric id")
|
||||
}
|
||||
}
|
||||
+230
@@ -0,0 +1,230 @@
|
||||
package onlyoffice
|
||||
|
||||
// restStore implements FileStore on top of the REST Documents methods in
|
||||
// files.go. It is a thin adapter: no endpoint logic lives here, and every call
|
||||
// is wrapped in DoRetry.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// restStore is a FileStore over the REST Documents API.
|
||||
type restStore struct{ c *Client }
|
||||
|
||||
// Name reports the backend name.
|
||||
func (s *restStore) Name() string { return ProviderREST }
|
||||
|
||||
// List returns the files and folders directly below parentID.
|
||||
func (s *restStore) List(ctx context.Context, parentID string) ([]Entry, error) {
|
||||
var out []Entry
|
||||
err := retryStoreOp(ctx, func() error {
|
||||
raw, err := s.c.ListFolder(ctx, parentID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
entries, err := entriesFromFolderMap(raw, ProviderREST)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out = entries
|
||||
return nil
|
||||
})
|
||||
return out, err
|
||||
}
|
||||
|
||||
// Stat returns file metadata. The REST adapter resolves files only; folders
|
||||
// are listed by their parent (use List).
|
||||
func (s *restStore) Stat(ctx context.Context, id string) (Entry, error) {
|
||||
var out Entry
|
||||
err := retryStoreOp(ctx, func() error {
|
||||
f, err := s.c.GetFile(ctx, id)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out = FileEntryToEntry(f, ProviderREST)
|
||||
return nil
|
||||
})
|
||||
return out, err
|
||||
}
|
||||
|
||||
// CreateFolder creates a subfolder under parentID.
|
||||
func (s *restStore) CreateFolder(ctx context.Context, parentID, title string) (Entry, error) {
|
||||
var out Entry
|
||||
err := retryStoreOp(ctx, func() error {
|
||||
m, err := s.c.CreateFolder(ctx, parentID, title)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
e, err := folderEntryFromMap(m, parentID, ProviderREST)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if e.ParentID == "" {
|
||||
e.ParentID = parentID
|
||||
}
|
||||
if e.Title == "" {
|
||||
e.Title = title
|
||||
}
|
||||
out = e
|
||||
return nil
|
||||
})
|
||||
return out, err
|
||||
}
|
||||
|
||||
// Upload streams r into parentID as title. UploadToFolder is path based, so
|
||||
// the reader is spooled to a temporary file first (ponytail: OnlyOffice
|
||||
// multipart upload buffers the whole body anyway).
|
||||
func (s *restStore) Upload(ctx context.Context, parentID, title string, r io.Reader) (Entry, error) {
|
||||
dir, err := os.MkdirTemp("", "oo-rest-upload-")
|
||||
if err != nil {
|
||||
return Entry{}, err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
local := filepath.Join(dir, SafeLocalFileName(title))
|
||||
f, err := os.Create(local)
|
||||
if err != nil {
|
||||
return Entry{}, err
|
||||
}
|
||||
if _, err := io.Copy(f, r); err != nil {
|
||||
f.Close()
|
||||
return Entry{}, err
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
return Entry{}, err
|
||||
}
|
||||
|
||||
var out Entry
|
||||
err = retryStoreOp(ctx, func() error {
|
||||
fe, err := s.c.UploadToFolder(ctx, parentID, local)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out = FileEntryToEntry(fe, ProviderREST)
|
||||
return nil
|
||||
})
|
||||
return out, err
|
||||
}
|
||||
|
||||
// Download streams the file bytes into w.
|
||||
func (s *restStore) Download(ctx context.Context, id string, w io.Writer) (int64, error) {
|
||||
var n int64
|
||||
err := retryStoreOp(ctx, func() error {
|
||||
var e error
|
||||
n, e = s.c.DownloadFile(ctx, id, w)
|
||||
return e
|
||||
})
|
||||
return n, err
|
||||
}
|
||||
|
||||
// Move moves file ids into parentID. The REST MoveFiles endpoint handles files
|
||||
// only; folder moves are not exposed by this adapter.
|
||||
func (s *restStore) Move(ctx context.Context, ids []string, parentID string) error {
|
||||
dest, err := strconv.Atoi(strings.TrimSpace(parentID))
|
||||
if err != nil {
|
||||
return fmt.Errorf("onlyoffice: rest store: move: non-numeric destination folder id %q", parentID)
|
||||
}
|
||||
fileIDs, err := numericIDs(ids)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return retryStoreOp(ctx, func() error {
|
||||
_, err := s.c.MoveFiles(ctx, dest, fileIDs)
|
||||
return err
|
||||
})
|
||||
}
|
||||
|
||||
// Copy copies file ids into parentID. files.go has no copy method, so the
|
||||
// shared REST fileops copy endpoint (CopyDavItems) is used.
|
||||
func (s *restStore) Copy(ctx context.Context, ids []string, parentID string) error {
|
||||
if len(ids) == 0 {
|
||||
return nil
|
||||
}
|
||||
return retryStoreOp(ctx, func() error {
|
||||
return s.c.CopyDavItems(ctx, nil, ids, parentID)
|
||||
})
|
||||
}
|
||||
|
||||
// Rename sets a new title (including extension) for a file.
|
||||
func (s *restStore) Rename(ctx context.Context, id, title string) error {
|
||||
return retryStoreOp(ctx, func() error {
|
||||
_, err := s.c.RenameFile(ctx, id, title)
|
||||
return err
|
||||
})
|
||||
}
|
||||
|
||||
// Delete permanently deletes file ids.
|
||||
func (s *restStore) Delete(ctx context.Context, ids []string) error {
|
||||
fileIDs, err := numericIDs(ids)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(fileIDs) == 0 {
|
||||
return nil
|
||||
}
|
||||
return retryStoreOp(ctx, func() error {
|
||||
return s.c.DeleteFiles(ctx, fileIDs)
|
||||
})
|
||||
}
|
||||
|
||||
// entriesFromFolderMap converts a ListFolder response map into canonical
|
||||
// entries, reusing the DavFile/DavFolder decoders for robust size handling.
|
||||
func entriesFromFolderMap(m map[string]any, provider string) ([]Entry, error) {
|
||||
if m == nil {
|
||||
return nil, nil
|
||||
}
|
||||
b, err := json.Marshal(m)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var listing DavListing
|
||||
if err := json.Unmarshal(b, &listing); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]Entry, 0, len(listing.Folders)+len(listing.Files))
|
||||
for _, f := range listing.Folders {
|
||||
out = append(out, DavFolderToEntry(f, provider))
|
||||
}
|
||||
for _, f := range listing.Files {
|
||||
out = append(out, DavFileToEntry(f, provider))
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// folderEntryFromMap converts a CreateFolder response map into a folder Entry.
|
||||
func folderEntryFromMap(m map[string]any, parentID, provider string) (Entry, error) {
|
||||
e := Entry{Kind: Folder, Provider: provider, ParentID: parentID}
|
||||
if m == nil {
|
||||
return e, nil
|
||||
}
|
||||
b, err := json.Marshal(m)
|
||||
if err != nil {
|
||||
return e, err
|
||||
}
|
||||
var f DavFolder
|
||||
if err := json.Unmarshal(b, &f); err != nil {
|
||||
return e, err
|
||||
}
|
||||
e = DavFolderToEntry(f, provider)
|
||||
return e, nil
|
||||
}
|
||||
|
||||
// numericIDs parses Documents numeric ids from strings.
|
||||
func numericIDs(ids []string) ([]int, error) {
|
||||
out := make([]int, 0, len(ids))
|
||||
for _, id := range ids {
|
||||
n, err := strconv.Atoi(strings.TrimSpace(id))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: rest store: non-numeric id %q", id)
|
||||
}
|
||||
out = append(out, n)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,209 @@
|
||||
//go:build integration
|
||||
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"strconv"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestIntegrationFileStores runs the same operation set (create folder, upload,
|
||||
// list, stat, download, move, copy, rename, delete) through the REST and DAV
|
||||
// FileStore adapters against a throwaway project Documents folder. Destructive
|
||||
// — only run against instances you own.
|
||||
//
|
||||
// The Documents fileops API is asynchronous: a move/copy/delete is accepted
|
||||
// immediately and becomes visible a moment later, so effects are polled.
|
||||
func TestIntegrationFileStores(t *testing.T) {
|
||||
c := liveClient(t)
|
||||
t.Cleanup(func() { cleanupTestProjects(t, c) })
|
||||
ctx := context.Background()
|
||||
|
||||
suffix := time.Now().UTC().Format("20060102-150405")
|
||||
project, err := c.CreateProject(NewProjectRequest{
|
||||
Title: testProjectPrefix + "store-" + suffix,
|
||||
Description: "go-onlyoffice file store integration",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("CreateProject: %v", err)
|
||||
}
|
||||
if project.ID == nil {
|
||||
t.Fatal("created project without id")
|
||||
}
|
||||
root, err := c.projectFolderID(ctx, strconv.Itoa(*project.ID))
|
||||
if err != nil {
|
||||
t.Fatalf("projectFolderID: %v", err)
|
||||
}
|
||||
|
||||
for _, backend := range []string{ProviderREST, ProviderDAV} {
|
||||
t.Run(backend, func(t *testing.T) {
|
||||
testFileStoreOps(t, ctx, c, c.FileStore(backend), root, suffix)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func testFileStoreOps(t *testing.T, ctx context.Context, c *Client, store FileStore, root, suffix string) {
|
||||
t.Helper()
|
||||
content := []byte("file store " + store.Name() + " " + suffix + "\n")
|
||||
|
||||
src, err := store.CreateFolder(ctx, root, "fs-src-"+suffix)
|
||||
if err != nil {
|
||||
t.Fatalf("CreateFolder src: %v", err)
|
||||
}
|
||||
if src.Kind != Folder || src.ID == "" {
|
||||
t.Fatalf("created src folder: %+v", src)
|
||||
}
|
||||
dst, err := store.CreateFolder(ctx, root, "fs-dst-"+suffix)
|
||||
if err != nil {
|
||||
t.Fatalf("CreateFolder dst: %v", err)
|
||||
}
|
||||
if dst.Kind != Folder || dst.ID == "" {
|
||||
t.Fatalf("created dst folder: %+v", dst)
|
||||
}
|
||||
t.Cleanup(func() {
|
||||
if err := c.DeleteDavItems(ctx, []string{src.ID, dst.ID}, nil); err != nil {
|
||||
t.Logf("cleanup folders: %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
up, err := store.Upload(ctx, src.ID, "doc-"+suffix+".txt", bytes.NewReader(content))
|
||||
if err != nil {
|
||||
t.Fatalf("Upload: %v", err)
|
||||
}
|
||||
if up.Kind != File || up.ID == "" {
|
||||
t.Fatalf("uploaded entry: %+v", up)
|
||||
}
|
||||
if !waitEntry(ctx, store, src.ID, up.ID, 15*time.Second) {
|
||||
t.Fatalf("uploaded %s not listed in src", up.ID)
|
||||
}
|
||||
|
||||
st, err := store.Stat(ctx, up.ID)
|
||||
if err != nil {
|
||||
t.Fatalf("Stat: %v", err)
|
||||
}
|
||||
if st.ID != up.ID || st.Kind != File {
|
||||
t.Fatalf("stat = %+v", st)
|
||||
}
|
||||
|
||||
var buf bytes.Buffer
|
||||
n, err := store.Download(ctx, up.ID, &buf)
|
||||
if err != nil {
|
||||
t.Fatalf("Download: %v", err)
|
||||
}
|
||||
if n != int64(len(content)) || !bytes.Equal(buf.Bytes(), content) {
|
||||
t.Fatalf("download mismatch: got %d bytes %q want %d", n, buf.String(), len(content))
|
||||
}
|
||||
|
||||
moveEventually(t, ctx, store, up.ID, dst.ID)
|
||||
if !waitEntry(ctx, store, dst.ID, up.ID, 20*time.Second) {
|
||||
t.Fatalf("moved file %s not in dst", up.ID)
|
||||
}
|
||||
|
||||
if err := store.Copy(ctx, []string{up.ID}, src.ID); err != nil {
|
||||
t.Fatalf("Copy: %v", err)
|
||||
}
|
||||
copied := waitOtherFile(ctx, store, src.ID, up.ID, 20*time.Second)
|
||||
if copied == nil {
|
||||
t.Fatalf("no copy found in src after Copy")
|
||||
}
|
||||
|
||||
newTitle := "renamed-" + suffix + ".txt"
|
||||
renameEventually(t, ctx, store, up.ID, newTitle)
|
||||
|
||||
if err := store.Delete(ctx, []string{up.ID, copied.ID}); err != nil {
|
||||
t.Fatalf("Delete: %v", err)
|
||||
}
|
||||
if !waitNoEntry(ctx, store, dst.ID, up.ID, 20*time.Second) {
|
||||
t.Fatalf("file %s still present in dst after delete", up.ID)
|
||||
}
|
||||
if !waitNoEntry(ctx, store, src.ID, copied.ID, 20*time.Second) {
|
||||
t.Fatalf("copy %s still present in src after delete", copied.ID)
|
||||
}
|
||||
}
|
||||
|
||||
// moveEventually issues Move and retries while the operation is not visible yet
|
||||
// (the fileops API accepts asynchronously and occasionally rejects a move that
|
||||
// raced the just-finished upload).
|
||||
func moveEventually(t *testing.T, ctx context.Context, store FileStore, id, dstID string) {
|
||||
t.Helper()
|
||||
var lastErr error
|
||||
for attempt := 0; attempt < 5; attempt++ {
|
||||
if lastErr = store.Move(ctx, []string{id}, dstID); lastErr == nil {
|
||||
if waitEntry(ctx, store, dstID, id, 6*time.Second) {
|
||||
return
|
||||
}
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
t.Fatalf("Move %s -> %s: %v", id, dstID, lastErr)
|
||||
}
|
||||
|
||||
func renameEventually(t *testing.T, ctx context.Context, store FileStore, id, title string) {
|
||||
t.Helper()
|
||||
var lastErr error
|
||||
for attempt := 0; attempt < 5; attempt++ {
|
||||
if lastErr = store.Rename(ctx, id, title); lastErr == nil {
|
||||
if e, err := store.Stat(ctx, id); err == nil && e.Title == title {
|
||||
return
|
||||
}
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
t.Fatalf("Rename %s -> %q: %v", id, title, lastErr)
|
||||
}
|
||||
|
||||
func waitEntry(ctx context.Context, store FileStore, parentID, id string, d time.Duration) bool {
|
||||
deadline := time.Now().Add(d)
|
||||
for time.Now().Before(deadline) {
|
||||
if list, err := store.List(ctx, parentID); err == nil && entryByID(list, id) != nil {
|
||||
return true
|
||||
}
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func waitNoEntry(ctx context.Context, store FileStore, parentID, id string, d time.Duration) bool {
|
||||
deadline := time.Now().Add(d)
|
||||
for time.Now().Before(deadline) {
|
||||
if list, err := store.List(ctx, parentID); err == nil && entryByID(list, id) == nil {
|
||||
return true
|
||||
}
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func waitOtherFile(ctx context.Context, store FileStore, parentID, id string, d time.Duration) *Entry {
|
||||
deadline := time.Now().Add(d)
|
||||
for time.Now().Before(deadline) {
|
||||
if list, err := store.List(ctx, parentID); err == nil {
|
||||
if e := firstFileOtherThan(list, id); e != nil {
|
||||
return e
|
||||
}
|
||||
}
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func entryByID(entries []Entry, id string) *Entry {
|
||||
for i := range entries {
|
||||
if entries[i].ID == id {
|
||||
return &entries[i]
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func firstFileOtherThan(entries []Entry, id string) *Entry {
|
||||
for i := range entries {
|
||||
if entries[i].Kind == File && entries[i].ID != id {
|
||||
return &entries[i]
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,337 @@
|
||||
package onlyoffice
|
||||
|
||||
// Text extraction pipeline for the own full-text index (epic #34, F6 #42).
|
||||
//
|
||||
// TextIndexer downloads stored documents, extracts text through docpipe
|
||||
// (pdftotext; OCR for scans) and writes the result to a TextIndex. For PDFs it
|
||||
// also indexes the text of embedded attachments (pdfdetach), so a scan filed
|
||||
// as an attachment is searchable too. It is the write side of ESTextIndex and
|
||||
// never touches the OnlyOffice server's own ES index.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"github.com/eslider/go-onlyoffice/internal/docpipe"
|
||||
)
|
||||
|
||||
// defaultTextIndexExts are the formats extracted by default. The OnlyOffice
|
||||
// index already covers docx/xlsx/pptx; F6 adds PDF.
|
||||
var defaultTextIndexExts = []string{"pdf"}
|
||||
|
||||
const (
|
||||
defaultTextIndexWorkers = 3
|
||||
defaultTextIndexLang = "deu+eng"
|
||||
maxTextIndexErrors = 20
|
||||
)
|
||||
|
||||
// TextExtractor turns a local file into indexable plain text. The default uses
|
||||
// docpipe (pdftotext + OCR); tests inject a fake to stay offline.
|
||||
type TextExtractor interface {
|
||||
Extract(path, workDir, lang string, minChars int) (string, error)
|
||||
}
|
||||
|
||||
// docpipeExtractor is the production TextExtractor.
|
||||
type docpipeExtractor struct{ tools docpipe.Tools }
|
||||
|
||||
// Extract renders the file as Markdown, OCRing PDFs/images with a weak text
|
||||
// layer first and appending the text of embedded PDF attachments
|
||||
// (docpipe.ToMarkdownWithAttachments).
|
||||
func (d docpipeExtractor) Extract(path, workDir, lang string, minChars int) (string, error) {
|
||||
text, err := d.tools.ToMarkdownWithAttachments(path, workDir, lang, minChars)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return strings.TrimSpace(text), nil
|
||||
}
|
||||
|
||||
// IndexOptions controls a TextIndexer run.
|
||||
type IndexOptions struct {
|
||||
Recursive bool // IndexFolder: descend into subfolders
|
||||
Extensions []string // empty = defaultTextIndexExts (pdf)
|
||||
Limit int // max files to index, 0 = all
|
||||
Lang string // OCR language(s), default deu+eng
|
||||
MinChars int // OCR threshold, default docpipe.DefaultMinTextChars
|
||||
Workers int // parallel downloads/extractions, default 3
|
||||
}
|
||||
|
||||
// IndexResult summarises a run.
|
||||
type IndexResult struct {
|
||||
Scanned int
|
||||
Indexed int
|
||||
Skipped int
|
||||
Failed int
|
||||
Errors []string
|
||||
}
|
||||
|
||||
// TextIndexer wires a FileStore, a TextIndex and an extractor together.
|
||||
type TextIndexer struct {
|
||||
Store FileStore
|
||||
Index TextIndex
|
||||
Extractor TextExtractor // nil = local docpipe tools
|
||||
WorkDir string // temp dir for downloads/extraction
|
||||
}
|
||||
|
||||
// NewTextIndexer returns a TextIndexer over the given store and index.
|
||||
func NewTextIndexer(store FileStore, index TextIndex) *TextIndexer {
|
||||
return &TextIndexer{Store: store, Index: index}
|
||||
}
|
||||
|
||||
// textIndexEnsurer is implemented by indexes that can be created up front.
|
||||
type textIndexEnsurer interface {
|
||||
Ensure(ctx context.Context) error
|
||||
}
|
||||
|
||||
// Ensure creates the backing index when the TextIndex supports it.
|
||||
func (ix *TextIndexer) Ensure(ctx context.Context) error {
|
||||
if e, ok := ix.Index.(textIndexEnsurer); ok {
|
||||
return e.Ensure(ctx)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// IndexFiles stats the given file ids and indexes them.
|
||||
func (ix *TextIndexer) IndexFiles(ctx context.Context, ids []string, opts IndexOptions) (IndexResult, error) {
|
||||
entries := make([]Entry, 0, len(ids))
|
||||
for _, id := range ids {
|
||||
e, err := ix.Store.Stat(ctx, id)
|
||||
if err != nil {
|
||||
return IndexResult{}, fmt.Errorf("onlyoffice: stat %s: %w", id, err)
|
||||
}
|
||||
entries = append(entries, e)
|
||||
}
|
||||
return ix.IndexEntries(ctx, entries, opts)
|
||||
}
|
||||
|
||||
// IndexFolder lists a folder and indexes every matching file.
|
||||
func (ix *TextIndexer) IndexFolder(ctx context.Context, folderID string, opts IndexOptions) (IndexResult, error) {
|
||||
entries, err := ix.collect(ctx, folderID, opts.Recursive)
|
||||
if err != nil {
|
||||
return IndexResult{}, err
|
||||
}
|
||||
return ix.IndexEntries(ctx, entries, opts)
|
||||
}
|
||||
|
||||
// PlanFolder lists the files IndexFolder would process, without downloading or
|
||||
// extracting anything.
|
||||
func (ix *TextIndexer) PlanFolder(ctx context.Context, folderID string, opts IndexOptions) ([]Entry, error) {
|
||||
entries, err := ix.collect(ctx, folderID, opts.Recursive)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return selectEntries(entries, opts), nil
|
||||
}
|
||||
|
||||
// PlanFiles stats the ids and returns those that would be indexed.
|
||||
func (ix *TextIndexer) PlanFiles(ctx context.Context, ids []string, opts IndexOptions) ([]Entry, error) {
|
||||
entries := make([]Entry, 0, len(ids))
|
||||
for _, id := range ids {
|
||||
e, err := ix.Store.Stat(ctx, id)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: stat %s: %w", id, err)
|
||||
}
|
||||
entries = append(entries, e)
|
||||
}
|
||||
return selectEntries(entries, opts), nil
|
||||
}
|
||||
|
||||
// IndexEntries extracts and indexes the given files (folders are ignored).
|
||||
func (ix *TextIndexer) IndexEntries(ctx context.Context, entries []Entry, opts IndexOptions) (IndexResult, error) {
|
||||
opts = opts.withDefaults()
|
||||
var res IndexResult
|
||||
work := selectEntries(entries, opts)
|
||||
res.Scanned = len(entries)
|
||||
res.Skipped = len(entries) - len(work)
|
||||
if len(work) == 0 {
|
||||
return res, nil
|
||||
}
|
||||
|
||||
workers := opts.Workers
|
||||
if workers > len(work) {
|
||||
workers = len(work)
|
||||
}
|
||||
if workers < 1 {
|
||||
workers = 1
|
||||
}
|
||||
|
||||
type outcome struct {
|
||||
doc TextDoc
|
||||
err error
|
||||
}
|
||||
jobs := make(chan Entry)
|
||||
results := make(chan outcome, workers)
|
||||
var wg sync.WaitGroup
|
||||
for i := 0; i < workers; i++ {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for e := range jobs {
|
||||
if err := ctx.Err(); err != nil {
|
||||
results <- outcome{err: err}
|
||||
continue
|
||||
}
|
||||
doc, err := ix.indexOne(ctx, e, opts)
|
||||
results <- outcome{doc: doc, err: err}
|
||||
}
|
||||
}()
|
||||
}
|
||||
go func() {
|
||||
defer close(jobs)
|
||||
for _, e := range work {
|
||||
select {
|
||||
case jobs <- e:
|
||||
case <-ctx.Done():
|
||||
return
|
||||
}
|
||||
}
|
||||
}()
|
||||
go func() {
|
||||
wg.Wait()
|
||||
close(results)
|
||||
}()
|
||||
|
||||
var docs []TextDoc
|
||||
for r := range results {
|
||||
if r.err != nil {
|
||||
res.Failed++
|
||||
if len(res.Errors) < maxTextIndexErrors {
|
||||
res.Errors = append(res.Errors, r.err.Error())
|
||||
}
|
||||
continue
|
||||
}
|
||||
docs = append(docs, r.doc)
|
||||
}
|
||||
if err := ctx.Err(); err != nil {
|
||||
return res, err
|
||||
}
|
||||
if len(docs) > 0 {
|
||||
if err := ix.Index.Put(ctx, docs); err != nil {
|
||||
return res, fmt.Errorf("onlyoffice: index %d docs: %w", len(docs), err)
|
||||
}
|
||||
res.Indexed = len(docs)
|
||||
}
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// indexOne downloads and extracts a single file.
|
||||
func (ix *TextIndexer) indexOne(ctx context.Context, e Entry, opts IndexOptions) (TextDoc, error) {
|
||||
ext := fileExt(e.Title)
|
||||
dir := ix.WorkDir
|
||||
if dir == "" {
|
||||
dir = os.TempDir()
|
||||
}
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
return TextDoc{}, err
|
||||
}
|
||||
tmp, err := os.CreateTemp(dir, "ooidx-*."+ext)
|
||||
if err != nil {
|
||||
return TextDoc{}, err
|
||||
}
|
||||
tmpPath := tmp.Name()
|
||||
defer os.Remove(tmpPath)
|
||||
|
||||
if _, err := ix.Store.Download(ctx, e.ID, tmp); err != nil {
|
||||
tmp.Close()
|
||||
return TextDoc{}, fmt.Errorf("download %s (%s): %w", e.ID, e.Title, err)
|
||||
}
|
||||
if err := tmp.Close(); err != nil {
|
||||
return TextDoc{}, err
|
||||
}
|
||||
text, err := ix.extractor().Extract(tmpPath, dir, opts.Lang, opts.MinChars)
|
||||
if err != nil {
|
||||
return TextDoc{}, fmt.Errorf("extract %s: %w", e.Title, err)
|
||||
}
|
||||
return TextDoc{ID: e.ID, Title: e.Title, FolderID: e.ParentID, Ext: ext, Content: text}, nil
|
||||
}
|
||||
|
||||
// collect lists files under folderID, breadth-first when recursive.
|
||||
func (ix *TextIndexer) collect(ctx context.Context, folderID string, recursive bool) ([]Entry, error) {
|
||||
var files []Entry
|
||||
queue := []string{folderID}
|
||||
for len(queue) > 0 {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
id := queue[0]
|
||||
queue = queue[1:]
|
||||
entries, err := ix.Store.List(ctx, id)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: list folder %s: %w", id, err)
|
||||
}
|
||||
for _, e := range entries {
|
||||
if e.Kind == Folder {
|
||||
if recursive {
|
||||
queue = append(queue, e.ID)
|
||||
}
|
||||
continue
|
||||
}
|
||||
if e.ParentID == "" {
|
||||
e.ParentID = id
|
||||
}
|
||||
files = append(files, e)
|
||||
}
|
||||
}
|
||||
return files, nil
|
||||
}
|
||||
|
||||
// selectEntries filters files by extension and applies the limit.
|
||||
func selectEntries(entries []Entry, opts IndexOptions) []Entry {
|
||||
allowed := extensionSet(opts.Extensions)
|
||||
work := make([]Entry, 0, len(entries))
|
||||
for _, e := range entries {
|
||||
if e.Kind != File {
|
||||
continue
|
||||
}
|
||||
if opts.Limit > 0 && len(work) >= opts.Limit {
|
||||
break
|
||||
}
|
||||
if !allowed[fileExt(e.Title)] {
|
||||
continue
|
||||
}
|
||||
work = append(work, e)
|
||||
}
|
||||
return work
|
||||
}
|
||||
|
||||
// extensionSet normalises the extension allow-list (default: pdf).
|
||||
func extensionSet(exts []string) map[string]bool {
|
||||
if len(exts) == 0 {
|
||||
exts = defaultTextIndexExts
|
||||
}
|
||||
set := make(map[string]bool, len(exts))
|
||||
for _, e := range normalizeExtensions(exts) {
|
||||
set[e] = true
|
||||
}
|
||||
return set
|
||||
}
|
||||
|
||||
// fileExt returns the lower-case extension without the dot.
|
||||
func fileExt(title string) string {
|
||||
return strings.ToLower(strings.TrimPrefix(filepath.Ext(strings.TrimSpace(title)), "."))
|
||||
}
|
||||
|
||||
// withDefaults fills zero-valued options.
|
||||
func (o IndexOptions) withDefaults() IndexOptions {
|
||||
if o.Workers <= 0 {
|
||||
o.Workers = defaultTextIndexWorkers
|
||||
}
|
||||
if o.MinChars <= 0 {
|
||||
o.MinChars = docpipe.DefaultMinTextChars
|
||||
}
|
||||
if strings.TrimSpace(o.Lang) == "" {
|
||||
o.Lang = defaultTextIndexLang
|
||||
}
|
||||
return o
|
||||
}
|
||||
|
||||
// extractor returns the configured extractor or the local docpipe default.
|
||||
func (ix *TextIndexer) extractor() TextExtractor {
|
||||
if ix.Extractor != nil {
|
||||
return ix.Extractor
|
||||
}
|
||||
return docpipeExtractor{tools: docpipe.LookPath()}
|
||||
}
|
||||
@@ -237,7 +237,18 @@ func (c *Client) UploadProjectFile(ctx context.Context, projectID, localPath str
|
||||
return decodeResponseFileEntry(raw)
|
||||
}
|
||||
|
||||
// UploadProjectFileReplacing upserts by stem|ext in the project Documents folder.
|
||||
func (c *Client) UploadProjectFileReplacing(ctx context.Context, projectID, localPath string) (*FileEntry, []int, error) {
|
||||
folderID, err := c.projectFolderID(ctx, projectID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
return c.UploadToFolderReplacing(ctx, folderID, localPath)
|
||||
}
|
||||
|
||||
// GetFile returns file metadata including viewUrl for download.
|
||||
//
|
||||
// Deprecated: use FileStore.Stat via Client.Files()/Client.FileStore.
|
||||
func (c *Client) GetFile(ctx context.Context, fileID string) (*FileEntry, error) {
|
||||
if fileID == "" {
|
||||
return nil, fmt.Errorf("file id is required")
|
||||
@@ -251,6 +262,8 @@ func (c *Client) GetFile(ctx context.Context, fileID string) (*FileEntry, error)
|
||||
}
|
||||
|
||||
// RenameFile sets a new title (including extension) for the file.
|
||||
//
|
||||
// Deprecated: use FileStore.Rename via Client.Files()/Client.FileStore.
|
||||
func (c *Client) RenameFile(ctx context.Context, fileID, newTitle string) (*FileEntry, error) {
|
||||
if fileID == "" || newTitle == "" {
|
||||
return nil, fmt.Errorf("file id and new title are required")
|
||||
@@ -263,55 +276,150 @@ func (c *Client) RenameFile(ctx context.Context, fileID, newTitle string) (*File
|
||||
return decodeResponseFileEntry(raw)
|
||||
}
|
||||
|
||||
type deleteFilesBody struct {
|
||||
FileIDs []int `json:"fileIds"`
|
||||
FolderIDs []int `json:"folderIds"`
|
||||
}
|
||||
|
||||
// DeleteFiles permanently deletes files by numeric id (Documents module).
|
||||
// Uses per-file DELETE (DeleteDavItems); fileops/delete returns 200 on some
|
||||
// portals (e.g. produktor.io) without removing the file.
|
||||
//
|
||||
// Deprecated: use FileStore.Delete via Client.Files()/Client.FileStore.
|
||||
func (c *Client) DeleteFiles(ctx context.Context, fileIDs []int) error {
|
||||
if len(fileIDs) == 0 {
|
||||
return fmt.Errorf("no file ids to delete")
|
||||
}
|
||||
body := deleteFilesBody{FileIDs: fileIDs, FolderIDs: nil}
|
||||
_, err := c.putJSON(ctx, "/api/2.0/files/fileops/delete.json", body)
|
||||
if err != nil {
|
||||
_, err = c.putJSON(ctx, "/api/2.0/files/fileops/delete", body)
|
||||
strIDs := make([]string, len(fileIDs))
|
||||
for i, id := range fileIDs {
|
||||
strIDs[i] = strconv.Itoa(id)
|
||||
}
|
||||
return err
|
||||
return c.DeleteDavItems(ctx, nil, strIDs)
|
||||
}
|
||||
|
||||
// ListFolder returns the Documents module listing for a folder id
|
||||
// (GET /api/2.0/files/{folderId}).
|
||||
//
|
||||
// Deprecated: use FileStore.List via Client.Files()/Client.FileStore.
|
||||
func (c *Client) ListFolder(ctx context.Context, folderID string) (map[string]any, error) {
|
||||
if folderID == "" {
|
||||
return nil, fmt.Errorf("folder id is required")
|
||||
}
|
||||
out, err := c.ResponseObject(ctx, "/api/2.0/files/"+url.PathEscape(folderID)+".json")
|
||||
if err != nil {
|
||||
out, err = c.ResponseObject(ctx, "/api/2.0/files/"+url.PathEscape(folderID))
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
|
||||
// CreateFolder creates a subfolder under parentFolderID.
|
||||
func (c *Client) CreateFolder(ctx context.Context, parentFolderID, title string) (map[string]any, error) {
|
||||
if parentFolderID == "" || title == "" {
|
||||
return nil, fmt.Errorf("parent folder id and title are required")
|
||||
}
|
||||
body := map[string]any{"title": title}
|
||||
out, err := c.postJSONObject(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentFolderID)+".json", body)
|
||||
if err != nil {
|
||||
out, err = c.postJSONObject(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentFolderID), body)
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
|
||||
// MoveFiles moves file ids into destFolderID (Documents fileops/move).
|
||||
//
|
||||
// Deprecated: use FileStore.Move via Client.Files()/Client.FileStore.
|
||||
func (c *Client) MoveFiles(ctx context.Context, destFolderID int, fileIDs []int) (map[string]any, error) {
|
||||
if destFolderID == 0 || len(fileIDs) == 0 {
|
||||
return nil, fmt.Errorf("dest folder and file ids are required")
|
||||
}
|
||||
body := map[string]any{
|
||||
"folderIds": []int{},
|
||||
"fileIds": fileIDs,
|
||||
"destFolderId": destFolderID,
|
||||
"resolveType": "Skip",
|
||||
"holdResult": true,
|
||||
}
|
||||
// fileops/move answers an operations envelope (like MoveDavItems), not a
|
||||
// single object, so parse the raw body before unwrapping and surface any
|
||||
// per-operation error. Unwrapping first (putJSONObject) made fileopsError
|
||||
// look for a "response" key that was already stripped.
|
||||
raw, err := c.putJSON(ctx, "/api/2.0/files/fileops/move", body)
|
||||
if err != nil {
|
||||
raw, err = c.putJSON(ctx, "/api/2.0/files/fileops/move.json", body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
if ferr := fileopsError(raw); ferr != nil {
|
||||
return nil, ferr
|
||||
}
|
||||
out, _ := unmarshalResponseObject(raw)
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// UploadToFolder uploads a local file into an arbitrary Documents folder id.
|
||||
//
|
||||
// Deprecated: use FileStore.Upload via Client.Files()/Client.FileStore.
|
||||
func (c *Client) UploadToFolder(ctx context.Context, folderID, localPath string) (*FileEntry, error) {
|
||||
if folderID == "" || localPath == "" {
|
||||
return nil, fmt.Errorf("folder id and local path are required")
|
||||
}
|
||||
uploadPath := fmt.Sprintf("/api/2.0/files/%s/upload.json", url.PathEscape(folderID))
|
||||
raw, err := c.uploadMultipart(ctx, uploadPath, "file", localPath)
|
||||
if err != nil {
|
||||
uploadPath = fmt.Sprintf("/api/2.0/files/%s/upload", url.PathEscape(folderID))
|
||||
raw, err = c.uploadMultipart(ctx, uploadPath, "file", localPath)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
return decodeResponseFileEntry(raw)
|
||||
}
|
||||
|
||||
// UpdateFile uploads a new version of an existing file (same id, name and
|
||||
// folder). It does not delete and does not create a second file.
|
||||
//
|
||||
// The Documents API method is PUT /api/2.0/files/{id}/update; POST is kept as
|
||||
// a fallback for older servers. The path is tried with and without .json.
|
||||
func (c *Client) UpdateFile(ctx context.Context, fileID, localPath string) (*FileEntry, error) {
|
||||
if fileID == "" || localPath == "" {
|
||||
return nil, fmt.Errorf("file id and local path are required")
|
||||
}
|
||||
base := fmt.Sprintf("/api/2.0/files/%s/update", url.PathEscape(fileID))
|
||||
attempts := []struct {
|
||||
method, path string
|
||||
}{
|
||||
{http.MethodPut, base},
|
||||
{http.MethodPut, base + ".json"},
|
||||
{http.MethodPost, base},
|
||||
{http.MethodPost, base + ".json"},
|
||||
}
|
||||
var lastErr error
|
||||
for _, a := range attempts {
|
||||
raw, err := c.uploadMultipartMethod(ctx, a.method, a.path, "file", localPath)
|
||||
if err == nil {
|
||||
return decodeResponseFileEntry(raw)
|
||||
}
|
||||
lastErr = err
|
||||
}
|
||||
return nil, lastErr
|
||||
}
|
||||
|
||||
// FileFolderID returns the parent folder id string for a file entry, if known.
|
||||
func FileFolderID(f *FileEntry) string {
|
||||
if f == nil || f.FolderID == nil {
|
||||
return ""
|
||||
}
|
||||
return f.FolderID.String()
|
||||
}
|
||||
|
||||
// DownloadFile streams file bytes from the file's viewUrl using the same auth
|
||||
// as API calls. Writes into dst.
|
||||
// as API calls. Writes into dst. When the portal serves the file from its stale
|
||||
// AWS S3 consumer, the bytes are fetched from the local MinIO store instead
|
||||
// (see storage_fallback.go).
|
||||
//
|
||||
// Deprecated: use FileStore.Download via Client.Files()/Client.FileStore.
|
||||
func (c *Client) DownloadFile(ctx context.Context, fileID string, dst io.Writer) (int64, error) {
|
||||
f, err := c.GetFile(ctx, fileID)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if f.ViewURL == nil || *f.ViewURL == "" {
|
||||
return 0, fmt.Errorf("file %s has no viewUrl", fileID)
|
||||
}
|
||||
downloadURL := c.resolveAPIURL(*f.ViewURL)
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, downloadURL, nil)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
req.Header.Set("Authorization", auth)
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode >= 400 {
|
||||
b, _ := io.ReadAll(io.LimitReader(resp.Body, 512))
|
||||
return 0, fmt.Errorf("GET viewUrl: %d %s", resp.StatusCode, truncate(string(b), 400))
|
||||
}
|
||||
n, err := io.Copy(dst, resp.Body)
|
||||
return n, err
|
||||
return c.downloadFileEntry(ctx, f, dst)
|
||||
}
|
||||
|
||||
func (c *Client) resolveAPIURL(ref string) string {
|
||||
@@ -319,8 +427,18 @@ func (c *Client) resolveAPIURL(ref string) string {
|
||||
if ref == "" {
|
||||
return ref
|
||||
}
|
||||
if strings.HasPrefix(ref, "http://") || strings.HasPrefix(ref, "https://") {
|
||||
return ref
|
||||
// Rewrite any host to the configured API base so downloads stay on the
|
||||
// internal network and keep the Authorization header (no cross-host
|
||||
// redirect that would strip it). Scheme-relative URLs are handled too.
|
||||
if strings.HasPrefix(ref, "//") {
|
||||
ref = "http:" + ref
|
||||
}
|
||||
if u, err := url.Parse(ref); err == nil && u.IsAbs() {
|
||||
if base, err2 := url.Parse(c.baseURL()); err2 == nil {
|
||||
u.Scheme = base.Scheme
|
||||
u.Host = base.Host
|
||||
return u.String()
|
||||
}
|
||||
}
|
||||
base := c.baseURL()
|
||||
if strings.HasPrefix(ref, "/") {
|
||||
|
||||
+389
@@ -0,0 +1,389 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// FileEntryExt returns a normalized extension (lowercase, with leading dot).
|
||||
func FileEntryExt(f *FileEntry) string {
|
||||
if f == nil {
|
||||
return ""
|
||||
}
|
||||
exst := ""
|
||||
if f.FileExst != nil {
|
||||
exst = strings.TrimSpace(*f.FileExst)
|
||||
}
|
||||
if exst != "" {
|
||||
if !strings.HasPrefix(exst, ".") {
|
||||
exst = "." + exst
|
||||
}
|
||||
return strings.ToLower(exst)
|
||||
}
|
||||
if f.Title != nil {
|
||||
if ext := filepath.Ext(*f.Title); ext != "" {
|
||||
return strings.ToLower(ext)
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// FileDedupKey is stem|ext — two files with the same key are duplicates.
|
||||
func FileDedupKey(f *FileEntry) string {
|
||||
st := FileEntryStem(f)
|
||||
ext := FileEntryExt(f)
|
||||
if st == "" {
|
||||
return ""
|
||||
}
|
||||
if ext == "" {
|
||||
return st
|
||||
}
|
||||
return st + "|" + strings.TrimPrefix(ext, ".")
|
||||
}
|
||||
|
||||
// FindFilesByDedupKey returns folder files matching stem and extension.
|
||||
func FindFilesByDedupKey(files []*FileEntry, stem, ext string) []*FileEntry {
|
||||
key := dedupKeyFromParts(stem, ext)
|
||||
if key == "" {
|
||||
return nil
|
||||
}
|
||||
var out []*FileEntry
|
||||
for _, f := range files {
|
||||
if FileDedupKey(f) == key {
|
||||
out = append(out, f)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func dedupKeyFromParts(stem, ext string) string {
|
||||
stem = strings.TrimSpace(stem)
|
||||
if stem == "" {
|
||||
return ""
|
||||
}
|
||||
ext = strings.ToLower(strings.TrimSpace(ext))
|
||||
if ext != "" && !strings.HasPrefix(ext, ".") {
|
||||
ext = "." + ext
|
||||
}
|
||||
if ext == "" {
|
||||
return stem
|
||||
}
|
||||
return stem + "|" + strings.TrimPrefix(ext, ".")
|
||||
}
|
||||
|
||||
// UploadExtFromLocal returns the lowercase extension from a local path.
|
||||
func UploadExtFromLocal(localPath string) string {
|
||||
ext := filepath.Ext(localPath)
|
||||
if ext == "" {
|
||||
return ""
|
||||
}
|
||||
return strings.ToLower(ext)
|
||||
}
|
||||
|
||||
// IsTrashFolderTitle reports staging/trash folders (e.g. _trash-md).
|
||||
func IsTrashFolderTitle(title string) bool {
|
||||
t := strings.ToLower(strings.TrimSpace(title))
|
||||
return strings.HasPrefix(t, "_") || strings.Contains(t, "trash")
|
||||
}
|
||||
|
||||
// ProjectFolderFile ties a file to its project Documents subfolder.
|
||||
type ProjectFolderFile struct {
|
||||
FolderID string
|
||||
FolderTitle string
|
||||
File *FileEntry
|
||||
}
|
||||
|
||||
// DedupGroup is one duplicate set: keep the newest (or non-trash) file.
|
||||
type DedupGroup struct {
|
||||
Key string
|
||||
FolderID string
|
||||
FolderTitle string
|
||||
Keep *FileEntry
|
||||
Remove []*FileEntry
|
||||
}
|
||||
|
||||
// DedupOptions controls project-wide duplicate scans.
|
||||
type DedupOptions struct {
|
||||
CrossFolder bool
|
||||
}
|
||||
|
||||
// FindProjectDuplicates scans project folders for duplicate files.
|
||||
func FindProjectDuplicates(folders []*FolderEntry, filesByFolder map[string][]*FileEntry, opts DedupOptions) []DedupGroup {
|
||||
var indexed []ProjectFolderFile
|
||||
for _, folder := range folders {
|
||||
if folder == nil || folder.ID == nil {
|
||||
continue
|
||||
}
|
||||
fid := folder.ID.String()
|
||||
title := ""
|
||||
if folder.Title != nil {
|
||||
title = *folder.Title
|
||||
}
|
||||
for _, f := range filesByFolder[fid] {
|
||||
if f == nil {
|
||||
continue
|
||||
}
|
||||
indexed = append(indexed, ProjectFolderFile{
|
||||
FolderID: fid, FolderTitle: title, File: f,
|
||||
})
|
||||
}
|
||||
}
|
||||
if opts.CrossFolder {
|
||||
return findCrossFolderDuplicates(indexed)
|
||||
}
|
||||
return findWithinFolderDuplicates(indexed)
|
||||
}
|
||||
|
||||
func findWithinFolderDuplicates(indexed []ProjectFolderFile) []DedupGroup {
|
||||
byFolder := map[string][]ProjectFolderFile{}
|
||||
for _, it := range indexed {
|
||||
byFolder[it.FolderID] = append(byFolder[it.FolderID], it)
|
||||
}
|
||||
var out []DedupGroup
|
||||
for fid, items := range byFolder {
|
||||
title := ""
|
||||
if len(items) > 0 {
|
||||
title = items[0].FolderTitle
|
||||
}
|
||||
byKey := map[string][]*FileEntry{}
|
||||
for _, it := range items {
|
||||
k := FileDedupKey(it.File)
|
||||
byKey[k] = append(byKey[k], it.File)
|
||||
}
|
||||
for k, group := range byKey {
|
||||
if k == "" {
|
||||
continue // dotfiles etc. have no stem: never treat as duplicates
|
||||
}
|
||||
if len(group) < 2 {
|
||||
continue
|
||||
}
|
||||
keep, remove := pickDuplicateKeeper(group, false)
|
||||
if keep == nil || len(remove) == 0 {
|
||||
continue
|
||||
}
|
||||
out = append(out, DedupGroup{
|
||||
Key: k, FolderID: fid, FolderTitle: title, Keep: keep, Remove: remove,
|
||||
})
|
||||
}
|
||||
}
|
||||
sortDedupGroups(out)
|
||||
return out
|
||||
}
|
||||
|
||||
func findCrossFolderDuplicates(indexed []ProjectFolderFile) []DedupGroup {
|
||||
byKey := map[string][]ProjectFolderFile{}
|
||||
for _, it := range indexed {
|
||||
k := FileDedupKey(it.File)
|
||||
byKey[k] = append(byKey[k], it)
|
||||
}
|
||||
var out []DedupGroup
|
||||
for k, items := range byKey {
|
||||
if k == "" {
|
||||
continue // dotfiles etc. have no stem: never treat as duplicates
|
||||
}
|
||||
if len(items) < 2 {
|
||||
continue
|
||||
}
|
||||
files := make([]*FileEntry, len(items))
|
||||
folders := make([]string, len(items))
|
||||
folderTitles := make([]string, len(items))
|
||||
for i, it := range items {
|
||||
files[i] = it.File
|
||||
folders[i] = it.FolderID
|
||||
folderTitles[i] = it.FolderTitle
|
||||
}
|
||||
keep, remove := pickDuplicateKeeperWithFolders(files, folders, folderTitles)
|
||||
if keep == nil || len(remove) == 0 {
|
||||
continue
|
||||
}
|
||||
fid, ftitle := "", ""
|
||||
for _, it := range items {
|
||||
if it.File == keep {
|
||||
fid, ftitle = it.FolderID, it.FolderTitle
|
||||
break
|
||||
}
|
||||
}
|
||||
out = append(out, DedupGroup{
|
||||
Key: k, FolderID: fid, FolderTitle: ftitle, Keep: keep, Remove: remove,
|
||||
})
|
||||
}
|
||||
sortDedupGroups(out)
|
||||
return out
|
||||
}
|
||||
|
||||
func pickDuplicateKeeper(files []*FileEntry, _ bool) (*FileEntry, []*FileEntry) {
|
||||
return pickDuplicateKeeperWithFolders(files, nil, nil)
|
||||
}
|
||||
|
||||
func pickDuplicateKeeperWithFolders(files []*FileEntry, folderIDs, folderTitles []string) (*FileEntry, []*FileEntry) {
|
||||
if len(files) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
type ranked struct {
|
||||
file *FileEntry
|
||||
trash bool
|
||||
}
|
||||
rankedFiles := make([]ranked, len(files))
|
||||
for i, f := range files {
|
||||
trash := false
|
||||
if folderTitles != nil && i < len(folderTitles) {
|
||||
trash = IsTrashFolderTitle(folderTitles[i])
|
||||
}
|
||||
rankedFiles[i] = ranked{file: f, trash: trash}
|
||||
}
|
||||
sort.SliceStable(rankedFiles, func(i, j int) bool {
|
||||
ri, rj := rankedFiles[i], rankedFiles[j]
|
||||
if ri.trash != rj.trash {
|
||||
return !ri.trash // non-trash first
|
||||
}
|
||||
ti, tj := rankedFiles[i].file.Updated, rankedFiles[j].file.Updated
|
||||
if ti == nil {
|
||||
return false
|
||||
}
|
||||
if tj == nil {
|
||||
return true
|
||||
}
|
||||
return ti.After(*tj) // newest first
|
||||
})
|
||||
keep := rankedFiles[0].file
|
||||
var remove []*FileEntry
|
||||
for _, r := range rankedFiles[1:] {
|
||||
remove = append(remove, r.file)
|
||||
}
|
||||
return keep, remove
|
||||
}
|
||||
|
||||
func sortDedupGroups(groups []DedupGroup) {
|
||||
sort.Slice(groups, func(i, j int) bool {
|
||||
if groups[i].FolderTitle != groups[j].FolderTitle {
|
||||
return groups[i].FolderTitle < groups[j].FolderTitle
|
||||
}
|
||||
return groups[i].Key < groups[j].Key
|
||||
})
|
||||
}
|
||||
|
||||
// ApplyDedupGroups deletes Remove files from each group.
|
||||
func (c *Client) ApplyDedupGroups(ctx context.Context, groups []DedupGroup) ([]int, error) {
|
||||
seen := map[int]struct{}{}
|
||||
var ids []int
|
||||
for _, g := range groups {
|
||||
for _, f := range g.Remove {
|
||||
n := int(FileEntryNumericID(f))
|
||||
if n == 0 {
|
||||
continue
|
||||
}
|
||||
if _, ok := seen[n]; ok {
|
||||
continue
|
||||
}
|
||||
seen[n] = struct{}{}
|
||||
ids = append(ids, n)
|
||||
}
|
||||
}
|
||||
if len(ids) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
if err := c.DeleteFiles(ctx, ids); err != nil {
|
||||
return ids, err
|
||||
}
|
||||
return ids, nil
|
||||
}
|
||||
|
||||
// DeleteFilesByDedupKey removes all files in folderID matching stem+ext.
|
||||
func (c *Client) DeleteFilesByDedupKey(ctx context.Context, folderID, stem, ext string) ([]int, error) {
|
||||
files, err := c.FolderFiles(ctx, folderID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
matches := FindFilesByDedupKey(files, stem, ext)
|
||||
if len(matches) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
ids := make([]int, 0, len(matches))
|
||||
for _, f := range matches {
|
||||
n := int(FileEntryNumericID(f))
|
||||
if n != 0 {
|
||||
ids = append(ids, n)
|
||||
}
|
||||
}
|
||||
if len(ids) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
if err := c.DeleteFiles(ctx, ids); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ids, nil
|
||||
}
|
||||
|
||||
// mergeProjectRootForDedupe includes projectFolder files in dedupe scans. OO often lists
|
||||
// root documents only in pf.Files while pf.Folders is empty.
|
||||
func mergeProjectRootForDedupe(rootID string, folders []*FolderEntry, filesByFolder map[string][]*FileEntry, rootFiles []*FileEntry) ([]*FolderEntry, map[string][]*FileEntry) {
|
||||
if rootID == "" {
|
||||
return folders, filesByFolder
|
||||
}
|
||||
if filesByFolder == nil {
|
||||
filesByFolder = map[string][]*FileEntry{}
|
||||
}
|
||||
for _, folder := range folders {
|
||||
if folder != nil && folder.ID != nil && folder.ID.String() == rootID {
|
||||
if len(rootFiles) > 0 {
|
||||
filesByFolder[rootID] = rootFiles
|
||||
}
|
||||
return folders, filesByFolder
|
||||
}
|
||||
}
|
||||
if len(rootFiles) == 0 {
|
||||
return folders, filesByFolder
|
||||
}
|
||||
id := json.Number(rootID)
|
||||
title := "(project root)"
|
||||
folders = append(folders, &FolderEntry{ID: &id, Title: &title})
|
||||
filesByFolder[rootID] = rootFiles
|
||||
return folders, filesByFolder
|
||||
}
|
||||
|
||||
// DedupeProject scans project folders and optionally deletes duplicates.
|
||||
func (c *Client) DedupeProject(ctx context.Context, projectID string, opts DedupOptions, apply bool) ([]DedupGroup, []int, error) {
|
||||
pf, err := c.GetProjectFiles(ctx, projectID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
rootID, err := c.projectFolderID(ctx, projectID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
var rootFiles []*FileEntry
|
||||
if rootID != "" {
|
||||
rootFiles, err = c.FolderFiles(ctx, rootID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
}
|
||||
filesByFolder := make(map[string][]*FileEntry, len(pf.Folders)+1)
|
||||
folders := make([]*FolderEntry, 0, len(pf.Folders)+1)
|
||||
for _, folder := range pf.Folders {
|
||||
if folder == nil || folder.ID == nil {
|
||||
continue
|
||||
}
|
||||
fid := folder.ID.String()
|
||||
if fid == rootID {
|
||||
filesByFolder[fid] = rootFiles
|
||||
} else {
|
||||
files, err := c.FolderFiles(ctx, fid)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
filesByFolder[fid] = files
|
||||
}
|
||||
folders = append(folders, folder)
|
||||
}
|
||||
folders, filesByFolder = mergeProjectRootForDedupe(rootID, folders, filesByFolder, rootFiles)
|
||||
groups := FindProjectDuplicates(folders, filesByFolder, opts)
|
||||
if !apply || len(groups) == 0 {
|
||||
return groups, nil, nil
|
||||
}
|
||||
deleted, err := c.ApplyDedupGroups(ctx, groups)
|
||||
return groups, deleted, err
|
||||
}
|
||||
@@ -0,0 +1,416 @@
|
||||
//go:build integration
|
||||
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestIntegrationFileDedup proves on a live OnlyOffice portal that the file
|
||||
// dedup helpers find real duplicates and delete only the redundant copies.
|
||||
// It creates a throwaway "go-onlyoffice-test-" project (removed by cleanup),
|
||||
// places same stem|ext files in two subfolders and in the project root, then
|
||||
// exercises FindProjectDuplicates, mergeProjectRootForDedupe,
|
||||
// ApplyDedupGroups and DeleteFilesByDedupKey. Destructive — run only against
|
||||
// an instance you own.
|
||||
func TestIntegrationFileDedup(t *testing.T) {
|
||||
c := liveClient(t)
|
||||
t.Cleanup(func() { cleanupTestProjects(t, c) })
|
||||
ctx := context.Background()
|
||||
|
||||
suffix := time.Now().UTC().Format("20060102-150405")
|
||||
project, err := c.CreateProject(NewProjectRequest{
|
||||
Title: testProjectPrefix + "dedup-" + suffix,
|
||||
Description: "go-onlyoffice file dedup integration",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("CreateProject: %v", err)
|
||||
}
|
||||
if project.ID == nil {
|
||||
t.Fatal("created project without id")
|
||||
}
|
||||
pid := strconv.Itoa(*project.ID)
|
||||
|
||||
root := projectFolderEventually(t, ctx, c, pid)
|
||||
|
||||
aID := createFolderLive(t, ctx, c, root, "A-"+suffix)
|
||||
bID := createFolderLive(t, ctx, c, root, "B-"+suffix)
|
||||
createFolderLive(t, ctx, c, root, "_trash-"+suffix)
|
||||
|
||||
// Check IsTrashFolderTitle against a real live folder title.
|
||||
if !IsTrashFolderTitle("_trash-" + suffix) {
|
||||
t.Fatalf("IsTrashFolderTitle(%q) = false for a live _trash folder", "_trash-"+suffix)
|
||||
}
|
||||
|
||||
stem := "dedup-" + suffix
|
||||
local := writeLocalFile(t, stem+".txt", []byte("dedup integration "+suffix+"\n"))
|
||||
|
||||
a1 := uploadFolderLive(t, ctx, c, aID, local)
|
||||
a2 := uploadFolderLive(t, ctx, c, aID, local)
|
||||
b1 := uploadFolderLive(t, ctx, c, bID, local)
|
||||
r1 := uploadFolderLive(t, ctx, c, root, local)
|
||||
r2 := uploadFolderLive(t, ctx, c, root, local)
|
||||
t.Logf("uploaded a1=%d a2=%d b1=%d r1=%d r2=%d",
|
||||
FileEntryNumericID(a1), FileEntryNumericID(a2), FileEntryNumericID(b1),
|
||||
FileEntryNumericID(r1), FileEntryNumericID(r2))
|
||||
|
||||
if n := dedupWaitCount(t, ctx, c, aID, stem, ".txt", 2, 30*time.Second); n != 2 {
|
||||
t.Fatalf("folder A has %d copies after upload, want 2", n)
|
||||
}
|
||||
if n := dedupWaitCount(t, ctx, c, bID, stem, ".txt", 1, 30*time.Second); n != 1 {
|
||||
t.Fatalf("folder B has %d copies after upload, want 1", n)
|
||||
}
|
||||
if n := dedupWaitCount(t, ctx, c, root, stem, ".txt", 2, 30*time.Second); n != 2 {
|
||||
t.Fatalf("project root has %d copies after upload, want 2", n)
|
||||
}
|
||||
|
||||
// #2 mergeProjectRootForDedupe on the live tree: the project root that
|
||||
// carries documents must be part of the scan exactly once.
|
||||
folders, byFolder, rootFiles := liveProjectIndex(t, ctx, c, pid, root)
|
||||
merged, mergedBy := mergeProjectRootForDedupe(root, folders, byFolder, rootFiles)
|
||||
rootCount := 0
|
||||
for _, folder := range merged {
|
||||
if folder != nil && folder.ID != nil && folder.ID.String() == root {
|
||||
rootCount++
|
||||
}
|
||||
}
|
||||
if rootCount != 1 {
|
||||
t.Fatalf("mergeProjectRootForDedupe: project root appears %d times in live tree, want 1", rootCount)
|
||||
}
|
||||
if got := len(FindFilesByDedupKey(mergedBy[root], stem, ".txt")); got != 2 {
|
||||
t.Fatalf("mergeProjectRootForDedupe: root carries %d matching files, want 2", got)
|
||||
}
|
||||
|
||||
// Within-folder scan: a duplicate pair in A and in the project root.
|
||||
within := FindProjectDuplicates(merged, mergedBy, DedupOptions{})
|
||||
removesByFolder := map[string]int{}
|
||||
for _, g := range within {
|
||||
removesByFolder[g.FolderID] = len(g.Remove)
|
||||
}
|
||||
if len(within) != 2 || removesByFolder[aID] != 1 || removesByFolder[root] != 1 {
|
||||
t.Fatalf("within-folder groups = %d (%v), want exactly A:1 root:1", len(within), removesByFolder)
|
||||
}
|
||||
|
||||
// DeleteFilesByDedupKey removes every stem|ext copy in one folder.
|
||||
removed, err := c.DeleteFilesByDedupKey(ctx, aID, stem, ".txt")
|
||||
if err != nil {
|
||||
t.Fatalf("DeleteFilesByDedupKey(A): %v", err)
|
||||
}
|
||||
if len(removed) != 2 {
|
||||
t.Fatalf("DeleteFilesByDedupKey(A) removed %v, want 2 ids", removed)
|
||||
}
|
||||
if n := dedupWaitCount(t, ctx, c, aID, stem, ".txt", 0, 30*time.Second); n != 0 {
|
||||
t.Fatalf("folder A still has %d copies after DeleteFilesByDedupKey", n)
|
||||
}
|
||||
|
||||
// Re-create the A duplicates so the cross-folder project scan can be
|
||||
// applied and a single survivor proven by polling.
|
||||
uploadFolderLive(t, ctx, c, aID, local)
|
||||
uploadFolderLive(t, ctx, c, aID, local)
|
||||
if n := dedupWaitCount(t, ctx, c, aID, stem, ".txt", 2, 30*time.Second); n != 2 {
|
||||
t.Fatalf("folder A has %d re-uploaded copies, want 2", n)
|
||||
}
|
||||
|
||||
// Cross-folder dry-run over the live project: one key, five copies, and
|
||||
// the project-root copies must be part of the group (root merge live).
|
||||
groups, deleted, err := dedupeProjectEventually(t, ctx, c, pid, DedupOptions{CrossFolder: true}, false)
|
||||
if err != nil {
|
||||
t.Fatalf("DedupeProject: %v", err)
|
||||
}
|
||||
if len(deleted) != 0 {
|
||||
t.Fatalf("dry-run DedupeProject deleted %v", deleted)
|
||||
}
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("DedupeProject cross-folder groups = %d, want 1 (%+v)", len(groups), groups)
|
||||
}
|
||||
if len(groups[0].Remove) != 4 {
|
||||
t.Fatalf("cross-folder group removes %d files, want 4", len(groups[0].Remove))
|
||||
}
|
||||
if !dedupGroupHasID(groups[0], FileEntryNumericID(r1)) && !dedupGroupHasID(groups[0], FileEntryNumericID(r2)) {
|
||||
t.Fatalf("cross-folder group does not include a project-root copy (merge not applied)")
|
||||
}
|
||||
|
||||
keepID := FileEntryNumericID(groups[0].Keep)
|
||||
deleted, err = c.ApplyDedupGroups(ctx, groups)
|
||||
if err != nil {
|
||||
t.Fatalf("ApplyDedupGroups: %v", err)
|
||||
}
|
||||
if len(deleted) != 4 {
|
||||
t.Fatalf("ApplyDedupGroups deleted %v, want 4 ids", deleted)
|
||||
}
|
||||
|
||||
survivorID, total := dedupWaitTotal(t, ctx, c, []string{aID, bID, root}, stem, ".txt", 1, 40*time.Second)
|
||||
if total != 1 {
|
||||
t.Fatalf("after ApplyDedupGroups %d copies survive, want 1", total)
|
||||
}
|
||||
if survivorID != keepID {
|
||||
t.Fatalf("remaining copy id = %d, want kept id %d", survivorID, keepID)
|
||||
}
|
||||
|
||||
// The removed ids must really be gone from every folder.
|
||||
gone := map[int64]bool{
|
||||
FileEntryNumericID(a2): true,
|
||||
FileEntryNumericID(b1): true,
|
||||
FileEntryNumericID(r1): true,
|
||||
FileEntryNumericID(r2): true,
|
||||
}
|
||||
for _, fid := range []string{aID, bID, root} {
|
||||
files, err := c.FolderFiles(ctx, fid)
|
||||
if err != nil {
|
||||
t.Fatalf("FolderFiles %s: %v", fid, err)
|
||||
}
|
||||
for _, f := range FindFilesByDedupKey(files, stem, ".txt") {
|
||||
id := FileEntryNumericID(f)
|
||||
if gone[id] {
|
||||
t.Fatalf("removed id %d still present in folder %s", id, fid)
|
||||
}
|
||||
}
|
||||
}
|
||||
t.Logf("survivor id=%d keep id=%d, deleted=%v", survivorID, keepID, deleted)
|
||||
}
|
||||
|
||||
// createFolderLive creates a subfolder (retrying transient 5xx) and returns
|
||||
// its Documents folder id.
|
||||
func createFolderLive(t *testing.T, ctx context.Context, c *Client, parentID, title string) string {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(30 * time.Second)
|
||||
var (
|
||||
m map[string]any
|
||||
err error
|
||||
)
|
||||
for {
|
||||
m, err = c.CreateFolder(ctx, parentID, title)
|
||||
if err == nil || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("CreateFolder %q: %v", title, err)
|
||||
}
|
||||
if m == nil {
|
||||
t.Fatalf("CreateFolder %q: empty response", title)
|
||||
}
|
||||
switch v := m["id"].(type) {
|
||||
case float64:
|
||||
return strconv.FormatInt(int64(v), 10)
|
||||
case json.Number:
|
||||
return v.String()
|
||||
case string:
|
||||
if v != "" {
|
||||
return v
|
||||
}
|
||||
}
|
||||
t.Fatalf("CreateFolder %q: no id in response %#v", title, m)
|
||||
return ""
|
||||
}
|
||||
|
||||
// writeLocalFile writes content to a temp file and returns its path.
|
||||
func writeLocalFile(t *testing.T, name string, content []byte) string {
|
||||
t.Helper()
|
||||
p := filepath.Join(t.TempDir(), name)
|
||||
if err := os.WriteFile(p, content, 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
// uploadFolderLive uploads localPath into folderID (retrying the portal's
|
||||
// transient post-create 500) and returns the file entry.
|
||||
func uploadFolderLive(t *testing.T, ctx context.Context, c *Client, folderID, localPath string) *FileEntry {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(30 * time.Second)
|
||||
var (
|
||||
e *FileEntry
|
||||
err error
|
||||
)
|
||||
for {
|
||||
e, err = c.UploadToFolder(ctx, folderID, localPath)
|
||||
if err == nil || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("UploadToFolder %s: %v", folderID, err)
|
||||
}
|
||||
if e == nil || e.ID == nil {
|
||||
t.Fatalf("UploadToFolder %s: no file entry (%+v)", folderID, e)
|
||||
}
|
||||
return e
|
||||
}
|
||||
|
||||
// liveProjectIndex rebuilds the project folder/file index the same way
|
||||
// DedupeProject does, for direct mergeProjectRootForDedupe assertions.
|
||||
func liveProjectIndex(t *testing.T, ctx context.Context, c *Client, projectID, rootID string) ([]*FolderEntry, map[string][]*FileEntry, []*FileEntry) {
|
||||
t.Helper()
|
||||
pf := getProjectFilesEventually(t, ctx, c, projectID)
|
||||
rootFiles := folderFilesEventually(t, ctx, c, rootID)
|
||||
folders := make([]*FolderEntry, 0, len(pf.Folders)+1)
|
||||
byFolder := make(map[string][]*FileEntry, len(pf.Folders)+1)
|
||||
for _, folder := range pf.Folders {
|
||||
if folder == nil || folder.ID == nil {
|
||||
continue
|
||||
}
|
||||
fid := folder.ID.String()
|
||||
if fid == rootID {
|
||||
byFolder[fid] = rootFiles
|
||||
} else {
|
||||
byFolder[fid] = folderFilesEventually(t, ctx, c, fid)
|
||||
}
|
||||
folders = append(folders, folder)
|
||||
}
|
||||
return folders, byFolder, rootFiles
|
||||
}
|
||||
|
||||
// projectFolderEventually resolves the project Documents root id, retrying on
|
||||
// a transient portal answer.
|
||||
func projectFolderEventually(t *testing.T, ctx context.Context, c *Client, projectID string) string {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(30 * time.Second)
|
||||
var (
|
||||
root string
|
||||
err error
|
||||
)
|
||||
for {
|
||||
root, err = c.projectFolderID(ctx, projectID)
|
||||
if err == nil || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("projectFolderID: %v", err)
|
||||
}
|
||||
if root == "" {
|
||||
t.Fatal("projectFolderID returned empty id")
|
||||
}
|
||||
return root
|
||||
}
|
||||
|
||||
// getProjectFilesEventually lists a project's files/folders, retrying on a
|
||||
// transient portal answer.
|
||||
func getProjectFilesEventually(t *testing.T, ctx context.Context, c *Client, projectID string) *ProjectFilesResponse {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(30 * time.Second)
|
||||
var last error
|
||||
for {
|
||||
pf, err := c.GetProjectFiles(ctx, projectID)
|
||||
if err == nil {
|
||||
return pf
|
||||
}
|
||||
last = err
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("GetProjectFiles %s: %v", projectID, last)
|
||||
}
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
// folderFilesEventually lists a folder, retrying while the portal answers
|
||||
// transiently (a freshly created folder can 500 until its parent map settles).
|
||||
func folderFilesEventually(t *testing.T, ctx context.Context, c *Client, folderID string) []*FileEntry {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(30 * time.Second)
|
||||
var last error
|
||||
for {
|
||||
files, err := c.FolderFiles(ctx, folderID)
|
||||
if err == nil {
|
||||
return files
|
||||
}
|
||||
last = err
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("FolderFiles %s: %v", folderID, last)
|
||||
}
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
// dedupeProjectEventually runs a project dedup scan, retrying the whole scan on
|
||||
// a transient portal error. It is used for dry-runs only (apply must stay a
|
||||
// single deliberate call).
|
||||
func dedupeProjectEventually(t *testing.T, ctx context.Context, c *Client, projectID string, opts DedupOptions, apply bool) ([]DedupGroup, []int, error) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(30 * time.Second)
|
||||
var (
|
||||
groups []DedupGroup
|
||||
deleted []int
|
||||
err error
|
||||
)
|
||||
for {
|
||||
groups, deleted, err = c.DedupeProject(ctx, projectID, opts, apply)
|
||||
if err == nil || time.Now().After(deadline) {
|
||||
return groups, deleted, err
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
}
|
||||
|
||||
// dedupWaitCount polls folderID until want stem|ext copies are visible.
|
||||
func dedupWaitCount(t *testing.T, ctx context.Context, c *Client, folderID, stem, ext string, want int, d time.Duration) int {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(d)
|
||||
got := -1
|
||||
for {
|
||||
if files, err := c.FolderFiles(ctx, folderID); err == nil {
|
||||
got = len(FindFilesByDedupKey(files, stem, ext))
|
||||
if got == want {
|
||||
return got
|
||||
}
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
return got
|
||||
}
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
// dedupWaitTotal polls the given folders until the total number of stem|ext
|
||||
// copies reaches want, returning the last seen file id and count.
|
||||
func dedupWaitTotal(t *testing.T, ctx context.Context, c *Client, folderIDs []string, stem, ext string, want int, d time.Duration) (int64, int) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(d)
|
||||
var survivor int64
|
||||
total := -1
|
||||
for {
|
||||
survivor, total = 0, 0
|
||||
for _, fid := range folderIDs {
|
||||
files, err := c.FolderFiles(ctx, fid)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
for _, f := range FindFilesByDedupKey(files, stem, ext) {
|
||||
total++
|
||||
survivor = FileEntryNumericID(f)
|
||||
}
|
||||
}
|
||||
if total == want {
|
||||
return survivor, total
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
return survivor, total
|
||||
}
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
func dedupGroupHasID(g DedupGroup, id int64) bool {
|
||||
if id == 0 {
|
||||
return false
|
||||
}
|
||||
if FileEntryNumericID(g.Keep) == id {
|
||||
return true
|
||||
}
|
||||
for _, f := range g.Remove {
|
||||
if FileEntryNumericID(f) == id {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestFileDedupKey(t *testing.T) {
|
||||
title := "OO-HONDA-7-INDEX.docx"
|
||||
exst := ".docx"
|
||||
f := &FileEntry{Title: &title, FileExst: &exst}
|
||||
if got := FileDedupKey(f); got != "OO-HONDA-7-INDEX|docx" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFindFilesByDedupKey(t *testing.T) {
|
||||
a := &FileEntry{Title: strPtr("foo.docx"), FileExst: strPtr(".docx")}
|
||||
b := &FileEntry{Title: strPtr("foo.md"), FileExst: strPtr(".md")}
|
||||
files := []*FileEntry{a, b}
|
||||
got := FindFilesByDedupKey(files, "foo", ".docx")
|
||||
if len(got) != 1 || got[0] != a {
|
||||
t.Fatalf("got %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFindWithinFolderDuplicates(t *testing.T) {
|
||||
t1 := time.Date(2026, 8, 27, 16, 0, 0, 0, time.UTC)
|
||||
t2 := t1.Add(time.Hour)
|
||||
old := &FileEntry{ID: jsonNum("1"), Title: strPtr("idx.docx"), FileExst: strPtr(".docx"), Updated: &t1}
|
||||
new := &FileEntry{ID: jsonNum("2"), Title: strPtr("idx.docx"), FileExst: strPtr(".docx"), Updated: &t2}
|
||||
indexed := []ProjectFolderFile{
|
||||
{FolderID: "490", FolderTitle: "00-Index", File: old},
|
||||
{FolderID: "490", FolderTitle: "00-Index", File: new},
|
||||
}
|
||||
groups := findWithinFolderDuplicates(indexed)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("groups=%d", len(groups))
|
||||
}
|
||||
if FileEntryNumericID(groups[0].Keep) != 2 {
|
||||
t.Fatalf("keep id=%d", FileEntryNumericID(groups[0].Keep))
|
||||
}
|
||||
if len(groups[0].Remove) != 1 || FileEntryNumericID(groups[0].Remove[0]) != 1 {
|
||||
t.Fatalf("remove=%v", groups[0].Remove)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCrossFolderPrefersNonTrash(t *testing.T) {
|
||||
t1 := time.Date(2026, 8, 27, 18, 0, 0, 0, time.UTC)
|
||||
t2 := t1.Add(-time.Hour)
|
||||
trash := &FileEntry{ID: jsonNum("10"), Title: strPtr("INDEX.md"), FileExst: strPtr(".md"), Updated: &t1}
|
||||
good := &FileEntry{ID: jsonNum("20"), Title: strPtr("INDEX.md"), FileExst: strPtr(".md"), Updated: &t2}
|
||||
indexed := []ProjectFolderFile{
|
||||
{FolderID: "493", FolderTitle: "_trash-md", File: trash},
|
||||
{FolderID: "492", FolderTitle: "OCR", File: good},
|
||||
}
|
||||
groups := findCrossFolderDuplicates(indexed)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("groups=%d", len(groups))
|
||||
}
|
||||
if FileEntryNumericID(groups[0].Keep) != 20 {
|
||||
t.Fatalf("keep id=%d", FileEntryNumericID(groups[0].Keep))
|
||||
}
|
||||
}
|
||||
|
||||
func TestMergeProjectRootForDedupe(t *testing.T) {
|
||||
old := &FileEntry{ID: jsonNum("1"), Title: strPtr("a.docx"), FileExst: strPtr(".docx")}
|
||||
newer := &FileEntry{ID: jsonNum("2"), Title: strPtr("a.docx"), FileExst: strPtr(".docx")}
|
||||
rootFiles := []*FileEntry{old, newer}
|
||||
folders, byFolder := mergeProjectRootForDedupe("489", nil, nil, rootFiles)
|
||||
if len(folders) != 1 || folders[0].ID.String() != "489" {
|
||||
t.Fatalf("folders=%+v", folders)
|
||||
}
|
||||
if len(byFolder["489"]) != 2 {
|
||||
t.Fatalf("root files=%d", len(byFolder["489"]))
|
||||
}
|
||||
groups := findWithinFolderDuplicates([]ProjectFolderFile{
|
||||
{FolderID: "489", FolderTitle: "(project root)", File: old},
|
||||
{FolderID: "489", FolderTitle: "(project root)", File: newer},
|
||||
})
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("groups=%d", len(groups))
|
||||
}
|
||||
}
|
||||
|
||||
func TestFindDuplicatesSkipsEmptyKey(t *testing.T) {
|
||||
// Dotfiles (".env", ".gitignore", ".npmrc") normalize to an empty stem, so
|
||||
// FileDedupKey is "". They are not duplicates of each other and must never
|
||||
// form a dedup group that would delete one of them.
|
||||
env := &FileEntry{ID: jsonNum("1"), Title: strPtr(".env")}
|
||||
gitignore := &FileEntry{ID: jsonNum("2"), Title: strPtr(".gitignore")}
|
||||
npmrc := &FileEntry{ID: jsonNum("3"), Title: strPtr(".npmrc")}
|
||||
if FileDedupKey(env) != "" || FileDedupKey(gitignore) != "" || FileDedupKey(npmrc) != "" {
|
||||
t.Fatalf("dotfiles should have empty dedup key")
|
||||
}
|
||||
within := []ProjectFolderFile{
|
||||
{FolderID: "500", FolderTitle: "Cfg", File: env},
|
||||
{FolderID: "500", FolderTitle: "Cfg", File: gitignore},
|
||||
}
|
||||
if groups := findWithinFolderDuplicates(within); len(groups) != 0 {
|
||||
t.Fatalf("within-folder empty-key groups = %d, want 0 (%+v)", len(groups), groups)
|
||||
}
|
||||
cross := []ProjectFolderFile{
|
||||
{FolderID: "500", FolderTitle: "Cfg", File: env},
|
||||
{FolderID: "501", FolderTitle: "Other", File: npmrc},
|
||||
}
|
||||
if groups := findCrossFolderDuplicates(cross); len(groups) != 0 {
|
||||
t.Fatalf("cross-folder empty-key groups = %d, want 0 (%+v)", len(groups), groups)
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsTrashFolderTitle(t *testing.T) {
|
||||
if !IsTrashFolderTitle("_trash-md") {
|
||||
t.Fatal("expected trash")
|
||||
}
|
||||
if IsTrashFolderTitle("00-Index") {
|
||||
t.Fatal("expected not trash")
|
||||
}
|
||||
}
|
||||
|
||||
func strPtr(s string) *string { return &s }
|
||||
|
||||
func jsonNum(s string) *json.Number {
|
||||
n := json.Number(s)
|
||||
return &n
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
package onlyoffice
|
||||
|
||||
// Human-readable folder paths for search results (F9). The OnlyOffice ES
|
||||
// index stores only ancestor folder ids; titles live in the Documents tree, so
|
||||
// resolving a path costs one GET /api/2.0/files/{id} per distinct folder,
|
||||
// cached on the client. Folders that cannot be listed (e.g. a section root)
|
||||
// fall back to their id, so a path is always produced.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// FolderTitle returns the title of a Documents folder id, cached on the client.
|
||||
// An empty id yields an empty title. Unknown/unlistable ids (section roots)
|
||||
// return ("", nil) so callers can fall back to the id.
|
||||
func (c *Client) FolderTitle(ctx context.Context, folderID string) (string, error) {
|
||||
folderID = strings.TrimSpace(folderID)
|
||||
if folderID == "" {
|
||||
return "", nil
|
||||
}
|
||||
c.folderTitlesMu.Lock()
|
||||
if c.folderTitles != nil {
|
||||
if t, ok := c.folderTitles[folderID]; ok {
|
||||
c.folderTitlesMu.Unlock()
|
||||
return t, nil
|
||||
}
|
||||
}
|
||||
c.folderTitlesMu.Unlock()
|
||||
|
||||
title := ""
|
||||
out, err := c.ListFolder(ctx, folderID)
|
||||
if err == nil {
|
||||
if cur, ok := out["current"].(map[string]any); ok {
|
||||
if s, ok := cur["title"].(string); ok {
|
||||
title = strings.TrimSpace(s)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
c.folderTitlesMu.Lock()
|
||||
if c.folderTitles == nil {
|
||||
c.folderTitles = map[string]string{}
|
||||
}
|
||||
c.folderTitles[folderID] = title
|
||||
c.folderTitlesMu.Unlock()
|
||||
return title, nil
|
||||
}
|
||||
|
||||
// FolderPath resolves an ancestor folder id chain (root → leaf, as the ES
|
||||
// backend reports it) into folder titles, falling back to the id when a title
|
||||
// cannot be read. The result never fails on a single lookup: only the whole
|
||||
// call honours ctx cancellation.
|
||||
func (c *Client) FolderPath(ctx context.Context, ids []string) []string {
|
||||
out := make([]string, 0, len(ids))
|
||||
for _, id := range ids {
|
||||
if err := ctx.Err(); err != nil {
|
||||
break
|
||||
}
|
||||
title, err := c.FolderTitle(ctx, id)
|
||||
if err != nil || title == "" {
|
||||
title = id
|
||||
}
|
||||
out = append(out, title)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// UniquePath builds a stable, human-readable, unique path for a result: the
|
||||
// resolved folder chain plus the file title. "." separates nothing — the
|
||||
// segments are joined with "/", matching the Documents breadcrumb the web UI
|
||||
// shows.
|
||||
func (c *Client) UniquePath(ctx context.Context, folderPath []string, title string) string {
|
||||
parts := c.FolderPath(ctx, folderPath)
|
||||
if t := strings.TrimSpace(title); t != "" {
|
||||
parts = append(parts, t)
|
||||
}
|
||||
return strings.Join(parts, "/")
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
//go:build integration
|
||||
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestIntegrationFolderPath resolves the real Fibu EDL folder chain
|
||||
// (project root 522 → Eingangsrechnungen 647 → 2025 649) to titles.
|
||||
func TestIntegrationFolderPath(t *testing.T) {
|
||||
creds := GetEnvironmentCredentials()
|
||||
if strings.TrimSpace(creds.Url) == "" || strings.TrimSpace(creds.User) == "" {
|
||||
t.Skip("no ONLYOFFICE_URL/USER credentials")
|
||||
}
|
||||
c := NewClient(creds)
|
||||
ctx := context.Background()
|
||||
|
||||
path := c.FolderPath(ctx, []string{"522", "647", "649"})
|
||||
if len(path) != 3 {
|
||||
t.Fatalf("FolderPath returned %v, want 3 segments", path)
|
||||
}
|
||||
for i, seg := range path {
|
||||
if strings.TrimSpace(seg) == "" {
|
||||
t.Errorf("segment %d empty: %v", i, path)
|
||||
}
|
||||
}
|
||||
full := c.UniquePath(ctx, []string{"522", "647", "649"}, "Rechnung-x.pdf")
|
||||
if !strings.HasSuffix(full, "Rechnung-x.pdf") || !strings.Contains(full, "/") {
|
||||
t.Errorf("UniquePath = %q, want a slash-joined path ending in the file", full)
|
||||
}
|
||||
t.Logf("path=%v full=%q", path, full)
|
||||
}
|
||||
+166
@@ -0,0 +1,166 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ErrFileExists is returned when --no-replace / no-clobber upload hits an existing stem|ext.
|
||||
var ErrFileExists = errors.New("onlyoffice: file already exists in folder (use replace or delete first)")
|
||||
|
||||
// FileEntryStem returns the logical basename without duplicated extensions.
|
||||
// OO often stores title="foo.docx" and fileExst=".docx" (UI shows foo.docx.docx).
|
||||
func FileEntryStem(f *FileEntry) string {
|
||||
if f == nil || f.Title == nil {
|
||||
return ""
|
||||
}
|
||||
exst := ""
|
||||
if f.FileExst != nil {
|
||||
exst = *f.FileExst
|
||||
}
|
||||
return NormalizeUploadStem(*f.Title, exst)
|
||||
}
|
||||
|
||||
// NormalizeUploadStem derives a stable stem for matching uploads.
|
||||
func NormalizeUploadStem(title, exst string) string {
|
||||
t := strings.TrimSpace(title)
|
||||
t = strings.TrimSuffix(t, ".")
|
||||
if exst != "" && strings.HasSuffix(t, exst) {
|
||||
t = strings.TrimSuffix(t, exst)
|
||||
}
|
||||
if ext := filepath.Ext(t); ext != "" {
|
||||
t = strings.TrimSuffix(t, ext)
|
||||
}
|
||||
return strings.TrimSpace(t)
|
||||
}
|
||||
|
||||
// UploadStemFromLocal returns the stem used to match/replace folder files.
|
||||
func UploadStemFromLocal(localPath string) string {
|
||||
base := filepath.Base(localPath)
|
||||
ext := filepath.Ext(base)
|
||||
if ext != "" {
|
||||
base = strings.TrimSuffix(base, ext)
|
||||
}
|
||||
return base
|
||||
}
|
||||
|
||||
// FolderFiles returns file entries in a Documents folder.
|
||||
func (c *Client) FolderFiles(ctx context.Context, folderID string) ([]*FileEntry, error) {
|
||||
raw, err := c.ListFolder(ctx, folderID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ParseFolderFileEntries(raw), nil
|
||||
}
|
||||
|
||||
// ParseFolderFileEntries extracts []*FileEntry from ListFolder JSON.
|
||||
func ParseFolderFileEntries(raw map[string]any) []*FileEntry {
|
||||
items, _ := raw["files"].([]any)
|
||||
out := make([]*FileEntry, 0, len(items))
|
||||
for _, it := range items {
|
||||
m, ok := it.(map[string]any)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
b, err := json.Marshal(m)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
var f FileEntry
|
||||
if err := json.Unmarshal(b, &f); err != nil {
|
||||
continue
|
||||
}
|
||||
out = append(out, &f)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// FindFilesByStem returns folder files whose logical stem matches.
|
||||
func FindFilesByStem(files []*FileEntry, stem string) []*FileEntry {
|
||||
stem = strings.TrimSpace(stem)
|
||||
if stem == "" {
|
||||
return nil
|
||||
}
|
||||
var out []*FileEntry
|
||||
for _, f := range files {
|
||||
if FileEntryStem(f) == stem {
|
||||
out = append(out, f)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// DeleteFilesByStem removes all files in folderID matching stem (any extension).
|
||||
// Prefer DeleteFilesByDedupKey when the upload extension is known.
|
||||
func (c *Client) DeleteFilesByStem(ctx context.Context, folderID, stem string) ([]int, error) {
|
||||
files, err := c.FolderFiles(ctx, folderID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
matches := FindFilesByStem(files, stem)
|
||||
if len(matches) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
ids := make([]int, 0, len(matches))
|
||||
for _, f := range matches {
|
||||
n := int(FileEntryNumericID(f))
|
||||
if n != 0 {
|
||||
ids = append(ids, n)
|
||||
}
|
||||
}
|
||||
if len(ids) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
if err := c.DeleteFiles(ctx, ids); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ids, nil
|
||||
}
|
||||
|
||||
// AssertNoFileConflict reports ErrFileExists when localPath stem|ext is already in folderID.
|
||||
func (c *Client) AssertNoFileConflict(ctx context.Context, folderID, localPath string) error {
|
||||
files, err := c.FolderFiles(ctx, folderID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
stem := UploadStemFromLocal(localPath)
|
||||
ext := UploadExtFromLocal(localPath)
|
||||
matches := FindFilesByDedupKey(files, stem, ext)
|
||||
if len(matches) == 0 {
|
||||
return nil
|
||||
}
|
||||
ids := make([]string, 0, len(matches))
|
||||
for _, f := range matches {
|
||||
ids = append(ids, fmt.Sprintf("%d", FileEntryNumericID(f)))
|
||||
}
|
||||
return fmt.Errorf("%w: %s%s in folder %s (existing file ids: %s)",
|
||||
ErrFileExists, stem, ext, folderID, strings.Join(ids, ", "))
|
||||
}
|
||||
|
||||
// UploadProjectFileNoClobber uploads only when stem|ext is not already in the project folder.
|
||||
func (c *Client) UploadProjectFileNoClobber(ctx context.Context, projectID, localPath string) (*FileEntry, error) {
|
||||
folderID, err := c.projectFolderID(ctx, projectID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := c.AssertNoFileConflict(ctx, folderID, localPath); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return c.UploadProjectFile(ctx, projectID, localPath)
|
||||
}
|
||||
|
||||
// UploadToFolderReplacing deletes same stem+ext files then uploads localPath.
|
||||
func (c *Client) UploadToFolderReplacing(ctx context.Context, folderID, localPath string) (*FileEntry, []int, error) {
|
||||
stem := UploadStemFromLocal(localPath)
|
||||
ext := UploadExtFromLocal(localPath)
|
||||
deleted, err := c.DeleteFilesByDedupKey(ctx, folderID, stem, ext)
|
||||
if err != nil {
|
||||
return nil, deleted, err
|
||||
}
|
||||
ent, err := c.UploadToFolder(ctx, folderID, localPath)
|
||||
return ent, deleted, err
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package onlyoffice
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestNormalizeUploadStem(t *testing.T) {
|
||||
tests := []struct {
|
||||
title, exst, want string
|
||||
}{
|
||||
{"OO-HONDA-7-INDEX.docx", ".docx", "OO-HONDA-7-INDEX"},
|
||||
{"README.txt", ".txt", "README"},
|
||||
{"car-docs-print.docx", ".docx", "car-docs-print"},
|
||||
{"plain.", "", "plain"},
|
||||
{"foo", ".docx", "foo"},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
if got := NormalizeUploadStem(tc.title, tc.exst); got != tc.want {
|
||||
t.Fatalf("NormalizeUploadStem(%q,%q)=%q want %q", tc.title, tc.exst, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileEntryStem(t *testing.T) {
|
||||
title := "00-INDEX.docx"
|
||||
exst := ".docx"
|
||||
f := &FileEntry{Title: &title, FileExst: &exst}
|
||||
if got := FileEntryStem(f); got != "00-INDEX" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUploadStemFromLocal(t *testing.T) {
|
||||
if got := UploadStemFromLocal("/tmp/OO-HONDA-7-INDEX.docx"); got != "OO-HONDA-7-INDEX" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
}
|
||||
+482
@@ -0,0 +1,482 @@
|
||||
package onlyoffice
|
||||
|
||||
// WebDAV-oriented Files operations. These expose the Documents module through
|
||||
// value types and cover everything needed to back a filesystem mapping:
|
||||
// listing (including the virtual @root sections), folder/file CRUD, move/copy,
|
||||
// and streaming upload/download. They are intentionally small and dependency
|
||||
// free (only net/http), so callers are not forced to import heavier parts of
|
||||
// the library.
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"mime/multipart"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// DavFolder is a folder row from the Files module.
|
||||
type DavFolder struct {
|
||||
ID string
|
||||
Title string
|
||||
ParentID string
|
||||
RootType int // 1=Common, 3=Trash, 5=My, 6=Share, 8=Projects, ...
|
||||
FilesCount int
|
||||
FoldersCount int
|
||||
Access int
|
||||
Shared bool
|
||||
Updated string
|
||||
}
|
||||
|
||||
// DavFile is a file row from the Files module.
|
||||
type DavFile struct {
|
||||
ID string
|
||||
Title string
|
||||
Size int64
|
||||
Updated string
|
||||
ViewURL string
|
||||
}
|
||||
|
||||
// DavListing is the contents of one folder.
|
||||
type DavListing struct {
|
||||
Current DavFolder
|
||||
Files []DavFile
|
||||
Folders []DavFolder
|
||||
}
|
||||
|
||||
// ListDavFolder returns the contents of a folder by id, which may be a
|
||||
// symbolic root such as "@my". For "@root" use ListDavSections.
|
||||
//
|
||||
// Deprecated: use FileStore.List via Client.Files()/Client.FileStore.
|
||||
func (c *Client) ListDavFolder(ctx context.Context, id string) (*DavListing, error) {
|
||||
raw, err := c.getJSON(ctx, "/api/2.0/files/"+url.PathEscape(id))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
resp, err := responseField(raw, "response")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// @root returns an array with a single blob; a normal folder returns an
|
||||
// object. Normalize both.
|
||||
if len(resp) > 0 && resp[0] == '[' {
|
||||
var arr []*DavListing
|
||||
if err := json.Unmarshal(resp, &arr); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(arr) == 0 {
|
||||
return &DavListing{}, nil
|
||||
}
|
||||
return arr[0], nil
|
||||
}
|
||||
var l DavListing
|
||||
if err := json.Unmarshal(resp, &l); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &l, nil
|
||||
}
|
||||
|
||||
// ListDavSections returns the virtual top-level sections shown by @root
|
||||
// ("In projects", "My documents", "Shared with me", "Common", "Favorites",
|
||||
// "Recent", "Trash"). Each is the `current` folder of one @root element.
|
||||
func (c *Client) ListDavSections(ctx context.Context) ([]DavFolder, error) {
|
||||
raw, err := c.getJSON(ctx, "/api/2.0/files/@root")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
resp, err := responseField(raw, "response")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var arr []struct {
|
||||
Current DavFolder `json:"current"`
|
||||
}
|
||||
if err := json.Unmarshal(resp, &arr); err != nil {
|
||||
// Tolerate a non-array (single listing) response.
|
||||
var single DavListing
|
||||
if err2 := json.Unmarshal(resp, &single); err2 != nil {
|
||||
return nil, err
|
||||
}
|
||||
return []DavFolder{single.Current}, nil
|
||||
}
|
||||
sections := make([]DavFolder, 0, len(arr))
|
||||
for i := range arr {
|
||||
sections = append(sections, arr[i].Current)
|
||||
}
|
||||
return sections, nil
|
||||
}
|
||||
|
||||
// CreateDavFolder creates a folder titled title inside parentID.
|
||||
//
|
||||
// Deprecated: use FileStore.CreateFolder via Client.Files()/Client.FileStore.
|
||||
func (c *Client) CreateDavFolder(ctx context.Context, parentID, title string) (*DavFolder, error) {
|
||||
raw, err := c.postJSON(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentID),
|
||||
map[string]string{"title": title})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var env struct {
|
||||
Response *DavFolder `json:"response"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &env); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if env.Response == nil {
|
||||
return nil, fmt.Errorf("onlyoffice: empty create-folder response")
|
||||
}
|
||||
return env.Response, nil
|
||||
}
|
||||
|
||||
// RenameDavFolder renames a folder.
|
||||
//
|
||||
// Deprecated: use FileStore.Rename via Client.Files()/Client.FileStore.
|
||||
func (c *Client) RenameDavFolder(ctx context.Context, id, title string) error {
|
||||
_, err := c.putJSON(ctx, "/api/2.0/files/folder/"+url.PathEscape(id),
|
||||
map[string]string{"title": title})
|
||||
return err
|
||||
}
|
||||
|
||||
// RenameDavFile renames a file (title includes the extension).
|
||||
//
|
||||
// Deprecated: use FileStore.Rename via Client.Files()/Client.FileStore.
|
||||
func (c *Client) RenameDavFile(ctx context.Context, id, title string) error {
|
||||
_, err := c.putJSON(ctx, "/api/2.0/files/file/"+url.PathEscape(id),
|
||||
map[string]string{"title": title})
|
||||
return err
|
||||
}
|
||||
|
||||
// MoveDavItems moves the given folders and/or files into destFolderID.
|
||||
// The fileops API answers 200 with per-operation error strings even when
|
||||
// nothing moves (e.g. missing permission), so the response is parsed and the
|
||||
// first operation error is returned instead of a silent nil.
|
||||
//
|
||||
// Deprecated: use FileStore.Move via Client.Files()/Client.FileStore.
|
||||
func (c *Client) MoveDavItems(ctx context.Context, folderIDs, fileIDs []string, destFolderID string) error {
|
||||
raw, err := c.putJSON(ctx, "/api/2.0/files/fileops/move", map[string]any{
|
||||
"folderIds": nums(folderIDs),
|
||||
"fileIds": nums(fileIDs),
|
||||
"destFolderId": num(destFolderID),
|
||||
"resolveType": "Skip",
|
||||
"holdResult": true,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return fileopsError(raw)
|
||||
}
|
||||
|
||||
// CopyDavItems copies the given folders and/or files into destFolderID.
|
||||
// Per-operation errors are surfaced like in MoveDavItems.
|
||||
//
|
||||
// Deprecated: use FileStore.Copy via Client.Files()/Client.FileStore.
|
||||
func (c *Client) CopyDavItems(ctx context.Context, folderIDs, fileIDs []string, destFolderID string) error {
|
||||
raw, err := c.putJSON(ctx, "/api/2.0/files/fileops/copy", map[string]any{
|
||||
"folderIds": nums(folderIDs),
|
||||
"fileIds": nums(fileIDs),
|
||||
"destFolderId": num(destFolderID),
|
||||
"conflictResolveType": "Skip",
|
||||
"deleteAfter": true,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return fileopsError(raw)
|
||||
}
|
||||
|
||||
// ListFileOps returns the currently active file operations
|
||||
// (GET /api/2.0/files/fileops) for status polling.
|
||||
func (c *Client) ListFileOps(ctx context.Context) ([]map[string]any, error) {
|
||||
raw, err := c.getJSON(ctx, "/api/2.0/files/fileops")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
resp, err := responseField(raw, "response")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(resp) == 0 || string(resp) == "null" {
|
||||
return nil, nil
|
||||
}
|
||||
var ops []map[string]any
|
||||
if err := json.Unmarshal(resp, &ops); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ops, nil
|
||||
}
|
||||
|
||||
// fileopsError extracts per-operation "error" strings from a fileops/move or
|
||||
// fileops/copy envelope. A 200 with error entries means nothing moved.
|
||||
func fileopsError(raw json.RawMessage) error {
|
||||
resp, err := responseField(raw, "response")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
var ops []struct {
|
||||
Error *string `json:"error"`
|
||||
Finished *bool `json:"finished"`
|
||||
Progress *int `json:"progress"`
|
||||
}
|
||||
if err := json.Unmarshal(resp, &ops); err != nil {
|
||||
return nil // not an operations envelope — nothing to report
|
||||
}
|
||||
var errs []string
|
||||
for _, op := range ops {
|
||||
if op.Error != nil && *op.Error != "" {
|
||||
errs = append(errs, *op.Error)
|
||||
}
|
||||
}
|
||||
if len(errs) > 0 {
|
||||
return fmt.Errorf("onlyoffice: fileops: %s", strings.Join(errs, "; "))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// DeleteDavItems deletes the given folders and/or files.
|
||||
//
|
||||
// Deprecated: use FileStore.Delete via Client.Files()/Client.FileStore.
|
||||
func (c *Client) DeleteDavItems(ctx context.Context, folderIDs, fileIDs []string) error {
|
||||
body := map[string]any{"DeleteAfter": true, "Immediately": true}
|
||||
for _, id := range folderIDs {
|
||||
if err := c.deleteDavItem(ctx, "/api/2.0/files/folder/"+url.PathEscape(id), body); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
for _, id := range fileIDs {
|
||||
if err := c.deleteDavItem(ctx, "/api/2.0/files/file/"+url.PathEscape(id), body); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// deleteDavItem deletes one item, retrying transient 429/502/503/504 answers
|
||||
// through DoRetry like every other bulk path (deletes are idempotent).
|
||||
func (c *Client) deleteDavItem(ctx context.Context, path string, body any) error {
|
||||
return DoRetry(ctx, DefaultRetryPolicy(), func() error {
|
||||
_, err := c.deleteJSON(ctx, path, body)
|
||||
return err
|
||||
})
|
||||
}
|
||||
|
||||
// UploadDavFile uploads src (fileName) into folderID, streaming from src.
|
||||
//
|
||||
// Deprecated: use FileStore.Upload via Client.Files()/Client.FileStore.
|
||||
func (c *Client) UploadDavFile(ctx context.Context, folderID, fileName string, src io.Reader) (*DavFile, error) {
|
||||
raw, err := c.uploadReader(ctx, "/api/2.0/files/"+url.PathEscape(folderID)+"/upload", "file", fileName, src)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var env struct {
|
||||
Response *DavFile `json:"response"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &env); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if env.Response == nil {
|
||||
return nil, fmt.Errorf("onlyoffice: empty upload response")
|
||||
}
|
||||
return env.Response, nil
|
||||
}
|
||||
|
||||
// DownloadDavFile streams the file identified by id to w, returning bytes
|
||||
// copied. It shares the MinIO stale-S3 fallback with DownloadFile.
|
||||
//
|
||||
// Deprecated: use FileStore.Download via Client.Files()/Client.FileStore.
|
||||
func (c *Client) DownloadDavFile(ctx context.Context, id string, w io.Writer) (int64, error) {
|
||||
return c.DownloadFile(ctx, id, w)
|
||||
}
|
||||
|
||||
// --- internal helpers -------------------------------------------------------
|
||||
|
||||
// deleteJSON performs an authenticated DELETE with an optional JSON body.
|
||||
func (c *Client) deleteJSON(ctx context.Context, path string, body any) (json.RawMessage, error) {
|
||||
var reader io.Reader
|
||||
if body != nil {
|
||||
buf, err := json.Marshal(body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
reader = bytes.NewReader(buf)
|
||||
}
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodDelete, c.baseURL()+path, reader)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Authorization", auth)
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
req.Header.Set("Accept", "application/json")
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
raw, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("DELETE %s: %d %s", path, resp.StatusCode, truncate(string(raw), 400))
|
||||
}
|
||||
return raw, nil
|
||||
}
|
||||
|
||||
// uploadReader uploads a stream to path under the given form field name.
|
||||
func (c *Client) uploadReader(ctx context.Context, path, fieldName, fileName string, src io.Reader) (json.RawMessage, error) {
|
||||
var buf bytes.Buffer
|
||||
mw := multipart.NewWriter(&buf)
|
||||
part, err := mw.CreateFormFile(fieldName, fileName)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if _, err := io.Copy(part, src); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := mw.Close(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.baseURL()+path, &buf)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Authorization", auth)
|
||||
req.Header.Set("Content-Type", mw.FormDataContentType())
|
||||
req.Header.Set("Accept", "application/json")
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
raw, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("upload %s: %d %s", path, resp.StatusCode, truncate(string(raw), 400))
|
||||
}
|
||||
return raw, nil
|
||||
}
|
||||
|
||||
func nums(ids []string) []json.Number {
|
||||
out := make([]json.Number, 0, len(ids))
|
||||
for _, id := range ids {
|
||||
if _, err := strconv.Atoi(id); err == nil {
|
||||
out = append(out, json.Number(id))
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func num(id string) any {
|
||||
if _, err := strconv.Atoi(id); err == nil {
|
||||
return json.Number(id)
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
// UnmarshalJSON decodes a folder from the portal envelope, including fields
|
||||
// that the base FolderEntry omits (parentId, rootFolderType, access, ...).
|
||||
func (f *DavFolder) UnmarshalJSON(b []byte) error {
|
||||
var raw struct {
|
||||
ID *json.Number `json:"id"`
|
||||
Title *string `json:"title"`
|
||||
ParentID *json.Number `json:"parentId"`
|
||||
RootType *int `json:"rootFolderType"`
|
||||
FilesCount *int `json:"filesCount"`
|
||||
FoldersCount *int `json:"foldersCount"`
|
||||
Access *int `json:"access"`
|
||||
Shared *bool `json:"shared"`
|
||||
Updated *string `json:"updated"`
|
||||
}
|
||||
if err := json.Unmarshal(b, &raw); err != nil {
|
||||
return err
|
||||
}
|
||||
if raw.ID != nil {
|
||||
f.ID = raw.ID.String()
|
||||
}
|
||||
if raw.Title != nil {
|
||||
f.Title = *raw.Title
|
||||
}
|
||||
if raw.ParentID != nil {
|
||||
f.ParentID = raw.ParentID.String()
|
||||
}
|
||||
if raw.RootType != nil {
|
||||
f.RootType = *raw.RootType
|
||||
}
|
||||
if raw.FilesCount != nil {
|
||||
f.FilesCount = *raw.FilesCount
|
||||
}
|
||||
if raw.FoldersCount != nil {
|
||||
f.FoldersCount = *raw.FoldersCount
|
||||
}
|
||||
if raw.Access != nil {
|
||||
f.Access = *raw.Access
|
||||
}
|
||||
if raw.Shared != nil {
|
||||
f.Shared = *raw.Shared
|
||||
}
|
||||
if raw.Updated != nil {
|
||||
f.Updated = *raw.Updated
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// UnmarshalJSON decodes a file row, capturing size and timestamps.
|
||||
func (f *DavFile) UnmarshalJSON(b []byte) error {
|
||||
var raw struct {
|
||||
ID *json.Number `json:"id"`
|
||||
Title *string `json:"title"`
|
||||
PureSize *int64 `json:"pureContentLength"`
|
||||
SizeStr *string `json:"contentLength"`
|
||||
Updated *string `json:"updated"`
|
||||
ViewURL *string `json:"viewUrl"`
|
||||
}
|
||||
if err := json.Unmarshal(b, &raw); err != nil {
|
||||
return err
|
||||
}
|
||||
if raw.ID != nil {
|
||||
f.ID = raw.ID.String()
|
||||
}
|
||||
if raw.Title != nil {
|
||||
f.Title = *raw.Title
|
||||
}
|
||||
if raw.PureSize != nil {
|
||||
f.Size = *raw.PureSize
|
||||
} else if raw.SizeStr != nil {
|
||||
if n, err := strconv.ParseInt(strings.Fields(*raw.SizeStr)[0], 10, 64); err == nil {
|
||||
f.Size = n
|
||||
}
|
||||
}
|
||||
if raw.Updated != nil {
|
||||
f.Updated = *raw.Updated
|
||||
}
|
||||
if raw.ViewURL != nil {
|
||||
f.ViewURL = *raw.ViewURL
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ModTime parses the folder's updated timestamp.
|
||||
func (f *DavFolder) ModTime() time.Time {
|
||||
t, _ := time.Parse("2006-01-02T15:04:05.0000000-07:00", f.Updated)
|
||||
return t
|
||||
}
|
||||
|
||||
// ModTime parses the file's updated timestamp.
|
||||
func (f *DavFile) ModTime() time.Time {
|
||||
t, _ := time.Parse("2006-01-02T15:04:05.0000000-07:00", f.Updated)
|
||||
return t
|
||||
}
|
||||
@@ -4,26 +4,33 @@ go 1.25.0
|
||||
|
||||
require (
|
||||
github.com/JohannesKaufmann/html-to-markdown/v2 v2.5.2
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.1
|
||||
github.com/charmbracelet/bubbles v0.18.0
|
||||
github.com/charmbracelet/bubbletea v0.25.0
|
||||
github.com/charmbracelet/glamour v0.8.0
|
||||
github.com/charmbracelet/lipgloss v0.12.1
|
||||
github.com/charmbracelet/x/ansi v0.1.4
|
||||
github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3
|
||||
github.com/eslider/go-hocr v0.2.2-0.20260827163626-8ff01582b002
|
||||
github.com/eslider/go-xls/v2 v2.1.0
|
||||
github.com/go-sql-driver/mysql v1.10.1
|
||||
github.com/google/go-querystring v1.2.0
|
||||
github.com/jackc/pgx/v5 v5.11.0
|
||||
github.com/joho/godotenv v1.5.1
|
||||
github.com/mattn/go-runewidth v0.0.15
|
||||
github.com/muesli/termenv v0.16.0
|
||||
github.com/spf13/cobra v1.10.2
|
||||
github.com/xuri/excelize/v2 v2.11.0
|
||||
gopkg.in/yaml.v3 v3.0.1
|
||||
modernc.org/sqlite v1.54.0
|
||||
modernc.org/sqlite v1.56.0
|
||||
)
|
||||
|
||||
require (
|
||||
filippo.io/edwards25519 v1.2.0 // indirect
|
||||
github.com/JohannesKaufmann/dom v0.3.1 // indirect
|
||||
github.com/alecthomas/chroma/v2 v2.14.0 // indirect
|
||||
github.com/atotto/clipboard v0.1.4 // indirect
|
||||
github.com/aws/smithy-go v1.24.0 // indirect
|
||||
github.com/aymanbagabas/go-osc52/v2 v2.0.1 // indirect
|
||||
github.com/aymerick/douceur v0.2.0 // indirect
|
||||
github.com/containerd/console v1.0.4-0.20230313162750-1ae8d489ac81 // indirect
|
||||
@@ -32,8 +39,12 @@ require (
|
||||
github.com/google/uuid v1.6.0 // indirect
|
||||
github.com/gorilla/css v1.0.1 // indirect
|
||||
github.com/inconshreveable/mousetrap v1.1.0 // indirect
|
||||
github.com/jackc/pgpassfile v1.0.0 // indirect
|
||||
github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect
|
||||
github.com/jackc/puddle/v2 v2.2.2 // indirect
|
||||
github.com/kr/text v0.2.0 // indirect
|
||||
github.com/lucasb-eyer/go-colorful v1.4.0 // indirect
|
||||
github.com/mattn/go-isatty v0.0.22 // indirect
|
||||
github.com/mattn/go-isatty v0.0.24 // indirect
|
||||
github.com/mattn/go-localereader v0.0.1 // indirect
|
||||
github.com/microcosm-cc/bluemonday v1.0.27 // indirect
|
||||
github.com/muesli/ansi v0.0.0-20211018074035-2e021307bc4b // indirect
|
||||
@@ -41,16 +52,23 @@ require (
|
||||
github.com/muesli/reflow v0.3.0 // indirect
|
||||
github.com/ncruces/go-strftime v1.0.0 // indirect
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
github.com/richardlehane/mscfb v1.0.7 // indirect
|
||||
github.com/richardlehane/msoleps v1.0.6 // indirect
|
||||
github.com/rivo/uniseg v0.4.7 // indirect
|
||||
github.com/rogpeppe/go-internal v1.16.0 // indirect
|
||||
github.com/spf13/pflag v1.0.9 // indirect
|
||||
github.com/tiendc/go-deepcopy v1.7.2 // indirect
|
||||
github.com/xuri/efp v0.0.1 // indirect
|
||||
github.com/xuri/nfp v0.0.2-0.20250530014748-2ddeb826f9a9 // indirect
|
||||
github.com/yuin/goldmark v1.8.2 // indirect
|
||||
github.com/yuin/goldmark-emoji v1.0.3 // indirect
|
||||
golang.org/x/net v0.55.0 // indirect
|
||||
golang.org/x/crypto v0.53.0 // indirect
|
||||
golang.org/x/net v0.56.0 // indirect
|
||||
golang.org/x/sync v0.21.0 // indirect
|
||||
golang.org/x/sys v0.46.0 // indirect
|
||||
golang.org/x/term v0.43.0 // indirect
|
||||
golang.org/x/text v0.37.0 // indirect
|
||||
modernc.org/libc v1.74.1 // indirect
|
||||
golang.org/x/sys v0.47.0 // indirect
|
||||
golang.org/x/term v0.44.0 // indirect
|
||||
golang.org/x/text v0.38.0 // indirect
|
||||
modernc.org/libc v1.74.4 // indirect
|
||||
modernc.org/mathutil v1.7.1 // indirect
|
||||
modernc.org/memory v1.11.0 // indirect
|
||||
)
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
filippo.io/edwards25519 v1.2.0 h1:crnVqOiS4jqYleHd9vaKZ+HKtHfllngJIiOpNpoJsjo=
|
||||
filippo.io/edwards25519 v1.2.0/go.mod h1:xzAOLCNug/yB62zG1bQ8uziwrIqIuxhctzJT18Q77mc=
|
||||
github.com/JohannesKaufmann/dom v0.3.1 h1:J16l9JAHWgkFPR3VIPbQ1gvS0cWab6laK1q7PFL3qh0=
|
||||
github.com/JohannesKaufmann/dom v0.3.1/go.mod h1:BZPkf8ZeYrBgABjwJn9iiKt8aiCtkxpHkevms+Yp2DE=
|
||||
github.com/JohannesKaufmann/html-to-markdown/v2 v2.5.2 h1:XFJZFWESIWlUEHHjzBuv8RvrtCWnSGlimEX17ysSDb8=
|
||||
@@ -10,6 +12,10 @@ github.com/alecthomas/repr v0.4.0 h1:GhI2A8MACjfegCPVq9f1FLvIBS+DrQ2KQBFZP1iFzXc
|
||||
github.com/alecthomas/repr v0.4.0/go.mod h1:Fr0507jx4eOXV7AlPV6AVZLYrLIuIeSOWtW57eE/O/4=
|
||||
github.com/atotto/clipboard v0.1.4 h1:EH0zSVneZPSuFR11BlR9YppQTVDbh5+16AmcJi4g1z4=
|
||||
github.com/atotto/clipboard v0.1.4/go.mod h1:ZY9tmq7sm5xIbd9bOK4onWV4S6X0u6GY7Vn0Yu86PYI=
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.1 h1:ABlyEARCDLN034NhxlRUSZr4l71mh+T5KAeGh6cerhU=
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.1/go.mod h1:MayyLB8y+buD9hZqkCW3kX1AKq07Y5pXxtgB+rRFhz0=
|
||||
github.com/aws/smithy-go v1.24.0 h1:LpilSUItNPFr1eY85RYgTIg5eIEPtvFbskaFcmmIUnk=
|
||||
github.com/aws/smithy-go v1.24.0/go.mod h1:LEj2LM3rBRQJxPZTB4KuzZkaZYnZPnvgIhb4pu07mx0=
|
||||
github.com/aymanbagabas/go-osc52/v2 v2.0.1 h1:HwpRHbFMcZLEVr42D4p7XBqjyuxQH5SMiErDT4WkJ2k=
|
||||
github.com/aymanbagabas/go-osc52/v2 v2.0.1/go.mod h1:uYgXzlJ7ZpABp8OJ+exZzJJhRNQ2ASbcXHWsFqH8hp8=
|
||||
github.com/aymanbagabas/go-udiff v0.2.0 h1:TK0fH4MteXUDspT88n8CKzvK0X9O2xu9yQjWpi6yML8=
|
||||
@@ -31,20 +37,28 @@ github.com/charmbracelet/x/exp/golden v0.0.0-20240715153702-9ba8adf781c4/go.mod
|
||||
github.com/containerd/console v1.0.4-0.20230313162750-1ae8d489ac81 h1:q2hJAaP1k2wIvVRd/hEHD7lacgqrCPS+k8g1MndzfWY=
|
||||
github.com/containerd/console v1.0.4-0.20230313162750-1ae8d489ac81/go.mod h1:YynlIjWYF8myEu6sdkwKIvGQq+cOckRm6So2avqoYAk=
|
||||
github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g=
|
||||
github.com/creack/pty v1.1.9/go.mod h1:oKZEueFk5CKHvIhNR5MUki03XCEU+Q6VDXinZuGJ33E=
|
||||
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
|
||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/dlclark/regexp2 v1.11.0 h1:G/nrcoOa7ZXlpoa/91N3X7mM3r8eIlMBBJZvsz/mxKI=
|
||||
github.com/dlclark/regexp2 v1.11.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8=
|
||||
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
|
||||
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
|
||||
github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3 h1:B9YK+Tck5mTccyDhtxBzWyqGYcFxLyB6+noMNW4/VgI=
|
||||
github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3/go.mod h1:HMJKR5wlh/ziNp+sHEDV2ltblO4JD2+IdDOWtGcQBTM=
|
||||
github.com/eslider/go-hocr v0.2.2-0.20260827163626-8ff01582b002 h1:LOFxQG4mxvlH7+lffbO2SIn2ThlClJlrHHfs1OqMNXs=
|
||||
github.com/eslider/go-hocr v0.2.2-0.20260827163626-8ff01582b002/go.mod h1:fIgfH/E1j3rU8du4X4+7mxTD0GPtPQibTzytgitdJWU=
|
||||
github.com/eslider/go-xls/v2 v2.1.0 h1:HszWKqYQbXxACmAXXWdMsfNl1NDBfGVBnJUPtyUHQ7A=
|
||||
github.com/eslider/go-xls/v2 v2.1.0/go.mod h1:xgxO6JrfuBr9jGUB+0z5l/yDmFFZ5diGk0ATGihxlMU=
|
||||
github.com/go-sql-driver/mysql v1.10.1 h1:arlSnNLq6a5yxGxV7qg9lF4j0C+KwD6NbQyKr9QL6ME=
|
||||
github.com/go-sql-driver/mysql v1.10.1/go.mod h1:M+cqaI7+xxXGG9swrdeUIoPG3Y3KCkF0pZej+SK+nWk=
|
||||
github.com/google/go-cmp v0.6.0 h1:ofyhxvXcZhMsU5ulbFiLKl/XBFqE1GSq7atu8tAmTRI=
|
||||
github.com/google/go-cmp v0.6.0/go.mod h1:17dUlkBOakJ0+DkrSSNjCkIjxS6bF9zb3elmeNGIjoY=
|
||||
github.com/google/go-querystring v1.2.0 h1:yhqkPbu2/OH+V9BfpCVPZkNmUXhb2gBxJArfhIxNtP0=
|
||||
github.com/google/go-querystring v1.2.0/go.mod h1:8IFJqpSRITyJ8QhQ13bmbeMBDfmeEJZD5A0egEOmkqU=
|
||||
github.com/google/pprof v0.0.0-20250317173921-a4b03ec1a45e h1:ijClszYn+mADRFY17kjQEVQ1XRhq2/JR1M3sGqeJoxs=
|
||||
github.com/google/pprof v0.0.0-20250317173921-a4b03ec1a45e/go.mod h1:boTsfXsheKC2y+lKOCMpSfarhxDeIzfZG1jqGcPl3cA=
|
||||
github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3 h1:LMLX+LgTNWpfvCBdFebv6EsYotImrt/Ppc5cXIriCSo=
|
||||
github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3/go.mod h1:jl5iWTm0/hd5PjEYEOuwAJ57L/CibdZfrqZ5XA5GrCk=
|
||||
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
|
||||
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
|
||||
github.com/gorilla/css v1.0.1 h1:ntNaBIghp6JmvWnxbZKANoLyuXTPZ4cAMlo6RyhlbO8=
|
||||
@@ -55,12 +69,24 @@ github.com/hexops/gotextdiff v1.0.3 h1:gitA9+qJrrTCsiCl7+kh75nPqQt1cx4ZkudSTLoUq
|
||||
github.com/hexops/gotextdiff v1.0.3/go.mod h1:pSWU5MAI3yDq+fZBTazCSJysOMbxWL1BSow5/V2vxeg=
|
||||
github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8=
|
||||
github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw=
|
||||
github.com/jackc/pgpassfile v1.0.0 h1:/6Hmqy13Ss2zCq62VdNG8tM1wchn8zjSGOBJ6icpsIM=
|
||||
github.com/jackc/pgpassfile v1.0.0/go.mod h1:CEx0iS5ambNFdcRtxPj5JhEz+xB6uRky5eyVu/W2HEg=
|
||||
github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 h1:iCEnooe7UlwOQYpKFhBabPMi4aNAfoODPEFNiAnClxo=
|
||||
github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761/go.mod h1:5TJZWKEWniPve33vlWYSoGYefn3gLQRzjfDlhSJ9ZKM=
|
||||
github.com/jackc/pgx/v5 v5.11.0 h1:IzBBtyK9AHqf98cctWFifYSci2hgQR/cd56wB4p+ogg=
|
||||
github.com/jackc/pgx/v5 v5.11.0/go.mod h1:mal1tBGAFfLHvZzaYh77YS/eC6IX9OWbRV1QIIM0Jn4=
|
||||
github.com/jackc/puddle/v2 v2.2.2 h1:PR8nw+E/1w0GLuRFSmiioY6UooMp6KJv0/61nB7icHo=
|
||||
github.com/jackc/puddle/v2 v2.2.2/go.mod h1:vriiEXHvEE654aYKXXjOvZM39qJ0q+azkZFrfEOc3H4=
|
||||
github.com/joho/godotenv v1.5.1 h1:7eLL/+HRGLY0ldzfGMeQkb7vMd0as4CfYvUVzLqw0N0=
|
||||
github.com/joho/godotenv v1.5.1/go.mod h1:f4LDr5Voq0i2e/R5DDNOoa2zzDfwtkZa6DnEwAbqwq4=
|
||||
github.com/kr/pretty v0.3.0 h1:WgNl7dwNpEZ6jJ9k1snq4pZsg7DOEN8hP9Xw0Tsjwk0=
|
||||
github.com/kr/pretty v0.3.0/go.mod h1:640gp4NfQd8pI5XOwp5fnNeVWj67G7CFk/SaSQn7NBk=
|
||||
github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY=
|
||||
github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE=
|
||||
github.com/lucasb-eyer/go-colorful v1.4.0 h1:UtrWVfLdarDgc44HcS7pYloGHJUjHV/4FwW4TvVgFr4=
|
||||
github.com/lucasb-eyer/go-colorful v1.4.0/go.mod h1:R4dSotOR9KMtayYi1e77YzuveK+i7ruzyGqttikkLy0=
|
||||
github.com/mattn/go-isatty v0.0.22 h1:j8l17JJ9i6VGPUFUYoTUKPSgKe/83EYU2zBC7YNKMw4=
|
||||
github.com/mattn/go-isatty v0.0.22/go.mod h1:ZXfXG4SQHsB/w3ZeOYbR0PrPwLy+n6xiMrJlRFqopa4=
|
||||
github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI=
|
||||
github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A=
|
||||
github.com/mattn/go-localereader v0.0.1 h1:ygSAOl7ZXTx4RdPYinUpg6W99U8jWvWi9Ye2JC/oIi4=
|
||||
github.com/mattn/go-localereader v0.0.1/go.mod h1:8fBrzywKY7BI3czFoHkuzRoWE9C+EiG4R1k4Cjx5p88=
|
||||
github.com/mattn/go-runewidth v0.0.12/go.mod h1:RAqKPSqVFrSLVXbA8x7dzmKdmGzieGRCM46jaSJTDAk=
|
||||
@@ -82,10 +108,16 @@ github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZb
|
||||
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
|
||||
github.com/richardlehane/mscfb v1.0.7 h1:oeoiM0WE79vHwE8RpIYYvIAc8ajTH2mb6UZm55/+EB0=
|
||||
github.com/richardlehane/mscfb v1.0.7/go.mod h1:pe0+IUIc0AHh0+teNzBlJCtSyZdFOGgV4ZK9bsoV+Jo=
|
||||
github.com/richardlehane/msoleps v1.0.6 h1:9BvkpjvD+iUBalUY4esMwv6uBkfOip/Lzvd93jvR9gg=
|
||||
github.com/richardlehane/msoleps v1.0.6/go.mod h1:BWev5JBpU9Ko2WAgmZEuiz4/u3ZYTKbjLycmwiWUfWg=
|
||||
github.com/rivo/uniseg v0.1.0/go.mod h1:J6wj4VEh+S6ZtnVlnTBMWIodfgj8LQOQFoIToxlJtxc=
|
||||
github.com/rivo/uniseg v0.2.0/go.mod h1:J6wj4VEh+S6ZtnVlnTBMWIodfgj8LQOQFoIToxlJtxc=
|
||||
github.com/rivo/uniseg v0.4.7 h1:WUdvkW8uEhrYfLC4ZzdpI2ztxP1I582+49Oc5Mq64VQ=
|
||||
github.com/rivo/uniseg v0.4.7/go.mod h1:FN3SvrM+Zdj16jyLfmOkMNblXMcoc8DfTHruCPUcx88=
|
||||
github.com/rogpeppe/go-internal v1.16.0 h1:O9DK+vNMDVGLr2BeZqmpLeMjiMNkuXfcqntWbZV6S5g=
|
||||
github.com/rogpeppe/go-internal v1.16.0/go.mod h1:DrUVZyrJU+txYW5/1kwtXQSMFio52ZOxX7yM1VHvnxs=
|
||||
github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM=
|
||||
github.com/sebdah/goldie/v2 v2.8.0 h1:dZb9wR8q5++oplmEiJT+U/5KyotVD+HNGCAc5gNr8rc=
|
||||
github.com/sebdah/goldie/v2 v2.8.0/go.mod h1:oZ9fp0+se1eapSRjfYbsV/0Hqhbuu3bJVvKI/NNtssI=
|
||||
@@ -95,33 +127,52 @@ github.com/spf13/cobra v1.10.2 h1:DMTTonx5m65Ic0GOoRY2c16WCbHxOOw6xxezuLaBpcU=
|
||||
github.com/spf13/cobra v1.10.2/go.mod h1:7C1pvHqHw5A4vrJfjNwvOdzYu0Gml16OCs2GRiTUUS4=
|
||||
github.com/spf13/pflag v1.0.9 h1:9exaQaMOCwffKiiiYk6/BndUBv+iRViNW+4lEMi0PvY=
|
||||
github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
|
||||
github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME=
|
||||
github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI=
|
||||
github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg=
|
||||
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
|
||||
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
|
||||
github.com/tiendc/go-deepcopy v1.7.2 h1:Ut2yYR7W9tWjTQitganoIue4UGxZwCcJy3orjrrIj44=
|
||||
github.com/tiendc/go-deepcopy v1.7.2/go.mod h1:4bKjNC2r7boYOkD2IOuZpYjmlDdzjbpTRyCx+goBCJQ=
|
||||
github.com/xuri/efp v0.0.1 h1:fws5Rv3myXyYni8uwj2qKjVaRP30PdjeYe2Y6FDsCL8=
|
||||
github.com/xuri/efp v0.0.1/go.mod h1:ybY/Jr0T0GTCnYjKqmdwxyxn2BQf2RcQIIvex5QldPI=
|
||||
github.com/xuri/excelize/v2 v2.11.0 h1:HxaEFl6sRN2+8J5a8HaKq+0M4FsjBGMnWWtjOCPSG88=
|
||||
github.com/xuri/excelize/v2 v2.11.0/go.mod h1:jxFLbzaIwGQ5ufFNvYfUOHqXhfPaNmP14KWfmNz2Uak=
|
||||
github.com/xuri/nfp v0.0.2-0.20250530014748-2ddeb826f9a9 h1:+C0TIdyyYmzadGaL/HBLbf3WdLgC29pgyhTjAT/0nuE=
|
||||
github.com/xuri/nfp v0.0.2-0.20250530014748-2ddeb826f9a9/go.mod h1:WwHg+CVyzlv/TX9xqBFXEZAuxOPxn2k1GNHwG41IIUQ=
|
||||
github.com/yuin/goldmark v1.7.1/go.mod h1:uzxRWxtg69N339t3louHJ7+O03ezfj6PlliRlaOzY1E=
|
||||
github.com/yuin/goldmark v1.8.2 h1:kEGpgqJXdgbkhcOgBxkC0X0PmoPG1ZyoZ117rDVp4zE=
|
||||
github.com/yuin/goldmark v1.8.2/go.mod h1:ip/1k0VRfGynBgxOz0yCqHrbZXhcjxyuS66Brc7iBKg=
|
||||
github.com/yuin/goldmark-emoji v1.0.3 h1:aLRkLHOuBR2czCY4R8olwMjID+tENfhyFDMCRhbIQY4=
|
||||
github.com/yuin/goldmark-emoji v1.0.3/go.mod h1:tTkZEbwu5wkPmgTcitqddVxY9osFZiavD+r4AzQrh1U=
|
||||
go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
|
||||
golang.org/x/crypto v0.53.0 h1:QZ4Muo8THX6CizN2vPPd5fBGHyogrdK9fG4wLPFUsto=
|
||||
golang.org/x/crypto v0.53.0/go.mod h1:DNLU434OwVakk9PzuwV8w62mAJpRJL3vsgcfp4Qnsio=
|
||||
golang.org/x/image v0.38.0 h1:5l+q+Y9JDC7mBOMjo4/aPhMDcxEptsX+Tt3GgRQRPuE=
|
||||
golang.org/x/image v0.38.0/go.mod h1:/3f6vaXC+6CEanU4KJxbcUZyEePbyKbaLoDOe4ehFYY=
|
||||
golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ=
|
||||
golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0=
|
||||
golang.org/x/net v0.55.0 h1:bcvxaJn3e1U6InsFWt1JUq1aSjnRxLzT2rtD2KfkDF8=
|
||||
golang.org/x/net v0.55.0/go.mod h1:L5U2KuzuOe1lY7Z+aWVIKK6qEeJXnXV9yzGA+WCHJww=
|
||||
golang.org/x/net v0.56.0 h1:Rw8j/hFzGvJUZwNBXnAtf5sVDVt+65SK2C7IxCxZt5o=
|
||||
golang.org/x/net v0.56.0/go.mod h1:D3Ku6r+V6JROoZK144D2XfMHFcMq/0zSfLelVTCFKec=
|
||||
golang.org/x/sync v0.21.0 h1:HLII4xRRTtCRkxYp4HNFF0Js/Og6q2i++KXbg0gHCwM=
|
||||
golang.org/x/sync v0.21.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sys v0.1.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.46.0 h1:noSf2Fq6F8DBgS+LysIkx7rIExoNHJsxOAtPp4rthXw=
|
||||
golang.org/x/sys v0.46.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/term v0.43.0 h1:S4RLU2sB31O/NCl+zFN9Aru9A/Cq2aqKpTZJ6B+DwT4=
|
||||
golang.org/x/term v0.43.0/go.mod h1:lrhlHNdQJHO+1qVYiHfFKVuVioJIheAc3fBSMFYEIsk=
|
||||
golang.org/x/text v0.37.0 h1:Cqjiwd9eSg8e0QAkyCaQTNHFIIzWtidPahFWR83rTrc=
|
||||
golang.org/x/text v0.37.0/go.mod h1:a5sjxXGs9hsn/AJVwuElvCAo9v8QYLzvavO5z2PiM38=
|
||||
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/term v0.44.0 h1:0rLvDRCtNj0gZkyIXhCyOb2OAzEhLVqc4B+hrsBhrmc=
|
||||
golang.org/x/term v0.44.0/go.mod h1:7ze4MdzUzLXpSAoFP1H0bOI9aXDqveSvatT5vKcFh2Y=
|
||||
golang.org/x/text v0.38.0 h1:sXmwo9DwP3OK9EZ7PqAdaooSGozfl/3a6/xJcbzPRhE=
|
||||
golang.org/x/text v0.38.0/go.mod h1:YXZt3QhHUKYT53r2lLKFIVi6Ao1jdzrTR/KQ09qyxF4=
|
||||
golang.org/x/tools v0.47.0 h1:7Kn5x/d1svx/PzryTsqeoZN4TZwqeH5pGWjefhLi/1Q=
|
||||
golang.org/x/tools v0.47.0/go.mod h1:dFHnyTvFWY212G+h7ZY4Vsp/K3U4/7W9TyVaAul8uCA=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk=
|
||||
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q=
|
||||
gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
||||
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||
modernc.org/cc/v4 v4.29.0 h1:CXgwL8cvxmyzBQZzbSl/6xFtMCryb6u8IOqDci39cgc=
|
||||
modernc.org/cc/v4 v4.29.0/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI=
|
||||
modernc.org/cc/v4 v4.29.1 h1:MKgdCV3WykTSPqpVrnxdEDS0HEd2FHpKZDzxzU5LyeI=
|
||||
modernc.org/cc/v4 v4.29.1/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI=
|
||||
modernc.org/ccgo/v4 v4.34.6 h1:sBgfIwyN0TQ9C5hwIeuqyeAKyMWnbvj2fvpF4L11uzU=
|
||||
modernc.org/ccgo/v4 v4.34.6/go.mod h1:SZ8YcN9NG7XVsQYdm6jYBvi8PQP1qi+kqB6OhjqI3Fk=
|
||||
modernc.org/fileutil v1.4.0 h1:j6ZzNTftVS054gi281TyLjHPp6CPHr2KCxEXjEbD6SM=
|
||||
@@ -132,8 +183,8 @@ modernc.org/gc/v3 v3.1.4 h1:2g65LGVSmFQrXeITAw97x7hCRvZFcyE1uDP+7Vng7JI=
|
||||
modernc.org/gc/v3 v3.1.4/go.mod h1:HFK/6AGESC7Ex+EZJhJ2Gni6cTaYpSMmU/cT9RmlfYY=
|
||||
modernc.org/goabi0 v0.2.0 h1:HvEowk7LxcPd0eq6mVOAEMai46V+i7Jrj13t4AzuNks=
|
||||
modernc.org/goabi0 v0.2.0/go.mod h1:CEFRnnJhKvWT1c1JTI3Avm+tgOWbkOu5oPA8eH8LnMI=
|
||||
modernc.org/libc v1.74.1 h1:bdR4VTKFMC4966QSNZ05XLGI/VwzVa2kTUX51Dm0riQ=
|
||||
modernc.org/libc v1.74.1/go.mod h1:uH4t5bOx3G3g9Xcmj10YKlTcVISlRDwv8VoQJG9n8Os=
|
||||
modernc.org/libc v1.74.4 h1:fX1Omw4o2/1C2iRkkIsrQTasJQldLhRmuPreXLoWs9k=
|
||||
modernc.org/libc v1.74.4/go.mod h1:eeQAS9W3sZeKYMFubydxJpII9ybHWshk+7or7bLG9co=
|
||||
modernc.org/mathutil v1.7.1 h1:GCZVGXdaN8gTqB1Mf/usp1Y/hSqgI2vAGGP4jZMCxOU=
|
||||
modernc.org/mathutil v1.7.1/go.mod h1:4p5IwJITfppl0G4sUEDtCr4DthTaT47/N3aT6MhfgJg=
|
||||
modernc.org/memory v1.11.0 h1:o4QC8aMQzmcwCK3t3Ux/ZHmwFPzE6hf2Y5LbkRs+hbI=
|
||||
@@ -142,8 +193,8 @@ modernc.org/opt v0.2.0 h1:tGyef5ApycA7FSEOMraay9SaTk5zmbx7Tu+cJs4QKZg=
|
||||
modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns=
|
||||
modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w=
|
||||
modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE=
|
||||
modernc.org/sqlite v1.54.0 h1:JCxR4qwkJvOaqAoYcgDoO25Nc+ROg6EJ2LfBVzdrgog=
|
||||
modernc.org/sqlite v1.54.0/go.mod h1:4ntCLuNmnH8+GNqjka1wNg7KJd5/Hi5FYp8K+XQ7GZw=
|
||||
modernc.org/sqlite v1.56.0 h1:/D8e2RfFqoy/Zc6PuC76U28zFwmI/sYx1Kjm4yEn9e0=
|
||||
modernc.org/sqlite v1.56.0/go.mod h1:yCJ2cmAaIkHQ25oXWrF8H4O1lIfPYPR26yCEDj2P3pQ=
|
||||
modernc.org/strutil v1.2.1 h1:UneZBkQA+DX2Rp35KcM69cSsNES9ly8mQWD71HKlOA0=
|
||||
modernc.org/strutil v1.2.1/go.mod h1:EHkiggD70koQxjVdSBM3JKM7k6L0FbGE5eymy9i3B9A=
|
||||
modernc.org/token v1.1.0 h1:Xl7Ap9dKaEs5kLoOQeQmPWevfnk/DM5qcLcYlA8ys6Y=
|
||||
|
||||
@@ -144,8 +144,13 @@ func unmarshalResponseObject(raw json.RawMessage) (map[string]any, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// getJSON issues an authenticated GET and returns the raw response body.
|
||||
// getJSON issues an authenticated GET and returns the raw response body,
|
||||
// retrying transient answers (see retryRaw).
|
||||
func (c *Client) getJSON(ctx context.Context, path string) (json.RawMessage, error) {
|
||||
return retryRaw(ctx, func() (json.RawMessage, error) { return c.getJSONOnce(ctx, path) })
|
||||
}
|
||||
|
||||
func (c *Client) getJSONOnce(ctx context.Context, path string) (json.RawMessage, error) {
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -181,7 +186,18 @@ func (c *Client) putForm(ctx context.Context, path string, fields url.Values) (j
|
||||
return c.formRequest(ctx, http.MethodPut, path, fields)
|
||||
}
|
||||
|
||||
// deleteForm issues an authenticated DELETE with application/x-www-form-urlencoded body.
|
||||
func (c *Client) deleteForm(ctx context.Context, path string, fields url.Values) (json.RawMessage, error) {
|
||||
return c.formRequest(ctx, http.MethodDelete, path, fields)
|
||||
}
|
||||
|
||||
func (c *Client) formRequest(ctx context.Context, method, path string, fields url.Values) (json.RawMessage, error) {
|
||||
return retryRaw(ctx, func() (json.RawMessage, error) {
|
||||
return c.formRequestOnce(ctx, method, path, fields)
|
||||
})
|
||||
}
|
||||
|
||||
func (c *Client) formRequestOnce(ctx context.Context, method, path string, fields url.Values) (json.RawMessage, error) {
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -210,6 +226,10 @@ func (c *Client) formRequest(ctx context.Context, method, path string, fields ur
|
||||
|
||||
// deleteReq issues an authenticated DELETE.
|
||||
func (c *Client) deleteReq(ctx context.Context, path string) (json.RawMessage, error) {
|
||||
return retryRaw(ctx, func() (json.RawMessage, error) { return c.deleteReqOnce(ctx, path) })
|
||||
}
|
||||
|
||||
func (c *Client) deleteReqOnce(ctx context.Context, path string) (json.RawMessage, error) {
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -244,8 +264,68 @@ func (c *Client) putJSONObject(ctx context.Context, path string, body any) (map[
|
||||
return unmarshalResponseObject(raw)
|
||||
}
|
||||
|
||||
// postJSONObject issues an authenticated POST with JSON body and decodes response.
|
||||
func (c *Client) postJSONObject(ctx context.Context, path string, body any) (map[string]any, error) {
|
||||
raw, err := c.postJSON(ctx, path, body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return unmarshalResponseObject(raw)
|
||||
}
|
||||
|
||||
// postJSON issues an authenticated POST with application/json body.
|
||||
func (c *Client) postJSON(ctx context.Context, path string, body any) (json.RawMessage, error) {
|
||||
return retryRaw(ctx, func() (json.RawMessage, error) { return c.postJSONOnce(ctx, path, body) })
|
||||
}
|
||||
|
||||
func (c *Client) postJSONOnce(ctx context.Context, path string, body any) (json.RawMessage, error) {
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var rdr io.Reader
|
||||
switch b := body.(type) {
|
||||
case nil:
|
||||
rdr = strings.NewReader("{}")
|
||||
case []byte:
|
||||
rdr = bytes.NewReader(b)
|
||||
case string:
|
||||
rdr = strings.NewReader(b)
|
||||
default:
|
||||
buf, err := json.Marshal(b)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
rdr = bytes.NewReader(buf)
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.baseURL()+path, rdr)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Authorization", auth)
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
req.Header.Set("Accept", "application/json")
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
raw, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("POST JSON %s: %d %s", path, resp.StatusCode, truncate(string(raw), 400))
|
||||
}
|
||||
return raw, nil
|
||||
}
|
||||
|
||||
// putJSON issues an authenticated PUT with application/json body.
|
||||
func (c *Client) putJSON(ctx context.Context, path string, body any) (json.RawMessage, error) {
|
||||
return retryRaw(ctx, func() (json.RawMessage, error) { return c.putJSONOnce(ctx, path, body) })
|
||||
}
|
||||
|
||||
func (c *Client) putJSONOnce(ctx context.Context, path string, body any) (json.RawMessage, error) {
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -289,6 +369,21 @@ func (c *Client) putJSON(ctx context.Context, path string, body any) (json.RawMe
|
||||
|
||||
// uploadMultipart posts a single file to path under the given form field name.
|
||||
func (c *Client) uploadMultipart(ctx context.Context, path, fieldName, filePath string) (json.RawMessage, error) {
|
||||
return c.uploadMultipartMethod(ctx, http.MethodPost, path, fieldName, filePath)
|
||||
}
|
||||
|
||||
// uploadMultipartMethod sends a single-file multipart request with the given
|
||||
// HTTP method. The OnlyOffice Documents API needs PUT for /update (a new
|
||||
// version) and POST for /upload (a new file); sending POST to /update answers
|
||||
// 500 on current servers. The file is re-opened per attempt, so transient
|
||||
// answers are retried like every other request.
|
||||
func (c *Client) uploadMultipartMethod(ctx context.Context, method, path, fieldName, filePath string) (json.RawMessage, error) {
|
||||
return retryRaw(ctx, func() (json.RawMessage, error) {
|
||||
return c.uploadMultipartOnce(ctx, method, path, fieldName, filePath)
|
||||
})
|
||||
}
|
||||
|
||||
func (c *Client) uploadMultipartOnce(ctx context.Context, method, path, fieldName, filePath string) (json.RawMessage, error) {
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -311,7 +406,7 @@ func (c *Client) uploadMultipart(ctx context.Context, path, fieldName, filePath
|
||||
if err := mw.Close(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.baseURL()+path, &buf)
|
||||
req, err := http.NewRequestWithContext(ctx, method, c.baseURL()+path, &buf)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -0,0 +1,422 @@
|
||||
// Package docpipe converts documents for OnlyOffice agent workflows:
|
||||
// Markdown ↔ DOCX (pandoc) and image/PDF OCR → searchable PDF + Markdown text.
|
||||
//
|
||||
// External tools (optional at runtime; helpers skip/error clearly when missing):
|
||||
// - pandoc — md↔docx
|
||||
// - ocrmypdf — OCR into a searchable PDF
|
||||
// - pdftotext — extract text layer
|
||||
// - pdfdetach — list/save embedded PDF attachments
|
||||
// - tesseract — OCR single images when ocrmypdf is unsuitable
|
||||
// - ghostscript (gs) — PDF rewrite/optimize via PostScript (pdfwrite)
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// DefaultMinTextChars: below this, a PDF is treated as needing OCR.
|
||||
const DefaultMinTextChars = 200
|
||||
|
||||
// Tools reports which converters are available on PATH.
|
||||
type Tools struct {
|
||||
Pandoc string
|
||||
OCRMyPDF string
|
||||
PDFToText string
|
||||
PDFDetach string
|
||||
Tesseract string
|
||||
Ghostscript string
|
||||
}
|
||||
|
||||
// LookPath resolves converter binaries (empty string if missing).
|
||||
func LookPath() Tools {
|
||||
find := func(names ...string) string {
|
||||
for _, n := range names {
|
||||
if p, err := exec.LookPath(n); err == nil {
|
||||
return p
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
return Tools{
|
||||
Pandoc: find("pandoc"),
|
||||
OCRMyPDF: find("ocrmypdf"),
|
||||
PDFToText: find("pdftotext"),
|
||||
PDFDetach: find("pdfdetach"),
|
||||
Tesseract: find("tesseract"),
|
||||
Ghostscript: find("gs", "ghostscript"),
|
||||
}
|
||||
}
|
||||
|
||||
func (t Tools) requirePandoc() error {
|
||||
if t.Pandoc == "" {
|
||||
return fmt.Errorf("pandoc not found on PATH (needed for md↔docx)")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Ext returns lower-case extension including dot (".pdf").
|
||||
func Ext(path string) string {
|
||||
return strings.ToLower(filepath.Ext(path))
|
||||
}
|
||||
|
||||
// ConvertFile converts between md and docx (and other pandoc formats) via pandoc.
|
||||
// outExt may be ".md", ".docx", or a full output path.
|
||||
func (t Tools) ConvertFile(inPath, outPath string) error {
|
||||
if err := t.requirePandoc(); err != nil {
|
||||
return err
|
||||
}
|
||||
if strings.TrimSpace(outPath) == "" {
|
||||
return fmt.Errorf("output path required")
|
||||
}
|
||||
cmd := exec.Command(t.Pandoc, inPath, "-o", outPath)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return fmt.Errorf("pandoc %s → %s: %w (%s)", inPath, outPath, err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// TXTToDOCX converts plain text to DOCX preserving line breaks (via markdown hard breaks).
|
||||
func (t Tools) TXTToDOCX(txtPath, docxPath string) error {
|
||||
if Ext(txtPath) != ".txt" {
|
||||
return fmt.Errorf("expected .txt input, got %q", txtPath)
|
||||
}
|
||||
if docxPath == "" {
|
||||
docxPath = strings.TrimSuffix(txtPath, Ext(txtPath)) + ".docx"
|
||||
}
|
||||
b, err := os.ReadFile(txtPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dir := filepath.Dir(docxPath)
|
||||
if dir == "" || dir == "." {
|
||||
dir = os.TempDir()
|
||||
}
|
||||
tmpMD := filepath.Join(dir, trimExt(filepath.Base(txtPath))+".txt2docx.md")
|
||||
md := TxtToMarkdown(string(b))
|
||||
if err := os.WriteFile(tmpMD, []byte(md), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.Remove(tmpMD)
|
||||
return t.MDToDOCX(tmpMD, docxPath)
|
||||
}
|
||||
|
||||
// TxtToMarkdown converts plain text to Markdown for DOCX output.
|
||||
// Prose text: each line is a hard break. Fixed-width extracts (INE, pdftotext -layout):
|
||||
// wrapped in a fenced code block (monospace, columns preserved).
|
||||
func TxtToMarkdown(content string) string {
|
||||
content = normalizeTxtNewlines(content)
|
||||
if isFixedWidthTxt(content) {
|
||||
return "```\n" + content + "\n```\n"
|
||||
}
|
||||
var b strings.Builder
|
||||
for _, line := range strings.Split(content, "\n") {
|
||||
if strings.TrimSpace(line) == "" {
|
||||
b.WriteByte('\n')
|
||||
continue
|
||||
}
|
||||
b.WriteString(line)
|
||||
b.WriteString(" \n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func normalizeTxtNewlines(content string) string {
|
||||
content = strings.ReplaceAll(content, "\r\n", "\n")
|
||||
return strings.ReplaceAll(content, "\r", "\n")
|
||||
}
|
||||
|
||||
// isFixedWidthTxt detects pdftotext -layout style extracts (many indented/spaced columns).
|
||||
func isFixedWidthTxt(content string) bool {
|
||||
lines := strings.Split(content, "\n")
|
||||
if len(lines) < 8 {
|
||||
return false
|
||||
}
|
||||
indented, long := 0, 0
|
||||
for _, line := range lines {
|
||||
if strings.TrimSpace(line) == "" {
|
||||
continue
|
||||
}
|
||||
if len(line) >= 72 {
|
||||
long++
|
||||
}
|
||||
if len(line) > 0 && (line[0] == ' ' || line[0] == '\t') {
|
||||
indented++
|
||||
}
|
||||
}
|
||||
n := len(lines)
|
||||
return indented*100/n >= 20 || (long >= 5 && indented*100/n >= 10)
|
||||
}
|
||||
|
||||
// MDToDOCX writes a DOCX next to or at outPath from a Markdown file.
|
||||
func (t Tools) MDToDOCX(mdPath, docxPath string) error {
|
||||
if Ext(mdPath) != ".md" && Ext(mdPath) != ".markdown" {
|
||||
return fmt.Errorf("expected markdown input, got %q", mdPath)
|
||||
}
|
||||
if docxPath == "" {
|
||||
docxPath = strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx"
|
||||
}
|
||||
return t.ConvertFile(mdPath, docxPath)
|
||||
}
|
||||
|
||||
// DOCXToMD writes Markdown from a DOCX file.
|
||||
func (t Tools) DOCXToMD(docxPath, mdPath string) error {
|
||||
if Ext(docxPath) != ".docx" {
|
||||
return fmt.Errorf("expected .docx input, got %q", docxPath)
|
||||
}
|
||||
if mdPath == "" {
|
||||
mdPath = strings.TrimSuffix(docxPath, Ext(docxPath)) + ".md"
|
||||
}
|
||||
return t.ConvertFile(docxPath, mdPath)
|
||||
}
|
||||
|
||||
// OptimizePDF rewrites a PDF through Ghostscript (PostScript pdfwrite).
|
||||
// Preserves native text layers; strips broken OCR overlays; shrinks for OO preview.
|
||||
// Use instead of ocrmypdf when pdftotext already extracts enough text.
|
||||
func (t Tools) OptimizePDF(inPath, outPath string) error {
|
||||
if t.Ghostscript == "" {
|
||||
return fmt.Errorf("ghostscript (gs) not found on PATH")
|
||||
}
|
||||
if outPath == "" {
|
||||
return fmt.Errorf("output PDF path required")
|
||||
}
|
||||
args := []string{
|
||||
"-sDEVICE=pdfwrite",
|
||||
"-dCompatibilityLevel=1.5",
|
||||
"-dNOPAUSE", "-dQUIET", "-dBATCH",
|
||||
"-dPDFSETTINGS=/ebook",
|
||||
"-dEmbedAllFonts=true",
|
||||
"-dSubsetFonts=true",
|
||||
"-dCompressFonts=true",
|
||||
"-dCompressPages=true",
|
||||
"-dDetectDuplicateImages=true",
|
||||
"-dAutoRotatePages=/None",
|
||||
"-sOutputFile=" + outPath,
|
||||
inPath,
|
||||
}
|
||||
cmd := exec.Command(t.Ghostscript, args...)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return fmt.Errorf("ghostscript pdfwrite: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// PDFTextLayerChars returns approximate extracted character count (0 if unavailable).
|
||||
func (t Tools) PDFTextLayerChars(pdfPath string) (int, error) {
|
||||
if t.PDFToText == "" {
|
||||
return 0, fmt.Errorf("pdftotext not found on PATH")
|
||||
}
|
||||
cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-")
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(bytes.TrimSpace(out)), nil
|
||||
}
|
||||
|
||||
// NeedsOCR reports whether path likely needs OCR before text extraction.
|
||||
func (t Tools) NeedsOCR(path string, minChars int) (bool, error) {
|
||||
if minChars <= 0 {
|
||||
minChars = DefaultMinTextChars
|
||||
}
|
||||
switch Ext(path) {
|
||||
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
|
||||
return true, nil
|
||||
case ".pdf":
|
||||
n, err := t.PDFTextLayerChars(path)
|
||||
if err != nil {
|
||||
// If we cannot measure, prefer OCR.
|
||||
return true, nil
|
||||
}
|
||||
return n < minChars, nil
|
||||
default:
|
||||
return false, nil
|
||||
}
|
||||
}
|
||||
|
||||
// OCRToPDF runs ocrmypdf into outPDF (searchable). Forces OCR when force is true.
|
||||
func (t Tools) OCRToPDF(inPath, outPDF string, force bool, lang string) error {
|
||||
if t.OCRMyPDF == "" {
|
||||
return fmt.Errorf("ocrmypdf not found on PATH")
|
||||
}
|
||||
if outPDF == "" {
|
||||
return fmt.Errorf("output PDF path required")
|
||||
}
|
||||
if lang == "" {
|
||||
lang = "eng"
|
||||
}
|
||||
args := []string{"-l", lang, "--skip-big", "100"}
|
||||
if force {
|
||||
args = append(args, "--force-ocr")
|
||||
} else {
|
||||
args = append(args, "--skip-text")
|
||||
}
|
||||
args = append(args, inPath, outPDF)
|
||||
cmd := exec.Command(t.OCRMyPDF, args...)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
// Retry with force if skip-text refused.
|
||||
if !force && strings.Contains(stderr.String(), "PriorOcrFoundError") == false {
|
||||
args2 := []string{"-l", lang, "--force-ocr", inPath, outPDF}
|
||||
cmd2 := exec.Command(t.OCRMyPDF, args2...)
|
||||
var stderr2 bytes.Buffer
|
||||
cmd2.Stderr = &stderr2
|
||||
if err2 := cmd2.Run(); err2 == nil {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
return fmt.Errorf("ocrmypdf: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ImageToText OCRs a raster image with tesseract (stdout text).
|
||||
func (t Tools) ImageToText(imgPath, lang string) (string, error) {
|
||||
if t.Tesseract == "" {
|
||||
return "", fmt.Errorf("tesseract not found on PATH")
|
||||
}
|
||||
if lang == "" {
|
||||
lang = "eng"
|
||||
}
|
||||
cmd := exec.Command(t.Tesseract, imgPath, "stdout", "-l", lang)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("tesseract: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return string(out), nil
|
||||
}
|
||||
|
||||
// ExtractPDFText returns layout text from a PDF via pdftotext.
|
||||
func (t Tools) ExtractPDFText(pdfPath string) (string, error) {
|
||||
if t.PDFToText == "" {
|
||||
return "", fmt.Errorf("pdftotext not found on PATH")
|
||||
}
|
||||
cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-")
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return string(out), nil
|
||||
}
|
||||
|
||||
// Result of ToMarkdown.
|
||||
type Result struct {
|
||||
Markdown string
|
||||
OCRPDFPath string // set when a searchable PDF was produced
|
||||
DidOCR bool
|
||||
Source string
|
||||
}
|
||||
|
||||
// ToMarkdown turns a local file into Markdown text.
|
||||
// PDFs/images with weak/no text layer are OCR'd to a searchable PDF first (when tools exist).
|
||||
func (t Tools) ToMarkdown(path string, workDir string, lang string, minChars int) (Result, error) {
|
||||
res := Result{Source: path}
|
||||
ext := Ext(path)
|
||||
switch ext {
|
||||
case ".md", ".markdown", ".txt":
|
||||
b, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = string(b)
|
||||
return res, nil
|
||||
case ".docx", ".odt", ".rtf", ".html", ".htm":
|
||||
if err := t.requirePandoc(); err != nil {
|
||||
return res, err
|
||||
}
|
||||
tmp := filepath.Join(workDir, "out.md")
|
||||
if err := t.ConvertFile(path, tmp); err != nil {
|
||||
return res, err
|
||||
}
|
||||
b, err := os.ReadFile(tmp)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = string(b)
|
||||
return res, nil
|
||||
case ".pdf":
|
||||
need, _ := t.NeedsOCR(path, minChars)
|
||||
pdf := path
|
||||
if need {
|
||||
if workDir == "" {
|
||||
workDir = os.TempDir()
|
||||
}
|
||||
outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf")
|
||||
if err := t.OCRToPDF(path, outPDF, true, lang); err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.DidOCR = true
|
||||
res.OCRPDFPath = outPDF
|
||||
pdf = outPDF
|
||||
}
|
||||
text, err := t.ExtractPDFText(pdf)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = wrapMD(filepath.Base(path), text)
|
||||
return res, nil
|
||||
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
|
||||
if workDir == "" {
|
||||
workDir = os.TempDir()
|
||||
}
|
||||
outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf")
|
||||
if t.OCRMyPDF != "" {
|
||||
if err := t.OCRToPDF(path, outPDF, true, lang); err == nil {
|
||||
res.DidOCR = true
|
||||
res.OCRPDFPath = outPDF
|
||||
text, err := t.ExtractPDFText(outPDF)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = wrapMD(filepath.Base(path), text)
|
||||
return res, nil
|
||||
}
|
||||
}
|
||||
text, err := t.ImageToText(path, lang)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.DidOCR = true
|
||||
res.Markdown = wrapMD(filepath.Base(path), text)
|
||||
return res, nil
|
||||
default:
|
||||
return res, fmt.Errorf("unsupported type %q for markdown extraction", ext)
|
||||
}
|
||||
}
|
||||
|
||||
func wrapMD(title, body string) string {
|
||||
body = strings.TrimSpace(body)
|
||||
if body == "" {
|
||||
return "# " + title + "\n\n_(empty text layer)_\n"
|
||||
}
|
||||
return "# " + title + "\n\n" + body + "\n"
|
||||
}
|
||||
|
||||
func trimExt(name string) string {
|
||||
return strings.TrimSuffix(name, filepath.Ext(name))
|
||||
}
|
||||
|
||||
// SiblingDOCX returns path with .docx extension replacing the original ext.
|
||||
func SiblingDOCX(mdPath string) string {
|
||||
return strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx"
|
||||
}
|
||||
|
||||
// EnsureDir creates parent directories for path.
|
||||
func EnsureDir(path string) error {
|
||||
dir := filepath.Dir(path)
|
||||
if dir == "" || dir == "." {
|
||||
return nil
|
||||
}
|
||||
return os.MkdirAll(dir, 0o755)
|
||||
}
|
||||
@@ -0,0 +1,167 @@
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestExt(t *testing.T) {
|
||||
if Ext("Foo.PDF") != ".pdf" {
|
||||
t.Fatalf("Ext: %q", Ext("Foo.PDF"))
|
||||
}
|
||||
}
|
||||
|
||||
func TestSiblingDOCX(t *testing.T) {
|
||||
if got := SiblingDOCX("notes.md"); got != "notes.docx" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWrapMD(t *testing.T) {
|
||||
s := wrapMD("a.pdf", " hello ")
|
||||
if !strings.HasPrefix(s, "# a.pdf\n") || !strings.Contains(s, "hello") {
|
||||
t.Fatalf("wrap: %q", s)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNeedsOCR_Image(t *testing.T) {
|
||||
tools := LookPath()
|
||||
need, err := tools.NeedsOCR("x.jpg", 0)
|
||||
if err != nil || !need {
|
||||
t.Fatalf("jpg should need OCR: need=%v err=%v", need, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTxtToMarkdown_FixedWidthUsesCodeBlock(t *testing.T) {
|
||||
var lines []string
|
||||
for i := 0; i < 12; i++ {
|
||||
lines = append(lines, " column layout line "+strings.Repeat("x", 40))
|
||||
}
|
||||
in := strings.Join(lines, "\n")
|
||||
md := TxtToMarkdown(in)
|
||||
if !strings.HasPrefix(md, "```\n") || !strings.Contains(md, "```") {
|
||||
t.Fatalf("expected code block: %q", md[:min(80, len(md))])
|
||||
}
|
||||
}
|
||||
|
||||
func min(a, b int) int {
|
||||
if a < b {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
func TestTxtToMarkdownPreservesLines(t *testing.T) {
|
||||
in := "line1\nline2\n\nline4"
|
||||
md := TxtToMarkdown(in)
|
||||
if !strings.Contains(md, "line1 \n") || !strings.Contains(md, "line2 \n") {
|
||||
t.Fatalf("hard breaks missing: %q", md)
|
||||
}
|
||||
if !strings.Contains(md, "line4 \n") {
|
||||
t.Fatalf("last line: %q", md)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTXTToDOCXPreservesLines(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.Pandoc == "" {
|
||||
t.Skip("pandoc not installed")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
txt := filepath.Join(dir, "sample.txt")
|
||||
docx := filepath.Join(dir, "sample.docx")
|
||||
body := "MyBox Auto — resumen\nNº contrato: 123\n\nEstado: Vigente\n"
|
||||
if err := os.WriteFile(txt, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := tools.TXTToDOCX(txt, docx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mdOut := filepath.Join(dir, "out.md")
|
||||
if err := tools.DOCXToMD(docx, mdOut); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := os.ReadFile(mdOut)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s := string(got)
|
||||
for _, want := range []string{"MyBox Auto", "Nº contrato", "Estado: Vigente"} {
|
||||
if !strings.Contains(s, want) {
|
||||
t.Fatalf("missing %q in %q", want, s)
|
||||
}
|
||||
}
|
||||
if strings.Contains(s, "MyBox Auto — resumen Nº") {
|
||||
t.Fatalf("lines collapsed: %q", s)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMDDocxRoundTrip(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.Pandoc == "" {
|
||||
t.Skip("pandoc not installed")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
md := filepath.Join(dir, "n.md")
|
||||
docx := filepath.Join(dir, "n.docx")
|
||||
md2 := filepath.Join(dir, "n2.md")
|
||||
if err := os.WriteFile(md, []byte("# Title\n\nHello **world**.\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := tools.MDToDOCX(md, docx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := os.Stat(docx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := tools.DOCXToMD(docx, md2); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
b, err := os.ReadFile(md2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(string(b), "Hello") {
|
||||
t.Fatalf("round-trip missing Hello: %s", b)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOptimizePDF(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.Ghostscript == "" || tools.PDFToText == "" {
|
||||
t.Skip("ghostscript/pdftotext not installed")
|
||||
}
|
||||
in := "/tmp/ccgg-original.pdf"
|
||||
if _, err := os.Stat(in); err != nil {
|
||||
t.Skip("local fixture not present")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
out := filepath.Join(dir, "out.pdf")
|
||||
charsIn, err := tools.PDFTextLayerChars(in)
|
||||
if err != nil || charsIn < 1000 {
|
||||
t.Skip("fixture has no text layer")
|
||||
}
|
||||
if err := tools.OptimizePDF(in, out); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
charsOut, err := tools.PDFTextLayerChars(out)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if charsOut < charsIn/2 {
|
||||
t.Fatalf("text layer lost: in=%d out=%d", charsIn, charsOut)
|
||||
}
|
||||
}
|
||||
|
||||
func TestToMarkdown_PlainMD(t *testing.T) {
|
||||
tools := LookPath()
|
||||
dir := t.TempDir()
|
||||
p := filepath.Join(dir, "a.md")
|
||||
_ = os.WriteFile(p, []byte("hi"), 0o644)
|
||||
res, err := tools.ToMarkdown(p, dir, "eng", 0)
|
||||
if err != nil || res.Markdown != "hi" {
|
||||
t.Fatalf("got %+v err=%v", res, err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,153 @@
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
hocr "github.com/eslider/go-hocr"
|
||||
)
|
||||
|
||||
// HOCRResult is structured OCR output for agents.
|
||||
type HOCRResult struct {
|
||||
HOCRPath string
|
||||
Markdown string
|
||||
YAML string
|
||||
DidOCR bool
|
||||
Source string
|
||||
}
|
||||
|
||||
// ImageToHOCR runs tesseract hOCR into outBase+".hocr" (tesseract adds the extension).
|
||||
// outBase must not include ".hocr". dpi 0 uses tesseract default; phone photos often need 200–300.
|
||||
func (t Tools) ImageToHOCR(imgPath, outBase, lang string, dpi int) (string, error) {
|
||||
if t.Tesseract == "" {
|
||||
return "", fmt.Errorf("tesseract not found on PATH")
|
||||
}
|
||||
if lang == "" {
|
||||
lang = "eng"
|
||||
}
|
||||
if outBase == "" {
|
||||
return "", fmt.Errorf("hOCR output base path required")
|
||||
}
|
||||
args := []string{imgPath, outBase, "-l", lang}
|
||||
if dpi > 0 {
|
||||
args = append(args, "--dpi", fmt.Sprintf("%d", dpi))
|
||||
}
|
||||
args = append(args, resolveHOCRConfig())
|
||||
cmd := exec.Command(t.Tesseract, args...)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return "", fmt.Errorf("tesseract hocr: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
out := outBase + ".hocr"
|
||||
if _, err := os.Stat(out); err != nil {
|
||||
// Some builds write .html
|
||||
alt := outBase + ".html"
|
||||
if _, err2 := os.Stat(alt); err2 == nil {
|
||||
return alt, nil
|
||||
}
|
||||
return "", fmt.Errorf("tesseract hocr: missing output %s (%s)", out, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// resolveHOCRConfig returns a tesseract config path that works when TESSDATA_PREFIX
|
||||
// points at a custom traineddata dir without relative "hocr" configs.
|
||||
func resolveHOCRConfig() string {
|
||||
var candidates []string
|
||||
if p := strings.TrimSpace(os.Getenv("TESSDATA_PREFIX")); p != "" {
|
||||
candidates = append(candidates,
|
||||
filepath.Join(p, "configs", "hocr"),
|
||||
filepath.Join(p, "tessdata", "configs", "hocr"),
|
||||
)
|
||||
}
|
||||
candidates = append(candidates,
|
||||
"/usr/share/tesseract-ocr/5/tessdata/configs/hocr",
|
||||
"/usr/share/tesseract-ocr/4.00/tessdata/configs/hocr",
|
||||
"/usr/share/tessdata/configs/hocr",
|
||||
)
|
||||
for _, c := range candidates {
|
||||
if _, err := os.Stat(c); err == nil {
|
||||
return c
|
||||
}
|
||||
}
|
||||
return "hocr"
|
||||
}
|
||||
|
||||
// HOCRToMarkdown parses an hOCR file via go-hocr and returns Markdown (+ optional YAML).
|
||||
// Words with confidence in (0, minConf) are dropped; minConf 0 keeps all.
|
||||
func HOCRToMarkdown(hocrPath string, minConf float32) (md string, yml string, err error) {
|
||||
doc, err := hocr.ReadFile(hocrPath)
|
||||
if err != nil {
|
||||
return "", "", fmt.Errorf("go-hocr: %w", err)
|
||||
}
|
||||
md = doc.ToMarkdown(minConf)
|
||||
yml, err = doc.ToYaml()
|
||||
if err != nil {
|
||||
return md, "", err
|
||||
}
|
||||
return md, yml, nil
|
||||
}
|
||||
|
||||
// ToHOCRMarkdown OCRs an image (or rasterizes first PDF page) to hOCR → Markdown/YAML.
|
||||
func (t Tools) ToHOCRMarkdown(path, workDir, lang string, dpi int, minConf float32) (HOCRResult, error) {
|
||||
res := HOCRResult{Source: path}
|
||||
if workDir == "" {
|
||||
workDir = os.TempDir()
|
||||
}
|
||||
ext := Ext(path)
|
||||
img := path
|
||||
switch ext {
|
||||
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
|
||||
// ok
|
||||
case ".pdf":
|
||||
raster, err := t.pdfFirstPagePNG(path, workDir)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
img = raster
|
||||
if dpi == 0 {
|
||||
dpi = 300
|
||||
}
|
||||
default:
|
||||
return res, fmt.Errorf("hOCR path expects image or PDF, got %q", ext)
|
||||
}
|
||||
base := filepath.Join(workDir, trimExt(filepath.Base(path))+".hocr-out")
|
||||
hocrPath, err := t.ImageToHOCR(img, base, lang, dpi)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.DidOCR = true
|
||||
res.HOCRPath = hocrPath
|
||||
md, yml, err := HOCRToMarkdown(hocrPath, minConf)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = wrapMD(filepath.Base(path), strings.TrimSpace(md))
|
||||
res.YAML = yml
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// pdfFirstPagePNG uses pdftoppm when available.
|
||||
func (t Tools) pdfFirstPagePNG(pdfPath, workDir string) (string, error) {
|
||||
pdftoppm, err := exec.LookPath("pdftoppm")
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("pdftoppm not found (needed to rasterize PDF for hOCR)")
|
||||
}
|
||||
outBase := filepath.Join(workDir, trimExt(filepath.Base(pdfPath))+".page")
|
||||
cmd := exec.Command(pdftoppm, "-png", "-f", "1", "-singlefile", "-r", "200", pdfPath, outBase)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return "", fmt.Errorf("pdftoppm: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
png := outBase + ".png"
|
||||
if _, err := os.Stat(png); err != nil {
|
||||
return "", fmt.Errorf("pdftoppm: missing %s", png)
|
||||
}
|
||||
return png, nil
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestHOCRToMarkdown_Fixture(t *testing.T) {
|
||||
// Minimal hOCR 1.2 snippet
|
||||
hocrXML := `<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||
<head>
|
||||
<title>tesseract</title>
|
||||
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
||||
<meta name="ocr-system" content="tesseract 5"/>
|
||||
<meta name="ocr-capabilities" content="ocr_page ocr_carea ocr_par ocr_line ocrx_word"/>
|
||||
</head>
|
||||
<body>
|
||||
<div class="ocr_page" id="page_1" title="image "x.png"; bbox 0 0 100 50; ppageno 0">
|
||||
<div class="ocr_carea" id="block_1_1" title="bbox 0 0 100 50">
|
||||
<p class="ocr_par" id="par_1_1" lang="eng" title="bbox 0 0 100 50">
|
||||
<span class="ocr_line" id="line_1_1" title="bbox 0 0 100 20; baseline 0 0; x_size 20">
|
||||
<span class="ocrx_word" id="word_1_1" title="bbox 0 0 40 20; x_wconf 96">Hello</span>
|
||||
<span class="ocrx_word" id="word_1_2" title="bbox 45 0 100 20; x_wconf 92">world</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>`
|
||||
dir := t.TempDir()
|
||||
p := filepath.Join(dir, "sample.hocr")
|
||||
if err := os.WriteFile(p, []byte(hocrXML), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
md, yml, err := HOCRToMarkdown(p, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(md, "Hello") || !strings.Contains(md, "world") {
|
||||
t.Fatalf("md=%q", md)
|
||||
}
|
||||
if !strings.Contains(yml, "Hello") {
|
||||
t.Fatalf("yaml missing word: %s", yml)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,278 @@
|
||||
package docpipe
|
||||
|
||||
// Embedded PDF attachments (F6 #42). Digitised invoices often carry the
|
||||
// original scan as a PDF attachment; the searchable body may hold only a
|
||||
// summary. pdfdetach (poppler) lists/saves them; each saved attachment is run
|
||||
// through the normal docpipe extraction (pdftotext/OCR).
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/xml"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"unicode"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
// PDFAttachment is one embedded file in a PDF.
|
||||
type PDFAttachment struct {
|
||||
Index int // 1-based number as `pdfdetach -list` reports it
|
||||
Name string // embedded file name
|
||||
}
|
||||
|
||||
// AttachmentText is the extracted text of one embedded attachment.
|
||||
type AttachmentText struct {
|
||||
Name string
|
||||
Text string
|
||||
}
|
||||
|
||||
// parseAttachmentList parses `pdfdetach -list` output. The first line is a
|
||||
// count ("N embedded files"); every following line is "<index>: <name>".
|
||||
// Pure, so it is unit-tested.
|
||||
func parseAttachmentList(out string) []PDFAttachment {
|
||||
var atts []PDFAttachment
|
||||
for _, line := range strings.Split(out, "\n") {
|
||||
line = strings.TrimSpace(line)
|
||||
if line == "" {
|
||||
continue
|
||||
}
|
||||
colon := strings.Index(line, ":")
|
||||
if colon <= 0 {
|
||||
continue
|
||||
}
|
||||
n, err := strconv.Atoi(strings.TrimSpace(line[:colon]))
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
name := strings.TrimSpace(line[colon+1:])
|
||||
if name == "" {
|
||||
continue
|
||||
}
|
||||
atts = append(atts, PDFAttachment{Index: n, Name: name})
|
||||
}
|
||||
return atts
|
||||
}
|
||||
|
||||
// safeAttachmentName strips directories and leading dots so a hostile
|
||||
// attachment name cannot escape the extraction directory.
|
||||
func safeAttachmentName(name string) string {
|
||||
name = strings.ReplaceAll(strings.TrimSpace(name), "\\", "/")
|
||||
name = filepath.Base(name)
|
||||
name = strings.TrimLeft(name, ".")
|
||||
if name == "" || name == "." || name == "/" {
|
||||
return ""
|
||||
}
|
||||
return name
|
||||
}
|
||||
|
||||
// JoinWithAttachments appends attachment text to the document body, each
|
||||
// section preceded by an "[attachment: <name>]" marker so a search hit shows
|
||||
// its source. Empty attachments are skipped. Pure, so it is unit-tested.
|
||||
func JoinWithAttachments(body string, atts []AttachmentText) string {
|
||||
var b strings.Builder
|
||||
b.WriteString(strings.TrimRight(body, "\n"))
|
||||
for _, a := range atts {
|
||||
text := strings.TrimSpace(a.Text)
|
||||
if text == "" {
|
||||
continue
|
||||
}
|
||||
b.WriteString("\n\n[attachment: ")
|
||||
b.WriteString(a.Name)
|
||||
b.WriteString("]\n\n")
|
||||
b.WriteString(text)
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// ListAttachments returns the embedded files of a PDF. A PDF without
|
||||
// attachments yields an empty slice and no error.
|
||||
func (t Tools) ListAttachments(pdfPath string) ([]PDFAttachment, error) {
|
||||
if t.PDFDetach == "" {
|
||||
return nil, fmt.Errorf("pdfdetach not found on PATH")
|
||||
}
|
||||
cmd := exec.Command(t.PDFDetach, "-list", pdfPath)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("pdfdetach -list %s: %w (%s)", filepath.Base(pdfPath), err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return parseAttachmentList(string(out)), nil
|
||||
}
|
||||
|
||||
// SaveAttachment writes the n-th embedded file (1-based) to outPath.
|
||||
func (t Tools) SaveAttachment(pdfPath string, index int, outPath string) error {
|
||||
if t.PDFDetach == "" {
|
||||
return fmt.Errorf("pdfdetach not found on PATH")
|
||||
}
|
||||
if strings.TrimSpace(outPath) == "" {
|
||||
return fmt.Errorf("output path required")
|
||||
}
|
||||
if err := EnsureDir(outPath); err != nil {
|
||||
return err
|
||||
}
|
||||
cmd := exec.Command(t.PDFDetach, "-save", strconv.Itoa(index), "-o", outPath, pdfPath)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return fmt.Errorf("pdfdetach -save %d: %w (%s)", index, err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ToMarkdownWithAttachments extracts the file as ToMarkdown does, then — for
|
||||
// PDFs — appends the text of every embedded attachment under an
|
||||
// "[attachment: <name>]" marker. Attachment failures are non-fatal: the body
|
||||
// is returned unchanged.
|
||||
func (t Tools) ToMarkdownWithAttachments(path, workDir, lang string, minChars int) (string, error) {
|
||||
res, err := t.ToMarkdown(path, workDir, lang, minChars)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
if Ext(path) != ".pdf" {
|
||||
return res.Markdown, nil
|
||||
}
|
||||
atts, err := t.attachmentTexts(path, workDir, lang, minChars)
|
||||
if err != nil {
|
||||
return res.Markdown, nil
|
||||
}
|
||||
return JoinWithAttachments(res.Markdown, atts), nil
|
||||
}
|
||||
|
||||
// attachmentTexts saves and extracts every embedded attachment, skipping the
|
||||
// ones that cannot be read. It returns an error only when the attachment list
|
||||
// itself cannot be obtained.
|
||||
func (t Tools) attachmentTexts(pdfPath, workDir, lang string, minChars int) ([]AttachmentText, error) {
|
||||
list, err := t.ListAttachments(pdfPath)
|
||||
if err != nil || len(list) == 0 {
|
||||
return nil, err
|
||||
}
|
||||
if workDir == "" {
|
||||
workDir = os.TempDir()
|
||||
}
|
||||
dir := filepath.Join(workDir, "att-"+trimExt(filepath.Base(pdfPath)))
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
out := make([]AttachmentText, 0, len(list))
|
||||
for _, a := range list {
|
||||
name := safeAttachmentName(a.Name)
|
||||
if name == "" {
|
||||
continue
|
||||
}
|
||||
saved := filepath.Join(dir, fmt.Sprintf("%d-%s", a.Index, name))
|
||||
if err := t.SaveAttachment(pdfPath, a.Index, saved); err != nil {
|
||||
continue
|
||||
}
|
||||
text, err := t.attachmentMarkdown(saved, dir, lang, minChars)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
out = append(out, AttachmentText{Name: a.Name, Text: text})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// attachmentMarkdown extracts a saved attachment with the regular pipeline.
|
||||
// Structured attachments that docpipe does not convert (e-invoice XML,
|
||||
// CuraSoft JSON, CSV/HTML) fall back to their text content, so the embedded
|
||||
// original is still searchable. Other unreadable formats return an error and
|
||||
// the caller skips them.
|
||||
func (t Tools) attachmentMarkdown(path, workDir, lang string, minChars int) (string, error) {
|
||||
if res, err := t.ToMarkdown(path, workDir, lang, minChars); err == nil {
|
||||
return res.Markdown, nil
|
||||
}
|
||||
switch Ext(path) {
|
||||
case ".xml", ".html", ".htm":
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return xmlToText(raw), nil
|
||||
case ".json", ".csv", ".yaml", ".yml", ".toml", ".txt", ".md", ".markdown":
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return string(raw), nil
|
||||
default:
|
||||
return textFallback(path)
|
||||
}
|
||||
}
|
||||
|
||||
// textFallback reads an attachment of an unknown or missing extension as plain
|
||||
// text when it looks textual (valid UTF-8, mostly printable runes). Binary
|
||||
// payloads (images, archives, NUL-padded blobs) are rejected with an error so
|
||||
// the caller skips them instead of poisoning the index. Classified digitised
|
||||
// PDFs (Scanner-*.ocr.pdf) carry .yaml/.md attachments; some exporters omit the
|
||||
// extension, which this covers.
|
||||
func textFallback(path string) (string, error) {
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer f.Close()
|
||||
raw, err := io.ReadAll(io.LimitReader(f, 1<<20))
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
if !utf8.Valid(raw) {
|
||||
return "", fmt.Errorf("unsupported attachment type %q", Ext(path))
|
||||
}
|
||||
if !mostlyPrintable(raw) {
|
||||
return "", fmt.Errorf("unsupported attachment type %q", Ext(path))
|
||||
}
|
||||
return string(raw), nil
|
||||
}
|
||||
|
||||
// mostlyPrintable reports whether at least 90% of the runes are printable text
|
||||
// (newlines, carriage returns and tabs count as text). Pure, so it is tested.
|
||||
func mostlyPrintable(b []byte) bool {
|
||||
if len(b) == 0 {
|
||||
return false
|
||||
}
|
||||
printable, total := 0, 0
|
||||
for _, r := range string(b) {
|
||||
if r == utf8.RuneError {
|
||||
continue
|
||||
}
|
||||
total++
|
||||
if unicode.IsPrint(r) || r == '\n' || r == '\r' || r == '\t' {
|
||||
printable++
|
||||
}
|
||||
}
|
||||
return total > 0 && printable*10 >= total*9
|
||||
}
|
||||
|
||||
// xmlToText returns the character data of an XML/HTML document: element text
|
||||
// values with decoded entities, one per line. Used for invoice XML (EN 16931
|
||||
// CII / ZUGFeRD) and HTML attachments. Pure, so it is unit-tested.
|
||||
func xmlToText(raw []byte) string {
|
||||
dec := xml.NewDecoder(bytes.NewReader(raw))
|
||||
dec.Strict = false
|
||||
var b strings.Builder
|
||||
for {
|
||||
tok, err := dec.Token()
|
||||
if err != nil {
|
||||
break
|
||||
}
|
||||
cd, ok := tok.(xml.CharData)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
s := strings.TrimSpace(string(cd))
|
||||
if s == "" {
|
||||
continue
|
||||
}
|
||||
b.WriteString(s)
|
||||
b.WriteByte('\n')
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
@@ -0,0 +1,189 @@
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestParseAttachmentList(t *testing.T) {
|
||||
out := "2 embedded files\n1: original.pdf\n2: scan_001.png\n"
|
||||
got := parseAttachmentList(out)
|
||||
want := []PDFAttachment{{Index: 1, Name: "original.pdf"}, {Index: 2, Name: "scan_001.png"}}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("got %+v, want %+v", got, want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("att[%d] = %+v, want %+v", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseAttachmentListEmptyAndMalformed(t *testing.T) {
|
||||
for _, in := range []string{"", "0 embedded files\n", "garbage\n\n \n"} {
|
||||
if got := parseAttachmentList(in); len(got) != 0 {
|
||||
t.Errorf("parseAttachmentList(%q) = %+v, want empty", in, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSafeAttachmentName(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"note.txt": "note.txt",
|
||||
"../../evil.pdf": "evil.pdf",
|
||||
`..\..\evil.pdf`: "evil.pdf",
|
||||
"/abs/scan_001.pdf": "scan_001.pdf",
|
||||
".hidden": "hidden",
|
||||
" spaced name.txt ": "spaced name.txt",
|
||||
"..": "",
|
||||
"": "",
|
||||
}
|
||||
for in, want := range cases {
|
||||
if got := safeAttachmentName(in); got != want {
|
||||
t.Errorf("safeAttachmentName(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestJoinWithAttachments(t *testing.T) {
|
||||
body := "# scan.pdf\n\nbody token\n"
|
||||
atts := []AttachmentText{
|
||||
{Name: "original.pdf", Text: " original token "},
|
||||
{Name: "empty.txt", Text: " "},
|
||||
}
|
||||
got := JoinWithAttachments(body, atts)
|
||||
if !strings.Contains(got, "body token") {
|
||||
t.Errorf("body text lost: %q", got)
|
||||
}
|
||||
if !strings.Contains(got, "[attachment: original.pdf]") {
|
||||
t.Errorf("marker missing: %q", got)
|
||||
}
|
||||
if !strings.Contains(got, "original token") {
|
||||
t.Errorf("attachment text missing: %q", got)
|
||||
}
|
||||
if strings.Contains(got, "empty.txt") {
|
||||
t.Errorf("empty attachment must be skipped: %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestJoinWithAttachmentsNoAttachments(t *testing.T) {
|
||||
got := JoinWithAttachments("# a.pdf\n\ntext\n\n", nil)
|
||||
if got != "# a.pdf\n\ntext" {
|
||||
t.Errorf("got %q, want trimmed body only", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestListAttachmentsWithoutTool(t *testing.T) {
|
||||
if _, err := (Tools{}).ListAttachments("x.pdf"); err == nil || !strings.Contains(err.Error(), "pdfdetach") {
|
||||
t.Fatalf("want pdfdetach error, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestToMarkdownWithAttachmentsFixture exercises the real pdfdetach + pdftotext
|
||||
// pipeline on testdata/pdf-with-attachment.pdf (body token + embedded
|
||||
// goo-note.txt). Skips when poppler is not installed.
|
||||
func TestToMarkdownWithAttachmentsFixture(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.PDFDetach == "" || tools.PDFToText == "" {
|
||||
t.Skip("pdfdetach/pdftotext not on PATH — skipping attachment extraction test")
|
||||
}
|
||||
fixture := filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf")
|
||||
got, err := tools.ToMarkdownWithAttachments(fixture, t.TempDir(), "eng", 1)
|
||||
if err != nil {
|
||||
t.Fatalf("ToMarkdownWithAttachments: %v", err)
|
||||
}
|
||||
for _, want := range []string{"goobodytoken", "[attachment: goo-note.txt]", "gooattachmenttoken"} {
|
||||
if !strings.Contains(got, want) {
|
||||
t.Errorf("result missing %q:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestXMLToText(t *testing.T) {
|
||||
raw := []byte(`<?xml version="1.0" encoding="UTF-8"?>
|
||||
<rsm:CrossIndustryInvoice><rsm:ExchangedDocument>
|
||||
<ram:ID>S1063</ram:ID></rsm:ExchangedDocument>
|
||||
<ram:Name>Edelweiss & Co</ram:Name><ram:GrandTotalAmount>42.00</ram:GrandTotalAmount>
|
||||
</rsm:CrossIndustryInvoice>`)
|
||||
got := xmlToText(raw)
|
||||
for _, want := range []string{"S1063", "Edelweiss & Co", "42.00"} {
|
||||
if !strings.Contains(got, want) {
|
||||
t.Errorf("xmlToText missing %q:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
if strings.ContainsAny(got, "<>") {
|
||||
t.Errorf("xmlToText left markup: %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAttachmentMarkdownFallback verifies structured attachments that docpipe
|
||||
// cannot convert are still reduced to searchable text, and unknown binary
|
||||
// formats error (so the caller skips them).
|
||||
func TestAttachmentMarkdownFallback(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
xmlPath := filepath.Join(dir, "factur-x.xml")
|
||||
if err := os.WriteFile(xmlPath, []byte(`<Invoice><Number>S1063</Number></Invoice>`), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := (Tools{}).attachmentMarkdown(xmlPath, dir, "", 0)
|
||||
if err != nil {
|
||||
t.Fatalf("attachmentMarkdown(xml): %v", err)
|
||||
}
|
||||
if !strings.Contains(got, "S1063") {
|
||||
t.Errorf("xml attachment text = %q, want S1063", got)
|
||||
}
|
||||
|
||||
// Classified digitised PDFs (Scanner-*.ocr.pdf) carry .yaml metadata.
|
||||
yamlPath := filepath.Join(dir, "Scanner-123-003.ocr.yaml")
|
||||
if err := os.WriteFile(yamlPath, []byte("document:\n type: Rechnung\nnumber: S1063\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
gotYAML, err := (Tools{}).attachmentMarkdown(yamlPath, dir, "", 0)
|
||||
if err != nil {
|
||||
t.Fatalf("attachmentMarkdown(yaml): %v", err)
|
||||
}
|
||||
if !strings.Contains(gotYAML, "S1063") {
|
||||
t.Errorf("yaml attachment text = %q, want S1063", gotYAML)
|
||||
}
|
||||
|
||||
// Extensionless textual attachment falls back to raw text.
|
||||
noExt := filepath.Join(dir, "attachment")
|
||||
if err := os.WriteFile(noExt, []byte("plain attachment token goonoext"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
gotNoExt, err := (Tools{}).attachmentMarkdown(noExt, dir, "", 0)
|
||||
if err != nil {
|
||||
t.Fatalf("attachmentMarkdown(no extension): %v", err)
|
||||
}
|
||||
if !strings.Contains(gotNoExt, "goonoext") {
|
||||
t.Errorf("extensionless attachment text = %q, want goonoext", gotNoExt)
|
||||
}
|
||||
|
||||
binPath := filepath.Join(dir, "data.bin")
|
||||
if err := os.WriteFile(binPath, []byte{0, 1, 2, 3}, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := (Tools{}).attachmentMarkdown(binPath, dir, "", 0); err == nil {
|
||||
t.Error("unsupported attachment: want error, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
// TestToMarkdownWithAttachmentsPlainPDF ensures a PDF without attachments
|
||||
// returns just the body (pdfdetach prints "0 embedded files").
|
||||
func TestToMarkdownWithAttachmentsPlainPDF(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.PDFDetach == "" || tools.PDFToText == "" {
|
||||
t.Skip("pdfdetach/pdftotext not on PATH")
|
||||
}
|
||||
// The fixture itself is a PDF with one attachment; strip it by extracting
|
||||
// the body only through ToMarkdown and compare JoinWithAttachments(nil).
|
||||
res, err := tools.ToMarkdown(filepath.Join("..", "..", "testdata", "pdf-with-attachment.pdf"), t.TempDir(), "eng", 1)
|
||||
if err != nil {
|
||||
t.Fatalf("ToMarkdown: %v", err)
|
||||
}
|
||||
if strings.Contains(res.Markdown, "gooattachmenttoken") {
|
||||
t.Fatalf("body must not contain attachment text: %q", res.Markdown)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,393 @@
|
||||
package xlspipe
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
// Sheet names (Russian tabs) for cutover Portugal workbook.
|
||||
const (
|
||||
SheetInputs = "Ввод"
|
||||
SheetDom6 = "Вс 6.09"
|
||||
SheetJue3 = "Чт 3.09"
|
||||
SheetWedHyp = "Ср гип"
|
||||
SheetSummary = "Сводка"
|
||||
)
|
||||
|
||||
// CutoverPortugalDefaults holds live FO values (2026-08-28).
|
||||
type CutoverPortugalDefaults struct {
|
||||
FerryDom6 float64
|
||||
FerryJue3 float64
|
||||
FerrySuperiorDelta float64
|
||||
HousingLow float64
|
||||
HousingMid float64
|
||||
HousingHigh float64
|
||||
DriveLow float64
|
||||
DriveMid float64
|
||||
DriveHigh float64
|
||||
BoardLow float64
|
||||
BoardMid float64
|
||||
BoardHigh float64
|
||||
FoodLow float64
|
||||
FoodMid float64
|
||||
FoodHigh float64
|
||||
SimLow float64
|
||||
SimMid float64
|
||||
SimHigh float64
|
||||
MonthCap float64
|
||||
ExtraNightsDom6 float64
|
||||
ExtraNightsJue3 float64
|
||||
ExtraNightsWed float64
|
||||
}
|
||||
|
||||
// DefaultCutoverPortugal returns FO live snapshot from portugal track (28.08.2026).
|
||||
func DefaultCutoverPortugal() CutoverPortugalDefaults {
|
||||
return CutoverPortugalDefaults{
|
||||
FerryDom6: 457.89,
|
||||
FerryJue3: 484.09,
|
||||
FerrySuperiorDelta: 19.64,
|
||||
HousingLow: 509,
|
||||
HousingMid: 600,
|
||||
HousingHigh: 650,
|
||||
DriveLow: 70,
|
||||
DriveMid: 85,
|
||||
DriveHigh: 100,
|
||||
BoardLow: 40,
|
||||
BoardMid: 60,
|
||||
BoardHigh: 80,
|
||||
FoodLow: 150,
|
||||
FoodMid: 220,
|
||||
FoodHigh: 300,
|
||||
SimLow: 50,
|
||||
SimMid: 100,
|
||||
SimHigh: 150,
|
||||
MonthCap: 2500,
|
||||
ExtraNightsDom6: 0,
|
||||
ExtraNightsJue3: 0,
|
||||
ExtraNightsWed: 4,
|
||||
}
|
||||
}
|
||||
|
||||
type inputField struct {
|
||||
name string // defined name (ASCII, for formulas)
|
||||
label string
|
||||
value float64
|
||||
note string
|
||||
comment string
|
||||
}
|
||||
|
||||
// BuildCutoverPortugalWorkbook creates a multi-sheet cutover budget with Russian labels,
|
||||
// cell comments on non-obvious inputs, named ranges, and cross-sheet formulas.
|
||||
func BuildCutoverPortugalWorkbook(d CutoverPortugalDefaults) (*excelize.File, error) {
|
||||
f := excelize.NewFile()
|
||||
defaultSheet := f.GetSheetName(0)
|
||||
if err := f.SetSheetName(defaultSheet, SheetInputs); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
for _, name := range []string{SheetDom6, SheetJue3, SheetWedHyp, SheetSummary} {
|
||||
if _, err := f.NewSheet(name); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
|
||||
if err := writeInputsSheet(f, d); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
scenarios := []struct {
|
||||
sheet, ferry, extra, note string
|
||||
}{
|
||||
{SheetDom6, "ferry_dom6", "extra_nights_dom6", "Живой слот вс 6.09 20:30; заезд в квартиру вт 8.09"},
|
||||
{SheetJue3, "ferry_jue3", "extra_nights_jue3", "Живой чт 3.09; T1a 05–12 если TF-крыша кончается раньше вс"},
|
||||
{SheetWedHyp, "ferry_jue3", "extra_nights_wed", "Гипотеза: ср 2.09 20:00 по тарифу Jue3; заезд пт 4.09"},
|
||||
}
|
||||
for _, sc := range scenarios {
|
||||
if err := writeScenarioSheet(f, sc.sheet, sc.ferry, sc.extra, sc.note); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
if err := writeSummarySheet(f); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
f.SetActiveSheet(0)
|
||||
if err := finalizeWorkbook(f); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
|
||||
func writeInputsSheet(f *excelize.File, d CutoverPortugalDefaults) error {
|
||||
if err := setHeaders(f, SheetInputs, "Параметр", "Значение €", "Кратко"); err != nil {
|
||||
return err
|
||||
}
|
||||
fields := []inputField{
|
||||
{
|
||||
name: "ferry_dom6", label: "Паром вс 6.09 (Básica, без residencia)",
|
||||
value: d.FerryDom6, note: "FO live: 2взр+младенец+авто",
|
||||
comment: "Fred Olsen Dom 6.09 20:30 SC→Huelva, прибытие Mar 8 09:00. Butaca Normal/Básica без субсидии канарского residencia (−183€). Меняйте после нового live FO.",
|
||||
},
|
||||
{
|
||||
name: "ferry_jue3", label: "Паром чт 3.09 (Básica, без residencia)",
|
||||
value: d.FerryJue3, note: "FO live Jue 3 20:00",
|
||||
comment: "Прямой рейс 35 ч. Дороже вс на ~26€. Используется также для листа «Ср гип», если своего рейса ср 2.09 нет в продаже.",
|
||||
},
|
||||
{
|
||||
name: "ferry_superior_uplift", label: "Доплата VIP / Butaca Superior",
|
||||
value: d.FerrySuperiorDelta, note: "Superior − Normal (Jue3 live)",
|
||||
comment: "Разница между Superior и Normal на live Jue3 ≈19,64€. На Dom6 Superior в сессии не перевыбирали — оценка по этой дельте. VIP = salón, не каюта.",
|
||||
},
|
||||
{
|
||||
name: "housing_7n_low", label: "Крыша 7 ночей Setúbal — минимум",
|
||||
value: d.HousingLow, note: "Airbnb low band",
|
||||
comment: "Короткая аренда T1a (#11): 08–15.09, 1–2BR с парковкой. Низкая граница live Airbnb Setúbal.",
|
||||
},
|
||||
{
|
||||
name: "housing_7n_mid", label: "Крыша 7 ночей Setúbal — целевой mid",
|
||||
value: d.HousingMid, note: "Цель #11: 500–650€/нед",
|
||||
comment: "Рабочая оценка для брони. Основной столбец mid на листах сценариев.",
|
||||
},
|
||||
{
|
||||
name: "housing_7n_high", label: "Крыша 7 ночей Setúbal — максимум",
|
||||
value: d.HousingHigh, note: "Airbnb high band",
|
||||
},
|
||||
{
|
||||
name: "drive_low", label: "Проезд Huelva → Setúbal — мин",
|
||||
value: d.DriveLow, note: "OSRM ~3,8 ч",
|
||||
comment: "≈342 км: топливо + платные дороги. Низкая/средняя/высокая оценка.",
|
||||
},
|
||||
{name: "drive_mid", label: "Проезд Huelva → Setúbal — mid", value: d.DriveMid},
|
||||
{name: "drive_high", label: "Проезд Huelva → Setúbal — макс", value: d.DriveHigh},
|
||||
{
|
||||
name: "board_low", label: "Еда на пароме — мин",
|
||||
value: d.BoardLow, note: "Меню FO",
|
||||
comment: "Питание на борту (Fred Olsen). George 0–3 обычно бесплатно как пассажир — еда отдельно.",
|
||||
},
|
||||
{name: "board_mid", label: "Еда на пароме — mid", value: d.BoardMid},
|
||||
{name: "board_high", label: "Еда на пароме — макс", value: d.BoardHigh},
|
||||
{
|
||||
name: "food_7d_low", label: "Еда 7 дней в PT — мин",
|
||||
value: d.FoodLow, note: "Готовим в apt",
|
||||
comment: "Первая неделя в Setúbal после парома — продукты, не рестораны.",
|
||||
},
|
||||
{name: "food_7d_mid", label: "Еда 7 дней в PT — mid", value: d.FoodMid},
|
||||
{name: "food_7d_high", label: "Еда 7 дней в PT — макс", value: d.FoodHigh},
|
||||
{
|
||||
name: "sim_low", label: "SIM + документы — мин",
|
||||
value: d.SimLow, note: "Разовые cutover",
|
||||
comment: "eSIM, копии, мелкие госпошлины при cutover. Не включает депозит аренды.",
|
||||
},
|
||||
{name: "sim_mid", label: "SIM + документы — mid", value: d.SimMid},
|
||||
{name: "sim_high", label: "SIM + документы — макс", value: d.SimHigh},
|
||||
{
|
||||
name: "month_cap", label: "Потолок бюджета на месяц (€)",
|
||||
value: d.MonthCap, note: "SoT: 2500€",
|
||||
comment: "Жёсткий потолок Sep из source-of-truth. «Остаток» = потолок − итого mid сценария.",
|
||||
},
|
||||
{
|
||||
name: "extra_nights_dom6", label: "Лишние ночи в PT (вс 6.09)",
|
||||
value: d.ExtraNightsDom6, note: "0 = TF до вс",
|
||||
comment: "Платные ночи в PT до начала 7-дневной крыши. Для Dom6 обычно 0: остаёмся на Тенерифе до вс, заезд вт 8.09.",
|
||||
},
|
||||
{
|
||||
name: "extra_nights_jue3", label: "Лишние ночи в PT (чт 3.09)",
|
||||
value: d.ExtraNightsJue3, note: "0 если TF до вс",
|
||||
comment: "Если крыша TF кончается раньше вс — нужны ночи 05–07.09 до Airbnb. По умолчанию 0.",
|
||||
},
|
||||
{
|
||||
name: "extra_nights_wed", label: "Лишние ночи в PT (ср гип)",
|
||||
value: d.ExtraNightsWed, note: "Гип: заезд пт 4.09",
|
||||
comment: "Гипотетический слот ср 2.09 → заезд пт 4.09 = 4 лишних ночи до типичного 7н блока. Рейса ср в FO нет — тариф как Jue3.",
|
||||
},
|
||||
}
|
||||
for i, fld := range fields {
|
||||
row := i + 2
|
||||
if err := f.SetCellStr(SheetInputs, fmt.Sprintf("A%d", row), fld.label); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := defineInput(f, SheetInputs, fld.name, row, fld.value, fld.note, fld.comment); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(SheetInputs, "A25", "—"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(SheetInputs, "B25", "Редактируйте жёлтые ячейки"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(SheetInputs, "C25", "Формулы на листах сценариев и «Сводка» пересчитаются в OnlyOffice"); err != nil {
|
||||
return err
|
||||
}
|
||||
return styleSheet(f, SheetInputs, SheetInputs)
|
||||
}
|
||||
|
||||
func writeScenarioSheet(f *excelize.File, sheet, ferryName, extraNightsName, scenarioNote string) error {
|
||||
if err := setHeaders(f, sheet, "Статья", "Мин €", "Mid €", "Макс €", "Пояснение"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "A2", "Паром Básica"); err != nil {
|
||||
return err
|
||||
}
|
||||
for col, ref := range []string{ferryName, ferryName, ferryName} {
|
||||
cell, _ := excelize.CoordinatesToCellName(col+2, 2)
|
||||
if err := f.SetCellFormula(sheet, cell, "="+ref); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := addCellComment(f, sheet, "A2", "Тариф парома для этого сценария. Берётся с листа «Ввод»."); err != nil {
|
||||
return err
|
||||
}
|
||||
lines := []struct {
|
||||
label string
|
||||
low, mid, high, note string
|
||||
comment string
|
||||
}{
|
||||
{"Крыша 7 н Setúbal", "housing_7n_low", "housing_7n_mid", "housing_7n_high", "Airbnb T1a #11", ""},
|
||||
{"Huelva → Setúbal", "drive_low", "drive_mid", "drive_high", "OSRM ~3,8 ч", ""},
|
||||
{"Еда на борту", "board_low", "board_mid", "board_high", "Меню FO", ""},
|
||||
{"Еда 7 д в PT", "food_7d_low", "food_7d_mid", "food_7d_high", "Готовим дома", ""},
|
||||
{"SIM / документы", "sim_low", "sim_mid", "sim_high", "Cutover", ""},
|
||||
}
|
||||
for i, ln := range lines {
|
||||
row := i + 3
|
||||
if err := setFormulaRow(f, sheet, row, ln.label,
|
||||
"="+ln.low, "="+ln.mid, "="+ln.high, ln.note); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "A8", "Итого Básica"); err != nil {
|
||||
return err
|
||||
}
|
||||
for col := 2; col <= 4; col++ {
|
||||
cell, _ := excelize.CoordinatesToCellName(col, 8)
|
||||
colL, _ := excelize.CoordinatesToCellName(col, 2)
|
||||
colH, _ := excelize.CoordinatesToCellName(col, 7)
|
||||
if err := f.SetCellFormula(sheet, cell, fmt.Sprintf("=SUM(%s:%s)", colL, colH)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "E8", "SUM строк 2–7"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := setFormulaRow(f, sheet, 9, "Итого Superior",
|
||||
"=B8+ferry_superior_uplift", "=C8+ferry_superior_uplift", "=D8+ferry_superior_uplift",
|
||||
"Básica + VIP"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := addCellComment(f, sheet, "A9", "Butaca Superior / VIP salón. Доплата с листа «Ввод»."); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "A10", "Лишние ночи PT"); err != nil {
|
||||
return err
|
||||
}
|
||||
for col, housing := range []string{"housing_7n_low", "housing_7n_mid", "housing_7n_high"} {
|
||||
cell, _ := excelize.CoordinatesToCellName(col+2, 10)
|
||||
formula := fmt.Sprintf("=%s*%s/7", extraNightsName, housing)
|
||||
if err := f.SetCellFormula(sheet, cell, formula); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "E10", "ночей × (крыша/7)"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := addCellComment(f, sheet, "A10", "Платное жильё до начала 7-дневной брони. Число ночей — на листе «Ввод» для этого сценария."); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := setFormulaRow(f, sheet, 11, "Итого с ночами",
|
||||
"=B8+B10", "=C8+C10", "=D8+D10", scenarioNote); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "A12", "Остаток от потолка"); err != nil {
|
||||
return err
|
||||
}
|
||||
for _, col := range []string{"B", "C", "D"} {
|
||||
if err := f.SetCellFormula(sheet, col+"12", "=month_cap-C11"); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "E12", "потолок − mid итого"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := addCellComment(f, sheet, "C12", "Сколько остаётся от месячного потолка 2500€ после cutover (mid)."); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "A13", "Среднее по строкам"); err != nil {
|
||||
return err
|
||||
}
|
||||
for col := 2; col <= 4; col++ {
|
||||
cell, _ := excelize.CoordinatesToCellName(col, 13)
|
||||
colL, _ := excelize.CoordinatesToCellName(col, 2)
|
||||
colH, _ := excelize.CoordinatesToCellName(col, 7)
|
||||
if err := f.SetCellFormula(sheet, cell, fmt.Sprintf("=AVERAGE(%s:%s)", colL, colH)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "E13", "AVG статей 2–7"); err != nil {
|
||||
return err
|
||||
}
|
||||
return styleSheet(f, sheet, SheetInputs)
|
||||
}
|
||||
|
||||
func writeSummarySheet(f *excelize.File) error {
|
||||
if err := setHeaders(f, SheetSummary,
|
||||
"Сценарий", "Básica mid", "Superior mid", "Итого mid", "Остаток", "Δ vs вс"); err != nil {
|
||||
return err
|
||||
}
|
||||
rows := []struct {
|
||||
label, sheet string
|
||||
}{
|
||||
{"Вс 6.09 (live)", SheetDom6},
|
||||
{"Чт 3.09 (live)", SheetJue3},
|
||||
{"Ср 2.09 (гипотеза)", SheetWedHyp},
|
||||
}
|
||||
qs := quoteSheet
|
||||
for i, r := range rows {
|
||||
row := i + 2
|
||||
if err := f.SetCellStr(SheetSummary, fmt.Sprintf("A%d", row), r.label); err != nil {
|
||||
return err
|
||||
}
|
||||
pairs := []struct {
|
||||
col int
|
||||
ref string
|
||||
}{
|
||||
{2, fmt.Sprintf("%s!C8", qs(r.sheet))},
|
||||
{3, fmt.Sprintf("%s!C9", qs(r.sheet))},
|
||||
{4, fmt.Sprintf("%s!C11", qs(r.sheet))},
|
||||
{5, fmt.Sprintf("%s!C12", qs(r.sheet))},
|
||||
}
|
||||
for _, p := range pairs {
|
||||
cell, _ := excelize.CoordinatesToCellName(p.col, row)
|
||||
if err := f.SetCellFormula(SheetSummary, cell, "="+p.ref); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
cell, _ := excelize.CoordinatesToCellName(6, row)
|
||||
if err := f.SetCellFormula(SheetSummary, cell, fmt.Sprintf("=D%d-$D$2", row)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(SheetSummary, "A6", "Потолок месяца"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellFormula(SheetSummary, "D6", "=month_cap"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(SheetSummary, "A7", "Лучший итого mid"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellFormula(SheetSummary, "D7", "=MIN(D2:D4)"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(SheetSummary, "E7", "MIN по сценариям"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := addCellComment(f, SheetSummary, "D7", "Минимальный mid «Итого с ночами» среди трёх сценариев. Сейчас обычно вс 6.09."); err != nil {
|
||||
return err
|
||||
}
|
||||
return styleSheet(f, SheetSummary, SheetInputs)
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
package xlspipe
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
func parseCalcFloat(s string) (float64, error) {
|
||||
s = strings.ReplaceAll(s, ",", "")
|
||||
s = strings.TrimSpace(s)
|
||||
var val float64
|
||||
_, err := fmt.Sscan(s, &val)
|
||||
return val, err
|
||||
}
|
||||
|
||||
func TestBuildCutoverPortugalWorkbook_Formulas(t *testing.T) {
|
||||
f, err := BuildCutoverPortugalWorkbook(DefaultCutoverPortugal())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "cutover.xlsx")
|
||||
if err := Save(f, path); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st, err := os.Stat(path)
|
||||
if err != nil || st.Size() < 4096 {
|
||||
t.Fatalf("xlsx too small: %v", err)
|
||||
}
|
||||
|
||||
opened, err := excelize.OpenFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer opened.Close()
|
||||
|
||||
cases := []struct {
|
||||
sheet, cell string
|
||||
want float64
|
||||
tol float64
|
||||
}{
|
||||
{SheetDom6, "C8", 1522.89, 0.02},
|
||||
{SheetJue3, "C8", 1549.09, 0.02},
|
||||
{SheetWedHyp, "C11", 1891.95, 0.05},
|
||||
{SheetSummary, "D2", 1522.89, 0.02},
|
||||
{SheetSummary, "D7", 1522.89, 0.02},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
got, err := opened.CalcCellValue(tc.sheet, tc.cell)
|
||||
if err != nil {
|
||||
t.Fatalf("%s!%s calc: %v", tc.sheet, tc.cell, err)
|
||||
}
|
||||
val, err := parseCalcFloat(got)
|
||||
if err != nil {
|
||||
t.Fatalf("%s!%s parse %q: %v", tc.sheet, tc.cell, got, err)
|
||||
}
|
||||
if diff := val - tc.want; diff < -tc.tol || diff > tc.tol {
|
||||
t.Fatalf("%s!%s = %v want ~%v", tc.sheet, tc.cell, val, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user