Compare commits
80
Commits
v0.11.0
...
33b5c1ea97
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
33b5c1ea97 | ||
|
|
9a6da1a4f7 | ||
|
|
504d13ed08 | ||
|
|
efc0864d42 | ||
|
|
59caf5e560 | ||
|
|
e7803c5269 | ||
|
|
a08e7c49ad | ||
|
|
4310aa7002 | ||
|
|
1a7a962183 | ||
|
|
af8e3b053a | ||
|
|
8ac777c031 | ||
|
|
c576bfe2ee | ||
|
|
7f56332285 | ||
|
|
a5807b9c31 | ||
|
|
d650a16a36 | ||
|
|
3da11586f9 | ||
|
|
e9c969a89c | ||
|
|
4d91726179 | ||
|
|
4d8af7a2fe | ||
|
|
34745e349e | ||
|
|
dfea57a57b | ||
|
|
8595c17f25 | ||
|
|
61da2fb88b | ||
|
|
24ca144b22 | ||
|
|
93828ee19d | ||
|
|
35f0cb8d20 | ||
|
|
ca5c85a4e2 | ||
|
|
220c7e8265 | ||
|
|
ff4c0481ba | ||
|
|
68445b0dfd | ||
|
|
8739d45b1a | ||
|
|
f3150436fd | ||
|
|
a30ae20f32 | ||
|
|
5a0b2ab412 | ||
|
|
69844c9122 | ||
|
|
b04d610275 | ||
|
|
6b3f40e1a1 | ||
|
|
ba8148a9e1 | ||
|
|
f1739dc9dc | ||
|
|
5b812ce346 | ||
|
|
309ae44ee4 | ||
|
|
ff2bc539f5 | ||
|
|
d9a7adcd94 | ||
|
|
3713ed6c51 | ||
|
|
ac06d86363 | ||
|
|
7d397ac680 | ||
|
|
e52cb31bb7 | ||
|
|
a8cb4e9780 | ||
|
|
622dcdb7bf | ||
|
|
2b267d36a8 | ||
|
|
50bd47570b | ||
|
|
096a996726 | ||
|
|
625dd0a60a | ||
|
|
e531de280f | ||
|
|
129e3a58cc | ||
|
|
70e605b4ca | ||
|
|
a264cbd61c | ||
|
|
a97a160c44 | ||
|
|
df447f2673 | ||
|
|
db9ef12ff4 | ||
|
|
2c0df0d55f | ||
|
|
239c5ea672 | ||
|
|
6ea2fbadf7 | ||
|
|
50d8974154 | ||
|
|
3fd3f797dc | ||
|
|
0e299d0692 | ||
|
|
9ead554f5c | ||
|
|
77c188c640 | ||
|
|
2a59817f51 | ||
|
|
9c450a0352 | ||
|
|
ab7dfbd859 | ||
|
|
dda2bd3ca5 | ||
|
|
ceea6d909e | ||
|
|
ecf34ba51a | ||
|
|
ebbd5d5373 | ||
|
|
20a09530cd | ||
|
|
666be883dc | ||
|
|
65cd3f5c74 | ||
|
|
1ef0670624 | ||
|
|
10c1cfc6a7 |
@@ -25,3 +25,19 @@ ONLYOFFICE_PROJECT_ID=33
|
||||
# cmd/office TUI — optional Document Server for DOCX→HTML preview:
|
||||
# ONLYOFFICE_DOCS_URL=https://docs.example.com
|
||||
# ONLYOFFICE_DOCS_SECRET=
|
||||
|
||||
# MinIO download fallback for the portal's stale AWS S3 consumer (older
|
||||
# Documents folders). When the portal redirects to amazonaws.com with access
|
||||
# key "minio" (403 InvalidAccessKeyId), files are fetched from the local MinIO
|
||||
# store instead. Without a key/secret the fallback is disabled.
|
||||
# MINIO_ENDPOINT=http://192.168.188.10:9000
|
||||
# MINIO_BUCKET=office
|
||||
# MINIO_ACCESS_KEY=
|
||||
# MINIO_SECRET_KEY=
|
||||
|
||||
# oo search — direct Elasticsearch access for name + content search. ES lives
|
||||
# inside the OnlyOffice VM on localhost:9200; expose it with an SSH tunnel
|
||||
# (see docs/elasticsearch.md). ONLYOFFICE_ES_INDEX defaults to files_file.
|
||||
# ONLYOFFICE_ES_URL=http://127.0.0.1:9200
|
||||
# ONLYOFFICE_ES_INDEX=files_file
|
||||
# ONLYOFFICE_TENANT=
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
github: eSlider
|
||||
ko_fi: eslider
|
||||
liberapay: eslider
|
||||
patreon: eslider
|
||||
custom:
|
||||
- https://polar.sh/eslider
|
||||
@@ -14,6 +14,7 @@ permissions:
|
||||
jobs:
|
||||
release-please:
|
||||
name: Release Please
|
||||
if: github.server_url == 'https://github.com'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Run release-please
|
||||
|
||||
@@ -16,6 +16,7 @@ permissions:
|
||||
jobs:
|
||||
goreleaser:
|
||||
name: GoReleaser
|
||||
if: github.server_url == 'https://github.com'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
# Always clone default branch first. workflow_dispatch often races with
|
||||
|
||||
@@ -10,6 +10,51 @@ permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
secret-scan:
|
||||
name: Secret scan (gitleaks)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Compute scan range (diff of new commits only)
|
||||
id: range
|
||||
run: |
|
||||
if [ "$GITHUB_EVENT_NAME" = "pull_request" ]; then
|
||||
RANGE="${{ github.event.pull_request.base.sha }}...${{ github.event.pull_request.head.sha }}"
|
||||
else
|
||||
BEFORE="${{ github.event.before }}"
|
||||
if [ "$BEFORE" = "0000000000000000000000000000000000000000" ]; then
|
||||
RANGE="$(git rev-list --max-parents=0 HEAD | tail -1)..$GITHUB_SHA"
|
||||
else
|
||||
RANGE="$BEFORE..$GITHUB_SHA"
|
||||
fi
|
||||
fi
|
||||
echo "RANGE=$RANGE" >> "$GITHUB_ENV"
|
||||
echo "Scanning range: $RANGE"
|
||||
|
||||
# Install the gitleaks binary instead of a docker action: the
|
||||
# docker://zricethezav/gitleaks action hardcodes /github/workspace,
|
||||
# which does not exist on the Gitea (act) runner. $GITHUB_WORKSPACE is
|
||||
# the checkout dir on BOTH runners (GitHub and Gitea act). Mirrors the
|
||||
# fix applied to 2dph (issue #142).
|
||||
- name: Gitleaks (diff-only, fail on leak)
|
||||
env:
|
||||
GITLEAKS_RANGE: ${{ env.RANGE }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
curl -fsSLo /tmp/gitleaks.tar.gz \
|
||||
https://github.com/gitleaks/gitleaks/releases/download/v8.30.1/gitleaks_8.30.1_linux_x64.tar.gz
|
||||
tar -xzf /tmp/gitleaks.tar.gz -C /tmp gitleaks
|
||||
chmod +x /tmp/gitleaks
|
||||
/tmp/gitleaks detect \
|
||||
--source "$GITHUB_WORKSPACE" \
|
||||
--log-opts="$GITLEAKS_RANGE" \
|
||||
--redact \
|
||||
--verbose
|
||||
|
||||
test:
|
||||
name: Test (Go ${{ matrix.go }})
|
||||
runs-on: ubuntu-latest
|
||||
@@ -22,7 +67,7 @@ jobs:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
uses: actions/setup-go@v4
|
||||
with:
|
||||
go-version: ${{ matrix.go }}
|
||||
cache: true
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
{
|
||||
".": "0.11.0"
|
||||
".": "0.17.0"
|
||||
}
|
||||
|
||||
@@ -9,16 +9,17 @@ Canonical Go client for OnlyOffice Workspace (Projects + Calendar + CRM) and the
|
||||
- `request.go` — `Request`, `Query`, `Time`, `Token`, `MetaResponse`, `Permissions`.
|
||||
- `auth.go` — `Authenticate`, `AuthenticateContext`, `InvalidateToken`, `Auth`, token lifecycle.
|
||||
- `http.go` — transport + DRY response decoders (`ResponseArray`/`ResponseObject`/`postFormObject`/`putFormObject`/`deleteObject`).
|
||||
- `projects.go`, `tasks.go`, `users.go`, `calendar.go`, `crm.go`, `files.go`, `mails.go`, `invoices.go` — typed / untyped domain methods. **`files.go`** — CRM opportunity upload plus **project/task Documents**. **`mails.go`** — OnlyOffice Workspace Mail. **`invoices.go`** — CRM invoices, PDF regen/cleanup, status. Association rules: [`docs/crm-associations.md`](docs/crm-associations.md).
|
||||
- `projects.go`, `tasks.go`, `users.go`, `calendar.go`, `crm.go`, `files.go`, `files_webdav.go`, `files_stem.go`, `retry.go`, `mails.go`, `invoices.go` — typed / untyped domain methods. **`files.go`** — CRM opportunity upload plus **project/task Documents** (`UpdateFile`, `UploadToFolderReplacing`). **`files_webdav.go`** — Documents module by id (`ListDavFolder`, `MoveDavItems`/`CopyDavItems` with per-operation error surfacing, `ListFileOps`). **`retry.go`** — `DoRetry`: deterministic linear backoff (no jitter) on 429/502/503/504; every bulk tool routes API calls through it. **`mails.go`** — OnlyOffice Workspace Mail. **`invoices.go`** — CRM invoices, PDF regen/cleanup, status. Association rules: [`docs/crm-associations.md`](docs/crm-associations.md).
|
||||
- Pure stdlib + `google/go-querystring`; no UI, no dotenv.
|
||||
- **CLI — `cmd/oo/` as `package main`.** Cobra wrapper that loads `.env` via `godotenv` at startup. **Subject-based command tree** mirroring [`tea`](https://gitea.com/gitea/tea):
|
||||
- `main.go` — entry point (docstring lists the command tree).
|
||||
- `common.go` — `rootCmd`, `newOO`, `printTable`/`printObject`, `--output table|json` flag.
|
||||
- `calendar.go`, `projects.go`, `projects_files.go`, `tasks.go`, `tasks_files.go`, `users.go`, `contacts.go`, `opportunities.go`, `cases.go`, `crm_tasks.go`, `mails.go`, `invoices.go` — one file per subject (or per subject facet), each registers in `init()`.
|
||||
- `calendar.go`, `projects.go`, `projects_files.go`, `tasks.go`, `tasks_files.go`, `users.go`, `contacts.go`, `opportunities.go`, `cases.go`, `crm.go`, `crm_tasks.go`, `catalog.go`, `docs.go`, `dav.go`, `mails.go`, `invoices.go` — one file per subject (or per subject facet), each registers in `init()`. `dav.go` exposes the Documents module by id (`oo dav ls|move|copy|mkdir|rename-file|rename-folder|download|fileops`).
|
||||
- CLI-only deps (`spf13/cobra`, `joho/godotenv`) stay out of the library.
|
||||
- **TUI — `cmd/office/` as `package main`.** Bubble Tea three-pane browser (module tree, selectable list, markdown preview). Reuses `cmd/internal/bootstrap` for env/auth and the root `onlyoffice` library for all API calls. UI logic in `cmd/office/ui/`; preview/formatting in `cmd/office/preview/`; list loaders in `cmd/office/fetch/`.
|
||||
- **List table (`DataTable`)** — `cmd/office/ui/table*.go`. Column layout policies live in `cmd/office/model/table_layout.go` (`TableFlexLayoutFor`); cell rendering uses the bubbles/table inline pattern in `table_render.go` (`renderTableCell`, `padANSIWidth`). See `.cursor/skills/office-tui-table/SKILL.md` before changing center-pane tables.
|
||||
- **Shared bootstrap — `cmd/internal/bootstrap/`.** `LoadEnv()` + `NewClient(ctx)` extracted from `oo`; both binaries import it.
|
||||
- **Bulk Documents tools — `cmd/ooscan/`, `cmd/pdfamount/`, `cmd/kontoblatt/`, `cmd/kontolink/`.** Single-purpose binaries (folder index, PDF amounts, Kontoblatt summary/linking). Pace requests, route API calls through `DoRetry`; usage in README.
|
||||
- **Personal ops tooling** (disk inventory, dossier→CRM sync, SearXNG) lives in private [`eSlider/oo-workspace`](https://git.produktor.io/eSlider/oo-workspace) (`oow`), not in this public tree.
|
||||
|
||||
## Rules
|
||||
@@ -27,7 +28,8 @@ Canonical Go client for OnlyOffice Workspace (Projects + Calendar + CRM) and the
|
||||
- New endpoints go into the library first; CLI commands are thin wrappers.
|
||||
- Prefer `ResponseObject` / `postFormObject` / `putFormObject` / `deleteObject` over hand-rolled `json.Unmarshal(responseField(...))` blocks — they exist for DRY, use them.
|
||||
- Domain split is by file, **not** by subpackage. Don't introduce `internal/` or `pkg/*` subpackages inside the library — it flattens the `*Client` call surface for a reason.
|
||||
- CLI commands follow **subject → verb** structure (`oo <subject> <verb>`), never `oo <verb>-<subject>`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/<subject>.go` for a genuinely new domain.
|
||||
- CLI commands follow **subject → verb** structure (`oo <subject> <verb>`), never `oo <verb>-<subject>`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/<subject>.go` for a genuinely new domain. The subject→verb tree in `cmd/oo/main.go` and the README table are documentation — update them with the code.
|
||||
- **Documents for agents:** prefer Markdown in git; OnlyOffice UI is weak for `.md`/`.txt`. Use `oo docs put-md` (md→docx) and `oo docs put-txt` (txt→docx, preserves line breaks). All upload paths default to **upsert** by `stem|ext` (`--replace`, default true); `--no-replace` fails on conflict; `--allow-duplicate` opts into raw OO append. `oo projects files dedupe PROJECT_ID` reports/removes duplicate stem|ext copies (`--apply`, `--cross`; includes project root folder).
|
||||
- Every table output goes through `printTable(headers, rows)`; every single-object through `printObject(v)`. Do not `fmt.Println` rows ad-hoc or the `--output json` flag breaks for that command.
|
||||
- No secrets in the repo; use `.env` (gitignored). Commit `.env.example` only.
|
||||
- Follow SemVer on tags; this repo is tagged at GitHub under `git@github.com:eSlider/go-onlyoffice.git`.
|
||||
|
||||
+115
@@ -6,6 +6,26 @@ adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||
|
||||
## Unreleased
|
||||
|
||||
## [0.17.0](https://github.com/eSlider/go-onlyoffice/compare/v0.16.0...v0.17.0) (2026-08-31)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **mailsync:** FetchMailFolder — integration-layer walk for ETL consumers ([35f0cb8](https://github.com/eSlider/go-onlyoffice/commit/35f0cb8d20076244141065c07e203e633bc3612a))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **files:** upsert uploads by default and dedupe project root ([24ca144](https://github.com/eSlider/go-onlyoffice/commit/24ca144b22abc5d056a5fd1ed9a1887f26a79d15))
|
||||
|
||||
## [0.16.0](https://github.com/eSlider/go-onlyoffice/compare/v0.15.0...v0.16.0) (2026-08-30)
|
||||
|
||||
### Features
|
||||
|
||||
* **docs:** `put-xlsx` — multi-sheet бюджеты с named inputs, SUM/AVG/MIN
|
||||
формулами, cross-sheet ссылками и cell comments (`internal/xlspipe`,
|
||||
excelize) ([68445b0](https://github.com/eSlider/go-onlyoffice/commit/68445b0))
|
||||
|
||||
### Added
|
||||
|
||||
* **docs:** CRM association graph and OO quirks (`docs/crm-associations.md`)
|
||||
@@ -16,6 +36,101 @@ adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||
|
||||
* Document that invoice→deal must be set at create (`update --opportunity` often HTTP 400)
|
||||
|
||||
## [0.15.0](https://github.com/eSlider/go-onlyoffice/compare/v0.14.0...v0.15.0) (2026-08-29)
|
||||
|
||||
### Features
|
||||
|
||||
* **files:** `ListFolder`, `CreateFolder`, `MoveFiles`, `UploadToFolder` — Documents folder helpers for OO Documents ingestion ([6ea2fba](https://github.com/eSlider/go-onlyoffice/commit/6ea2fba))
|
||||
* **files:** dedupe by stem|ext (`files_stem.go`) + `oo projects files dedupe` ([ac06d86](https://github.com/eSlider/go-onlyoffice/commit/ac06d86))
|
||||
* **docs:** `internal/docpipe` — md↔docx convert, OCR→PDF, hOCR→Markdown via go-hocr, as-md/put-md for agents ([6ea2fba](https://github.com/eSlider/go-onlyoffice/commit/6ea2fba), [a264cbd](https://github.com/eSlider/go-onlyoffice/commit/a264cbd))
|
||||
* **docs:** optimize PDF via Ghostscript pdfwrite ([6b3f40e](https://github.com/eSlider/go-onlyoffice/commit/6b3f40e))
|
||||
* **security:** gitleaks secret-scan in CI + pre-push/pre-commit hooks ([9ead554](https://github.com/eSlider/go-onlyoffice/commit/9ead554))
|
||||
|
||||
### Fixes
|
||||
|
||||
* **files:** `DeleteFiles` via per-file DELETE API; `DeleteDavItems` with `Immediately` true ([622dcdb](https://github.com/eSlider/go-onlyoffice/commit/622dcdb), [d9a7adc](https://github.com/eSlider/go-onlyoffice/commit/d9a7adc))
|
||||
* **docs:** put-txt preserves line breaks in DOCX; fixed-width extracts in code block; put-md upsert by stem ([309ae44](https://github.com/eSlider/go-onlyoffice/commit/309ae44), [f1739dc](https://github.com/eSlider/go-onlyoffice/commit/f1739dc), [50bd475](https://github.com/eSlider/go-onlyoffice/commit/50bd475))
|
||||
* **ci:** point gitleaks at `/github/workspace` in docker action ([2c0df0d](https://github.com/eSlider/go-onlyoffice/commit/2c0df0d))
|
||||
|
||||
## [0.14.0](https://github.com/eSlider/go-onlyoffice/compare/v0.13.0...v0.14.0) (2026-08-27)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **crm:** contact email & person-opportunity indexes, history entity whitelist ([9c450a0](https://github.com/eSlider/go-onlyoffice/commit/9c450a0352c25f12ac16d91062744ab026ffe660))
|
||||
* **docs:** hOCR → Markdown via go-hocr ([e531de2](https://github.com/eSlider/go-onlyoffice/commit/e531de280f2c521a7411f75212803649b177db58))
|
||||
* **docs:** md↔docx convert, OCR→PDF, as-md/put-md for agents ([6ea2fba](https://github.com/eSlider/go-onlyoffice/commit/6ea2fbadf768d7d85f6c7a49a9bc1050e642bc56))
|
||||
* **docs:** md↔docx, OCR→PDF, as-md/put-md ([db9ef12](https://github.com/eSlider/go-onlyoffice/commit/db9ef12ff4096f8aadb557ca128b6edcb503028c))
|
||||
* **docs:** oo docs hocr — tesseract hOCR → go-hocr Markdown/YAML ([a264cbd](https://github.com/eSlider/go-onlyoffice/commit/a264cbd61c12be9098c8d16e7c1c1253eb460816))
|
||||
* **mail:** oo mails send — SendMail + guard empty-by-id send ([dda2bd3](https://github.com/eSlider/go-onlyoffice/commit/dda2bd3ca5cea6ebb89e1ac02b785b9838727bda))
|
||||
* **mail:** oo mails send — SendMail client + guard empty-by-id ([ab7dfbd](https://github.com/eSlider/go-onlyoffice/commit/ab7dfbd8594de5724286e2b460e789f4acb114bc))
|
||||
* **security:** secret-scan via gitleaks in CI + pre-push/pre-commit hooks ([#142](https://github.com/eSlider/go-onlyoffice/issues/142)) ([9ead554](https://github.com/eSlider/go-onlyoffice/commit/9ead554f5cc229687d3a27934a023fb1dddaef15))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **ci:** point gitleaks at /github/workspace in docker action ([2c0df0d](https://github.com/eSlider/go-onlyoffice/commit/2c0df0d55fb03555b60481695554d75dceffce48))
|
||||
* **docs:** put-md upsert — no duplicate folder files ([2b267d3](https://github.com/eSlider/go-onlyoffice/commit/2b267d36a8086b14a94706468f7e92e9179e62d4))
|
||||
* **docs:** put-md upsert by stem to avoid duplicate folder files ([50bd475](https://github.com/eSlider/go-onlyoffice/commit/50bd47570b6430cebde43641a7b1d2f8d54d32f2))
|
||||
* **files:** DeleteFiles actually removes files on produktor OO ([a8cb4e9](https://github.com/eSlider/go-onlyoffice/commit/a8cb4e97805226e8af694ec6a64ee3ac6a5e9fdd))
|
||||
* **files:** DeleteFiles via per-file DELETE API ([622dcdb](https://github.com/eSlider/go-onlyoffice/commit/622dcdb7bffd49acbea5ef07f5d5d009f44b1358))
|
||||
|
||||
|
||||
### Documentation
|
||||
|
||||
* **readme:** document oo docs hocr ([70e605b](https://github.com/eSlider/go-onlyoffice/commit/70e605b4ca8ee005e4351b2fc2d2eec081604998))
|
||||
* **readme:** document oo docs md↔docx / OCR agent workflow ([239c5ea](https://github.com/eSlider/go-onlyoffice/commit/239c5ea67201e62de6ba067f0e7963c7c9bda188))
|
||||
|
||||
## [0.13.0](https://github.com/eSlider/go-onlyoffice/compare/v0.12.0...v0.13.0) (2026-08-27)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **crm:** contact email & person-opportunity indexes, history entity whitelist ([9c450a0](https://github.com/eSlider/go-onlyoffice/commit/9c450a0352c25f12ac16d91062744ab026ffe660))
|
||||
* **docs:** hOCR → Markdown via go-hocr ([e531de2](https://github.com/eSlider/go-onlyoffice/commit/e531de280f2c521a7411f75212803649b177db58))
|
||||
* **docs:** md↔docx convert, OCR→PDF, as-md/put-md for agents ([6ea2fba](https://github.com/eSlider/go-onlyoffice/commit/6ea2fbadf768d7d85f6c7a49a9bc1050e642bc56))
|
||||
* **docs:** md↔docx, OCR→PDF, as-md/put-md ([db9ef12](https://github.com/eSlider/go-onlyoffice/commit/db9ef12ff4096f8aadb557ca128b6edcb503028c))
|
||||
* **docs:** oo docs hocr — tesseract hOCR → go-hocr Markdown/YAML ([a264cbd](https://github.com/eSlider/go-onlyoffice/commit/a264cbd61c12be9098c8d16e7c1c1253eb460816))
|
||||
* **mail:** oo mails send — SendMail + guard empty-by-id send ([dda2bd3](https://github.com/eSlider/go-onlyoffice/commit/dda2bd3ca5cea6ebb89e1ac02b785b9838727bda))
|
||||
* **mail:** oo mails send — SendMail client + guard empty-by-id ([ab7dfbd](https://github.com/eSlider/go-onlyoffice/commit/ab7dfbd8594de5724286e2b460e789f4acb114bc))
|
||||
* **security:** secret-scan via gitleaks in CI + pre-push/pre-commit hooks ([#142](https://github.com/eSlider/go-onlyoffice/issues/142)) ([9ead554](https://github.com/eSlider/go-onlyoffice/commit/9ead554f5cc229687d3a27934a023fb1dddaef15))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **ci:** point gitleaks at /github/workspace in docker action ([2c0df0d](https://github.com/eSlider/go-onlyoffice/commit/2c0df0d55fb03555b60481695554d75dceffce48))
|
||||
* **files:** rewrite viewUrl host to API base on download ([ecf34ba](https://github.com/eSlider/go-onlyoffice/commit/ecf34ba51aae77f1626a60795f7329f25d9344e5))
|
||||
|
||||
|
||||
### Documentation
|
||||
|
||||
* **readme:** document oo docs hocr ([70e605b](https://github.com/eSlider/go-onlyoffice/commit/70e605b4ca8ee005e4351b2fc2d2eec081604998))
|
||||
* **readme:** document oo docs md↔docx / OCR agent workflow ([239c5ea](https://github.com/eSlider/go-onlyoffice/commit/239c5ea67201e62de6ba067f0e7963c7c9bda188))
|
||||
|
||||
## [0.12.0](https://github.com/eSlider/go-onlyoffice/compare/v0.11.0...v0.12.0) (2026-08-27)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **crm:** contact email & person-opportunity indexes, history entity whitelist ([9c450a0](https://github.com/eSlider/go-onlyoffice/commit/9c450a0352c25f12ac16d91062744ab026ffe660))
|
||||
* **docs:** md↔docx convert, OCR→PDF, as-md/put-md for agents ([6ea2fba](https://github.com/eSlider/go-onlyoffice/commit/6ea2fbadf768d7d85f6c7a49a9bc1050e642bc56))
|
||||
* **docs:** md↔docx, OCR→PDF, as-md/put-md ([db9ef12](https://github.com/eSlider/go-onlyoffice/commit/db9ef12ff4096f8aadb557ca128b6edcb503028c))
|
||||
* **files:** add WebDAV-oriented Files operations ([ebbd5d5](https://github.com/eSlider/go-onlyoffice/commit/ebbd5d5373abfeca5f160312f201cbd683f55e42))
|
||||
* **mail:** oo mails send — SendMail + guard empty-by-id send ([dda2bd3](https://github.com/eSlider/go-onlyoffice/commit/dda2bd3ca5cea6ebb89e1ac02b785b9838727bda))
|
||||
* **mail:** oo mails send — SendMail client + guard empty-by-id ([ab7dfbd](https://github.com/eSlider/go-onlyoffice/commit/ab7dfbd8594de5724286e2b460e789f4acb114bc))
|
||||
* **security:** secret-scan via gitleaks in CI + pre-push/pre-commit hooks ([#142](https://github.com/eSlider/go-onlyoffice/issues/142)) ([9ead554](https://github.com/eSlider/go-onlyoffice/commit/9ead554f5cc229687d3a27934a023fb1dddaef15))
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **ci:** point gitleaks at /github/workspace in docker action ([2c0df0d](https://github.com/eSlider/go-onlyoffice/commit/2c0df0d55fb03555b60481695554d75dceffce48))
|
||||
* **files:** rewrite viewUrl host to API base on download ([ecf34ba](https://github.com/eSlider/go-onlyoffice/commit/ecf34ba51aae77f1626a60795f7329f25d9344e5))
|
||||
|
||||
|
||||
### Documentation
|
||||
|
||||
* **readme:** document oo docs md↔docx / OCR agent workflow ([239c5ea](https://github.com/eSlider/go-onlyoffice/commit/239c5ea67201e62de6ba067f0e7963c7c9bda188))
|
||||
|
||||
## [0.11.0](https://github.com/eSlider/go-onlyoffice/compare/v0.10.0...v0.11.0) (2026-08-18)
|
||||
|
||||
|
||||
|
||||
@@ -503,6 +503,29 @@ type Task struct {
|
||||
|---|---|
|
||||
| `GetUsers()` | List all users with profiles |
|
||||
|
||||
### Documents Files
|
||||
|
||||
| Method | Description |
|
||||
|---|---|
|
||||
| `ListDavFolder(ctx, id)` | List a Documents folder (`@root` for virtual sections) |
|
||||
| `ListDavSections(ctx)` | Virtual sections (Documents, Projects, …) |
|
||||
| `CreateDavFolder(ctx, parentID, title)` | Create a subfolder |
|
||||
| `RenameDavFolder(ctx, id, title)` / `RenameDavFile(ctx, id, title)` | Rename folder / file |
|
||||
| `DownloadFile(ctx, id, dst)` / `DownloadDavFile(ctx, id, w)` | Download file bytes |
|
||||
| `UploadDavFile(ctx, folderID, fileName, src)` | Upload from a reader |
|
||||
| `UploadToFolder(ctx, folderID, localPath)` | Upload a local file into a folder |
|
||||
| `UploadToFolderReplacing(ctx, folderID, localPath)` | Upsert by `stem\|ext`; returns replaced ids |
|
||||
| `UpdateFile(ctx, fileID, localPath)` | New version of an existing file (same id, no copy) |
|
||||
| `MoveDavItems(ctx, folderIDs, fileIDs, dest)` | Move (`resolveType=Skip`); per-operation errors surfaced, not silent nil |
|
||||
| `CopyDavItems(ctx, folderIDs, fileIDs, dest)` | Copy (`conflictResolveType=Skip`); errors surfaced |
|
||||
| `MoveFiles(ctx, destFolderID, fileIDs)` | Move with `resolveType=Skip` + `holdResult`; errors surfaced |
|
||||
| `ListFileOps(ctx)` | Active file operations (move/copy status polling) |
|
||||
| `FolderFiles(ctx, folderID)` | Flat file list of a folder (stem helpers) |
|
||||
| `DeleteFilesByStem(ctx, folderID, stem)` | Remove `stem\|ext` copies |
|
||||
| `DoRetry(ctx, policy, fn)` | Deterministic linear backoff (N·Base, no jitter) on 429/502/503/504 |
|
||||
| `DefaultRetryPolicy()` | 5 attempts, 1s·2s·3s·4s waits, 30s cap |
|
||||
| `Transient(err)` | True for retriable OnlyOffice answers |
|
||||
|
||||
### Helper Types
|
||||
|
||||
| Type | Description |
|
||||
@@ -607,30 +630,110 @@ go test -tags=integration ./cmd/office/fetch/... ./cmd/office/preview/...
|
||||
```bash
|
||||
# Project Documents (files module)
|
||||
oo projects files list 33
|
||||
oo projects files upload 33 ./notes.md
|
||||
oo projects files download 12345 --to ./copy.md
|
||||
oo projects files rename 12345 notes-v2.md
|
||||
oo projects files upload 33 ./notes.docx
|
||||
oo projects files download 12345 --to ./copy.docx
|
||||
oo projects files rename 12345 notes-v2.docx
|
||||
oo projects files delete 12345
|
||||
oo projects files dedupe 7 # dry-run duplicate report
|
||||
oo projects files dedupe 7 --apply # remove older stem|ext copies per folder
|
||||
oo projects files dedupe 7 --cross --apply # cross-folder; keep non-_trash
|
||||
|
||||
# Agent document pipeline (md in git ↔ docx in OO; OCR scans)
|
||||
oo docs tools
|
||||
oo docs convert ./note.md # → note.docx
|
||||
oo docs convert ./note.docx # → note.md
|
||||
oo docs ocr ./scan.jpg --md ./scan.md # searchable PDF + markdown
|
||||
oo docs hocr ./scan.jpg --lang spa --md ./scan.hocr.md --yaml ./scan.yml
|
||||
oo docs put-md 7 ./OO-HONDA-7-INDEX.md --folder 490
|
||||
oo docs put-txt 7 ./notes.txt --folder 490
|
||||
oo docs put-xlsx 7 ./table.xlsx --folder 490
|
||||
oo docs as-md 2815 --to ./parte.md # download OO file as MD (OCR if needed)
|
||||
oo docs as-md 307 --hocr --lang spa # OO download via go-hocr structure
|
||||
oo projects files put-md 7 ./note.md # alias
|
||||
oo projects files as-md 2815 # alias
|
||||
oo tasks files list 208
|
||||
oo tasks files upload 208 ./notes.pdf
|
||||
oo tasks files detach 208 12345
|
||||
```
|
||||
|
||||
### Documents module (`oo dav`)
|
||||
|
||||
Direct access to the Documents module by folder/file id — the same calls that
|
||||
back `oo-webdav` and the project/task file commands. `move` sends
|
||||
`resolveType=Skip` + `holdResult=true`: without those params the legacy
|
||||
`fileops/move` endpoint answers 200 without moving anything, and the library
|
||||
surfaces such per-operation errors instead of a silent nil
|
||||
(`MoveDavItems` / `CopyDavItems` / `MoveFiles`).
|
||||
|
||||
```bash
|
||||
oo dav ls 659
|
||||
oo dav ls @root # virtual sections (Documents, Projects, …)
|
||||
oo dav mkdir 659 "2026 inbox"
|
||||
oo dav move 659 22881 22882 # DEST_FOLDER_ID FILE_ID…
|
||||
oo dav move 659 22881 --folders 670 # move folders along with files
|
||||
oo dav copy 659 22881
|
||||
oo dav rename-file 22881 invoice-v2.pdf
|
||||
oo dav rename-folder 671 o2-archive
|
||||
oo dav download 22881 --to ./copy.pdf # default path: ./<server title>
|
||||
oo dav fileops # active move/copy operations (status polling)
|
||||
```
|
||||
|
||||
### Search (`oo search`)
|
||||
|
||||
Full-text search over the Documents index. The REST endpoint
|
||||
`/api/2.0/files/@search/{query}` only searches file names in the database, so
|
||||
`oo search` talks to the OnlyOffice **Elasticsearch** directly (index
|
||||
`files_file`). Name search is default; `--content` also matches extracted
|
||||
document text (`document.attachment.content`, Office formats only).
|
||||
See [`docs/elasticsearch.md`](docs/elasticsearch.md) for the tunnel setup.
|
||||
|
||||
```bash
|
||||
oo search "Rechnung" # names only
|
||||
oo search "Mahngebühr" --content # names + document text
|
||||
oo search "Rechnung" --folder 649 --limit 50
|
||||
oo search "Rechnung" --json # shorthand for -o json
|
||||
```
|
||||
|
||||
Requires `ONLYOFFICE_ES_URL` (plus optional `ONLYOFFICE_ES_INDEX`,
|
||||
`ONLYOFFICE_TENANT`).
|
||||
|
||||
### Bulk tools (`cmd/`)
|
||||
|
||||
Small single-purpose binaries for bulk Documents work. All of them pace
|
||||
requests and retry transient OnlyOffice answers (429/502/503/504) with a
|
||||
deterministic linear backoff — no jitter, same waits on every run
|
||||
(see `DoRetry` below). Build with `go build ./cmd/<tool>`.
|
||||
|
||||
```bash
|
||||
ooscan 659 # recursive index → TSV: file_id, folder_id, path, title
|
||||
ooscan 659 666 > oo-index.tsv # several roots into one index
|
||||
pdfamount 671 # "Zu zahlender Betrag" per PDF → TSV: file_id, title, amount
|
||||
kontoblatt 3906 ./kontoblatt.xlsx # summary (Gegenkonto/Monat) uploaded next to source
|
||||
kontolink IN.xlsx oo-index.tsv OUT.xlsx [FILE_ID] [AMOUNTS_TSV]
|
||||
# kontolink writes DocEditor links into the Link column: Beleg → supplier+month
|
||||
# → amount+date (5th arg = pdfamount output); with FILE_ID it updates the
|
||||
# source file in place, else uploads an "(links)" copy next to it.
|
||||
```
|
||||
|
||||
| Subject | Verbs |
|
||||
|---|---|
|
||||
| `calendar` | `list`, `events`, `add`, `delete` |
|
||||
| `projects` | `list`, `get`, `milestones`, `create`, `update`, `delete`, **`files`** (`list`, `upload`, `download`, `rename`, `delete`) |
|
||||
| `projects` | `list`, `get`, `milestones`, `milestone-create`, `create`, `update`, `delete`, `contacts` (`add`, `remove`), `link-authors`, `link-git`, **`files`** (`list`, `upload`, `download`, `rename`, `delete`, `dedupe`, `as-md`, `put-md`, `put-txt`, `put-xlsx`) |
|
||||
| `tasks` | `list`, `get`, `create`, `update`, `delete`, `subtask add`, **`files`** (`list`, `upload`, `detach`) |
|
||||
| `users` | `list`, `self` (alias: `oo whoami`) |
|
||||
| `contacts` | `list`, `get`, `delete`, `info-add`, `merge`, `dedupe-info` |
|
||||
| `contacts` | `list`, `get`, `delete`, `info-add`, `merge`, `dedupe-info`, `tags`, `tag-add`, `tag-create`, `tag-remove` |
|
||||
| `persons` | `list`, `create`, `delete`, `dedupe` |
|
||||
| `companies` | `list`, `create`, `delete`, `dedupe`, `dedupe-persons` |
|
||||
| `opportunities` | `list`, `get`, `create`, `delete`, `stages`, `member-add`, `dedupe`, `dedupe-members`, `fix-titles` |
|
||||
| `opportunities` | `list`, `get`, `create`, `update`, `delete`, `stages`, `member-add`, `dedupe`, `dedupe-members`, `fix-titles` |
|
||||
| `invoices` | `list`, `get`, `create`, `update`, `pdf`, `pdf-cleanup`, `status`, `delete`, `items …` |
|
||||
| `crm` | `cleanup` |
|
||||
| `mails` | `accounts`, `folders`, `list`, `get`, `draft`, `attach`, `draft-invoice`, `delete` |
|
||||
| `mails` | `accounts`, `folders`, `list`, `get`, `download-attachment`, `draft`, `attach`, `draft-invoice`, `send`, `delete` |
|
||||
| `cases` | `list`, `create`, `delete`, `member-add` |
|
||||
| `crm-tasks` | `list`, `create`, `delete`, `categories` |
|
||||
| `crm-tasks` | `list`, `create`, `delete`, `categories`, `reassign-self` |
|
||||
| `docs` | `tools`, `convert`, `optimize`, `ocr`, `hocr`, `as-md`, `put-md`, `put-txt`, `put-xlsx` |
|
||||
| `catalog` | `match`, `merge`, `apply`, `scan-contacts`, `scan-projects`, `scan-thunderbird` |
|
||||
| `dav` | `ls`, `move`, `copy`, `mkdir`, `rename-file`, `rename-folder`, `download`, `fileops` |
|
||||
| `search` | `QUERY` (`--content`, `--folder ID`, `--limit N`, `--json`) |
|
||||
|
||||
The CLI reads only `.env` from the current working directory (godotenv is a
|
||||
CLI-only concern — the library itself never loads dotfiles).
|
||||
|
||||
@@ -11,6 +11,7 @@ package onlyoffice
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/cookiejar"
|
||||
"os"
|
||||
"strings"
|
||||
)
|
||||
@@ -35,8 +36,9 @@ type Client struct {
|
||||
|
||||
// NewClient returns a new Client backed by http.DefaultClient.
|
||||
func NewClient(c Credentials) *Client {
|
||||
jar, _ := cookiejar.New(nil)
|
||||
return &Client{
|
||||
client: http.DefaultClient,
|
||||
client: &http.Client{Jar: jar},
|
||||
credentials: &c,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,220 @@
|
||||
// Command kontoblatt builds a summary ("сводная таблица") of a Kontoblatt XLSX
|
||||
// (Datum, Gegenkonto, Buchungstext, Beleg, Soll, Haben, Bemerkung) and uploads
|
||||
// it back to the same OnlyOffice folder as the source file.
|
||||
//
|
||||
// Usage: kontoblatt <FILE_ID> <LOCAL_XLSX>
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
type agg struct {
|
||||
count int
|
||||
soll float64
|
||||
haben float64
|
||||
reFehlt int
|
||||
}
|
||||
|
||||
type rec struct {
|
||||
date, month, konto, text string
|
||||
soll, haben float64
|
||||
reFehlt bool
|
||||
}
|
||||
|
||||
var dateRe = regexp.MustCompile(`^\d{2}\.\d{2}\.\d{4}$`)
|
||||
|
||||
func parseAmount(s string) float64 {
|
||||
s = strings.TrimSpace(s)
|
||||
s = strings.ReplaceAll(s, "€", "")
|
||||
s = strings.ReplaceAll(s, " ", "")
|
||||
s = strings.ReplaceAll(s, ",", "") // German thousands separator
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return 0
|
||||
}
|
||||
v, err := strconv.ParseFloat(s, 64)
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
func cell(row []string, i int) string {
|
||||
if i < len(row) {
|
||||
return strings.TrimSpace(row[i])
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func main() {
|
||||
if len(os.Args) < 3 {
|
||||
fmt.Fprintln(os.Stderr, "usage: kontoblatt <FILE_ID> <LOCAL_XLSX>")
|
||||
os.Exit(2)
|
||||
}
|
||||
fileID, path := os.Args[1], os.Args[2]
|
||||
ctx := context.Background()
|
||||
|
||||
f, err := excelize.OpenFile(path)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
var recs []rec
|
||||
for _, sh := range f.GetSheetList() {
|
||||
rows, err := f.GetRows(sh)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
for _, r := range rows {
|
||||
d := cell(r, 0)
|
||||
if !dateRe.MatchString(d) {
|
||||
continue
|
||||
}
|
||||
text := cell(r, 2)
|
||||
recs = append(recs, rec{
|
||||
date: d,
|
||||
month: d[3:10],
|
||||
konto: cell(r, 1),
|
||||
text: text,
|
||||
soll: parseAmount(cell(r, 4)),
|
||||
haben: parseAmount(cell(r, 5)),
|
||||
reFehlt: strings.Contains(strings.ToUpper(text), "FEHLT"),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
byKonto := map[string]*agg{}
|
||||
byMonth := map[string]*agg{}
|
||||
getK := func(k string) *agg {
|
||||
if byKonto[k] == nil {
|
||||
byKonto[k] = &agg{}
|
||||
}
|
||||
return byKonto[k]
|
||||
}
|
||||
getM := func(k string) *agg {
|
||||
if byMonth[k] == nil {
|
||||
byMonth[k] = &agg{}
|
||||
}
|
||||
return byMonth[k]
|
||||
}
|
||||
var tot agg
|
||||
for _, r := range recs {
|
||||
k := getK(r.konto)
|
||||
k.count++
|
||||
k.soll += r.soll
|
||||
k.haben += r.haben
|
||||
if r.reFehlt {
|
||||
k.reFehlt++
|
||||
}
|
||||
m := getM(r.month)
|
||||
m.count++
|
||||
m.soll += r.soll
|
||||
m.haben += r.haben
|
||||
if r.reFehlt {
|
||||
m.reFehlt++
|
||||
}
|
||||
tot.count++
|
||||
tot.soll += r.soll
|
||||
tot.haben += r.haben
|
||||
if r.reFehlt {
|
||||
tot.reFehlt++
|
||||
}
|
||||
}
|
||||
|
||||
out := excelize.NewFile()
|
||||
defer out.Close()
|
||||
writeSheet(out, "Nach Gegenkonto", "Gegenkonto", byKonto, tot)
|
||||
writeSheet(out, "Nach Monat", "Monat", byMonth, tot)
|
||||
outPath := "/tmp/opencode/kontoblatt-zusammenfassung.xlsx"
|
||||
if err := out.SaveAs(outPath); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
|
||||
// upload next to the source file
|
||||
creds := onlyoffice.GetEnvironmentCredentials()
|
||||
c := onlyoffice.NewClient(creds)
|
||||
var src *onlyoffice.FileEntry
|
||||
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
var err error
|
||||
src, err = c.GetFile(ctx, fileID)
|
||||
return err
|
||||
}); derr != nil {
|
||||
panic(derr)
|
||||
}
|
||||
folder := ""
|
||||
if src.FolderID != nil {
|
||||
folder = src.FolderID.String()
|
||||
}
|
||||
title := ""
|
||||
if src.Title != nil {
|
||||
title = *src.Title
|
||||
}
|
||||
fmt.Printf("source: id=%s title=%q folder=%s\n", fileID, title, folder)
|
||||
|
||||
name := "Kontoblatt-1591-2025-Zusammenfassung.xlsx"
|
||||
tmp := "/tmp/opencode/" + name
|
||||
data, _ := os.ReadFile(outPath)
|
||||
if err := os.WriteFile(tmp, data, 0o600); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
var entry *onlyoffice.FileEntry
|
||||
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
var err error
|
||||
entry, _, err = c.UploadToFolderReplacing(ctx, folder, tmp)
|
||||
return err
|
||||
}); derr != nil {
|
||||
panic(derr)
|
||||
}
|
||||
fmt.Printf("uploaded: %s -> folder %s (id %v)\n", name, folder, entry.ID)
|
||||
|
||||
// print the summary
|
||||
printAgg("Nach Gegenkonto", byKonto, tot)
|
||||
printAgg("Nach Monat", byMonth, tot)
|
||||
}
|
||||
|
||||
func writeSheet(f *excelize.File, sheet, key string, m map[string]*agg, tot agg) {
|
||||
f.NewSheet(sheet)
|
||||
rows := [][]any{{key, "Anzahl", "Soll", "Haben", "Saldo", `davon "fehlt"`}}
|
||||
keys := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
for _, k := range keys {
|
||||
a := m[k]
|
||||
rows = append(rows, []any{k, a.count, a.soll, a.haben, a.soll - a.haben, a.reFehlt})
|
||||
}
|
||||
rows = append(rows, []any{"GESAMT", tot.count, tot.soll, tot.haben, tot.soll - tot.haben, tot.reFehlt})
|
||||
for i, row := range rows {
|
||||
for j, v := range row {
|
||||
cellRef, _ := excelize.CoordinatesToCellName(j+1, i+1)
|
||||
_ = f.SetCellValue(sheet, cellRef, v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func printAgg(title string, m map[string]*agg, tot agg) {
|
||||
fmt.Printf("\n== %s ==\n", title)
|
||||
keys := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
fmt.Printf("%-12s %6s %12s %12s %12s %7s\n", "key", "count", "soll", "haben", "saldo", "fehlt")
|
||||
for _, k := range keys {
|
||||
a := m[k]
|
||||
fmt.Printf("%-12s %6d %12.2f %12.2f %12.2f %7d\n", k, a.count, a.soll, a.haben, a.soll-a.haben, a.reFehlt)
|
||||
}
|
||||
fmt.Printf("%-12s %6d %12.2f %12.2f %12.2f %7d\n", "GESAMT", tot.count, tot.soll, tot.haben, tot.soll-tot.haben, tot.reFehlt)
|
||||
}
|
||||
@@ -0,0 +1,478 @@
|
||||
// Command kontolink fills the "Link" column of a Kontoblatt ("ungeklärte
|
||||
// Posten") XLSX by matching each row to an OnlyOffice document.
|
||||
//
|
||||
// Strategy (deterministic, conservative — no LLM):
|
||||
// 1. Beleg token (letters/digits from the "Beleg" column) appears in the file
|
||||
// title; among candidates prefer (a) the row's month, (b) real invoices over
|
||||
// copies/dupes, and require the result to be unique;
|
||||
// 2. else supplier + row month + "rechnung", again unique.
|
||||
//
|
||||
// A file is linked at most once (rows already carrying a link are kept and their
|
||||
// file counts as used). Ambiguous rows are left UNLINKED for manual review.
|
||||
//
|
||||
// Usage: kontolink <IN_XLSX> <INDEX_TSV> <OUT_XLSX>
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
var (
|
||||
dateRe = regexp.MustCompile(`^\d{2}\.\d{2}\.\d{4}$`)
|
||||
nonAln = regexp.MustCompile(`[^0-9a-z]+`)
|
||||
fileID = regexp.MustCompile(`fileid=(\d+)`)
|
||||
)
|
||||
|
||||
func parseDay(s string) (time.Time, bool) {
|
||||
t, err := time.Parse("02.01.2006", strings.TrimSpace(s))
|
||||
return t, err == nil
|
||||
}
|
||||
|
||||
func titleDay(title string) (time.Time, bool) {
|
||||
if len(title) >= 10 {
|
||||
if t, err := time.Parse("2006-01-02", title[:10]); err == nil {
|
||||
return t, true
|
||||
}
|
||||
}
|
||||
return time.Time{}, false
|
||||
}
|
||||
|
||||
// nearest picks the candidate whose title date is closest to rd. Ties and
|
||||
// undated candidates (when >1) are rejected.
|
||||
func nearest(cands []entry, rd time.Time) (entry, bool) {
|
||||
if len(cands) == 1 {
|
||||
return cands[0], true
|
||||
}
|
||||
best, bestD, tie := -1, 0.0, false
|
||||
for i, e := range cands {
|
||||
td, ok := titleDay(e.title)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
d := td.Sub(rd).Hours() / 24
|
||||
if d < 0 {
|
||||
d = -d
|
||||
}
|
||||
if best < 0 || d < bestD {
|
||||
best, bestD, tie = i, d, false
|
||||
} else if d == bestD {
|
||||
tie = true
|
||||
}
|
||||
}
|
||||
if best < 0 || tie {
|
||||
return entry{}, false
|
||||
}
|
||||
return cands[best], true
|
||||
}
|
||||
|
||||
const linkPrefix = "https://office.pro-dukt.de/Products/Files/DocEditor.aspx?fileid="
|
||||
|
||||
type entry struct {
|
||||
id, path, title, norm string
|
||||
}
|
||||
|
||||
func norm(s string) string { return nonAln.ReplaceAllString(strings.ToLower(s), "") }
|
||||
|
||||
func main() {
|
||||
if len(os.Args) < 4 {
|
||||
fmt.Fprintln(os.Stderr, "usage: kontolink <IN_XLSX> <INDEX_TSV> <OUT_XLSX>")
|
||||
os.Exit(2)
|
||||
}
|
||||
in, idxPath, out := os.Args[1], os.Args[2], os.Args[3]
|
||||
|
||||
idxRaw, err := os.ReadFile(idxPath)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
var entries []entry
|
||||
for _, line := range strings.Split(string(idxRaw), "\n") {
|
||||
parts := strings.Split(line, "\t")
|
||||
if len(parts) < 4 || parts[0] == "" {
|
||||
continue
|
||||
}
|
||||
entries = append(entries, entry{id: parts[0], path: parts[2], title: parts[3], norm: norm(parts[3])})
|
||||
}
|
||||
|
||||
f, err := excelize.OpenFile(in)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
defer f.Close()
|
||||
sheet := f.GetSheetList()[0]
|
||||
rows, err := f.GetRows(sheet)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
|
||||
// optional 5th arg: amounts TSV "file_id\ttitle\tamount" (see cmd/pdfamount)
|
||||
var amts []amtEntry
|
||||
if len(os.Args) >= 6 && os.Args[5] != "" {
|
||||
amts = loadAmounts(os.Args[5])
|
||||
}
|
||||
|
||||
used := map[string]bool{}
|
||||
for _, r := range rows {
|
||||
if m := fileID.FindStringSubmatch(cell(r, 7)); m != nil {
|
||||
used[m[1]] = true
|
||||
}
|
||||
}
|
||||
|
||||
var linked, byBeleg, bySupplier, byAmount, unmatched, ambiguous int
|
||||
for i, r := range rows {
|
||||
if i == 0 || !dateRe.MatchString(cell(r, 0)) || strings.TrimSpace(cell(r, 7)) != "" {
|
||||
continue
|
||||
}
|
||||
beleg := norm(cell(r, 3))
|
||||
supplier := supplierNorm(cell(r, 2))
|
||||
month := monthYear(cell(r, 0))
|
||||
rd, _ := parseDay(cell(r, 0))
|
||||
|
||||
e, kind, ok := pick(entries, used, beleg, supplier, month, rd)
|
||||
if !ok {
|
||||
if ae, aok := amountPick(amts, used, supplier, rowAmount(r), rd); aok {
|
||||
e, kind, ok = entry{id: ae.id, title: ae.title}, "amount", true
|
||||
}
|
||||
}
|
||||
if !ok {
|
||||
if beleg != "" {
|
||||
ambiguous++
|
||||
} else {
|
||||
unmatched++
|
||||
}
|
||||
continue
|
||||
}
|
||||
ref, _ := excelize.CoordinatesToCellName(8, i+1)
|
||||
if err := f.SetCellValue(sheet, ref, linkPrefix+e.id); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
used[e.id] = true
|
||||
linked++
|
||||
switch kind {
|
||||
case "beleg":
|
||||
byBeleg++
|
||||
case "supplier":
|
||||
bySupplier++
|
||||
case "amount":
|
||||
byAmount++
|
||||
}
|
||||
fmt.Printf("row %3d %-30s -> %s [%s]\n", i+1, cell(r, 2), e.title, kind)
|
||||
}
|
||||
|
||||
if err := f.SaveAs(out); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
fmt.Printf("\nlinked=%d (beleg=%d, supplier=%d, amount=%d), ambiguous=%d, no-candidate=%d\n",
|
||||
linked, byBeleg, bySupplier, byAmount, ambiguous, unmatched)
|
||||
|
||||
// Optional 4th arg: source OnlyOffice file id. Try to update it in place;
|
||||
// if it is locked (OnlyOffice 500), upload a "(links)" copy next to it.
|
||||
if len(os.Args) >= 5 && os.Args[4] != "" {
|
||||
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 120*time.Second)
|
||||
defer cancel()
|
||||
var src *onlyoffice.FileEntry
|
||||
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
var err error
|
||||
src, err = c.GetFile(ctx, os.Args[4])
|
||||
return err
|
||||
}); derr != nil {
|
||||
panic(derr)
|
||||
}
|
||||
folder, title := "", ""
|
||||
if src.FolderID != nil {
|
||||
folder = src.FolderID.String()
|
||||
}
|
||||
if src.Title != nil {
|
||||
title = *src.Title
|
||||
}
|
||||
uderr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
_, err := c.UpdateFile(ctx, os.Args[4], out)
|
||||
return err
|
||||
})
|
||||
if uderr == nil {
|
||||
fmt.Printf("updated file %s in place\n", os.Args[4])
|
||||
return
|
||||
}
|
||||
fmt.Printf("in-place update failed (locked?); uploading a copy to folder %s\n", folder)
|
||||
ext := filepath.Ext(title)
|
||||
name := strings.TrimSuffix(title, ext) + " (links)" + ext
|
||||
tmp := filepath.Join(os.TempDir(), name)
|
||||
data, _ := os.ReadFile(out)
|
||||
if err := os.WriteFile(tmp, data, 0o600); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
if derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
_, _, err := c.UploadToFolderReplacing(ctx, folder, tmp)
|
||||
return err
|
||||
}); derr != nil {
|
||||
panic(derr)
|
||||
}
|
||||
fmt.Printf("uploaded copy: %s -> folder %s\n", name, folder)
|
||||
}
|
||||
}
|
||||
|
||||
func cell(r []string, i int) string {
|
||||
if i < len(r) {
|
||||
return strings.TrimSpace(r[i])
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func supplierNorm(s string) string {
|
||||
s = strings.ToUpper(s)
|
||||
if i := strings.Index(s, ","); i >= 0 {
|
||||
s = s[:i]
|
||||
}
|
||||
for _, w := range []string{"RE FEHLT", "GS FEHLT", "WOFR", "WOFÜR"} {
|
||||
s = strings.ReplaceAll(s, w, "")
|
||||
}
|
||||
return norm(s)
|
||||
}
|
||||
|
||||
func monthYear(date string) string {
|
||||
if len(date) == 10 {
|
||||
return date[6:10] + "-" + date[3:5]
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// pick returns an unused candidate. Beleg match wins; supplier+month is a
|
||||
// fallback. When several candidates qualify, the one closest in time to the row
|
||||
// date wins; a tie is rejected (ambiguous) rather than guessed.
|
||||
func pick(entries []entry, used map[string]bool, beleg, supplier, month string, rd time.Time) (entry, string, bool) {
|
||||
free := func(e entry) bool { return !used[e.id] }
|
||||
|
||||
if len(beleg) >= 5 {
|
||||
var inMonth []entry
|
||||
for _, e := range entries {
|
||||
if free(e) && belegMatches(e.norm, beleg) &&
|
||||
(month == "" || strings.Contains(e.title, month)) {
|
||||
inMonth = append(inMonth, e)
|
||||
}
|
||||
}
|
||||
if supplier != "" {
|
||||
var s []entry
|
||||
for _, e := range inMonth {
|
||||
if strings.Contains(e.norm, supplier) {
|
||||
s = append(s, e)
|
||||
}
|
||||
}
|
||||
if len(s) > 0 {
|
||||
inMonth = s
|
||||
}
|
||||
}
|
||||
inMonth = topRank(inMonth)
|
||||
if e, ok := nearest(inMonth, rd); ok {
|
||||
return e, "beleg", true
|
||||
}
|
||||
// A Beleg is present but no file carries it: do NOT fall back to a
|
||||
// supplier guess (that links the wrong invoice).
|
||||
return entry{}, "", false
|
||||
}
|
||||
|
||||
if supplier != "" && month != "" {
|
||||
var c []entry
|
||||
for _, e := range entries {
|
||||
if free(e) && strings.Contains(e.norm, supplier) &&
|
||||
strings.Contains(e.title, month) && strings.Contains(e.norm, "rechnung") {
|
||||
c = append(c, e)
|
||||
}
|
||||
}
|
||||
c = topRank(c)
|
||||
if e, ok := nearest(c, rd); ok {
|
||||
return e, "supplier", true
|
||||
}
|
||||
}
|
||||
return entry{}, "", false
|
||||
}
|
||||
|
||||
// topRank keeps only the highest-ranked candidates (real invoice over copy /
|
||||
// dupe / op), so a tie with a duplicate does not mask the real file.
|
||||
func topRank(cands []entry) []entry {
|
||||
if len(cands) < 2 {
|
||||
return cands
|
||||
}
|
||||
best := 0
|
||||
for _, e := range cands {
|
||||
if rank(e) > best {
|
||||
best = rank(e)
|
||||
}
|
||||
}
|
||||
out := cands[:0]
|
||||
for _, e := range cands {
|
||||
if rank(e) == best {
|
||||
out = append(out, e)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func rank(e entry) int {
|
||||
s := 0
|
||||
if strings.Contains(e.path, "/2025") || strings.Contains(e.path, "/2024") {
|
||||
s += 4
|
||||
}
|
||||
if strings.Contains(e.norm, "rechnung") {
|
||||
s += 2
|
||||
}
|
||||
if strings.Contains(e.norm, "dupe") || strings.Contains(e.norm, "copy") ||
|
||||
strings.Contains(e.norm, "op") {
|
||||
s--
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// belegMatches reports whether a Beleg identifies the file: the whole normalized
|
||||
// Beleg appears, or (for long numeric Belege, e.g. "24/641393110") an 8-digit
|
||||
// window of its longest digit run appears.
|
||||
func belegMatches(titleNorm, beleg string) bool {
|
||||
if strings.Contains(titleNorm, beleg) {
|
||||
return true
|
||||
}
|
||||
run := longestDigitRun(beleg)
|
||||
for i := 0; i+8 <= len(run); i++ {
|
||||
if strings.Contains(titleNorm, run[i:i+8]) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func longestDigitRun(s string) string {
|
||||
var best, cur strings.Builder
|
||||
for _, r := range s {
|
||||
if r >= '0' && r <= '9' {
|
||||
cur.WriteRune(r)
|
||||
if cur.Len() > best.Len() {
|
||||
best.Reset()
|
||||
best.WriteString(cur.String())
|
||||
}
|
||||
} else {
|
||||
cur.Reset()
|
||||
}
|
||||
}
|
||||
return best.String()
|
||||
}
|
||||
|
||||
type amtEntry struct {
|
||||
id string
|
||||
title string
|
||||
norm string
|
||||
amount float64
|
||||
date time.Time
|
||||
hasDate bool
|
||||
}
|
||||
|
||||
func loadAmounts(path string) []amtEntry {
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
var out []amtEntry
|
||||
for _, line := range strings.Split(string(raw), "\n") {
|
||||
p := strings.Split(line, "\t")
|
||||
if len(p) < 3 {
|
||||
continue
|
||||
}
|
||||
v, err := strconv.ParseFloat(strings.TrimSpace(p[2]), 64)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
e := amtEntry{id: p[0], title: p[1], norm: norm(p[1]), amount: v}
|
||||
if len(p[1]) >= 10 {
|
||||
if t, err := time.Parse("2006-01-02", p[1][:10]); err == nil {
|
||||
e.date, e.hasDate = t, true
|
||||
}
|
||||
}
|
||||
out = append(out, e)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func rowAmount(r []string) float64 {
|
||||
if v := parseAmount(cell(r, 4)); v != 0 {
|
||||
return v
|
||||
}
|
||||
return parseAmount(cell(r, 5))
|
||||
}
|
||||
|
||||
func parseAmount(s string) float64 {
|
||||
s = strings.ReplaceAll(s, "€", "")
|
||||
s = strings.ReplaceAll(s, " ", "")
|
||||
s = strings.ReplaceAll(s, ",", ".")
|
||||
if s == "" {
|
||||
return 0
|
||||
}
|
||||
v, err := strconv.ParseFloat(s, 64)
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// amountPick matches a row to an O2 invoice by amount + nearest date. Scoped to
|
||||
// Telefonica/O2 rows and O2 files, so it cannot cross-link other suppliers.
|
||||
func amountPick(amts []amtEntry, used map[string]bool, supplier string, amt float64, rd time.Time) (amtEntry, bool) {
|
||||
if amt <= 0 || len(amts) == 0 {
|
||||
return amtEntry{}, false
|
||||
}
|
||||
if !strings.Contains(supplier, "telefonica") && !strings.Contains(supplier, "o2") {
|
||||
return amtEntry{}, false
|
||||
}
|
||||
var cands []amtEntry
|
||||
for _, a := range amts {
|
||||
if used[a.id] || !strings.Contains(a.norm, "o2") {
|
||||
continue
|
||||
}
|
||||
d := a.amount - amt
|
||||
if d < 0 {
|
||||
d = -d
|
||||
}
|
||||
if d > 0.005 {
|
||||
continue
|
||||
}
|
||||
if a.hasDate && !rd.IsZero() {
|
||||
days := a.date.Sub(rd).Hours() / 24
|
||||
if days < 0 {
|
||||
days = -days
|
||||
}
|
||||
if days > 75 {
|
||||
continue
|
||||
}
|
||||
}
|
||||
cands = append(cands, a)
|
||||
}
|
||||
if len(cands) == 1 {
|
||||
return cands[0], true
|
||||
}
|
||||
best, bestD, tie := -1, 0.0, false
|
||||
for i, a := range cands {
|
||||
if !a.hasDate {
|
||||
continue
|
||||
}
|
||||
d := a.date.Sub(rd).Hours() / 24
|
||||
if d < 0 {
|
||||
d = -d
|
||||
}
|
||||
if best < 0 || d < bestD {
|
||||
best, bestD, tie = i, d, false
|
||||
} else if d == bestD {
|
||||
tie = true
|
||||
}
|
||||
}
|
||||
if best < 0 || tie {
|
||||
return amtEntry{}, false
|
||||
}
|
||||
return cands[best], true
|
||||
}
|
||||
+291
@@ -0,0 +1,291 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
func init() {
|
||||
rootCmd.AddCommand(davCmd())
|
||||
}
|
||||
|
||||
// davCmd exposes the Documents module through the same Dav calls that back
|
||||
// oo-webdav (ListDavFolder / MoveDavItems / CopyDavItems / DownloadDavFile).
|
||||
// MoveDavItems sends resolveType=Skip + holdResult=true, which the legacy
|
||||
// fileops/move call without those params silently ignores (200 without move).
|
||||
func davCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "dav",
|
||||
Short: "Documents module by folder/file id (oo-webdav proven path)",
|
||||
}
|
||||
cmd.AddCommand(davLsCmd())
|
||||
cmd.AddCommand(davMoveCmd())
|
||||
cmd.AddCommand(davCopyCmd())
|
||||
cmd.AddCommand(davMkdirCmd())
|
||||
cmd.AddCommand(davRenameFileCmd())
|
||||
cmd.AddCommand(davRenameFolderCmd())
|
||||
cmd.AddCommand(davDownloadCmd())
|
||||
cmd.AddCommand(davFileOpsCmd())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func davLsCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "ls FOLDER_ID",
|
||||
Short: "List a Documents folder (@root for virtual sections)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
if args[0] == "@root" {
|
||||
sections, err := c.ListDavSections(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(sections))
|
||||
for _, s := range sections {
|
||||
rows = append(rows, map[string]any{
|
||||
"id": s.ID,
|
||||
"title": s.Title,
|
||||
"filesCount": s.FilesCount,
|
||||
"foldersCount": s.FoldersCount,
|
||||
})
|
||||
}
|
||||
printTable([]string{"id", "title", "filesCount", "foldersCount"}, rows)
|
||||
return nil
|
||||
}
|
||||
l, err := c.ListDavFolder(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if outputFormat == "json" {
|
||||
folders := make([]map[string]any, 0, len(l.Folders))
|
||||
for _, f := range l.Folders {
|
||||
folders = append(folders, map[string]any{
|
||||
"id": f.ID,
|
||||
"title": f.Title,
|
||||
"filesCount": f.FilesCount,
|
||||
"foldersCount": f.FoldersCount,
|
||||
})
|
||||
}
|
||||
files := make([]map[string]any, 0, len(l.Files))
|
||||
for _, f := range l.Files {
|
||||
files = append(files, map[string]any{
|
||||
"id": f.ID,
|
||||
"title": f.Title,
|
||||
"size": f.Size,
|
||||
"updated": f.Updated,
|
||||
})
|
||||
}
|
||||
printObject(map[string]any{"folders": folders, "files": files})
|
||||
return nil
|
||||
}
|
||||
if len(l.Folders) > 0 {
|
||||
frows := make([]map[string]any, 0, len(l.Folders))
|
||||
for _, f := range l.Folders {
|
||||
frows = append(frows, map[string]any{
|
||||
"id": f.ID,
|
||||
"title": f.Title,
|
||||
"filesCount": f.FilesCount,
|
||||
"foldersCount": f.FoldersCount,
|
||||
})
|
||||
}
|
||||
if outputFormat == "table" {
|
||||
fmt.Println("folders:")
|
||||
}
|
||||
printTable([]string{"id", "title", "filesCount", "foldersCount"}, frows)
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(l.Files))
|
||||
for _, f := range l.Files {
|
||||
rows = append(rows, map[string]any{
|
||||
"id": f.ID,
|
||||
"title": f.Title,
|
||||
"size": f.Size,
|
||||
"updated": f.Updated,
|
||||
})
|
||||
}
|
||||
if outputFormat == "table" {
|
||||
fmt.Println("files:")
|
||||
}
|
||||
printTable([]string{"id", "title", "size", "updated"}, rows)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
return cmd
|
||||
}
|
||||
|
||||
func davMoveCmd() *cobra.Command {
|
||||
var folderIDs []string
|
||||
cmd := &cobra.Command{
|
||||
Use: "move DEST_FOLDER_ID FILE_ID [FILE_ID...]",
|
||||
Short: "Move file(s) into a Documents folder (resolveType=Skip, holdResult)",
|
||||
Args: cobra.MinimumNArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := c.MoveDavItems(cmd.Context(), folderIDs, args[1:], args[0]); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"moved_files": args[1:], "moved_folders": folderIDs, "dest": args[0]})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringSliceVar(&folderIDs, "folders", nil, "folder ids to move along with the files")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func davCopyCmd() *cobra.Command {
|
||||
var folderIDs []string
|
||||
cmd := &cobra.Command{
|
||||
Use: "copy DEST_FOLDER_ID FILE_ID [FILE_ID...]",
|
||||
Short: "Copy file(s) into a Documents folder (conflictResolveType=Skip)",
|
||||
Args: cobra.MinimumNArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := c.CopyDavItems(cmd.Context(), folderIDs, args[1:], args[0]); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"copied_files": args[1:], "copied_folders": folderIDs, "dest": args[0]})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringSliceVar(&folderIDs, "folders", nil, "folder ids to copy along with the files")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func davMkdirCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "mkdir PARENT_FOLDER_ID TITLE",
|
||||
Short: "Create a subfolder in Documents",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
f, err := c.CreateDavFolder(cmd.Context(), args[0], args[1])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"id": f.ID, "title": f.Title, "parent": args[0]})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func davRenameFileCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "rename-file FILE_ID NEW_TITLE",
|
||||
Short: "Rename a Documents file (include extension in NEW_TITLE)",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := c.RenameDavFile(cmd.Context(), args[0], args[1]); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"id": args[0], "title": args[1]})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func davRenameFolderCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "rename-folder FOLDER_ID NEW_TITLE",
|
||||
Short: "Rename a Documents folder",
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := c.RenameDavFolder(cmd.Context(), args[0], args[1]); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"id": args[0], "title": args[1]})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func davDownloadCmd() *cobra.Command {
|
||||
var to string
|
||||
cmd := &cobra.Command{
|
||||
Use: "download FILE_ID",
|
||||
Short: "Download Documents file bytes (default path: ./<title>)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
f, err := c.GetFile(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
path := to
|
||||
if path == "" {
|
||||
path = onlyoffice.SafeLocalFileName(onlyoffice.FileEntryTitle(f))
|
||||
}
|
||||
out, err := os.Create(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer out.Close()
|
||||
n, err := c.DownloadDavFile(ctx, args[0], out)
|
||||
if err != nil {
|
||||
_ = os.Remove(path)
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"path": path, "bytes": n})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&to, "to", "", "output path (default: ./<server title>)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func davFileOpsCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "fileops",
|
||||
Short: "List active file operations (move/copy status polling)",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ops, err := c.ListFileOps(cmd.Context())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(ops))
|
||||
for _, op := range ops {
|
||||
rows = append(rows, map[string]any{
|
||||
"id": fmt.Sprint(op["id"]),
|
||||
"operation": fmt.Sprint(op["operation"]),
|
||||
"progress": fmt.Sprint(op["progress"]),
|
||||
"finished": fmt.Sprint(op["finished"]),
|
||||
"error": fmt.Sprint(op["error"]),
|
||||
})
|
||||
}
|
||||
printTable([]string{"id", "operation", "progress", "finished", "error"}, rows)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
+634
@@ -0,0 +1,634 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/eslider/go-onlyoffice/internal/docpipe"
|
||||
"github.com/eslider/go-onlyoffice/internal/xlspipe"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
func init() {
|
||||
rootCmd.AddCommand(docsCmd())
|
||||
}
|
||||
|
||||
func docsCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "docs",
|
||||
Short: "Local document pipeline: md↔docx, OCR→PDF, extract Markdown",
|
||||
Long: `Agent-friendly conversions (requires pandoc / ocrmypdf / pdftotext on PATH).
|
||||
|
||||
OnlyOffice Documents UI is poor for .md/.txt — keep sources in git, store .docx in OO.
|
||||
Upload Markdown as DOCX: oo docs put-md PROJECT_ID file.md
|
||||
Upload plain text: oo docs put-txt PROJECT_ID file.txt (preserves line breaks)
|
||||
Upload/generate XLSX: oo docs put-xlsx PROJECT_ID [--template cutover-portugal | FILE.xlsx]
|
||||
Read an OO file as MD: oo docs as-md FILE_ID
|
||||
OCR a scan locally: oo docs ocr scan.pdf --md out.md
|
||||
Structured OCR (hOCR→MD): oo docs hocr scan.jpg --md out.md --yaml out.yml`,
|
||||
}
|
||||
cmd.AddCommand(docsConvertCmd())
|
||||
cmd.AddCommand(docsOptimizeCmd())
|
||||
cmd.AddCommand(docsOCRCmd())
|
||||
cmd.AddCommand(docsHOCRCmd())
|
||||
cmd.AddCommand(docsAsMDCmd())
|
||||
cmd.AddCommand(docsPutMDCmd())
|
||||
cmd.AddCommand(docsPutTxtCmd())
|
||||
cmd.AddCommand(docsPutXlsxCmd())
|
||||
cmd.AddCommand(docsToolsCmd())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsToolsCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "tools",
|
||||
Short: "Show which converter binaries are on PATH",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
t := docpipe.LookPath()
|
||||
printObject(map[string]any{
|
||||
"pandoc": strOrNil(t.Pandoc),
|
||||
"ocrmypdf": strOrNil(t.OCRMyPDF),
|
||||
"pdftotext": strOrNil(t.PDFToText),
|
||||
"tesseract": strOrNil(t.Tesseract),
|
||||
"ghostscript": strOrNil(t.Ghostscript),
|
||||
})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func strOrNil(s string) any {
|
||||
if s == "" {
|
||||
return nil
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func docsConvertCmd() *cobra.Command {
|
||||
var to string
|
||||
cmd := &cobra.Command{
|
||||
Use: "convert PATH",
|
||||
Short: "Convert a local file with pandoc (md↔docx by default)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
in := args[0]
|
||||
t := docpipe.LookPath()
|
||||
out := to
|
||||
if out == "" {
|
||||
switch docpipe.Ext(in) {
|
||||
case ".md", ".markdown":
|
||||
out = docpipe.SiblingDOCX(in)
|
||||
case ".docx":
|
||||
out = strings.TrimSuffix(in, docpipe.Ext(in)) + ".md"
|
||||
default:
|
||||
return fmt.Errorf("--to required for input type %s", docpipe.Ext(in))
|
||||
}
|
||||
}
|
||||
if err := docpipe.EnsureDir(out); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := t.ConvertFile(in, out); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"in": in, "out": out})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&to, "to", "", "output path (default: sibling .docx or .md)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsOptimizeCmd() *cobra.Command {
|
||||
var out string
|
||||
cmd := &cobra.Command{
|
||||
Use: "optimize PDF_PATH",
|
||||
Short: "Rewrite PDF via Ghostscript (PostScript pdfwrite) for OO-friendly size/text",
|
||||
Long: `Use for InDesign/iText PDFs with a good text layer — avoids ocrmypdf invisible
|
||||
text overlays that break OnlyOffice DocEditor. Skips OCR; rewrites via gs pdfwrite.`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
in := args[0]
|
||||
t := docpipe.LookPath()
|
||||
if out == "" {
|
||||
base := strings.TrimSuffix(filepath.Base(in), filepath.Ext(in))
|
||||
out = filepath.Join(filepath.Dir(in), base+".optimized.pdf")
|
||||
}
|
||||
if err := docpipe.EnsureDir(out); err != nil {
|
||||
return err
|
||||
}
|
||||
chars, _ := t.PDFTextLayerChars(in)
|
||||
if err := t.OptimizePDF(in, out); err != nil {
|
||||
return err
|
||||
}
|
||||
outChars, _ := t.PDFTextLayerChars(out)
|
||||
printObject(map[string]any{
|
||||
"in": in,
|
||||
"pdf": out,
|
||||
"text_chars_in": chars,
|
||||
"text_chars_out": outChars,
|
||||
"note": "native text layer preserved; no OCR overlay",
|
||||
})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&out, "out", "", "output PDF (default: <name>.optimized.pdf)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsOCRCmd() *cobra.Command {
|
||||
var out, mdOut, lang string
|
||||
var force bool
|
||||
var writeMD bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "ocr PATH",
|
||||
Short: "OCR image/PDF → searchable PDF (and optional Markdown)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
in := args[0]
|
||||
t := docpipe.LookPath()
|
||||
if out == "" {
|
||||
base := strings.TrimSuffix(filepath.Base(in), filepath.Ext(in))
|
||||
out = filepath.Join(filepath.Dir(in), base+".ocr.pdf")
|
||||
}
|
||||
if err := docpipe.EnsureDir(out); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := t.OCRToPDF(in, out, force, lang); err != nil {
|
||||
return err
|
||||
}
|
||||
res := map[string]any{"in": in, "pdf": out}
|
||||
if writeMD || mdOut != "" {
|
||||
if mdOut == "" {
|
||||
mdOut = strings.TrimSuffix(out, filepath.Ext(out)) + ".md"
|
||||
}
|
||||
text, err := t.ExtractPDFText(out)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
body := "# " + filepath.Base(in) + "\n\n" + strings.TrimSpace(text) + "\n"
|
||||
if err := os.WriteFile(mdOut, []byte(body), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
res["md"] = mdOut
|
||||
}
|
||||
printObject(res)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&out, "out", "", "output searchable PDF (default: <name>.ocr.pdf)")
|
||||
cmd.Flags().StringVar(&mdOut, "md", "", "write Markdown extraction to this path")
|
||||
cmd.Flags().BoolVar(&writeMD, "markdown", false, "also write sibling .md next to OCR PDF")
|
||||
cmd.Flags().StringVar(&lang, "lang", "eng", "OCR language(s) for tesseract/ocrmypdf")
|
||||
cmd.Flags().BoolVar(&force, "force", false, "force OCR even if a text layer exists (default: skip when pdftotext finds enough text)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsHOCRCmd() *cobra.Command {
|
||||
var mdOut, yamlOut, hocrOut, lang string
|
||||
var dpi int
|
||||
var minConf float32
|
||||
cmd := &cobra.Command{
|
||||
Use: "hocr PATH",
|
||||
Short: "Tesseract hOCR → structured Markdown/YAML (via go-hocr)",
|
||||
Long: `Runs tesseract with hOCR output, parses with go-hocr, writes Markdown
|
||||
(and optional YAML). Better reading order / confidence than plain pdftotext.
|
||||
|
||||
For Spanish scans: --lang spa or spa+eng (needs tesseract-ocr-spa / TESSDATA_PREFIX).
|
||||
Phone photos: --dpi 200..300.`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
in := args[0]
|
||||
t := docpipe.LookPath()
|
||||
dir, err := os.MkdirTemp("", "oo-docs-hocr-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
res, err := t.ToHOCRMarkdown(in, dir, lang, dpi, minConf)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if mdOut == "" {
|
||||
base := strings.TrimSuffix(filepath.Base(in), filepath.Ext(in))
|
||||
mdOut = filepath.Join(filepath.Dir(in), base+".hocr.md")
|
||||
}
|
||||
if err := docpipe.EnsureDir(mdOut); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(mdOut, []byte(res.Markdown), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
obj := map[string]any{
|
||||
"in": in,
|
||||
"md": mdOut,
|
||||
"did_ocr": res.DidOCR,
|
||||
}
|
||||
if hocrOut != "" {
|
||||
if err := docpipe.EnsureDir(hocrOut); err != nil {
|
||||
return err
|
||||
}
|
||||
b, err := os.ReadFile(res.HOCRPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(hocrOut, b, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
obj["hocr"] = hocrOut
|
||||
} else {
|
||||
obj["hocr_tmp"] = res.HOCRPath
|
||||
}
|
||||
if yamlOut != "" {
|
||||
if err := docpipe.EnsureDir(yamlOut); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(yamlOut, []byte(res.YAML), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
obj["yaml"] = yamlOut
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&mdOut, "md", "", "Markdown output (default: <name>.hocr.md)")
|
||||
cmd.Flags().StringVar(&yamlOut, "yaml", "", "also write structured YAML from go-hocr")
|
||||
cmd.Flags().StringVar(&hocrOut, "hocr", "", "also keep raw .hocr file at this path")
|
||||
cmd.Flags().StringVar(&lang, "lang", "eng", "tesseract language(s), e.g. spa+eng")
|
||||
cmd.Flags().IntVar(&dpi, "dpi", 220, "hint DPI for phone photos / scans (0 = tesseract default)")
|
||||
cmd.Flags().Float32Var(&minConf, "min-conf", 0, "drop words with OCR confidence below this (0 = keep all)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsAsMDCmd() *cobra.Command {
|
||||
var to, lang string
|
||||
var minChars int
|
||||
var uploadOCR bool
|
||||
var useHOCR bool
|
||||
var dpi int
|
||||
var minConf float32
|
||||
cmd := &cobra.Command{
|
||||
Use: "as-md FILE_ID",
|
||||
Short: "Download an OO Documents file and emit Markdown (OCR PDF/image if needed)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
meta, err := c.GetFile(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
title := onlyoffice.FileEntryTitle(meta)
|
||||
dir, err := os.MkdirTemp("", "oo-docs-as-md-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
local := filepath.Join(dir, onlyoffice.SafeLocalFileName(title))
|
||||
f, err := os.Create(local)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := c.DownloadFile(ctx, args[0], f); err != nil {
|
||||
_ = f.Close()
|
||||
return err
|
||||
}
|
||||
_ = f.Close()
|
||||
|
||||
tools := docpipe.LookPath()
|
||||
var res docpipe.Result
|
||||
if useHOCR {
|
||||
hr, err := tools.ToHOCRMarkdown(local, dir, lang, dpi, minConf)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
res = docpipe.Result{Markdown: hr.Markdown, DidOCR: hr.DidOCR, Source: hr.Source}
|
||||
} else {
|
||||
res, err = tools.ToMarkdown(local, dir, lang, minChars)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
outPath := to
|
||||
if outPath == "" {
|
||||
base := strings.TrimSuffix(onlyoffice.SafeLocalFileName(title), filepath.Ext(onlyoffice.SafeLocalFileName(title)))
|
||||
if base == "" || base == "download" {
|
||||
base = "file-" + args[0]
|
||||
}
|
||||
outPath = base + ".md"
|
||||
}
|
||||
if err := docpipe.EnsureDir(outPath); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(outPath, []byte(res.Markdown), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
obj := map[string]any{
|
||||
"file_id": args[0],
|
||||
"title": title,
|
||||
"md": outPath,
|
||||
"did_ocr": res.DidOCR,
|
||||
}
|
||||
if res.OCRPDFPath != "" && uploadOCR {
|
||||
folderID := onlyoffice.FileFolderID(meta)
|
||||
upName := strings.TrimSuffix(onlyoffice.SafeLocalFileName(title), filepath.Ext(onlyoffice.SafeLocalFileName(title))) + ".ocr.pdf"
|
||||
tmpUp := filepath.Join(dir, upName)
|
||||
data, err := os.ReadFile(res.OCRPDFPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(tmpUp, data, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
if folderID == "" {
|
||||
obj["ocr_pdf_local"] = res.OCRPDFPath
|
||||
obj["note"] = "file has no folderId; OCR PDF left local — pass after moving into a folder"
|
||||
} else {
|
||||
ent, err := c.UploadToFolder(ctx, folderID, tmpUp)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
obj["ocr_pdf_file_id"] = fileIDStr(ent)
|
||||
obj["ocr_pdf_title"] = onlyoffice.FileEntryTitle(ent)
|
||||
}
|
||||
} else if res.OCRPDFPath != "" {
|
||||
// Keep OCR PDF outside temp by copying beside md if requested via env-less default:
|
||||
kept := strings.TrimSuffix(outPath, filepath.Ext(outPath)) + ".ocr.pdf"
|
||||
if b, err := os.ReadFile(res.OCRPDFPath); err == nil {
|
||||
_ = os.WriteFile(kept, b, 0o644)
|
||||
obj["ocr_pdf_local"] = kept
|
||||
} else {
|
||||
obj["ocr_pdf_local"] = res.OCRPDFPath
|
||||
}
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&to, "to", "", "write Markdown to this path (default: ./<title>.md)")
|
||||
cmd.Flags().StringVar(&lang, "lang", "eng", "OCR language")
|
||||
cmd.Flags().IntVar(&minChars, "min-chars", docpipe.DefaultMinTextChars, "OCR PDF if text layer shorter than this")
|
||||
cmd.Flags().BoolVar(&uploadOCR, "upload-ocr", false, "upload searchable OCR PDF back into the same OO folder")
|
||||
cmd.Flags().BoolVar(&useHOCR, "hocr", false, "use tesseract hOCR + go-hocr instead of ocrmypdf/pdftotext")
|
||||
cmd.Flags().IntVar(&dpi, "dpi", 220, "DPI hint when --hocr (phone photos)")
|
||||
cmd.Flags().Float32Var(&minConf, "min-conf", 0, "drop low-confidence words when --hocr")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsPutMDCmd() *cobra.Command {
|
||||
var folderID string
|
||||
var keepLocalDOCX string
|
||||
var replace bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "put-md PROJECT_ID MARKDOWN_PATH",
|
||||
Short: "Convert Markdown→DOCX and upload DOCX into a project (OO-friendly)",
|
||||
Long: `Agents edit .md locally; this uploads .docx so OnlyOffice can open/version it.`,
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
pid, mdPath := args[0], args[1]
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
tools := docpipe.LookPath()
|
||||
dir, err := os.MkdirTemp("", "oo-docs-put-md-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
docxName := strings.TrimSuffix(filepath.Base(mdPath), filepath.Ext(mdPath)) + ".docx"
|
||||
docxPath := filepath.Join(dir, docxName)
|
||||
if err := tools.MDToDOCX(mdPath, docxPath); err != nil {
|
||||
return err
|
||||
}
|
||||
if keepLocalDOCX != "" {
|
||||
if err := docpipe.EnsureDir(keepLocalDOCX); err != nil {
|
||||
return err
|
||||
}
|
||||
b, err := os.ReadFile(docxPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(keepLocalDOCX, b, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
ent, deleted, err := uploadProjectDoc(ctx, c, pid, docxPath, folderID, replace)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
obj := map[string]any{
|
||||
"project_id": pid,
|
||||
"md": mdPath,
|
||||
"uploaded": fileEntryToMap(ent),
|
||||
}
|
||||
if folderID != "" {
|
||||
obj["folder_id"] = folderID
|
||||
}
|
||||
if len(deleted) > 0 {
|
||||
obj["replaced_file_ids"] = deleted
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&folderID, "folder", "", "Documents folder id (default: project root)")
|
||||
cmd.Flags().StringVar(&keepLocalDOCX, "keep-docx", "", "also write the generated DOCX to this local path")
|
||||
cmd.Flags().BoolVar(&replace, "replace", true, "replace same stem|ext before upload (default); false = fail if name taken")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsPutTxtCmd() *cobra.Command {
|
||||
var folderID string
|
||||
var keepLocalDOCX string
|
||||
var replace bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "put-txt PROJECT_ID TEXT_PATH",
|
||||
Short: "Convert plain text→DOCX (preserve line breaks) and upload into a project",
|
||||
Long: `OnlyOffice cannot render .txt well. This keeps each source line on its own DOCX line.`,
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
pid, txtPath := args[0], args[1]
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
tools := docpipe.LookPath()
|
||||
dir, err := os.MkdirTemp("", "oo-docs-put-txt-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
docxName := strings.TrimSuffix(filepath.Base(txtPath), filepath.Ext(txtPath)) + ".docx"
|
||||
docxPath := filepath.Join(dir, docxName)
|
||||
if err := tools.TXTToDOCX(txtPath, docxPath); err != nil {
|
||||
return err
|
||||
}
|
||||
if keepLocalDOCX != "" {
|
||||
if err := docpipe.EnsureDir(keepLocalDOCX); err != nil {
|
||||
return err
|
||||
}
|
||||
b, err := os.ReadFile(docxPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(keepLocalDOCX, b, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
ent, deleted, err := uploadProjectDoc(ctx, c, pid, docxPath, folderID, replace)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
obj := map[string]any{
|
||||
"project_id": pid,
|
||||
"txt": txtPath,
|
||||
"uploaded": fileEntryToMap(ent),
|
||||
}
|
||||
if folderID != "" {
|
||||
obj["folder_id"] = folderID
|
||||
}
|
||||
if len(deleted) > 0 {
|
||||
obj["replaced_file_ids"] = deleted
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&folderID, "folder", "", "Documents folder id (default: project root)")
|
||||
cmd.Flags().StringVar(&keepLocalDOCX, "keep-docx", "", "also write the generated DOCX to this local path")
|
||||
cmd.Flags().BoolVar(&replace, "replace", true, "replace same stem|ext before upload (default); false = fail if name taken")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsPutXlsxCmd() *cobra.Command {
|
||||
var folderID, template, title, keepLocal string
|
||||
var replace bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "put-xlsx PROJECT_ID [LOCAL_XLSX]",
|
||||
Short: "Upload or generate an XLSX workbook into a project (excelize templates with formulas)",
|
||||
Long: `Spreadsheets live in OnlyOffice — not in git. Generate multi-sheet workbooks with
|
||||
formulas (SUM/AVG, cross-sheet refs, named inputs) via --template, or upload an existing .xlsx.
|
||||
|
||||
oo docs put-xlsx 218 --template cutover-portugal
|
||||
oo docs put-xlsx 218 --template cutover-portugal --title 2026-08-28-cutover-budget.xlsx
|
||||
oo docs put-xlsx 218 ./my.xlsx`,
|
||||
Args: cobra.RangeArgs(1, 2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
pid := args[0]
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dir, err := os.MkdirTemp("", "oo-docs-put-xlsx-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
var xlsxPath string
|
||||
var srcLabel string
|
||||
switch {
|
||||
case template != "":
|
||||
wb, err := xlspipe.BuildTemplate(template)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
name := title
|
||||
if name == "" {
|
||||
name = "cutover-budget.xlsx"
|
||||
if template == xlspipe.TemplateCutoverPortugal {
|
||||
name = "2026-08-28-cutover-budget.xlsx"
|
||||
}
|
||||
}
|
||||
if !strings.HasSuffix(strings.ToLower(name), ".xlsx") {
|
||||
name += ".xlsx"
|
||||
}
|
||||
xlsxPath = filepath.Join(dir, name)
|
||||
if err := xlspipe.Save(wb, xlsxPath); err != nil {
|
||||
wb.Close()
|
||||
return err
|
||||
}
|
||||
wb.Close()
|
||||
srcLabel = "template:" + template
|
||||
case len(args) == 2:
|
||||
xlsxPath = args[1]
|
||||
if docpipe.Ext(xlsxPath) != ".xlsx" {
|
||||
return fmt.Errorf("expected .xlsx, got %s", docpipe.Ext(xlsxPath))
|
||||
}
|
||||
srcLabel = xlsxPath
|
||||
default:
|
||||
return fmt.Errorf("pass LOCAL_XLSX or --template")
|
||||
}
|
||||
|
||||
if keepLocal != "" {
|
||||
b, err := os.ReadFile(xlsxPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := docpipe.EnsureDir(keepLocal); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(keepLocal, b, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
ctx := cmd.Context()
|
||||
ent, deleted, err := uploadProjectDoc(ctx, c, pid, xlsxPath, folderID, replace)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
obj := map[string]any{
|
||||
"project_id": pid,
|
||||
"source": srcLabel,
|
||||
"uploaded": fileEntryToMap(ent),
|
||||
}
|
||||
if folderID != "" {
|
||||
obj["folder_id"] = folderID
|
||||
}
|
||||
if len(deleted) > 0 {
|
||||
obj["replaced_file_ids"] = deleted
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&folderID, "folder", "", "Documents folder id (default: project root)")
|
||||
cmd.Flags().StringVar(&template, "template", "", "built-in workbook template (cutover-portugal)")
|
||||
cmd.Flags().StringVar(&title, "title", "", "upload file name when using --template")
|
||||
cmd.Flags().StringVar(&keepLocal, "keep-xlsx", "", "also write generated/uploaded bytes to this local path")
|
||||
cmd.Flags().BoolVar(&replace, "replace", true, "replace same stem|ext before upload (default); false = fail if name taken")
|
||||
return cmd
|
||||
}
|
||||
|
||||
// uploadProjectDoc upserts (--replace, default) or no-clobbers into project/folder Documents.
|
||||
func uploadProjectDoc(ctx context.Context, c *onlyoffice.Client, pid, localPath, folderID string, replace bool) (*onlyoffice.FileEntry, []int, error) {
|
||||
if folderID != "" {
|
||||
if replace {
|
||||
return c.UploadToFolderReplacing(ctx, folderID, localPath)
|
||||
}
|
||||
if err := c.AssertNoFileConflict(ctx, folderID, localPath); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
ent, err := c.UploadToFolder(ctx, folderID, localPath)
|
||||
return ent, nil, err
|
||||
}
|
||||
if replace {
|
||||
return c.UploadProjectFileReplacing(ctx, pid, localPath)
|
||||
}
|
||||
ent, err := c.UploadProjectFileNoClobber(ctx, pid, localPath)
|
||||
return ent, nil, err
|
||||
}
|
||||
+116
@@ -22,9 +22,11 @@ func init() {
|
||||
mailsCmd.AddCommand(mailsFoldersCmd())
|
||||
mailsCmd.AddCommand(mailsListCmd())
|
||||
mailsCmd.AddCommand(mailsGetCmd())
|
||||
mailsCmd.AddCommand(mailsDownloadAttachmentCmd())
|
||||
mailsCmd.AddCommand(mailsDraftCmd())
|
||||
mailsCmd.AddCommand(mailsAttachCmd())
|
||||
mailsCmd.AddCommand(mailsDraftInvoiceCmd())
|
||||
mailsCmd.AddCommand(mailsSendCmd())
|
||||
mailsCmd.AddCommand(mailsDeleteCmd())
|
||||
}
|
||||
|
||||
@@ -131,6 +133,48 @@ func mailsGetCmd() *cobra.Command {
|
||||
}
|
||||
}
|
||||
|
||||
func mailsDownloadAttachmentCmd() *cobra.Command {
|
||||
var outPath string
|
||||
cmd := &cobra.Command{
|
||||
Use: "download-attachment ATTACHMENT_ID",
|
||||
Short: "Download a mail attachment by attachment id",
|
||||
Long: `Download a raw attachment from OnlyOffice Mail's download.ashx handler.
|
||||
|
||||
Example:
|
||||
oo mails download-attachment 12345 --out /tmp/attach.bin
|
||||
`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if strings.TrimSpace(outPath) == "" {
|
||||
return fmt.Errorf("--out is required")
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
body, err := c.DownloadMailAttachment(cmd.Context(), args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := writeMailAttachment(outPath, body); err != nil {
|
||||
return err
|
||||
}
|
||||
if outputFormat == "json" {
|
||||
printObject(map[string]any{
|
||||
"attachmentId": args[0],
|
||||
"bytes": len(body),
|
||||
"path": outPath,
|
||||
})
|
||||
return nil
|
||||
}
|
||||
fmt.Printf("saved %d bytes to %s\n", len(body), outPath)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&outPath, "out", "", "output file path")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func mailsDraftCmd() *cobra.Command {
|
||||
var from, to, cc, bcc, subject, body, html string
|
||||
var id int64
|
||||
@@ -297,6 +341,78 @@ func formatInvoiceCostEUR(v any) string {
|
||||
return s
|
||||
}
|
||||
|
||||
func writeMailAttachment(path string, body []byte) error {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return fmt.Errorf("attachment output path is required")
|
||||
}
|
||||
return os.WriteFile(path, body, 0o644)
|
||||
}
|
||||
|
||||
func mailsSendCmd() *cobra.Command {
|
||||
var from, to, cc, bcc, subject, body, html string
|
||||
var id int64
|
||||
cmd := &cobra.Command{
|
||||
Use: "send",
|
||||
Short: "Send a mail message (OnlyOffice Mail)",
|
||||
Long: `Send via PUT /api/2.0/mail/messages/send.json.
|
||||
|
||||
oo mails send --id 7803 --body "…" # send referencing a draft id
|
||||
oo mails send --to a@b.com --subject "…" --body "…"
|
||||
oo mails send --id 7803 --to a@b.com --subject "…" --body "…" --cc x@y.com
|
||||
|
||||
IMPORTANT: send.json does NOT copy subject/body from the referenced draft — the
|
||||
content must be in this request (--subject/--body). Cc/Bcc are omitted when empty
|
||||
(the API 400s on empty strings). The API send does not append the UI signature —
|
||||
put the chat line in --body if needed.
|
||||
`,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if to == "" && id == 0 {
|
||||
return fmt.Errorf("--to is required (or --id of an existing draft)")
|
||||
}
|
||||
htmlBody := body
|
||||
if html != "" {
|
||||
htmlBody = html
|
||||
}
|
||||
if htmlBody == "" && id != 0 {
|
||||
// The send.json endpoint does NOT copy subject/body from the
|
||||
// referenced draft — an empty body here sends an empty message.
|
||||
// Warn instead of silently mailing an empty email.
|
||||
return fmt.Errorf("--body/--html is required when sending by --id (send.json needs the content in the request)")
|
||||
}
|
||||
if htmlBody == "" && to == "" {
|
||||
return fmt.Errorf("--body is required for a fresh message")
|
||||
}
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
raw, err := c.SendMail(cmd.Context(), onlyoffice.SendMailParams{
|
||||
ID: id,
|
||||
From: from,
|
||||
To: to,
|
||||
Cc: cc,
|
||||
Bcc: bcc,
|
||||
Subject: subject,
|
||||
Body: htmlBody,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Println(string(raw))
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().Int64Var(&id, "id", 0, "existing draft id to send (0 = fresh message)")
|
||||
cmd.Flags().StringVar(&from, "from", "", "from address (default: first enabled mailbox)")
|
||||
cmd.Flags().StringVar(&to, "to", "", "recipient (required unless --id)")
|
||||
cmd.Flags().StringVar(&cc, "cc", "", "cc")
|
||||
cmd.Flags().StringVar(&bcc, "bcc", "", "bcc")
|
||||
cmd.Flags().StringVar(&subject, "subject", "", "subject")
|
||||
cmd.Flags().StringVar(&body, "body", "", "plain text or HTML body")
|
||||
cmd.Flags().StringVar(&html, "html", "", "HTML body (alias of --body when set)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func mailsDeleteCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "delete ID [ID...]",
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestWriteMailAttachment(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "attach.bin")
|
||||
body := []byte("payload")
|
||||
if err := writeMailAttachment(path, body); err != nil {
|
||||
t.Fatalf("writeMailAttachment: %v", err)
|
||||
}
|
||||
got, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadFile: %v", err)
|
||||
}
|
||||
if string(got) != string(body) {
|
||||
t.Fatalf("body = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteMailAttachmentRequiresPath(t *testing.T) {
|
||||
if err := writeMailAttachment("", []byte("x")); err == nil {
|
||||
t.Fatal("expected error for empty path")
|
||||
}
|
||||
}
|
||||
+9
-5
@@ -3,18 +3,22 @@
|
||||
// Command tree is subject-based (mirrors the library split and the `tea` CLI):
|
||||
//
|
||||
// oo calendar list | events | add | delete
|
||||
// oo projects list | get | milestones | create | update | delete | files (list|upload|download|rename|delete)
|
||||
// oo projects list | get | milestones | milestone-create | create | update | delete | contacts (add|remove) | link-authors | link-git | files (list|upload|download|rename|delete|dedupe|as-md|put-md|put-txt|put-xlsx)
|
||||
// oo tasks list | get | create | update | delete | subtask add | files (list|upload|detach)
|
||||
// oo users list | self (alias: oo whoami)
|
||||
// oo contacts list | get | delete | info-add | merge | dedupe-info
|
||||
// oo contacts list | get | delete | info-add | merge | dedupe-info | tags | tag-add | tag-create | tag-remove
|
||||
// oo persons list | create | delete | dedupe
|
||||
// oo companies list | create | delete | dedupe | dedupe-persons
|
||||
// oo opportunities list | get | create | delete | stages | member-add | dedupe | dedupe-members | fix-titles
|
||||
// oo opportunities list | get | create | update | delete | stages | member-add | dedupe | dedupe-members | fix-titles
|
||||
// oo cases list | create | delete | member-add
|
||||
// oo crm-tasks list | create | delete | categories
|
||||
// oo crm-tasks list | create | delete | categories | reassign-self
|
||||
// oo crm cleanup
|
||||
// oo mails accounts | folders | list | get | draft | attach | draft-invoice | delete
|
||||
// oo mails accounts | folders | list | get | download-attachment | draft | attach | draft-invoice | send | delete
|
||||
// oo invoices list | get | create | update | pdf | pdf-cleanup | status | delete | items …
|
||||
// oo docs tools | convert | optimize | ocr | hocr | as-md | put-md | put-txt | put-xlsx
|
||||
// oo catalog match | merge | apply | scan-contacts | scan-projects | scan-thunderbird
|
||||
// oo dav ls | move | copy | mkdir | rename-file | rename-folder | download | fileops
|
||||
// oo search QUERY [--content] [--folder ID] [--limit N] [--json]
|
||||
//
|
||||
// CRM association rules: docs/crm-associations.md
|
||||
//
|
||||
|
||||
+116
-7
@@ -24,9 +24,43 @@ func projectFilesCmd() *cobra.Command {
|
||||
cmd.AddCommand(prjFilesDownloadCmd())
|
||||
cmd.AddCommand(prjFilesRenameCmd())
|
||||
cmd.AddCommand(prjFilesDeleteCmd())
|
||||
cmd.AddCommand(prjFilesDedupeCmd())
|
||||
// Convenience aliases into oo docs (md↔docx / OCR pipeline).
|
||||
cmd.AddCommand(aliasDocsAsMD())
|
||||
cmd.AddCommand(aliasDocsPutMD())
|
||||
cmd.AddCommand(aliasDocsPutTxt())
|
||||
cmd.AddCommand(aliasDocsPutXlsx())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func aliasDocsAsMD() *cobra.Command {
|
||||
c := docsAsMDCmd()
|
||||
c.Use = "as-md FILE_ID"
|
||||
c.Short = "Alias of `oo docs as-md` — download OO file as Markdown (OCR if needed)"
|
||||
return c
|
||||
}
|
||||
|
||||
func aliasDocsPutMD() *cobra.Command {
|
||||
c := docsPutMDCmd()
|
||||
c.Use = "put-md PROJECT_ID MARKDOWN_PATH"
|
||||
c.Short = "Alias of `oo docs put-md` — Markdown→DOCX upload into project"
|
||||
return c
|
||||
}
|
||||
|
||||
func aliasDocsPutTxt() *cobra.Command {
|
||||
c := docsPutTxtCmd()
|
||||
c.Use = "put-txt PROJECT_ID TEXT_PATH"
|
||||
c.Short = "Alias of `oo docs put-txt` — plain text→DOCX upload into project"
|
||||
return c
|
||||
}
|
||||
|
||||
func aliasDocsPutXlsx() *cobra.Command {
|
||||
c := docsPutXlsxCmd()
|
||||
c.Use = "put-xlsx PROJECT_ID [LOCAL_XLSX]"
|
||||
c.Short = "Alias of `oo docs put-xlsx` — generate/upload XLSX with formulas"
|
||||
return c
|
||||
}
|
||||
|
||||
func prjFilesListCmd() *cobra.Command {
|
||||
var showFolders bool
|
||||
cmd := &cobra.Command{
|
||||
@@ -49,8 +83,8 @@ func prjFilesListCmd() *cobra.Command {
|
||||
continue
|
||||
}
|
||||
frows = append(frows, map[string]any{
|
||||
"id": folderIDStr(f),
|
||||
"title": derefString(f.Title),
|
||||
"id": folderIDStr(f),
|
||||
"title": derefString(f.Title),
|
||||
"filesCount": derefInt(f.FilesCount),
|
||||
"foldersCount": derefInt(f.FoldersCount),
|
||||
})
|
||||
@@ -73,10 +107,13 @@ func prjFilesListCmd() *cobra.Command {
|
||||
}
|
||||
|
||||
func prjFilesUploadCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
var replace, allowDuplicate bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "upload PROJECT_ID LOCAL_PATH [LOCAL_PATH...]",
|
||||
Short: "Upload file(s) into the project's Documents folder",
|
||||
Args: cobra.MinimumNArgs(2),
|
||||
Short: "Upload file(s) into the project's Documents folder (upsert by stem|ext)",
|
||||
Long: `Default: replace an existing file with the same logical name (stem|ext), like cp overwrite.
|
||||
Pass --no-replace to fail when the name is taken; --allow-duplicate to always create a new file id.`,
|
||||
Args: cobra.MinimumNArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
@@ -84,15 +121,31 @@ func prjFilesUploadCmd() *cobra.Command {
|
||||
}
|
||||
pid := args[0]
|
||||
for _, p := range args[1:] {
|
||||
entry, err := c.UploadProjectFile(cmd.Context(), pid, p)
|
||||
var entry *onlyoffice.FileEntry
|
||||
var deleted []int
|
||||
switch {
|
||||
case allowDuplicate:
|
||||
entry, err = c.UploadProjectFile(cmd.Context(), pid, p)
|
||||
case replace:
|
||||
entry, deleted, err = c.UploadProjectFileReplacing(cmd.Context(), pid, p)
|
||||
default:
|
||||
entry, err = c.UploadProjectFileNoClobber(cmd.Context(), pid, p)
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(fileEntryToMap(entry))
|
||||
obj := fileEntryToMap(entry)
|
||||
if len(deleted) > 0 {
|
||||
obj["replaced_file_ids"] = deleted
|
||||
}
|
||||
printObject(obj)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().BoolVar(&replace, "replace", true, "replace same stem|ext in project folder (default)")
|
||||
cmd.Flags().BoolVar(&allowDuplicate, "allow-duplicate", false, "always create a new file even when the name exists")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func prjFilesDownloadCmd() *cobra.Command {
|
||||
@@ -184,6 +237,62 @@ func prjFilesDeleteCmd() *cobra.Command {
|
||||
}
|
||||
}
|
||||
|
||||
func prjFilesDedupeCmd() *cobra.Command {
|
||||
var apply, cross bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "dedupe PROJECT_ID",
|
||||
Short: "Find (and optionally remove) duplicate files in project Documents folders",
|
||||
Long: `Duplicates share the same logical name: stem|ext (OO title+fileExst).
|
||||
|
||||
Default: dry-run report. Pass --apply to delete older copies (keeps newest; --cross prefers non-trash folders).`,
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
groups, deleted, err := c.DedupeProject(cmd.Context(), args[0], onlyoffice.DedupOptions{
|
||||
CrossFolder: cross,
|
||||
}, apply)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(groups))
|
||||
for _, g := range groups {
|
||||
row := map[string]any{
|
||||
"key": g.Key,
|
||||
"folder_id": g.FolderID,
|
||||
"folder_title": g.FolderTitle,
|
||||
"keep_id": fileIDStr(g.Keep),
|
||||
"keep_title": onlyoffice.FileEntryTitle(g.Keep),
|
||||
"remove_count": len(g.Remove),
|
||||
}
|
||||
removeIDs := make([]string, 0, len(g.Remove))
|
||||
for _, f := range g.Remove {
|
||||
removeIDs = append(removeIDs, fileIDStr(f))
|
||||
}
|
||||
row["remove_ids"] = removeIDs
|
||||
rows = append(rows, row)
|
||||
}
|
||||
out := map[string]any{
|
||||
"project_id": args[0],
|
||||
"dry_run": !apply,
|
||||
"cross": cross,
|
||||
"groups": len(groups),
|
||||
"duplicates": rows,
|
||||
}
|
||||
if apply {
|
||||
out["deleted_ids"] = deleted
|
||||
}
|
||||
printObject(out)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().BoolVar(&apply, "apply", false, "delete duplicate files (default: report only)")
|
||||
cmd.Flags().BoolVar(&cross, "cross", false, "also dedupe same stem|ext across folders (prefers non-_trash)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func fileEntryRows(files []*onlyoffice.FileEntry) []map[string]any {
|
||||
rows := make([]map[string]any, 0, len(files))
|
||||
for _, f := range files {
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
func init() {
|
||||
rootCmd.AddCommand(searchCmd())
|
||||
}
|
||||
|
||||
// searchCmd queries the OnlyOffice Elasticsearch index directly. The REST
|
||||
// /api/2.0/files/@search endpoint only searches file names in the database;
|
||||
// content search needs ES (see docs/elasticsearch.md).
|
||||
func searchCmd() *cobra.Command {
|
||||
var (
|
||||
content bool
|
||||
folder string
|
||||
limit int
|
||||
asJSON bool
|
||||
)
|
||||
cmd := &cobra.Command{
|
||||
Use: "search QUERY",
|
||||
Short: "Full-text search over documents by name, optionally by content (Elasticsearch)",
|
||||
Long: "Search the OnlyOffice Documents index.\n\n" +
|
||||
"By default only file names are matched. With --content the query also\n" +
|
||||
"matches extracted document text (document.attachment.content); this covers\n" +
|
||||
"Office formats (docx/xlsx/pptx) and is slower.\n\n" +
|
||||
"Requires ONLYOFFICE_ES_URL (and optionally ONLYOFFICE_ES_INDEX,\n" +
|
||||
"ONLYOFFICE_TENANT). See docs/elasticsearch.md for the tunnel setup.",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if asJSON {
|
||||
outputFormat = "json"
|
||||
}
|
||||
es, err := onlyoffice.NewESSearcher(onlyoffice.ESConfigFromEnv())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
hits, err := es.Search(cmd.Context(), onlyoffice.SearchQuery{
|
||||
Text: args[0],
|
||||
InContent: content,
|
||||
FolderID: folder,
|
||||
Limit: limit,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
rows := make([]map[string]any, 0, len(hits))
|
||||
for _, h := range hits {
|
||||
rows = append(rows, map[string]any{
|
||||
"id": h.ID,
|
||||
"title": h.Title,
|
||||
"folder": h.ParentID,
|
||||
"score": h.Score,
|
||||
"highlight": h.Highlight,
|
||||
})
|
||||
}
|
||||
if outputFormat == "json" {
|
||||
printJSON(rows)
|
||||
return nil
|
||||
}
|
||||
printTable([]string{"id", "title", "folder", "score", "highlight"}, rows)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().BoolVar(&content, "content", false, "also match extracted document content")
|
||||
cmd.Flags().StringVar(&folder, "folder", "", "limit to a Documents folder id")
|
||||
cmd.Flags().IntVar(&limit, "limit", 20, "maximum number of results")
|
||||
cmd.Flags().BoolVar(&asJSON, "json", false, "shorthand for --output json")
|
||||
return cmd
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestSearchCommandRegisteredWithFlags(t *testing.T) {
|
||||
cmd, _, err := rootCmd.Find([]string{"search"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if cmd.Name() != "search" {
|
||||
t.Fatalf("search resolved to %q", cmd.Name())
|
||||
}
|
||||
for _, name := range []string{"content", "folder", "limit", "json"} {
|
||||
if cmd.Flags().Lookup(name) == nil {
|
||||
t.Errorf("search: missing --%s flag", name)
|
||||
}
|
||||
}
|
||||
if got := cmd.Flags().Lookup("limit").DefValue; got != "20" {
|
||||
t.Errorf("--limit default = %q, want 20", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSearchWithoutESURLIsClearError(t *testing.T) {
|
||||
clearEnv(t, "ONLYOFFICE_ES_URL", "ONLYOFFICE_ES_INDEX", "ONLYOFFICE_TENANT")
|
||||
errBuf := &bytes.Buffer{}
|
||||
rootCmd.SetErr(errBuf)
|
||||
rootCmd.SetOut(&bytes.Buffer{})
|
||||
rootCmd.SetArgs([]string{"search", "Rechnung"})
|
||||
t.Cleanup(func() {
|
||||
rootCmd.SetArgs(nil)
|
||||
rootCmd.SetOut(nil)
|
||||
rootCmd.SetErr(nil)
|
||||
})
|
||||
|
||||
err := rootCmd.Execute()
|
||||
if err == nil {
|
||||
t.Fatal("expected error without ONLYOFFICE_ES_URL")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "ONLYOFFICE_ES_URL") {
|
||||
t.Fatalf("error %q missing ONLYOFFICE_ES_URL", err.Error())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
// Command ooscan recursively lists OnlyOffice Documents folders into a TSV
|
||||
// index: file_id, folder_id, path, title.
|
||||
//
|
||||
// Usage: ooscan <FOLDER_ID> [<FOLDER_ID>...]
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
)
|
||||
|
||||
func main() {
|
||||
ctx := context.Background()
|
||||
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
|
||||
seen := map[string]bool{}
|
||||
for _, root := range os.Args[1:] {
|
||||
walk(ctx, c, root, "", 0, seen)
|
||||
}
|
||||
}
|
||||
|
||||
func walk(ctx context.Context, c *onlyoffice.Client, folderID, path string, depth int, seen map[string]bool) {
|
||||
if depth > 8 || seen[folderID] {
|
||||
return
|
||||
}
|
||||
seen[folderID] = true
|
||||
// Throttle: OnlyOffice rate-limits (429) and the host must not be flooded.
|
||||
time.Sleep(350 * time.Millisecond)
|
||||
ctx, cancel := context.WithTimeout(ctx, 60*time.Second)
|
||||
defer cancel()
|
||||
var l *onlyoffice.DavListing
|
||||
derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
var err error
|
||||
l, err = c.ListDavFolder(ctx, folderID)
|
||||
return err
|
||||
})
|
||||
if derr != nil {
|
||||
fmt.Fprintf(os.Stderr, "list %s (%s): %v\n", path, folderID, derr)
|
||||
return
|
||||
}
|
||||
for _, f := range l.Files {
|
||||
fmt.Printf("%s\t%s\t%s\t%s\n", f.ID, folderID, path, f.Title)
|
||||
}
|
||||
for _, sub := range l.Folders {
|
||||
walk(ctx, c, sub.ID, path+"/"+sub.Title, depth+1, seen)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
// Command pdfamount walks a Documents folder, downloads matching PDFs and
|
||||
// extracts the payable amount, printing "file_id\ttitle\tamount".
|
||||
//
|
||||
// Usage: pdfamount <FOLDER_ID> [TITLE_FILTER_REGEX]
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
)
|
||||
|
||||
var amountRe = regexp.MustCompile(`(?i)(zu zahlender betrag|rechnungsbetrag)\s*[:\s]*([0-9][0-9.]*,[0-9]{2})`)
|
||||
|
||||
func main() {
|
||||
if len(os.Args) < 2 {
|
||||
fmt.Fprintln(os.Stderr, "usage: pdfamount <FOLDER_ID> [TITLE_FILTER_REGEX]")
|
||||
os.Exit(2)
|
||||
}
|
||||
folder := os.Args[1]
|
||||
filter := regexp.MustCompile(`(?i)rechnung`)
|
||||
if len(os.Args) >= 3 {
|
||||
filter = regexp.MustCompile(os.Args[2])
|
||||
}
|
||||
ctx := context.Background()
|
||||
c := onlyoffice.NewClient(onlyoffice.GetEnvironmentCredentials())
|
||||
|
||||
files := listAll(ctx, c, folder)
|
||||
for _, f := range files {
|
||||
if !filter.MatchString(f.title) {
|
||||
continue
|
||||
}
|
||||
if !strings.HasSuffix(strings.ToLower(f.title), ".pdf") {
|
||||
continue
|
||||
}
|
||||
amount, err := pdfAmount(ctx, c, f.id)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "%s: %v\n", f.title, err)
|
||||
continue
|
||||
}
|
||||
if amount == "" {
|
||||
continue
|
||||
}
|
||||
fmt.Printf("%s\t%s\t%s\n", f.id, f.title, amount)
|
||||
}
|
||||
}
|
||||
|
||||
type file struct{ id, title string }
|
||||
|
||||
func listAll(ctx context.Context, c *onlyoffice.Client, folder string) []file {
|
||||
seen := map[string]bool{}
|
||||
var out []file
|
||||
var walk func(string)
|
||||
walk = func(id string) {
|
||||
if seen[id] {
|
||||
return
|
||||
}
|
||||
seen[id] = true
|
||||
time.Sleep(300 * time.Millisecond)
|
||||
l, err := c.ListDavFolder(ctx, id)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "list %s: %v\n", id, err)
|
||||
return
|
||||
}
|
||||
for _, f := range l.Files {
|
||||
out = append(out, file{f.ID, f.Title})
|
||||
}
|
||||
for _, sub := range l.Folders {
|
||||
walk(sub.ID)
|
||||
}
|
||||
}
|
||||
walk(folder)
|
||||
return out
|
||||
}
|
||||
|
||||
func pdfAmount(ctx context.Context, c *onlyoffice.Client, id string) (string, error) {
|
||||
time.Sleep(time.Second)
|
||||
tmp, err := os.CreateTemp("", "pdf-*.pdf")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer os.Remove(tmp.Name())
|
||||
derr := onlyoffice.DoRetry(ctx, onlyoffice.DefaultRetryPolicy(), func() error {
|
||||
_ = tmp.Truncate(0)
|
||||
_, _ = tmp.Seek(0, 0)
|
||||
_, err := c.DownloadFile(ctx, id, tmp)
|
||||
return err
|
||||
})
|
||||
if derr != nil {
|
||||
tmp.Close()
|
||||
return "", derr
|
||||
}
|
||||
tmp.Close()
|
||||
var buf bytes.Buffer
|
||||
cmd := exec.CommandContext(ctx, "pdftotext", "-layout", tmp.Name(), "-")
|
||||
cmd.Stdout = &buf
|
||||
if err := cmd.Run(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
m := amountRe.FindStringSubmatch(buf.String())
|
||||
if m == nil {
|
||||
return "", nil
|
||||
}
|
||||
return parseDe(m[2]), nil
|
||||
}
|
||||
|
||||
// parseDe turns "1.234,56" into 1234.56.
|
||||
func parseDe(s string) string {
|
||||
s = strings.ReplaceAll(s, ".", "")
|
||||
s = strings.ReplaceAll(s, ",", ".")
|
||||
v, err := strconv.ParseFloat(s, 64)
|
||||
if err != nil {
|
||||
return s
|
||||
}
|
||||
return strconv.FormatFloat(v, 'f', 2, 64)
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// History entities that OnlyOffice CRM actually accepts for history notes.
|
||||
// There is NO person/contact history in this API version: POST /api/2.0/crm/history.json
|
||||
// returns 400 "Value does not fall within the expected range." for entityType
|
||||
// contact/person/people/client/member. Verified against a live instance (#74).
|
||||
const (
|
||||
HistoryEntityOpportunity = "opportunity"
|
||||
HistoryEntityCase = "case"
|
||||
)
|
||||
|
||||
// IsCompany reports whether a CRM contact row is a company (vs a person).
|
||||
// The field arrives as JSON bool; be liberal about what we accept.
|
||||
func IsCompany(person map[string]any) bool {
|
||||
b, _ := person["isCompany"].(bool)
|
||||
return b
|
||||
}
|
||||
|
||||
// ContactID returns the CRM id of a contact row as a plain string.
|
||||
func ContactID(row map[string]any) string {
|
||||
return fmt.Sprint(row["id"])
|
||||
}
|
||||
|
||||
// BuildContactEmailIndex scans all persons once and maps lowercase email →
|
||||
// contact id. Use this instead of calling FindPersonByEmail per address:
|
||||
// the index is O(N) over the whole CRM, the per-address lookup is O(N×M).
|
||||
func (c *Client) BuildContactEmailIndex(ctx context.Context) (map[string]string, error) {
|
||||
all, err := c.ListAllContacts(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
index := make(map[string]string, len(all)*2)
|
||||
for _, person := range all {
|
||||
if IsCompany(person) {
|
||||
continue
|
||||
}
|
||||
id := ContactID(person)
|
||||
for _, row := range ContactInfoRows(person) {
|
||||
if NormalizeContactInfoType(fmt.Sprint(row["infoType"])) != "email" {
|
||||
continue
|
||||
}
|
||||
email := strings.ToLower(strings.TrimSpace(fmt.Sprint(row["data"])))
|
||||
if email != "" && email != "<nil>" {
|
||||
index[email] = id
|
||||
}
|
||||
}
|
||||
}
|
||||
return index, nil
|
||||
}
|
||||
|
||||
// BuildPersonOpportunityIndex maps every opportunity member's contact id to a
|
||||
// deterministic representative opportunity: the one with the lowest numeric id.
|
||||
// OnlyOffice has no person-level history, so notes for a person go on their
|
||||
// deal — this index answers "which deal" in one pass.
|
||||
func (c *Client) BuildPersonOpportunityIndex(ctx context.Context) (map[string]string, error) {
|
||||
opps, err := c.ListAllOpportunities(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
index := map[string]string{}
|
||||
for _, opp := range opps {
|
||||
oppID := ContactID(opp)
|
||||
for _, member := range OpportunityMembers(opp) {
|
||||
pid := ContactID(member)
|
||||
if cur, ok := index[pid]; !ok || NumericIDLess(oppID, cur) {
|
||||
index[pid] = oppID
|
||||
}
|
||||
}
|
||||
}
|
||||
return index, nil
|
||||
}
|
||||
|
||||
// NumericIDLess compares two string ids numerically when possible, falling
|
||||
// back to lexicographic order so results stay deterministic either way.
|
||||
func NumericIDLess(a, b string) bool {
|
||||
na, errA := strconv.Atoi(strings.TrimSpace(a))
|
||||
nb, errB := strconv.Atoi(strings.TrimSpace(b))
|
||||
if errA == nil && errB == nil && na != nb {
|
||||
return na < nb
|
||||
}
|
||||
return a < b
|
||||
}
|
||||
|
||||
// SortIDs orders id strings deterministically (numeric first, then lexical).
|
||||
func SortIDs(ids []string) {
|
||||
sort.Strings(ids)
|
||||
sort.SliceStable(ids, func(i, j int) bool { return NumericIDLess(ids[i], ids[j]) })
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
package onlyoffice
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestIsCompany(t *testing.T) {
|
||||
if IsCompany(map[string]any{"isCompany": true}) != true {
|
||||
t.Fatal("true row not detected")
|
||||
}
|
||||
if IsCompany(map[string]any{"isCompany": false}) {
|
||||
t.Fatal("false row detected as company")
|
||||
}
|
||||
if IsCompany(map[string]any{}) {
|
||||
t.Fatal("missing field detected as company")
|
||||
}
|
||||
if IsCompany(nil) {
|
||||
t.Fatal("nil row detected as company")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNumericIDLess(t *testing.T) {
|
||||
cases := []struct {
|
||||
a, b string
|
||||
want bool
|
||||
}{
|
||||
{"9", "10", true},
|
||||
{"1747", "1748", true},
|
||||
{"abc", "abd", true},
|
||||
{"10", "9", false},
|
||||
{" 12 ", "13", true},
|
||||
{"x1", "2", false}, // non-numeric falls back lexical: "x1" > "2"
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := NumericIDLess(c.a, c.b); got != c.want {
|
||||
t.Errorf("NumericIDLess(%q,%q)=%v want %v", c.a, c.b, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSortIDs(t *testing.T) {
|
||||
ids := []string{"20", "3", "100", "1"}
|
||||
SortIDs(ids)
|
||||
want := "1 3 20 100"
|
||||
got := ""
|
||||
for i, id := range ids {
|
||||
if i > 0 {
|
||||
got += " "
|
||||
}
|
||||
got += id
|
||||
}
|
||||
if got != want {
|
||||
t.Fatalf("SortIDs=%q want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestContactID(t *testing.T) {
|
||||
if ContactID(map[string]any{"id": float64(42)}) != "42" {
|
||||
t.Fatal("numeric id formatting broken")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHistoryEntityConstants(t *testing.T) {
|
||||
if HistoryEntityOpportunity != "opportunity" || HistoryEntityCase != "case" {
|
||||
t.Fatal("history entity whitelist drifted from live-verified values")
|
||||
}
|
||||
}
|
||||
@@ -15,10 +15,15 @@ import (
|
||||
)
|
||||
|
||||
// ListContacts returns a page of CRM contacts and the total count.
|
||||
// sortBy=id is always set: OnlyOffice filter.json without an explicit sort
|
||||
// order is non-deterministic on large contact sets, so a paged walk
|
||||
// (ListAllContacts, ListContactsByTag, FindCompany, FindPerson) can skip or
|
||||
// duplicate contacts across page boundaries.
|
||||
func (c *Client) ListContacts(ctx context.Context, count, startIndex int, search string) ([]map[string]any, int, error) {
|
||||
q := url.Values{}
|
||||
q.Set("count", strconv.Itoa(count))
|
||||
q.Set("startIndex", strconv.Itoa(startIndex))
|
||||
q.Set("sortBy", "id")
|
||||
if search != "" {
|
||||
q.Set("filterValue", search)
|
||||
}
|
||||
@@ -243,6 +248,24 @@ func (c *Client) DeleteContact(ctx context.Context, contactID string) (map[strin
|
||||
return c.deleteObject(ctx, fmt.Sprintf("/api/2.0/crm/contact/%s.json", url.PathEscape(contactID)))
|
||||
}
|
||||
|
||||
// UpdateContactName renames a CRM company contact displayName.
|
||||
// Uses the company endpoint (person names go through /crm/contact/person/{id}).
|
||||
func (c *Client) UpdateContactName(ctx context.Context, contactID, newName string) (map[string]any, error) {
|
||||
body := map[string]any{
|
||||
"displayName": newName,
|
||||
"companyName": newName,
|
||||
"isCompany": true,
|
||||
}
|
||||
out, err := c.putJSONObject(ctx, fmt.Sprintf("/api/2.0/crm/contact/company/%s.json", url.PathEscape(contactID)), body)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if fresh, gerr := c.GetContact(ctx, contactID); gerr == nil && fresh != nil {
|
||||
out = fresh
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// ListContactTags returns all CRM contact tags (title + relativeItemsCount).
|
||||
func (c *Client) ListContactTags(ctx context.Context) ([]map[string]any, error) {
|
||||
return c.ResponseArray(ctx, "/api/2.0/crm/contact/tag.json")
|
||||
@@ -288,6 +311,7 @@ func (c *Client) ListContactsByTag(ctx context.Context, tagName string, count, s
|
||||
q.Set("count", strconv.Itoa(count))
|
||||
q.Set("startIndex", strconv.Itoa(startIndex))
|
||||
q.Set("tags", tagName)
|
||||
q.Set("sortBy", "id")
|
||||
raw, err := c.getJSON(ctx, "/api/2.0/crm/contact/filter.json?"+q.Encode())
|
||||
if err != nil {
|
||||
return nil, 0, err
|
||||
@@ -717,6 +741,11 @@ func (c *Client) DeleteCRMTask(ctx context.Context, id string) (map[string]any,
|
||||
return c.deleteObject(ctx, fmt.Sprintf("/api/2.0/crm/task/%s.json", url.PathEscape(id)))
|
||||
}
|
||||
|
||||
// CloseCRMTask closes (completes) a CRM task via the task close endpoint.
|
||||
func (c *Client) CloseCRMTask(ctx context.Context, id string) (map[string]any, error) {
|
||||
return c.putFormObject(ctx, fmt.Sprintf("/api/2.0/crm/task/%s/close.json", url.PathEscape(id)), url.Values{})
|
||||
}
|
||||
|
||||
// ListTaskCategories returns CRM task categories.
|
||||
func (c *Client) ListTaskCategories(ctx context.Context) ([]map[string]any, error) {
|
||||
return c.ResponseArray(ctx, "/api/2.0/crm/task/category.json")
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
---
|
||||
type: reference
|
||||
status: current
|
||||
related:
|
||||
- README.md
|
||||
- file_es.go
|
||||
---
|
||||
|
||||
# Elasticsearch — полнотекстовый поиск OnlyOffice
|
||||
|
||||
## Что это
|
||||
|
||||
Полнотекстовый поиск OnlyOffice Workspace работает на **Elasticsearch**.
|
||||
Клиент на сервере — NEST. Индекс — имя таблицы.
|
||||
|
||||
Для файлов индекс `files_file`:
|
||||
|
||||
| поле | тип | смысл |
|
||||
|------|-----|-------|
|
||||
| `id` | integer | id файла (тот же, что в REST/Documents) |
|
||||
| `title` | text (`whitespacecustom`) | имя файла |
|
||||
| `tenantId` | integer | тенант (портал) |
|
||||
| `folders` | nested | список папок: `folderId` (строка), `id`, `tenantId` |
|
||||
| `document.attachment.content` | text (`document`) | извлеченный текст (ingest-attachment) |
|
||||
| `document.attachment.content_type` | text | MIME |
|
||||
|
||||
Важно:
|
||||
- Живой сервер — **Elasticsearch 7.16.3**, кластер `elasticsearch`.
|
||||
- REST `GET /api/2.0/files/@search/{query}` ищет **только по имени в БД**
|
||||
(`fileDao.Search`), ES не задействует. Для поиска по содержимому нужен
|
||||
прямой ES — это и делает `oo search`.
|
||||
- `title` analyzer `whitespacecustom` режет по пробелам и lower-case. Полное
|
||||
имя файла — один токен (`Rechnung-4711.pdf`), поэтому поиск по имени ищет
|
||||
слово целиком, а не подстроку.
|
||||
- `document.attachment.content` заполняется **только для Office-форматов**
|
||||
(docx / xlsx / pptx). У PDF/txt, залитых через API, контент не извлекается.
|
||||
- Индексация асинхронная (TeamLabSvc) — файл появляется в ES не мгновенно.
|
||||
|
||||
## Доступ
|
||||
|
||||
ES слушает `127.0.0.1:9200` **внутри** VM OnlyOffice. Снаружи порт закрыт,
|
||||
SSH в VM открыт на хосте как `127.0.0.1:32` (контейнер `onlyoffice-v2`,
|
||||
QEMU). Схема — SSH-туннель.
|
||||
|
||||
```bash
|
||||
# из корня go-onlyoffice (ключ и хост — как в infra-доках)
|
||||
ssh -f -N -o ControlMaster=no -o ControlPath=none \
|
||||
-p 32 -i ~/.ssh/id_ed25519 \
|
||||
-L 9200:127.0.0.1:9200 root@127.0.0.1
|
||||
|
||||
curl -s http://127.0.0.1:9200/ | head # tagline + version
|
||||
curl -s 'http://127.0.0.1:9200/_cat/indices?h=index,docs.count'
|
||||
```
|
||||
|
||||
`-o ControlMaster=no -o ControlPath=none` обязательны: иначе forward уходит
|
||||
в persistent master-соединение из `~/.ssh/config` и порт остаётся занят.
|
||||
|
||||
Проверить, что туннель жив:
|
||||
|
||||
```bash
|
||||
curl -s http://127.0.0.1:9200/files_file/_count
|
||||
```
|
||||
|
||||
## Переменные
|
||||
|
||||
| env | default | смысл |
|
||||
|-----|---------|-------|
|
||||
| `ONLYOFFICE_ES_URL` | — (обязателен) | `scheme://host:port` ES |
|
||||
| `ONLYOFFICE_ES_INDEX` | `files_file` | индекс |
|
||||
| `ONLYOFFICE_TENANT` | пусто (все) | фильтр `tenantId` |
|
||||
|
||||
Имена — в [`.env.example`](../.env.example). Секретов нет: ES без пароля.
|
||||
|
||||
## CLI
|
||||
|
||||
```bash
|
||||
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 oo search "Rechnung"
|
||||
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 oo search "Mahngebühr" --content
|
||||
oo search "Rechnung" --folder 649 --limit 50 --json
|
||||
```
|
||||
|
||||
Флаги: `--content` (искать и по тексту), `--folder ID` (папка
|
||||
`folders.folderId`), `--limit N` (по умолчанию 20, максимум 200),
|
||||
`--json` = `-o json`.
|
||||
|
||||
## Библиотека
|
||||
|
||||
`file_es.go` — `ESSearcher` (`Name() = "elasticsearch"`), прямой ES REST на
|
||||
stdlib `net/http`:
|
||||
|
||||
```go
|
||||
es, _ := onlyoffice.NewESSearcher(onlyoffice.ESConfigFromEnv())
|
||||
hits, _ := es.Search(ctx, onlyoffice.SearchQuery{
|
||||
Text: "Rechnung", InContent: true, Limit: 20,
|
||||
})
|
||||
```
|
||||
|
||||
Запрос: `multi_match` по `title^2` (+ `document.attachment.content` при
|
||||
`InContent`), фильтры `tenantId` и `folders.folderId`, `_source`
|
||||
id/title/folders, `highlight` для фрагмента. Ответ → `[]SearchHit` (модель из
|
||||
эпика #34; пока объявлена в `file_es.go`, переедет в `file_core.go` с F1 #35).
|
||||
|
||||
## Тесты
|
||||
|
||||
```bash
|
||||
# unit — чистые builders/парсеры, без сети
|
||||
go test ./ -run ES
|
||||
|
||||
# integration — нужен ONLYOFFICE_ES_URL (+ креды REST для залива)
|
||||
set -a; . .env; set +a
|
||||
ONLYOFFICE_ES_URL=http://127.0.0.1:9200 ONLYOFFICE_TENANT=1 \
|
||||
go test -tags=integration -run TestIntegrationESSearch -v .
|
||||
```
|
||||
|
||||
Интеграционный тест заливает временный xlsx (в имени и в ячейке — уникальные
|
||||
токены), ждёт индексации, проверяет поиск по имени и по содержимому, затем
|
||||
удаляет проект.
|
||||
|
||||
## Грабли
|
||||
|
||||
- `locale`/версия ES: 7.16.3, `_search` совместим с REST 7.x.
|
||||
- ES без auth и слушает только localhost — туннель обязателен.
|
||||
- Фильтр `tenantId` сузит выдачу; без него видны документы всех тенантов.
|
||||
- Поиск по содержимому PDF, залитых через API, не работает (нет
|
||||
`attachment.content`) — только Office-форматы.
|
||||
+329
@@ -0,0 +1,329 @@
|
||||
package onlyoffice
|
||||
|
||||
// Elasticsearch backend of the unified file client (epic #34, F3 #37).
|
||||
//
|
||||
// OnlyOffice full-text search runs on Elasticsearch (index `files_file`, NEST
|
||||
// client on the server). The REST endpoint GET /api/2.0/files/@search/{query}
|
||||
// only searches file names in the database, so content search needs a direct
|
||||
// ES query. The live server is Elasticsearch 7.16.3; the request shape below
|
||||
// is plain REST and stays stdlib-only, matching the repo's no-extra-deps rule.
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Canonical file/search model (epic #34, F1 #35). Declared here because F1 is
|
||||
// not merged yet; move to file_core.go and drop these when it lands. Keep the
|
||||
// shape identical to the contract in #35.
|
||||
type (
|
||||
// Kind distinguishes a file from a folder.
|
||||
Kind int
|
||||
// Entry is a canonical file/folder record.
|
||||
Entry struct {
|
||||
ID string
|
||||
ParentID string
|
||||
Title string
|
||||
Kind Kind
|
||||
Size int64
|
||||
MIME string
|
||||
Created time.Time
|
||||
Modified time.Time
|
||||
Version int
|
||||
Provider string
|
||||
}
|
||||
// SearchQuery is a backend-agnostic search request.
|
||||
SearchQuery struct {
|
||||
Text string
|
||||
InContent bool
|
||||
FolderID string
|
||||
Extensions []string
|
||||
Limit int
|
||||
}
|
||||
// SearchHit is a search result entry plus its relevance data.
|
||||
SearchHit struct {
|
||||
Entry
|
||||
Score float64
|
||||
Highlight string
|
||||
Path []string
|
||||
}
|
||||
// Searcher searches a document store by name and optionally content.
|
||||
Searcher interface {
|
||||
Search(ctx context.Context, q SearchQuery) ([]SearchHit, error)
|
||||
Name() string
|
||||
}
|
||||
)
|
||||
|
||||
// Kind values (epic #34).
|
||||
const (
|
||||
File Kind = iota
|
||||
Folder
|
||||
)
|
||||
|
||||
const (
|
||||
defaultESIndex = "files_file"
|
||||
defaultESLimit = 20
|
||||
maxESLimit = 200
|
||||
maxESResponseSize = 8 << 20
|
||||
)
|
||||
|
||||
// ESConfig configures the direct Elasticsearch searcher.
|
||||
type ESConfig struct {
|
||||
URL string // scheme://host:port of the ES HTTP endpoint
|
||||
Index string // index name, default files_file
|
||||
Tenant string // tenantId filter, empty means all tenants
|
||||
}
|
||||
|
||||
// ESConfigFromEnv reads ONLYOFFICE_ES_URL, ONLYOFFICE_ES_INDEX (default
|
||||
// files_file) and ONLYOFFICE_TENANT. The library never loads dotfiles — the
|
||||
// CLI does that.
|
||||
func ESConfigFromEnv() ESConfig {
|
||||
return ESConfig{
|
||||
URL: strings.TrimRight(strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL")), "/"),
|
||||
Index: firstNonEmpty(os.Getenv("ONLYOFFICE_ES_INDEX"), defaultESIndex),
|
||||
Tenant: strings.TrimSpace(os.Getenv("ONLYOFFICE_TENANT")),
|
||||
}
|
||||
}
|
||||
|
||||
// ESSearcher queries OnlyOffice's Elasticsearch index directly for file name
|
||||
// and document content.
|
||||
type ESSearcher struct {
|
||||
cfg ESConfig
|
||||
http *http.Client
|
||||
}
|
||||
|
||||
// NewESSearcher returns a searcher for the OnlyOffice Elasticsearch index.
|
||||
// The URL is required; an empty index falls back to files_file.
|
||||
func NewESSearcher(cfg ESConfig) (*ESSearcher, error) {
|
||||
if strings.TrimSpace(cfg.URL) == "" {
|
||||
return nil, fmt.Errorf("onlyoffice: elasticsearch URL is empty (set ONLYOFFICE_ES_URL)")
|
||||
}
|
||||
cfg.URL = strings.TrimRight(cfg.URL, "/")
|
||||
if cfg.Index == "" {
|
||||
cfg.Index = defaultESIndex
|
||||
}
|
||||
return &ESSearcher{cfg: cfg, http: &http.Client{Timeout: 30 * time.Second}}, nil
|
||||
}
|
||||
|
||||
// Name implements Searcher.
|
||||
func (s *ESSearcher) Name() string { return "elasticsearch" }
|
||||
|
||||
// Search runs a multi_match over title (and, when q.InContent is set,
|
||||
// document.attachment.content), filtered by tenant and optional folder.
|
||||
func (s *ESSearcher) Search(ctx context.Context, q SearchQuery) ([]SearchHit, error) {
|
||||
q.Text = strings.TrimSpace(q.Text)
|
||||
if q.Text == "" {
|
||||
return nil, fmt.Errorf("onlyoffice: empty search query")
|
||||
}
|
||||
body, err := json.Marshal(esSearchRequest(q, s.cfg.Tenant))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: build elasticsearch query: %w", err)
|
||||
}
|
||||
endpoint := s.cfg.URL + "/" + s.cfg.Index + "/_search"
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, endpoint, bytes.NewReader(body))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
req.Header.Set("Accept", "application/json")
|
||||
resp, err := s.http.Do(req)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: elasticsearch search: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
raw, err := io.ReadAll(io.LimitReader(resp.Body, maxESResponseSize))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("onlyoffice: elasticsearch search: %d %s", resp.StatusCode, truncate(string(raw), 400))
|
||||
}
|
||||
return parseESSearchResponse(raw)
|
||||
}
|
||||
|
||||
// esSearchRequest builds the ES query body. Pure, so it is unit-tested.
|
||||
func esSearchRequest(q SearchQuery, tenant string) esRequest {
|
||||
limit := q.Limit
|
||||
if limit <= 0 {
|
||||
limit = defaultESLimit
|
||||
}
|
||||
if limit > maxESLimit {
|
||||
limit = maxESLimit
|
||||
}
|
||||
fields := []string{"title^2"}
|
||||
if q.InContent {
|
||||
fields = append(fields, "document.attachment.content")
|
||||
}
|
||||
must := []esClause{{MultiMatch: &esMultiMatch{Query: q.Text, Fields: fields}}}
|
||||
|
||||
var filter []esClause
|
||||
if t := strings.TrimSpace(tenant); t != "" {
|
||||
filter = append(filter, esClause{Term: map[string]any{"tenantId": numericOrString(t)}})
|
||||
}
|
||||
if f := strings.TrimSpace(q.FolderID); f != "" {
|
||||
filter = append(filter, esClause{Term: map[string]any{"folders.folderId": f}})
|
||||
}
|
||||
for _, ext := range normalizeExtensions(q.Extensions) {
|
||||
filter = append(filter, esClause{Wildcard: map[string]any{"title": "*." + ext}})
|
||||
}
|
||||
|
||||
highlightFields := map[string]struct{}{"title": {}}
|
||||
if q.InContent {
|
||||
highlightFields["document.attachment.content"] = struct{}{}
|
||||
}
|
||||
return esRequest{
|
||||
Size: limit,
|
||||
Source: []string{"id", "title", "folders"},
|
||||
Query: esQuery{Bool: esBool{Must: must, Filter: filter}},
|
||||
Highlight: esHighlight{PreTags: []string{"<em>"}, PostTags: []string{"</em>"}, Fields: highlightFields},
|
||||
}
|
||||
}
|
||||
|
||||
// normalizeExtensions lowercases, trims leading dots and drops empties.
|
||||
func normalizeExtensions(exts []string) []string {
|
||||
out := make([]string, 0, len(exts))
|
||||
seen := map[string]bool{}
|
||||
for _, e := range exts {
|
||||
e = strings.ToLower(strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(e), ".")))
|
||||
if e == "" || seen[e] {
|
||||
continue
|
||||
}
|
||||
seen[e] = true
|
||||
out = append(out, e)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// numericOrString keeps an integer-looking filter value numeric (tenantId is
|
||||
// a long) and leaves anything else as a string (folderId is a text token).
|
||||
func numericOrString(s string) any {
|
||||
if n, err := strconv.ParseInt(s, 10, 64); err == nil {
|
||||
return n
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// esRequest is the subset of the ES query DSL this client emits.
|
||||
type esRequest struct {
|
||||
Size int `json:"size"`
|
||||
Source []string `json:"_source"`
|
||||
Query esQuery `json:"query"`
|
||||
Highlight esHighlight `json:"highlight"`
|
||||
}
|
||||
|
||||
type esQuery struct {
|
||||
Bool esBool `json:"bool"`
|
||||
}
|
||||
|
||||
type esBool struct {
|
||||
Must []esClause `json:"must,omitempty"`
|
||||
Filter []esClause `json:"filter,omitempty"`
|
||||
}
|
||||
|
||||
type esClause struct {
|
||||
MultiMatch *esMultiMatch `json:"multi_match,omitempty"`
|
||||
Term map[string]any `json:"term,omitempty"`
|
||||
Wildcard map[string]any `json:"wildcard,omitempty"`
|
||||
}
|
||||
|
||||
type esMultiMatch struct {
|
||||
Query string `json:"query"`
|
||||
Fields []string `json:"fields"`
|
||||
}
|
||||
|
||||
type esHighlight struct {
|
||||
PreTags []string `json:"pre_tags,omitempty"`
|
||||
PostTags []string `json:"post_tags,omitempty"`
|
||||
Fields map[string]struct{} `json:"fields"`
|
||||
}
|
||||
|
||||
// esResponse is the subset of an ES search response we consume.
|
||||
type esResponse struct {
|
||||
Took int `json:"took"`
|
||||
Hits struct {
|
||||
Total struct {
|
||||
Value int `json:"value"`
|
||||
Relation string `json:"relation"`
|
||||
} `json:"total"`
|
||||
Hits []esResponseHit `json:"hits"`
|
||||
} `json:"hits"`
|
||||
}
|
||||
|
||||
type esResponseHit struct {
|
||||
ID string `json:"_id"`
|
||||
Score float64 `json:"_score"`
|
||||
Source struct {
|
||||
ID int `json:"id"`
|
||||
Title string `json:"title"`
|
||||
Folders []struct {
|
||||
FolderID string `json:"folderId"`
|
||||
ID int `json:"id"`
|
||||
} `json:"folders"`
|
||||
} `json:"_source"`
|
||||
Highlight map[string][]string `json:"highlight"`
|
||||
}
|
||||
|
||||
// parseESSearchResponse converts an ES search response into SearchHit values.
|
||||
// Pure, so it is unit-tested.
|
||||
func parseESSearchResponse(raw []byte) ([]SearchHit, error) {
|
||||
var r esResponse
|
||||
if err := json.Unmarshal(raw, &r); err != nil {
|
||||
return nil, fmt.Errorf("onlyoffice: decode elasticsearch response: %w", err)
|
||||
}
|
||||
hits := make([]SearchHit, 0, len(r.Hits.Hits))
|
||||
for _, h := range r.Hits.Hits {
|
||||
id := strconv.Itoa(h.Source.ID)
|
||||
if h.Source.ID == 0 {
|
||||
id = h.ID
|
||||
}
|
||||
var parent string
|
||||
path := make([]string, 0, len(h.Source.Folders))
|
||||
for i, f := range h.Source.Folders {
|
||||
path = append(path, f.FolderID)
|
||||
if i == 0 {
|
||||
parent = f.FolderID
|
||||
}
|
||||
}
|
||||
hits = append(hits, SearchHit{
|
||||
Entry: Entry{
|
||||
ID: id,
|
||||
ParentID: parent,
|
||||
Title: h.Source.Title,
|
||||
Kind: File,
|
||||
Provider: "elasticsearch",
|
||||
},
|
||||
Score: h.Score,
|
||||
Highlight: esHighlightText(h.Highlight),
|
||||
Path: path,
|
||||
})
|
||||
}
|
||||
return hits, nil
|
||||
}
|
||||
|
||||
var esHighlightTag = regexp.MustCompile(`</?em[^>]*>`)
|
||||
|
||||
// esHighlightText flattens a highlight map into one plain-text snippet,
|
||||
// preferring the content fragment over the title.
|
||||
func esHighlightText(hl map[string][]string) string {
|
||||
for _, key := range []string{"document.attachment.content", "title"} {
|
||||
frags := hl[key]
|
||||
if len(frags) == 0 {
|
||||
continue
|
||||
}
|
||||
clean := make([]string, 0, len(frags))
|
||||
for _, f := range frags {
|
||||
clean = append(clean, esHighlightTag.ReplaceAllString(f, ""))
|
||||
}
|
||||
return strings.Join(clean, " … ")
|
||||
}
|
||||
return ""
|
||||
}
|
||||
@@ -0,0 +1,135 @@
|
||||
//go:build integration
|
||||
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
// TestIntegrationESSearch uploads a throwaway workbook and verifies that the
|
||||
// direct Elasticsearch search finds it by file name and by content.
|
||||
//
|
||||
// The content index (document.attachment.content) is only populated for Office
|
||||
// formats (docx/xlsx/pptx), so the fixture is an xlsx whose cell carries a
|
||||
// unique token. Requires ONLYOFFICE_ES_URL (a reachable ES endpoint — in the
|
||||
// current setup a tunnel to the ES inside the OnlyOffice VM, see
|
||||
// docs/elasticsearch.md) plus the regular REST credentials for the upload.
|
||||
// Skips when either is missing.
|
||||
func TestIntegrationESSearch(t *testing.T) {
|
||||
esURL := strings.TrimSpace(os.Getenv("ONLYOFFICE_ES_URL"))
|
||||
if esURL == "" {
|
||||
t.Skip("ONLYOFFICE_ES_URL not set — skipping Elasticsearch integration test")
|
||||
}
|
||||
c := liveClient(t)
|
||||
t.Cleanup(func() { cleanupTestProjects(t, c) })
|
||||
|
||||
stamp := time.Now().UTC().Format("20060102-150405")
|
||||
nameToken := "goesname" + stamp
|
||||
contentToken := "goescontent" + stamp
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
|
||||
defer cancel()
|
||||
|
||||
project, err := c.CreateProject(NewProjectRequest{
|
||||
Title: testProjectPrefix + "es-" + stamp,
|
||||
Description: "go-onlyoffice elasticsearch integration",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("CreateProject: %v", err)
|
||||
}
|
||||
if project.ID == nil {
|
||||
t.Fatal("created project without id")
|
||||
}
|
||||
pid := strconv.Itoa(*project.ID)
|
||||
|
||||
title := nameToken + ".xlsx"
|
||||
localPath := filepath.Join(t.TempDir(), title)
|
||||
book := excelize.NewFile()
|
||||
if err := book.SetCellValue("Sheet1", "A1", "OnlyOffice Elasticsearch content fixture "+contentToken); err != nil {
|
||||
t.Fatalf("SetCellValue: %v", err)
|
||||
}
|
||||
if err := book.SaveAs(localPath); err != nil {
|
||||
t.Fatalf("SaveAs: %v", err)
|
||||
}
|
||||
|
||||
entry, err := c.UploadProjectFile(ctx, pid, localPath)
|
||||
if err != nil {
|
||||
t.Fatalf("UploadProjectFile: %v", err)
|
||||
}
|
||||
fileID := strconv.Itoa(int(FileEntryNumericID(entry)))
|
||||
if fileID == "0" {
|
||||
t.Fatalf("upload returned no file id: %+v", entry)
|
||||
}
|
||||
|
||||
es, err := NewESSearcher(ESConfig{
|
||||
URL: esURL,
|
||||
Index: os.Getenv("ONLYOFFICE_ES_INDEX"),
|
||||
Tenant: os.Getenv("ONLYOFFICE_TENANT"),
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("NewESSearcher: %v", err)
|
||||
}
|
||||
|
||||
// Indexing is asynchronous on the server; poll until the file shows up.
|
||||
// The server's title analyzer splits on whitespace, so the name query is
|
||||
// the full file name token (including extension), as a user would type it.
|
||||
nameHit := waitForHit(t, ctx, es, SearchQuery{Text: title}, fileID)
|
||||
if nameHit.Title != title {
|
||||
t.Errorf("name hit title = %q, want %q", nameHit.Title, title)
|
||||
}
|
||||
contentHit := waitForHit(t, ctx, es, SearchQuery{Text: contentToken, InContent: true}, fileID)
|
||||
if contentHit.Highlight == "" {
|
||||
t.Error("content hit has no highlight fragment")
|
||||
}
|
||||
if !strings.Contains(contentHit.Title, nameToken) {
|
||||
t.Errorf("content hit title = %q, want the uploaded workbook", contentHit.Title)
|
||||
}
|
||||
|
||||
// The content token is absent from the title, so a name-only search must
|
||||
// not return the file — this proves the content field is really queried.
|
||||
if hits := searchQuiet(t, es, SearchQuery{Text: contentToken}); len(hits) != 0 {
|
||||
t.Errorf("name-only search for content token returned %d hits, want 0", len(hits))
|
||||
}
|
||||
}
|
||||
|
||||
// waitForHit polls ES until the file with fileID appears and returns that hit.
|
||||
func waitForHit(t *testing.T, ctx context.Context, s *ESSearcher, q SearchQuery, fileID string) SearchHit {
|
||||
t.Helper()
|
||||
var lastErr error
|
||||
for {
|
||||
hits, err := s.Search(ctx, q)
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
} else {
|
||||
for _, h := range hits {
|
||||
if h.ID == fileID {
|
||||
return h
|
||||
}
|
||||
}
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
t.Fatalf("search %q: file %s not indexed in time (last err: %v)", q.Text, fileID, lastErr)
|
||||
case <-time.After(3 * time.Second):
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func searchQuiet(t *testing.T, s *ESSearcher, q SearchQuery) []SearchHit {
|
||||
t.Helper()
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
hits, err := s.Search(ctx, q)
|
||||
if err != nil {
|
||||
t.Fatalf("Search(%q): %v", q.Text, err)
|
||||
}
|
||||
return hits
|
||||
}
|
||||
+188
@@ -0,0 +1,188 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"reflect"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestESSearchRequestNameOnly(t *testing.T) {
|
||||
got := esSearchRequest(SearchQuery{Text: "Rechnung"}, "1")
|
||||
if got.Size != defaultESLimit {
|
||||
t.Errorf("size = %d, want %d", got.Size, defaultESLimit)
|
||||
}
|
||||
if !reflect.DeepEqual(got.Source, []string{"id", "title", "folders"}) {
|
||||
t.Errorf("_source = %v", got.Source)
|
||||
}
|
||||
if len(got.Query.Bool.Must) != 1 || got.Query.Bool.Must[0].MultiMatch == nil {
|
||||
t.Fatalf("must = %+v, want one multi_match", got.Query.Bool.Must)
|
||||
}
|
||||
mm := got.Query.Bool.Must[0].MultiMatch
|
||||
if mm.Query != "Rechnung" {
|
||||
t.Errorf("query = %q", mm.Query)
|
||||
}
|
||||
if !reflect.DeepEqual(mm.Fields, []string{"title^2"}) {
|
||||
t.Errorf("fields = %v, want title only", mm.Fields)
|
||||
}
|
||||
if _, ok := got.Highlight.Fields["document.attachment.content"]; ok {
|
||||
t.Error("content highlight present without InContent")
|
||||
}
|
||||
if _, ok := got.Highlight.Fields["title"]; !ok {
|
||||
t.Error("title highlight missing")
|
||||
}
|
||||
if len(got.Query.Bool.Filter) != 1 || got.Query.Bool.Filter[0].Term["tenantId"] != int64(1) {
|
||||
t.Errorf("tenant filter = %+v, want numeric tenantId=1", got.Query.Bool.Filter)
|
||||
}
|
||||
}
|
||||
|
||||
func TestESSearchRequestContentFields(t *testing.T) {
|
||||
got := esSearchRequest(SearchQuery{Text: "Mahnung", InContent: true}, "")
|
||||
mm := got.Query.Bool.Must[0].MultiMatch
|
||||
want := []string{"title^2", "document.attachment.content"}
|
||||
if !reflect.DeepEqual(mm.Fields, want) {
|
||||
t.Errorf("fields = %v, want %v", mm.Fields, want)
|
||||
}
|
||||
if _, ok := got.Highlight.Fields["document.attachment.content"]; !ok {
|
||||
t.Error("content highlight missing with InContent")
|
||||
}
|
||||
if len(got.Query.Bool.Filter) != 0 {
|
||||
t.Errorf("filter = %+v, want none without tenant/folder", got.Query.Bool.Filter)
|
||||
}
|
||||
}
|
||||
|
||||
func TestESSearchRequestFiltersAndLimit(t *testing.T) {
|
||||
got := esSearchRequest(SearchQuery{
|
||||
Text: "Storchen",
|
||||
FolderID: "649",
|
||||
Extensions: []string{".PDF", "pdf", "docx"},
|
||||
Limit: 999,
|
||||
}, "42")
|
||||
if got.Size != maxESLimit {
|
||||
t.Errorf("size = %d, want cap %d", got.Size, maxESLimit)
|
||||
}
|
||||
var tenant, folder, wildcards int
|
||||
for _, f := range got.Query.Bool.Filter {
|
||||
switch {
|
||||
case f.Term != nil && f.Term["tenantId"] != nil:
|
||||
tenant++
|
||||
case f.Term != nil && f.Term["folders.folderId"] != nil:
|
||||
folder++
|
||||
if f.Term["folders.folderId"] != "649" {
|
||||
t.Errorf("folder filter = %+v", f.Term)
|
||||
}
|
||||
case f.Wildcard != nil:
|
||||
wildcards++
|
||||
}
|
||||
}
|
||||
if tenant != 1 || folder != 1 {
|
||||
t.Errorf("term filters tenant=%d folder=%d, want 1 each", tenant, folder)
|
||||
}
|
||||
if wildcards != 2 {
|
||||
t.Errorf("wildcard filters = %d, want deduped PDF+docx", wildcards)
|
||||
}
|
||||
}
|
||||
|
||||
func TestESSearchRequestRejectsEmptyTextAtSearch(t *testing.T) {
|
||||
s, err := NewESSearcher(ESConfig{URL: "http://localhost:9200"})
|
||||
if err != nil {
|
||||
t.Fatalf("NewESSearcher: %v", err)
|
||||
}
|
||||
if _, err := s.Search(t.Context(), SearchQuery{Text: " "}); err == nil {
|
||||
t.Error("empty query: want error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewESSearcherRequiresURL(t *testing.T) {
|
||||
if _, err := NewESSearcher(ESConfig{}); err == nil {
|
||||
t.Error("empty URL: want error")
|
||||
}
|
||||
s, err := NewESSearcher(ESConfig{URL: "http://es:9200/"})
|
||||
if err != nil {
|
||||
t.Fatalf("NewESSearcher: %v", err)
|
||||
}
|
||||
if s.cfg.Index != defaultESIndex {
|
||||
t.Errorf("index = %q, want %q", s.cfg.Index, defaultESIndex)
|
||||
}
|
||||
if s.cfg.URL != "http://es:9200" {
|
||||
t.Errorf("url = %q, want trimmed", s.cfg.URL)
|
||||
}
|
||||
if s.Name() != "elasticsearch" {
|
||||
t.Errorf("Name() = %q", s.Name())
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeExtensions(t *testing.T) {
|
||||
got := normalizeExtensions([]string{" .PDF ", "pdf", "", "xlsx"})
|
||||
want := []string{"pdf", "xlsx"}
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("normalizeExtensions = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseESSearchResponse(t *testing.T) {
|
||||
raw := []byte(`{
|
||||
"took": 12,
|
||||
"hits": {
|
||||
"total": {"value": 2, "relation": "eq"},
|
||||
"hits": [
|
||||
{
|
||||
"_id": "2395",
|
||||
"_score": 7.31,
|
||||
"_source": {"id": 2395, "title": "Rechnung-4711.pdf",
|
||||
"folders": [{"folderId": "438", "id": 0}, {"folderId": "11", "id": 0}]},
|
||||
"highlight": {
|
||||
"title": ["<em>Rechnung</em>-4711.pdf"],
|
||||
"document.attachment.content": ["… Zahlung der <em>Rechnung</em> …"]
|
||||
}
|
||||
},
|
||||
{
|
||||
"_id": "2318",
|
||||
"_score": 6.02,
|
||||
"_source": {"id": 2318, "title": "Mahnung.pdf", "folders": []},
|
||||
"highlight": {"title": ["<em>Mahnung</em>.pdf"]}
|
||||
}
|
||||
]
|
||||
}
|
||||
}`)
|
||||
hits, err := parseESSearchResponse(raw)
|
||||
if err != nil {
|
||||
t.Fatalf("parseESSearchResponse: %v", err)
|
||||
}
|
||||
if len(hits) != 2 {
|
||||
t.Fatalf("hits = %d, want 2", len(hits))
|
||||
}
|
||||
h0 := hits[0]
|
||||
if h0.ID != "2395" || h0.Title != "Rechnung-4711.pdf" || h0.Kind != File {
|
||||
t.Errorf("hit0 entry = %+v", h0.Entry)
|
||||
}
|
||||
if h0.ParentID != "438" || !reflect.DeepEqual(h0.Path, []string{"438", "11"}) {
|
||||
t.Errorf("hit0 path = %v parent = %q", h0.Path, h0.ParentID)
|
||||
}
|
||||
if h0.Score != 7.31 {
|
||||
t.Errorf("hit0 score = %v", h0.Score)
|
||||
}
|
||||
if h0.Highlight != "… Zahlung der Rechnung …" {
|
||||
t.Errorf("hit0 highlight = %q, want content fragment", h0.Highlight)
|
||||
}
|
||||
if hits[1].Highlight != "Mahnung.pdf" {
|
||||
t.Errorf("hit1 highlight = %q, want title without tags", hits[1].Highlight)
|
||||
}
|
||||
if hits[1].ParentID != "" || len(hits[1].Path) != 0 {
|
||||
t.Errorf("hit1 path = %v", hits[1].Path)
|
||||
}
|
||||
}
|
||||
|
||||
func TestESSearchRequestJSONShape(t *testing.T) {
|
||||
got := esSearchRequest(SearchQuery{Text: "Rechnung", InContent: true}, "1")
|
||||
b, err := json.Marshal(got)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal: %v", err)
|
||||
}
|
||||
var back map[string]any
|
||||
if err := json.Unmarshal(b, &back); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if _, ok := back["query"].(map[string]any)["bool"]; !ok {
|
||||
t.Errorf("query.bool missing: %s", b)
|
||||
}
|
||||
}
|
||||
@@ -237,6 +237,15 @@ func (c *Client) UploadProjectFile(ctx context.Context, projectID, localPath str
|
||||
return decodeResponseFileEntry(raw)
|
||||
}
|
||||
|
||||
// UploadProjectFileReplacing upserts by stem|ext in the project Documents folder.
|
||||
func (c *Client) UploadProjectFileReplacing(ctx context.Context, projectID, localPath string) (*FileEntry, []int, error) {
|
||||
folderID, err := c.projectFolderID(ctx, projectID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
return c.UploadToFolderReplacing(ctx, folderID, localPath)
|
||||
}
|
||||
|
||||
// GetFile returns file metadata including viewUrl for download.
|
||||
func (c *Client) GetFile(ctx context.Context, fileID string) (*FileEntry, error) {
|
||||
if fileID == "" {
|
||||
@@ -263,55 +272,137 @@ func (c *Client) RenameFile(ctx context.Context, fileID, newTitle string) (*File
|
||||
return decodeResponseFileEntry(raw)
|
||||
}
|
||||
|
||||
type deleteFilesBody struct {
|
||||
FileIDs []int `json:"fileIds"`
|
||||
FolderIDs []int `json:"folderIds"`
|
||||
}
|
||||
|
||||
// DeleteFiles permanently deletes files by numeric id (Documents module).
|
||||
// Uses per-file DELETE (DeleteDavItems); fileops/delete returns 200 on some
|
||||
// portals (e.g. produktor.io) without removing the file.
|
||||
func (c *Client) DeleteFiles(ctx context.Context, fileIDs []int) error {
|
||||
if len(fileIDs) == 0 {
|
||||
return fmt.Errorf("no file ids to delete")
|
||||
}
|
||||
body := deleteFilesBody{FileIDs: fileIDs, FolderIDs: nil}
|
||||
_, err := c.putJSON(ctx, "/api/2.0/files/fileops/delete.json", body)
|
||||
if err != nil {
|
||||
_, err = c.putJSON(ctx, "/api/2.0/files/fileops/delete", body)
|
||||
strIDs := make([]string, len(fileIDs))
|
||||
for i, id := range fileIDs {
|
||||
strIDs[i] = strconv.Itoa(id)
|
||||
}
|
||||
return err
|
||||
return c.DeleteDavItems(ctx, nil, strIDs)
|
||||
}
|
||||
|
||||
// ListFolder returns the Documents module listing for a folder id
|
||||
// (GET /api/2.0/files/{folderId}).
|
||||
func (c *Client) ListFolder(ctx context.Context, folderID string) (map[string]any, error) {
|
||||
if folderID == "" {
|
||||
return nil, fmt.Errorf("folder id is required")
|
||||
}
|
||||
out, err := c.ResponseObject(ctx, "/api/2.0/files/"+url.PathEscape(folderID)+".json")
|
||||
if err != nil {
|
||||
out, err = c.ResponseObject(ctx, "/api/2.0/files/"+url.PathEscape(folderID))
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
|
||||
// CreateFolder creates a subfolder under parentFolderID.
|
||||
func (c *Client) CreateFolder(ctx context.Context, parentFolderID, title string) (map[string]any, error) {
|
||||
if parentFolderID == "" || title == "" {
|
||||
return nil, fmt.Errorf("parent folder id and title are required")
|
||||
}
|
||||
body := map[string]any{"title": title}
|
||||
out, err := c.postJSONObject(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentFolderID)+".json", body)
|
||||
if err != nil {
|
||||
out, err = c.postJSONObject(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentFolderID), body)
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
|
||||
// MoveFiles moves file ids into destFolderID (Documents fileops/move).
|
||||
func (c *Client) MoveFiles(ctx context.Context, destFolderID int, fileIDs []int) (map[string]any, error) {
|
||||
if destFolderID == 0 || len(fileIDs) == 0 {
|
||||
return nil, fmt.Errorf("dest folder and file ids are required")
|
||||
}
|
||||
body := map[string]any{
|
||||
"folderIds": []int{},
|
||||
"fileIds": fileIDs,
|
||||
"destFolderId": destFolderID,
|
||||
"resolveType": "Skip",
|
||||
"holdResult": true,
|
||||
}
|
||||
out, err := c.putJSONObject(ctx, "/api/2.0/files/fileops/move.json", body)
|
||||
if err != nil {
|
||||
out, err = c.putJSONObject(ctx, "/api/2.0/files/fileops/move", body)
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if raw, merr := json.Marshal(out); merr == nil {
|
||||
if ferr := fileopsError(raw); ferr != nil {
|
||||
return nil, ferr
|
||||
}
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
|
||||
// UploadToFolder uploads a local file into an arbitrary Documents folder id.
|
||||
func (c *Client) UploadToFolder(ctx context.Context, folderID, localPath string) (*FileEntry, error) {
|
||||
if folderID == "" || localPath == "" {
|
||||
return nil, fmt.Errorf("folder id and local path are required")
|
||||
}
|
||||
uploadPath := fmt.Sprintf("/api/2.0/files/%s/upload.json", url.PathEscape(folderID))
|
||||
raw, err := c.uploadMultipart(ctx, uploadPath, "file", localPath)
|
||||
if err != nil {
|
||||
uploadPath = fmt.Sprintf("/api/2.0/files/%s/upload", url.PathEscape(folderID))
|
||||
raw, err = c.uploadMultipart(ctx, uploadPath, "file", localPath)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
return decodeResponseFileEntry(raw)
|
||||
}
|
||||
|
||||
// UpdateFile uploads a new version of an existing file (same id, name and
|
||||
// folder). It does not delete and does not create a second file.
|
||||
//
|
||||
// The Documents API method is PUT /api/2.0/files/{id}/update; POST is kept as
|
||||
// a fallback for older servers. The path is tried with and without .json.
|
||||
func (c *Client) UpdateFile(ctx context.Context, fileID, localPath string) (*FileEntry, error) {
|
||||
if fileID == "" || localPath == "" {
|
||||
return nil, fmt.Errorf("file id and local path are required")
|
||||
}
|
||||
base := fmt.Sprintf("/api/2.0/files/%s/update", url.PathEscape(fileID))
|
||||
attempts := []struct {
|
||||
method, path string
|
||||
}{
|
||||
{http.MethodPut, base},
|
||||
{http.MethodPut, base + ".json"},
|
||||
{http.MethodPost, base},
|
||||
{http.MethodPost, base + ".json"},
|
||||
}
|
||||
var lastErr error
|
||||
for _, a := range attempts {
|
||||
raw, err := c.uploadMultipartMethod(ctx, a.method, a.path, "file", localPath)
|
||||
if err == nil {
|
||||
return decodeResponseFileEntry(raw)
|
||||
}
|
||||
lastErr = err
|
||||
}
|
||||
return nil, lastErr
|
||||
}
|
||||
|
||||
// FileFolderID returns the parent folder id string for a file entry, if known.
|
||||
func FileFolderID(f *FileEntry) string {
|
||||
if f == nil || f.FolderID == nil {
|
||||
return ""
|
||||
}
|
||||
return f.FolderID.String()
|
||||
}
|
||||
|
||||
// DownloadFile streams file bytes from the file's viewUrl using the same auth
|
||||
// as API calls. Writes into dst.
|
||||
// as API calls. Writes into dst. When the portal serves the file from its stale
|
||||
// AWS S3 consumer, the bytes are fetched from the local MinIO store instead
|
||||
// (see storage_fallback.go).
|
||||
func (c *Client) DownloadFile(ctx context.Context, fileID string, dst io.Writer) (int64, error) {
|
||||
f, err := c.GetFile(ctx, fileID)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if f.ViewURL == nil || *f.ViewURL == "" {
|
||||
return 0, fmt.Errorf("file %s has no viewUrl", fileID)
|
||||
}
|
||||
downloadURL := c.resolveAPIURL(*f.ViewURL)
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, downloadURL, nil)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
req.Header.Set("Authorization", auth)
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode >= 400 {
|
||||
b, _ := io.ReadAll(io.LimitReader(resp.Body, 512))
|
||||
return 0, fmt.Errorf("GET viewUrl: %d %s", resp.StatusCode, truncate(string(b), 400))
|
||||
}
|
||||
n, err := io.Copy(dst, resp.Body)
|
||||
return n, err
|
||||
return c.downloadFileEntry(ctx, f, dst)
|
||||
}
|
||||
|
||||
func (c *Client) resolveAPIURL(ref string) string {
|
||||
@@ -319,8 +410,18 @@ func (c *Client) resolveAPIURL(ref string) string {
|
||||
if ref == "" {
|
||||
return ref
|
||||
}
|
||||
if strings.HasPrefix(ref, "http://") || strings.HasPrefix(ref, "https://") {
|
||||
return ref
|
||||
// Rewrite any host to the configured API base so downloads stay on the
|
||||
// internal network and keep the Authorization header (no cross-host
|
||||
// redirect that would strip it). Scheme-relative URLs are handled too.
|
||||
if strings.HasPrefix(ref, "//") {
|
||||
ref = "http:" + ref
|
||||
}
|
||||
if u, err := url.Parse(ref); err == nil && u.IsAbs() {
|
||||
if base, err2 := url.Parse(c.baseURL()); err2 == nil {
|
||||
u.Scheme = base.Scheme
|
||||
u.Host = base.Host
|
||||
return u.String()
|
||||
}
|
||||
}
|
||||
base := c.baseURL()
|
||||
if strings.HasPrefix(ref, "/") {
|
||||
|
||||
+383
@@ -0,0 +1,383 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// FileEntryExt returns a normalized extension (lowercase, with leading dot).
|
||||
func FileEntryExt(f *FileEntry) string {
|
||||
if f == nil {
|
||||
return ""
|
||||
}
|
||||
exst := ""
|
||||
if f.FileExst != nil {
|
||||
exst = strings.TrimSpace(*f.FileExst)
|
||||
}
|
||||
if exst != "" {
|
||||
if !strings.HasPrefix(exst, ".") {
|
||||
exst = "." + exst
|
||||
}
|
||||
return strings.ToLower(exst)
|
||||
}
|
||||
if f.Title != nil {
|
||||
if ext := filepath.Ext(*f.Title); ext != "" {
|
||||
return strings.ToLower(ext)
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// FileDedupKey is stem|ext — two files with the same key are duplicates.
|
||||
func FileDedupKey(f *FileEntry) string {
|
||||
st := FileEntryStem(f)
|
||||
ext := FileEntryExt(f)
|
||||
if st == "" {
|
||||
return ""
|
||||
}
|
||||
if ext == "" {
|
||||
return st
|
||||
}
|
||||
return st + "|" + strings.TrimPrefix(ext, ".")
|
||||
}
|
||||
|
||||
// FindFilesByDedupKey returns folder files matching stem and extension.
|
||||
func FindFilesByDedupKey(files []*FileEntry, stem, ext string) []*FileEntry {
|
||||
key := dedupKeyFromParts(stem, ext)
|
||||
if key == "" {
|
||||
return nil
|
||||
}
|
||||
var out []*FileEntry
|
||||
for _, f := range files {
|
||||
if FileDedupKey(f) == key {
|
||||
out = append(out, f)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func dedupKeyFromParts(stem, ext string) string {
|
||||
stem = strings.TrimSpace(stem)
|
||||
if stem == "" {
|
||||
return ""
|
||||
}
|
||||
ext = strings.ToLower(strings.TrimSpace(ext))
|
||||
if ext != "" && !strings.HasPrefix(ext, ".") {
|
||||
ext = "." + ext
|
||||
}
|
||||
if ext == "" {
|
||||
return stem
|
||||
}
|
||||
return stem + "|" + strings.TrimPrefix(ext, ".")
|
||||
}
|
||||
|
||||
// UploadExtFromLocal returns the lowercase extension from a local path.
|
||||
func UploadExtFromLocal(localPath string) string {
|
||||
ext := filepath.Ext(localPath)
|
||||
if ext == "" {
|
||||
return ""
|
||||
}
|
||||
return strings.ToLower(ext)
|
||||
}
|
||||
|
||||
// IsTrashFolderTitle reports staging/trash folders (e.g. _trash-md).
|
||||
func IsTrashFolderTitle(title string) bool {
|
||||
t := strings.ToLower(strings.TrimSpace(title))
|
||||
return strings.HasPrefix(t, "_") || strings.Contains(t, "trash")
|
||||
}
|
||||
|
||||
// ProjectFolderFile ties a file to its project Documents subfolder.
|
||||
type ProjectFolderFile struct {
|
||||
FolderID string
|
||||
FolderTitle string
|
||||
File *FileEntry
|
||||
}
|
||||
|
||||
// DedupGroup is one duplicate set: keep the newest (or non-trash) file.
|
||||
type DedupGroup struct {
|
||||
Key string
|
||||
FolderID string
|
||||
FolderTitle string
|
||||
Keep *FileEntry
|
||||
Remove []*FileEntry
|
||||
}
|
||||
|
||||
// DedupOptions controls project-wide duplicate scans.
|
||||
type DedupOptions struct {
|
||||
CrossFolder bool
|
||||
}
|
||||
|
||||
// FindProjectDuplicates scans project folders for duplicate files.
|
||||
func FindProjectDuplicates(folders []*FolderEntry, filesByFolder map[string][]*FileEntry, opts DedupOptions) []DedupGroup {
|
||||
var indexed []ProjectFolderFile
|
||||
for _, folder := range folders {
|
||||
if folder == nil || folder.ID == nil {
|
||||
continue
|
||||
}
|
||||
fid := folder.ID.String()
|
||||
title := ""
|
||||
if folder.Title != nil {
|
||||
title = *folder.Title
|
||||
}
|
||||
for _, f := range filesByFolder[fid] {
|
||||
if f == nil {
|
||||
continue
|
||||
}
|
||||
indexed = append(indexed, ProjectFolderFile{
|
||||
FolderID: fid, FolderTitle: title, File: f,
|
||||
})
|
||||
}
|
||||
}
|
||||
if opts.CrossFolder {
|
||||
return findCrossFolderDuplicates(indexed)
|
||||
}
|
||||
return findWithinFolderDuplicates(indexed)
|
||||
}
|
||||
|
||||
func findWithinFolderDuplicates(indexed []ProjectFolderFile) []DedupGroup {
|
||||
byFolder := map[string][]ProjectFolderFile{}
|
||||
for _, it := range indexed {
|
||||
byFolder[it.FolderID] = append(byFolder[it.FolderID], it)
|
||||
}
|
||||
var out []DedupGroup
|
||||
for fid, items := range byFolder {
|
||||
title := ""
|
||||
if len(items) > 0 {
|
||||
title = items[0].FolderTitle
|
||||
}
|
||||
byKey := map[string][]*FileEntry{}
|
||||
for _, it := range items {
|
||||
k := FileDedupKey(it.File)
|
||||
byKey[k] = append(byKey[k], it.File)
|
||||
}
|
||||
for k, group := range byKey {
|
||||
if len(group) < 2 {
|
||||
continue
|
||||
}
|
||||
keep, remove := pickDuplicateKeeper(group, false)
|
||||
if keep == nil || len(remove) == 0 {
|
||||
continue
|
||||
}
|
||||
out = append(out, DedupGroup{
|
||||
Key: k, FolderID: fid, FolderTitle: title, Keep: keep, Remove: remove,
|
||||
})
|
||||
}
|
||||
}
|
||||
sortDedupGroups(out)
|
||||
return out
|
||||
}
|
||||
|
||||
func findCrossFolderDuplicates(indexed []ProjectFolderFile) []DedupGroup {
|
||||
byKey := map[string][]ProjectFolderFile{}
|
||||
for _, it := range indexed {
|
||||
k := FileDedupKey(it.File)
|
||||
byKey[k] = append(byKey[k], it)
|
||||
}
|
||||
var out []DedupGroup
|
||||
for k, items := range byKey {
|
||||
if len(items) < 2 {
|
||||
continue
|
||||
}
|
||||
files := make([]*FileEntry, len(items))
|
||||
folders := make([]string, len(items))
|
||||
folderTitles := make([]string, len(items))
|
||||
for i, it := range items {
|
||||
files[i] = it.File
|
||||
folders[i] = it.FolderID
|
||||
folderTitles[i] = it.FolderTitle
|
||||
}
|
||||
keep, remove := pickDuplicateKeeperWithFolders(files, folders, folderTitles)
|
||||
if keep == nil || len(remove) == 0 {
|
||||
continue
|
||||
}
|
||||
fid, ftitle := "", ""
|
||||
for _, it := range items {
|
||||
if it.File == keep {
|
||||
fid, ftitle = it.FolderID, it.FolderTitle
|
||||
break
|
||||
}
|
||||
}
|
||||
out = append(out, DedupGroup{
|
||||
Key: k, FolderID: fid, FolderTitle: ftitle, Keep: keep, Remove: remove,
|
||||
})
|
||||
}
|
||||
sortDedupGroups(out)
|
||||
return out
|
||||
}
|
||||
|
||||
func pickDuplicateKeeper(files []*FileEntry, _ bool) (*FileEntry, []*FileEntry) {
|
||||
return pickDuplicateKeeperWithFolders(files, nil, nil)
|
||||
}
|
||||
|
||||
func pickDuplicateKeeperWithFolders(files []*FileEntry, folderIDs, folderTitles []string) (*FileEntry, []*FileEntry) {
|
||||
if len(files) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
type ranked struct {
|
||||
file *FileEntry
|
||||
trash bool
|
||||
}
|
||||
rankedFiles := make([]ranked, len(files))
|
||||
for i, f := range files {
|
||||
trash := false
|
||||
if folderTitles != nil && i < len(folderTitles) {
|
||||
trash = IsTrashFolderTitle(folderTitles[i])
|
||||
}
|
||||
rankedFiles[i] = ranked{file: f, trash: trash}
|
||||
}
|
||||
sort.SliceStable(rankedFiles, func(i, j int) bool {
|
||||
ri, rj := rankedFiles[i], rankedFiles[j]
|
||||
if ri.trash != rj.trash {
|
||||
return !ri.trash // non-trash first
|
||||
}
|
||||
ti, tj := rankedFiles[i].file.Updated, rankedFiles[j].file.Updated
|
||||
if ti == nil {
|
||||
return false
|
||||
}
|
||||
if tj == nil {
|
||||
return true
|
||||
}
|
||||
return ti.After(*tj) // newest first
|
||||
})
|
||||
keep := rankedFiles[0].file
|
||||
var remove []*FileEntry
|
||||
for _, r := range rankedFiles[1:] {
|
||||
remove = append(remove, r.file)
|
||||
}
|
||||
return keep, remove
|
||||
}
|
||||
|
||||
func sortDedupGroups(groups []DedupGroup) {
|
||||
sort.Slice(groups, func(i, j int) bool {
|
||||
if groups[i].FolderTitle != groups[j].FolderTitle {
|
||||
return groups[i].FolderTitle < groups[j].FolderTitle
|
||||
}
|
||||
return groups[i].Key < groups[j].Key
|
||||
})
|
||||
}
|
||||
|
||||
// ApplyDedupGroups deletes Remove files from each group.
|
||||
func (c *Client) ApplyDedupGroups(ctx context.Context, groups []DedupGroup) ([]int, error) {
|
||||
seen := map[int]struct{}{}
|
||||
var ids []int
|
||||
for _, g := range groups {
|
||||
for _, f := range g.Remove {
|
||||
n := int(FileEntryNumericID(f))
|
||||
if n == 0 {
|
||||
continue
|
||||
}
|
||||
if _, ok := seen[n]; ok {
|
||||
continue
|
||||
}
|
||||
seen[n] = struct{}{}
|
||||
ids = append(ids, n)
|
||||
}
|
||||
}
|
||||
if len(ids) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
if err := c.DeleteFiles(ctx, ids); err != nil {
|
||||
return ids, err
|
||||
}
|
||||
return ids, nil
|
||||
}
|
||||
|
||||
// DeleteFilesByDedupKey removes all files in folderID matching stem+ext.
|
||||
func (c *Client) DeleteFilesByDedupKey(ctx context.Context, folderID, stem, ext string) ([]int, error) {
|
||||
files, err := c.FolderFiles(ctx, folderID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
matches := FindFilesByDedupKey(files, stem, ext)
|
||||
if len(matches) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
ids := make([]int, 0, len(matches))
|
||||
for _, f := range matches {
|
||||
n := int(FileEntryNumericID(f))
|
||||
if n != 0 {
|
||||
ids = append(ids, n)
|
||||
}
|
||||
}
|
||||
if len(ids) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
if err := c.DeleteFiles(ctx, ids); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ids, nil
|
||||
}
|
||||
|
||||
// mergeProjectRootForDedupe includes projectFolder files in dedupe scans. OO often lists
|
||||
// root documents only in pf.Files while pf.Folders is empty.
|
||||
func mergeProjectRootForDedupe(rootID string, folders []*FolderEntry, filesByFolder map[string][]*FileEntry, rootFiles []*FileEntry) ([]*FolderEntry, map[string][]*FileEntry) {
|
||||
if rootID == "" {
|
||||
return folders, filesByFolder
|
||||
}
|
||||
if filesByFolder == nil {
|
||||
filesByFolder = map[string][]*FileEntry{}
|
||||
}
|
||||
for _, folder := range folders {
|
||||
if folder != nil && folder.ID != nil && folder.ID.String() == rootID {
|
||||
if len(rootFiles) > 0 {
|
||||
filesByFolder[rootID] = rootFiles
|
||||
}
|
||||
return folders, filesByFolder
|
||||
}
|
||||
}
|
||||
if len(rootFiles) == 0 {
|
||||
return folders, filesByFolder
|
||||
}
|
||||
id := json.Number(rootID)
|
||||
title := "(project root)"
|
||||
folders = append(folders, &FolderEntry{ID: &id, Title: &title})
|
||||
filesByFolder[rootID] = rootFiles
|
||||
return folders, filesByFolder
|
||||
}
|
||||
|
||||
// DedupeProject scans project folders and optionally deletes duplicates.
|
||||
func (c *Client) DedupeProject(ctx context.Context, projectID string, opts DedupOptions, apply bool) ([]DedupGroup, []int, error) {
|
||||
pf, err := c.GetProjectFiles(ctx, projectID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
rootID, err := c.projectFolderID(ctx, projectID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
var rootFiles []*FileEntry
|
||||
if rootID != "" {
|
||||
rootFiles, err = c.FolderFiles(ctx, rootID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
}
|
||||
filesByFolder := make(map[string][]*FileEntry, len(pf.Folders)+1)
|
||||
folders := make([]*FolderEntry, 0, len(pf.Folders)+1)
|
||||
for _, folder := range pf.Folders {
|
||||
if folder == nil || folder.ID == nil {
|
||||
continue
|
||||
}
|
||||
fid := folder.ID.String()
|
||||
if fid == rootID {
|
||||
filesByFolder[fid] = rootFiles
|
||||
} else {
|
||||
files, err := c.FolderFiles(ctx, fid)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
filesByFolder[fid] = files
|
||||
}
|
||||
folders = append(folders, folder)
|
||||
}
|
||||
folders, filesByFolder = mergeProjectRootForDedupe(rootID, folders, filesByFolder, rootFiles)
|
||||
groups := FindProjectDuplicates(folders, filesByFolder, opts)
|
||||
if !apply || len(groups) == 0 {
|
||||
return groups, nil, nil
|
||||
}
|
||||
deleted, err := c.ApplyDedupGroups(ctx, groups)
|
||||
return groups, deleted, err
|
||||
}
|
||||
@@ -0,0 +1,101 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestFileDedupKey(t *testing.T) {
|
||||
title := "OO-HONDA-7-INDEX.docx"
|
||||
exst := ".docx"
|
||||
f := &FileEntry{Title: &title, FileExst: &exst}
|
||||
if got := FileDedupKey(f); got != "OO-HONDA-7-INDEX|docx" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFindFilesByDedupKey(t *testing.T) {
|
||||
a := &FileEntry{Title: strPtr("foo.docx"), FileExst: strPtr(".docx")}
|
||||
b := &FileEntry{Title: strPtr("foo.md"), FileExst: strPtr(".md")}
|
||||
files := []*FileEntry{a, b}
|
||||
got := FindFilesByDedupKey(files, "foo", ".docx")
|
||||
if len(got) != 1 || got[0] != a {
|
||||
t.Fatalf("got %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFindWithinFolderDuplicates(t *testing.T) {
|
||||
t1 := time.Date(2026, 8, 27, 16, 0, 0, 0, time.UTC)
|
||||
t2 := t1.Add(time.Hour)
|
||||
old := &FileEntry{ID: jsonNum("1"), Title: strPtr("idx.docx"), FileExst: strPtr(".docx"), Updated: &t1}
|
||||
new := &FileEntry{ID: jsonNum("2"), Title: strPtr("idx.docx"), FileExst: strPtr(".docx"), Updated: &t2}
|
||||
indexed := []ProjectFolderFile{
|
||||
{FolderID: "490", FolderTitle: "00-Index", File: old},
|
||||
{FolderID: "490", FolderTitle: "00-Index", File: new},
|
||||
}
|
||||
groups := findWithinFolderDuplicates(indexed)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("groups=%d", len(groups))
|
||||
}
|
||||
if FileEntryNumericID(groups[0].Keep) != 2 {
|
||||
t.Fatalf("keep id=%d", FileEntryNumericID(groups[0].Keep))
|
||||
}
|
||||
if len(groups[0].Remove) != 1 || FileEntryNumericID(groups[0].Remove[0]) != 1 {
|
||||
t.Fatalf("remove=%v", groups[0].Remove)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCrossFolderPrefersNonTrash(t *testing.T) {
|
||||
t1 := time.Date(2026, 8, 27, 18, 0, 0, 0, time.UTC)
|
||||
t2 := t1.Add(-time.Hour)
|
||||
trash := &FileEntry{ID: jsonNum("10"), Title: strPtr("INDEX.md"), FileExst: strPtr(".md"), Updated: &t1}
|
||||
good := &FileEntry{ID: jsonNum("20"), Title: strPtr("INDEX.md"), FileExst: strPtr(".md"), Updated: &t2}
|
||||
indexed := []ProjectFolderFile{
|
||||
{FolderID: "493", FolderTitle: "_trash-md", File: trash},
|
||||
{FolderID: "492", FolderTitle: "OCR", File: good},
|
||||
}
|
||||
groups := findCrossFolderDuplicates(indexed)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("groups=%d", len(groups))
|
||||
}
|
||||
if FileEntryNumericID(groups[0].Keep) != 20 {
|
||||
t.Fatalf("keep id=%d", FileEntryNumericID(groups[0].Keep))
|
||||
}
|
||||
}
|
||||
|
||||
func TestMergeProjectRootForDedupe(t *testing.T) {
|
||||
old := &FileEntry{ID: jsonNum("1"), Title: strPtr("a.docx"), FileExst: strPtr(".docx")}
|
||||
newer := &FileEntry{ID: jsonNum("2"), Title: strPtr("a.docx"), FileExst: strPtr(".docx")}
|
||||
rootFiles := []*FileEntry{old, newer}
|
||||
folders, byFolder := mergeProjectRootForDedupe("489", nil, nil, rootFiles)
|
||||
if len(folders) != 1 || folders[0].ID.String() != "489" {
|
||||
t.Fatalf("folders=%+v", folders)
|
||||
}
|
||||
if len(byFolder["489"]) != 2 {
|
||||
t.Fatalf("root files=%d", len(byFolder["489"]))
|
||||
}
|
||||
groups := findWithinFolderDuplicates([]ProjectFolderFile{
|
||||
{FolderID: "489", FolderTitle: "(project root)", File: old},
|
||||
{FolderID: "489", FolderTitle: "(project root)", File: newer},
|
||||
})
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("groups=%d", len(groups))
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsTrashFolderTitle(t *testing.T) {
|
||||
if !IsTrashFolderTitle("_trash-md") {
|
||||
t.Fatal("expected trash")
|
||||
}
|
||||
if IsTrashFolderTitle("00-Index") {
|
||||
t.Fatal("expected not trash")
|
||||
}
|
||||
}
|
||||
|
||||
func strPtr(s string) *string { return &s }
|
||||
|
||||
func jsonNum(s string) *json.Number {
|
||||
n := json.Number(s)
|
||||
return &n
|
||||
}
|
||||
+166
@@ -0,0 +1,166 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ErrFileExists is returned when --no-replace / no-clobber upload hits an existing stem|ext.
|
||||
var ErrFileExists = errors.New("onlyoffice: file already exists in folder (use replace or delete first)")
|
||||
|
||||
// FileEntryStem returns the logical basename without duplicated extensions.
|
||||
// OO often stores title="foo.docx" and fileExst=".docx" (UI shows foo.docx.docx).
|
||||
func FileEntryStem(f *FileEntry) string {
|
||||
if f == nil || f.Title == nil {
|
||||
return ""
|
||||
}
|
||||
exst := ""
|
||||
if f.FileExst != nil {
|
||||
exst = *f.FileExst
|
||||
}
|
||||
return NormalizeUploadStem(*f.Title, exst)
|
||||
}
|
||||
|
||||
// NormalizeUploadStem derives a stable stem for matching uploads.
|
||||
func NormalizeUploadStem(title, exst string) string {
|
||||
t := strings.TrimSpace(title)
|
||||
t = strings.TrimSuffix(t, ".")
|
||||
if exst != "" && strings.HasSuffix(t, exst) {
|
||||
t = strings.TrimSuffix(t, exst)
|
||||
}
|
||||
if ext := filepath.Ext(t); ext != "" {
|
||||
t = strings.TrimSuffix(t, ext)
|
||||
}
|
||||
return strings.TrimSpace(t)
|
||||
}
|
||||
|
||||
// UploadStemFromLocal returns the stem used to match/replace folder files.
|
||||
func UploadStemFromLocal(localPath string) string {
|
||||
base := filepath.Base(localPath)
|
||||
ext := filepath.Ext(base)
|
||||
if ext != "" {
|
||||
base = strings.TrimSuffix(base, ext)
|
||||
}
|
||||
return base
|
||||
}
|
||||
|
||||
// FolderFiles returns file entries in a Documents folder.
|
||||
func (c *Client) FolderFiles(ctx context.Context, folderID string) ([]*FileEntry, error) {
|
||||
raw, err := c.ListFolder(ctx, folderID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ParseFolderFileEntries(raw), nil
|
||||
}
|
||||
|
||||
// ParseFolderFileEntries extracts []*FileEntry from ListFolder JSON.
|
||||
func ParseFolderFileEntries(raw map[string]any) []*FileEntry {
|
||||
items, _ := raw["files"].([]any)
|
||||
out := make([]*FileEntry, 0, len(items))
|
||||
for _, it := range items {
|
||||
m, ok := it.(map[string]any)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
b, err := json.Marshal(m)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
var f FileEntry
|
||||
if err := json.Unmarshal(b, &f); err != nil {
|
||||
continue
|
||||
}
|
||||
out = append(out, &f)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// FindFilesByStem returns folder files whose logical stem matches.
|
||||
func FindFilesByStem(files []*FileEntry, stem string) []*FileEntry {
|
||||
stem = strings.TrimSpace(stem)
|
||||
if stem == "" {
|
||||
return nil
|
||||
}
|
||||
var out []*FileEntry
|
||||
for _, f := range files {
|
||||
if FileEntryStem(f) == stem {
|
||||
out = append(out, f)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// DeleteFilesByStem removes all files in folderID matching stem (any extension).
|
||||
// Prefer DeleteFilesByDedupKey when the upload extension is known.
|
||||
func (c *Client) DeleteFilesByStem(ctx context.Context, folderID, stem string) ([]int, error) {
|
||||
files, err := c.FolderFiles(ctx, folderID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
matches := FindFilesByStem(files, stem)
|
||||
if len(matches) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
ids := make([]int, 0, len(matches))
|
||||
for _, f := range matches {
|
||||
n := int(FileEntryNumericID(f))
|
||||
if n != 0 {
|
||||
ids = append(ids, n)
|
||||
}
|
||||
}
|
||||
if len(ids) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
if err := c.DeleteFiles(ctx, ids); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ids, nil
|
||||
}
|
||||
|
||||
// AssertNoFileConflict reports ErrFileExists when localPath stem|ext is already in folderID.
|
||||
func (c *Client) AssertNoFileConflict(ctx context.Context, folderID, localPath string) error {
|
||||
files, err := c.FolderFiles(ctx, folderID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
stem := UploadStemFromLocal(localPath)
|
||||
ext := UploadExtFromLocal(localPath)
|
||||
matches := FindFilesByDedupKey(files, stem, ext)
|
||||
if len(matches) == 0 {
|
||||
return nil
|
||||
}
|
||||
ids := make([]string, 0, len(matches))
|
||||
for _, f := range matches {
|
||||
ids = append(ids, fmt.Sprintf("%d", FileEntryNumericID(f)))
|
||||
}
|
||||
return fmt.Errorf("%w: %s%s in folder %s (existing file ids: %s)",
|
||||
ErrFileExists, stem, ext, folderID, strings.Join(ids, ", "))
|
||||
}
|
||||
|
||||
// UploadProjectFileNoClobber uploads only when stem|ext is not already in the project folder.
|
||||
func (c *Client) UploadProjectFileNoClobber(ctx context.Context, projectID, localPath string) (*FileEntry, error) {
|
||||
folderID, err := c.projectFolderID(ctx, projectID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := c.AssertNoFileConflict(ctx, folderID, localPath); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return c.UploadProjectFile(ctx, projectID, localPath)
|
||||
}
|
||||
|
||||
// UploadToFolderReplacing deletes same stem+ext files then uploads localPath.
|
||||
func (c *Client) UploadToFolderReplacing(ctx context.Context, folderID, localPath string) (*FileEntry, []int, error) {
|
||||
stem := UploadStemFromLocal(localPath)
|
||||
ext := UploadExtFromLocal(localPath)
|
||||
deleted, err := c.DeleteFilesByDedupKey(ctx, folderID, stem, ext)
|
||||
if err != nil {
|
||||
return nil, deleted, err
|
||||
}
|
||||
ent, err := c.UploadToFolder(ctx, folderID, localPath)
|
||||
return ent, deleted, err
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package onlyoffice
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestNormalizeUploadStem(t *testing.T) {
|
||||
tests := []struct {
|
||||
title, exst, want string
|
||||
}{
|
||||
{"OO-HONDA-7-INDEX.docx", ".docx", "OO-HONDA-7-INDEX"},
|
||||
{"README.txt", ".txt", "README"},
|
||||
{"car-docs-print.docx", ".docx", "car-docs-print"},
|
||||
{"plain.", "", "plain"},
|
||||
{"foo", ".docx", "foo"},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
if got := NormalizeUploadStem(tc.title, tc.exst); got != tc.want {
|
||||
t.Fatalf("NormalizeUploadStem(%q,%q)=%q want %q", tc.title, tc.exst, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileEntryStem(t *testing.T) {
|
||||
title := "00-INDEX.docx"
|
||||
exst := ".docx"
|
||||
f := &FileEntry{Title: &title, FileExst: &exst}
|
||||
if got := FileEntryStem(f); got != "00-INDEX" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUploadStemFromLocal(t *testing.T) {
|
||||
if got := UploadStemFromLocal("/tmp/OO-HONDA-7-INDEX.docx"); got != "OO-HONDA-7-INDEX" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
}
|
||||
+455
@@ -0,0 +1,455 @@
|
||||
package onlyoffice
|
||||
|
||||
// WebDAV-oriented Files operations. These expose the Documents module through
|
||||
// value types and cover everything needed to back a filesystem mapping:
|
||||
// listing (including the virtual @root sections), folder/file CRUD, move/copy,
|
||||
// and streaming upload/download. They are intentionally small and dependency
|
||||
// free (only net/http), so callers are not forced to import heavier parts of
|
||||
// the library.
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"mime/multipart"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// DavFolder is a folder row from the Files module.
|
||||
type DavFolder struct {
|
||||
ID string
|
||||
Title string
|
||||
ParentID string
|
||||
RootType int // 1=Common, 3=Trash, 5=My, 6=Share, 8=Projects, ...
|
||||
FilesCount int
|
||||
FoldersCount int
|
||||
Access int
|
||||
Shared bool
|
||||
Updated string
|
||||
}
|
||||
|
||||
// DavFile is a file row from the Files module.
|
||||
type DavFile struct {
|
||||
ID string
|
||||
Title string
|
||||
Size int64
|
||||
Updated string
|
||||
ViewURL string
|
||||
}
|
||||
|
||||
// DavListing is the contents of one folder.
|
||||
type DavListing struct {
|
||||
Current DavFolder
|
||||
Files []DavFile
|
||||
Folders []DavFolder
|
||||
}
|
||||
|
||||
// ListDavFolder returns the contents of a folder by id, which may be a
|
||||
// symbolic root such as "@my". For "@root" use ListDavSections.
|
||||
func (c *Client) ListDavFolder(ctx context.Context, id string) (*DavListing, error) {
|
||||
raw, err := c.getJSON(ctx, "/api/2.0/files/"+url.PathEscape(id))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
resp, err := responseField(raw, "response")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// @root returns an array with a single blob; a normal folder returns an
|
||||
// object. Normalize both.
|
||||
if len(resp) > 0 && resp[0] == '[' {
|
||||
var arr []*DavListing
|
||||
if err := json.Unmarshal(resp, &arr); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(arr) == 0 {
|
||||
return &DavListing{}, nil
|
||||
}
|
||||
return arr[0], nil
|
||||
}
|
||||
var l DavListing
|
||||
if err := json.Unmarshal(resp, &l); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &l, nil
|
||||
}
|
||||
|
||||
// ListDavSections returns the virtual top-level sections shown by @root
|
||||
// ("In projects", "My documents", "Shared with me", "Common", "Favorites",
|
||||
// "Recent", "Trash"). Each is the `current` folder of one @root element.
|
||||
func (c *Client) ListDavSections(ctx context.Context) ([]DavFolder, error) {
|
||||
raw, err := c.getJSON(ctx, "/api/2.0/files/@root")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
resp, err := responseField(raw, "response")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var arr []struct {
|
||||
Current DavFolder `json:"current"`
|
||||
}
|
||||
if err := json.Unmarshal(resp, &arr); err != nil {
|
||||
// Tolerate a non-array (single listing) response.
|
||||
var single DavListing
|
||||
if err2 := json.Unmarshal(resp, &single); err2 != nil {
|
||||
return nil, err
|
||||
}
|
||||
return []DavFolder{single.Current}, nil
|
||||
}
|
||||
sections := make([]DavFolder, 0, len(arr))
|
||||
for i := range arr {
|
||||
sections = append(sections, arr[i].Current)
|
||||
}
|
||||
return sections, nil
|
||||
}
|
||||
|
||||
// CreateDavFolder creates a folder titled title inside parentID.
|
||||
func (c *Client) CreateDavFolder(ctx context.Context, parentID, title string) (*DavFolder, error) {
|
||||
raw, err := c.postJSON(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentID),
|
||||
map[string]string{"title": title})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var env struct {
|
||||
Response *DavFolder `json:"response"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &env); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if env.Response == nil {
|
||||
return nil, fmt.Errorf("onlyoffice: empty create-folder response")
|
||||
}
|
||||
return env.Response, nil
|
||||
}
|
||||
|
||||
// RenameDavFolder renames a folder.
|
||||
func (c *Client) RenameDavFolder(ctx context.Context, id, title string) error {
|
||||
_, err := c.putJSON(ctx, "/api/2.0/files/folder/"+url.PathEscape(id),
|
||||
map[string]string{"title": title})
|
||||
return err
|
||||
}
|
||||
|
||||
// RenameDavFile renames a file (title includes the extension).
|
||||
func (c *Client) RenameDavFile(ctx context.Context, id, title string) error {
|
||||
_, err := c.putJSON(ctx, "/api/2.0/files/file/"+url.PathEscape(id),
|
||||
map[string]string{"title": title})
|
||||
return err
|
||||
}
|
||||
|
||||
// MoveDavItems moves the given folders and/or files into destFolderID.
|
||||
// The fileops API answers 200 with per-operation error strings even when
|
||||
// nothing moves (e.g. missing permission), so the response is parsed and the
|
||||
// first operation error is returned instead of a silent nil.
|
||||
func (c *Client) MoveDavItems(ctx context.Context, folderIDs, fileIDs []string, destFolderID string) error {
|
||||
raw, err := c.putJSON(ctx, "/api/2.0/files/fileops/move", map[string]any{
|
||||
"folderIds": nums(folderIDs),
|
||||
"fileIds": nums(fileIDs),
|
||||
"destFolderId": num(destFolderID),
|
||||
"resolveType": "Skip",
|
||||
"holdResult": true,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return fileopsError(raw)
|
||||
}
|
||||
|
||||
// CopyDavItems copies the given folders and/or files into destFolderID.
|
||||
// Per-operation errors are surfaced like in MoveDavItems.
|
||||
func (c *Client) CopyDavItems(ctx context.Context, folderIDs, fileIDs []string, destFolderID string) error {
|
||||
raw, err := c.putJSON(ctx, "/api/2.0/files/fileops/copy", map[string]any{
|
||||
"folderIds": nums(folderIDs),
|
||||
"fileIds": nums(fileIDs),
|
||||
"destFolderId": num(destFolderID),
|
||||
"conflictResolveType": "Skip",
|
||||
"deleteAfter": true,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return fileopsError(raw)
|
||||
}
|
||||
|
||||
// ListFileOps returns the currently active file operations
|
||||
// (GET /api/2.0/files/fileops) for status polling.
|
||||
func (c *Client) ListFileOps(ctx context.Context) ([]map[string]any, error) {
|
||||
raw, err := c.getJSON(ctx, "/api/2.0/files/fileops")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
resp, err := responseField(raw, "response")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(resp) == 0 || string(resp) == "null" {
|
||||
return nil, nil
|
||||
}
|
||||
var ops []map[string]any
|
||||
if err := json.Unmarshal(resp, &ops); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ops, nil
|
||||
}
|
||||
|
||||
// fileopsError extracts per-operation "error" strings from a fileops/move or
|
||||
// fileops/copy envelope. A 200 with error entries means nothing moved.
|
||||
func fileopsError(raw json.RawMessage) error {
|
||||
resp, err := responseField(raw, "response")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
var ops []struct {
|
||||
Error *string `json:"error"`
|
||||
Finished *bool `json:"finished"`
|
||||
Progress *int `json:"progress"`
|
||||
}
|
||||
if err := json.Unmarshal(resp, &ops); err != nil {
|
||||
return nil // not an operations envelope — nothing to report
|
||||
}
|
||||
var errs []string
|
||||
for _, op := range ops {
|
||||
if op.Error != nil && *op.Error != "" {
|
||||
errs = append(errs, *op.Error)
|
||||
}
|
||||
}
|
||||
if len(errs) > 0 {
|
||||
return fmt.Errorf("onlyoffice: fileops: %s", strings.Join(errs, "; "))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// DeleteDavItems deletes the given folders and/or files.
|
||||
func (c *Client) DeleteDavItems(ctx context.Context, folderIDs, fileIDs []string) error {
|
||||
body := map[string]any{"DeleteAfter": true, "Immediately": true}
|
||||
for _, id := range folderIDs {
|
||||
if _, err := c.deleteJSON(ctx, "/api/2.0/files/folder/"+url.PathEscape(id), body); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
for _, id := range fileIDs {
|
||||
if _, err := c.deleteJSON(ctx, "/api/2.0/files/file/"+url.PathEscape(id), body); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// UploadDavFile uploads src (fileName) into folderID, streaming from src.
|
||||
func (c *Client) UploadDavFile(ctx context.Context, folderID, fileName string, src io.Reader) (*DavFile, error) {
|
||||
raw, err := c.uploadReader(ctx, "/api/2.0/files/"+url.PathEscape(folderID)+"/upload", "file", fileName, src)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var env struct {
|
||||
Response *DavFile `json:"response"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &env); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if env.Response == nil {
|
||||
return nil, fmt.Errorf("onlyoffice: empty upload response")
|
||||
}
|
||||
return env.Response, nil
|
||||
}
|
||||
|
||||
// DownloadDavFile streams the file identified by id to w, returning bytes
|
||||
// copied. It shares the MinIO stale-S3 fallback with DownloadFile.
|
||||
func (c *Client) DownloadDavFile(ctx context.Context, id string, w io.Writer) (int64, error) {
|
||||
return c.DownloadFile(ctx, id, w)
|
||||
}
|
||||
|
||||
// --- internal helpers -------------------------------------------------------
|
||||
|
||||
// deleteJSON performs an authenticated DELETE with an optional JSON body.
|
||||
func (c *Client) deleteJSON(ctx context.Context, path string, body any) (json.RawMessage, error) {
|
||||
var reader io.Reader
|
||||
if body != nil {
|
||||
buf, err := json.Marshal(body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
reader = bytes.NewReader(buf)
|
||||
}
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodDelete, c.baseURL()+path, reader)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Authorization", auth)
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
req.Header.Set("Accept", "application/json")
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
raw, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("DELETE %s: %d %s", path, resp.StatusCode, truncate(string(raw), 400))
|
||||
}
|
||||
return raw, nil
|
||||
}
|
||||
|
||||
// uploadReader uploads a stream to path under the given form field name.
|
||||
func (c *Client) uploadReader(ctx context.Context, path, fieldName, fileName string, src io.Reader) (json.RawMessage, error) {
|
||||
var buf bytes.Buffer
|
||||
mw := multipart.NewWriter(&buf)
|
||||
part, err := mw.CreateFormFile(fieldName, fileName)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if _, err := io.Copy(part, src); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := mw.Close(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.baseURL()+path, &buf)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Authorization", auth)
|
||||
req.Header.Set("Content-Type", mw.FormDataContentType())
|
||||
req.Header.Set("Accept", "application/json")
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
raw, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("upload %s: %d %s", path, resp.StatusCode, truncate(string(raw), 400))
|
||||
}
|
||||
return raw, nil
|
||||
}
|
||||
|
||||
func nums(ids []string) []json.Number {
|
||||
out := make([]json.Number, 0, len(ids))
|
||||
for _, id := range ids {
|
||||
if _, err := strconv.Atoi(id); err == nil {
|
||||
out = append(out, json.Number(id))
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func num(id string) any {
|
||||
if _, err := strconv.Atoi(id); err == nil {
|
||||
return json.Number(id)
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
// UnmarshalJSON decodes a folder from the portal envelope, including fields
|
||||
// that the base FolderEntry omits (parentId, rootFolderType, access, ...).
|
||||
func (f *DavFolder) UnmarshalJSON(b []byte) error {
|
||||
var raw struct {
|
||||
ID *json.Number `json:"id"`
|
||||
Title *string `json:"title"`
|
||||
ParentID *json.Number `json:"parentId"`
|
||||
RootType *int `json:"rootFolderType"`
|
||||
FilesCount *int `json:"filesCount"`
|
||||
FoldersCount *int `json:"foldersCount"`
|
||||
Access *int `json:"access"`
|
||||
Shared *bool `json:"shared"`
|
||||
Updated *string `json:"updated"`
|
||||
}
|
||||
if err := json.Unmarshal(b, &raw); err != nil {
|
||||
return err
|
||||
}
|
||||
if raw.ID != nil {
|
||||
f.ID = raw.ID.String()
|
||||
}
|
||||
if raw.Title != nil {
|
||||
f.Title = *raw.Title
|
||||
}
|
||||
if raw.ParentID != nil {
|
||||
f.ParentID = raw.ParentID.String()
|
||||
}
|
||||
if raw.RootType != nil {
|
||||
f.RootType = *raw.RootType
|
||||
}
|
||||
if raw.FilesCount != nil {
|
||||
f.FilesCount = *raw.FilesCount
|
||||
}
|
||||
if raw.FoldersCount != nil {
|
||||
f.FoldersCount = *raw.FoldersCount
|
||||
}
|
||||
if raw.Access != nil {
|
||||
f.Access = *raw.Access
|
||||
}
|
||||
if raw.Shared != nil {
|
||||
f.Shared = *raw.Shared
|
||||
}
|
||||
if raw.Updated != nil {
|
||||
f.Updated = *raw.Updated
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// UnmarshalJSON decodes a file row, capturing size and timestamps.
|
||||
func (f *DavFile) UnmarshalJSON(b []byte) error {
|
||||
var raw struct {
|
||||
ID *json.Number `json:"id"`
|
||||
Title *string `json:"title"`
|
||||
PureSize *int64 `json:"pureContentLength"`
|
||||
SizeStr *string `json:"contentLength"`
|
||||
Updated *string `json:"updated"`
|
||||
ViewURL *string `json:"viewUrl"`
|
||||
}
|
||||
if err := json.Unmarshal(b, &raw); err != nil {
|
||||
return err
|
||||
}
|
||||
if raw.ID != nil {
|
||||
f.ID = raw.ID.String()
|
||||
}
|
||||
if raw.Title != nil {
|
||||
f.Title = *raw.Title
|
||||
}
|
||||
if raw.PureSize != nil {
|
||||
f.Size = *raw.PureSize
|
||||
} else if raw.SizeStr != nil {
|
||||
if n, err := strconv.ParseInt(strings.Fields(*raw.SizeStr)[0], 10, 64); err == nil {
|
||||
f.Size = n
|
||||
}
|
||||
}
|
||||
if raw.Updated != nil {
|
||||
f.Updated = *raw.Updated
|
||||
}
|
||||
if raw.ViewURL != nil {
|
||||
f.ViewURL = *raw.ViewURL
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ModTime parses the folder's updated timestamp.
|
||||
func (f *DavFolder) ModTime() time.Time {
|
||||
t, _ := time.Parse("2006-01-02T15:04:05.0000000-07:00", f.Updated)
|
||||
return t
|
||||
}
|
||||
|
||||
// ModTime parses the file's updated timestamp.
|
||||
func (f *DavFile) ModTime() time.Time {
|
||||
t, _ := time.Parse("2006-01-02T15:04:05.0000000-07:00", f.Updated)
|
||||
return t
|
||||
}
|
||||
@@ -4,18 +4,21 @@ go 1.25.0
|
||||
|
||||
require (
|
||||
github.com/JohannesKaufmann/html-to-markdown/v2 v2.5.2
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.1
|
||||
github.com/charmbracelet/bubbles v0.18.0
|
||||
github.com/charmbracelet/bubbletea v0.25.0
|
||||
github.com/charmbracelet/glamour v0.8.0
|
||||
github.com/charmbracelet/lipgloss v0.12.1
|
||||
github.com/charmbracelet/x/ansi v0.1.4
|
||||
github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3
|
||||
github.com/eslider/go-hocr v0.2.2-0.20260827163626-8ff01582b002
|
||||
github.com/eslider/go-xls/v2 v2.1.0
|
||||
github.com/google/go-querystring v1.2.0
|
||||
github.com/joho/godotenv v1.5.1
|
||||
github.com/mattn/go-runewidth v0.0.15
|
||||
github.com/muesli/termenv v0.16.0
|
||||
github.com/spf13/cobra v1.10.2
|
||||
github.com/xuri/excelize/v2 v2.11.0
|
||||
gopkg.in/yaml.v3 v3.0.1
|
||||
modernc.org/sqlite v1.56.0
|
||||
)
|
||||
@@ -24,6 +27,7 @@ require (
|
||||
github.com/JohannesKaufmann/dom v0.3.1 // indirect
|
||||
github.com/alecthomas/chroma/v2 v2.14.0 // indirect
|
||||
github.com/atotto/clipboard v0.1.4 // indirect
|
||||
github.com/aws/smithy-go v1.24.0 // indirect
|
||||
github.com/aymanbagabas/go-osc52/v2 v2.0.1 // indirect
|
||||
github.com/aymerick/douceur v0.2.0 // indirect
|
||||
github.com/containerd/console v1.0.4-0.20230313162750-1ae8d489ac81 // indirect
|
||||
@@ -41,15 +45,21 @@ require (
|
||||
github.com/muesli/reflow v0.3.0 // indirect
|
||||
github.com/ncruces/go-strftime v1.0.0 // indirect
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
github.com/richardlehane/mscfb v1.0.7 // indirect
|
||||
github.com/richardlehane/msoleps v1.0.6 // indirect
|
||||
github.com/rivo/uniseg v0.4.7 // indirect
|
||||
github.com/spf13/pflag v1.0.9 // indirect
|
||||
github.com/tiendc/go-deepcopy v1.7.2 // indirect
|
||||
github.com/xuri/efp v0.0.1 // indirect
|
||||
github.com/xuri/nfp v0.0.2-0.20250530014748-2ddeb826f9a9 // indirect
|
||||
github.com/yuin/goldmark v1.8.2 // indirect
|
||||
github.com/yuin/goldmark-emoji v1.0.3 // indirect
|
||||
golang.org/x/net v0.55.0 // indirect
|
||||
golang.org/x/crypto v0.53.0 // indirect
|
||||
golang.org/x/net v0.56.0 // indirect
|
||||
golang.org/x/sync v0.21.0 // indirect
|
||||
golang.org/x/sys v0.47.0 // indirect
|
||||
golang.org/x/term v0.43.0 // indirect
|
||||
golang.org/x/text v0.37.0 // indirect
|
||||
golang.org/x/term v0.44.0 // indirect
|
||||
golang.org/x/text v0.38.0 // indirect
|
||||
modernc.org/libc v1.74.4 // indirect
|
||||
modernc.org/mathutil v1.7.1 // indirect
|
||||
modernc.org/memory v1.11.0 // indirect
|
||||
|
||||
@@ -10,6 +10,10 @@ github.com/alecthomas/repr v0.4.0 h1:GhI2A8MACjfegCPVq9f1FLvIBS+DrQ2KQBFZP1iFzXc
|
||||
github.com/alecthomas/repr v0.4.0/go.mod h1:Fr0507jx4eOXV7AlPV6AVZLYrLIuIeSOWtW57eE/O/4=
|
||||
github.com/atotto/clipboard v0.1.4 h1:EH0zSVneZPSuFR11BlR9YppQTVDbh5+16AmcJi4g1z4=
|
||||
github.com/atotto/clipboard v0.1.4/go.mod h1:ZY9tmq7sm5xIbd9bOK4onWV4S6X0u6GY7Vn0Yu86PYI=
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.1 h1:ABlyEARCDLN034NhxlRUSZr4l71mh+T5KAeGh6cerhU=
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.1/go.mod h1:MayyLB8y+buD9hZqkCW3kX1AKq07Y5pXxtgB+rRFhz0=
|
||||
github.com/aws/smithy-go v1.24.0 h1:LpilSUItNPFr1eY85RYgTIg5eIEPtvFbskaFcmmIUnk=
|
||||
github.com/aws/smithy-go v1.24.0/go.mod h1:LEj2LM3rBRQJxPZTB4KuzZkaZYnZPnvgIhb4pu07mx0=
|
||||
github.com/aymanbagabas/go-osc52/v2 v2.0.1 h1:HwpRHbFMcZLEVr42D4p7XBqjyuxQH5SMiErDT4WkJ2k=
|
||||
github.com/aymanbagabas/go-osc52/v2 v2.0.1/go.mod h1:uYgXzlJ7ZpABp8OJ+exZzJJhRNQ2ASbcXHWsFqH8hp8=
|
||||
github.com/aymanbagabas/go-udiff v0.2.0 h1:TK0fH4MteXUDspT88n8CKzvK0X9O2xu9yQjWpi6yML8=
|
||||
@@ -31,12 +35,16 @@ github.com/charmbracelet/x/exp/golden v0.0.0-20240715153702-9ba8adf781c4/go.mod
|
||||
github.com/containerd/console v1.0.4-0.20230313162750-1ae8d489ac81 h1:q2hJAaP1k2wIvVRd/hEHD7lacgqrCPS+k8g1MndzfWY=
|
||||
github.com/containerd/console v1.0.4-0.20230313162750-1ae8d489ac81/go.mod h1:YynlIjWYF8myEu6sdkwKIvGQq+cOckRm6So2avqoYAk=
|
||||
github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g=
|
||||
github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
|
||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/dlclark/regexp2 v1.11.0 h1:G/nrcoOa7ZXlpoa/91N3X7mM3r8eIlMBBJZvsz/mxKI=
|
||||
github.com/dlclark/regexp2 v1.11.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8=
|
||||
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
|
||||
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
|
||||
github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3 h1:B9YK+Tck5mTccyDhtxBzWyqGYcFxLyB6+noMNW4/VgI=
|
||||
github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3/go.mod h1:HMJKR5wlh/ziNp+sHEDV2ltblO4JD2+IdDOWtGcQBTM=
|
||||
github.com/eslider/go-hocr v0.2.2-0.20260827163626-8ff01582b002 h1:LOFxQG4mxvlH7+lffbO2SIn2ThlClJlrHHfs1OqMNXs=
|
||||
github.com/eslider/go-hocr v0.2.2-0.20260827163626-8ff01582b002/go.mod h1:fIgfH/E1j3rU8du4X4+7mxTD0GPtPQibTzytgitdJWU=
|
||||
github.com/eslider/go-xls/v2 v2.1.0 h1:HszWKqYQbXxACmAXXWdMsfNl1NDBfGVBnJUPtyUHQ7A=
|
||||
github.com/eslider/go-xls/v2 v2.1.0/go.mod h1:xgxO6JrfuBr9jGUB+0z5l/yDmFFZ5diGk0ATGihxlMU=
|
||||
github.com/google/go-cmp v0.6.0 h1:ofyhxvXcZhMsU5ulbFiLKl/XBFqE1GSq7atu8tAmTRI=
|
||||
@@ -82,6 +90,10 @@ github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZb
|
||||
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
|
||||
github.com/richardlehane/mscfb v1.0.7 h1:oeoiM0WE79vHwE8RpIYYvIAc8ajTH2mb6UZm55/+EB0=
|
||||
github.com/richardlehane/mscfb v1.0.7/go.mod h1:pe0+IUIc0AHh0+teNzBlJCtSyZdFOGgV4ZK9bsoV+Jo=
|
||||
github.com/richardlehane/msoleps v1.0.6 h1:9BvkpjvD+iUBalUY4esMwv6uBkfOip/Lzvd93jvR9gg=
|
||||
github.com/richardlehane/msoleps v1.0.6/go.mod h1:BWev5JBpU9Ko2WAgmZEuiz4/u3ZYTKbjLycmwiWUfWg=
|
||||
github.com/rivo/uniseg v0.1.0/go.mod h1:J6wj4VEh+S6ZtnVlnTBMWIodfgj8LQOQFoIToxlJtxc=
|
||||
github.com/rivo/uniseg v0.2.0/go.mod h1:J6wj4VEh+S6ZtnVlnTBMWIodfgj8LQOQFoIToxlJtxc=
|
||||
github.com/rivo/uniseg v0.4.7 h1:WUdvkW8uEhrYfLC4ZzdpI2ztxP1I582+49Oc5Mq64VQ=
|
||||
@@ -95,25 +107,39 @@ github.com/spf13/cobra v1.10.2 h1:DMTTonx5m65Ic0GOoRY2c16WCbHxOOw6xxezuLaBpcU=
|
||||
github.com/spf13/cobra v1.10.2/go.mod h1:7C1pvHqHw5A4vrJfjNwvOdzYu0Gml16OCs2GRiTUUS4=
|
||||
github.com/spf13/pflag v1.0.9 h1:9exaQaMOCwffKiiiYk6/BndUBv+iRViNW+4lEMi0PvY=
|
||||
github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
|
||||
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
|
||||
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
|
||||
github.com/tiendc/go-deepcopy v1.7.2 h1:Ut2yYR7W9tWjTQitganoIue4UGxZwCcJy3orjrrIj44=
|
||||
github.com/tiendc/go-deepcopy v1.7.2/go.mod h1:4bKjNC2r7boYOkD2IOuZpYjmlDdzjbpTRyCx+goBCJQ=
|
||||
github.com/xuri/efp v0.0.1 h1:fws5Rv3myXyYni8uwj2qKjVaRP30PdjeYe2Y6FDsCL8=
|
||||
github.com/xuri/efp v0.0.1/go.mod h1:ybY/Jr0T0GTCnYjKqmdwxyxn2BQf2RcQIIvex5QldPI=
|
||||
github.com/xuri/excelize/v2 v2.11.0 h1:HxaEFl6sRN2+8J5a8HaKq+0M4FsjBGMnWWtjOCPSG88=
|
||||
github.com/xuri/excelize/v2 v2.11.0/go.mod h1:jxFLbzaIwGQ5ufFNvYfUOHqXhfPaNmP14KWfmNz2Uak=
|
||||
github.com/xuri/nfp v0.0.2-0.20250530014748-2ddeb826f9a9 h1:+C0TIdyyYmzadGaL/HBLbf3WdLgC29pgyhTjAT/0nuE=
|
||||
github.com/xuri/nfp v0.0.2-0.20250530014748-2ddeb826f9a9/go.mod h1:WwHg+CVyzlv/TX9xqBFXEZAuxOPxn2k1GNHwG41IIUQ=
|
||||
github.com/yuin/goldmark v1.7.1/go.mod h1:uzxRWxtg69N339t3louHJ7+O03ezfj6PlliRlaOzY1E=
|
||||
github.com/yuin/goldmark v1.8.2 h1:kEGpgqJXdgbkhcOgBxkC0X0PmoPG1ZyoZ117rDVp4zE=
|
||||
github.com/yuin/goldmark v1.8.2/go.mod h1:ip/1k0VRfGynBgxOz0yCqHrbZXhcjxyuS66Brc7iBKg=
|
||||
github.com/yuin/goldmark-emoji v1.0.3 h1:aLRkLHOuBR2czCY4R8olwMjID+tENfhyFDMCRhbIQY4=
|
||||
github.com/yuin/goldmark-emoji v1.0.3/go.mod h1:tTkZEbwu5wkPmgTcitqddVxY9osFZiavD+r4AzQrh1U=
|
||||
go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
|
||||
golang.org/x/crypto v0.53.0 h1:QZ4Muo8THX6CizN2vPPd5fBGHyogrdK9fG4wLPFUsto=
|
||||
golang.org/x/crypto v0.53.0/go.mod h1:DNLU434OwVakk9PzuwV8w62mAJpRJL3vsgcfp4Qnsio=
|
||||
golang.org/x/image v0.38.0 h1:5l+q+Y9JDC7mBOMjo4/aPhMDcxEptsX+Tt3GgRQRPuE=
|
||||
golang.org/x/image v0.38.0/go.mod h1:/3f6vaXC+6CEanU4KJxbcUZyEePbyKbaLoDOe4ehFYY=
|
||||
golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ=
|
||||
golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0=
|
||||
golang.org/x/net v0.55.0 h1:bcvxaJn3e1U6InsFWt1JUq1aSjnRxLzT2rtD2KfkDF8=
|
||||
golang.org/x/net v0.55.0/go.mod h1:L5U2KuzuOe1lY7Z+aWVIKK6qEeJXnXV9yzGA+WCHJww=
|
||||
golang.org/x/net v0.56.0 h1:Rw8j/hFzGvJUZwNBXnAtf5sVDVt+65SK2C7IxCxZt5o=
|
||||
golang.org/x/net v0.56.0/go.mod h1:D3Ku6r+V6JROoZK144D2XfMHFcMq/0zSfLelVTCFKec=
|
||||
golang.org/x/sync v0.21.0 h1:HLII4xRRTtCRkxYp4HNFF0Js/Og6q2i++KXbg0gHCwM=
|
||||
golang.org/x/sync v0.21.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sys v0.1.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/term v0.43.0 h1:S4RLU2sB31O/NCl+zFN9Aru9A/Cq2aqKpTZJ6B+DwT4=
|
||||
golang.org/x/term v0.43.0/go.mod h1:lrhlHNdQJHO+1qVYiHfFKVuVioJIheAc3fBSMFYEIsk=
|
||||
golang.org/x/text v0.37.0 h1:Cqjiwd9eSg8e0QAkyCaQTNHFIIzWtidPahFWR83rTrc=
|
||||
golang.org/x/text v0.37.0/go.mod h1:a5sjxXGs9hsn/AJVwuElvCAo9v8QYLzvavO5z2PiM38=
|
||||
golang.org/x/term v0.44.0 h1:0rLvDRCtNj0gZkyIXhCyOb2OAzEhLVqc4B+hrsBhrmc=
|
||||
golang.org/x/term v0.44.0/go.mod h1:7ze4MdzUzLXpSAoFP1H0bOI9aXDqveSvatT5vKcFh2Y=
|
||||
golang.org/x/text v0.38.0 h1:sXmwo9DwP3OK9EZ7PqAdaooSGozfl/3a6/xJcbzPRhE=
|
||||
golang.org/x/text v0.38.0/go.mod h1:YXZt3QhHUKYT53r2lLKFIVi6Ao1jdzrTR/KQ09qyxF4=
|
||||
golang.org/x/tools v0.47.0 h1:7Kn5x/d1svx/PzryTsqeoZN4TZwqeH5pGWjefhLi/1Q=
|
||||
golang.org/x/tools v0.47.0/go.mod h1:dFHnyTvFWY212G+h7ZY4Vsp/K3U4/7W9TyVaAul8uCA=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
|
||||
|
||||
@@ -346,6 +346,14 @@ func (c *Client) putJSON(ctx context.Context, path string, body any) (json.RawMe
|
||||
|
||||
// uploadMultipart posts a single file to path under the given form field name.
|
||||
func (c *Client) uploadMultipart(ctx context.Context, path, fieldName, filePath string) (json.RawMessage, error) {
|
||||
return c.uploadMultipartMethod(ctx, http.MethodPost, path, fieldName, filePath)
|
||||
}
|
||||
|
||||
// uploadMultipartMethod sends a single-file multipart request with the given
|
||||
// HTTP method. The OnlyOffice Documents API needs PUT for /update (a new
|
||||
// version) and POST for /upload (a new file); sending POST to /update answers
|
||||
// 500 on current servers.
|
||||
func (c *Client) uploadMultipartMethod(ctx context.Context, method, path, fieldName, filePath string) (json.RawMessage, error) {
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -368,7 +376,7 @@ func (c *Client) uploadMultipart(ctx context.Context, path, fieldName, filePath
|
||||
if err := mw.Close(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.baseURL()+path, &buf)
|
||||
req, err := http.NewRequestWithContext(ctx, method, c.baseURL()+path, &buf)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -0,0 +1,419 @@
|
||||
// Package docpipe converts documents for OnlyOffice agent workflows:
|
||||
// Markdown ↔ DOCX (pandoc) and image/PDF OCR → searchable PDF + Markdown text.
|
||||
//
|
||||
// External tools (optional at runtime; helpers skip/error clearly when missing):
|
||||
// - pandoc — md↔docx
|
||||
// - ocrmypdf — OCR into a searchable PDF
|
||||
// - pdftotext — extract text layer
|
||||
// - tesseract — OCR single images when ocrmypdf is unsuitable
|
||||
// - ghostscript (gs) — PDF rewrite/optimize via PostScript (pdfwrite)
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// DefaultMinTextChars: below this, a PDF is treated as needing OCR.
|
||||
const DefaultMinTextChars = 200
|
||||
|
||||
// Tools reports which converters are available on PATH.
|
||||
type Tools struct {
|
||||
Pandoc string
|
||||
OCRMyPDF string
|
||||
PDFToText string
|
||||
Tesseract string
|
||||
Ghostscript string
|
||||
}
|
||||
|
||||
// LookPath resolves converter binaries (empty string if missing).
|
||||
func LookPath() Tools {
|
||||
find := func(names ...string) string {
|
||||
for _, n := range names {
|
||||
if p, err := exec.LookPath(n); err == nil {
|
||||
return p
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
return Tools{
|
||||
Pandoc: find("pandoc"),
|
||||
OCRMyPDF: find("ocrmypdf"),
|
||||
PDFToText: find("pdftotext"),
|
||||
Tesseract: find("tesseract"),
|
||||
Ghostscript: find("gs", "ghostscript"),
|
||||
}
|
||||
}
|
||||
|
||||
func (t Tools) requirePandoc() error {
|
||||
if t.Pandoc == "" {
|
||||
return fmt.Errorf("pandoc not found on PATH (needed for md↔docx)")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Ext returns lower-case extension including dot (".pdf").
|
||||
func Ext(path string) string {
|
||||
return strings.ToLower(filepath.Ext(path))
|
||||
}
|
||||
|
||||
// ConvertFile converts between md and docx (and other pandoc formats) via pandoc.
|
||||
// outExt may be ".md", ".docx", or a full output path.
|
||||
func (t Tools) ConvertFile(inPath, outPath string) error {
|
||||
if err := t.requirePandoc(); err != nil {
|
||||
return err
|
||||
}
|
||||
if strings.TrimSpace(outPath) == "" {
|
||||
return fmt.Errorf("output path required")
|
||||
}
|
||||
cmd := exec.Command(t.Pandoc, inPath, "-o", outPath)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return fmt.Errorf("pandoc %s → %s: %w (%s)", inPath, outPath, err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// TXTToDOCX converts plain text to DOCX preserving line breaks (via markdown hard breaks).
|
||||
func (t Tools) TXTToDOCX(txtPath, docxPath string) error {
|
||||
if Ext(txtPath) != ".txt" {
|
||||
return fmt.Errorf("expected .txt input, got %q", txtPath)
|
||||
}
|
||||
if docxPath == "" {
|
||||
docxPath = strings.TrimSuffix(txtPath, Ext(txtPath)) + ".docx"
|
||||
}
|
||||
b, err := os.ReadFile(txtPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dir := filepath.Dir(docxPath)
|
||||
if dir == "" || dir == "." {
|
||||
dir = os.TempDir()
|
||||
}
|
||||
tmpMD := filepath.Join(dir, trimExt(filepath.Base(txtPath))+".txt2docx.md")
|
||||
md := TxtToMarkdown(string(b))
|
||||
if err := os.WriteFile(tmpMD, []byte(md), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.Remove(tmpMD)
|
||||
return t.MDToDOCX(tmpMD, docxPath)
|
||||
}
|
||||
|
||||
// TxtToMarkdown converts plain text to Markdown for DOCX output.
|
||||
// Prose text: each line is a hard break. Fixed-width extracts (INE, pdftotext -layout):
|
||||
// wrapped in a fenced code block (monospace, columns preserved).
|
||||
func TxtToMarkdown(content string) string {
|
||||
content = normalizeTxtNewlines(content)
|
||||
if isFixedWidthTxt(content) {
|
||||
return "```\n" + content + "\n```\n"
|
||||
}
|
||||
var b strings.Builder
|
||||
for _, line := range strings.Split(content, "\n") {
|
||||
if strings.TrimSpace(line) == "" {
|
||||
b.WriteByte('\n')
|
||||
continue
|
||||
}
|
||||
b.WriteString(line)
|
||||
b.WriteString(" \n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func normalizeTxtNewlines(content string) string {
|
||||
content = strings.ReplaceAll(content, "\r\n", "\n")
|
||||
return strings.ReplaceAll(content, "\r", "\n")
|
||||
}
|
||||
|
||||
// isFixedWidthTxt detects pdftotext -layout style extracts (many indented/spaced columns).
|
||||
func isFixedWidthTxt(content string) bool {
|
||||
lines := strings.Split(content, "\n")
|
||||
if len(lines) < 8 {
|
||||
return false
|
||||
}
|
||||
indented, long := 0, 0
|
||||
for _, line := range lines {
|
||||
if strings.TrimSpace(line) == "" {
|
||||
continue
|
||||
}
|
||||
if len(line) >= 72 {
|
||||
long++
|
||||
}
|
||||
if len(line) > 0 && (line[0] == ' ' || line[0] == '\t') {
|
||||
indented++
|
||||
}
|
||||
}
|
||||
n := len(lines)
|
||||
return indented*100/n >= 20 || (long >= 5 && indented*100/n >= 10)
|
||||
}
|
||||
|
||||
// MDToDOCX writes a DOCX next to or at outPath from a Markdown file.
|
||||
func (t Tools) MDToDOCX(mdPath, docxPath string) error {
|
||||
if Ext(mdPath) != ".md" && Ext(mdPath) != ".markdown" {
|
||||
return fmt.Errorf("expected markdown input, got %q", mdPath)
|
||||
}
|
||||
if docxPath == "" {
|
||||
docxPath = strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx"
|
||||
}
|
||||
return t.ConvertFile(mdPath, docxPath)
|
||||
}
|
||||
|
||||
// DOCXToMD writes Markdown from a DOCX file.
|
||||
func (t Tools) DOCXToMD(docxPath, mdPath string) error {
|
||||
if Ext(docxPath) != ".docx" {
|
||||
return fmt.Errorf("expected .docx input, got %q", docxPath)
|
||||
}
|
||||
if mdPath == "" {
|
||||
mdPath = strings.TrimSuffix(docxPath, Ext(docxPath)) + ".md"
|
||||
}
|
||||
return t.ConvertFile(docxPath, mdPath)
|
||||
}
|
||||
|
||||
// OptimizePDF rewrites a PDF through Ghostscript (PostScript pdfwrite).
|
||||
// Preserves native text layers; strips broken OCR overlays; shrinks for OO preview.
|
||||
// Use instead of ocrmypdf when pdftotext already extracts enough text.
|
||||
func (t Tools) OptimizePDF(inPath, outPath string) error {
|
||||
if t.Ghostscript == "" {
|
||||
return fmt.Errorf("ghostscript (gs) not found on PATH")
|
||||
}
|
||||
if outPath == "" {
|
||||
return fmt.Errorf("output PDF path required")
|
||||
}
|
||||
args := []string{
|
||||
"-sDEVICE=pdfwrite",
|
||||
"-dCompatibilityLevel=1.5",
|
||||
"-dNOPAUSE", "-dQUIET", "-dBATCH",
|
||||
"-dPDFSETTINGS=/ebook",
|
||||
"-dEmbedAllFonts=true",
|
||||
"-dSubsetFonts=true",
|
||||
"-dCompressFonts=true",
|
||||
"-dCompressPages=true",
|
||||
"-dDetectDuplicateImages=true",
|
||||
"-dAutoRotatePages=/None",
|
||||
"-sOutputFile=" + outPath,
|
||||
inPath,
|
||||
}
|
||||
cmd := exec.Command(t.Ghostscript, args...)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return fmt.Errorf("ghostscript pdfwrite: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// PDFTextLayerChars returns approximate extracted character count (0 if unavailable).
|
||||
func (t Tools) PDFTextLayerChars(pdfPath string) (int, error) {
|
||||
if t.PDFToText == "" {
|
||||
return 0, fmt.Errorf("pdftotext not found on PATH")
|
||||
}
|
||||
cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-")
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(bytes.TrimSpace(out)), nil
|
||||
}
|
||||
|
||||
// NeedsOCR reports whether path likely needs OCR before text extraction.
|
||||
func (t Tools) NeedsOCR(path string, minChars int) (bool, error) {
|
||||
if minChars <= 0 {
|
||||
minChars = DefaultMinTextChars
|
||||
}
|
||||
switch Ext(path) {
|
||||
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
|
||||
return true, nil
|
||||
case ".pdf":
|
||||
n, err := t.PDFTextLayerChars(path)
|
||||
if err != nil {
|
||||
// If we cannot measure, prefer OCR.
|
||||
return true, nil
|
||||
}
|
||||
return n < minChars, nil
|
||||
default:
|
||||
return false, nil
|
||||
}
|
||||
}
|
||||
|
||||
// OCRToPDF runs ocrmypdf into outPDF (searchable). Forces OCR when force is true.
|
||||
func (t Tools) OCRToPDF(inPath, outPDF string, force bool, lang string) error {
|
||||
if t.OCRMyPDF == "" {
|
||||
return fmt.Errorf("ocrmypdf not found on PATH")
|
||||
}
|
||||
if outPDF == "" {
|
||||
return fmt.Errorf("output PDF path required")
|
||||
}
|
||||
if lang == "" {
|
||||
lang = "eng"
|
||||
}
|
||||
args := []string{"-l", lang, "--skip-big", "100"}
|
||||
if force {
|
||||
args = append(args, "--force-ocr")
|
||||
} else {
|
||||
args = append(args, "--skip-text")
|
||||
}
|
||||
args = append(args, inPath, outPDF)
|
||||
cmd := exec.Command(t.OCRMyPDF, args...)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
// Retry with force if skip-text refused.
|
||||
if !force && strings.Contains(stderr.String(), "PriorOcrFoundError") == false {
|
||||
args2 := []string{"-l", lang, "--force-ocr", inPath, outPDF}
|
||||
cmd2 := exec.Command(t.OCRMyPDF, args2...)
|
||||
var stderr2 bytes.Buffer
|
||||
cmd2.Stderr = &stderr2
|
||||
if err2 := cmd2.Run(); err2 == nil {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
return fmt.Errorf("ocrmypdf: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ImageToText OCRs a raster image with tesseract (stdout text).
|
||||
func (t Tools) ImageToText(imgPath, lang string) (string, error) {
|
||||
if t.Tesseract == "" {
|
||||
return "", fmt.Errorf("tesseract not found on PATH")
|
||||
}
|
||||
if lang == "" {
|
||||
lang = "eng"
|
||||
}
|
||||
cmd := exec.Command(t.Tesseract, imgPath, "stdout", "-l", lang)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("tesseract: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return string(out), nil
|
||||
}
|
||||
|
||||
// ExtractPDFText returns layout text from a PDF via pdftotext.
|
||||
func (t Tools) ExtractPDFText(pdfPath string) (string, error) {
|
||||
if t.PDFToText == "" {
|
||||
return "", fmt.Errorf("pdftotext not found on PATH")
|
||||
}
|
||||
cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-")
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return string(out), nil
|
||||
}
|
||||
|
||||
// Result of ToMarkdown.
|
||||
type Result struct {
|
||||
Markdown string
|
||||
OCRPDFPath string // set when a searchable PDF was produced
|
||||
DidOCR bool
|
||||
Source string
|
||||
}
|
||||
|
||||
// ToMarkdown turns a local file into Markdown text.
|
||||
// PDFs/images with weak/no text layer are OCR'd to a searchable PDF first (when tools exist).
|
||||
func (t Tools) ToMarkdown(path string, workDir string, lang string, minChars int) (Result, error) {
|
||||
res := Result{Source: path}
|
||||
ext := Ext(path)
|
||||
switch ext {
|
||||
case ".md", ".markdown", ".txt":
|
||||
b, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = string(b)
|
||||
return res, nil
|
||||
case ".docx", ".odt", ".rtf", ".html", ".htm":
|
||||
if err := t.requirePandoc(); err != nil {
|
||||
return res, err
|
||||
}
|
||||
tmp := filepath.Join(workDir, "out.md")
|
||||
if err := t.ConvertFile(path, tmp); err != nil {
|
||||
return res, err
|
||||
}
|
||||
b, err := os.ReadFile(tmp)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = string(b)
|
||||
return res, nil
|
||||
case ".pdf":
|
||||
need, _ := t.NeedsOCR(path, minChars)
|
||||
pdf := path
|
||||
if need {
|
||||
if workDir == "" {
|
||||
workDir = os.TempDir()
|
||||
}
|
||||
outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf")
|
||||
if err := t.OCRToPDF(path, outPDF, true, lang); err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.DidOCR = true
|
||||
res.OCRPDFPath = outPDF
|
||||
pdf = outPDF
|
||||
}
|
||||
text, err := t.ExtractPDFText(pdf)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = wrapMD(filepath.Base(path), text)
|
||||
return res, nil
|
||||
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
|
||||
if workDir == "" {
|
||||
workDir = os.TempDir()
|
||||
}
|
||||
outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf")
|
||||
if t.OCRMyPDF != "" {
|
||||
if err := t.OCRToPDF(path, outPDF, true, lang); err == nil {
|
||||
res.DidOCR = true
|
||||
res.OCRPDFPath = outPDF
|
||||
text, err := t.ExtractPDFText(outPDF)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = wrapMD(filepath.Base(path), text)
|
||||
return res, nil
|
||||
}
|
||||
}
|
||||
text, err := t.ImageToText(path, lang)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.DidOCR = true
|
||||
res.Markdown = wrapMD(filepath.Base(path), text)
|
||||
return res, nil
|
||||
default:
|
||||
return res, fmt.Errorf("unsupported type %q for markdown extraction", ext)
|
||||
}
|
||||
}
|
||||
|
||||
func wrapMD(title, body string) string {
|
||||
body = strings.TrimSpace(body)
|
||||
if body == "" {
|
||||
return "# " + title + "\n\n_(empty text layer)_\n"
|
||||
}
|
||||
return "# " + title + "\n\n" + body + "\n"
|
||||
}
|
||||
|
||||
func trimExt(name string) string {
|
||||
return strings.TrimSuffix(name, filepath.Ext(name))
|
||||
}
|
||||
|
||||
// SiblingDOCX returns path with .docx extension replacing the original ext.
|
||||
func SiblingDOCX(mdPath string) string {
|
||||
return strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx"
|
||||
}
|
||||
|
||||
// EnsureDir creates parent directories for path.
|
||||
func EnsureDir(path string) error {
|
||||
dir := filepath.Dir(path)
|
||||
if dir == "" || dir == "." {
|
||||
return nil
|
||||
}
|
||||
return os.MkdirAll(dir, 0o755)
|
||||
}
|
||||
@@ -0,0 +1,167 @@
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestExt(t *testing.T) {
|
||||
if Ext("Foo.PDF") != ".pdf" {
|
||||
t.Fatalf("Ext: %q", Ext("Foo.PDF"))
|
||||
}
|
||||
}
|
||||
|
||||
func TestSiblingDOCX(t *testing.T) {
|
||||
if got := SiblingDOCX("notes.md"); got != "notes.docx" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWrapMD(t *testing.T) {
|
||||
s := wrapMD("a.pdf", " hello ")
|
||||
if !strings.HasPrefix(s, "# a.pdf\n") || !strings.Contains(s, "hello") {
|
||||
t.Fatalf("wrap: %q", s)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNeedsOCR_Image(t *testing.T) {
|
||||
tools := LookPath()
|
||||
need, err := tools.NeedsOCR("x.jpg", 0)
|
||||
if err != nil || !need {
|
||||
t.Fatalf("jpg should need OCR: need=%v err=%v", need, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTxtToMarkdown_FixedWidthUsesCodeBlock(t *testing.T) {
|
||||
var lines []string
|
||||
for i := 0; i < 12; i++ {
|
||||
lines = append(lines, " column layout line "+strings.Repeat("x", 40))
|
||||
}
|
||||
in := strings.Join(lines, "\n")
|
||||
md := TxtToMarkdown(in)
|
||||
if !strings.HasPrefix(md, "```\n") || !strings.Contains(md, "```") {
|
||||
t.Fatalf("expected code block: %q", md[:min(80, len(md))])
|
||||
}
|
||||
}
|
||||
|
||||
func min(a, b int) int {
|
||||
if a < b {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
func TestTxtToMarkdownPreservesLines(t *testing.T) {
|
||||
in := "line1\nline2\n\nline4"
|
||||
md := TxtToMarkdown(in)
|
||||
if !strings.Contains(md, "line1 \n") || !strings.Contains(md, "line2 \n") {
|
||||
t.Fatalf("hard breaks missing: %q", md)
|
||||
}
|
||||
if !strings.Contains(md, "line4 \n") {
|
||||
t.Fatalf("last line: %q", md)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTXTToDOCXPreservesLines(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.Pandoc == "" {
|
||||
t.Skip("pandoc not installed")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
txt := filepath.Join(dir, "sample.txt")
|
||||
docx := filepath.Join(dir, "sample.docx")
|
||||
body := "MyBox Auto — resumen\nNº contrato: 123\n\nEstado: Vigente\n"
|
||||
if err := os.WriteFile(txt, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := tools.TXTToDOCX(txt, docx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mdOut := filepath.Join(dir, "out.md")
|
||||
if err := tools.DOCXToMD(docx, mdOut); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := os.ReadFile(mdOut)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s := string(got)
|
||||
for _, want := range []string{"MyBox Auto", "Nº contrato", "Estado: Vigente"} {
|
||||
if !strings.Contains(s, want) {
|
||||
t.Fatalf("missing %q in %q", want, s)
|
||||
}
|
||||
}
|
||||
if strings.Contains(s, "MyBox Auto — resumen Nº") {
|
||||
t.Fatalf("lines collapsed: %q", s)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMDDocxRoundTrip(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.Pandoc == "" {
|
||||
t.Skip("pandoc not installed")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
md := filepath.Join(dir, "n.md")
|
||||
docx := filepath.Join(dir, "n.docx")
|
||||
md2 := filepath.Join(dir, "n2.md")
|
||||
if err := os.WriteFile(md, []byte("# Title\n\nHello **world**.\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := tools.MDToDOCX(md, docx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := os.Stat(docx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := tools.DOCXToMD(docx, md2); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
b, err := os.ReadFile(md2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(string(b), "Hello") {
|
||||
t.Fatalf("round-trip missing Hello: %s", b)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOptimizePDF(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.Ghostscript == "" || tools.PDFToText == "" {
|
||||
t.Skip("ghostscript/pdftotext not installed")
|
||||
}
|
||||
in := "/tmp/ccgg-original.pdf"
|
||||
if _, err := os.Stat(in); err != nil {
|
||||
t.Skip("local fixture not present")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
out := filepath.Join(dir, "out.pdf")
|
||||
charsIn, err := tools.PDFTextLayerChars(in)
|
||||
if err != nil || charsIn < 1000 {
|
||||
t.Skip("fixture has no text layer")
|
||||
}
|
||||
if err := tools.OptimizePDF(in, out); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
charsOut, err := tools.PDFTextLayerChars(out)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if charsOut < charsIn/2 {
|
||||
t.Fatalf("text layer lost: in=%d out=%d", charsIn, charsOut)
|
||||
}
|
||||
}
|
||||
|
||||
func TestToMarkdown_PlainMD(t *testing.T) {
|
||||
tools := LookPath()
|
||||
dir := t.TempDir()
|
||||
p := filepath.Join(dir, "a.md")
|
||||
_ = os.WriteFile(p, []byte("hi"), 0o644)
|
||||
res, err := tools.ToMarkdown(p, dir, "eng", 0)
|
||||
if err != nil || res.Markdown != "hi" {
|
||||
t.Fatalf("got %+v err=%v", res, err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,153 @@
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
hocr "github.com/eslider/go-hocr"
|
||||
)
|
||||
|
||||
// HOCRResult is structured OCR output for agents.
|
||||
type HOCRResult struct {
|
||||
HOCRPath string
|
||||
Markdown string
|
||||
YAML string
|
||||
DidOCR bool
|
||||
Source string
|
||||
}
|
||||
|
||||
// ImageToHOCR runs tesseract hOCR into outBase+".hocr" (tesseract adds the extension).
|
||||
// outBase must not include ".hocr". dpi 0 uses tesseract default; phone photos often need 200–300.
|
||||
func (t Tools) ImageToHOCR(imgPath, outBase, lang string, dpi int) (string, error) {
|
||||
if t.Tesseract == "" {
|
||||
return "", fmt.Errorf("tesseract not found on PATH")
|
||||
}
|
||||
if lang == "" {
|
||||
lang = "eng"
|
||||
}
|
||||
if outBase == "" {
|
||||
return "", fmt.Errorf("hOCR output base path required")
|
||||
}
|
||||
args := []string{imgPath, outBase, "-l", lang}
|
||||
if dpi > 0 {
|
||||
args = append(args, "--dpi", fmt.Sprintf("%d", dpi))
|
||||
}
|
||||
args = append(args, resolveHOCRConfig())
|
||||
cmd := exec.Command(t.Tesseract, args...)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return "", fmt.Errorf("tesseract hocr: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
out := outBase + ".hocr"
|
||||
if _, err := os.Stat(out); err != nil {
|
||||
// Some builds write .html
|
||||
alt := outBase + ".html"
|
||||
if _, err2 := os.Stat(alt); err2 == nil {
|
||||
return alt, nil
|
||||
}
|
||||
return "", fmt.Errorf("tesseract hocr: missing output %s (%s)", out, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// resolveHOCRConfig returns a tesseract config path that works when TESSDATA_PREFIX
|
||||
// points at a custom traineddata dir without relative "hocr" configs.
|
||||
func resolveHOCRConfig() string {
|
||||
var candidates []string
|
||||
if p := strings.TrimSpace(os.Getenv("TESSDATA_PREFIX")); p != "" {
|
||||
candidates = append(candidates,
|
||||
filepath.Join(p, "configs", "hocr"),
|
||||
filepath.Join(p, "tessdata", "configs", "hocr"),
|
||||
)
|
||||
}
|
||||
candidates = append(candidates,
|
||||
"/usr/share/tesseract-ocr/5/tessdata/configs/hocr",
|
||||
"/usr/share/tesseract-ocr/4.00/tessdata/configs/hocr",
|
||||
"/usr/share/tessdata/configs/hocr",
|
||||
)
|
||||
for _, c := range candidates {
|
||||
if _, err := os.Stat(c); err == nil {
|
||||
return c
|
||||
}
|
||||
}
|
||||
return "hocr"
|
||||
}
|
||||
|
||||
// HOCRToMarkdown parses an hOCR file via go-hocr and returns Markdown (+ optional YAML).
|
||||
// Words with confidence in (0, minConf) are dropped; minConf 0 keeps all.
|
||||
func HOCRToMarkdown(hocrPath string, minConf float32) (md string, yml string, err error) {
|
||||
doc, err := hocr.ReadFile(hocrPath)
|
||||
if err != nil {
|
||||
return "", "", fmt.Errorf("go-hocr: %w", err)
|
||||
}
|
||||
md = doc.ToMarkdown(minConf)
|
||||
yml, err = doc.ToYaml()
|
||||
if err != nil {
|
||||
return md, "", err
|
||||
}
|
||||
return md, yml, nil
|
||||
}
|
||||
|
||||
// ToHOCRMarkdown OCRs an image (or rasterizes first PDF page) to hOCR → Markdown/YAML.
|
||||
func (t Tools) ToHOCRMarkdown(path, workDir, lang string, dpi int, minConf float32) (HOCRResult, error) {
|
||||
res := HOCRResult{Source: path}
|
||||
if workDir == "" {
|
||||
workDir = os.TempDir()
|
||||
}
|
||||
ext := Ext(path)
|
||||
img := path
|
||||
switch ext {
|
||||
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
|
||||
// ok
|
||||
case ".pdf":
|
||||
raster, err := t.pdfFirstPagePNG(path, workDir)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
img = raster
|
||||
if dpi == 0 {
|
||||
dpi = 300
|
||||
}
|
||||
default:
|
||||
return res, fmt.Errorf("hOCR path expects image or PDF, got %q", ext)
|
||||
}
|
||||
base := filepath.Join(workDir, trimExt(filepath.Base(path))+".hocr-out")
|
||||
hocrPath, err := t.ImageToHOCR(img, base, lang, dpi)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.DidOCR = true
|
||||
res.HOCRPath = hocrPath
|
||||
md, yml, err := HOCRToMarkdown(hocrPath, minConf)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = wrapMD(filepath.Base(path), strings.TrimSpace(md))
|
||||
res.YAML = yml
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// pdfFirstPagePNG uses pdftoppm when available.
|
||||
func (t Tools) pdfFirstPagePNG(pdfPath, workDir string) (string, error) {
|
||||
pdftoppm, err := exec.LookPath("pdftoppm")
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("pdftoppm not found (needed to rasterize PDF for hOCR)")
|
||||
}
|
||||
outBase := filepath.Join(workDir, trimExt(filepath.Base(pdfPath))+".page")
|
||||
cmd := exec.Command(pdftoppm, "-png", "-f", "1", "-singlefile", "-r", "200", pdfPath, outBase)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return "", fmt.Errorf("pdftoppm: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
png := outBase + ".png"
|
||||
if _, err := os.Stat(png); err != nil {
|
||||
return "", fmt.Errorf("pdftoppm: missing %s", png)
|
||||
}
|
||||
return png, nil
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestHOCRToMarkdown_Fixture(t *testing.T) {
|
||||
// Minimal hOCR 1.2 snippet
|
||||
hocrXML := `<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||
<head>
|
||||
<title>tesseract</title>
|
||||
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
||||
<meta name="ocr-system" content="tesseract 5"/>
|
||||
<meta name="ocr-capabilities" content="ocr_page ocr_carea ocr_par ocr_line ocrx_word"/>
|
||||
</head>
|
||||
<body>
|
||||
<div class="ocr_page" id="page_1" title="image "x.png"; bbox 0 0 100 50; ppageno 0">
|
||||
<div class="ocr_carea" id="block_1_1" title="bbox 0 0 100 50">
|
||||
<p class="ocr_par" id="par_1_1" lang="eng" title="bbox 0 0 100 50">
|
||||
<span class="ocr_line" id="line_1_1" title="bbox 0 0 100 20; baseline 0 0; x_size 20">
|
||||
<span class="ocrx_word" id="word_1_1" title="bbox 0 0 40 20; x_wconf 96">Hello</span>
|
||||
<span class="ocrx_word" id="word_1_2" title="bbox 45 0 100 20; x_wconf 92">world</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>`
|
||||
dir := t.TempDir()
|
||||
p := filepath.Join(dir, "sample.hocr")
|
||||
if err := os.WriteFile(p, []byte(hocrXML), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
md, yml, err := HOCRToMarkdown(p, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(md, "Hello") || !strings.Contains(md, "world") {
|
||||
t.Fatalf("md=%q", md)
|
||||
}
|
||||
if !strings.Contains(yml, "Hello") {
|
||||
t.Fatalf("yaml missing word: %s", yml)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,393 @@
|
||||
package xlspipe
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
// Sheet names (Russian tabs) for cutover Portugal workbook.
|
||||
const (
|
||||
SheetInputs = "Ввод"
|
||||
SheetDom6 = "Вс 6.09"
|
||||
SheetJue3 = "Чт 3.09"
|
||||
SheetWedHyp = "Ср гип"
|
||||
SheetSummary = "Сводка"
|
||||
)
|
||||
|
||||
// CutoverPortugalDefaults holds live FO values (2026-08-28).
|
||||
type CutoverPortugalDefaults struct {
|
||||
FerryDom6 float64
|
||||
FerryJue3 float64
|
||||
FerrySuperiorDelta float64
|
||||
HousingLow float64
|
||||
HousingMid float64
|
||||
HousingHigh float64
|
||||
DriveLow float64
|
||||
DriveMid float64
|
||||
DriveHigh float64
|
||||
BoardLow float64
|
||||
BoardMid float64
|
||||
BoardHigh float64
|
||||
FoodLow float64
|
||||
FoodMid float64
|
||||
FoodHigh float64
|
||||
SimLow float64
|
||||
SimMid float64
|
||||
SimHigh float64
|
||||
MonthCap float64
|
||||
ExtraNightsDom6 float64
|
||||
ExtraNightsJue3 float64
|
||||
ExtraNightsWed float64
|
||||
}
|
||||
|
||||
// DefaultCutoverPortugal returns FO live snapshot from portugal track (28.08.2026).
|
||||
func DefaultCutoverPortugal() CutoverPortugalDefaults {
|
||||
return CutoverPortugalDefaults{
|
||||
FerryDom6: 457.89,
|
||||
FerryJue3: 484.09,
|
||||
FerrySuperiorDelta: 19.64,
|
||||
HousingLow: 509,
|
||||
HousingMid: 600,
|
||||
HousingHigh: 650,
|
||||
DriveLow: 70,
|
||||
DriveMid: 85,
|
||||
DriveHigh: 100,
|
||||
BoardLow: 40,
|
||||
BoardMid: 60,
|
||||
BoardHigh: 80,
|
||||
FoodLow: 150,
|
||||
FoodMid: 220,
|
||||
FoodHigh: 300,
|
||||
SimLow: 50,
|
||||
SimMid: 100,
|
||||
SimHigh: 150,
|
||||
MonthCap: 2500,
|
||||
ExtraNightsDom6: 0,
|
||||
ExtraNightsJue3: 0,
|
||||
ExtraNightsWed: 4,
|
||||
}
|
||||
}
|
||||
|
||||
type inputField struct {
|
||||
name string // defined name (ASCII, for formulas)
|
||||
label string
|
||||
value float64
|
||||
note string
|
||||
comment string
|
||||
}
|
||||
|
||||
// BuildCutoverPortugalWorkbook creates a multi-sheet cutover budget with Russian labels,
|
||||
// cell comments on non-obvious inputs, named ranges, and cross-sheet formulas.
|
||||
func BuildCutoverPortugalWorkbook(d CutoverPortugalDefaults) (*excelize.File, error) {
|
||||
f := excelize.NewFile()
|
||||
defaultSheet := f.GetSheetName(0)
|
||||
if err := f.SetSheetName(defaultSheet, SheetInputs); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
for _, name := range []string{SheetDom6, SheetJue3, SheetWedHyp, SheetSummary} {
|
||||
if _, err := f.NewSheet(name); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
|
||||
if err := writeInputsSheet(f, d); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
scenarios := []struct {
|
||||
sheet, ferry, extra, note string
|
||||
}{
|
||||
{SheetDom6, "ferry_dom6", "extra_nights_dom6", "Живой слот вс 6.09 20:30; заезд в квартиру вт 8.09"},
|
||||
{SheetJue3, "ferry_jue3", "extra_nights_jue3", "Живой чт 3.09; T1a 05–12 если TF-крыша кончается раньше вс"},
|
||||
{SheetWedHyp, "ferry_jue3", "extra_nights_wed", "Гипотеза: ср 2.09 20:00 по тарифу Jue3; заезд пт 4.09"},
|
||||
}
|
||||
for _, sc := range scenarios {
|
||||
if err := writeScenarioSheet(f, sc.sheet, sc.ferry, sc.extra, sc.note); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
if err := writeSummarySheet(f); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
f.SetActiveSheet(0)
|
||||
if err := finalizeWorkbook(f); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
|
||||
func writeInputsSheet(f *excelize.File, d CutoverPortugalDefaults) error {
|
||||
if err := setHeaders(f, SheetInputs, "Параметр", "Значение €", "Кратко"); err != nil {
|
||||
return err
|
||||
}
|
||||
fields := []inputField{
|
||||
{
|
||||
name: "ferry_dom6", label: "Паром вс 6.09 (Básica, без residencia)",
|
||||
value: d.FerryDom6, note: "FO live: 2взр+младенец+авто",
|
||||
comment: "Fred Olsen Dom 6.09 20:30 SC→Huelva, прибытие Mar 8 09:00. Butaca Normal/Básica без субсидии канарского residencia (−183€). Меняйте после нового live FO.",
|
||||
},
|
||||
{
|
||||
name: "ferry_jue3", label: "Паром чт 3.09 (Básica, без residencia)",
|
||||
value: d.FerryJue3, note: "FO live Jue 3 20:00",
|
||||
comment: "Прямой рейс 35 ч. Дороже вс на ~26€. Используется также для листа «Ср гип», если своего рейса ср 2.09 нет в продаже.",
|
||||
},
|
||||
{
|
||||
name: "ferry_superior_uplift", label: "Доплата VIP / Butaca Superior",
|
||||
value: d.FerrySuperiorDelta, note: "Superior − Normal (Jue3 live)",
|
||||
comment: "Разница между Superior и Normal на live Jue3 ≈19,64€. На Dom6 Superior в сессии не перевыбирали — оценка по этой дельте. VIP = salón, не каюта.",
|
||||
},
|
||||
{
|
||||
name: "housing_7n_low", label: "Крыша 7 ночей Setúbal — минимум",
|
||||
value: d.HousingLow, note: "Airbnb low band",
|
||||
comment: "Короткая аренда T1a (#11): 08–15.09, 1–2BR с парковкой. Низкая граница live Airbnb Setúbal.",
|
||||
},
|
||||
{
|
||||
name: "housing_7n_mid", label: "Крыша 7 ночей Setúbal — целевой mid",
|
||||
value: d.HousingMid, note: "Цель #11: 500–650€/нед",
|
||||
comment: "Рабочая оценка для брони. Основной столбец mid на листах сценариев.",
|
||||
},
|
||||
{
|
||||
name: "housing_7n_high", label: "Крыша 7 ночей Setúbal — максимум",
|
||||
value: d.HousingHigh, note: "Airbnb high band",
|
||||
},
|
||||
{
|
||||
name: "drive_low", label: "Проезд Huelva → Setúbal — мин",
|
||||
value: d.DriveLow, note: "OSRM ~3,8 ч",
|
||||
comment: "≈342 км: топливо + платные дороги. Низкая/средняя/высокая оценка.",
|
||||
},
|
||||
{name: "drive_mid", label: "Проезд Huelva → Setúbal — mid", value: d.DriveMid},
|
||||
{name: "drive_high", label: "Проезд Huelva → Setúbal — макс", value: d.DriveHigh},
|
||||
{
|
||||
name: "board_low", label: "Еда на пароме — мин",
|
||||
value: d.BoardLow, note: "Меню FO",
|
||||
comment: "Питание на борту (Fred Olsen). George 0–3 обычно бесплатно как пассажир — еда отдельно.",
|
||||
},
|
||||
{name: "board_mid", label: "Еда на пароме — mid", value: d.BoardMid},
|
||||
{name: "board_high", label: "Еда на пароме — макс", value: d.BoardHigh},
|
||||
{
|
||||
name: "food_7d_low", label: "Еда 7 дней в PT — мин",
|
||||
value: d.FoodLow, note: "Готовим в apt",
|
||||
comment: "Первая неделя в Setúbal после парома — продукты, не рестораны.",
|
||||
},
|
||||
{name: "food_7d_mid", label: "Еда 7 дней в PT — mid", value: d.FoodMid},
|
||||
{name: "food_7d_high", label: "Еда 7 дней в PT — макс", value: d.FoodHigh},
|
||||
{
|
||||
name: "sim_low", label: "SIM + документы — мин",
|
||||
value: d.SimLow, note: "Разовые cutover",
|
||||
comment: "eSIM, копии, мелкие госпошлины при cutover. Не включает депозит аренды.",
|
||||
},
|
||||
{name: "sim_mid", label: "SIM + документы — mid", value: d.SimMid},
|
||||
{name: "sim_high", label: "SIM + документы — макс", value: d.SimHigh},
|
||||
{
|
||||
name: "month_cap", label: "Потолок бюджета на месяц (€)",
|
||||
value: d.MonthCap, note: "SoT: 2500€",
|
||||
comment: "Жёсткий потолок Sep из source-of-truth. «Остаток» = потолок − итого mid сценария.",
|
||||
},
|
||||
{
|
||||
name: "extra_nights_dom6", label: "Лишние ночи в PT (вс 6.09)",
|
||||
value: d.ExtraNightsDom6, note: "0 = TF до вс",
|
||||
comment: "Платные ночи в PT до начала 7-дневной крыши. Для Dom6 обычно 0: остаёмся на Тенерифе до вс, заезд вт 8.09.",
|
||||
},
|
||||
{
|
||||
name: "extra_nights_jue3", label: "Лишние ночи в PT (чт 3.09)",
|
||||
value: d.ExtraNightsJue3, note: "0 если TF до вс",
|
||||
comment: "Если крыша TF кончается раньше вс — нужны ночи 05–07.09 до Airbnb. По умолчанию 0.",
|
||||
},
|
||||
{
|
||||
name: "extra_nights_wed", label: "Лишние ночи в PT (ср гип)",
|
||||
value: d.ExtraNightsWed, note: "Гип: заезд пт 4.09",
|
||||
comment: "Гипотетический слот ср 2.09 → заезд пт 4.09 = 4 лишних ночи до типичного 7н блока. Рейса ср в FO нет — тариф как Jue3.",
|
||||
},
|
||||
}
|
||||
for i, fld := range fields {
|
||||
row := i + 2
|
||||
if err := f.SetCellStr(SheetInputs, fmt.Sprintf("A%d", row), fld.label); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := defineInput(f, SheetInputs, fld.name, row, fld.value, fld.note, fld.comment); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(SheetInputs, "A25", "—"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(SheetInputs, "B25", "Редактируйте жёлтые ячейки"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(SheetInputs, "C25", "Формулы на листах сценариев и «Сводка» пересчитаются в OnlyOffice"); err != nil {
|
||||
return err
|
||||
}
|
||||
return styleSheet(f, SheetInputs, SheetInputs)
|
||||
}
|
||||
|
||||
func writeScenarioSheet(f *excelize.File, sheet, ferryName, extraNightsName, scenarioNote string) error {
|
||||
if err := setHeaders(f, sheet, "Статья", "Мин €", "Mid €", "Макс €", "Пояснение"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "A2", "Паром Básica"); err != nil {
|
||||
return err
|
||||
}
|
||||
for col, ref := range []string{ferryName, ferryName, ferryName} {
|
||||
cell, _ := excelize.CoordinatesToCellName(col+2, 2)
|
||||
if err := f.SetCellFormula(sheet, cell, "="+ref); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := addCellComment(f, sheet, "A2", "Тариф парома для этого сценария. Берётся с листа «Ввод»."); err != nil {
|
||||
return err
|
||||
}
|
||||
lines := []struct {
|
||||
label string
|
||||
low, mid, high, note string
|
||||
comment string
|
||||
}{
|
||||
{"Крыша 7 н Setúbal", "housing_7n_low", "housing_7n_mid", "housing_7n_high", "Airbnb T1a #11", ""},
|
||||
{"Huelva → Setúbal", "drive_low", "drive_mid", "drive_high", "OSRM ~3,8 ч", ""},
|
||||
{"Еда на борту", "board_low", "board_mid", "board_high", "Меню FO", ""},
|
||||
{"Еда 7 д в PT", "food_7d_low", "food_7d_mid", "food_7d_high", "Готовим дома", ""},
|
||||
{"SIM / документы", "sim_low", "sim_mid", "sim_high", "Cutover", ""},
|
||||
}
|
||||
for i, ln := range lines {
|
||||
row := i + 3
|
||||
if err := setFormulaRow(f, sheet, row, ln.label,
|
||||
"="+ln.low, "="+ln.mid, "="+ln.high, ln.note); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "A8", "Итого Básica"); err != nil {
|
||||
return err
|
||||
}
|
||||
for col := 2; col <= 4; col++ {
|
||||
cell, _ := excelize.CoordinatesToCellName(col, 8)
|
||||
colL, _ := excelize.CoordinatesToCellName(col, 2)
|
||||
colH, _ := excelize.CoordinatesToCellName(col, 7)
|
||||
if err := f.SetCellFormula(sheet, cell, fmt.Sprintf("=SUM(%s:%s)", colL, colH)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "E8", "SUM строк 2–7"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := setFormulaRow(f, sheet, 9, "Итого Superior",
|
||||
"=B8+ferry_superior_uplift", "=C8+ferry_superior_uplift", "=D8+ferry_superior_uplift",
|
||||
"Básica + VIP"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := addCellComment(f, sheet, "A9", "Butaca Superior / VIP salón. Доплата с листа «Ввод»."); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "A10", "Лишние ночи PT"); err != nil {
|
||||
return err
|
||||
}
|
||||
for col, housing := range []string{"housing_7n_low", "housing_7n_mid", "housing_7n_high"} {
|
||||
cell, _ := excelize.CoordinatesToCellName(col+2, 10)
|
||||
formula := fmt.Sprintf("=%s*%s/7", extraNightsName, housing)
|
||||
if err := f.SetCellFormula(sheet, cell, formula); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "E10", "ночей × (крыша/7)"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := addCellComment(f, sheet, "A10", "Платное жильё до начала 7-дневной брони. Число ночей — на листе «Ввод» для этого сценария."); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := setFormulaRow(f, sheet, 11, "Итого с ночами",
|
||||
"=B8+B10", "=C8+C10", "=D8+D10", scenarioNote); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "A12", "Остаток от потолка"); err != nil {
|
||||
return err
|
||||
}
|
||||
for _, col := range []string{"B", "C", "D"} {
|
||||
if err := f.SetCellFormula(sheet, col+"12", "=month_cap-C11"); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "E12", "потолок − mid итого"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := addCellComment(f, sheet, "C12", "Сколько остаётся от месячного потолка 2500€ после cutover (mid)."); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "A13", "Среднее по строкам"); err != nil {
|
||||
return err
|
||||
}
|
||||
for col := 2; col <= 4; col++ {
|
||||
cell, _ := excelize.CoordinatesToCellName(col, 13)
|
||||
colL, _ := excelize.CoordinatesToCellName(col, 2)
|
||||
colH, _ := excelize.CoordinatesToCellName(col, 7)
|
||||
if err := f.SetCellFormula(sheet, cell, fmt.Sprintf("=AVERAGE(%s:%s)", colL, colH)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(sheet, "E13", "AVG статей 2–7"); err != nil {
|
||||
return err
|
||||
}
|
||||
return styleSheet(f, sheet, SheetInputs)
|
||||
}
|
||||
|
||||
func writeSummarySheet(f *excelize.File) error {
|
||||
if err := setHeaders(f, SheetSummary,
|
||||
"Сценарий", "Básica mid", "Superior mid", "Итого mid", "Остаток", "Δ vs вс"); err != nil {
|
||||
return err
|
||||
}
|
||||
rows := []struct {
|
||||
label, sheet string
|
||||
}{
|
||||
{"Вс 6.09 (live)", SheetDom6},
|
||||
{"Чт 3.09 (live)", SheetJue3},
|
||||
{"Ср 2.09 (гипотеза)", SheetWedHyp},
|
||||
}
|
||||
qs := quoteSheet
|
||||
for i, r := range rows {
|
||||
row := i + 2
|
||||
if err := f.SetCellStr(SheetSummary, fmt.Sprintf("A%d", row), r.label); err != nil {
|
||||
return err
|
||||
}
|
||||
pairs := []struct {
|
||||
col int
|
||||
ref string
|
||||
}{
|
||||
{2, fmt.Sprintf("%s!C8", qs(r.sheet))},
|
||||
{3, fmt.Sprintf("%s!C9", qs(r.sheet))},
|
||||
{4, fmt.Sprintf("%s!C11", qs(r.sheet))},
|
||||
{5, fmt.Sprintf("%s!C12", qs(r.sheet))},
|
||||
}
|
||||
for _, p := range pairs {
|
||||
cell, _ := excelize.CoordinatesToCellName(p.col, row)
|
||||
if err := f.SetCellFormula(SheetSummary, cell, "="+p.ref); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
cell, _ := excelize.CoordinatesToCellName(6, row)
|
||||
if err := f.SetCellFormula(SheetSummary, cell, fmt.Sprintf("=D%d-$D$2", row)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := f.SetCellStr(SheetSummary, "A6", "Потолок месяца"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellFormula(SheetSummary, "D6", "=month_cap"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(SheetSummary, "A7", "Лучший итого mid"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellFormula(SheetSummary, "D7", "=MIN(D2:D4)"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(SheetSummary, "E7", "MIN по сценариям"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := addCellComment(f, SheetSummary, "D7", "Минимальный mid «Итого с ночами» среди трёх сценариев. Сейчас обычно вс 6.09."); err != nil {
|
||||
return err
|
||||
}
|
||||
return styleSheet(f, SheetSummary, SheetInputs)
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
package xlspipe
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
func parseCalcFloat(s string) (float64, error) {
|
||||
s = strings.ReplaceAll(s, ",", "")
|
||||
s = strings.TrimSpace(s)
|
||||
var val float64
|
||||
_, err := fmt.Sscan(s, &val)
|
||||
return val, err
|
||||
}
|
||||
|
||||
func TestBuildCutoverPortugalWorkbook_Formulas(t *testing.T) {
|
||||
f, err := BuildCutoverPortugalWorkbook(DefaultCutoverPortugal())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "cutover.xlsx")
|
||||
if err := Save(f, path); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st, err := os.Stat(path)
|
||||
if err != nil || st.Size() < 4096 {
|
||||
t.Fatalf("xlsx too small: %v", err)
|
||||
}
|
||||
|
||||
opened, err := excelize.OpenFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer opened.Close()
|
||||
|
||||
cases := []struct {
|
||||
sheet, cell string
|
||||
want float64
|
||||
tol float64
|
||||
}{
|
||||
{SheetDom6, "C8", 1522.89, 0.02},
|
||||
{SheetJue3, "C8", 1549.09, 0.02},
|
||||
{SheetWedHyp, "C11", 1891.95, 0.05},
|
||||
{SheetSummary, "D2", 1522.89, 0.02},
|
||||
{SheetSummary, "D7", 1522.89, 0.02},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
got, err := opened.CalcCellValue(tc.sheet, tc.cell)
|
||||
if err != nil {
|
||||
t.Fatalf("%s!%s calc: %v", tc.sheet, tc.cell, err)
|
||||
}
|
||||
val, err := parseCalcFloat(got)
|
||||
if err != nil {
|
||||
t.Fatalf("%s!%s parse %q: %v", tc.sheet, tc.cell, got, err)
|
||||
}
|
||||
if diff := val - tc.want; diff < -tc.tol || diff > tc.tol {
|
||||
t.Fatalf("%s!%s = %v want ~%v", tc.sheet, tc.cell, val, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
package xlspipe
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
// Template names for oo docs put-xlsx --template.
|
||||
const (
|
||||
TemplateCutoverPortugal = "cutover-portugal"
|
||||
)
|
||||
|
||||
// BuildTemplate returns an xlsx workbook for a known template name.
|
||||
func BuildTemplate(name string) (*File, error) {
|
||||
switch strings.ToLower(strings.TrimSpace(name)) {
|
||||
case TemplateCutoverPortugal, "portugal-cutover", "cutover":
|
||||
return BuildCutoverPortugalWorkbook(DefaultCutoverPortugal())
|
||||
default:
|
||||
return nil, fmt.Errorf("unknown xlsx template %q (try: %s)", name, TemplateCutoverPortugal)
|
||||
}
|
||||
}
|
||||
|
||||
// File is an alias so cmd/oo can refer to excelize.File without importing excelize in every handler.
|
||||
type File = excelize.File
|
||||
@@ -0,0 +1,135 @@
|
||||
package xlspipe
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
const commentAuthor = "oo"
|
||||
|
||||
// Save writes the workbook to path (creates parent dirs).
|
||||
func Save(f *excelize.File, path string) error {
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
return f.SaveAs(path)
|
||||
}
|
||||
|
||||
// quoteSheet returns an Excel sheet reference safe for formulas ('Name'!A1).
|
||||
func quoteSheet(name string) string {
|
||||
return "'" + strings.ReplaceAll(name, "'", "''") + "'"
|
||||
}
|
||||
|
||||
// defineInput registers a named range on sheet!B{row}, sets value, note, optional comment.
|
||||
func defineInput(f *excelize.File, sheet, name string, row int, value float64, note, comment string) error {
|
||||
cell := fmt.Sprintf("B%d", row)
|
||||
if err := f.SetCellFloat(sheet, cell, value, 2, 64); err != nil {
|
||||
return err
|
||||
}
|
||||
if note != "" {
|
||||
if err := f.SetCellStr(sheet, fmt.Sprintf("C%d", row), note); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
ref := fmt.Sprintf("%s!$B$%d", quoteSheet(sheet), row)
|
||||
if err := f.SetDefinedName(&excelize.DefinedName{Name: name, RefersTo: ref}); err != nil {
|
||||
return err
|
||||
}
|
||||
if comment != "" {
|
||||
return addCellComment(f, sheet, cell, comment)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func addCellComment(f *excelize.File, sheet, cell, text string) error {
|
||||
return f.AddComment(sheet, excelize.Comment{
|
||||
Cell: cell,
|
||||
Author: commentAuthor,
|
||||
Text: text,
|
||||
Width: 280,
|
||||
Height: 120,
|
||||
})
|
||||
}
|
||||
|
||||
func setHeaders(f *excelize.File, sheet string, headers ...string) error {
|
||||
for i, h := range headers {
|
||||
cell, err := excelize.CoordinatesToCellName(i+1, 1)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStr(sheet, cell, h); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func setFormulaRow(f *excelize.File, sheet string, row int, label string, low, mid, high, note string) error {
|
||||
if err := f.SetCellStr(sheet, fmt.Sprintf("A%d", row), label); err != nil {
|
||||
return err
|
||||
}
|
||||
for col, formula := range []string{low, mid, high} {
|
||||
cell, err := excelize.CoordinatesToCellName(col+2, row)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellFormula(sheet, cell, formula); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if note != "" {
|
||||
return f.SetCellStr(sheet, fmt.Sprintf("E%d", row), note)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func styleSheet(f *excelize.File, sheet, inputsSheet string) error {
|
||||
widths := map[string]float64{"A": 34, "B": 12, "C": 12, "D": 12, "E": 40}
|
||||
for col, w := range widths {
|
||||
if err := f.SetColWidth(sheet, col, col, w); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
headerStyle, err := f.NewStyle(&excelize.Style{
|
||||
Font: &excelize.Font{Bold: true},
|
||||
Fill: excelize.Fill{Type: "pattern", Color: []string{"#E8F0FE"}, Pattern: 1},
|
||||
Alignment: &excelize.Alignment{Horizontal: "center", WrapText: true},
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
euroStyle, err := f.NewStyle(&excelize.Style{NumFmt: 4})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
inputStyle, err := f.NewStyle(&excelize.Style{
|
||||
NumFmt: 4,
|
||||
Fill: excelize.Fill{Type: "pattern", Color: []string{"#FFF9E6"}, Pattern: 1},
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := f.SetCellStyle(sheet, "A1", "E1", headerStyle); err != nil {
|
||||
return err
|
||||
}
|
||||
if sheet == inputsSheet {
|
||||
return f.SetCellStyle(sheet, "B2", "B30", inputStyle)
|
||||
}
|
||||
if err := f.SetCellStyle(sheet, "B2", "D20", euroStyle); err != nil {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func finalizeWorkbook(f *excelize.File) error {
|
||||
if err := f.SetCalcProps(&excelize.CalcPropsOptions{FullCalcOnLoad: boolPtr(true)}); err != nil {
|
||||
return err
|
||||
}
|
||||
return f.UpdateLinkedValue()
|
||||
}
|
||||
|
||||
func boolPtr(v bool) *bool { return &v }
|
||||
@@ -7,6 +7,8 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/mail"
|
||||
"net/url"
|
||||
"strconv"
|
||||
@@ -96,6 +98,38 @@ func (c *Client) GetMailMessage(ctx context.Context, messageID string) (map[stri
|
||||
return c.ResponseObject(ctx, "/api/2.0/mail/messages/"+url.PathEscape(id))
|
||||
}
|
||||
|
||||
// DownloadMailAttachment fetches raw attachment bytes by mail attachment id via
|
||||
// the mail addon's download.ashx handler. This path relies on the session
|
||||
// cookie captured during authentication, so NewClient configures a cookie jar.
|
||||
func (c *Client) DownloadMailAttachment(ctx context.Context, attachmentID string) ([]byte, error) {
|
||||
id := strings.TrimSpace(attachmentID)
|
||||
if id == "" {
|
||||
return nil, fmt.Errorf("DownloadMailAttachment: attachment id is required")
|
||||
}
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, c.baseURL()+"/addons/mail/httphandlers/download.ashx?attachid="+url.QueryEscape(id), nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Authorization", auth)
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
raw, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("DownloadMailAttachment %s: %d %s", id, resp.StatusCode, truncate(string(raw), 400))
|
||||
}
|
||||
return raw, nil
|
||||
}
|
||||
|
||||
// RemoveMailMessages deletes messages by id (PUT /api/2.0/mail/messages/remove).
|
||||
// The API response "response" field may be a number or object; success is HTTP 2xx.
|
||||
func (c *Client) RemoveMailMessages(ctx context.Context, ids ...int) (map[string]any, error) {
|
||||
@@ -159,6 +193,49 @@ func (c *Client) SaveMailDraft(ctx context.Context, p SaveMailDraftParams) (map[
|
||||
return c.putJSONObject(ctx, "/api/2.0/mail/drafts/save", body)
|
||||
}
|
||||
|
||||
// SendMailParams describes a message to send via PUT /api/2.0/mail/messages/send.
|
||||
// ID refers to an existing draft/message id; From falls back to the first enabled
|
||||
// mailbox. Cc/Bcc are omitted when empty (the API 400s on empty strings). Chat
|
||||
// line goes into Body (API send does not append the UI signature).
|
||||
type SendMailParams struct {
|
||||
ID int64
|
||||
From string
|
||||
To string
|
||||
Cc string
|
||||
Bcc string
|
||||
Subject string
|
||||
Body string // HTML
|
||||
}
|
||||
|
||||
// SendMail sends an existing draft (or a fresh message) via the OnlyOffice Mail
|
||||
// send endpoint. Returns the raw send response.
|
||||
func (c *Client) SendMail(ctx context.Context, p SendMailParams) (json.RawMessage, error) {
|
||||
if strings.TrimSpace(p.To) == "" {
|
||||
return nil, fmt.Errorf("SendMail: to is required")
|
||||
}
|
||||
if strings.TrimSpace(p.From) == "" {
|
||||
from, err := c.defaultMailFrom(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
p.From = from
|
||||
}
|
||||
body := map[string]any{
|
||||
"id": p.ID,
|
||||
"from": p.From,
|
||||
"to": p.To,
|
||||
"subject": p.Subject,
|
||||
"body": p.Body,
|
||||
}
|
||||
if strings.TrimSpace(p.Cc) != "" {
|
||||
body["cc"] = p.Cc
|
||||
}
|
||||
if strings.TrimSpace(p.Bcc) != "" {
|
||||
body["bcc"] = p.Bcc
|
||||
}
|
||||
return c.putJSON(ctx, "/api/2.0/mail/messages/send.json", body)
|
||||
}
|
||||
|
||||
func (c *Client) defaultMailFrom(ctx context.Context) (string, error) {
|
||||
accounts, err := c.ListMailAccounts(ctx)
|
||||
if err != nil {
|
||||
|
||||
+118
-3
@@ -1,8 +1,14 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/cookiejar"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestResolveMailFolder(t *testing.T) {
|
||||
@@ -44,9 +50,9 @@ func TestMailMessagesPath(t *testing.T) {
|
||||
|
||||
func TestParseMailAddress(t *testing.T) {
|
||||
tests := []struct {
|
||||
raw string
|
||||
wantName string
|
||||
wantAddress string
|
||||
raw string
|
||||
wantName string
|
||||
wantAddress string
|
||||
}{
|
||||
{`"LinkedIn Jobbenachrichtigungen" <jobalerts-noreply@linkedin.com>`, "LinkedIn Jobbenachrichtigungen", "jobalerts-noreply@linkedin.com"},
|
||||
{`"Bitfinex" <no-reply@bitfinex.com>`, "Bitfinex", "no-reply@bitfinex.com"},
|
||||
@@ -97,3 +103,112 @@ func TestInt64FromMap(t *testing.T) {
|
||||
t.Fatal("string")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewClientSetsCookieJar(t *testing.T) {
|
||||
c := NewClient(Credentials{Url: "https://example.test", User: "u", Password: "p"})
|
||||
if c.client == nil {
|
||||
t.Fatal("client is nil")
|
||||
}
|
||||
if c.client.Jar == nil {
|
||||
t.Fatal("cookie jar is nil")
|
||||
}
|
||||
if _, ok := c.client.Jar.(*cookiejar.Jar); !ok {
|
||||
t.Fatalf("unexpected jar type %T", c.client.Jar)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDownloadMailAttachmentUsesAuthCookie(t *testing.T) {
|
||||
var gotAuth, gotCookie, gotPath string
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
switch r.URL.Path {
|
||||
case "/api/2.0/authentication.json":
|
||||
http.SetCookie(w, &http.Cookie{Name: "sessionid", Value: "abc123", Path: "/"})
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"response":{"token":"tok","expires":"2099-01-01T00:00:00.0000000+00:00"}}`))
|
||||
case "/addons/mail/httphandlers/download.ashx":
|
||||
gotAuth = r.Header.Get("Authorization")
|
||||
gotCookie = r.Header.Get("Cookie")
|
||||
gotPath = r.URL.RequestURI()
|
||||
if gotCookie == "" {
|
||||
http.Error(w, "missing cookie", http.StatusUnauthorized)
|
||||
return
|
||||
}
|
||||
_, _ = w.Write([]byte("payload"))
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
}
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
c := NewClient(Credentials{Url: srv.URL, User: "u", Password: "p"})
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
body, err := c.DownloadMailAttachment(ctx, "42")
|
||||
if err != nil {
|
||||
t.Fatalf("DownloadMailAttachment: %v", err)
|
||||
}
|
||||
if string(body) != "payload" {
|
||||
t.Fatalf("body = %q", body)
|
||||
}
|
||||
if gotAuth != "tok" {
|
||||
t.Fatalf("auth header = %q", gotAuth)
|
||||
}
|
||||
if !strings.Contains(gotCookie, "sessionid=abc123") {
|
||||
t.Fatalf("cookie header = %q", gotCookie)
|
||||
}
|
||||
if gotPath != "/addons/mail/httphandlers/download.ashx?attachid=42" {
|
||||
t.Fatalf("path = %q", gotPath)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSendMailOmitsEmptyCcBcc(t *testing.T) {
|
||||
var gotBody map[string]any
|
||||
var gotPath string
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
switch r.URL.Path {
|
||||
case "/api/2.0/authentication.json":
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"response":{"token":"tok","expires":"2099-01-01T00:00:00.0000000+00:00"}}`))
|
||||
case "/api/2.0/mail/messages/send.json":
|
||||
gotPath = r.URL.Path
|
||||
dec := json.NewDecoder(r.Body)
|
||||
_ = dec.Decode(&gotBody)
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"response":{"id":1}}`))
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
}
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
c := NewClient(Credentials{Url: srv.URL, User: "u", Password: "p"})
|
||||
ctx := context.Background()
|
||||
raw, err := c.SendMail(ctx, SendMailParams{
|
||||
ID: 99,
|
||||
From: "me@x.com",
|
||||
To: "a@b.com",
|
||||
Subject: "hi",
|
||||
Body: "<p>hello</p>",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("SendMail: %v", err)
|
||||
}
|
||||
if gotPath != "/api/2.0/mail/messages/send.json" {
|
||||
t.Fatalf("path = %q", gotPath)
|
||||
}
|
||||
if _, hasCC := gotBody["cc"]; hasCC {
|
||||
t.Fatalf("empty cc should be omitted: %v", gotBody)
|
||||
}
|
||||
if _, hasBcc := gotBody["bcc"]; hasBcc {
|
||||
t.Fatalf("empty bcc should be omitted: %v", gotBody)
|
||||
}
|
||||
if gotBody["to"] != "a@b.com" {
|
||||
t.Fatalf("to = %v", gotBody["to"])
|
||||
}
|
||||
if gotBody["id"] != float64(99) {
|
||||
t.Fatalf("id = %v", gotBody["id"])
|
||||
}
|
||||
if !strings.Contains(string(raw), `"id"`) {
|
||||
t.Fatalf("raw = %s", raw)
|
||||
}
|
||||
}
|
||||
|
||||
+175
@@ -0,0 +1,175 @@
|
||||
package onlyoffice
|
||||
|
||||
// High-level mail folder walk for ETL consumers (2dph brain mail-ingest,
|
||||
// cv tools). This is the "integration layer" half of reusing the canonical
|
||||
// client instead of private per-project OOClient copies: the caller gets a
|
||||
// single hydrated stream instead of hand-rolling list → get → download
|
||||
// against the raw API.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strconv"
|
||||
"time"
|
||||
)
|
||||
|
||||
// MailSyncAttachment is one attachment of a hydrated mail message.
|
||||
type MailSyncAttachment struct {
|
||||
ID string // id accepted by Client.DownloadMailAttachment
|
||||
Name string
|
||||
Size int64
|
||||
Body []byte // non-nil only when MailSyncOptions.FetchBodies is set
|
||||
}
|
||||
|
||||
// MailSyncMessage is a hydrated mail message for sync pipelines.
|
||||
type MailSyncMessage struct {
|
||||
ID int64
|
||||
Folder int
|
||||
Subject string
|
||||
From string // raw RFC 5322 header value ("Name" <addr>)
|
||||
Date time.Time
|
||||
IsNew bool
|
||||
HasAttachments bool
|
||||
Attachments []MailSyncAttachment
|
||||
}
|
||||
|
||||
// MailSyncOptions controls FetchMailFolder.
|
||||
type MailSyncOptions struct {
|
||||
Limit int // max messages to hydrate; 0 = whole folder
|
||||
StartIndex int // skip this many messages before collecting
|
||||
FetchBodies bool // eagerly download attachment bytes
|
||||
}
|
||||
|
||||
// FetchMailFolder walks a mail folder page by page and hydrates every
|
||||
// message: list → get → (optionally) download attachments. It is the single
|
||||
// entry point sync pipelines need on top of the mail API.
|
||||
//
|
||||
// Messages are returned in API order (newest first). The folder walk stops
|
||||
// at the first empty or short page.
|
||||
func (c *Client) FetchMailFolder(ctx context.Context, folderID int, opts MailSyncOptions) ([]MailSyncMessage, error) {
|
||||
if folderID <= 0 {
|
||||
folderID = MailFolderInbox
|
||||
}
|
||||
var out []MailSyncMessage
|
||||
skipped := 0
|
||||
for page := 1; ; page++ {
|
||||
batch, err := c.ResponseArray(ctx,
|
||||
mailMessagesPath(MailMessagesFilter{Folder: folderID}, page, mailMessagesPageSize))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("FetchMailFolder: %w", err)
|
||||
}
|
||||
if len(batch) == 0 {
|
||||
break
|
||||
}
|
||||
for _, raw := range batch {
|
||||
if skipped < opts.StartIndex {
|
||||
skipped++
|
||||
continue
|
||||
}
|
||||
msg, err := c.hydrateMailMessage(ctx, raw, opts)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, *msg)
|
||||
if opts.Limit > 0 && len(out) >= opts.Limit {
|
||||
return out, nil
|
||||
}
|
||||
}
|
||||
if len(batch) < mailMessagesPageSize {
|
||||
break
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// hydrateMailMessage converts one raw API message into a MailSyncMessage,
|
||||
// fetching the full record when the list item does not carry the attachment
|
||||
// metadata, and downloading bodies when requested.
|
||||
func (c *Client) hydrateMailMessage(ctx context.Context, m map[string]any, opts MailSyncOptions) (*MailSyncMessage, error) {
|
||||
msg := &MailSyncMessage{
|
||||
ID: Int64FromMap(m, "id"),
|
||||
Folder: int(Int64FromMap(m, "folder")),
|
||||
Subject: stringFromMap(m, "subject"),
|
||||
From: stringFromMap(m, "from"),
|
||||
IsNew: boolFromMap(m, "isNew") == "true",
|
||||
}
|
||||
msg.Date = parseMailTime(stringFromMap(m, "date"))
|
||||
|
||||
atts, _ := m["attachments"].([]any)
|
||||
hasFlag := boolFromMap(m, "hasAttachments") == "true"
|
||||
if hasFlag && len(atts) == 0 {
|
||||
// List items may omit the attachment array; pull the full record.
|
||||
full, err := c.GetMailMessage(ctx, strconv.FormatInt(msg.ID, 10))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("FetchMailFolder: hydrate message %d: %w", msg.ID, err)
|
||||
}
|
||||
atts, _ = full["attachments"].([]any)
|
||||
}
|
||||
for _, a := range atts {
|
||||
am, ok := a.(map[string]any)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
att := MailSyncAttachment{
|
||||
ID: mailAttachmentID(am),
|
||||
Name: stringFromMap(am, "fileName"),
|
||||
Size: Int64FromMap(am, "size"),
|
||||
}
|
||||
if att.Name == "" {
|
||||
att.Name = stringFromMap(am, "name")
|
||||
}
|
||||
if att.ID != "" {
|
||||
msg.Attachments = append(msg.Attachments, att)
|
||||
}
|
||||
}
|
||||
msg.HasAttachments = hasFlag || len(msg.Attachments) > 0
|
||||
|
||||
if opts.FetchBodies {
|
||||
for i := range msg.Attachments {
|
||||
body, err := c.DownloadMailAttachment(ctx, msg.Attachments[i].ID)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("FetchMailFolder: message %d attachment %q: %w",
|
||||
msg.ID, msg.Attachments[i].Name, err)
|
||||
}
|
||||
msg.Attachments[i].Body = body
|
||||
}
|
||||
}
|
||||
return msg, nil
|
||||
}
|
||||
|
||||
// mailAttachmentID extracts the download id from an attachment object.
|
||||
// OnlyOffice variants use "id", "fileId" or "attachmentId".
|
||||
func mailAttachmentID(am map[string]any) string {
|
||||
for _, key := range []string{"id", "fileId", "attachmentId"} {
|
||||
switch v := am[key].(type) {
|
||||
case string:
|
||||
if s := v; s != "" {
|
||||
return s
|
||||
}
|
||||
case float64:
|
||||
if n := int64(v); n != 0 {
|
||||
return strconv.FormatInt(n, 10)
|
||||
}
|
||||
case int64:
|
||||
if v != 0 {
|
||||
return strconv.FormatInt(v, 10)
|
||||
}
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// parseMailTime accepts the OnlyOffice timestamp shapes seen in the wild:
|
||||
// RFC3339 (with any fractional digits) and second-precision local form.
|
||||
func parseMailTime(s string) time.Time {
|
||||
if s == "" {
|
||||
return time.Time{}
|
||||
}
|
||||
if t, err := time.Parse(time.RFC3339, s); err == nil {
|
||||
return t
|
||||
}
|
||||
if t, err := time.Parse("2006-01-02T15:04:05", s); err == nil {
|
||||
return t
|
||||
}
|
||||
return time.Time{}
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// mailsSyncMock serves a two-page inbox: page 1 has two list items (one
|
||||
// reporting hasAttachments but omitting the attachment array, forcing the
|
||||
// full-record fetch), page 2 is empty. The full record for message 102
|
||||
// carries one attachment whose body is served by download.ashx.
|
||||
func newMailSyncTestServer(t *testing.T, msgsPage1 string) *httptest.Server {
|
||||
t.Helper()
|
||||
return httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
switch {
|
||||
case r.URL.Path == "/api/2.0/authentication.json":
|
||||
http.SetCookie(w, &http.Cookie{Name: "sessionid", Value: "abc", Path: "/"})
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"response":{"token":"tok","expires":"2099-01-01T00:00:00.0000000+00:00"}}`))
|
||||
|
||||
case r.URL.Path == "/api/2.0/mail/messages":
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
if r.URL.Query().Get("page") > "1" {
|
||||
_, _ = w.Write([]byte(`{"response":[]}`))
|
||||
return
|
||||
}
|
||||
_, _ = w.Write([]byte(`{"response":[` + msgsPage1 + `]}`))
|
||||
|
||||
case r.URL.Path == "/api/2.0/mail/messages/102":
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"response":{
|
||||
"id":102,"subject":"Full record","from":"\"A\" <a@b.com>",
|
||||
"date":"2026-08-22T10:15:00+02:00","folder":1,"isNew":false,
|
||||
"hasAttachments":true,
|
||||
"attachments":[{"id":77,"fileName":"report.pdf","size":3}]}}`))
|
||||
|
||||
case r.URL.Path == "/addons/mail/httphandlers/download.ashx":
|
||||
if r.Header.Get("Cookie") == "" {
|
||||
http.Error(w, "missing cookie", http.StatusUnauthorized)
|
||||
return
|
||||
}
|
||||
_, _ = w.Write([]byte("PDF!"))
|
||||
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
}
|
||||
}))
|
||||
}
|
||||
|
||||
func TestFetchMailFolderHydratesAndDownloads(t *testing.T) {
|
||||
page1 := `
|
||||
{"id":101,"subject":"Plain","from":"x@y.z","date":"2026-08-21T09:00:00Z",
|
||||
"folder":1,"isNew":true,"hasAttachments":false},
|
||||
{"id":102,"subject":"With attachment (list item)","from":"a@b.com",
|
||||
"date":"2026-08-22T10:15:00+02:00","folder":1,"isNew":false,
|
||||
"hasAttachments":true}
|
||||
`
|
||||
srv := newMailSyncTestServer(t, page1)
|
||||
defer srv.Close()
|
||||
|
||||
c := NewClient(Credentials{Url: srv.URL, User: "u", Password: "p"})
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
|
||||
msgs, err := c.FetchMailFolder(ctx, MailFolderInbox, MailSyncOptions{FetchBodies: true})
|
||||
if err != nil {
|
||||
t.Fatalf("FetchMailFolder: %v", err)
|
||||
}
|
||||
if len(msgs) != 2 {
|
||||
t.Fatalf("got %d messages, want 2", len(msgs))
|
||||
}
|
||||
|
||||
first := msgs[0]
|
||||
if first.ID != 101 || first.Subject != "Plain" || !first.IsNew {
|
||||
t.Fatalf("first = %+v", first)
|
||||
}
|
||||
if first.Date.IsZero() || first.Date.Year() != 2026 {
|
||||
t.Fatalf("first date = %v", first.Date)
|
||||
}
|
||||
if first.HasAttachments {
|
||||
t.Fatalf("first should have no attachments")
|
||||
}
|
||||
|
||||
second := msgs[1]
|
||||
if !second.HasAttachments || len(second.Attachments) != 1 {
|
||||
t.Fatalf("second attachments = %+v", second.Attachments)
|
||||
}
|
||||
att := second.Attachments[0]
|
||||
if att.ID != "77" || att.Name != "report.pdf" || att.Size != 3 || string(att.Body) != "PDF!" {
|
||||
t.Fatalf("attachment = %+v", att)
|
||||
}
|
||||
if second.Date.Location() == time.UTC && second.Date.Hour() != 8 {
|
||||
t.Fatalf("second date = %v (want +02:00 offset preserved)", second.Date)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFetchMailFolderLimitAndStartIndex(t *testing.T) {
|
||||
var items []string
|
||||
for i := 1; i <= 5; i++ {
|
||||
items = append(items, `{"id":`+string(rune('0'+i))+`,"subject":"m`+string(rune('0'+i))+`",
|
||||
"from":"x@y.z","date":"2026-08-20T00:00:00Z","folder":1}`)
|
||||
}
|
||||
srv := newMailSyncTestServer(t, strings.Join(items, ","))
|
||||
defer srv.Close()
|
||||
|
||||
c := NewClient(Credentials{Url: srv.URL, User: "u", Password: "p"})
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
|
||||
got, err := c.FetchMailFolder(ctx, MailFolderInbox, MailSyncOptions{StartIndex: 1, Limit: 2})
|
||||
if err != nil {
|
||||
t.Fatalf("FetchMailFolder: %v", err)
|
||||
}
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("got %d messages, want 2", len(got))
|
||||
}
|
||||
if got[0].ID != 2 || got[1].ID != 3 {
|
||||
t.Fatalf("ids = %d,%d want 2,3", got[0].ID, got[1].ID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseMailTime(t *testing.T) {
|
||||
fractions := "2026-08-22T10:15:00.1234567+02:00"
|
||||
if parseMailTime(fractions).IsZero() {
|
||||
t.Fatalf("RFC3339 with 7-digit fraction failed: %q", fractions)
|
||||
}
|
||||
if parseMailTime("2026-08-22T10:15:00").IsZero() {
|
||||
t.Fatal("second-precision form failed")
|
||||
}
|
||||
if !parseMailTime("").IsZero() || !parseMailTime("garbage").IsZero() {
|
||||
t.Fatal("unparseable input must yield zero time")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"regexp"
|
||||
"time"
|
||||
)
|
||||
|
||||
// RetryPolicy controls deterministic retries against OnlyOffice: fixed linear
|
||||
// backoff without jitter, so repeated runs wait exactly the same schedule.
|
||||
// OnlyOffice throttles bulk reads/writes with 429 (and occasional 502/503/504
|
||||
// from openresty), so every bulk tool routes API calls through DoRetry.
|
||||
type RetryPolicy struct {
|
||||
Attempts int // total attempts, including the first try
|
||||
Base time.Duration // wait before retry N is N*Base
|
||||
Max time.Duration // per-wait cap
|
||||
}
|
||||
|
||||
// DefaultRetryPolicy retries up to 5 times with 1s, 2s, 3s, 4s waits.
|
||||
func DefaultRetryPolicy() RetryPolicy {
|
||||
return RetryPolicy{Attempts: 5, Base: time.Second, Max: 30 * time.Second}
|
||||
}
|
||||
|
||||
var transientRe = regexp.MustCompile(`:\s*(429|502|503|504)\b`)
|
||||
|
||||
// Transient reports whether err looks like a transient OnlyOffice answer
|
||||
// (an HTTP 429/502/503/504 surfaced as "...: <code> ...").
|
||||
func Transient(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
return transientRe.MatchString(err.Error())
|
||||
}
|
||||
|
||||
// DoRetry runs fn until it succeeds, fails non-transiently, or attempts run
|
||||
// out. Waits are deterministic: N*Base capped at Max, no jitter.
|
||||
func DoRetry(ctx context.Context, p RetryPolicy, fn func() error) error {
|
||||
if p.Attempts < 1 {
|
||||
p.Attempts = 1
|
||||
}
|
||||
var err error
|
||||
for attempt := 1; attempt <= p.Attempts; attempt++ {
|
||||
if ctx.Err() != nil {
|
||||
return ctx.Err()
|
||||
}
|
||||
if err = fn(); err == nil || !Transient(err) || attempt == p.Attempts {
|
||||
return err
|
||||
}
|
||||
wait := time.Duration(attempt) * p.Base
|
||||
if wait > p.Max {
|
||||
wait = p.Max
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-time.After(wait):
|
||||
}
|
||||
}
|
||||
return err
|
||||
}
|
||||
Executable
+26
@@ -0,0 +1,26 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# install.sh — symlinks the shared githooks (pre-push, pre-commit) into .git/hooks
|
||||
# for this repository. Safe to run repeatedly.
|
||||
#
|
||||
# Usage:
|
||||
# ./scripts/githooks/install.sh
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(git rev-parse --show-toplevel)"
|
||||
SRC="$ROOT/scripts/githooks"
|
||||
HOOKS="$ROOT/.git/hooks"
|
||||
|
||||
mkdir -p "$HOOKS"
|
||||
chmod +x "$SRC"/secret-scan.sh "$SRC"/pre-push "$SRC"/pre-commit
|
||||
|
||||
for h in pre-push pre-commit; do
|
||||
if [[ -e "$HOOKS/$h" ]] && [[ ! -L "$HOOKS/$h" ]]; then
|
||||
echo "error: $HOOKS/$h already exists and is not a symlink; remove it first" >&2
|
||||
exit 1
|
||||
fi
|
||||
ln -sfn "$SRC/$h" "$HOOKS/$h"
|
||||
echo "installed $h -> $SRC/$h"
|
||||
done
|
||||
echo "githooks installed for $(basename "$ROOT")"
|
||||
Executable
+20
@@ -0,0 +1,20 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# pre-commit git hook — blocks a commit if staged changes contain a secret.
|
||||
# Scans only the staged (index) diff with gitleaks (via secret-scan.sh).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$(readlink -f "${BASH_SOURCE[0]}")")" && pwd)"
|
||||
ROOT="$(git rev-parse --show-toplevel)"
|
||||
if [[ "$SCRIPT_DIR" == "$ROOT/scripts/githooks" ]]; then
|
||||
SCAN="$SCRIPT_DIR/secret-scan.sh"
|
||||
else
|
||||
SCAN="$ROOT/scripts/githooks/secret-scan.sh"
|
||||
fi
|
||||
|
||||
if ! "$SCAN" --staged; then
|
||||
echo "pre-commit: LEAK FOUND in staged changes; commit BLOCKED. Remove the secret first." >&2
|
||||
exit 1
|
||||
fi
|
||||
exit 0
|
||||
Executable
+66
@@ -0,0 +1,66 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# pre-push git hook — blocks a push if any NEW commit leaks a secret.
|
||||
#
|
||||
# Reads the refs being pushed from stdin (format:
|
||||
# <local ref> <local sha> <remote ref> <remote sha>
|
||||
# per ref). Scans only the new commits with gitleaks (via secret-scan.sh) and
|
||||
# aborts (exit 1) if anything is found.
|
||||
#
|
||||
# Install: ln -s ../../scripts/githooks/pre-push .git/hooks/pre-push
|
||||
# (or run scripts/githooks/install.sh)
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# Resolve symlinks so the hook works whether copied into .git/hooks or
|
||||
# symlinked from scripts/githooks (and in worktrees sharing the main repo hooks).
|
||||
SCRIPT_DIR="$(cd "$(dirname "$(readlink -f "${BASH_SOURCE[0]}")")" && pwd)"
|
||||
ROOT="$(git rev-parse --show-toplevel)"
|
||||
if [[ "$SCRIPT_DIR" == "$ROOT/scripts/githooks" ]]; then
|
||||
SCAN="$SCRIPT_DIR/secret-scan.sh"
|
||||
else
|
||||
SCAN="$ROOT/scripts/githooks/secret-scan.sh"
|
||||
fi
|
||||
|
||||
zero=0000000000000000000000000000000000000000
|
||||
failed=0
|
||||
|
||||
while read -r local_ref local_sha remote_ref remote_sha; do
|
||||
[[ -n "$local_sha" ]] || continue
|
||||
|
||||
# Deletion push — nothing to scan.
|
||||
if [[ "$local_sha" == "$zero" ]]; then
|
||||
continue
|
||||
fi
|
||||
|
||||
# New branch (no remote ref yet): scan only the commits this branch ADDS over
|
||||
# its merge-base with the integration branch (PR target), NOT all history.
|
||||
# This keeps pre-existing historical leaks (see #140) from blocking new work.
|
||||
if [[ "$remote_sha" == "$zero" ]]; then
|
||||
base=""
|
||||
for base_ref in origin/release/v1 origin/release/v2 origin/master origin/main; do
|
||||
if git rev-parse --verify "$base_ref" >/dev/null 2>&1; then
|
||||
base="$(git merge-base "$base_ref" "$local_sha" 2>/dev/null)"
|
||||
break
|
||||
fi
|
||||
done
|
||||
if [[ -z "$base" ]] && git rev-parse --verify origin/HEAD >/dev/null 2>&1; then
|
||||
base="$(git merge-base origin/HEAD "$local_sha" 2>/dev/null)"
|
||||
fi
|
||||
[[ -z "$base" ]] && base="$(git rev-list --max-parents=0 "$local_sha" 2>/dev/null | tail -1)"
|
||||
range="${base}..${local_sha}"
|
||||
else
|
||||
range="${remote_sha}..${local_sha}"
|
||||
fi
|
||||
|
||||
echo "secret-scan: scanning new commits ${range} (ref ${local_ref})"
|
||||
if ! "$SCAN" --range "$range"; then
|
||||
echo "secret-scan: LEAK FOUND in ${local_ref}; push BLOCKED. Remove the secret before pushing." >&2
|
||||
failed=1
|
||||
fi
|
||||
done
|
||||
|
||||
if [[ "$failed" -ne 0 ]]; then
|
||||
exit 1
|
||||
fi
|
||||
exit 0
|
||||
Executable
+55
@@ -0,0 +1,55 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# secret-scan.sh — shared secret scanner used by git hooks (pre-push, pre-commit)
|
||||
# and by SE agents before any push.
|
||||
#
|
||||
# Scans ONLY the diff of new commits (or staged changes) with gitleaks, never the
|
||||
# full history. Fails (exit non-zero) on any finding, so a leaking push is blocked.
|
||||
#
|
||||
# Uses the gitleaks Docker image (gitleaks/gitleaks) if docker is available,
|
||||
# otherwise a locally installed `gitleaks` binary. No secret VALUES are ever
|
||||
# printed: findings are emitted redacted.
|
||||
#
|
||||
# Usage:
|
||||
# secret-scan.sh <range> scan a git log range, e.g. origin/release/v1..HEAD
|
||||
# secret-scan.sh --staged scan staged (index) changes
|
||||
# secret-scan.sh --all scan full history (warning: not for normal use)
|
||||
#
|
||||
# Exit codes: 0 = clean, 1 = leaks found (caller should abort), 2 = scan failed.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
GITLEAKS_IMAGE="zricethezav/gitleaks:latest"
|
||||
|
||||
run_gitleaks() {
|
||||
# $@ = gitleaks args; runs in current dir (a git repo).
|
||||
if command -v gitleaks >/dev/null 2>&1; then
|
||||
gitleaks "$@"
|
||||
elif command -v docker >/dev/null 2>&1 && docker info >/dev/null 2>&1; then
|
||||
docker run --rm -v "$PWD:/repo" -w /repo "$GITLEAKS_IMAGE" "$@"
|
||||
else
|
||||
echo "error: secret-scan: neither 'gitleaks' binary nor docker image available" >&2
|
||||
exit 2
|
||||
fi
|
||||
}
|
||||
|
||||
mode="${1:---range}"
|
||||
shift || true
|
||||
|
||||
case "$mode" in
|
||||
--range)
|
||||
range="${1:?usage: secret-scan.sh <range>}"
|
||||
run_gitleaks detect --source "$PWD" --no-banner --redact --log-opts="$range" >&2
|
||||
;;
|
||||
--staged)
|
||||
# Scan only staged (index) content: pipe `git diff --cached` through gitleaks --pipe.
|
||||
git diff --cached --binary | run_gitleaks detect --pipe --no-banner --redact >&2
|
||||
;;
|
||||
--all)
|
||||
run_gitleaks detect --source "$PWD" --no-banner --redact >&2
|
||||
;;
|
||||
*)
|
||||
echo "error: secret-scan: unknown mode '$mode'" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,200 @@
|
||||
package onlyoffice
|
||||
|
||||
// MinIO download fallback for the portal's stale AWS S3 consumer.
|
||||
//
|
||||
// On the Fibu EDL portal some older Documents files live in S3/MinIO, but the
|
||||
// portal's storage consumer still points at s3.us-east-1.amazonaws.com with
|
||||
// access key "minio". Downloads of those files answer 403 InvalidAccessKeyId.
|
||||
// The bytes are present in the local MinIO store under a deterministic object
|
||||
// key, so the client retries the GET there.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/aws/aws-sdk-go-v2/aws"
|
||||
"github.com/aws/aws-sdk-go-v2/aws/signer/v4"
|
||||
)
|
||||
|
||||
const (
|
||||
defaultMinioEndpoint = "http://192.168.188.10:9000"
|
||||
defaultMinioBucket = "office"
|
||||
minioRegion = "us-east-1"
|
||||
)
|
||||
|
||||
// minioObjectKey is the fallback object key layout the portal's S3 consumer
|
||||
// writes for Documents files: 00/00/01/files/folder_<folderId>/file_<fileId>/v1/content.pdf.
|
||||
// Prefer minioObjectKeyFromURL: the portal stores all files below its storage
|
||||
// root folder, which is not the API folderId returned by GetFile.
|
||||
func minioObjectKey(fileID, folderID string) string {
|
||||
return "00/00/01/files/folder_" + folderID + "/file_" + fileID + "/v1/content.pdf"
|
||||
}
|
||||
|
||||
// minioObjectKeyFromURL extracts the object key from an S3 download URL. Path
|
||||
// style URLs (bucket as first path segment) have that segment removed; virtual
|
||||
// host style URLs are returned as-is. This is authoritative: the portal signs
|
||||
// the exact key, so no folder-id guessing is needed.
|
||||
func minioObjectKeyFromURL(rawURL, bucket string) (string, bool) {
|
||||
u, err := url.Parse(strings.TrimSpace(rawURL))
|
||||
if err != nil || u.Path == "" {
|
||||
return "", false
|
||||
}
|
||||
segs := strings.Split(strings.Trim(u.Path, "/"), "/")
|
||||
// Path-style URLs carry the bucket as leading segment; the portal's S3
|
||||
// consumer can emit it twice (serviceurl already includes the bucket), so
|
||||
// strip every leading segment equal to the bucket.
|
||||
for len(segs) > 0 && bucket != "" && segs[0] == bucket {
|
||||
segs = segs[1:]
|
||||
}
|
||||
if len(segs) == 0 {
|
||||
return "", false
|
||||
}
|
||||
for _, s := range segs {
|
||||
if s == "" || s == "." || s == ".." {
|
||||
return "", false
|
||||
}
|
||||
}
|
||||
return strings.Join(segs, "/"), true
|
||||
}
|
||||
|
||||
// isStaleS3Redirect reports whether a download landed on the portal's stale AWS
|
||||
// S3 consumer. Such responses either carry an S3 InvalidAccessKeyId XML body or
|
||||
// point at amazonaws.com with the "minio" access key id in the query.
|
||||
func isStaleS3Redirect(rawURL string, body []byte) bool {
|
||||
if strings.Contains(strings.ToLower(string(body)), "invalidaccesskeyid") {
|
||||
return true
|
||||
}
|
||||
u, err := url.Parse(strings.TrimSpace(rawURL))
|
||||
if err != nil || u.Host == "" {
|
||||
return false
|
||||
}
|
||||
host := strings.ToLower(u.Host)
|
||||
if !strings.Contains(host, "amazonaws.com") {
|
||||
return false
|
||||
}
|
||||
q := strings.ToLower(u.RawQuery)
|
||||
return strings.Contains(q, "accesskeyid=minio") || strings.Contains(q, "x-amz-credential=minio")
|
||||
}
|
||||
|
||||
// minioConfig is the runtime configuration for the local MinIO fallback.
|
||||
type minioConfig struct {
|
||||
Endpoint string
|
||||
Bucket string
|
||||
AccessKey string
|
||||
SecretKey string
|
||||
}
|
||||
|
||||
// loadMinioConfig reads the fallback configuration from the environment.
|
||||
// Secrets are never defaulted; without access/secret keys the fallback is off.
|
||||
func loadMinioConfig() minioConfig {
|
||||
return minioConfig{
|
||||
Endpoint: strings.TrimRight(firstNonEmpty(os.Getenv("MINIO_ENDPOINT"), defaultMinioEndpoint), "/"),
|
||||
Bucket: firstNonEmpty(os.Getenv("MINIO_BUCKET"), defaultMinioBucket),
|
||||
AccessKey: os.Getenv("MINIO_ACCESS_KEY"),
|
||||
SecretKey: os.Getenv("MINIO_SECRET_KEY"),
|
||||
}
|
||||
}
|
||||
|
||||
// downloadFileEntry downloads f's bytes to dst. It transparently falls back to
|
||||
// the local MinIO store when the portal redirects the download to its stale AWS
|
||||
// S3 consumer.
|
||||
func (c *Client) downloadFileEntry(ctx context.Context, f *FileEntry, dst io.Writer) (int64, error) {
|
||||
if f == nil {
|
||||
return 0, fmt.Errorf("onlyoffice: download: nil file entry")
|
||||
}
|
||||
if f.ViewURL == nil || *f.ViewURL == "" {
|
||||
return 0, fmt.Errorf("onlyoffice: file has no viewUrl")
|
||||
}
|
||||
downloadURL := c.resolveAPIURL(*f.ViewURL)
|
||||
auth, err := c.authHeader()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, downloadURL, nil)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
req.Header.Set("Authorization", auth)
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode >= 400 {
|
||||
b, _ := io.ReadAll(io.LimitReader(resp.Body, 4096))
|
||||
finalURL := downloadURL
|
||||
if resp.Request != nil && resp.Request.URL != nil {
|
||||
finalURL = resp.Request.URL.String()
|
||||
}
|
||||
if isStaleS3Redirect(finalURL, b) {
|
||||
key, ok := minioObjectKeyFromURL(finalURL, loadMinioConfig().Bucket)
|
||||
if !ok {
|
||||
fileID := ""
|
||||
if f.ID != nil {
|
||||
fileID = f.ID.String()
|
||||
}
|
||||
key = minioObjectKey(fileID, FileFolderID(f))
|
||||
}
|
||||
n, merr := c.downloadFromMinio(ctx, key, dst)
|
||||
if merr == nil {
|
||||
return n, nil
|
||||
}
|
||||
return 0, fmt.Errorf("GET viewUrl: %d (stale S3) and minio fallback: %w", resp.StatusCode, merr)
|
||||
}
|
||||
return 0, fmt.Errorf("GET viewUrl: %d %s", resp.StatusCode, truncate(string(b), 400))
|
||||
}
|
||||
return io.Copy(dst, resp.Body)
|
||||
}
|
||||
|
||||
// downloadFromMinio streams objectKey from the configured MinIO bucket.
|
||||
func (c *Client) downloadFromMinio(ctx context.Context, objectKey string, dst io.Writer) (int64, error) {
|
||||
if objectKey == "" {
|
||||
return 0, fmt.Errorf("onlyoffice: minio fallback: empty object key")
|
||||
}
|
||||
cfg := loadMinioConfig()
|
||||
if cfg.AccessKey == "" || cfg.SecretKey == "" {
|
||||
return 0, fmt.Errorf("onlyoffice: minio fallback: MINIO_ACCESS_KEY/MINIO_SECRET_KEY not set")
|
||||
}
|
||||
base, err := url.Parse(cfg.Endpoint)
|
||||
if err != nil || base.Host == "" {
|
||||
return 0, fmt.Errorf("onlyoffice: minio fallback: bad MINIO_ENDPOINT %q", cfg.Endpoint)
|
||||
}
|
||||
u := *base
|
||||
u.Path = "/" + cfg.Bucket + "/" + objectKey
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, u.String(), nil)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if err := signMinioRequest(ctx, cfg, req); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("onlyoffice: minio fallback: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode >= 400 {
|
||||
b, _ := io.ReadAll(io.LimitReader(resp.Body, 512))
|
||||
return 0, fmt.Errorf("onlyoffice: minio fallback: %d %s", resp.StatusCode, truncate(string(b), 300))
|
||||
}
|
||||
return io.Copy(dst, resp.Body)
|
||||
}
|
||||
|
||||
// signMinioRequest signs req with AWS Signature V4 for the S3 service.
|
||||
func signMinioRequest(ctx context.Context, cfg minioConfig, req *http.Request) error {
|
||||
sum := sha256.Sum256(nil)
|
||||
creds := aws.Credentials{AccessKeyID: cfg.AccessKey, SecretAccessKey: cfg.SecretKey}
|
||||
if err := v4.NewSigner().SignHTTP(ctx, creds, req, hex.EncodeToString(sum[:]), "s3", minioRegion, time.Now()); err != nil {
|
||||
return fmt.Errorf("onlyoffice: minio fallback: sign: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
//go:build integration
|
||||
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestIntegrationMinioFallback downloads known stale-S3 files through the local
|
||||
// MinIO fallback. Requires ONLYOFFICE_URL/USER/PASS (as all integration tests),
|
||||
// MINIO_ACCESS_KEY/MINIO_SECRET_KEY and MINIO_TEST_FILE_IDS="3785,3859,3666";
|
||||
// skips when any of those are missing.
|
||||
func TestIntegrationMinioFallback(t *testing.T) {
|
||||
if os.Getenv("MINIO_ACCESS_KEY") == "" || os.Getenv("MINIO_SECRET_KEY") == "" {
|
||||
t.Skip("MINIO_ACCESS_KEY/MINIO_SECRET_KEY not set — skipping integration test")
|
||||
}
|
||||
raw := strings.TrimSpace(os.Getenv("MINIO_TEST_FILE_IDS"))
|
||||
if raw == "" {
|
||||
t.Skip("MINIO_TEST_FILE_IDS not set — skipping integration test")
|
||||
}
|
||||
c := liveClient(t)
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
|
||||
defer cancel()
|
||||
for _, id := range strings.Split(raw, ",") {
|
||||
id = strings.TrimSpace(id)
|
||||
if id == "" {
|
||||
continue
|
||||
}
|
||||
n, err := c.DownloadFile(ctx, id, io.Discard)
|
||||
if err != nil {
|
||||
t.Errorf("DownloadFile(%s): %v", id, err)
|
||||
continue
|
||||
}
|
||||
if n == 0 {
|
||||
t.Errorf("DownloadFile(%s): 0 bytes", id)
|
||||
} else {
|
||||
t.Logf("DownloadFile(%s): %d bytes", id, n)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,219 @@
|
||||
package onlyoffice
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestMinioObjectKey(t *testing.T) {
|
||||
cases := []struct {
|
||||
fileID string
|
||||
folderID string
|
||||
want string
|
||||
}{
|
||||
{"3785", "652", "00/00/01/files/folder_652/file_3785/v1/content.pdf"},
|
||||
{"1", "2", "00/00/01/files/folder_2/file_1/v1/content.pdf"},
|
||||
{"3666", "4000", "00/00/01/files/folder_4000/file_3666/v1/content.pdf"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
if got := minioObjectKey(tc.fileID, tc.folderID); got != tc.want {
|
||||
t.Errorf("minioObjectKey(%q, %q) = %q, want %q", tc.fileID, tc.folderID, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestMinioObjectKeyFromURL(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
url string
|
||||
bucket string
|
||||
want string
|
||||
ok bool
|
||||
}{
|
||||
{
|
||||
name: "path style drops bucket segment",
|
||||
url: "https://s3.us-east-1.amazonaws.com/office/00/00/01/files/folder_4000/file_3785/v1/content.pdf?AWSAccessKeyId=minio",
|
||||
bucket: "office",
|
||||
want: "00/00/01/files/folder_4000/file_3785/v1/content.pdf",
|
||||
ok: true,
|
||||
},
|
||||
{
|
||||
name: "doubled bucket segment (portal serviceurl includes bucket)",
|
||||
url: "https://s3.us-east-1.amazonaws.com/office/office/00/00/01/files/folder_4000/file_3785/v1/content.pdf?AWSAccessKeyId=minio",
|
||||
bucket: "office",
|
||||
want: "00/00/01/files/folder_4000/file_3785/v1/content.pdf",
|
||||
ok: true,
|
||||
},
|
||||
{
|
||||
name: "virtual host style keeps path",
|
||||
url: "https://office.s3.us-east-1.amazonaws.com/00/00/01/files/folder_4000/file_3785/v1/content.pdf",
|
||||
bucket: "office",
|
||||
want: "00/00/01/files/folder_4000/file_3785/v1/content.pdf",
|
||||
ok: true,
|
||||
},
|
||||
{
|
||||
name: "foreign first segment kept",
|
||||
url: "https://example.com/other/file_1/v1/content.pdf",
|
||||
bucket: "office",
|
||||
want: "other/file_1/v1/content.pdf",
|
||||
ok: true,
|
||||
},
|
||||
{name: "empty path", url: "https://example.com", bucket: "office", ok: false},
|
||||
{name: "traversal", url: "https://example.com/office/../etc/passwd", bucket: "office", ok: false},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got, ok := minioObjectKeyFromURL(tc.url, tc.bucket)
|
||||
if ok != tc.ok || got != tc.want {
|
||||
t.Errorf("minioObjectKeyFromURL(%q, %q) = (%q, %v), want (%q, %v)", tc.url, tc.bucket, got, ok, tc.want, tc.ok)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsStaleS3Redirect(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
url string
|
||||
body []byte
|
||||
want bool
|
||||
}{
|
||||
{
|
||||
name: "aws redirect with minio access key",
|
||||
url: "https://s3.us-east-1.amazonaws.com/office/x/file_1?AWSAccessKeyId=minio&Expires=1",
|
||||
want: true,
|
||||
},
|
||||
{
|
||||
name: "aws redirect with minio x-amz-credential",
|
||||
url: "https://office.s3.us-east-1.amazonaws.com/00/00/01/files/folder_1/file_1?X-Amz-Credential=minio%2F20260914",
|
||||
want: true,
|
||||
},
|
||||
{
|
||||
name: "invalid access key xml body",
|
||||
url: "https://portal.internal/download/1",
|
||||
body: []byte(`<?xml version="1.0"?><Error><Code>InvalidAccessKeyId</Code><AWSAccessKeyId>minio</AWSAccessKeyId></Error>`),
|
||||
want: true,
|
||||
},
|
||||
{
|
||||
name: "regular pdf from portal",
|
||||
url: "https://portal.internal/download/1",
|
||||
body: []byte("%PDF-1.7 data"),
|
||||
want: false,
|
||||
},
|
||||
{
|
||||
name: "aws redirect with foreign key",
|
||||
url: "https://s3.us-east-1.amazonaws.com/office/x?AWSAccessKeyId=other",
|
||||
want: false,
|
||||
},
|
||||
{
|
||||
name: "amazonaws in path but foreign host",
|
||||
url: "https://example.com/amazonaws.com/file?AWSAccessKeyId=minio",
|
||||
want: false,
|
||||
},
|
||||
{
|
||||
name: "empty",
|
||||
want: false,
|
||||
},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if got := isStaleS3Redirect(tc.url, tc.body); got != tc.want {
|
||||
t.Errorf("isStaleS3Redirect(%q, %q) = %v, want %v", tc.url, tc.body, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
const staleS3Body = `<?xml version="1.0" encoding="UTF-8"?>` +
|
||||
`<Error><Code>InvalidAccessKeyId</Code>` +
|
||||
`<Message>The AWS Access Key Id you provided does not exist in our records.</Message>` +
|
||||
`<AWSAccessKeyId>minio</AWSAccessKeyId></Error>`
|
||||
|
||||
func TestDownloadFileMinioFallback(t *testing.T) {
|
||||
const payload = "PDFDATA-3785"
|
||||
|
||||
portal := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
switch r.URL.Path {
|
||||
case "/api/2.0/files/file/3785.json":
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
io.WriteString(w, `{"response":{"id":3785,"title":"04.pdf","folderId":655,"viewUrl":"/download/3785"}}`)
|
||||
case "/download/3785":
|
||||
http.Redirect(w, r, "/office/00/00/01/files/folder_4000/file_3785/v1/content.pdf?AWSAccessKeyId=minio", http.StatusTemporaryRedirect)
|
||||
case "/office/00/00/01/files/folder_4000/file_3785/v1/content.pdf":
|
||||
w.WriteHeader(http.StatusForbidden)
|
||||
io.WriteString(w, staleS3Body)
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
}
|
||||
}))
|
||||
defer portal.Close()
|
||||
|
||||
var minioPath, minioAuth string
|
||||
minio := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
minioPath, minioAuth = r.URL.Path, r.Header.Get("Authorization")
|
||||
io.WriteString(w, payload)
|
||||
}))
|
||||
defer minio.Close()
|
||||
|
||||
t.Setenv("MINIO_ENDPOINT", minio.URL)
|
||||
t.Setenv("MINIO_BUCKET", "office")
|
||||
t.Setenv("MINIO_ACCESS_KEY", "testkey")
|
||||
t.Setenv("MINIO_SECRET_KEY", "testsecret")
|
||||
|
||||
c := &Client{
|
||||
client: portal.Client(),
|
||||
credentials: &Credentials{Url: portal.URL},
|
||||
token: &Token{Value: "Bearer test", Expires: Time(time.Now().Add(time.Hour))},
|
||||
}
|
||||
|
||||
var buf bytes.Buffer
|
||||
n, err := c.DownloadFile(context.Background(), "3785", &buf)
|
||||
if err != nil {
|
||||
t.Fatalf("DownloadFile: %v", err)
|
||||
}
|
||||
if n != int64(len(payload)) || buf.String() != payload {
|
||||
t.Fatalf("got %d bytes %q, want %d bytes %q", n, buf.String(), len(payload), payload)
|
||||
}
|
||||
if want := "/office/00/00/01/files/folder_4000/file_3785/v1/content.pdf"; minioPath != want {
|
||||
t.Errorf("minio path = %q, want %q", minioPath, want)
|
||||
}
|
||||
if !strings.HasPrefix(minioAuth, "AWS4-HMAC-SHA256") {
|
||||
t.Errorf("minio request not SigV4-signed; Authorization=%q", minioAuth)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDownloadFileMinioFallbackWithoutCreds(t *testing.T) {
|
||||
portal := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
switch r.URL.Path {
|
||||
case "/api/2.0/files/file/3785.json":
|
||||
io.WriteString(w, `{"response":{"id":3785,"folderId":652,"viewUrl":"/download/3785"}}`)
|
||||
default:
|
||||
w.WriteHeader(http.StatusForbidden)
|
||||
io.WriteString(w, staleS3Body)
|
||||
}
|
||||
}))
|
||||
defer portal.Close()
|
||||
|
||||
t.Setenv("MINIO_ACCESS_KEY", "")
|
||||
t.Setenv("MINIO_SECRET_KEY", "")
|
||||
|
||||
c := &Client{
|
||||
client: portal.Client(),
|
||||
credentials: &Credentials{Url: portal.URL},
|
||||
token: &Token{Value: "Bearer test", Expires: Time(time.Now().Add(time.Hour))},
|
||||
}
|
||||
|
||||
_, err := c.DownloadFile(context.Background(), "3785", io.Discard)
|
||||
if err == nil {
|
||||
t.Fatal("expected error without minio credentials")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "MINIO_ACCESS_KEY/MINIO_SECRET_KEY not set") {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user