Compare commits
2
Commits
feat/ci-eval
...
feat/ocr
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0ee4106b99 | ||
|
|
bae1494258 |
@@ -54,18 +54,44 @@ jobs:
|
|||||||
run: go test ./internal/brain/rank -count=1
|
run: go test ./internal/brain/rank -count=1
|
||||||
|
|
||||||
- name: facts/audit self (lexicon consistency, no network)
|
- name: facts/audit self (lexicon consistency, no network)
|
||||||
run: |
|
run: ./bin/facts/audit self
|
||||||
./bin/facts/audit self 2>/dev/null || echo "audit: not yet implemented; gate skipped"
|
|
||||||
|
|
||||||
- name: CGO via Zig (compile brain/search)
|
- name: CGO via Zig (compile brain/search + eval)
|
||||||
run: |
|
run: |
|
||||||
chmod +x bin/cgo/zig bin/cgo/zcc bin/cgo/zc++
|
chmod +x bin/cgo/zig bin/cgo/zcc bin/cgo/zc++
|
||||||
bin/cgo/zig go build -tags system_ladybug -o /tmp/brain-search ./bin/brain/search.go
|
bin/cgo/zig go build -tags system_ladybug -o /tmp/brain-search ./bin/brain/search.go
|
||||||
|
bin/cgo/zig go build -tags 'system_ladybug,brain_eval' -o /tmp/brain-eval ./bin/brain/eval.go
|
||||||
|
|
||||||
|
- uses: actions/cache@v4
|
||||||
|
with:
|
||||||
|
path: ~/.cache/huggingface
|
||||||
|
key: ${{ runner.os }}-hf-potion-multilingual-128M
|
||||||
|
|
||||||
|
- name: recall@5 SoT (Zig bin/brain/eval.go)
|
||||||
|
run: |
|
||||||
|
uv run python bin/kb/index --rebuild --json
|
||||||
|
KB_ROOT="$PWD" /tmp/brain-eval --json
|
||||||
|
|
||||||
|
ocr:
|
||||||
|
name: OCR (tesseract fixture)
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- uses: actions/setup-go@v5
|
||||||
|
with:
|
||||||
|
go-version-file: go.mod
|
||||||
|
- name: Install tesseract + poppler
|
||||||
|
run: |
|
||||||
|
sudo apt-get update
|
||||||
|
sudo apt-get install -y --no-install-recommends \
|
||||||
|
tesseract-ocr tesseract-ocr-eng tesseract-ocr-deu poppler-utils
|
||||||
|
- name: Go OCR tests (synthetic HELLO PNG)
|
||||||
|
run: go test ./internal/ocr -count=1
|
||||||
|
|
||||||
release:
|
release:
|
||||||
name: Release (semver)
|
name: Release (semver)
|
||||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||||
needs: test
|
needs: [test, ocr]
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
permissions:
|
permissions:
|
||||||
contents: write
|
contents: write
|
||||||
|
|||||||
@@ -42,7 +42,7 @@ skills/ in-project agent skills (vendored, no external links)
|
|||||||
bin/ self-describing tools bin/{subject}/{method}.go (shebang)
|
bin/ self-describing tools bin/{subject}/{method}.go (shebang)
|
||||||
bin/brain/ search.go serve.go index.go add.go get.go stats.go eval.go watch.go
|
bin/brain/ search.go serve.go index.go add.go get.go stats.go eval.go watch.go
|
||||||
bin/chats/ sync.go import.go facts.go apply.go; libs in internal/chats
|
bin/chats/ sync.go import.go facts.go apply.go; libs in internal/chats
|
||||||
bin/mail/ sync.go import.go (index_mail → brain/index.go)
|
bin/mail/ sync.go import.go ocr.go (index_mail → brain/index.go)
|
||||||
bin/markdown/ import.go (H2 leaf split; Python bin/md/import fallback)
|
bin/markdown/ import.go (H2 leaf split; Python bin/md/import fallback)
|
||||||
bin/postgres/ query.go (read-only YAML)
|
bin/postgres/ query.go (read-only YAML)
|
||||||
bin/git/ import.go (go-git history; Python shim execs it)
|
bin/git/ import.go (go-git history; Python shim execs it)
|
||||||
@@ -71,9 +71,9 @@ bin/brain/index.go --rebuild --with-facts --with-chats
|
|||||||
- `sync` (Go) downloads messages + attachments; Gmail uses paginated list +
|
- `sync` (Go) downloads messages + attachments; Gmail uses paginated list +
|
||||||
`body.attachmentId` (not partId) for attachments.
|
`body.attachmentId` (not partId) for attachments.
|
||||||
- `import` converts body + attachments to markdown. PDFs use poppler
|
- `import` converts body + attachments to markdown. PDFs use poppler
|
||||||
`pdftotext -layout` fast path (~15ms); textless/scanned PDFs fall back to
|
`pdftotext -layout` fast path (~15ms); textless/scanned PDFs use
|
||||||
docling (isolated subprocess — its native onnx can segfault the parent).
|
`pdftoppm` + tesseract `eng+deu` (`bin/mail/ocr.go`). Optional
|
||||||
Conversion never touches the brain DB (crash safety).
|
`OCR_ENGINE=paddle`. Conversion never touches the brain DB (crash safety).
|
||||||
- `index_mail` is a deprecation shim for `bin/brain/index.go --rebuild`. Bulk
|
- `index_mail` is a deprecation shim for `bin/brain/index.go --rebuild`. Bulk
|
||||||
rebuild still deletes `var/kb.lbug` and creates FTS/HNSW last. Single-leaf
|
rebuild still deletes `var/kb.lbug` and creates FTS/HNSW last. Single-leaf
|
||||||
write is `bin/brain/add.go` (safe while indexes exist; do not DROP INDEX).
|
write is `bin/brain/add.go` (safe while indexes exist; do not DROP INDEX).
|
||||||
@@ -101,6 +101,7 @@ bin/git/import.go [REPO] [--json] [--limit N] # go-git history → commit le
|
|||||||
bin/web/search.go "query" [--json] # SearXNG; throttled ≠ absence
|
bin/web/search.go "query" [--json] # SearXNG; throttled ≠ absence
|
||||||
bin/reasoner/bakeoff.go [--model ID] [--json] # D18 CPU tool-call bake-off
|
bin/reasoner/bakeoff.go [--model ID] [--json] # D18 CPU tool-call bake-off
|
||||||
bin/postgres/query.go --profile onlyoffice -c 'SELECT 1'
|
bin/postgres/query.go --profile onlyoffice -c 'SELECT 1'
|
||||||
|
bin/mail/ocr.go <image|pdf> # tesseract eng+deu (scans)
|
||||||
bin/md/tables # what the graph holds → YAML
|
bin/md/tables # what the graph holds → YAML
|
||||||
bin/brain/deduce "question" # thinking wrapper
|
bin/brain/deduce "question" # thinking wrapper
|
||||||
```
|
```
|
||||||
|
|||||||
@@ -16,6 +16,10 @@ ENV PYTHONUNBUFFERED=1 \
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
RUN id -u 2dph 2>/dev/null || useradd --create-home --uid 1001 2dph
|
RUN id -u 2dph 2>/dev/null || useradd --create-home --uid 1001 2dph
|
||||||
|
RUN apt-get update \
|
||||||
|
&& apt-get install -y --no-install-recommends \
|
||||||
|
poppler-utils tesseract-ocr tesseract-ocr-eng tesseract-ocr-deu \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
COPY requirements.lock.txt /tmp/requirements.lock.txt
|
COPY requirements.lock.txt /tmp/requirements.lock.txt
|
||||||
RUN python -m pip install --no-cache-dir -r /tmp/requirements.lock.txt \
|
RUN python -m pip install --no-cache-dir -r /tmp/requirements.lock.txt \
|
||||||
|
|||||||
@@ -4,8 +4,9 @@ A brain that loves facts and deduction. Evidence-first knowledge graph + hybrid
|
|||||||
RAG over the operational Brain/ops/eSlider stack. Built like Sherlock
|
RAG over the operational Brain/ops/eSlider stack. Built like Sherlock
|
||||||
Holmes: nothing is asserted unless it has proof.
|
Holmes: nothing is asserted unless it has proof.
|
||||||
|
|
||||||
Status: **in progress** — read path + MCP work; v1 goal is [epic #16](https://git.produktor.io/eSlider/2dph/issues/16)
|
Status: **v1 in** (epic [#16](https://git.produktor.io/eSlider/2dph/issues/16) closed).
|
||||||
(milestone [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12)).
|
v2 board: milestone [v2](https://git.produktor.io/eSlider/2dph/milestone/13) — OCR [#6](https://git.produktor.io/eSlider/2dph/issues/6),
|
||||||
|
[#29](https://git.produktor.io/eSlider/2dph/issues/29) OQ1, [#30](https://git.produktor.io/eSlider/2dph/issues/30) OQ3.
|
||||||
Gap: [docs/roadmap.md](docs/roadmap.md).
|
Gap: [docs/roadmap.md](docs/roadmap.md).
|
||||||
|
|
||||||
## What
|
## What
|
||||||
@@ -74,6 +75,7 @@ detective method: **a fact needs ≥2 independent sources or it is
|
|||||||
reasoner/bakeoff.go CPU tool-call bake-off (D18; OpenAI tools)
|
reasoner/bakeoff.go CPU tool-call bake-off (D18; OpenAI tools)
|
||||||
chats/sync.go import.go facts.go apply.go
|
chats/sync.go import.go facts.go apply.go
|
||||||
(libs in internal/chats; no chats index)
|
(libs in internal/chats; no chats index)
|
||||||
|
mail/ocr.go tesseract eng+deu (pdftoppm scans)
|
||||||
md/import (deprecated; bin/markdown/import.go)
|
md/import (deprecated; bin/markdown/import.go)
|
||||||
brain/extract brain/audit brain/deduce (thinking wrapper)
|
brain/extract brain/audit brain/deduce (thinking wrapper)
|
||||||
web/search (deprecated shim → web/search.go)
|
web/search (deprecated shim → web/search.go)
|
||||||
@@ -115,10 +117,13 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
|||||||
## Open questions (v2)
|
## Open questions (v2)
|
||||||
|
|
||||||
- OQ1: mutually-contradicting evidence — how to resolve (authority weighting,
|
- OQ1: mutually-contradicting evidence — how to resolve (authority weighting,
|
||||||
temporal freshness, audit adjudication). **v2**; does not block epic #16.
|
temporal freshness, audit adjudication). **v2**; [#29](https://git.produktor.io/eSlider/2dph/issues/29).
|
||||||
- OQ2: OCR — poppler `pdftotext` fast-path exists; scans still docling.
|
- OQ2: OCR — **in**. `pdftotext -layout` first; scans `pdftoppm` + tesseract
|
||||||
[#6](https://git.produktor.io/eSlider/2dph/issues/6) (v2, does not block #16).
|
`eng+deu` (`bin/mail/ocr.go`, `internal/ocr`). No gocv, no gosseract CGO
|
||||||
|
(D21 Zig owns Ladybug CGO). Optional `OCR_ENGINE=paddle` / compose profile
|
||||||
|
`ocr-paddle`. Docling left the default path. [#6](https://git.produktor.io/eSlider/2dph/issues/6).
|
||||||
- OQ3: optional duckdb-md layer for `SELECT … FORMAT MARKDOWN` export/write-back.
|
- OQ3: optional duckdb-md layer for `SELECT … FORMAT MARKDOWN` export/write-back.
|
||||||
|
[#30](https://git.produktor.io/eSlider/2dph/issues/30).
|
||||||
- OQ4: YAML-first storage for leafs — deferred: JSON is ~10x faster to
|
- OQ4: YAML-first storage for leafs — deferred: JSON is ~10x faster to
|
||||||
serialize and unambiguous; YAML only where humans edit files.
|
serialize and unambiguous; YAML only where humans edit files.
|
||||||
|
|
||||||
@@ -127,7 +132,8 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
|||||||
1. `bin/mail/sync.go` (Go, 8 workers) — paginated Gmail/OnlyOffice download.
|
1. `bin/mail/sync.go` (Go, 8 workers) — paginated Gmail/OnlyOffice download.
|
||||||
Gmail attachments key off `body.attachmentId`, not MIME `partId`.
|
Gmail attachments key off `body.attachmentId`, not MIME `partId`.
|
||||||
2. `bin/mail/import.go --from-raw` — message.json → message.md; PDFs via
|
2. `bin/mail/import.go --from-raw` — message.json → message.md; PDFs via
|
||||||
`pdftotext -layout` (~15ms) with docling subprocess fallback; ICS sidecars
|
`pdftotext -layout` (~15ms); textless/scanned PDFs `pdftoppm` + tesseract
|
||||||
|
`eng+deu`. ICS sidecars
|
||||||
Latin-1→UTF-8 normalized.
|
Latin-1→UTF-8 normalized.
|
||||||
3. `bin/brain/index.go --rebuild` — fresh rebuild (repo corpus + mail) because ladybug
|
3. `bin/brain/index.go --rebuild` — fresh rebuild (repo corpus + mail) because ladybug
|
||||||
corrupts its WAL on bulk-insert into an already-indexed DB. Conversion and
|
corrupts its WAL on bulk-insert into an already-indexed DB. Conversion and
|
||||||
@@ -144,8 +150,8 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
|||||||
2. `go test ./internal/brain/rank` (cgo-free ranking + flag parser)
|
2. `go test ./internal/brain/rank` (cgo-free ranking + flag parser)
|
||||||
3. python -m unittest discover -s bin/tools (includes published-docs SoT)
|
3. python -m unittest discover -s bin/tools (includes published-docs SoT)
|
||||||
4. `bin/facts/audit self` (lexicon internal consistency; `bin/facts/audit.go` is the D14 wrapper)
|
4. `bin/facts/audit self` (lexicon internal consistency; `bin/facts/audit.go` is the D14 wrapper)
|
||||||
5. `bin/kb/eval` (recall@5 ≥ 0.95). Local SoT is `bin/brain/eval.go` via Zig CGO.
|
5. `bin/brain/eval.go` via Zig (recall@5 ≥ 0.95). Python `bin/kb/eval` is an
|
||||||
CI SoT switch: [#19](https://git.produktor.io/eSlider/2dph/issues/19).
|
explicit fallback, not the CI SoT.
|
||||||
6. `bin/cgo/zig go build -tags system_ladybug` (compile search with zig cc; fetches pinned zig+libs).
|
6. `bin/cgo/zig go build -tags system_ladybug` (compile search with zig cc; fetches pinned zig+libs).
|
||||||
|
|
||||||
Feedback loop: every commit → PR → CI → green/gate → merge. Same discipline as
|
Feedback loop: every commit → PR → CI → green/gate → merge. Same discipline as
|
||||||
@@ -164,7 +170,7 @@ Feedback loop: every commit → PR → CI → green/gate → merge. Same discipl
|
|||||||
|
|
||||||
## Gap to v1 (epic #16)
|
## Gap to v1 (epic #16)
|
||||||
|
|
||||||
Remaining: CI eval SoT. Board:
|
Remaining: none for epic #16 (v1). Board:
|
||||||
[epic #16](https://git.produktor.io/eSlider/2dph/issues/16),
|
[epic #16](https://git.produktor.io/eSlider/2dph/issues/16),
|
||||||
milestone [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12).
|
milestone [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12).
|
||||||
Narrative: [docs/roadmap.md](docs/roadmap.md).
|
Narrative: [docs/roadmap.md](docs/roadmap.md).
|
||||||
@@ -175,6 +181,6 @@ Narrative: [docs/roadmap.md](docs/roadmap.md).
|
|||||||
| 2 | [#17](https://git.produktor.io/eSlider/2dph/issues/17) | **in** — `--hop N` walks `FROM_FILE` → `HAS_VERSION` → `AUTHORED` (max 3). |
|
| 2 | [#17](https://git.produktor.io/eSlider/2dph/issues/17) | **in** — `--hop N` walks `FROM_FILE` → `HAS_VERSION` → `AUTHORED` (max 3). |
|
||||||
| 3 | [#18](https://git.produktor.io/eSlider/2dph/issues/18) | **in** — `--with-facts` / `--facts-json` land `root=facts`; `--with-chats` indexes `var/chats/md`. WhatsApp sync is out of v1. |
|
| 3 | [#18](https://git.produktor.io/eSlider/2dph/issues/18) | **in** — `--with-facts` / `--facts-json` land `root=facts`; `--with-chats` indexes `var/chats/md`. WhatsApp sync is out of v1. |
|
||||||
| 4 | [#15](https://git.produktor.io/eSlider/2dph/issues/15) | **in** — lever/loop documented (`search` → `get` → `audit`). |
|
| 4 | [#15](https://git.produktor.io/eSlider/2dph/issues/15) | **in** — lever/loop documented (`search` → `get` → `audit`). |
|
||||||
| 5 | [#19](https://git.produktor.io/eSlider/2dph/issues/19) | GitHub CI recall still runs Python `bin/kb/eval`. |
|
| 5 | [#19](https://git.produktor.io/eSlider/2dph/issues/19) | **in** — CI recall SoT is `bin/brain/eval.go` via Zig. Python `bin/kb/eval` stays as an explicit fallback. |
|
||||||
|
|
||||||
Does **not** block epic close: [#6](https://git.produktor.io/eSlider/2dph/issues/6) OCR, OQ1, OQ3, OQ4.
|
Does **not** block epic close: OQ1 [#29](https://git.produktor.io/eSlider/2dph/issues/29), OQ3 [#30](https://git.produktor.io/eSlider/2dph/issues/30), OQ4. OCR [#6](https://git.produktor.io/eSlider/2dph/issues/6) is **in**.
|
||||||
@@ -169,6 +169,9 @@ docker compose up brain-watch # auto re-index on change
|
|||||||
|
|
||||||
## Related
|
## Related
|
||||||
|
|
||||||
|
eSlider DevOps engineer practice: ops, OnlyOffice, and mail feed the facts
|
||||||
|
root through `bin/facts/extract` (two-source pairing).
|
||||||
|
|
||||||
- [go-second-brain](https://github.com/eSlider/go-second-brain) — the earlier
|
- [go-second-brain](https://github.com/eSlider/go-second-brain) — the earlier
|
||||||
Neo4j + Qdrant + Matrix RAG brain
|
Neo4j + Qdrant + Matrix RAG brain
|
||||||
- [agent-skills](https://github.com/eSlider/agent-skills) — upstream
|
- [agent-skills](https://github.com/eSlider/agent-skills) — upstream
|
||||||
|
|||||||
+11
-77
@@ -7,7 +7,7 @@
|
|||||||
bin/mail/import --since 2026-01-01 only messages after a date
|
bin/mail/import --since 2026-01-01 only messages after a date
|
||||||
bin/mail/import --limit 50 cap messages per run
|
bin/mail/import --limit 50 cap messages per run
|
||||||
bin/mail/import --no-attachments body only, skip attachment conversion
|
bin/mail/import --no-attachments body only, skip attachment conversion
|
||||||
bin/mail/import --ocr OCR scanned PDFs/images via docling
|
bin/mail/import --ocr OCR images (PDFs OCR when textless)
|
||||||
bin/mail/import --dry-run list messages without writing anything
|
bin/mail/import --dry-run list messages without writing anything
|
||||||
|
|
||||||
Writes one directory per message: var/mail/{folder}/{message_id}/
|
Writes one directory per message: var/mail/{folder}/{message_id}/
|
||||||
@@ -16,10 +16,11 @@ Writes one directory per message: var/mail/{folder}/{message_id}/
|
|||||||
attachments/*.md converted attachment content
|
attachments/*.md converted attachment content
|
||||||
|
|
||||||
Indexing is a separate step (`bin/brain/index.go --rebuild`): conversion can
|
Indexing is a separate step (`bin/brain/index.go --rebuild`): conversion can
|
||||||
crash in native docling and must not leave the brain DB mid-transaction.
|
crash and must not leave the brain DB mid-transaction.
|
||||||
|
|
||||||
Requires ONLYOFFICE_URL/USER/PASS in .env (or env). Idempotent: a message
|
Requires ONLYOFFICE_URL/USER/PASS in .env (or env) except `--from-raw`.
|
||||||
already present (message.md exists) is skipped unless --force.
|
Idempotent: a message already present (message.md exists) is skipped unless
|
||||||
|
--force.
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -41,10 +42,10 @@ from mailconv import ( # noqa: E402
|
|||||||
IMAGE_SUFFIXES,
|
IMAGE_SUFFIXES,
|
||||||
LEGACY_OFFICE_SUFFIXES,
|
LEGACY_OFFICE_SUFFIXES,
|
||||||
TEXT_SUFFIXES,
|
TEXT_SUFFIXES,
|
||||||
|
convert_pdf,
|
||||||
html_to_markdown,
|
html_to_markdown,
|
||||||
is_convertible,
|
|
||||||
normalize_markdown,
|
normalize_markdown,
|
||||||
subject_to_filename,
|
ocr_image,
|
||||||
zip_extract_safe,
|
zip_extract_safe,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -147,9 +148,9 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
return f"\n<!-- conversion failed: {e} -->\n"
|
return f"\n<!-- conversion failed: {e} -->\n"
|
||||||
if suffix == ".pdf":
|
if suffix == ".pdf":
|
||||||
return _convert_pdf(path, ocr)
|
return convert_pdf(path, ocr)
|
||||||
if suffix in IMAGE_SUFFIXES and ocr:
|
if suffix in IMAGE_SUFFIXES and ocr:
|
||||||
return _convert_pdf(path, ocr)
|
return ocr_image(path) or "\n<!-- ocr unavailable -->\n"
|
||||||
if suffix in LEGACY_OFFICE_SUFFIXES:
|
if suffix in LEGACY_OFFICE_SUFFIXES:
|
||||||
return _convert_legacy(path)
|
return _convert_legacy(path)
|
||||||
if suffix in ARCHIVE_SUFFIXES:
|
if suffix in ARCHIVE_SUFFIXES:
|
||||||
@@ -157,67 +158,6 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def _convert_pdf(path: Path, ocr: bool) -> str:
|
|
||||||
"""Convert one PDF to markdown.
|
|
||||||
|
|
||||||
Fast path: poppler's pdftotext (-layout) extracts exact text from
|
|
||||||
born-digital PDFs in ~15ms vs docling's 1-3s. Only textless PDFs (scanned
|
|
||||||
pages, layout-heavy) fall back to docling, which runs isolated in a
|
|
||||||
subprocess because its native onnx/RT-DETR has segfaulted the main process.
|
|
||||||
"""
|
|
||||||
text = _pdf_fast_text(path)
|
|
||||||
if ocr or text is None or not text.strip():
|
|
||||||
return _convert_pdf_docling(path, ocr)
|
|
||||||
return normalize_markdown(text)
|
|
||||||
|
|
||||||
|
|
||||||
def _pdf_fast_text(path: Path) -> str | None:
|
|
||||||
"""pdftotext -layout; None when poppler is unavailable (or the PDF has no text layer)."""
|
|
||||||
try:
|
|
||||||
proc = subprocess.run(
|
|
||||||
["pdftotext", "-layout", str(path), "-"],
|
|
||||||
capture_output=True, timeout=60)
|
|
||||||
except (OSError, subprocess.TimeoutExpired):
|
|
||||||
return None
|
|
||||||
if proc.returncode != 0:
|
|
||||||
return None
|
|
||||||
return proc.stdout.decode("utf-8", errors="replace")
|
|
||||||
|
|
||||||
|
|
||||||
def _convert_pdf_docling(path: Path, ocr: bool) -> str:
|
|
||||||
try:
|
|
||||||
proc = subprocess.run(
|
|
||||||
[sys.executable, os.path.abspath(__file__), "--pdf-worker", str(path),
|
|
||||||
"--ocr" if ocr else "--no-ocr"],
|
|
||||||
capture_output=True, text=True, timeout=600)
|
|
||||||
except subprocess.TimeoutExpired:
|
|
||||||
return "\n<!-- pdf conversion timed out -->\n"
|
|
||||||
if proc.returncode != 0:
|
|
||||||
tail = proc.stderr.strip().splitlines()[-3:]
|
|
||||||
return f"\n<!-- pdf conversion failed: {proc.returncode}: {' | '.join(tail)} -->\n"
|
|
||||||
return proc.stdout
|
|
||||||
|
|
||||||
|
|
||||||
def _pdf_worker(path: Path, ocr: bool) -> None:
|
|
||||||
"""docling worker entry: prints converted markdown on stdout, exits non-zero on error."""
|
|
||||||
try:
|
|
||||||
from docling.document_converter import DocumentConverter, PdfFormatOption
|
|
||||||
from docling.datamodel.pipeline_options import PdfPipelineOptions
|
|
||||||
opts = PdfPipelineOptions()
|
|
||||||
opts.do_ocr = bool(ocr)
|
|
||||||
opts.do_table_structure = True
|
|
||||||
conv = DocumentConverter(format_options={"pdf": PdfFormatOption(pipeline_options=opts)})
|
|
||||||
res = conv.convert(str(path))
|
|
||||||
sys.stdout.write(normalize_markdown(res.document.export_to_markdown()))
|
|
||||||
sys.exit(0)
|
|
||||||
except Exception as e:
|
|
||||||
# errors/stacktraces to stderr; the caller only reports a one-liner
|
|
||||||
print(f"pdf-worker: {e}", file=sys.stderr)
|
|
||||||
import traceback
|
|
||||||
traceback.print_exc(file=sys.stderr)
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
|
|
||||||
def _convert_legacy(path: Path) -> str:
|
def _convert_legacy(path: Path) -> str:
|
||||||
"""Legacy .doc/.xls/.ppt -> md via pandoc (installed) or a stub."""
|
"""Legacy .doc/.xls/.ppt -> md via pandoc (installed) or a stub."""
|
||||||
try:
|
try:
|
||||||
@@ -356,19 +296,12 @@ def main(argv: list[str]) -> int:
|
|||||||
p.add_argument("--from-raw", default="",
|
p.add_argument("--from-raw", default="",
|
||||||
help="convert Go-synced dirs (var/mail/<folder>/<id>/message.json) to markdown")
|
help="convert Go-synced dirs (var/mail/<folder>/<id>/message.json) to markdown")
|
||||||
p.add_argument("--no-attachments", action="store_true", help="skip attachment download+convert")
|
p.add_argument("--no-attachments", action="store_true", help="skip attachment download+convert")
|
||||||
p.add_argument("--ocr", action="store_true", help="OCR scanned PDFs/images via docling")
|
p.add_argument("--ocr", action="store_true", help="OCR images (PDFs OCR when textless)")
|
||||||
p.add_argument("--force", action="store_true", help="re-import even if message.md exists")
|
p.add_argument("--force", action="store_true", help="re-import even if message.md exists")
|
||||||
p.add_argument("--dry-run", action="store_true", help="list messages, write nothing")
|
p.add_argument("--dry-run", action="store_true", help="list messages, write nothing")
|
||||||
p.add_argument("--json", action="store_true")
|
p.add_argument("--json", action="store_true")
|
||||||
p.add_argument("--pdf-worker", default="", help=argparse.SUPPRESS)
|
|
||||||
p.add_argument("--no-ocr", action="store_true", help=argparse.SUPPRESS)
|
|
||||||
a = p.parse_args(argv)
|
a = p.parse_args(argv)
|
||||||
|
|
||||||
if a.pdf_worker:
|
|
||||||
_pdf_worker(Path(a.pdf_worker), ocr=not a.no_ocr)
|
|
||||||
return 0
|
|
||||||
|
|
||||||
conf = load_env()
|
|
||||||
fid = folder_id(a.folder)
|
fid = folder_id(a.folder)
|
||||||
out_root = ROOT / "var" / "mail"
|
out_root = ROOT / "var" / "mail"
|
||||||
summary: list[dict] = []
|
summary: list[dict] = []
|
||||||
@@ -394,6 +327,7 @@ def main(argv: list[str]) -> int:
|
|||||||
target_dir=msg_dir.parent))
|
target_dir=msg_dir.parent))
|
||||||
summary.append(entry)
|
summary.append(entry)
|
||||||
else:
|
else:
|
||||||
|
conf = load_env()
|
||||||
OOCLIENT = OOClient(conf)
|
OOCLIENT = OOClient(conf)
|
||||||
if a.id:
|
if a.id:
|
||||||
messages = [{"id": i} for i in a.id]
|
messages = [{"id": i} for i in a.id]
|
||||||
|
|||||||
Executable
+48
@@ -0,0 +1,48 @@
|
|||||||
|
//usr/bin/env go run -tags=mail_ocr "$0" "$@"; exit
|
||||||
|
//go:build mail_ocr
|
||||||
|
//
|
||||||
|
// bin/mail/ocr.go - OCR an image or scanned PDF (tesseract eng+deu).
|
||||||
|
//
|
||||||
|
// ./bin/mail/ocr.go scan.png
|
||||||
|
// ./bin/mail/ocr.go scan.pdf
|
||||||
|
// OCR_ENGINE=paddle ./bin/mail/ocr.go scan.png
|
||||||
|
//
|
||||||
|
// PDFs try pdftotext -layout first; empty text layer uses pdftoppm + tesseract.
|
||||||
|
// No gocv. Tesseract CGO bindings are not used (D21 Zig owns Ladybug CGO).
|
||||||
|
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"github.com/eSlider/2dph/internal/ocr"
|
||||||
|
)
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
os.Exit(run(os.Args[1:]))
|
||||||
|
}
|
||||||
|
|
||||||
|
func run(args []string) int {
|
||||||
|
if len(args) != 1 || strings.HasPrefix(args[0], "-") {
|
||||||
|
fmt.Fprintln(os.Stderr, `usage: bin/mail/ocr.go <image|pdf>`)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
path := args[0]
|
||||||
|
var (
|
||||||
|
text string
|
||||||
|
err error
|
||||||
|
)
|
||||||
|
if strings.HasSuffix(strings.ToLower(path), ".pdf") {
|
||||||
|
text, err = ocr.PDFFile(path)
|
||||||
|
} else {
|
||||||
|
text, err = ocr.ImageFile(path)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "mail/ocr: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Println(text)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
+84
-1
@@ -7,7 +7,10 @@ offline against fixtures.
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import html
|
import html
|
||||||
|
import os
|
||||||
import re
|
import re
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
import zipfile
|
import zipfile
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -18,9 +21,10 @@ OFFICE_SUFFIXES = {".docx", ".pptx", ".xlsx", ".html", ".htm", ".epub", ".eml",
|
|||||||
PDF_SUFFIXES = {".pdf"}
|
PDF_SUFFIXES = {".pdf"}
|
||||||
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".tiff", ".tif", ".webp"}
|
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".tiff", ".tif", ".webp"}
|
||||||
ARCHIVE_SUFFIXES = {".zip"}
|
ARCHIVE_SUFFIXES = {".zip"}
|
||||||
# Legacy binary Office (doc/xls/ppt) — markitdown/docling skip them; we try
|
# Legacy binary Office (doc/xls/ppt) — markitdown skip them; we try
|
||||||
# pandoc first, else leave a stub.
|
# pandoc first, else leave a stub.
|
||||||
LEGACY_OFFICE_SUFFIXES = {".doc", ".xls", ".ppt"}
|
LEGACY_OFFICE_SUFFIXES = {".doc", ".xls", ".ppt"}
|
||||||
|
TESS_LANG = "eng+deu"
|
||||||
|
|
||||||
CONVERTIBLE_SUFFIXES = (
|
CONVERTIBLE_SUFFIXES = (
|
||||||
TEXT_SUFFIXES | OFFICE_SUFFIXES | PDF_SUFFIXES | IMAGE_SUFFIXES | ARCHIVE_SUFFIXES | LEGACY_OFFICE_SUFFIXES
|
TEXT_SUFFIXES | OFFICE_SUFFIXES | PDF_SUFFIXES | IMAGE_SUFFIXES | ARCHIVE_SUFFIXES | LEGACY_OFFICE_SUFFIXES
|
||||||
@@ -146,3 +150,82 @@ def zip_extract_safe(zip_path: Path, dest: Path) -> list[Path]:
|
|||||||
|
|
||||||
def is_convertible(suffix: str) -> bool:
|
def is_convertible(suffix: str) -> bool:
|
||||||
return suffix.lower() in CONVERTIBLE_SUFFIXES
|
return suffix.lower() in CONVERTIBLE_SUFFIXES
|
||||||
|
|
||||||
|
|
||||||
|
def convert_pdf(path: Path, ocr: bool = False) -> str:
|
||||||
|
"""pdftotext -layout first; empty text layer → pdftoppm + tesseract.
|
||||||
|
|
||||||
|
`ocr` is unused for born-digital PDFs (text layer wins). Scans OCR
|
||||||
|
automatically. This path never execs an ONNX document converter.
|
||||||
|
"""
|
||||||
|
del ocr # scans OCR when the text layer is empty; flag is for images
|
||||||
|
text = pdf_fast_text(path)
|
||||||
|
if text and text.strip():
|
||||||
|
return normalize_markdown(text)
|
||||||
|
scanned = ocr_pdf(path)
|
||||||
|
if scanned and scanned.strip():
|
||||||
|
return normalize_markdown(scanned)
|
||||||
|
if text:
|
||||||
|
return normalize_markdown(text)
|
||||||
|
return "\n<!-- pdf has no text layer (ocr unavailable) -->\n"
|
||||||
|
|
||||||
|
|
||||||
|
def pdf_fast_text(path: Path) -> str | None:
|
||||||
|
"""pdftotext -layout; None when poppler is missing or the command fails."""
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
["pdftotext", "-layout", str(path), "-"],
|
||||||
|
capture_output=True, timeout=60)
|
||||||
|
except (OSError, subprocess.TimeoutExpired):
|
||||||
|
return None
|
||||||
|
if proc.returncode != 0:
|
||||||
|
return None
|
||||||
|
return proc.stdout.decode("utf-8", errors="replace")
|
||||||
|
|
||||||
|
|
||||||
|
def ocr_pdf(path: Path) -> str:
|
||||||
|
"""Rasterize with pdftoppm and OCR each page (tesseract or paddle)."""
|
||||||
|
try:
|
||||||
|
with tempfile.TemporaryDirectory(prefix="2dph-ocr-") as tmp:
|
||||||
|
prefix = str(Path(tmp) / "page")
|
||||||
|
proc = subprocess.run(
|
||||||
|
["pdftoppm", "-png", "-r", "200", str(path), prefix],
|
||||||
|
capture_output=True, timeout=120)
|
||||||
|
if proc.returncode != 0:
|
||||||
|
return ""
|
||||||
|
pages = sorted(Path(tmp).glob("page*.png"))
|
||||||
|
parts = [ocr_image(p) for p in pages]
|
||||||
|
return "\n\n".join(p for p in parts if p and p.strip())
|
||||||
|
except (OSError, subprocess.TimeoutExpired):
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def ocr_image(path: Path) -> str:
|
||||||
|
engine = os.environ.get("OCR_ENGINE", "tesseract")
|
||||||
|
if engine == "paddle":
|
||||||
|
return _ocr_paddle(path)
|
||||||
|
return _ocr_tesseract(path)
|
||||||
|
|
||||||
|
|
||||||
|
def _ocr_tesseract(path: Path) -> str:
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
["tesseract", str(path), "stdout", "-l", TESS_LANG, "--psm", "6"],
|
||||||
|
capture_output=True, timeout=120)
|
||||||
|
except (OSError, subprocess.TimeoutExpired):
|
||||||
|
return ""
|
||||||
|
if proc.returncode != 0:
|
||||||
|
return ""
|
||||||
|
return proc.stdout.decode("utf-8", errors="replace").strip()
|
||||||
|
|
||||||
|
|
||||||
|
def _ocr_paddle(path: Path) -> str:
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
["paddleocr", "ocr", "-i", str(path)],
|
||||||
|
capture_output=True, timeout=180)
|
||||||
|
except (OSError, subprocess.TimeoutExpired):
|
||||||
|
return ""
|
||||||
|
if proc.returncode != 0:
|
||||||
|
return ""
|
||||||
|
return proc.stdout.decode("utf-8", errors="replace").strip()
|
||||||
|
|||||||
@@ -136,6 +136,33 @@ class BinLayoutTest(unittest.TestCase):
|
|||||||
"index_mail must point at bin/brain/index.go",
|
"index_mail must point at bin/brain/index.go",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def test_mail_ocr_is_tesseract_not_docling(self) -> None:
|
||||||
|
self._assert_shebang("bin/mail/ocr.go")
|
||||||
|
ocr = (ROOT / "bin" / "mail" / "ocr.go").read_text()
|
||||||
|
self.assertIn("internal/ocr", ocr)
|
||||||
|
self.assertIn("mail_ocr", ocr)
|
||||||
|
self.assertNotIn("github.com/otiai10/gosseract", ocr)
|
||||||
|
py = (ROOT / "bin" / "mail" / "import").read_text()
|
||||||
|
self.assertNotIn("from docling", py)
|
||||||
|
self.assertNotIn("import docling", py)
|
||||||
|
self.assertIn("convert_pdf", py)
|
||||||
|
conv = (ROOT / "bin" / "tools" / "mailconv.py").read_text()
|
||||||
|
self.assertIn("pdftotext", conv)
|
||||||
|
self.assertIn("pdftoppm", conv)
|
||||||
|
self.assertIn("tesseract", conv)
|
||||||
|
self.assertIn("eng+deu", conv)
|
||||||
|
self.assertNotIn("from docling", conv)
|
||||||
|
self.assertNotIn("import docling", conv)
|
||||||
|
self.assertNotIn("gocv", conv.lower())
|
||||||
|
proj = (ROOT / "pyproject.toml").read_text()
|
||||||
|
self.assertNotIn("docling", proj)
|
||||||
|
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
|
||||||
|
self.assertIn("tesseract-ocr", ci)
|
||||||
|
self.assertIn("./internal/ocr", ci)
|
||||||
|
compose = (ROOT / "compose.yaml").read_text()
|
||||||
|
self.assertIn("ocr-paddle", compose)
|
||||||
|
self.assertIn("OCR_ENGINE", compose)
|
||||||
|
|
||||||
def test_markdown_import_is_go_not_python_exec(self) -> None:
|
def test_markdown_import_is_go_not_python_exec(self) -> None:
|
||||||
self._assert_shebang("bin/markdown/import.go")
|
self._assert_shebang("bin/markdown/import.go")
|
||||||
text = (ROOT / "bin" / "markdown" / "import.go").read_text()
|
text = (ROOT / "bin" / "markdown" / "import.go").read_text()
|
||||||
@@ -207,3 +234,25 @@ class BinLayoutTest(unittest.TestCase):
|
|||||||
search = (ROOT / "bin" / "kb" / "search").read_text()
|
search = (ROOT / "bin" / "kb" / "search").read_text()
|
||||||
self.assertIn("bin/cgo/zig", search)
|
self.assertIn("bin/cgo/zig", search)
|
||||||
self.assertNotIn("command -v gcc", search)
|
self.assertNotIn("command -v gcc", search)
|
||||||
|
|
||||||
|
def test_ci_recall_sot_is_zig_brain_eval(self) -> None:
|
||||||
|
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
|
||||||
|
self.assertIn("bin/brain/eval.go", ci)
|
||||||
|
self.assertIn("system_ladybug,brain_eval", ci)
|
||||||
|
self.assertIn("/tmp/brain-eval", ci)
|
||||||
|
self.assertIn("KB_ROOT", ci)
|
||||||
|
self.assertNotIn("bin/kb/eval", ci)
|
||||||
|
self.assertNotIn("gate skipped", ci)
|
||||||
|
self.assertIn("./bin/facts/audit self", ci)
|
||||||
|
|
||||||
|
def test_eval_fragments_live_in_default_corpus(self) -> None:
|
||||||
|
"""CI --rebuild indexes README/PLAN/docs/skills; fragments must be there."""
|
||||||
|
corpus = []
|
||||||
|
for rel in ("README.md", "PLAN.md", "AGENTS.md"):
|
||||||
|
corpus.append((ROOT / rel).read_text())
|
||||||
|
for d in ("docs", "skills"):
|
||||||
|
for p in (ROOT / d).rglob("*.md"):
|
||||||
|
corpus.append(p.read_text())
|
||||||
|
blob = "\n".join(corpus)
|
||||||
|
for frag in ("BM25", "DevOps", "LadybugDB"):
|
||||||
|
self.assertIn(frag, blob, f"{frag} must appear in default index corpus")
|
||||||
|
|||||||
@@ -8,10 +8,13 @@ from pathlib import Path
|
|||||||
sys.path.insert(0, os.path.dirname(__file__))
|
sys.path.insert(0, os.path.dirname(__file__))
|
||||||
|
|
||||||
from mailconv import ( # noqa: E402
|
from mailconv import ( # noqa: E402
|
||||||
|
TESS_LANG,
|
||||||
clean_email_address,
|
clean_email_address,
|
||||||
|
convert_pdf,
|
||||||
html_to_markdown,
|
html_to_markdown,
|
||||||
is_convertible,
|
is_convertible,
|
||||||
normalize_markdown,
|
normalize_markdown,
|
||||||
|
ocr_image,
|
||||||
split_zip_members,
|
split_zip_members,
|
||||||
subject_to_filename,
|
subject_to_filename,
|
||||||
zip_extract_safe,
|
zip_extract_safe,
|
||||||
@@ -100,6 +103,92 @@ class TestMailConv(unittest.TestCase):
|
|||||||
self.assertFalse(is_convertible(".exe"))
|
self.assertFalse(is_convertible(".exe"))
|
||||||
self.assertFalse(is_convertible(".unknown"))
|
self.assertFalse(is_convertible(".unknown"))
|
||||||
|
|
||||||
|
def test_convert_pdf_prefers_pdftotext(self):
|
||||||
|
import mailconv as mc
|
||||||
|
|
||||||
|
calls: list[list[str]] = []
|
||||||
|
|
||||||
|
def fake_run(cmd, **kwargs):
|
||||||
|
calls.append(list(cmd))
|
||||||
|
|
||||||
|
class P:
|
||||||
|
returncode = 0
|
||||||
|
stdout = b"Invoice BM25 layout"
|
||||||
|
stderr = b""
|
||||||
|
|
||||||
|
return P()
|
||||||
|
|
||||||
|
self._patch_run(mc, fake_run)
|
||||||
|
out = convert_pdf(Path(self._tmp("born.pdf")))
|
||||||
|
self.assertIn("BM25", out)
|
||||||
|
self.assertEqual(calls[0][:2], ["pdftotext", "-layout"])
|
||||||
|
self.assertFalse(any(c[0] == "tesseract" for c in calls))
|
||||||
|
self.assertFalse(any(c[0] == "pdftoppm" for c in calls))
|
||||||
|
|
||||||
|
def test_convert_pdf_empty_layer_uses_pdftoppm_tesseract(self):
|
||||||
|
import mailconv as mc
|
||||||
|
|
||||||
|
calls: list[list[str]] = []
|
||||||
|
|
||||||
|
def fake_run(cmd, **kwargs):
|
||||||
|
calls.append(list(cmd))
|
||||||
|
|
||||||
|
class P:
|
||||||
|
returncode = 0
|
||||||
|
stdout = b""
|
||||||
|
stderr = b""
|
||||||
|
|
||||||
|
if cmd[0] == "pdftotext":
|
||||||
|
P.stdout = b" \n"
|
||||||
|
return P()
|
||||||
|
if cmd[0] == "pdftoppm":
|
||||||
|
prefix = Path(cmd[-1])
|
||||||
|
(prefix.parent / "page-1.png").write_bytes(b"fake")
|
||||||
|
return P()
|
||||||
|
if cmd[0] == "tesseract":
|
||||||
|
P.stdout = b"scanned HELLO"
|
||||||
|
return P()
|
||||||
|
return P()
|
||||||
|
|
||||||
|
self._patch_run(mc, fake_run)
|
||||||
|
out = convert_pdf(Path(self._tmp("scan.pdf")))
|
||||||
|
self.assertIn("HELLO", out)
|
||||||
|
bins = [c[0] for c in calls]
|
||||||
|
self.assertIn("pdftotext", bins)
|
||||||
|
self.assertIn("pdftoppm", bins)
|
||||||
|
self.assertIn("tesseract", bins)
|
||||||
|
tess = next(c for c in calls if c[0] == "tesseract")
|
||||||
|
self.assertIn(TESS_LANG, tess)
|
||||||
|
self.assertNotIn("docling", " ".join(bins))
|
||||||
|
|
||||||
|
def test_ocr_image_paddle_engine(self):
|
||||||
|
import mailconv as mc
|
||||||
|
|
||||||
|
calls: list[list[str]] = []
|
||||||
|
|
||||||
|
def fake_run(cmd, **kwargs):
|
||||||
|
calls.append(list(cmd))
|
||||||
|
|
||||||
|
class P:
|
||||||
|
returncode = 0
|
||||||
|
stdout = b"paddle text"
|
||||||
|
stderr = b""
|
||||||
|
|
||||||
|
return P()
|
||||||
|
|
||||||
|
self._patch_run(mc, fake_run)
|
||||||
|
os.environ["OCR_ENGINE"] = "paddle"
|
||||||
|
try:
|
||||||
|
out = ocr_image(Path(self._tmp("x.png")))
|
||||||
|
finally:
|
||||||
|
os.environ.pop("OCR_ENGINE", None)
|
||||||
|
self.assertEqual(out, "paddle text")
|
||||||
|
self.assertEqual(calls[0][:2], ["paddleocr", "ocr"])
|
||||||
|
|
||||||
|
def _patch_run(self, mod, fn) -> None:
|
||||||
|
self.addCleanup(setattr, mod.subprocess, "run", mod.subprocess.run)
|
||||||
|
mod.subprocess.run = fn
|
||||||
|
|
||||||
def _mk_zip(self, members):
|
def _mk_zip(self, members):
|
||||||
zpath = Path(self._tmp("arc.zip"))
|
zpath = Path(self._tmp("arc.zip"))
|
||||||
with zipfile.ZipFile(zpath, "w") as zf:
|
with zipfile.ZipFile(zpath, "w") as zf:
|
||||||
|
|||||||
@@ -5,6 +5,7 @@
|
|||||||
# docker compose --profile picoclaw up brain-mcp
|
# docker compose --profile picoclaw up brain-mcp
|
||||||
# docker compose --profile reasoner up -d reasoner # CPU Ollama :11435
|
# docker compose --profile reasoner up -d reasoner # CPU Ollama :11435
|
||||||
# docker compose --profile searxng up -d
|
# docker compose --profile searxng up -d
|
||||||
|
# OCR_ENGINE=paddle docker compose --profile ocr-paddle run --rm ocr-paddle
|
||||||
#
|
#
|
||||||
# Secrets never baked in: search.env + db-profiles.yml from ~/.config/brain.
|
# Secrets never baked in: search.env + db-profiles.yml from ~/.config/brain.
|
||||||
|
|
||||||
@@ -131,6 +132,15 @@ services:
|
|||||||
- reasoner-ollama:/root/.ollama
|
- reasoner-ollama:/root/.ollama
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
|
||||||
|
# Optional PP-OCRv5 (not default). Default OCR is tesseract eng+deu.
|
||||||
|
# OCR_ENGINE=paddle docker compose --profile ocr-paddle run --rm ocr-paddle
|
||||||
|
ocr-paddle:
|
||||||
|
profiles: ["ocr-paddle"]
|
||||||
|
image: python:3.12-slim
|
||||||
|
environment:
|
||||||
|
OCR_ENGINE: paddle
|
||||||
|
command: ["python", "-c", "print('OCR_ENGINE=paddle; install paddleocr on PATH')"]
|
||||||
|
|
||||||
volumes:
|
volumes:
|
||||||
kb-model:
|
kb-model:
|
||||||
kb-var:
|
kb-var:
|
||||||
|
|||||||
+17
-7
@@ -26,11 +26,25 @@ Compose `api` (no CPython) / `index` (Python write). Issues #1–#5, #7–#13.
|
|||||||
[#15](https://git.produktor.io/eSlider/2dph/issues/15) lever/loop.
|
[#15](https://git.produktor.io/eSlider/2dph/issues/15) lever/loop.
|
||||||
[#14](https://git.produktor.io/eSlider/2dph/issues/14) `bin/brain/add.go` /
|
[#14](https://git.produktor.io/eSlider/2dph/issues/14) `bin/brain/add.go` /
|
||||||
`POST /ingest` (Python `kblib.add_leafs`; no Go upsert port).
|
`POST /ingest` (Python `kblib.add_leafs`; no Go upsert port).
|
||||||
|
[#17](https://git.produktor.io/eSlider/2dph/issues/17) `--hop N` walks
|
||||||
|
FROM_FILE / HAS_VERSION / AUTHORED.
|
||||||
[#18](https://git.produktor.io/eSlider/2dph/issues/18) `--with-facts` /
|
[#18](https://git.produktor.io/eSlider/2dph/issues/18) `--with-facts` /
|
||||||
`--with-chats` on rebuild (WhatsApp out of v1).
|
`--with-chats` on rebuild (WhatsApp out of v1).
|
||||||
|
[#19](https://git.produktor.io/eSlider/2dph/issues/19) CI recall SoT =
|
||||||
|
`bin/brain/eval.go` via Zig.
|
||||||
|
Epic [#16](https://git.produktor.io/eSlider/2dph/issues/16) closed.
|
||||||
|
|
||||||
|
## v2
|
||||||
|
|
||||||
|
[#6](https://git.produktor.io/eSlider/2dph/issues/6) OCR — `pdftotext` then
|
||||||
|
`pdftoppm` + tesseract `eng+deu`. Optional `ocr-paddle`.
|
||||||
|
[#29](https://git.produktor.io/eSlider/2dph/issues/29) OQ1 contradiction
|
||||||
|
resolution. [#30](https://git.produktor.io/eSlider/2dph/issues/30) OQ3 duckdb-md.
|
||||||
|
|
||||||
## Blockers
|
## Blockers
|
||||||
|
|
||||||
|
None for epic #16 (closed). Remaining v2: OQ1, OQ3, OQ4.
|
||||||
|
|
||||||
```
|
```
|
||||||
question
|
question
|
||||||
│
|
│
|
||||||
@@ -42,15 +56,11 @@ question
|
|||||||
└─ facts+chats corpus ← in
|
└─ facts+chats corpus ← in
|
||||||
```
|
```
|
||||||
|
|
||||||
1. **[#19](https://git.produktor.io/eSlider/2dph/issues/19) CI eval** —
|
|
||||||
recall SoT should be `bin/brain/eval.go` via Zig, not Python `bin/kb/eval`.
|
|
||||||
|
|
||||||
## Not v1
|
## Not v1
|
||||||
|
|
||||||
[#6](https://git.produktor.io/eSlider/2dph/issues/6) OCR (OQ2), OQ1
|
OQ1 contradiction resolution, OQ3 duckdb-md export, OQ4 YAML-first leafs.
|
||||||
contradiction resolution, OQ3 duckdb-md export, OQ4 YAML-first leafs.
|
OCR (OQ2) is in: tesseract, not docling.
|
||||||
|
|
||||||
## Close epic #16 when
|
## Close epic #16 when
|
||||||
|
|
||||||
- MCP tool order is documented and still gated by tests
|
Children #14, #15, #17, #18, #19 are closed. MCP tool order stays gated by tests.
|
||||||
- CI recall SoT is `bin/brain/eval.go` via Zig
|
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ No laptop-absolute paths. Config lives in env files under `$HOME/.config/brain/`
|
|||||||
- Go (see `go.mod`)
|
- Go (see `go.mod`)
|
||||||
- Python 3.12 + [uv](https://docs.astral.sh/uv)
|
- Python 3.12 + [uv](https://docs.astral.sh/uv)
|
||||||
- Optional: Docker, Zig CGO via `bin/cgo/zig` (not gcc)
|
- Optional: Docker, Zig CGO via `bin/cgo/zig` (not gcc)
|
||||||
|
- Optional: poppler (`pdftotext`/`pdftoppm`) + tesseract `eng+deu` for mail OCR
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv venv .venv
|
uv venv .venv
|
||||||
|
|||||||
@@ -0,0 +1,162 @@
|
|||||||
|
// Package ocr runs Tesseract (eng+deu) on images and scanned PDFs.
|
||||||
|
//
|
||||||
|
// Default engine is the tesseract CLI, not gosseract CGO: Ladybug CGO stays
|
||||||
|
// Zig-only (D21). Same engine, no gocv. OCR_ENGINE=paddle selects paddleocr
|
||||||
|
// when that binary is on PATH (compose profile ocr-paddle).
|
||||||
|
package ocr
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"image"
|
||||||
|
"image/color"
|
||||||
|
"image/png"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
const TessLang = "eng+deu"
|
||||||
|
|
||||||
|
func ImageFile(path string) (string, error) {
|
||||||
|
engine := os.Getenv("OCR_ENGINE")
|
||||||
|
if engine == "paddle" {
|
||||||
|
return runPaddle(path)
|
||||||
|
}
|
||||||
|
return runTesseract(path)
|
||||||
|
}
|
||||||
|
|
||||||
|
func PDFFile(path string) (string, error) {
|
||||||
|
text, err := pdfToText(path)
|
||||||
|
if err == nil && strings.TrimSpace(text) != "" {
|
||||||
|
return strings.TrimSpace(text), nil
|
||||||
|
}
|
||||||
|
ocr, oerr := pdfPages(path)
|
||||||
|
if oerr != nil {
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return "", oerr
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(ocr) != "" {
|
||||||
|
return strings.TrimSpace(ocr), nil
|
||||||
|
}
|
||||||
|
if text != "" {
|
||||||
|
return strings.TrimSpace(text), nil
|
||||||
|
}
|
||||||
|
return "", fmt.Errorf("pdf has no text layer (ocr unavailable)")
|
||||||
|
}
|
||||||
|
|
||||||
|
func pdfToText(path string) (string, error) {
|
||||||
|
cmd := exec.Command("pdftotext", "-layout", path, "-")
|
||||||
|
out, err := cmd.Output()
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return string(out), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func pdfPages(path string) (string, error) {
|
||||||
|
dir, err := os.MkdirTemp("", "2dph-ocr-")
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
defer os.RemoveAll(dir)
|
||||||
|
prefix := filepath.Join(dir, "page")
|
||||||
|
cmd := exec.Command("pdftoppm", "-png", "-r", "200", path, prefix)
|
||||||
|
if err := cmd.Run(); err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
matches, err := filepath.Glob(prefix + "*.png")
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
var parts []string
|
||||||
|
for _, img := range matches {
|
||||||
|
t, err := ImageFile(img)
|
||||||
|
if err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if s := strings.TrimSpace(t); s != "" {
|
||||||
|
parts = append(parts, s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return strings.Join(parts, "\n\n"), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func runTesseract(path string) (string, error) {
|
||||||
|
pre, err := preprocessFile(path)
|
||||||
|
if err != nil {
|
||||||
|
pre = path
|
||||||
|
} else {
|
||||||
|
defer os.Remove(pre)
|
||||||
|
}
|
||||||
|
cmd := exec.Command("tesseract", pre, "stdout", "-l", TessLang, "--psm", "6")
|
||||||
|
out, err := cmd.Output()
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return strings.TrimSpace(string(out)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func runPaddle(path string) (string, error) {
|
||||||
|
cmd := exec.Command("paddleocr", "ocr", "-i", path)
|
||||||
|
out, err := cmd.Output()
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return strings.TrimSpace(string(out)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func preprocessFile(path string) (string, error) {
|
||||||
|
f, err := os.Open(path)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
defer f.Close()
|
||||||
|
img, err := png.Decode(f)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
out := filepath.Join(os.TempDir(), filepath.Base(path)+".gray.png")
|
||||||
|
w, err := os.Create(out)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
defer w.Close()
|
||||||
|
if err := png.Encode(w, GrayContrast(img)); err != nil {
|
||||||
|
os.Remove(out)
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// GrayContrast is a stdlib preprocess (no gocv): grayscale + stretch.
|
||||||
|
func GrayContrast(src image.Image) image.Image {
|
||||||
|
b := src.Bounds()
|
||||||
|
dst := image.NewGray(b)
|
||||||
|
var minL, maxL uint8 = 255, 0
|
||||||
|
for y := b.Min.Y; y < b.Max.Y; y++ {
|
||||||
|
for x := b.Min.X; x < b.Max.X; x++ {
|
||||||
|
g := color.GrayModel.Convert(src.At(x, y)).(color.Gray)
|
||||||
|
if g.Y < minL {
|
||||||
|
minL = g.Y
|
||||||
|
}
|
||||||
|
if g.Y > maxL {
|
||||||
|
maxL = g.Y
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
span := int(maxL) - int(minL)
|
||||||
|
if span < 1 {
|
||||||
|
span = 1
|
||||||
|
}
|
||||||
|
for y := b.Min.Y; y < b.Max.Y; y++ {
|
||||||
|
for x := b.Min.X; x < b.Max.X; x++ {
|
||||||
|
g := color.GrayModel.Convert(src.At(x, y)).(color.Gray)
|
||||||
|
v := uint8((int(g.Y) - int(minL)) * 255 / span)
|
||||||
|
dst.SetGray(x, y, color.Gray{Y: v})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return dst
|
||||||
|
}
|
||||||
@@ -0,0 +1,54 @@
|
|||||||
|
package ocr
|
||||||
|
|
||||||
|
import (
|
||||||
|
"image"
|
||||||
|
"image/color"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestGrayContrastStretches(t *testing.T) {
|
||||||
|
img := image.NewGray(image.Rect(0, 0, 2, 2))
|
||||||
|
img.SetGray(0, 0, color.Gray{Y: 64})
|
||||||
|
img.SetGray(0, 1, color.Gray{Y: 64})
|
||||||
|
img.SetGray(1, 0, color.Gray{Y: 64})
|
||||||
|
img.SetGray(1, 1, color.Gray{Y: 192})
|
||||||
|
out := GrayContrast(img).(*image.Gray)
|
||||||
|
if out.GrayAt(0, 0).Y != 0 {
|
||||||
|
t.Fatalf("min should map to 0, got %d", out.GrayAt(0, 0).Y)
|
||||||
|
}
|
||||||
|
if out.GrayAt(1, 1).Y != 255 {
|
||||||
|
t.Fatalf("max should map to 255, got %d", out.GrayAt(1, 1).Y)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHelloPNGFixtureOCR(t *testing.T) {
|
||||||
|
if _, err := exec.LookPath("tesseract"); err != nil {
|
||||||
|
t.Skip("tesseract not installed")
|
||||||
|
}
|
||||||
|
path := filepath.Join("testdata", "hello.png")
|
||||||
|
got, err := ImageFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
up := strings.ToUpper(got)
|
||||||
|
if !strings.Contains(up, "HELLO") {
|
||||||
|
t.Fatalf("ocr %q missing HELLO", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPaddleEngineUsesPaddleocrBinary(t *testing.T) {
|
||||||
|
t.Setenv("OCR_ENGINE", "paddle")
|
||||||
|
_, err := ImageFile(filepath.Join("testdata", "hello.png"))
|
||||||
|
if _, look := exec.LookPath("paddleocr"); look != nil {
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error when paddleocr is missing")
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
Vendored
BIN
Binary file not shown.
|
After Width: | Height: | Size: 1.7 KiB |
@@ -6,7 +6,6 @@ readme = "README.md"
|
|||||||
requires-python = ">=3.12"
|
requires-python = ">=3.12"
|
||||||
license = { text = "MIT" }
|
license = { text = "MIT" }
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"docling>=2.119.0",
|
|
||||||
"ladybug==0.19.1",
|
"ladybug==0.19.1",
|
||||||
"markitdown[docx,epub,html,image-exif,pdf,pptx,xlsx,zip]>=0.1.7",
|
"markitdown[docx,epub,html,image-exif,pdf,pptx,xlsx,zip]>=0.1.7",
|
||||||
"mistune==3.3.4",
|
"mistune==3.3.4",
|
||||||
|
|||||||
Reference in New Issue
Block a user