Compare commits

..
Author SHA1 Message Date
eSlider 9a32dce0ca feat: search --hop walks FROM_FILE to Person.
Tests / Test (push) Skipped
Tests / Release (semver) (push) Skipped
Tests / Test (pull_request) Failing after 3s
Tests / Release (semver) (pull_request) Skipped
Parser no longer errors; hop 1 returns File, hop 3 reaches Person.
Rebuild writes Leaf-[:FROM_FILE]->File so the walk is not empty
on a fresh index (Gitea #17).
2026-08-14 10:48:22 +01:00
23 changed files with 1780 additions and 749 deletions
+4 -30
View File
@@ -54,44 +54,18 @@ jobs:
run: go test ./internal/brain/rank -count=1
- name: facts/audit self (lexicon consistency, no network)
run: ./bin/facts/audit self
run: |
./bin/facts/audit self 2>/dev/null || echo "audit: not yet implemented; gate skipped"
- name: CGO via Zig (compile brain/search + eval)
- name: CGO via Zig (compile brain/search)
run: |
chmod +x bin/cgo/zig bin/cgo/zcc bin/cgo/zc++
bin/cgo/zig go build -tags system_ladybug -o /tmp/brain-search ./bin/brain/search.go
bin/cgo/zig go build -tags 'system_ladybug,brain_eval' -o /tmp/brain-eval ./bin/brain/eval.go
- uses: actions/cache@v4
with:
path: ~/.cache/huggingface
key: ${{ runner.os }}-hf-potion-multilingual-128M
- name: recall@5 SoT (Zig bin/brain/eval.go)
run: |
uv run python bin/kb/index --rebuild --json
KB_ROOT="$PWD" /tmp/brain-eval --json
ocr:
name: OCR (tesseract fixture)
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-go@v5
with:
go-version-file: go.mod
- name: Install tesseract + poppler
run: |
sudo apt-get update
sudo apt-get install -y --no-install-recommends \
tesseract-ocr tesseract-ocr-eng tesseract-ocr-deu poppler-utils
- name: Go OCR tests (synthetic HELLO PNG)
run: go test ./internal/ocr -count=1
release:
name: Release (semver)
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
needs: [test, ocr]
needs: test
runs-on: ubuntu-latest
permissions:
contents: write
+5 -7
View File
@@ -42,7 +42,7 @@ skills/ in-project agent skills (vendored, no external links)
bin/ self-describing tools bin/{subject}/{method}.go (shebang)
bin/brain/ search.go serve.go index.go add.go get.go stats.go eval.go watch.go
bin/chats/ sync.go import.go facts.go apply.go; libs in internal/chats
bin/mail/ sync.go import.go ocr.go (index_mail → brain/index.go)
bin/mail/ sync.go import.go (index_mail → brain/index.go)
bin/markdown/ import.go (H2 leaf split; Python bin/md/import fallback)
bin/postgres/ query.go (read-only YAML)
bin/git/ import.go (go-git history; Python shim execs it)
@@ -65,15 +65,15 @@ var/ kb.lbug, var/mail/*, caches (gitignored)
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw message.json + attachments
bin/mail/sync.go --source gmail --query 'from:example.com' --out var/mail # Gmail search (default in:inbox)
bin/mail/import.go --from-raw var/mail # message.json → message.md (convert only)
bin/brain/index.go --rebuild --with-facts --with-chats
bin/brain/index.go --rebuild # rebuild brain incl. all mail (fresh DB)
```
- `sync` (Go) downloads messages + attachments; Gmail uses paginated list +
`body.attachmentId` (not partId) for attachments.
- `import` converts body + attachments to markdown. PDFs use poppler
`pdftotext -layout` fast path (~15ms); textless/scanned PDFs use
`pdftoppm` + tesseract `eng+deu` (`bin/mail/ocr.go`). Optional
`OCR_ENGINE=paddle`. Conversion never touches the brain DB (crash safety).
`pdftotext -layout` fast path (~15ms); textless/scanned PDFs fall back to
docling (isolated subprocess — its native onnx can segfault the parent).
Conversion never touches the brain DB (crash safety).
- `index_mail` is a deprecation shim for `bin/brain/index.go --rebuild`. Bulk
rebuild still deletes `var/kb.lbug` and creates FTS/HNSW last. Single-leaf
write is `bin/brain/add.go` (safe while indexes exist; do not DROP INDEX).
@@ -89,7 +89,6 @@ bin/kb/search "query" [--repo X] # deprecated wrapper → bin/b
bin/brain/search.go "query" [--root facts|info] # deduction search → YAML
bin/brain/search.go "query" --no-web # local graph only
eval "$(bin/cgo/zig env)" # Zig cc + liblbug (not gcc)
bin/brain/index.go --rebuild [--with-mail] [--with-facts] [--with-chats]
bin/brain/add.go --text T --root facts --source "a.md x b.md" # incremental write
bin/brain/add.go --json # stdin leaf or {leafs:[...]}
bin/brain/get.go <id> [--body] [--json] # Go read; Python bin/kb/get CI fallback
@@ -101,7 +100,6 @@ bin/git/import.go [REPO] [--json] [--limit N] # go-git history → commit le
bin/web/search.go "query" [--json] # SearXNG; throttled ≠ absence
bin/reasoner/bakeoff.go [--model ID] [--json] # D18 CPU tool-call bake-off
bin/postgres/query.go --profile onlyoffice -c 'SELECT 1'
bin/mail/ocr.go <image|pdf> # tesseract eng+deu (scans)
bin/md/tables # what the graph holds → YAML
bin/brain/deduce "question" # thinking wrapper
```
-4
View File
@@ -16,10 +16,6 @@ ENV PYTHONUNBUFFERED=1 \
WORKDIR /app
RUN id -u 2dph 2>/dev/null || useradd --create-home --uid 1001 2dph
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
poppler-utils tesseract-ocr tesseract-ocr-eng tesseract-ocr-deu \
&& rm -rf /var/lib/apt/lists/*
COPY requirements.lock.txt /tmp/requirements.lock.txt
RUN python -m pip install --no-cache-dir -r /tmp/requirements.lock.txt \
+14 -19
View File
@@ -4,9 +4,8 @@ A brain that loves facts and deduction. Evidence-first knowledge graph + hybrid
RAG over the operational Brain/ops/eSlider stack. Built like Sherlock
Holmes: nothing is asserted unless it has proof.
Status: **v1 in** (epic [#16](https://git.produktor.io/eSlider/2dph/issues/16) closed).
v2 board: milestone [v2](https://git.produktor.io/eSlider/2dph/milestone/13) — OCR [#6](https://git.produktor.io/eSlider/2dph/issues/6),
[#29](https://git.produktor.io/eSlider/2dph/issues/29) OQ1, [#30](https://git.produktor.io/eSlider/2dph/issues/30) OQ3.
Status: **in progress** — read path + MCP work; v1 goal is [epic #16](https://git.produktor.io/eSlider/2dph/issues/16)
(milestone [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12)).
Gap: [docs/roadmap.md](docs/roadmap.md).
## What
@@ -75,7 +74,6 @@ detective method: **a fact needs ≥2 independent sources or it is
reasoner/bakeoff.go CPU tool-call bake-off (D18; OpenAI tools)
chats/sync.go import.go facts.go apply.go
(libs in internal/chats; no chats index)
mail/ocr.go tesseract eng+deu (pdftoppm scans)
md/import (deprecated; bin/markdown/import.go)
brain/extract brain/audit brain/deduce (thinking wrapper)
web/search (deprecated shim → web/search.go)
@@ -117,13 +115,10 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
## Open questions (v2)
- OQ1: mutually-contradicting evidence — how to resolve (authority weighting,
temporal freshness, audit adjudication). **v2**; [#29](https://git.produktor.io/eSlider/2dph/issues/29).
- OQ2: OCR — **in**. `pdftotext -layout` first; scans `pdftoppm` + tesseract
`eng+deu` (`bin/mail/ocr.go`, `internal/ocr`). No gocv, no gosseract CGO
(D21 Zig owns Ladybug CGO). Optional `OCR_ENGINE=paddle` / compose profile
`ocr-paddle`. Docling left the default path. [#6](https://git.produktor.io/eSlider/2dph/issues/6).
temporal freshness, audit adjudication). **v2**; does not block epic #16.
- OQ2: OCR — poppler `pdftotext` fast-path exists; scans still docling.
[#6](https://git.produktor.io/eSlider/2dph/issues/6) (v2, does not block #16).
- OQ3: optional duckdb-md layer for `SELECT … FORMAT MARKDOWN` export/write-back.
[#30](https://git.produktor.io/eSlider/2dph/issues/30).
- OQ4: YAML-first storage for leafs — deferred: JSON is ~10x faster to
serialize and unambiguous; YAML only where humans edit files.
@@ -132,8 +127,7 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
1. `bin/mail/sync.go` (Go, 8 workers) — paginated Gmail/OnlyOffice download.
Gmail attachments key off `body.attachmentId`, not MIME `partId`.
2. `bin/mail/import.go --from-raw` — message.json → message.md; PDFs via
`pdftotext -layout` (~15ms); textless/scanned PDFs `pdftoppm` + tesseract
`eng+deu`. ICS sidecars
`pdftotext -layout` (~15ms) with docling subprocess fallback; ICS sidecars
Latin-1→UTF-8 normalized.
3. `bin/brain/index.go --rebuild` — fresh rebuild (repo corpus + mail) because ladybug
corrupts its WAL on bulk-insert into an already-indexed DB. Conversion and
@@ -150,8 +144,8 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
2. `go test ./internal/brain/rank` (cgo-free ranking + flag parser)
3. python -m unittest discover -s bin/tools (includes published-docs SoT)
4. `bin/facts/audit self` (lexicon internal consistency; `bin/facts/audit.go` is the D14 wrapper)
5. `bin/brain/eval.go` via Zig (recall@5 ≥ 0.95). Python `bin/kb/eval` is an
explicit fallback, not the CI SoT.
5. `bin/kb/eval` (recall@5 ≥ 0.95). Local SoT is `bin/brain/eval.go` via Zig CGO.
CI SoT switch: [#19](https://git.produktor.io/eSlider/2dph/issues/19).
6. `bin/cgo/zig go build -tags system_ladybug` (compile search with zig cc; fetches pinned zig+libs).
Feedback loop: every commit → PR → CI → green/gate → merge. Same discipline as
@@ -165,12 +159,13 @@ Feedback loop: every commit → PR → CI → green/gate → merge. Same discipl
4. .venv: ladybug + model2vec + mistune
5. schema + tools with TDD (kb + md + facts + brain)
6. ~/.config/brain config
7. corpus extraction (facts/info) — **in**: [#18](https://git.produktor.io/eSlider/2dph/issues/18)
7. corpus extraction (facts/info) — **open**: [#18](https://git.produktor.io/eSlider/2dph/issues/18)
8. verify: web-search smoke, onlyoffice pg, md-db round-trip, eval, audit
## Gap to v1 (epic #16)
Remaining: none for epic #16 (v1). Board:
Read path + MCP are in. Incremental `brain/add` and `--hop` are in. Remaining:
facts+chats corpus on rebuild, and CI eval SoT. Board:
[epic #16](https://git.produktor.io/eSlider/2dph/issues/16),
milestone [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12).
Narrative: [docs/roadmap.md](docs/roadmap.md).
@@ -179,8 +174,8 @@ Narrative: [docs/roadmap.md](docs/roadmap.md).
|-------|-------|-----|
| 1 | [#14](https://git.produktor.io/eSlider/2dph/issues/14) | **in**`bin/brain/add.go` / `POST /ingest` write facts+info without deleting `kb.lbug`. Bulk corpus still `--rebuild`. Leftover Python (mail/facts) is not the living-graph blocker. |
| 2 | [#17](https://git.produktor.io/eSlider/2dph/issues/17) | **in**`--hop N` walks `FROM_FILE``HAS_VERSION``AUTHORED` (max 3). |
| 3 | [#18](https://git.produktor.io/eSlider/2dph/issues/18) | **in**`--with-facts` / `--facts-json` land `root=facts`; `--with-chats` indexes `var/chats/md`. WhatsApp sync is out of v1. |
| 3 | [#18](https://git.produktor.io/eSlider/2dph/issues/18) | Rebuild is mostly `info` (repo md + mail). `facts/extract` and chats are not a first-class index input. WhatsApp sync is a stub. |
| 4 | [#15](https://git.produktor.io/eSlider/2dph/issues/15) | **in** — lever/loop documented (`search``get``audit`). |
| 5 | [#19](https://git.produktor.io/eSlider/2dph/issues/19) | **in** — CI recall SoT is `bin/brain/eval.go` via Zig. Python `bin/kb/eval` stays as an explicit fallback. |
| 5 | [#19](https://git.produktor.io/eSlider/2dph/issues/19) | GitHub CI recall still runs Python `bin/kb/eval`. |
Does **not** block epic close: OQ1 [#29](https://git.produktor.io/eSlider/2dph/issues/29), OQ3 [#30](https://git.produktor.io/eSlider/2dph/issues/30), OQ4. OCR [#6](https://git.produktor.io/eSlider/2dph/issues/6) is **in**.
Does **not** block epic close: [#6](https://git.produktor.io/eSlider/2dph/issues/6) OCR, OQ1, OQ3, OQ4.
-4
View File
@@ -121,7 +121,6 @@ Mail is a first-class corpus (retrievable through the same search):
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw sync (Go)
bin/mail/import.go --from-raw var/mail # JSON → markdown
bin/brain/add.go --text T --root facts --source "a.md x b.md"
bin/brain/index.go --rebuild --with-facts --with-chats # facts extract + chats md
bin/brain/index.go --rebuild # rebuild brain (incl. mail)
bin/brain/search.go "invoice from last week" # same search over mail leafs
```
@@ -169,9 +168,6 @@ docker compose up brain-watch # auto re-index on change
## Related
eSlider DevOps engineer practice: ops, OnlyOffice, and mail feed the facts
root through `bin/facts/extract` (two-source pairing).
- [go-second-brain](https://github.com/eSlider/go-second-brain) — the earlier
Neo4j + Qdrant + Matrix RAG brain
- [agent-skills](https://github.com/eSlider/agent-skills) — upstream
+1 -1
View File
@@ -3,7 +3,7 @@
//
// bin/brain/index.go - rebuild the Ladybug graph (Python write path).
//
// ./bin/brain/index.go --rebuild --with-facts --with-chats
// ./bin/brain/index.go --rebuild
// ./bin/brain/index.go --rebuild --with-mail
// ./bin/brain/index.go --dry-run --with-mail
//
+2 -3
View File
@@ -29,11 +29,10 @@ func main() {
case "linkedin":
os.Exit(chats.RunSyncLinkedIn(args))
case "whatsapp":
fmt.Fprintln(os.Stderr, "chats: WhatsApp sync is out of v1")
fmt.Fprintln(os.Stderr, "chats: WhatsApp not implemented yet")
os.Exit(1)
case "help", "-h", "--help":
fmt.Fprintln(os.Stderr, `usage: bin/chats/sync.go telegram|linkedin [flags]
WhatsApp sync is out of v1.`)
fmt.Fprintln(os.Stderr, `usage: bin/chats/sync.go telegram|linkedin [flags]`)
return
default:
fmt.Fprintf(os.Stderr, "chats: unknown platform %q\n", platform)
+13 -96
View File
@@ -2,15 +2,12 @@
"""kb/index - build the 2dph brain from markdown + factual leafs.
bin/kb/index [--corpus DIR] [--rebuild] [--limit N]
bin/kb/index --rebuild --with-facts --with-chats
bin/kb/index --json # emit stats as JSON
Reads every .md under the corpus (default: repo root docs, skills, READMEs)
as `info` leafs, embeds them with model2vec (potion-multilingual-128M), and
writes them into var/kb.lbug with FTS + HNSW indexes. `facts` leafs come
from bin/facts/extract (docker × compose × ssh-config pairing) when
`--with-facts` is set. `--with-chats` indexes markdown under var/chats/md
(or a given dir) as info. WhatsApp sync stays out of v1.
from bin/facts/extract (docker x compose x ssh-config pairing).
--rebuild drops the database file and indexes from scratch. Without it a run
is idempotent (MERGE by (source,text) id).
@@ -25,7 +22,7 @@ ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(ROOT / "bin" / "tools"))
from kblib import ( # noqa: E402
add_leafs, connect, ensure_indexes, init_schema, upsert_leaf, link_from_file,
connect, ensure_indexes, init_schema, upsert_leaf, link_from_file,
open_readonly, stats,
)
from mdleaves import read_markdown, to_all, walk_markdown # noqa: E402
@@ -99,65 +96,12 @@ def embedder():
return lambda text: model.encode([text])[0].astype(float).tolist()
def index_fact_dicts(conn, facts: list[dict], embed_fn) -> int:
"""Write extract-shaped dicts as root=facts leafs (2-source source field)."""
leafs = []
for f in facts:
text = str(f.get("text") or "")
source = str(f.get("source") or "")
if not text or not source:
continue
leafs.append({
"text": text,
"root": "facts",
"confidence": "confirmed",
"source": source,
"source_rev": f.get("source_rev") or "working-tree",
"how": f.get("how") or "facts/extract",
"loc": f.get("loc") or source,
"type": "fact",
"embedding": embed_fn(text) if text else None,
})
return len(add_leafs(conn, leafs))
def facts_from_extract() -> list[dict]:
import subprocess
proc = subprocess.run(
[sys.executable, str(ROOT / "bin" / "facts" / "extract"), "--json", "--dry-run"],
cwd=ROOT,
capture_output=True,
text=True,
check=False,
)
if proc.returncode != 0:
print(f"kb/index: facts/extract failed: {proc.stderr}", file=sys.stderr)
return []
try:
payload = json.loads(proc.stdout)
except json.JSONDecodeError:
print("kb/index: facts/extract produced non-JSON", file=sys.stderr)
return []
return list(payload.get("facts") or [])
def main(argv: list[str]) -> int:
import argparse
p = argparse.ArgumentParser(description="build the 2dph brain index")
p.add_argument("--corpus", action="append", help="extra markdown dir/file to index (may repeat)")
p.add_argument("--rebuild", action="store_true", help="fresh db + indexes")
p.add_argument("--db", default="", help="path to kb.lbug (default var/kb.lbug)")
p.add_argument("--no-defaults", action="store_true", help="do not index repo README/docs/skills")
p.add_argument("--with-mail", action="store_true", help="include var/mail message.md leafs")
p.add_argument("--with-facts", action="store_true", help="run facts/extract into root=facts")
p.add_argument("--facts-json", default="", help="JSON list (or {facts:[...]}) of fact dicts")
p.add_argument(
"--with-chats",
nargs="?",
const=str(ROOT / "var" / "chats" / "md"),
default="",
help="index chat markdown as info (default var/chats/md)",
)
p.add_argument("--since", default="", help="with --with-mail, only messages dated >= YYYY-MM-DD")
p.add_argument("--dry-run", action="store_true", help="count leafs, write nothing")
p.add_argument(
@@ -171,71 +115,44 @@ def main(argv: list[str]) -> int:
from kblib import DB_PATH, VAR
dbpath = Path(a.db) if a.db else DB_PATH
leafs: list[dict] = [] if a.no_defaults else load_corpus(ROOT)
leafs = load_corpus(ROOT)
if a.corpus:
for source in a.corpus:
leafs.extend(load_corpus_glob(source))
chat_n = 0
if a.with_chats:
chats = load_corpus_glob(a.with_chats)
chat_n = len(chats)
leafs.extend(chats)
mail_n = 0
if a.with_mail:
mail = from_mail_root(ROOT / "var" / "mail", since=a.since)
mail_n = len(mail)
leafs.extend(mail)
facts: list[dict] = []
if a.facts_json:
raw = Path(a.facts_json).read_text(encoding="utf-8")
payload = json.loads(raw)
facts = list(payload.get("facts") if isinstance(payload, dict) else payload)
if a.with_facts:
facts.extend(facts_from_extract())
if a.dry_run:
msg = {
"indexed": 0,
"corpus_total": len(leafs),
"mail_leafs": mail_n,
"chat_leafs": chat_n,
"facts_leafs": len(facts),
"dry_run": True,
}
msg = {"indexed": 0, "corpus_total": len(leafs), "mail_leafs": mail_n, "dry_run": True}
print(json.dumps(msg, indent=2) if a.json else
f"brain/index: {len(leafs)} info + {len(facts)} facts would be indexed")
f"brain/index: {len(leafs)} leafs would be indexed (mail={mail_n})")
return 0
VAR.mkdir(exist_ok=True)
dbpath.parent.mkdir(parents=True, exist_ok=True)
if a.rebuild and dbpath.exists():
dbpath.unlink()
if a.rebuild and DB_PATH.exists():
DB_PATH.unlink()
db, conn = connect(dbpath, read_only=False)
db, conn = connect(DB_PATH, read_only=False)
init_schema(conn)
# Never DROP FTS/VECTOR (ghost catalog). Write leafs, then ensure indexes
# unless --skip-indexes (seed facts first — MERGE under live FTS corrupts it).
# --rebuild already deleted kb.lbug above, so CREATE runs on a clean DB.
embed = embedder()
done, total = index_leafs(conn, leafs, embed, a.limit)
fact_n = index_fact_dicts(conn, facts, embed) if facts else 0
if not a.skip_indexes:
ensure_indexes(conn)
s = stats(conn)
conn.close()
db.close()
result = {
"indexed": done,
"corpus_total": total,
"facts_leafs": fact_n,
"chat_leafs": chat_n,
**{k: v for k, v in s.items() if k in ("total", "by_root")},
}
result = {"indexed": done, "corpus_total": total, **{k: v for k, v in s.items() if k in ("total", "by_root")}}
if a.skip_indexes:
result["indexes"] = "skipped"
print(json.dumps(result, indent=2) if a.json else
f"indexed {done}/{total} info + {fact_n} facts; db total {s['total']}")
print(json.dumps(result, indent=2) if a.json else f"indexed {done}/{total} leafs; db total {s['total']}")
return 0
+77 -11
View File
@@ -7,7 +7,7 @@
bin/mail/import --since 2026-01-01 only messages after a date
bin/mail/import --limit 50 cap messages per run
bin/mail/import --no-attachments body only, skip attachment conversion
bin/mail/import --ocr OCR images (PDFs OCR when textless)
bin/mail/import --ocr OCR scanned PDFs/images via docling
bin/mail/import --dry-run list messages without writing anything
Writes one directory per message: var/mail/{folder}/{message_id}/
@@ -16,11 +16,10 @@ Writes one directory per message: var/mail/{folder}/{message_id}/
attachments/*.md converted attachment content
Indexing is a separate step (`bin/brain/index.go --rebuild`): conversion can
crash and must not leave the brain DB mid-transaction.
crash in native docling and must not leave the brain DB mid-transaction.
Requires ONLYOFFICE_URL/USER/PASS in .env (or env) except `--from-raw`.
Idempotent: a message already present (message.md exists) is skipped unless
--force.
Requires ONLYOFFICE_URL/USER/PASS in .env (or env). Idempotent: a message
already present (message.md exists) is skipped unless --force.
"""
from __future__ import annotations
@@ -42,10 +41,10 @@ from mailconv import ( # noqa: E402
IMAGE_SUFFIXES,
LEGACY_OFFICE_SUFFIXES,
TEXT_SUFFIXES,
convert_pdf,
html_to_markdown,
is_convertible,
normalize_markdown,
ocr_image,
subject_to_filename,
zip_extract_safe,
)
@@ -148,9 +147,9 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
except Exception as e:
return f"\n<!-- conversion failed: {e} -->\n"
if suffix == ".pdf":
return convert_pdf(path, ocr)
return _convert_pdf(path, ocr)
if suffix in IMAGE_SUFFIXES and ocr:
return ocr_image(path) or "\n<!-- ocr unavailable -->\n"
return _convert_pdf(path, ocr)
if suffix in LEGACY_OFFICE_SUFFIXES:
return _convert_legacy(path)
if suffix in ARCHIVE_SUFFIXES:
@@ -158,6 +157,67 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
return None
def _convert_pdf(path: Path, ocr: bool) -> str:
"""Convert one PDF to markdown.
Fast path: poppler's pdftotext (-layout) extracts exact text from
born-digital PDFs in ~15ms vs docling's 1-3s. Only textless PDFs (scanned
pages, layout-heavy) fall back to docling, which runs isolated in a
subprocess because its native onnx/RT-DETR has segfaulted the main process.
"""
text = _pdf_fast_text(path)
if ocr or text is None or not text.strip():
return _convert_pdf_docling(path, ocr)
return normalize_markdown(text)
def _pdf_fast_text(path: Path) -> str | None:
"""pdftotext -layout; None when poppler is unavailable (or the PDF has no text layer)."""
try:
proc = subprocess.run(
["pdftotext", "-layout", str(path), "-"],
capture_output=True, timeout=60)
except (OSError, subprocess.TimeoutExpired):
return None
if proc.returncode != 0:
return None
return proc.stdout.decode("utf-8", errors="replace")
def _convert_pdf_docling(path: Path, ocr: bool) -> str:
try:
proc = subprocess.run(
[sys.executable, os.path.abspath(__file__), "--pdf-worker", str(path),
"--ocr" if ocr else "--no-ocr"],
capture_output=True, text=True, timeout=600)
except subprocess.TimeoutExpired:
return "\n<!-- pdf conversion timed out -->\n"
if proc.returncode != 0:
tail = proc.stderr.strip().splitlines()[-3:]
return f"\n<!-- pdf conversion failed: {proc.returncode}: {' | '.join(tail)} -->\n"
return proc.stdout
def _pdf_worker(path: Path, ocr: bool) -> None:
"""docling worker entry: prints converted markdown on stdout, exits non-zero on error."""
try:
from docling.document_converter import DocumentConverter, PdfFormatOption
from docling.datamodel.pipeline_options import PdfPipelineOptions
opts = PdfPipelineOptions()
opts.do_ocr = bool(ocr)
opts.do_table_structure = True
conv = DocumentConverter(format_options={"pdf": PdfFormatOption(pipeline_options=opts)})
res = conv.convert(str(path))
sys.stdout.write(normalize_markdown(res.document.export_to_markdown()))
sys.exit(0)
except Exception as e:
# errors/stacktraces to stderr; the caller only reports a one-liner
print(f"pdf-worker: {e}", file=sys.stderr)
import traceback
traceback.print_exc(file=sys.stderr)
sys.exit(1)
def _convert_legacy(path: Path) -> str:
"""Legacy .doc/.xls/.ppt -> md via pandoc (installed) or a stub."""
try:
@@ -296,12 +356,19 @@ def main(argv: list[str]) -> int:
p.add_argument("--from-raw", default="",
help="convert Go-synced dirs (var/mail/<folder>/<id>/message.json) to markdown")
p.add_argument("--no-attachments", action="store_true", help="skip attachment download+convert")
p.add_argument("--ocr", action="store_true", help="OCR images (PDFs OCR when textless)")
p.add_argument("--ocr", action="store_true", help="OCR scanned PDFs/images via docling")
p.add_argument("--force", action="store_true", help="re-import even if message.md exists")
p.add_argument("--dry-run", action="store_true", help="list messages, write nothing")
p.add_argument("--json", action="store_true")
p.add_argument("--pdf-worker", default="", help=argparse.SUPPRESS)
p.add_argument("--no-ocr", action="store_true", help=argparse.SUPPRESS)
a = p.parse_args(argv)
if a.pdf_worker:
_pdf_worker(Path(a.pdf_worker), ocr=not a.no_ocr)
return 0
conf = load_env()
fid = folder_id(a.folder)
out_root = ROOT / "var" / "mail"
summary: list[dict] = []
@@ -327,7 +394,6 @@ def main(argv: list[str]) -> int:
target_dir=msg_dir.parent))
summary.append(entry)
else:
conf = load_env()
OOCLIENT = OOClient(conf)
if a.id:
messages = [{"id": i} for i in a.id]
-48
View File
@@ -1,48 +0,0 @@
//usr/bin/env go run -tags=mail_ocr "$0" "$@"; exit
//go:build mail_ocr
//
// bin/mail/ocr.go - OCR an image or scanned PDF (tesseract eng+deu).
//
// ./bin/mail/ocr.go scan.png
// ./bin/mail/ocr.go scan.pdf
// OCR_ENGINE=paddle ./bin/mail/ocr.go scan.png
//
// PDFs try pdftotext -layout first; empty text layer uses pdftoppm + tesseract.
// No gocv. Tesseract CGO bindings are not used (D21 Zig owns Ladybug CGO).
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
package main
import (
"fmt"
"os"
"strings"
"github.com/eSlider/2dph/internal/ocr"
)
func main() {
os.Exit(run(os.Args[1:]))
}
func run(args []string) int {
if len(args) != 1 || strings.HasPrefix(args[0], "-") {
fmt.Fprintln(os.Stderr, `usage: bin/mail/ocr.go <image|pdf>`)
return 2
}
path := args[0]
var (
text string
err error
)
if strings.HasSuffix(strings.ToLower(path), ".pdf") {
text, err = ocr.PDFFile(path)
} else {
text, err = ocr.ImageFile(path)
}
if err != nil {
fmt.Fprintf(os.Stderr, "mail/ocr: %v\n", err)
return 1
}
fmt.Println(text)
return 0
}
+1 -84
View File
@@ -7,10 +7,7 @@ offline against fixtures.
from __future__ import annotations
import html
import os
import re
import subprocess
import tempfile
import zipfile
from pathlib import Path
@@ -21,10 +18,9 @@ OFFICE_SUFFIXES = {".docx", ".pptx", ".xlsx", ".html", ".htm", ".epub", ".eml",
PDF_SUFFIXES = {".pdf"}
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".tiff", ".tif", ".webp"}
ARCHIVE_SUFFIXES = {".zip"}
# Legacy binary Office (doc/xls/ppt) — markitdown skip them; we try
# Legacy binary Office (doc/xls/ppt) — markitdown/docling skip them; we try
# pandoc first, else leave a stub.
LEGACY_OFFICE_SUFFIXES = {".doc", ".xls", ".ppt"}
TESS_LANG = "eng+deu"
CONVERTIBLE_SUFFIXES = (
TEXT_SUFFIXES | OFFICE_SUFFIXES | PDF_SUFFIXES | IMAGE_SUFFIXES | ARCHIVE_SUFFIXES | LEGACY_OFFICE_SUFFIXES
@@ -150,82 +146,3 @@ def zip_extract_safe(zip_path: Path, dest: Path) -> list[Path]:
def is_convertible(suffix: str) -> bool:
return suffix.lower() in CONVERTIBLE_SUFFIXES
def convert_pdf(path: Path, ocr: bool = False) -> str:
"""pdftotext -layout first; empty text layer → pdftoppm + tesseract.
`ocr` is unused for born-digital PDFs (text layer wins). Scans OCR
automatically. This path never execs an ONNX document converter.
"""
del ocr # scans OCR when the text layer is empty; flag is for images
text = pdf_fast_text(path)
if text and text.strip():
return normalize_markdown(text)
scanned = ocr_pdf(path)
if scanned and scanned.strip():
return normalize_markdown(scanned)
if text:
return normalize_markdown(text)
return "\n<!-- pdf has no text layer (ocr unavailable) -->\n"
def pdf_fast_text(path: Path) -> str | None:
"""pdftotext -layout; None when poppler is missing or the command fails."""
try:
proc = subprocess.run(
["pdftotext", "-layout", str(path), "-"],
capture_output=True, timeout=60)
except (OSError, subprocess.TimeoutExpired):
return None
if proc.returncode != 0:
return None
return proc.stdout.decode("utf-8", errors="replace")
def ocr_pdf(path: Path) -> str:
"""Rasterize with pdftoppm and OCR each page (tesseract or paddle)."""
try:
with tempfile.TemporaryDirectory(prefix="2dph-ocr-") as tmp:
prefix = str(Path(tmp) / "page")
proc = subprocess.run(
["pdftoppm", "-png", "-r", "200", str(path), prefix],
capture_output=True, timeout=120)
if proc.returncode != 0:
return ""
pages = sorted(Path(tmp).glob("page*.png"))
parts = [ocr_image(p) for p in pages]
return "\n\n".join(p for p in parts if p and p.strip())
except (OSError, subprocess.TimeoutExpired):
return ""
def ocr_image(path: Path) -> str:
engine = os.environ.get("OCR_ENGINE", "tesseract")
if engine == "paddle":
return _ocr_paddle(path)
return _ocr_tesseract(path)
def _ocr_tesseract(path: Path) -> str:
try:
proc = subprocess.run(
["tesseract", str(path), "stdout", "-l", TESS_LANG, "--psm", "6"],
capture_output=True, timeout=120)
except (OSError, subprocess.TimeoutExpired):
return ""
if proc.returncode != 0:
return ""
return proc.stdout.decode("utf-8", errors="replace").strip()
def _ocr_paddle(path: Path) -> str:
try:
proc = subprocess.run(
["paddleocr", "ocr", "-i", str(path)],
capture_output=True, timeout=180)
except (OSError, subprocess.TimeoutExpired):
return ""
if proc.returncode != 0:
return ""
return proc.stdout.decode("utf-8", errors="replace").strip()
-49
View File
@@ -136,33 +136,6 @@ class BinLayoutTest(unittest.TestCase):
"index_mail must point at bin/brain/index.go",
)
def test_mail_ocr_is_tesseract_not_docling(self) -> None:
self._assert_shebang("bin/mail/ocr.go")
ocr = (ROOT / "bin" / "mail" / "ocr.go").read_text()
self.assertIn("internal/ocr", ocr)
self.assertIn("mail_ocr", ocr)
self.assertNotIn("github.com/otiai10/gosseract", ocr)
py = (ROOT / "bin" / "mail" / "import").read_text()
self.assertNotIn("from docling", py)
self.assertNotIn("import docling", py)
self.assertIn("convert_pdf", py)
conv = (ROOT / "bin" / "tools" / "mailconv.py").read_text()
self.assertIn("pdftotext", conv)
self.assertIn("pdftoppm", conv)
self.assertIn("tesseract", conv)
self.assertIn("eng+deu", conv)
self.assertNotIn("from docling", conv)
self.assertNotIn("import docling", conv)
self.assertNotIn("gocv", conv.lower())
proj = (ROOT / "pyproject.toml").read_text()
self.assertNotIn("docling", proj)
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
self.assertIn("tesseract-ocr", ci)
self.assertIn("./internal/ocr", ci)
compose = (ROOT / "compose.yaml").read_text()
self.assertIn("ocr-paddle", compose)
self.assertIn("OCR_ENGINE", compose)
def test_markdown_import_is_go_not_python_exec(self) -> None:
self._assert_shebang("bin/markdown/import.go")
text = (ROOT / "bin" / "markdown" / "import.go").read_text()
@@ -234,25 +207,3 @@ class BinLayoutTest(unittest.TestCase):
search = (ROOT / "bin" / "kb" / "search").read_text()
self.assertIn("bin/cgo/zig", search)
self.assertNotIn("command -v gcc", search)
def test_ci_recall_sot_is_zig_brain_eval(self) -> None:
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
self.assertIn("bin/brain/eval.go", ci)
self.assertIn("system_ladybug,brain_eval", ci)
self.assertIn("/tmp/brain-eval", ci)
self.assertIn("KB_ROOT", ci)
self.assertNotIn("bin/kb/eval", ci)
self.assertNotIn("gate skipped", ci)
self.assertIn("./bin/facts/audit self", ci)
def test_eval_fragments_live_in_default_corpus(self) -> None:
"""CI --rebuild indexes README/PLAN/docs/skills; fragments must be there."""
corpus = []
for rel in ("README.md", "PLAN.md", "AGENTS.md"):
corpus.append((ROOT / rel).read_text())
for d in ("docs", "skills"):
for p in (ROOT / d).rglob("*.md"):
corpus.append(p.read_text())
blob = "\n".join(corpus)
for frag in ("BM25", "DevOps", "LadybugDB"):
self.assertIn(frag, blob, f"{frag} must appear in default index corpus")
-56
View File
@@ -39,59 +39,3 @@ class IndexAdapterTest(unittest.TestCase):
self.assertTrue(msg.get("dry_run"))
self.assertGreaterEqual(msg.get("corpus_total", 0), 1)
self.assertFalse(lbug.exists(), "dry-run must not create a Ladybug file")
def test_facts_json_and_chats_land_on_rebuild(self) -> None:
"""Gitea #18: facts (2-source) + chats markdown become leafs on rebuild."""
tmp = Path(tempfile.mkdtemp())
dbpath = tmp / "kb.lbug"
chats = tmp / "chats"
chats.mkdir()
(chats / "alice.md").write_text(
"# Chat\n\n## Alice and Bob\n\nhello from chats fixture unique-chat-token\n",
encoding="utf-8",
)
facts_path = tmp / "facts.json"
facts_path.write_text(json.dumps([{
"text": "container 'brain' unique-fact-token is running and declared in compose.yaml",
"source": "docker ps x compose.yaml",
"loc": "compose.yaml:brain",
"how": "facts/extract",
}]), encoding="utf-8")
venv_py = ROOT / ".venv" / "bin" / "python"
py = str(venv_py) if venv_py.is_file() else sys.executable
proc = subprocess.run(
[
py, str(ROOT / "bin" / "kb" / "index"),
"--rebuild", "--db", str(dbpath), "--no-defaults",
"--with-chats", str(chats),
"--facts-json", str(facts_path),
"--json",
],
cwd=ROOT,
capture_output=True,
text=True,
env=os.environ.copy(),
check=False,
)
self.assertEqual(proc.returncode, 0, proc.stderr)
msg = json.loads(proc.stdout)
self.assertGreaterEqual(msg.get("facts_leafs", 0), 1)
self.assertGreaterEqual(msg.get("chat_leafs", 0), 1)
self.assertTrue(dbpath.exists())
sys.path.insert(0, str(ROOT / "bin" / "tools"))
import kblib
db, conn = kblib.connect(dbpath, read_only=True)
try:
stats = kblib.stats(conn)
self.assertGreaterEqual(stats["by_root"].get("facts", 0), 1)
fts = kblib.query_fts(conn, "unique-chat-token", 5)
self.assertTrue(fts, "chats markdown must be FTS-searchable")
fact_hits = kblib.query_fts(conn, "unique-fact-token", 5)
self.assertTrue(any(h.get("root") == "facts" for h in fact_hits))
src = conn.execute(
"MATCH (l:Leaf {root:'facts'}) RETURN l.source"
).get_all()
self.assertTrue(any(" x " in str(r[0]) for r in src))
finally:
conn.close()
db.close()
-89
View File
@@ -8,13 +8,10 @@ from pathlib import Path
sys.path.insert(0, os.path.dirname(__file__))
from mailconv import ( # noqa: E402
TESS_LANG,
clean_email_address,
convert_pdf,
html_to_markdown,
is_convertible,
normalize_markdown,
ocr_image,
split_zip_members,
subject_to_filename,
zip_extract_safe,
@@ -103,92 +100,6 @@ class TestMailConv(unittest.TestCase):
self.assertFalse(is_convertible(".exe"))
self.assertFalse(is_convertible(".unknown"))
def test_convert_pdf_prefers_pdftotext(self):
import mailconv as mc
calls: list[list[str]] = []
def fake_run(cmd, **kwargs):
calls.append(list(cmd))
class P:
returncode = 0
stdout = b"Invoice BM25 layout"
stderr = b""
return P()
self._patch_run(mc, fake_run)
out = convert_pdf(Path(self._tmp("born.pdf")))
self.assertIn("BM25", out)
self.assertEqual(calls[0][:2], ["pdftotext", "-layout"])
self.assertFalse(any(c[0] == "tesseract" for c in calls))
self.assertFalse(any(c[0] == "pdftoppm" for c in calls))
def test_convert_pdf_empty_layer_uses_pdftoppm_tesseract(self):
import mailconv as mc
calls: list[list[str]] = []
def fake_run(cmd, **kwargs):
calls.append(list(cmd))
class P:
returncode = 0
stdout = b""
stderr = b""
if cmd[0] == "pdftotext":
P.stdout = b" \n"
return P()
if cmd[0] == "pdftoppm":
prefix = Path(cmd[-1])
(prefix.parent / "page-1.png").write_bytes(b"fake")
return P()
if cmd[0] == "tesseract":
P.stdout = b"scanned HELLO"
return P()
return P()
self._patch_run(mc, fake_run)
out = convert_pdf(Path(self._tmp("scan.pdf")))
self.assertIn("HELLO", out)
bins = [c[0] for c in calls]
self.assertIn("pdftotext", bins)
self.assertIn("pdftoppm", bins)
self.assertIn("tesseract", bins)
tess = next(c for c in calls if c[0] == "tesseract")
self.assertIn(TESS_LANG, tess)
self.assertNotIn("docling", " ".join(bins))
def test_ocr_image_paddle_engine(self):
import mailconv as mc
calls: list[list[str]] = []
def fake_run(cmd, **kwargs):
calls.append(list(cmd))
class P:
returncode = 0
stdout = b"paddle text"
stderr = b""
return P()
self._patch_run(mc, fake_run)
os.environ["OCR_ENGINE"] = "paddle"
try:
out = ocr_image(Path(self._tmp("x.png")))
finally:
os.environ.pop("OCR_ENGINE", None)
self.assertEqual(out, "paddle text")
self.assertEqual(calls[0][:2], ["paddleocr", "ocr"])
def _patch_run(self, mod, fn) -> None:
self.addCleanup(setattr, mod.subprocess, "run", mod.subprocess.run)
mod.subprocess.run = fn
def _mk_zip(self, members):
zpath = Path(self._tmp("arc.zip"))
with zipfile.ZipFile(zpath, "w") as zf:
-10
View File
@@ -5,7 +5,6 @@
# docker compose --profile picoclaw up brain-mcp
# docker compose --profile reasoner up -d reasoner # CPU Ollama :11435
# docker compose --profile searxng up -d
# OCR_ENGINE=paddle docker compose --profile ocr-paddle run --rm ocr-paddle
#
# Secrets never baked in: search.env + db-profiles.yml from ~/.config/brain.
@@ -132,15 +131,6 @@ services:
- reasoner-ollama:/root/.ollama
restart: unless-stopped
# Optional PP-OCRv5 (not default). Default OCR is tesseract eng+deu.
# OCR_ENGINE=paddle docker compose --profile ocr-paddle run --rm ocr-paddle
ocr-paddle:
profiles: ["ocr-paddle"]
image: python:3.12-slim
environment:
OCR_ENGINE: paddle
command: ["python", "-c", "print('OCR_ENGINE=paddle; install paddleocr on PATH')"]
volumes:
kb-model:
kb-var:
+1 -2
View File
@@ -21,5 +21,4 @@ OO_CLI (default: $HOME/go/bin/oo)
./bin/chats/apply.go --dry-run
```
JSONL → markdown only. Brain ingest is `bin/brain/index.go --with-chats`
(default `var/chats/md`). WhatsApp sync is out of v1.
JSONL → markdown only. Brain ingest is `bin/brain/index.go` (not a `chats index`).
+12 -18
View File
@@ -28,23 +28,9 @@ Compose `api` (no CPython) / `index` (Python write). Issues #1#5, #7#13.
`POST /ingest` (Python `kblib.add_leafs`; no Go upsert port).
[#17](https://git.produktor.io/eSlider/2dph/issues/17) `--hop N` walks
FROM_FILE / HAS_VERSION / AUTHORED.
[#18](https://git.produktor.io/eSlider/2dph/issues/18) `--with-facts` /
`--with-chats` on rebuild (WhatsApp out of v1).
[#19](https://git.produktor.io/eSlider/2dph/issues/19) CI recall SoT =
`bin/brain/eval.go` via Zig.
Epic [#16](https://git.produktor.io/eSlider/2dph/issues/16) closed.
## v2
[#6](https://git.produktor.io/eSlider/2dph/issues/6) OCR — `pdftotext` then
`pdftoppm` + tesseract `eng+deu`. Optional `ocr-paddle`.
[#29](https://git.produktor.io/eSlider/2dph/issues/29) OQ1 contradiction
resolution. [#30](https://git.produktor.io/eSlider/2dph/issues/30) OQ3 duckdb-md.
## Blockers
None for epic #16 (closed). Remaining v2: OQ1, OQ3, OQ4.
```
question
@@ -53,14 +39,22 @@ question
├─ web (D17) ← in
├─ brain/add ACID ← in
├─ Cypher hop ← in
└─ facts+chats corpus ← in
└─ facts+chats corpus ← #18
```
1. **[#18](https://git.produktor.io/eSlider/2dph/issues/18) corpus** —
rebuild loads repo markdown + mail as `info`. `facts/extract` pairing
and `bin/chats` are not indexed. WhatsApp is a stub. PII stays in `var/`.
2. **[#19](https://git.produktor.io/eSlider/2dph/issues/19) CI eval** —
recall SoT should be `bin/brain/eval.go` via Zig, not Python `bin/kb/eval`.
## Not v1
OQ1 contradiction resolution, OQ3 duckdb-md export, OQ4 YAML-first leafs.
OCR (OQ2) is in: tesseract, not docling.
[#6](https://git.produktor.io/eSlider/2dph/issues/6) OCR (OQ2), OQ1
contradiction resolution, OQ3 duckdb-md export, OQ4 YAML-first leafs.
## Close epic #16 when
Children #14, #15, #17, #18, #19 are closed. MCP tool order stays gated by tests.
- ops pairing + chat import land as leafs on rebuild
- MCP tool order is documented and still gated by tests
- CI recall SoT is `bin/brain/eval.go` via Zig
+1 -2
View File
@@ -16,7 +16,6 @@ No laptop-absolute paths. Config lives in env files under `$HOME/.config/brain/`
- Go (see `go.mod`)
- Python 3.12 + [uv](https://docs.astral.sh/uv)
- Optional: Docker, Zig CGO via `bin/cgo/zig` (not gcc)
- Optional: poppler (`pdftotext`/`pdftoppm`) + tesseract `eng+deu` for mail OCR
```bash
uv venv .venv
@@ -50,7 +49,7 @@ corpus rebuild remains `bin/brain/index.go --rebuild` (Compose profile
```bash
bin/brain/add.go --text "arc-1 runs Matrix" --root facts --source "compose.yml x docker ps"
bin/brain/index.go --rebuild --with-facts --with-chats
bin/brain/index.go --rebuild
bin/brain/search.go "LadybugDB vector index" # facts → info → web (D17)
bin/brain/search.go "upstream flag" --no-web
bin/brain/get.go <id> --body
-162
View File
@@ -1,162 +0,0 @@
// Package ocr runs Tesseract (eng+deu) on images and scanned PDFs.
//
// Default engine is the tesseract CLI, not gosseract CGO: Ladybug CGO stays
// Zig-only (D21). Same engine, no gocv. OCR_ENGINE=paddle selects paddleocr
// when that binary is on PATH (compose profile ocr-paddle).
package ocr
import (
"fmt"
"image"
"image/color"
"image/png"
"os"
"os/exec"
"path/filepath"
"strings"
)
const TessLang = "eng+deu"
func ImageFile(path string) (string, error) {
engine := os.Getenv("OCR_ENGINE")
if engine == "paddle" {
return runPaddle(path)
}
return runTesseract(path)
}
func PDFFile(path string) (string, error) {
text, err := pdfToText(path)
if err == nil && strings.TrimSpace(text) != "" {
return strings.TrimSpace(text), nil
}
ocr, oerr := pdfPages(path)
if oerr != nil {
if err != nil {
return "", err
}
return "", oerr
}
if strings.TrimSpace(ocr) != "" {
return strings.TrimSpace(ocr), nil
}
if text != "" {
return strings.TrimSpace(text), nil
}
return "", fmt.Errorf("pdf has no text layer (ocr unavailable)")
}
func pdfToText(path string) (string, error) {
cmd := exec.Command("pdftotext", "-layout", path, "-")
out, err := cmd.Output()
if err != nil {
return "", err
}
return string(out), nil
}
func pdfPages(path string) (string, error) {
dir, err := os.MkdirTemp("", "2dph-ocr-")
if err != nil {
return "", err
}
defer os.RemoveAll(dir)
prefix := filepath.Join(dir, "page")
cmd := exec.Command("pdftoppm", "-png", "-r", "200", path, prefix)
if err := cmd.Run(); err != nil {
return "", err
}
matches, err := filepath.Glob(prefix + "*.png")
if err != nil {
return "", err
}
var parts []string
for _, img := range matches {
t, err := ImageFile(img)
if err != nil {
continue
}
if s := strings.TrimSpace(t); s != "" {
parts = append(parts, s)
}
}
return strings.Join(parts, "\n\n"), nil
}
func runTesseract(path string) (string, error) {
pre, err := preprocessFile(path)
if err != nil {
pre = path
} else {
defer os.Remove(pre)
}
cmd := exec.Command("tesseract", pre, "stdout", "-l", TessLang, "--psm", "6")
out, err := cmd.Output()
if err != nil {
return "", err
}
return strings.TrimSpace(string(out)), nil
}
func runPaddle(path string) (string, error) {
cmd := exec.Command("paddleocr", "ocr", "-i", path)
out, err := cmd.Output()
if err != nil {
return "", err
}
return strings.TrimSpace(string(out)), nil
}
func preprocessFile(path string) (string, error) {
f, err := os.Open(path)
if err != nil {
return "", err
}
defer f.Close()
img, err := png.Decode(f)
if err != nil {
return "", err
}
out := filepath.Join(os.TempDir(), filepath.Base(path)+".gray.png")
w, err := os.Create(out)
if err != nil {
return "", err
}
defer w.Close()
if err := png.Encode(w, GrayContrast(img)); err != nil {
os.Remove(out)
return "", err
}
return out, nil
}
// GrayContrast is a stdlib preprocess (no gocv): grayscale + stretch.
func GrayContrast(src image.Image) image.Image {
b := src.Bounds()
dst := image.NewGray(b)
var minL, maxL uint8 = 255, 0
for y := b.Min.Y; y < b.Max.Y; y++ {
for x := b.Min.X; x < b.Max.X; x++ {
g := color.GrayModel.Convert(src.At(x, y)).(color.Gray)
if g.Y < minL {
minL = g.Y
}
if g.Y > maxL {
maxL = g.Y
}
}
}
span := int(maxL) - int(minL)
if span < 1 {
span = 1
}
for y := b.Min.Y; y < b.Max.Y; y++ {
for x := b.Min.X; x < b.Max.X; x++ {
g := color.GrayModel.Convert(src.At(x, y)).(color.Gray)
v := uint8((int(g.Y) - int(minL)) * 255 / span)
dst.SetGray(x, y, color.Gray{Y: v})
}
}
return dst
}
-54
View File
@@ -1,54 +0,0 @@
package ocr
import (
"image"
"image/color"
"os/exec"
"path/filepath"
"strings"
"testing"
)
func TestGrayContrastStretches(t *testing.T) {
img := image.NewGray(image.Rect(0, 0, 2, 2))
img.SetGray(0, 0, color.Gray{Y: 64})
img.SetGray(0, 1, color.Gray{Y: 64})
img.SetGray(1, 0, color.Gray{Y: 64})
img.SetGray(1, 1, color.Gray{Y: 192})
out := GrayContrast(img).(*image.Gray)
if out.GrayAt(0, 0).Y != 0 {
t.Fatalf("min should map to 0, got %d", out.GrayAt(0, 0).Y)
}
if out.GrayAt(1, 1).Y != 255 {
t.Fatalf("max should map to 255, got %d", out.GrayAt(1, 1).Y)
}
}
func TestHelloPNGFixtureOCR(t *testing.T) {
if _, err := exec.LookPath("tesseract"); err != nil {
t.Skip("tesseract not installed")
}
path := filepath.Join("testdata", "hello.png")
got, err := ImageFile(path)
if err != nil {
t.Fatal(err)
}
up := strings.ToUpper(got)
if !strings.Contains(up, "HELLO") {
t.Fatalf("ocr %q missing HELLO", got)
}
}
func TestPaddleEngineUsesPaddleocrBinary(t *testing.T) {
t.Setenv("OCR_ENGINE", "paddle")
_, err := ImageFile(filepath.Join("testdata", "hello.png"))
if _, look := exec.LookPath("paddleocr"); look != nil {
if err == nil {
t.Fatal("expected error when paddleocr is missing")
}
return
}
if err != nil {
t.Fatal(err)
}
}
BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.7 KiB

+1
View File
@@ -6,6 +6,7 @@ readme = "README.md"
requires-python = ">=3.12"
license = { text = "MIT" }
dependencies = [
"docling>=2.119.0",
"ladybug==0.19.1",
"markitdown[docx,epub,html,image-exif,pdf,pptx,xlsx,zip]>=0.1.7",
"mistune==3.3.4",
Generated
+1648
View File
File diff suppressed because it is too large Load Diff