Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7db9e6b836 |
@@ -45,53 +45,27 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
uv run python -m unittest discover -s bin/tools -t .
|
uv run python -m unittest discover -s bin/tools -t .
|
||||||
|
|
||||||
- name: Go tests (root module; duckdb-go CGO via gcc, no ladybug)
|
- name: Go tests (root module, no ladybug cgo)
|
||||||
run: |
|
run: |
|
||||||
CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= go vet ./...
|
go vet ./...
|
||||||
CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= go test ./... -count=1
|
go test ./... -count=1
|
||||||
|
|
||||||
- name: brain ranking tests (no cgo / no ladybug)
|
- name: brain ranking tests (no cgo / no ladybug)
|
||||||
run: go test ./internal/brain/rank -count=1
|
run: go test ./internal/brain/rank -count=1
|
||||||
|
|
||||||
- name: facts/audit self (lexicon consistency, no network)
|
- name: facts/audit self (lexicon consistency, no network)
|
||||||
run: ./bin/facts/audit self
|
run: |
|
||||||
|
./bin/facts/audit self 2>/dev/null || echo "audit: not yet implemented; gate skipped"
|
||||||
|
|
||||||
- name: CGO via Zig (compile brain/search + eval)
|
- name: CGO via Zig (compile brain/search)
|
||||||
run: |
|
run: |
|
||||||
chmod +x bin/cgo/zig bin/cgo/zcc bin/cgo/zc++
|
chmod +x bin/cgo/zig bin/cgo/zcc bin/cgo/zc++
|
||||||
bin/cgo/zig go build -tags system_ladybug -o /tmp/brain-search ./bin/brain/search.go
|
bin/cgo/zig go build -tags system_ladybug -o /tmp/brain-search ./bin/brain/search.go
|
||||||
bin/cgo/zig go build -tags 'system_ladybug,brain_eval' -o /tmp/brain-eval ./bin/brain/eval.go
|
|
||||||
|
|
||||||
- uses: actions/cache@v4
|
|
||||||
with:
|
|
||||||
path: ~/.cache/huggingface
|
|
||||||
key: ${{ runner.os }}-hf-potion-multilingual-128M
|
|
||||||
|
|
||||||
- name: recall@5 SoT (Zig bin/brain/eval.go)
|
|
||||||
run: |
|
|
||||||
uv run python bin/kb/index --rebuild --json
|
|
||||||
KB_ROOT="$PWD" /tmp/brain-eval --json
|
|
||||||
|
|
||||||
ocr:
|
|
||||||
name: OCR (tesseract fixture)
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- uses: actions/setup-go@v5
|
|
||||||
with:
|
|
||||||
go-version-file: go.mod
|
|
||||||
- name: Install tesseract + poppler
|
|
||||||
run: |
|
|
||||||
sudo apt-get update
|
|
||||||
sudo apt-get install -y --no-install-recommends \
|
|
||||||
tesseract-ocr tesseract-ocr-eng tesseract-ocr-deu poppler-utils
|
|
||||||
- name: Go OCR tests (synthetic HELLO PNG)
|
|
||||||
run: go test ./internal/ocr -count=1
|
|
||||||
|
|
||||||
release:
|
release:
|
||||||
name: Release (semver)
|
name: Release (semver)
|
||||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||||
needs: [test, ocr]
|
needs: test
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
permissions:
|
permissions:
|
||||||
contents: write
|
contents: write
|
||||||
|
|||||||
@@ -12,5 +12,3 @@ __pycache__/
|
|||||||
lib-ladybug/
|
lib-ladybug/
|
||||||
go.work.local
|
go.work.local
|
||||||
models/
|
models/
|
||||||
# Purged from git history. Do not re-add.
|
|
||||||
docs/crm-associations-proof.md
|
|
||||||
|
|||||||
@@ -40,16 +40,15 @@ PLAN.md decisions + execution + open questions
|
|||||||
docs/ published docs
|
docs/ published docs
|
||||||
skills/ in-project agent skills (vendored, no external links)
|
skills/ in-project agent skills (vendored, no external links)
|
||||||
bin/ self-describing tools bin/{subject}/{method}.go (shebang)
|
bin/ self-describing tools bin/{subject}/{method}.go (shebang)
|
||||||
bin/brain/ search.go serve.go index.go add.go get.go stats.go eval.go watch.go
|
bin/brain/ search.go serve.go index.go get.go stats.go eval.go watch.go
|
||||||
bin/chats/ sync.go import.go facts.go apply.go; libs in internal/chats
|
bin/chats/ sync.go import.go facts.go apply.go; libs in internal/chats
|
||||||
bin/mail/ sync.go import.go ocr.go (index_mail → brain/index.go)
|
bin/mail/ sync.go import.go (index_mail → brain/index.go)
|
||||||
bin/markdown/ import.go (H2 leaf split; Python bin/md/import fallback)
|
bin/markdown/ import.go (H2 leaf split; Python bin/md/import fallback)
|
||||||
bin/postgres/ query.go (read-only YAML)
|
bin/postgres/ query.go (read-only YAML)
|
||||||
bin/git/ import.go (go-git history; Python shim execs it)
|
bin/git/ import.go (go-git history; Python shim execs it)
|
||||||
bin/web/ search.go (SearXNG; Python shim execs it)
|
bin/web/ search.go (SearXNG; Python shim execs it)
|
||||||
bin/reasoner/ bakeoff.go (D18 CPU OpenAI tool-call bake-off)
|
bin/reasoner/ bakeoff.go (D18 CPU OpenAI tool-call bake-off)
|
||||||
internal/ shared Go (brain/rank is cgo-free; chats parsers; gitlog; websearch; reasoner; duckstats)
|
internal/ shared Go (brain/rank is cgo-free; chats parsers; gitlog; websearch; reasoner)
|
||||||
bin/qa/ stats.go (DuckDB quantiles / JSONL count; gcc CGO, not Zig)
|
|
||||||
bin/watch/ corpus watcher (used by bin/brain/watch.go)
|
bin/watch/ corpus watcher (used by bin/brain/watch.go)
|
||||||
bin/tools/ vendored python libs behind bin/* (kblib, yamlout, websearch)
|
bin/tools/ vendored python libs behind bin/* (kblib, yamlout, websearch)
|
||||||
bin/cgo/ zig zcc zc++ (CGO via zig cc, not gcc)
|
bin/cgo/ zig zcc zc++ (CGO via zig cc, not gcc)
|
||||||
@@ -66,18 +65,18 @@ var/ kb.lbug, var/mail/*, caches (gitignored)
|
|||||||
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw message.json + attachments
|
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw message.json + attachments
|
||||||
bin/mail/sync.go --source gmail --query 'from:example.com' --out var/mail # Gmail search (default in:inbox)
|
bin/mail/sync.go --source gmail --query 'from:example.com' --out var/mail # Gmail search (default in:inbox)
|
||||||
bin/mail/import.go --from-raw var/mail # message.json → message.md (convert only)
|
bin/mail/import.go --from-raw var/mail # message.json → message.md (convert only)
|
||||||
bin/brain/index.go --rebuild --with-facts --with-chats
|
bin/brain/index.go --rebuild # rebuild brain incl. all mail (fresh DB)
|
||||||
```
|
```
|
||||||
|
|
||||||
- `sync` (Go) downloads messages + attachments; Gmail uses paginated list +
|
- `sync` (Go) downloads messages + attachments; Gmail uses paginated list +
|
||||||
`body.attachmentId` (not partId) for attachments.
|
`body.attachmentId` (not partId) for attachments.
|
||||||
- `import` converts body + attachments to markdown. PDFs use poppler
|
- `import` converts body + attachments to markdown. PDFs use poppler
|
||||||
`pdftotext -layout` fast path (~15ms); textless/scanned PDFs use
|
`pdftotext -layout` fast path (~15ms); textless/scanned PDFs fall back to
|
||||||
`pdftoppm` + tesseract `eng+deu` (`bin/mail/ocr.go`). Optional
|
docling (isolated subprocess — its native onnx can segfault the parent).
|
||||||
`OCR_ENGINE=paddle`. Conversion never touches the brain DB (crash safety).
|
Conversion never touches the brain DB (crash safety).
|
||||||
- `index_mail` is a deprecation shim for `bin/brain/index.go --rebuild`. Bulk
|
- `index_mail` is a deprecation shim for `bin/brain/index.go --rebuild`. Ladybug
|
||||||
rebuild still deletes `var/kb.lbug` and creates FTS/HNSW last. Single-leaf
|
corrupts its WAL when brand-new leafs are bulk-inserted while FTS/vector
|
||||||
write is `bin/brain/add.go` (safe while indexes exist; do not DROP INDEX).
|
indexes exist; a fresh DB with indexes created last is the only safe path.
|
||||||
Keep conversion + indexing separate so a conversion crash can't leave the
|
Keep conversion + indexing separate so a conversion crash can't leave the
|
||||||
DB mid-transaction.
|
DB mid-transaction.
|
||||||
|
|
||||||
@@ -90,9 +89,6 @@ bin/kb/search "query" [--repo X] # deprecated wrapper → bin/b
|
|||||||
bin/brain/search.go "query" [--root facts|info] # deduction search → YAML
|
bin/brain/search.go "query" [--root facts|info] # deduction search → YAML
|
||||||
bin/brain/search.go "query" --no-web # local graph only
|
bin/brain/search.go "query" --no-web # local graph only
|
||||||
eval "$(bin/cgo/zig env)" # Zig cc + liblbug (not gcc)
|
eval "$(bin/cgo/zig env)" # Zig cc + liblbug (not gcc)
|
||||||
bin/brain/index.go --rebuild [--with-mail] [--with-facts] [--with-chats]
|
|
||||||
bin/brain/add.go --text T --root facts --source "a.md x b.md" # incremental write
|
|
||||||
bin/brain/add.go --json # stdin leaf or {leafs:[...]}
|
|
||||||
bin/brain/get.go <id> [--body] [--json] # Go read; Python bin/kb/get CI fallback
|
bin/brain/get.go <id> [--body] [--json] # Go read; Python bin/kb/get CI fallback
|
||||||
bin/brain/stats.go [--json]
|
bin/brain/stats.go [--json]
|
||||||
bin/brain/eval.go [--json] # recall@5; questions in internal/brain/rank
|
bin/brain/eval.go [--json] # recall@5; questions in internal/brain/rank
|
||||||
@@ -102,16 +98,12 @@ bin/git/import.go [REPO] [--json] [--limit N] # go-git history → commit le
|
|||||||
bin/web/search.go "query" [--json] # SearXNG; throttled ≠ absence
|
bin/web/search.go "query" [--json] # SearXNG; throttled ≠ absence
|
||||||
bin/reasoner/bakeoff.go [--model ID] [--json] # D18 CPU tool-call bake-off
|
bin/reasoner/bakeoff.go [--model ID] [--json] # D18 CPU tool-call bake-off
|
||||||
bin/postgres/query.go --profile onlyoffice -c 'SELECT 1'
|
bin/postgres/query.go --profile onlyoffice -c 'SELECT 1'
|
||||||
bin/qa/stats.go # D22 DuckDB quantiles / JSONL (gcc CGO)
|
|
||||||
bin/mail/ocr.go <image|pdf> # tesseract eng+deu (scans)
|
|
||||||
bin/md/tables # what the graph holds → YAML
|
bin/md/tables # what the graph holds → YAML
|
||||||
bin/brain/deduce "question" # thinking wrapper
|
bin/brain/deduce "question" # thinking wrapper
|
||||||
```
|
```
|
||||||
|
|
||||||
Never start a shell command with `cd` — use the tool working-directory
|
Never start a shell command with `cd` — use the tool working-directory
|
||||||
parameter. Search before reading whole files. For YAML/JSON/XML/CSV/TOML/HCL
|
parameter. Search before reading whole files.
|
||||||
prefer mikefarah/yq (`skills/yq/SKILL.md`). For bulk rows and quantiles use
|
|
||||||
duckdb-go (`internal/duckstats`, `skills/duckdb/SKILL.md`), not Ladybug.
|
|
||||||
|
|
||||||
## GitHub safety rules (ABSOLUTE — never violate)
|
## GitHub safety rules (ABSOLUTE — never violate)
|
||||||
|
|
||||||
|
|||||||
@@ -16,10 +16,6 @@ ENV PYTHONUNBUFFERED=1 \
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
RUN id -u 2dph 2>/dev/null || useradd --create-home --uid 1001 2dph
|
RUN id -u 2dph 2>/dev/null || useradd --create-home --uid 1001 2dph
|
||||||
RUN apt-get update \
|
|
||||||
&& apt-get install -y --no-install-recommends \
|
|
||||||
poppler-utils tesseract-ocr tesseract-ocr-eng tesseract-ocr-deu \
|
|
||||||
&& rm -rf /var/lib/apt/lists/*
|
|
||||||
|
|
||||||
COPY requirements.lock.txt /tmp/requirements.lock.txt
|
COPY requirements.lock.txt /tmp/requirements.lock.txt
|
||||||
RUN python -m pip install --no-cache-dir -r /tmp/requirements.lock.txt \
|
RUN python -m pip install --no-cache-dir -r /tmp/requirements.lock.txt \
|
||||||
|
|||||||
@@ -4,9 +4,8 @@ A brain that loves facts and deduction. Evidence-first knowledge graph + hybrid
|
|||||||
RAG over the operational Brain/ops/eSlider stack. Built like Sherlock
|
RAG over the operational Brain/ops/eSlider stack. Built like Sherlock
|
||||||
Holmes: nothing is asserted unless it has proof.
|
Holmes: nothing is asserted unless it has proof.
|
||||||
|
|
||||||
Status: **v1 in** (epic [#16](https://git.produktor.io/eSlider/2dph/issues/16) closed).
|
Status: **in progress** — read path + MCP work; v1 goal is [epic #16](https://git.produktor.io/eSlider/2dph/issues/16)
|
||||||
v2 board: milestone [v2](https://git.produktor.io/eSlider/2dph/milestone/13) — OCR [#6](https://git.produktor.io/eSlider/2dph/issues/6),
|
(milestone [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12)).
|
||||||
[#29](https://git.produktor.io/eSlider/2dph/issues/29) OQ1, [#30](https://git.produktor.io/eSlider/2dph/issues/30) OQ3.
|
|
||||||
Gap: [docs/roadmap.md](docs/roadmap.md).
|
Gap: [docs/roadmap.md](docs/roadmap.md).
|
||||||
|
|
||||||
## What
|
## What
|
||||||
@@ -32,9 +31,9 @@ detective method: **a fact needs ≥2 independent sources or it is
|
|||||||
| D3 | web search | Go client `bin/web/search.go` (`internal/websearch`). SearXNG URL is config (`BRAIN_SEARCH_URL`). Optional Compose profile `searxng` (sanitized settings). Do not run a second copy on a host that already has one. Empty/`throttled` ≠ “nothing exists”. |
|
| D3 | web search | Go client `bin/web/search.go` (`internal/websearch`). SearXNG URL is config (`BRAIN_SEARCH_URL`). Optional Compose profile `searxng` (sanitized settings). Do not run a second copy on a host that already has one. Empty/`throttled` ≠ “nothing exists”. |
|
||||||
| D4 | embeddings | **model2vec** `minishlab/potion-multilingual-128M` instead of embeddinggemma. |
|
| D4 | embeddings | **model2vec** `minishlab/potion-multilingual-128M` instead of embeddinggemma. |
|
||||||
| D5 | parser | **mistune** for MD → leaf extraction (duckdb-md documented as future optional SQL/export layer, not v1). |
|
| D5 | parser | **mistune** for MD → leaf extraction (duckdb-md documented as future optional SQL/export layer, not v1). |
|
||||||
| D6 | graph engine | **LadybugDB**. Go is the service (`bin/brain/search.go`, `bin/brain/serve.go` in-process, `internal/brain`). Read path is Go + Zig CGO (D21). Python `bin/kb/{get,stats,eval}` is the CI fallback when Zig/libs are not fetched. Incremental write is Python `bin/kb/add` (`bin/brain/add.go`). Bulk rebuild stays `compose --profile index` until the Go write path is safe. |
|
| D6 | graph engine | **LadybugDB**. Go is the service (`bin/brain/search.go`, `bin/brain/serve.go` in-process, `internal/brain`). Read path is Go + Zig CGO (D21). Python `bin/kb/{get,stats,eval}` is the CI fallback when Zig/libs are not fetched. Index/write stays Python (`compose --profile index`) until the Go write path is safe. |
|
||||||
| D7 | db access | `db-yaml`/`psql-yq`-style, read-only, YAML out. OnlyOffice Postgres via SSH tunnel (`127.0.0.1:5433`). |
|
| D7 | db access | `db-yaml`/`psql-yq`-style, read-only, YAML out. OnlyOffice Postgres via SSH tunnel (`127.0.0.1:5433`). |
|
||||||
| D8 | evidence | detective method: ≥2 independent sources or `(not confirmed)`. 2-source auto-pair docker ps × compose × ssh-config × docs. |
|
| D8 | evidence | detective method: ≥2 independent sources or `(not confirmed)`. Auto-pair docker ps × compose × ssh-config × docs. |
|
||||||
| D9 | facts/goal model | Who / What / How / Where / When + evidence + confidence on every edge. |
|
| D9 | facts/goal model | Who / What / How / Where / When + evidence + confidence on every edge. |
|
||||||
| D10 | versioning | everything is a leaf with `sha256 + observed_at + source_rev`; `File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person`. Stale = `source_rev` < git HEAD. |
|
| D10 | versioning | everything is a leaf with `sha256 + observed_at + source_rev`; `File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person`. Stale = `source_rev` < git HEAD. |
|
||||||
| D11 | strong/weak | `root` column: `facts` (strong) vs `info` (weak). Answer is `confirmed` only from facts root. |
|
| D11 | strong/weak | `root` column: `facts` (strong) vs `info` (weak). Answer is `confirmed` only from facts root. |
|
||||||
@@ -46,9 +45,8 @@ detective method: **a fact needs ≥2 independent sources or it is
|
|||||||
| D17 | assertion gate | Fact-check every *claim* (facts → info → live → web), not every edit. `bin/brain/search.go` adds a `web` block when there is no facts hit (`throttled`/`skipped`/`refused` ≠ absence). `--root` and `--no-web` stay local. Missing graph ≠ “does not exist”. |
|
| D17 | assertion gate | Fact-check every *claim* (facts → info → live → web), not every edit. `bin/brain/search.go` adds a `web` block when there is no facts hit (`throttled`/`skipped`/`refused` ≠ absence). `--root` and `--no-web` stay local. Missing graph ≠ “does not exist”. |
|
||||||
| D18 | reasoner | Pluggable OpenAI-compatible URL (`REASONER_BASE_URL`). RAM: `Qwen/Qwen3.5-9B`. Quality: `prism-ml/Bonsai-27B-gguf` or `Qwen/Qwen3.6-27B`. No official Qwen3.6-9B. CPU bake-off: `bin/reasoner/bakeoff.go` + compose profile `reasoner` (`OLLAMA_NUM_GPU=0`, `:11435`). PicoClaw is compose profile `picoclaw`; tools are `search`/`get`/`audit`. Weights are not copied into the 2dph image. Agent lever/loop: [#15](https://git.produktor.io/eSlider/2dph/issues/15). |
|
| D18 | reasoner | Pluggable OpenAI-compatible URL (`REASONER_BASE_URL`). RAM: `Qwen/Qwen3.5-9B`. Quality: `prism-ml/Bonsai-27B-gguf` or `Qwen/Qwen3.6-27B`. No official Qwen3.6-9B. CPU bake-off: `bin/reasoner/bakeoff.go` + compose profile `reasoner` (`OLLAMA_NUM_GPU=0`, `:11435`). PicoClaw is compose profile `picoclaw`; tools are `search`/`get`/`audit`. Weights are not copied into the 2dph image. Agent lever/loop: [#15](https://git.produktor.io/eSlider/2dph/issues/15). |
|
||||||
| D19 | git history | [go-git](https://github.com/go-git/go-git) via `bin/git/import.go`. No subprocess of the git binary. Conversion prints commit leafs; brain write is `bin/brain/index.go`. |
|
| D19 | git history | [go-git](https://github.com/go-git/go-git) via `bin/git/import.go`. No subprocess of the git binary. Conversion prints commit leafs; brain write is `bin/brain/index.go`. |
|
||||||
| D20 | agent API | OpenAPI + MCP are generated from the same `internal/httpapi.Ops` table as `bin/brain/serve.go` handlers. `GET /openapi.json`, `POST /mcp` (JSON-RPC tools/list + tools/call). Tool names match OpenAPI paths (`search`/`get`/`stats`/`audit`/`ingest`). |
|
| D20 | agent API | OpenAPI + MCP are generated from the same `internal/httpapi.Ops` table as `bin/brain/serve.go` handlers. `GET /openapi.json`, `POST /mcp` (JSON-RPC tools/list + tools/call). Tool names match OpenAPI paths (`search`/`get`/`stats`/`audit`). |
|
||||||
| D21 | CGO | Ladybug/tokenizers CGO is compiled with **Zig** (`bin/cgo/zcc` → `zig cc -target …-linux-gnu`), not gcc. `bin/cgo/zig` pins Zig 0.14.1 + liblbug 0.19.1 + libtokenizers 1.27.0. Compose `target: api` has no CPython; write/rebuild is profile `index`. |
|
| D21 | CGO | Ladybug/tokenizers CGO is compiled with **Zig** (`bin/cgo/zcc` → `zig cc -target …-linux-gnu`), not gcc. `bin/cgo/zig` pins Zig 0.14.1 + liblbug 0.19.1 + libtokenizers 1.27.0. Compose `target: api` has no CPython; write/rebuild is profile `index`. |
|
||||||
| D22 | analytics | **duckdb-go** in-process (`internal/duckstats`, `bin/qa/stats.go`) for quantiles/JSONL. Links with **gcc/g++**, not Zig. Ladybug stays the graph; web-search cache stays modernc sqlite. Slice small structured docs with **mikefarah/yq**, not kislyuk/jq. [#30](https://git.produktor.io/eSlider/2dph/issues/30). |
|
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
@@ -59,10 +57,8 @@ detective method: **a fact needs ≥2 independent sources or it is
|
|||||||
skills/ in-project skills (web-search, postgres, brain, picoclaw, diataxis-docs)
|
skills/ in-project skills (web-search, postgres, brain, picoclaw, diataxis-docs)
|
||||||
bin/
|
bin/
|
||||||
facts/extract.go audit.go crm.go # D14 shebang; Python implementation
|
facts/extract.go audit.go crm.go # D14 shebang; Python implementation
|
||||||
kb/index Python bulk write (called by bin/brain/index.go)
|
kb/index Python write path (called by bin/brain/index.go)
|
||||||
kb/add Python incremental write (called by bin/brain/add.go)
|
|
||||||
brain/index.go rebuild FTS + HNSW (incl. --with-mail)
|
brain/index.go rebuild FTS + HNSW (incl. --with-mail)
|
||||||
brain/add.go incremental leaf write (no rebuild)
|
|
||||||
brain/get.go stats.go eval.go # Go read (cgo); Python bin/kb/* CI fallback
|
brain/get.go stats.go eval.go # Go read (cgo); Python bin/kb/* CI fallback
|
||||||
brain/watch.go
|
brain/watch.go
|
||||||
brain/search.go deduction: facts → info → web-search
|
brain/search.go deduction: facts → info → web-search
|
||||||
@@ -76,7 +72,6 @@ detective method: **a fact needs ≥2 independent sources or it is
|
|||||||
reasoner/bakeoff.go CPU tool-call bake-off (D18; OpenAI tools)
|
reasoner/bakeoff.go CPU tool-call bake-off (D18; OpenAI tools)
|
||||||
chats/sync.go import.go facts.go apply.go
|
chats/sync.go import.go facts.go apply.go
|
||||||
(libs in internal/chats; no chats index)
|
(libs in internal/chats; no chats index)
|
||||||
mail/ocr.go tesseract eng+deu (pdftoppm scans)
|
|
||||||
md/import (deprecated; bin/markdown/import.go)
|
md/import (deprecated; bin/markdown/import.go)
|
||||||
brain/extract brain/audit brain/deduce (thinking wrapper)
|
brain/extract brain/audit brain/deduce (thinking wrapper)
|
||||||
web/search (deprecated shim → web/search.go)
|
web/search (deprecated shim → web/search.go)
|
||||||
@@ -91,9 +86,7 @@ detective method: **a fact needs ≥2 independent sources or it is
|
|||||||
Node tables: `Person, Service, Host, Container, Repo, File, Commit, Leaf`.
|
Node tables: `Person, Service, Host, Container, Repo, File, Commit, Leaf`.
|
||||||
`Leaf(embedding FLOAT[N])` — FTS on `text`, HNSW vector index on `embedding`.
|
`Leaf(embedding FLOAT[N])` — FTS on `text`, HNSW vector index on `embedding`.
|
||||||
Edges: `RUNS / USES / FROM_FILE / HAS_VERSION / AUTHORED / ABOUT / ASSOCIATED / SIMILAR_0.85`.
|
Edges: `RUNS / USES / FROM_FILE / HAS_VERSION / AUTHORED / ABOUT / ASSOCIATED / SIMILAR_0.85`.
|
||||||
`FROM_FILE` / `HAS_VERSION` / `AUTHORED`: `bin/brain/search.go --hop N` walks
|
`FROM_FILE` / `HAS_VERSION` exist in schema; search `--hop` does not walk them yet ([#17](https://git.produktor.io/eSlider/2dph/issues/17)).
|
||||||
them from each hit (1=File, 2=Commit, 3=Person). Rebuild writes
|
|
||||||
`Leaf-[:FROM_FILE]->File`; git import writes the rest.
|
|
||||||
|
|
||||||
Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
||||||
`where`, `when`, `source_rev`.
|
`where`, `when`, `source_rev`.
|
||||||
@@ -111,20 +104,17 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
|||||||
|
|
||||||
- `bin/{subject}/{method}` — line 2 is a usage comment (mirrors `psql-yq`).
|
- `bin/{subject}/{method}` — line 2 is a usage comment (mirrors `psql-yq`).
|
||||||
- bash + python primary; golang via Go shebang when a compiled helper is right.
|
- bash + python primary; golang via Go shebang when a compiled helper is right.
|
||||||
- YAML default output, `--json` for machines. Slice with mikefarah/yq.
|
- YAML default output, `--json` for machines. Slice with `yq`.
|
||||||
- Everything that touches the network / DB is read-only, throttled, cached.
|
- Everything that touches the network / DB is read-only, throttled, cached.
|
||||||
- Tests (TDD) gate every commit; `gh` + CI/CD on every push.
|
- Tests (TDD) gate every commit; `gh` + CI/CD on every push.
|
||||||
|
|
||||||
## Open questions (v2)
|
## Open questions (v2)
|
||||||
|
|
||||||
- OQ1: mutually-contradicting evidence — how to resolve (authority weighting,
|
- OQ1: mutually-contradicting evidence — how to resolve (authority weighting,
|
||||||
temporal freshness, audit adjudication). **v2**; [#29](https://git.produktor.io/eSlider/2dph/issues/29).
|
temporal freshness, audit adjudication). **v2**; does not block epic #16.
|
||||||
- OQ2: OCR — **in**. `pdftotext -layout` first; scans `pdftoppm` + tesseract
|
- OQ2: OCR — poppler `pdftotext` fast-path exists; scans still docling.
|
||||||
`eng+deu` (`bin/mail/ocr.go`, `internal/ocr`). No gocv, no gosseract CGO
|
[#6](https://git.produktor.io/eSlider/2dph/issues/6) (v2, does not block #16).
|
||||||
(D21 Zig owns Ladybug CGO). Optional `OCR_ENGINE=paddle` / compose profile
|
- OQ3: optional duckdb-md layer for `SELECT … FORMAT MARKDOWN` export/write-back.
|
||||||
`ocr-paddle`. Docling left the default path. [#6](https://git.produktor.io/eSlider/2dph/issues/6).
|
|
||||||
- OQ3: **in** — duckdb-go (`internal/duckstats`, `bin/qa/stats.go`) for
|
|
||||||
quantiles / JSONL count. Not a second graph. [#30](https://git.produktor.io/eSlider/2dph/issues/30).
|
|
||||||
- OQ4: YAML-first storage for leafs — deferred: JSON is ~10x faster to
|
- OQ4: YAML-first storage for leafs — deferred: JSON is ~10x faster to
|
||||||
serialize and unambiguous; YAML only where humans edit files.
|
serialize and unambiguous; YAML only where humans edit files.
|
||||||
|
|
||||||
@@ -133,8 +123,7 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
|||||||
1. `bin/mail/sync.go` (Go, 8 workers) — paginated Gmail/OnlyOffice download.
|
1. `bin/mail/sync.go` (Go, 8 workers) — paginated Gmail/OnlyOffice download.
|
||||||
Gmail attachments key off `body.attachmentId`, not MIME `partId`.
|
Gmail attachments key off `body.attachmentId`, not MIME `partId`.
|
||||||
2. `bin/mail/import.go --from-raw` — message.json → message.md; PDFs via
|
2. `bin/mail/import.go --from-raw` — message.json → message.md; PDFs via
|
||||||
`pdftotext -layout` (~15ms); textless/scanned PDFs `pdftoppm` + tesseract
|
`pdftotext -layout` (~15ms) with docling subprocess fallback; ICS sidecars
|
||||||
`eng+deu`. ICS sidecars
|
|
||||||
Latin-1→UTF-8 normalized.
|
Latin-1→UTF-8 normalized.
|
||||||
3. `bin/brain/index.go --rebuild` — fresh rebuild (repo corpus + mail) because ladybug
|
3. `bin/brain/index.go --rebuild` — fresh rebuild (repo corpus + mail) because ladybug
|
||||||
corrupts its WAL on bulk-insert into an already-indexed DB. Conversion and
|
corrupts its WAL on bulk-insert into an already-indexed DB. Conversion and
|
||||||
@@ -151,8 +140,8 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
|||||||
2. `go test ./internal/brain/rank` (cgo-free ranking + flag parser)
|
2. `go test ./internal/brain/rank` (cgo-free ranking + flag parser)
|
||||||
3. python -m unittest discover -s bin/tools (includes published-docs SoT)
|
3. python -m unittest discover -s bin/tools (includes published-docs SoT)
|
||||||
4. `bin/facts/audit self` (lexicon internal consistency; `bin/facts/audit.go` is the D14 wrapper)
|
4. `bin/facts/audit self` (lexicon internal consistency; `bin/facts/audit.go` is the D14 wrapper)
|
||||||
5. `bin/brain/eval.go` via Zig (recall@5 ≥ 0.95). Python `bin/kb/eval` is an
|
5. `bin/kb/eval` (recall@5 ≥ 0.95). Local SoT is `bin/brain/eval.go` via Zig CGO.
|
||||||
explicit fallback, not the CI SoT.
|
CI SoT switch: [#19](https://git.produktor.io/eSlider/2dph/issues/19).
|
||||||
6. `bin/cgo/zig go build -tags system_ladybug` (compile search with zig cc; fetches pinned zig+libs).
|
6. `bin/cgo/zig go build -tags system_ladybug` (compile search with zig cc; fetches pinned zig+libs).
|
||||||
|
|
||||||
Feedback loop: every commit → PR → CI → green/gate → merge. Same discipline as
|
Feedback loop: every commit → PR → CI → green/gate → merge. Same discipline as
|
||||||
@@ -166,22 +155,23 @@ Feedback loop: every commit → PR → CI → green/gate → merge. Same discipl
|
|||||||
4. .venv: ladybug + model2vec + mistune
|
4. .venv: ladybug + model2vec + mistune
|
||||||
5. schema + tools with TDD (kb + md + facts + brain)
|
5. schema + tools with TDD (kb + md + facts + brain)
|
||||||
6. ~/.config/brain config
|
6. ~/.config/brain config
|
||||||
7. corpus extraction (facts/info) — **in**: [#18](https://git.produktor.io/eSlider/2dph/issues/18)
|
7. corpus extraction (facts/info) — **open**: [#18](https://git.produktor.io/eSlider/2dph/issues/18)
|
||||||
8. verify: web-search smoke, onlyoffice pg, md-db round-trip, eval, audit
|
8. verify: web-search smoke, onlyoffice pg, md-db round-trip, eval, audit
|
||||||
|
|
||||||
## Gap to v1 (epic #16)
|
## Gap to v1 (epic #16)
|
||||||
|
|
||||||
Remaining: none for epic #16 (v1). Board:
|
Read path + MCP are in. The detective brain is not closed until the graph is
|
||||||
|
**writable incrementally** and search can **walk** it. Board:
|
||||||
[epic #16](https://git.produktor.io/eSlider/2dph/issues/16),
|
[epic #16](https://git.produktor.io/eSlider/2dph/issues/16),
|
||||||
milestone [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12).
|
milestone [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12).
|
||||||
Narrative: [docs/roadmap.md](docs/roadmap.md).
|
Narrative: [docs/roadmap.md](docs/roadmap.md).
|
||||||
|
|
||||||
| Order | Issue | Gap |
|
| Order | Issue | Gap |
|
||||||
|-------|-------|-----|
|
|-------|-------|-----|
|
||||||
| 1 | [#14](https://git.produktor.io/eSlider/2dph/issues/14) | **in** — `bin/brain/add.go` / `POST /ingest` write facts+info without deleting `kb.lbug`. Bulk corpus still `--rebuild`. Leftover Python (mail/facts) is not the living-graph blocker. |
|
| 1 | [#14](https://git.produktor.io/eSlider/2dph/issues/14) | Write stays Python rebuild; `brain/add` / `POST /ingest` are hints. Ladybug 0.19 WAL corrupts if new leafs land while FTS/HNSW exist. |
|
||||||
| 2 | [#17](https://git.produktor.io/eSlider/2dph/issues/17) | **in** — `--hop N` walks `FROM_FILE` → `HAS_VERSION` → `AUTHORED` (max 3). |
|
| 2 | [#17](https://git.produktor.io/eSlider/2dph/issues/17) | `--hop` errors. `FROM_FILE` / `HAS_VERSION` are in schema; search does not walk them. |
|
||||||
| 3 | [#18](https://git.produktor.io/eSlider/2dph/issues/18) | **in** — `--with-facts` / `--facts-json` land `root=facts`; `--with-chats` indexes `var/chats/md`. WhatsApp sync is out of v1. |
|
| 3 | [#18](https://git.produktor.io/eSlider/2dph/issues/18) | Rebuild is mostly `info` (repo md + mail). `facts/extract` and chats are not a first-class index input. WhatsApp sync is a stub. |
|
||||||
| 4 | [#15](https://git.produktor.io/eSlider/2dph/issues/15) | **in** — lever/loop documented (`search` → `get` → `audit`). |
|
| 4 | [#15](https://git.produktor.io/eSlider/2dph/issues/15) | Lever = 2dph fact-check. Loop = PicoClaw/MCP `search` → `get` → `audit`. Specify in-repo, not only live config. |
|
||||||
| 5 | [#19](https://git.produktor.io/eSlider/2dph/issues/19) | **in** — CI recall SoT is `bin/brain/eval.go` via Zig. Python `bin/kb/eval` stays as an explicit fallback. |
|
| 5 | [#19](https://git.produktor.io/eSlider/2dph/issues/19) | GitHub CI recall still runs Python `bin/kb/eval`. |
|
||||||
|
|
||||||
Does **not** block epic close: OQ1 [#29](https://git.produktor.io/eSlider/2dph/issues/29), OQ3 [#30](https://git.produktor.io/eSlider/2dph/issues/30), OQ4. OCR [#6](https://git.produktor.io/eSlider/2dph/issues/6) is **in**.
|
Does **not** block epic close: [#6](https://git.produktor.io/eSlider/2dph/issues/6) OCR, OQ1, OQ3, OQ4.
|
||||||
@@ -96,7 +96,7 @@ bin/brain/stats.go # index health
|
|||||||
bin/brain/eval.go # recall@5 gate
|
bin/brain/eval.go # recall@5 gate
|
||||||
```
|
```
|
||||||
|
|
||||||
`--hop N` walks File/Commit/Person from each hit (max 3). `bin/kb/search` is a deprecated wrapper around `bin/brain/search.go`.
|
`--hop` is not implemented (needs File/FROM_FILE edges); the flag errors instead of walking. `bin/kb/search` is a deprecated wrapper around `bin/brain/search.go`.
|
||||||
|
|
||||||
Git history is read with [go-git](https://github.com/go-git/go-git) (no git binary):
|
Git history is read with [go-git](https://github.com/go-git/go-git) (no git binary):
|
||||||
|
|
||||||
@@ -120,8 +120,6 @@ Mail is a first-class corpus (retrievable through the same search):
|
|||||||
```bash
|
```bash
|
||||||
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw sync (Go)
|
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw sync (Go)
|
||||||
bin/mail/import.go --from-raw var/mail # JSON → markdown
|
bin/mail/import.go --from-raw var/mail # JSON → markdown
|
||||||
bin/brain/add.go --text T --root facts --source "a.md x b.md"
|
|
||||||
bin/brain/index.go --rebuild --with-facts --with-chats # facts extract + chats md
|
|
||||||
bin/brain/index.go --rebuild # rebuild brain (incl. mail)
|
bin/brain/index.go --rebuild # rebuild brain (incl. mail)
|
||||||
bin/brain/search.go "invoice from last week" # same search over mail leafs
|
bin/brain/search.go "invoice from last week" # same search over mail leafs
|
||||||
```
|
```
|
||||||
@@ -130,9 +128,8 @@ bin/brain/search.go "invoice from last week" # same s
|
|||||||
|
|
||||||
- **LadybugDB** — single `var/kb.lbug`, Cypher + HNSW + BM25, embedded.
|
- **LadybugDB** — single `var/kb.lbug`, Cypher + HNSW + BM25, embedded.
|
||||||
Read tools (`get` / `stats` / `eval`) are Go + Zig CGO (`bin/cgo/zcc`).
|
Read tools (`get` / `stats` / `eval`) are Go + Zig CGO (`bin/cgo/zcc`).
|
||||||
Python fallbacks stay for CI until the runner fetches Zig. Incremental
|
Python fallbacks stay for CI until the runner fetches Zig. Write is
|
||||||
write is `bin/brain/add.go` (Python `kblib.add_leafs`). Bulk rebuild is
|
Compose profile `index` (`bin/brain/index.go`).
|
||||||
Compose profile `index` (`bin/brain/index.go --rebuild`).
|
|
||||||
- **model2vec** — `potion-multilingual-128M` (256-dim), CPU, no Ollama
|
- **model2vec** — `potion-multilingual-128M` (256-dim), CPU, no Ollama
|
||||||
runtime dependency.
|
runtime dependency.
|
||||||
- facts and info split by `root` but written in the same transaction.
|
- facts and info split by `root` but written in the same transaction.
|
||||||
@@ -169,9 +166,6 @@ docker compose up brain-watch # auto re-index on change
|
|||||||
|
|
||||||
## Related
|
## Related
|
||||||
|
|
||||||
eSlider DevOps engineer practice: ops, OnlyOffice, and mail feed the facts
|
|
||||||
root through `bin/facts/extract` (two-source pairing).
|
|
||||||
|
|
||||||
- [go-second-brain](https://github.com/eSlider/go-second-brain) — the earlier
|
- [go-second-brain](https://github.com/eSlider/go-second-brain) — the earlier
|
||||||
Neo4j + Qdrant + Matrix RAG brain
|
Neo4j + Qdrant + Matrix RAG brain
|
||||||
- [agent-skills](https://github.com/eSlider/agent-skills) — upstream
|
- [agent-skills](https://github.com/eSlider/agent-skills) — upstream
|
||||||
|
|||||||
@@ -1,21 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=brain_add "$0" "$@"; exit
|
|
||||||
//go:build brain_add
|
|
||||||
//
|
|
||||||
// bin/brain/add.go - incremental leaf write (Python kblib, no rebuild).
|
|
||||||
//
|
|
||||||
// ./bin/brain/add.go --text T --root facts --source "a.md x b.md"
|
|
||||||
// ./bin/brain/add.go --json
|
|
||||||
//
|
|
||||||
// D6: write stays Python. Does not delete var/kb.lbug.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/cmdbin"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(cmdbin.ExecFile("bin/kb/add", os.Args[1:]))
|
|
||||||
}
|
|
||||||
+3
-3
@@ -3,12 +3,12 @@
|
|||||||
//
|
//
|
||||||
// bin/brain/index.go - rebuild the Ladybug graph (Python write path).
|
// bin/brain/index.go - rebuild the Ladybug graph (Python write path).
|
||||||
//
|
//
|
||||||
// ./bin/brain/index.go --rebuild --with-facts --with-chats
|
// ./bin/brain/index.go --rebuild
|
||||||
// ./bin/brain/index.go --rebuild --with-mail
|
// ./bin/brain/index.go --rebuild --with-mail
|
||||||
// ./bin/brain/index.go --dry-run --with-mail
|
// ./bin/brain/index.go --dry-run --with-mail
|
||||||
//
|
//
|
||||||
// v1 write: bin/brain/add.go for one/few leafs (indexes may already exist).
|
// v1 write is always a rebuild when mail is included (live FTS/HNSW + bulk
|
||||||
// Bulk mail/corpus still --rebuild (fresh file, indexes last).
|
// insert corrupts Ladybug 0.19 WAL). `add` is v2.
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||||
package main
|
package main
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -3,7 +3,7 @@
|
|||||||
//
|
//
|
||||||
// bin/brain/search.go - deduction search over the 2dph brain.
|
// bin/brain/search.go - deduction search over the 2dph brain.
|
||||||
//
|
//
|
||||||
// ./bin/brain/search.go "query" [--root facts|info] [--repo P] [-n N] [--hop N] [--json] [--no-web]
|
// ./bin/brain/search.go "query" [--root facts|info] [--repo P] [-n N] [--json] [--no-web]
|
||||||
// ./bin/brain/search.go serve [port]
|
// ./bin/brain/search.go serve [port]
|
||||||
// ./bin/brain/search.go --list-model
|
// ./bin/brain/search.go --list-model
|
||||||
//
|
//
|
||||||
|
|||||||
+2
-3
@@ -29,11 +29,10 @@ func main() {
|
|||||||
case "linkedin":
|
case "linkedin":
|
||||||
os.Exit(chats.RunSyncLinkedIn(args))
|
os.Exit(chats.RunSyncLinkedIn(args))
|
||||||
case "whatsapp":
|
case "whatsapp":
|
||||||
fmt.Fprintln(os.Stderr, "chats: WhatsApp sync is out of v1")
|
fmt.Fprintln(os.Stderr, "chats: WhatsApp not implemented yet")
|
||||||
os.Exit(1)
|
os.Exit(1)
|
||||||
case "help", "-h", "--help":
|
case "help", "-h", "--help":
|
||||||
fmt.Fprintln(os.Stderr, `usage: bin/chats/sync.go telegram|linkedin [flags]
|
fmt.Fprintln(os.Stderr, `usage: bin/chats/sync.go telegram|linkedin [flags]`)
|
||||||
WhatsApp sync is out of v1.`)
|
|
||||||
return
|
return
|
||||||
default:
|
default:
|
||||||
fmt.Fprintf(os.Stderr, "chats: unknown platform %q\n", platform)
|
fmt.Fprintf(os.Stderr, "chats: unknown platform %q\n", platform)
|
||||||
|
|||||||
-114
@@ -1,114 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""kb/add - incremental leaf write (no rebuild).
|
|
||||||
|
|
||||||
bin/kb/add --text T --root facts|info --source S
|
|
||||||
bin/kb/add --json # stdin: one object or {"leafs":[...]}
|
|
||||||
bin/kb/add --db PATH --json
|
|
||||||
|
|
||||||
Writes facts+info in one Ladybug transaction. Does not delete kb.lbug.
|
|
||||||
Embedding is used when provided; otherwise model2vec encodes the text.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[2]
|
|
||||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
|
||||||
|
|
||||||
from kblib import ( # noqa: E402
|
|
||||||
EMBED_DIM,
|
|
||||||
add_leafs,
|
|
||||||
connect,
|
|
||||||
ensure_indexes,
|
|
||||||
init_schema,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def _as_leafs(payload: object) -> list[dict]:
|
|
||||||
if isinstance(payload, list):
|
|
||||||
return [dict(x) for x in payload]
|
|
||||||
if isinstance(payload, dict):
|
|
||||||
if "leafs" in payload:
|
|
||||||
return [dict(x) for x in payload["leafs"]]
|
|
||||||
return [dict(payload)]
|
|
||||||
raise ValueError("json must be an object, a list, or {leafs:[...]}")
|
|
||||||
|
|
||||||
|
|
||||||
def _embed_missing(leafs: list[dict]) -> None:
|
|
||||||
missing = [lf for lf in leafs if not lf.get("embedding")]
|
|
||||||
if not missing:
|
|
||||||
return
|
|
||||||
from model2vec import StaticModel
|
|
||||||
|
|
||||||
model = StaticModel.from_pretrained("minishlab/potion-multilingual-128M")
|
|
||||||
for lf in missing:
|
|
||||||
text = str(lf.get("text") or "")
|
|
||||||
vec = model.encode([text])[0].astype(float).tolist()
|
|
||||||
if len(vec) != EMBED_DIM:
|
|
||||||
vec = (vec + [0.0] * EMBED_DIM)[:EMBED_DIM]
|
|
||||||
lf["embedding"] = vec
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str]) -> int:
|
|
||||||
import argparse
|
|
||||||
|
|
||||||
p = argparse.ArgumentParser(description="add leafs without rebuilding the brain")
|
|
||||||
p.add_argument("--db", default="", help="path to kb.lbug (default var/kb.lbug)")
|
|
||||||
p.add_argument("--json", action="store_true", help="read leaf JSON from stdin")
|
|
||||||
p.add_argument("--text", default="", help="leaf text")
|
|
||||||
p.add_argument("--root", default="info", choices=("facts", "info"))
|
|
||||||
p.add_argument("--source", default="")
|
|
||||||
p.add_argument("--confidence", default="confirmed")
|
|
||||||
p.add_argument("--source-rev", default="working-tree")
|
|
||||||
p.add_argument("--how", default="brain/add")
|
|
||||||
p.add_argument("--loc", default="")
|
|
||||||
p.add_argument("--type", default="reference", dest="type_")
|
|
||||||
args = p.parse_args(argv)
|
|
||||||
|
|
||||||
if args.json:
|
|
||||||
raw = sys.stdin.read()
|
|
||||||
if not raw.strip():
|
|
||||||
print("kb/add: empty stdin", file=sys.stderr)
|
|
||||||
return 2
|
|
||||||
leafs = _as_leafs(json.loads(raw))
|
|
||||||
else:
|
|
||||||
if not args.text or not args.source:
|
|
||||||
print("kb/add: --text and --source are required (or --json)", file=sys.stderr)
|
|
||||||
return 2
|
|
||||||
leafs = [{
|
|
||||||
"text": args.text,
|
|
||||||
"root": args.root,
|
|
||||||
"source": args.source,
|
|
||||||
"confidence": args.confidence,
|
|
||||||
"source_rev": args.source_rev,
|
|
||||||
"how": args.how,
|
|
||||||
"loc": args.loc or args.source,
|
|
||||||
"type": args.type_,
|
|
||||||
}]
|
|
||||||
|
|
||||||
for lf in leafs:
|
|
||||||
if not lf.get("text") or not lf.get("source"):
|
|
||||||
print("kb/add: each leaf needs text and source", file=sys.stderr)
|
|
||||||
return 2
|
|
||||||
|
|
||||||
_embed_missing(leafs)
|
|
||||||
|
|
||||||
from kblib import DB_PATH, VAR
|
|
||||||
|
|
||||||
dbpath = Path(args.db) if args.db else DB_PATH
|
|
||||||
dbpath.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
VAR.mkdir(exist_ok=True)
|
|
||||||
db, conn = connect(dbpath, read_only=False)
|
|
||||||
init_schema(conn)
|
|
||||||
ids = add_leafs(conn, leafs)
|
|
||||||
ensure_indexes(conn)
|
|
||||||
conn.close()
|
|
||||||
db.close()
|
|
||||||
print(json.dumps({"mode": "add", "ids": ids, "db": str(dbpath)}))
|
|
||||||
return 0
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
sys.exit(main(sys.argv[1:]))
|
|
||||||
+14
-98
@@ -2,15 +2,12 @@
|
|||||||
"""kb/index - build the 2dph brain from markdown + factual leafs.
|
"""kb/index - build the 2dph brain from markdown + factual leafs.
|
||||||
|
|
||||||
bin/kb/index [--corpus DIR] [--rebuild] [--limit N]
|
bin/kb/index [--corpus DIR] [--rebuild] [--limit N]
|
||||||
bin/kb/index --rebuild --with-facts --with-chats
|
|
||||||
bin/kb/index --json # emit stats as JSON
|
bin/kb/index --json # emit stats as JSON
|
||||||
|
|
||||||
Reads every .md under the corpus (default: repo root docs, skills, READMEs)
|
Reads every .md under the corpus (default: repo root docs, skills, READMEs)
|
||||||
as `info` leafs, embeds them with model2vec (potion-multilingual-128M), and
|
as `info` leafs, embeds them with model2vec (potion-multilingual-128M), and
|
||||||
writes them into var/kb.lbug with FTS + HNSW indexes. `facts` leafs come
|
writes them into var/kb.lbug with FTS + HNSW indexes. `facts` leafs come
|
||||||
from bin/facts/extract (docker × compose × ssh-config pairing) when
|
from bin/facts/extract (docker x compose x ssh-config pairing).
|
||||||
`--with-facts` is set. `--with-chats` indexes markdown under var/chats/md
|
|
||||||
(or a given dir) as info. WhatsApp sync stays out of v1.
|
|
||||||
|
|
||||||
--rebuild drops the database file and indexes from scratch. Without it a run
|
--rebuild drops the database file and indexes from scratch. Without it a run
|
||||||
is idempotent (MERGE by (source,text) id).
|
is idempotent (MERGE by (source,text) id).
|
||||||
@@ -25,7 +22,7 @@ ROOT = Path(__file__).resolve().parents[2]
|
|||||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
||||||
|
|
||||||
from kblib import ( # noqa: E402
|
from kblib import ( # noqa: E402
|
||||||
add_leafs, connect, ensure_indexes, init_schema, upsert_leaf, link_from_file,
|
connect, ensure_indexes, init_schema, upsert_leaf,
|
||||||
open_readonly, stats,
|
open_readonly, stats,
|
||||||
)
|
)
|
||||||
from mdleaves import read_markdown, to_all, walk_markdown # noqa: E402
|
from mdleaves import read_markdown, to_all, walk_markdown # noqa: E402
|
||||||
@@ -84,11 +81,10 @@ def index_leafs(conn, leafs: list[dict], embed_fn, limit: int) -> tuple[int, int
|
|||||||
for lf in leafs[:limit] if limit else leafs:
|
for lf in leafs[:limit] if limit else leafs:
|
||||||
query = f"{lf['heading']}\n\n{lf['text']}"
|
query = f"{lf['heading']}\n\n{lf['text']}"
|
||||||
emb = embed_fn(lf["text"]) if lf["text"] else None
|
emb = embed_fn(lf["text"]) if lf["text"] else None
|
||||||
lid = upsert_leaf(conn, text=query, root="info", confidence="confirmed",
|
upsert_leaf(conn, text=query, root="info", confidence="confirmed",
|
||||||
source=lf["source"], source_rev="working-tree",
|
source=lf["source"], source_rev="working-tree",
|
||||||
how="kb/index", loc=lf["source"], type_=lf.get("type", "reference"),
|
how="kb/index", loc=lf["source"], type_=lf.get("type", "reference"),
|
||||||
embedding=emb)
|
embedding=emb)
|
||||||
link_from_file(conn, lid, lf["source"], repo=str(lf.get("repo") or ""))
|
|
||||||
count += 1
|
count += 1
|
||||||
return count, len(leafs)
|
return count, len(leafs)
|
||||||
|
|
||||||
@@ -99,65 +95,12 @@ def embedder():
|
|||||||
return lambda text: model.encode([text])[0].astype(float).tolist()
|
return lambda text: model.encode([text])[0].astype(float).tolist()
|
||||||
|
|
||||||
|
|
||||||
def index_fact_dicts(conn, facts: list[dict], embed_fn) -> int:
|
|
||||||
"""Write extract-shaped dicts as root=facts leafs (2-source source field)."""
|
|
||||||
leafs = []
|
|
||||||
for f in facts:
|
|
||||||
text = str(f.get("text") or "")
|
|
||||||
source = str(f.get("source") or "")
|
|
||||||
if not text or not source:
|
|
||||||
continue
|
|
||||||
leafs.append({
|
|
||||||
"text": text,
|
|
||||||
"root": "facts",
|
|
||||||
"confidence": "confirmed",
|
|
||||||
"source": source,
|
|
||||||
"source_rev": f.get("source_rev") or "working-tree",
|
|
||||||
"how": f.get("how") or "facts/extract",
|
|
||||||
"loc": f.get("loc") or source,
|
|
||||||
"type": "fact",
|
|
||||||
"embedding": embed_fn(text) if text else None,
|
|
||||||
})
|
|
||||||
return len(add_leafs(conn, leafs))
|
|
||||||
|
|
||||||
|
|
||||||
def facts_from_extract() -> list[dict]:
|
|
||||||
import subprocess
|
|
||||||
proc = subprocess.run(
|
|
||||||
[sys.executable, str(ROOT / "bin" / "facts" / "extract"), "--json", "--dry-run"],
|
|
||||||
cwd=ROOT,
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
check=False,
|
|
||||||
)
|
|
||||||
if proc.returncode != 0:
|
|
||||||
print(f"kb/index: facts/extract failed: {proc.stderr}", file=sys.stderr)
|
|
||||||
return []
|
|
||||||
try:
|
|
||||||
payload = json.loads(proc.stdout)
|
|
||||||
except json.JSONDecodeError:
|
|
||||||
print("kb/index: facts/extract produced non-JSON", file=sys.stderr)
|
|
||||||
return []
|
|
||||||
return list(payload.get("facts") or [])
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str]) -> int:
|
def main(argv: list[str]) -> int:
|
||||||
import argparse
|
import argparse
|
||||||
p = argparse.ArgumentParser(description="build the 2dph brain index")
|
p = argparse.ArgumentParser(description="build the 2dph brain index")
|
||||||
p.add_argument("--corpus", action="append", help="extra markdown dir/file to index (may repeat)")
|
p.add_argument("--corpus", action="append", help="extra markdown dir/file to index (may repeat)")
|
||||||
p.add_argument("--rebuild", action="store_true", help="fresh db + indexes")
|
p.add_argument("--rebuild", action="store_true", help="fresh db + indexes")
|
||||||
p.add_argument("--db", default="", help="path to kb.lbug (default var/kb.lbug)")
|
|
||||||
p.add_argument("--no-defaults", action="store_true", help="do not index repo README/docs/skills")
|
|
||||||
p.add_argument("--with-mail", action="store_true", help="include var/mail message.md leafs")
|
p.add_argument("--with-mail", action="store_true", help="include var/mail message.md leafs")
|
||||||
p.add_argument("--with-facts", action="store_true", help="run facts/extract into root=facts")
|
|
||||||
p.add_argument("--facts-json", default="", help="JSON list (or {facts:[...]}) of fact dicts")
|
|
||||||
p.add_argument(
|
|
||||||
"--with-chats",
|
|
||||||
nargs="?",
|
|
||||||
const=str(ROOT / "var" / "chats" / "md"),
|
|
||||||
default="",
|
|
||||||
help="index chat markdown as info (default var/chats/md)",
|
|
||||||
)
|
|
||||||
p.add_argument("--since", default="", help="with --with-mail, only messages dated >= YYYY-MM-DD")
|
p.add_argument("--since", default="", help="with --with-mail, only messages dated >= YYYY-MM-DD")
|
||||||
p.add_argument("--dry-run", action="store_true", help="count leafs, write nothing")
|
p.add_argument("--dry-run", action="store_true", help="count leafs, write nothing")
|
||||||
p.add_argument(
|
p.add_argument(
|
||||||
@@ -171,71 +114,44 @@ def main(argv: list[str]) -> int:
|
|||||||
|
|
||||||
from kblib import DB_PATH, VAR
|
from kblib import DB_PATH, VAR
|
||||||
|
|
||||||
dbpath = Path(a.db) if a.db else DB_PATH
|
leafs = load_corpus(ROOT)
|
||||||
leafs: list[dict] = [] if a.no_defaults else load_corpus(ROOT)
|
|
||||||
if a.corpus:
|
if a.corpus:
|
||||||
for source in a.corpus:
|
for source in a.corpus:
|
||||||
leafs.extend(load_corpus_glob(source))
|
leafs.extend(load_corpus_glob(source))
|
||||||
chat_n = 0
|
|
||||||
if a.with_chats:
|
|
||||||
chats = load_corpus_glob(a.with_chats)
|
|
||||||
chat_n = len(chats)
|
|
||||||
leafs.extend(chats)
|
|
||||||
mail_n = 0
|
mail_n = 0
|
||||||
if a.with_mail:
|
if a.with_mail:
|
||||||
mail = from_mail_root(ROOT / "var" / "mail", since=a.since)
|
mail = from_mail_root(ROOT / "var" / "mail", since=a.since)
|
||||||
mail_n = len(mail)
|
mail_n = len(mail)
|
||||||
leafs.extend(mail)
|
leafs.extend(mail)
|
||||||
|
|
||||||
facts: list[dict] = []
|
|
||||||
if a.facts_json:
|
|
||||||
raw = Path(a.facts_json).read_text(encoding="utf-8")
|
|
||||||
payload = json.loads(raw)
|
|
||||||
facts = list(payload.get("facts") if isinstance(payload, dict) else payload)
|
|
||||||
if a.with_facts:
|
|
||||||
facts.extend(facts_from_extract())
|
|
||||||
|
|
||||||
if a.dry_run:
|
if a.dry_run:
|
||||||
msg = {
|
msg = {"indexed": 0, "corpus_total": len(leafs), "mail_leafs": mail_n, "dry_run": True}
|
||||||
"indexed": 0,
|
|
||||||
"corpus_total": len(leafs),
|
|
||||||
"mail_leafs": mail_n,
|
|
||||||
"chat_leafs": chat_n,
|
|
||||||
"facts_leafs": len(facts),
|
|
||||||
"dry_run": True,
|
|
||||||
}
|
|
||||||
print(json.dumps(msg, indent=2) if a.json else
|
print(json.dumps(msg, indent=2) if a.json else
|
||||||
f"brain/index: {len(leafs)} info + {len(facts)} facts would be indexed")
|
f"brain/index: {len(leafs)} leafs would be indexed (mail={mail_n})")
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
VAR.mkdir(exist_ok=True)
|
VAR.mkdir(exist_ok=True)
|
||||||
dbpath.parent.mkdir(parents=True, exist_ok=True)
|
if a.rebuild and DB_PATH.exists():
|
||||||
if a.rebuild and dbpath.exists():
|
DB_PATH.unlink()
|
||||||
dbpath.unlink()
|
|
||||||
|
|
||||||
db, conn = connect(dbpath, read_only=False)
|
db, conn = connect(DB_PATH, read_only=False)
|
||||||
init_schema(conn)
|
init_schema(conn)
|
||||||
|
|
||||||
|
# Never DROP FTS/VECTOR (ghost catalog). Write leafs, then ensure indexes
|
||||||
|
# unless --skip-indexes (seed facts first — MERGE under live FTS corrupts it).
|
||||||
|
# --rebuild already deleted kb.lbug above, so CREATE runs on a clean DB.
|
||||||
embed = embedder()
|
embed = embedder()
|
||||||
done, total = index_leafs(conn, leafs, embed, a.limit)
|
done, total = index_leafs(conn, leafs, embed, a.limit)
|
||||||
fact_n = index_fact_dicts(conn, facts, embed) if facts else 0
|
|
||||||
if not a.skip_indexes:
|
if not a.skip_indexes:
|
||||||
ensure_indexes(conn)
|
ensure_indexes(conn)
|
||||||
s = stats(conn)
|
s = stats(conn)
|
||||||
conn.close()
|
conn.close()
|
||||||
db.close()
|
db.close()
|
||||||
|
|
||||||
result = {
|
result = {"indexed": done, "corpus_total": total, **{k: v for k, v in s.items() if k in ("total", "by_root")}}
|
||||||
"indexed": done,
|
|
||||||
"corpus_total": total,
|
|
||||||
"facts_leafs": fact_n,
|
|
||||||
"chat_leafs": chat_n,
|
|
||||||
**{k: v for k, v in s.items() if k in ("total", "by_root")},
|
|
||||||
}
|
|
||||||
if a.skip_indexes:
|
if a.skip_indexes:
|
||||||
result["indexes"] = "skipped"
|
result["indexes"] = "skipped"
|
||||||
print(json.dumps(result, indent=2) if a.json else
|
print(json.dumps(result, indent=2) if a.json else f"indexed {done}/{total} leafs; db total {s['total']}")
|
||||||
f"indexed {done}/{total} info + {fact_n} facts; db total {s['total']}")
|
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+77
-11
@@ -7,7 +7,7 @@
|
|||||||
bin/mail/import --since 2026-01-01 only messages after a date
|
bin/mail/import --since 2026-01-01 only messages after a date
|
||||||
bin/mail/import --limit 50 cap messages per run
|
bin/mail/import --limit 50 cap messages per run
|
||||||
bin/mail/import --no-attachments body only, skip attachment conversion
|
bin/mail/import --no-attachments body only, skip attachment conversion
|
||||||
bin/mail/import --ocr OCR images (PDFs OCR when textless)
|
bin/mail/import --ocr OCR scanned PDFs/images via docling
|
||||||
bin/mail/import --dry-run list messages without writing anything
|
bin/mail/import --dry-run list messages without writing anything
|
||||||
|
|
||||||
Writes one directory per message: var/mail/{folder}/{message_id}/
|
Writes one directory per message: var/mail/{folder}/{message_id}/
|
||||||
@@ -16,11 +16,10 @@ Writes one directory per message: var/mail/{folder}/{message_id}/
|
|||||||
attachments/*.md converted attachment content
|
attachments/*.md converted attachment content
|
||||||
|
|
||||||
Indexing is a separate step (`bin/brain/index.go --rebuild`): conversion can
|
Indexing is a separate step (`bin/brain/index.go --rebuild`): conversion can
|
||||||
crash and must not leave the brain DB mid-transaction.
|
crash in native docling and must not leave the brain DB mid-transaction.
|
||||||
|
|
||||||
Requires ONLYOFFICE_URL/USER/PASS in .env (or env) except `--from-raw`.
|
Requires ONLYOFFICE_URL/USER/PASS in .env (or env). Idempotent: a message
|
||||||
Idempotent: a message already present (message.md exists) is skipped unless
|
already present (message.md exists) is skipped unless --force.
|
||||||
--force.
|
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -42,10 +41,10 @@ from mailconv import ( # noqa: E402
|
|||||||
IMAGE_SUFFIXES,
|
IMAGE_SUFFIXES,
|
||||||
LEGACY_OFFICE_SUFFIXES,
|
LEGACY_OFFICE_SUFFIXES,
|
||||||
TEXT_SUFFIXES,
|
TEXT_SUFFIXES,
|
||||||
convert_pdf,
|
|
||||||
html_to_markdown,
|
html_to_markdown,
|
||||||
|
is_convertible,
|
||||||
normalize_markdown,
|
normalize_markdown,
|
||||||
ocr_image,
|
subject_to_filename,
|
||||||
zip_extract_safe,
|
zip_extract_safe,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -148,9 +147,9 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
return f"\n<!-- conversion failed: {e} -->\n"
|
return f"\n<!-- conversion failed: {e} -->\n"
|
||||||
if suffix == ".pdf":
|
if suffix == ".pdf":
|
||||||
return convert_pdf(path, ocr)
|
return _convert_pdf(path, ocr)
|
||||||
if suffix in IMAGE_SUFFIXES and ocr:
|
if suffix in IMAGE_SUFFIXES and ocr:
|
||||||
return ocr_image(path) or "\n<!-- ocr unavailable -->\n"
|
return _convert_pdf(path, ocr)
|
||||||
if suffix in LEGACY_OFFICE_SUFFIXES:
|
if suffix in LEGACY_OFFICE_SUFFIXES:
|
||||||
return _convert_legacy(path)
|
return _convert_legacy(path)
|
||||||
if suffix in ARCHIVE_SUFFIXES:
|
if suffix in ARCHIVE_SUFFIXES:
|
||||||
@@ -158,6 +157,67 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _convert_pdf(path: Path, ocr: bool) -> str:
|
||||||
|
"""Convert one PDF to markdown.
|
||||||
|
|
||||||
|
Fast path: poppler's pdftotext (-layout) extracts exact text from
|
||||||
|
born-digital PDFs in ~15ms vs docling's 1-3s. Only textless PDFs (scanned
|
||||||
|
pages, layout-heavy) fall back to docling, which runs isolated in a
|
||||||
|
subprocess because its native onnx/RT-DETR has segfaulted the main process.
|
||||||
|
"""
|
||||||
|
text = _pdf_fast_text(path)
|
||||||
|
if ocr or text is None or not text.strip():
|
||||||
|
return _convert_pdf_docling(path, ocr)
|
||||||
|
return normalize_markdown(text)
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_fast_text(path: Path) -> str | None:
|
||||||
|
"""pdftotext -layout; None when poppler is unavailable (or the PDF has no text layer)."""
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
["pdftotext", "-layout", str(path), "-"],
|
||||||
|
capture_output=True, timeout=60)
|
||||||
|
except (OSError, subprocess.TimeoutExpired):
|
||||||
|
return None
|
||||||
|
if proc.returncode != 0:
|
||||||
|
return None
|
||||||
|
return proc.stdout.decode("utf-8", errors="replace")
|
||||||
|
|
||||||
|
|
||||||
|
def _convert_pdf_docling(path: Path, ocr: bool) -> str:
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
[sys.executable, os.path.abspath(__file__), "--pdf-worker", str(path),
|
||||||
|
"--ocr" if ocr else "--no-ocr"],
|
||||||
|
capture_output=True, text=True, timeout=600)
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
return "\n<!-- pdf conversion timed out -->\n"
|
||||||
|
if proc.returncode != 0:
|
||||||
|
tail = proc.stderr.strip().splitlines()[-3:]
|
||||||
|
return f"\n<!-- pdf conversion failed: {proc.returncode}: {' | '.join(tail)} -->\n"
|
||||||
|
return proc.stdout
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_worker(path: Path, ocr: bool) -> None:
|
||||||
|
"""docling worker entry: prints converted markdown on stdout, exits non-zero on error."""
|
||||||
|
try:
|
||||||
|
from docling.document_converter import DocumentConverter, PdfFormatOption
|
||||||
|
from docling.datamodel.pipeline_options import PdfPipelineOptions
|
||||||
|
opts = PdfPipelineOptions()
|
||||||
|
opts.do_ocr = bool(ocr)
|
||||||
|
opts.do_table_structure = True
|
||||||
|
conv = DocumentConverter(format_options={"pdf": PdfFormatOption(pipeline_options=opts)})
|
||||||
|
res = conv.convert(str(path))
|
||||||
|
sys.stdout.write(normalize_markdown(res.document.export_to_markdown()))
|
||||||
|
sys.exit(0)
|
||||||
|
except Exception as e:
|
||||||
|
# errors/stacktraces to stderr; the caller only reports a one-liner
|
||||||
|
print(f"pdf-worker: {e}", file=sys.stderr)
|
||||||
|
import traceback
|
||||||
|
traceback.print_exc(file=sys.stderr)
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
|
||||||
def _convert_legacy(path: Path) -> str:
|
def _convert_legacy(path: Path) -> str:
|
||||||
"""Legacy .doc/.xls/.ppt -> md via pandoc (installed) or a stub."""
|
"""Legacy .doc/.xls/.ppt -> md via pandoc (installed) or a stub."""
|
||||||
try:
|
try:
|
||||||
@@ -296,12 +356,19 @@ def main(argv: list[str]) -> int:
|
|||||||
p.add_argument("--from-raw", default="",
|
p.add_argument("--from-raw", default="",
|
||||||
help="convert Go-synced dirs (var/mail/<folder>/<id>/message.json) to markdown")
|
help="convert Go-synced dirs (var/mail/<folder>/<id>/message.json) to markdown")
|
||||||
p.add_argument("--no-attachments", action="store_true", help="skip attachment download+convert")
|
p.add_argument("--no-attachments", action="store_true", help="skip attachment download+convert")
|
||||||
p.add_argument("--ocr", action="store_true", help="OCR images (PDFs OCR when textless)")
|
p.add_argument("--ocr", action="store_true", help="OCR scanned PDFs/images via docling")
|
||||||
p.add_argument("--force", action="store_true", help="re-import even if message.md exists")
|
p.add_argument("--force", action="store_true", help="re-import even if message.md exists")
|
||||||
p.add_argument("--dry-run", action="store_true", help="list messages, write nothing")
|
p.add_argument("--dry-run", action="store_true", help="list messages, write nothing")
|
||||||
p.add_argument("--json", action="store_true")
|
p.add_argument("--json", action="store_true")
|
||||||
|
p.add_argument("--pdf-worker", default="", help=argparse.SUPPRESS)
|
||||||
|
p.add_argument("--no-ocr", action="store_true", help=argparse.SUPPRESS)
|
||||||
a = p.parse_args(argv)
|
a = p.parse_args(argv)
|
||||||
|
|
||||||
|
if a.pdf_worker:
|
||||||
|
_pdf_worker(Path(a.pdf_worker), ocr=not a.no_ocr)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
conf = load_env()
|
||||||
fid = folder_id(a.folder)
|
fid = folder_id(a.folder)
|
||||||
out_root = ROOT / "var" / "mail"
|
out_root = ROOT / "var" / "mail"
|
||||||
summary: list[dict] = []
|
summary: list[dict] = []
|
||||||
@@ -327,7 +394,6 @@ def main(argv: list[str]) -> int:
|
|||||||
target_dir=msg_dir.parent))
|
target_dir=msg_dir.parent))
|
||||||
summary.append(entry)
|
summary.append(entry)
|
||||||
else:
|
else:
|
||||||
conf = load_env()
|
|
||||||
OOCLIENT = OOClient(conf)
|
OOCLIENT = OOClient(conf)
|
||||||
if a.id:
|
if a.id:
|
||||||
messages = [{"id": i} for i in a.id]
|
messages = [{"id": i} for i in a.id]
|
||||||
|
|||||||
@@ -1,48 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=mail_ocr "$0" "$@"; exit
|
|
||||||
//go:build mail_ocr
|
|
||||||
//
|
|
||||||
// bin/mail/ocr.go - OCR an image or scanned PDF (tesseract eng+deu).
|
|
||||||
//
|
|
||||||
// ./bin/mail/ocr.go scan.png
|
|
||||||
// ./bin/mail/ocr.go scan.pdf
|
|
||||||
// OCR_ENGINE=paddle ./bin/mail/ocr.go scan.png
|
|
||||||
//
|
|
||||||
// PDFs try pdftotext -layout first; empty text layer uses pdftoppm + tesseract.
|
|
||||||
// No gocv. Tesseract CGO bindings are not used (D21 Zig owns Ladybug CGO).
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"fmt"
|
|
||||||
"os"
|
|
||||||
"strings"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/ocr"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(run(os.Args[1:]))
|
|
||||||
}
|
|
||||||
|
|
||||||
func run(args []string) int {
|
|
||||||
if len(args) != 1 || strings.HasPrefix(args[0], "-") {
|
|
||||||
fmt.Fprintln(os.Stderr, `usage: bin/mail/ocr.go <image|pdf>`)
|
|
||||||
return 2
|
|
||||||
}
|
|
||||||
path := args[0]
|
|
||||||
var (
|
|
||||||
text string
|
|
||||||
err error
|
|
||||||
)
|
|
||||||
if strings.HasSuffix(strings.ToLower(path), ".pdf") {
|
|
||||||
text, err = ocr.PDFFile(path)
|
|
||||||
} else {
|
|
||||||
text, err = ocr.ImageFile(path)
|
|
||||||
}
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "mail/ocr: %v\n", err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
fmt.Println(text)
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
@@ -1,73 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=qa_stats "$0" "$@"; exit
|
|
||||||
//go:build qa_stats
|
|
||||||
//
|
|
||||||
// bin/qa/stats.go - DuckDB quantiles over a JSON number array or JSONL count.
|
|
||||||
//
|
|
||||||
// ./bin/qa/stats.go <<< '[1,2,3,4,5]'
|
|
||||||
// ./bin/qa/stats.go --jsonl rows.jsonl
|
|
||||||
//
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
// DuckDB CGO needs gcc/g++ (not Zig). After eval "$(bin/cgo/zig env)":
|
|
||||||
// CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= ./bin/qa/stats.go
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"encoding/json"
|
|
||||||
"fmt"
|
|
||||||
"io"
|
|
||||||
"os"
|
|
||||||
"strings"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/duckstats"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(run(os.Args[1:]))
|
|
||||||
}
|
|
||||||
|
|
||||||
func run(args []string) int {
|
|
||||||
jsonl := ""
|
|
||||||
for i := 0; i < len(args); i++ {
|
|
||||||
a := args[i]
|
|
||||||
switch {
|
|
||||||
case a == "--jsonl" && i+1 < len(args):
|
|
||||||
i++
|
|
||||||
jsonl = args[i]
|
|
||||||
case strings.HasPrefix(a, "--jsonl="):
|
|
||||||
jsonl = strings.TrimPrefix(a, "--jsonl=")
|
|
||||||
case a == "-h" || a == "--help":
|
|
||||||
fmt.Fprintln(os.Stderr, "bin/qa/stats.go [--jsonl FILE] # stdin = JSON [float,…]")
|
|
||||||
return 0
|
|
||||||
default:
|
|
||||||
fmt.Fprintln(os.Stderr, "unknown arg:", a)
|
|
||||||
return 2
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if jsonl != "" {
|
|
||||||
n, err := duckstats.CountJSONL(jsonl)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintln(os.Stderr, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
fmt.Printf("n: %d\n", n)
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
raw, err := io.ReadAll(os.Stdin)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintln(os.Stderr, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
var samples []float64
|
|
||||||
if err := json.Unmarshal(raw, &samples); err != nil {
|
|
||||||
fmt.Fprintln(os.Stderr, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
s, err := duckstats.Quantiles(samples)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintln(os.Stderr, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
fmt.Printf("n: %d\nmin: %g\np50: %g\np95: %g\nmax: %g\navg: %g\n",
|
|
||||||
s.N, s.Min, s.P50, s.P95, s.Max, s.Avg)
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
+1
-12
@@ -7,7 +7,7 @@
|
|||||||
// ./bin/reasoner/bakeoff.go --model MichelRosselli/bonsai-27b:Q1_0 --json
|
// ./bin/reasoner/bakeoff.go --model MichelRosselli/bonsai-27b:Q1_0 --json
|
||||||
//
|
//
|
||||||
// Measures OpenAI tool_calls (search/get/audit) and RSS from Ollama /api/ps, not VRAM.
|
// Measures OpenAI tool_calls (search/get/audit) and RSS from Ollama /api/ps, not VRAM.
|
||||||
// PicoClaw is compose profile picoclaw; tool names match internal/httpapi MCP ops.
|
// PicoClaw is not in this repo; the tool names match internal/httpapi MCP ops.
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||||
package main
|
package main
|
||||||
|
|
||||||
@@ -17,7 +17,6 @@ import (
|
|||||||
"os"
|
"os"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/duckstats"
|
|
||||||
"github.com/eSlider/2dph/internal/reasoner"
|
"github.com/eSlider/2dph/internal/reasoner"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -62,14 +61,6 @@ func run(args []string) int {
|
|||||||
}
|
}
|
||||||
c := reasoner.Client{BaseURL: base, Model: model, Device: device}
|
c := reasoner.Client{BaseURL: base, Model: model, Device: device}
|
||||||
rep := reasoner.Run(c)
|
rep := reasoner.Run(c)
|
||||||
lat := make([]float64, 0, len(rep.Prompts))
|
|
||||||
for _, p := range rep.Prompts {
|
|
||||||
lat = append(lat, float64(p.LatencyMS))
|
|
||||||
}
|
|
||||||
if st, err := duckstats.Quantiles(lat); err == nil {
|
|
||||||
rep.LatencyP50MS = st.P50
|
|
||||||
rep.LatencyP95MS = st.P95
|
|
||||||
}
|
|
||||||
raw, err := json.MarshalIndent(rep, "", " ")
|
raw, err := json.MarshalIndent(rep, "", " ")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintln(os.Stderr, err)
|
fmt.Fprintln(os.Stderr, err)
|
||||||
@@ -85,8 +76,6 @@ func run(args []string) int {
|
|||||||
fmt.Printf("xml_leak: %d\n", rep.XMLLeak)
|
fmt.Printf("xml_leak: %d\n", rep.XMLLeak)
|
||||||
fmt.Printf("rss_mb: %d\n", rep.RSSMB)
|
fmt.Printf("rss_mb: %d\n", rep.RSSMB)
|
||||||
fmt.Printf("vram_mb: %d\n", rep.VRAMMB)
|
fmt.Printf("vram_mb: %d\n", rep.VRAMMB)
|
||||||
fmt.Printf("latency_p50_ms: %g\n", rep.LatencyP50MS)
|
|
||||||
fmt.Printf("latency_p95_ms: %g\n", rep.LatencyP95MS)
|
|
||||||
for _, p := range rep.Prompts {
|
for _, p := range rep.Prompts {
|
||||||
status := "fail"
|
status := "fail"
|
||||||
if p.OK {
|
if p.OK {
|
||||||
|
|||||||
+1
-92
@@ -4,7 +4,7 @@ Single embedded graph `var/kb.lbug`. Two roots: facts (assertions backed by
|
|||||||
>=2 independent sources) and info (narrative leafs). Hybrid retrieval: BM25
|
>=2 independent sources) and info (narrative leafs). Hybrid retrieval: BM25
|
||||||
(FTS extension) + HNSW cosine (VECTOR extension) + Cypher graph hops.
|
(FTS extension) + HNSW cosine (VECTOR extension) + Cypher graph hops.
|
||||||
|
|
||||||
All access is read-only unless `--rebuild` (kb/index) or `kb/add`.
|
All access is read-only unless `--rebuild` is passed to kb/index.
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -118,97 +118,6 @@ def upsert_leaf(conn: ladybug.Connection, *, text: str, root: str, confidence: s
|
|||||||
return lid
|
return lid
|
||||||
|
|
||||||
|
|
||||||
def add_leafs(conn: ladybug.Connection, leafs: list[dict]) -> list[str]:
|
|
||||||
"""Write facts+info leafs in one transaction. Safe while FTS/HNSW exist.
|
|
||||||
|
|
||||||
Each leaf dict: text, source, optional root/confidence/source_rev/how/loc/type/embedding.
|
|
||||||
Does not delete the database file. Measured on Ladybug 0.19: MERGE of new
|
|
||||||
ids (and updates) stays FTS+HNSW queryable; DROP INDEX is the fatal path.
|
|
||||||
"""
|
|
||||||
if not leafs:
|
|
||||||
return []
|
|
||||||
started = False
|
|
||||||
try:
|
|
||||||
conn.execute("BEGIN TRANSACTION")
|
|
||||||
started = True
|
|
||||||
except Exception:
|
|
||||||
started = False
|
|
||||||
ids: list[str] = []
|
|
||||||
try:
|
|
||||||
for lf in leafs:
|
|
||||||
ids.append(
|
|
||||||
upsert_leaf(
|
|
||||||
conn,
|
|
||||||
text=str(lf["text"]),
|
|
||||||
root=str(lf.get("root") or ROOT_INFO),
|
|
||||||
confidence=str(lf.get("confidence") or CONF_CONFIRMED),
|
|
||||||
source=str(lf["source"]),
|
|
||||||
source_rev=str(lf.get("source_rev") or "working-tree"),
|
|
||||||
how=str(lf.get("how") or "brain/add"),
|
|
||||||
loc=str(lf.get("loc") or lf.get("source") or ""),
|
|
||||||
type_=str(lf.get("type") or lf.get("type_") or "reference"),
|
|
||||||
embedding=lf.get("embedding"),
|
|
||||||
)
|
|
||||||
)
|
|
||||||
if started:
|
|
||||||
conn.execute("COMMIT")
|
|
||||||
except Exception:
|
|
||||||
if started:
|
|
||||||
try:
|
|
||||||
conn.execute("ROLLBACK")
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
raise
|
|
||||||
return ids
|
|
||||||
|
|
||||||
|
|
||||||
def file_id(repo: str, path: str) -> str:
|
|
||||||
"""Stable File.id matching gitimport (`repo:path`)."""
|
|
||||||
return f"{repo}:{path}" if repo else path
|
|
||||||
|
|
||||||
|
|
||||||
def link_from_file(conn: ladybug.Connection, leaf_id: str, path: str,
|
|
||||||
repo: str = "", mtime: str = "") -> str:
|
|
||||||
"""MERGE File and Leaf-[:FROM_FILE]->File so --hop 1 can walk."""
|
|
||||||
fid = file_id(repo, path)
|
|
||||||
conn.execute(
|
|
||||||
"MERGE (f:File {id:$id}) SET f.path=$path, f.repo=$repo, f.mtime=$mtime",
|
|
||||||
parameters={"id": fid, "path": path, "repo": repo, "mtime": mtime},
|
|
||||||
)
|
|
||||||
conn.execute(
|
|
||||||
"MATCH (l:Leaf {id:$lid}), (f:File {id:$fid}) "
|
|
||||||
"MERGE (l)-[:FROM_FILE]->(f)",
|
|
||||||
parameters={"lid": leaf_id, "fid": fid},
|
|
||||||
)
|
|
||||||
return fid
|
|
||||||
|
|
||||||
|
|
||||||
HOP_STMTS = {
|
|
||||||
1: "MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File) RETURN f.id, f.path, 1",
|
|
||||||
2: ("MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File)-[:HAS_VERSION]->(c:Commit) "
|
|
||||||
"RETURN c.id, c.subject, 2"),
|
|
||||||
3: ("MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File)-[:HAS_VERSION]->(c:Commit)"
|
|
||||||
"-[:AUTHORED]->(p:Person) RETURN p.id, p.name, 3"),
|
|
||||||
}
|
|
||||||
HOP_LABELS = {1: "File", 2: "Commit", 3: "Person"}
|
|
||||||
|
|
||||||
|
|
||||||
def hop_walk(conn: ladybug.Connection, leaf_id: str, n: int) -> list[dict]:
|
|
||||||
"""Walk Leaf → File → Commit → Person up to n hops (max 3)."""
|
|
||||||
depth = min(max(int(n), 0), 3)
|
|
||||||
out: list[dict] = []
|
|
||||||
for d in range(1, depth + 1):
|
|
||||||
rows = conn.execute(HOP_STMTS[d], parameters={"id": leaf_id}).get_all()
|
|
||||||
for row in rows:
|
|
||||||
out.append({
|
|
||||||
"id": row[0],
|
|
||||||
"label": HOP_LABELS[d],
|
|
||||||
"name": row[1],
|
|
||||||
"depth": int(row[2]),
|
|
||||||
})
|
|
||||||
return out
|
|
||||||
|
|
||||||
|
|
||||||
def leaf_index_names(conn: ladybug.Connection) -> set[str]:
|
def leaf_index_names(conn: ladybug.Connection) -> set[str]:
|
||||||
"""Return index names on the Leaf table (e.g. {'id', 'Leaf_vec', '_PK'})."""
|
"""Return index names on the Leaf table (e.g. {'id', 'Leaf_vec', '_PK'})."""
|
||||||
rows = conn.execute("CALL SHOW_INDEXES() RETURN *").get_all()
|
rows = conn.execute("CALL SHOW_INDEXES() RETURN *").get_all()
|
||||||
|
|||||||
+1
-84
@@ -7,10 +7,7 @@ offline against fixtures.
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import html
|
import html
|
||||||
import os
|
|
||||||
import re
|
import re
|
||||||
import subprocess
|
|
||||||
import tempfile
|
|
||||||
import zipfile
|
import zipfile
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -21,10 +18,9 @@ OFFICE_SUFFIXES = {".docx", ".pptx", ".xlsx", ".html", ".htm", ".epub", ".eml",
|
|||||||
PDF_SUFFIXES = {".pdf"}
|
PDF_SUFFIXES = {".pdf"}
|
||||||
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".tiff", ".tif", ".webp"}
|
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".tiff", ".tif", ".webp"}
|
||||||
ARCHIVE_SUFFIXES = {".zip"}
|
ARCHIVE_SUFFIXES = {".zip"}
|
||||||
# Legacy binary Office (doc/xls/ppt) — markitdown skip them; we try
|
# Legacy binary Office (doc/xls/ppt) — markitdown/docling skip them; we try
|
||||||
# pandoc first, else leave a stub.
|
# pandoc first, else leave a stub.
|
||||||
LEGACY_OFFICE_SUFFIXES = {".doc", ".xls", ".ppt"}
|
LEGACY_OFFICE_SUFFIXES = {".doc", ".xls", ".ppt"}
|
||||||
TESS_LANG = "eng+deu"
|
|
||||||
|
|
||||||
CONVERTIBLE_SUFFIXES = (
|
CONVERTIBLE_SUFFIXES = (
|
||||||
TEXT_SUFFIXES | OFFICE_SUFFIXES | PDF_SUFFIXES | IMAGE_SUFFIXES | ARCHIVE_SUFFIXES | LEGACY_OFFICE_SUFFIXES
|
TEXT_SUFFIXES | OFFICE_SUFFIXES | PDF_SUFFIXES | IMAGE_SUFFIXES | ARCHIVE_SUFFIXES | LEGACY_OFFICE_SUFFIXES
|
||||||
@@ -150,82 +146,3 @@ def zip_extract_safe(zip_path: Path, dest: Path) -> list[Path]:
|
|||||||
|
|
||||||
def is_convertible(suffix: str) -> bool:
|
def is_convertible(suffix: str) -> bool:
|
||||||
return suffix.lower() in CONVERTIBLE_SUFFIXES
|
return suffix.lower() in CONVERTIBLE_SUFFIXES
|
||||||
|
|
||||||
|
|
||||||
def convert_pdf(path: Path, ocr: bool = False) -> str:
|
|
||||||
"""pdftotext -layout first; empty text layer → pdftoppm + tesseract.
|
|
||||||
|
|
||||||
`ocr` is unused for born-digital PDFs (text layer wins). Scans OCR
|
|
||||||
automatically. This path never execs an ONNX document converter.
|
|
||||||
"""
|
|
||||||
del ocr # scans OCR when the text layer is empty; flag is for images
|
|
||||||
text = pdf_fast_text(path)
|
|
||||||
if text and text.strip():
|
|
||||||
return normalize_markdown(text)
|
|
||||||
scanned = ocr_pdf(path)
|
|
||||||
if scanned and scanned.strip():
|
|
||||||
return normalize_markdown(scanned)
|
|
||||||
if text:
|
|
||||||
return normalize_markdown(text)
|
|
||||||
return "\n<!-- pdf has no text layer (ocr unavailable) -->\n"
|
|
||||||
|
|
||||||
|
|
||||||
def pdf_fast_text(path: Path) -> str | None:
|
|
||||||
"""pdftotext -layout; None when poppler is missing or the command fails."""
|
|
||||||
try:
|
|
||||||
proc = subprocess.run(
|
|
||||||
["pdftotext", "-layout", str(path), "-"],
|
|
||||||
capture_output=True, timeout=60)
|
|
||||||
except (OSError, subprocess.TimeoutExpired):
|
|
||||||
return None
|
|
||||||
if proc.returncode != 0:
|
|
||||||
return None
|
|
||||||
return proc.stdout.decode("utf-8", errors="replace")
|
|
||||||
|
|
||||||
|
|
||||||
def ocr_pdf(path: Path) -> str:
|
|
||||||
"""Rasterize with pdftoppm and OCR each page (tesseract or paddle)."""
|
|
||||||
try:
|
|
||||||
with tempfile.TemporaryDirectory(prefix="2dph-ocr-") as tmp:
|
|
||||||
prefix = str(Path(tmp) / "page")
|
|
||||||
proc = subprocess.run(
|
|
||||||
["pdftoppm", "-png", "-r", "200", str(path), prefix],
|
|
||||||
capture_output=True, timeout=120)
|
|
||||||
if proc.returncode != 0:
|
|
||||||
return ""
|
|
||||||
pages = sorted(Path(tmp).glob("page*.png"))
|
|
||||||
parts = [ocr_image(p) for p in pages]
|
|
||||||
return "\n\n".join(p for p in parts if p and p.strip())
|
|
||||||
except (OSError, subprocess.TimeoutExpired):
|
|
||||||
return ""
|
|
||||||
|
|
||||||
|
|
||||||
def ocr_image(path: Path) -> str:
|
|
||||||
engine = os.environ.get("OCR_ENGINE", "tesseract")
|
|
||||||
if engine == "paddle":
|
|
||||||
return _ocr_paddle(path)
|
|
||||||
return _ocr_tesseract(path)
|
|
||||||
|
|
||||||
|
|
||||||
def _ocr_tesseract(path: Path) -> str:
|
|
||||||
try:
|
|
||||||
proc = subprocess.run(
|
|
||||||
["tesseract", str(path), "stdout", "-l", TESS_LANG, "--psm", "6"],
|
|
||||||
capture_output=True, timeout=120)
|
|
||||||
except (OSError, subprocess.TimeoutExpired):
|
|
||||||
return ""
|
|
||||||
if proc.returncode != 0:
|
|
||||||
return ""
|
|
||||||
return proc.stdout.decode("utf-8", errors="replace").strip()
|
|
||||||
|
|
||||||
|
|
||||||
def _ocr_paddle(path: Path) -> str:
|
|
||||||
try:
|
|
||||||
proc = subprocess.run(
|
|
||||||
["paddleocr", "ocr", "-i", str(path)],
|
|
||||||
capture_output=True, timeout=180)
|
|
||||||
except (OSError, subprocess.TimeoutExpired):
|
|
||||||
return ""
|
|
||||||
if proc.returncode != 0:
|
|
||||||
return ""
|
|
||||||
return proc.stdout.decode("utf-8", errors="replace").strip()
|
|
||||||
|
|||||||
@@ -75,20 +75,9 @@ class BinLayoutTest(unittest.TestCase):
|
|||||||
)
|
)
|
||||||
|
|
||||||
def test_brain_methods_are_shebangs(self) -> None:
|
def test_brain_methods_are_shebangs(self) -> None:
|
||||||
for method in ("index.go", "add.go", "get.go", "stats.go", "eval.go", "watch.go"):
|
for method in ("index.go", "get.go", "stats.go", "eval.go", "watch.go"):
|
||||||
self._assert_shebang(f"bin/brain/{method}")
|
self._assert_shebang(f"bin/brain/{method}")
|
||||||
|
|
||||||
def test_brain_add_is_python_write_not_rebuild(self) -> None:
|
|
||||||
self._assert_shebang("bin/brain/add.go")
|
|
||||||
text = (ROOT / "bin" / "brain" / "add.go").read_text()
|
|
||||||
self.assertIn("cmdbin.ExecFile", text)
|
|
||||||
self.assertIn("bin/kb/add", text)
|
|
||||||
self.assertNotIn("--rebuild", text)
|
|
||||||
py = (ROOT / "bin" / "kb" / "add").read_text()
|
|
||||||
self.assertIn("add_leafs", py)
|
|
||||||
self.assertIn("--json", py)
|
|
||||||
self.assertNotIn("unlink", py.lower())
|
|
||||||
|
|
||||||
def test_brain_get_stats_eval_are_not_python_exec(self) -> None:
|
def test_brain_get_stats_eval_are_not_python_exec(self) -> None:
|
||||||
for method in ("get.go", "stats.go", "eval.go"):
|
for method in ("get.go", "stats.go", "eval.go"):
|
||||||
text = (ROOT / "bin" / "brain" / method).read_text()
|
text = (ROOT / "bin" / "brain" / method).read_text()
|
||||||
@@ -136,33 +125,6 @@ class BinLayoutTest(unittest.TestCase):
|
|||||||
"index_mail must point at bin/brain/index.go",
|
"index_mail must point at bin/brain/index.go",
|
||||||
)
|
)
|
||||||
|
|
||||||
def test_mail_ocr_is_tesseract_not_docling(self) -> None:
|
|
||||||
self._assert_shebang("bin/mail/ocr.go")
|
|
||||||
ocr = (ROOT / "bin" / "mail" / "ocr.go").read_text()
|
|
||||||
self.assertIn("internal/ocr", ocr)
|
|
||||||
self.assertIn("mail_ocr", ocr)
|
|
||||||
self.assertNotIn("github.com/otiai10/gosseract", ocr)
|
|
||||||
py = (ROOT / "bin" / "mail" / "import").read_text()
|
|
||||||
self.assertNotIn("from docling", py)
|
|
||||||
self.assertNotIn("import docling", py)
|
|
||||||
self.assertIn("convert_pdf", py)
|
|
||||||
conv = (ROOT / "bin" / "tools" / "mailconv.py").read_text()
|
|
||||||
self.assertIn("pdftotext", conv)
|
|
||||||
self.assertIn("pdftoppm", conv)
|
|
||||||
self.assertIn("tesseract", conv)
|
|
||||||
self.assertIn("eng+deu", conv)
|
|
||||||
self.assertNotIn("from docling", conv)
|
|
||||||
self.assertNotIn("import docling", conv)
|
|
||||||
self.assertNotIn("gocv", conv.lower())
|
|
||||||
proj = (ROOT / "pyproject.toml").read_text()
|
|
||||||
self.assertNotIn("docling", proj)
|
|
||||||
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
|
|
||||||
self.assertIn("tesseract-ocr", ci)
|
|
||||||
self.assertIn("./internal/ocr", ci)
|
|
||||||
compose = (ROOT / "compose.yaml").read_text()
|
|
||||||
self.assertIn("ocr-paddle", compose)
|
|
||||||
self.assertIn("OCR_ENGINE", compose)
|
|
||||||
|
|
||||||
def test_markdown_import_is_go_not_python_exec(self) -> None:
|
def test_markdown_import_is_go_not_python_exec(self) -> None:
|
||||||
self._assert_shebang("bin/markdown/import.go")
|
self._assert_shebang("bin/markdown/import.go")
|
||||||
text = (ROOT / "bin" / "markdown" / "import.go").read_text()
|
text = (ROOT / "bin" / "markdown" / "import.go").read_text()
|
||||||
@@ -217,32 +179,6 @@ class BinLayoutTest(unittest.TestCase):
|
|||||||
if "go-git/go-git" in line:
|
if "go-git/go-git" in line:
|
||||||
self.assertNotIn("indirect", line)
|
self.assertNotIn("indirect", line)
|
||||||
|
|
||||||
def test_duckdb_go_is_direct_require(self) -> None:
|
|
||||||
text = (ROOT / "go.mod").read_text()
|
|
||||||
first = text.split("require (")[1].split(")")[0]
|
|
||||||
self.assertRegex(first, r"github.com/duckdb/duckdb-go/v2\s+v")
|
|
||||||
for line in first.splitlines():
|
|
||||||
if "duckdb/duckdb-go" in line:
|
|
||||||
self.assertNotIn("indirect", line)
|
|
||||||
skill = (ROOT / "skills" / "duckdb" / "SKILL.md").read_text()
|
|
||||||
self.assertIn("github.com/duckdb/duckdb-go", skill)
|
|
||||||
self.assertIn("Ladybug", skill)
|
|
||||||
self.assertIn("sqlite", skill.lower())
|
|
||||||
self.assertIn("gcc", skill.lower())
|
|
||||||
self.assertIn("Zig", skill)
|
|
||||||
plan = (ROOT / "PLAN.md").read_text()
|
|
||||||
self.assertIn("D22", plan)
|
|
||||||
self.assertIn("duckdb-go", plan)
|
|
||||||
self._assert_shebang("bin/qa/stats.go")
|
|
||||||
reasoner = (ROOT / "internal" / "reasoner" / "client.go").read_text()
|
|
||||||
self.assertNotIn("duckdb", reasoner)
|
|
||||||
self.assertNotIn("duckstats", reasoner)
|
|
||||||
bakeoff = (ROOT / "bin" / "reasoner" / "bakeoff.go").read_text()
|
|
||||||
self.assertIn("internal/duckstats", bakeoff)
|
|
||||||
webcache = (ROOT / "internal" / "websearch" / "cache.go").read_text()
|
|
||||||
self.assertNotIn("duckdb", webcache)
|
|
||||||
self.assertIn("modernc.org/sqlite", webcache)
|
|
||||||
|
|
||||||
def test_cgo_uses_zig_not_gcc(self) -> None:
|
def test_cgo_uses_zig_not_gcc(self) -> None:
|
||||||
for rel in ("bin/cgo/zig", "bin/cgo/zcc", "bin/cgo/zc++"):
|
for rel in ("bin/cgo/zig", "bin/cgo/zcc", "bin/cgo/zc++"):
|
||||||
p = ROOT / rel
|
p = ROOT / rel
|
||||||
@@ -260,25 +196,3 @@ class BinLayoutTest(unittest.TestCase):
|
|||||||
search = (ROOT / "bin" / "kb" / "search").read_text()
|
search = (ROOT / "bin" / "kb" / "search").read_text()
|
||||||
self.assertIn("bin/cgo/zig", search)
|
self.assertIn("bin/cgo/zig", search)
|
||||||
self.assertNotIn("command -v gcc", search)
|
self.assertNotIn("command -v gcc", search)
|
||||||
|
|
||||||
def test_ci_recall_sot_is_zig_brain_eval(self) -> None:
|
|
||||||
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
|
|
||||||
self.assertIn("bin/brain/eval.go", ci)
|
|
||||||
self.assertIn("system_ladybug,brain_eval", ci)
|
|
||||||
self.assertIn("/tmp/brain-eval", ci)
|
|
||||||
self.assertIn("KB_ROOT", ci)
|
|
||||||
self.assertNotIn("bin/kb/eval", ci)
|
|
||||||
self.assertNotIn("gate skipped", ci)
|
|
||||||
self.assertIn("./bin/facts/audit self", ci)
|
|
||||||
|
|
||||||
def test_eval_fragments_live_in_default_corpus(self) -> None:
|
|
||||||
"""CI --rebuild indexes README/PLAN/docs/skills; fragments must be there."""
|
|
||||||
corpus = []
|
|
||||||
for rel in ("README.md", "PLAN.md", "AGENTS.md"):
|
|
||||||
corpus.append((ROOT / rel).read_text())
|
|
||||||
for d in ("docs", "skills"):
|
|
||||||
for p in (ROOT / d).rglob("*.md"):
|
|
||||||
corpus.append(p.read_text())
|
|
||||||
blob = "\n".join(corpus)
|
|
||||||
for frag in ("BM25", "DevOps", "LadybugDB"):
|
|
||||||
self.assertIn(frag, blob, f"{frag} must appear in default index corpus")
|
|
||||||
|
|||||||
@@ -39,59 +39,3 @@ class IndexAdapterTest(unittest.TestCase):
|
|||||||
self.assertTrue(msg.get("dry_run"))
|
self.assertTrue(msg.get("dry_run"))
|
||||||
self.assertGreaterEqual(msg.get("corpus_total", 0), 1)
|
self.assertGreaterEqual(msg.get("corpus_total", 0), 1)
|
||||||
self.assertFalse(lbug.exists(), "dry-run must not create a Ladybug file")
|
self.assertFalse(lbug.exists(), "dry-run must not create a Ladybug file")
|
||||||
|
|
||||||
def test_facts_json_and_chats_land_on_rebuild(self) -> None:
|
|
||||||
"""Gitea #18: facts (2-source) + chats markdown become leafs on rebuild."""
|
|
||||||
tmp = Path(tempfile.mkdtemp())
|
|
||||||
dbpath = tmp / "kb.lbug"
|
|
||||||
chats = tmp / "chats"
|
|
||||||
chats.mkdir()
|
|
||||||
(chats / "alice.md").write_text(
|
|
||||||
"# Chat\n\n## Alice and Bob\n\nhello from chats fixture unique-chat-token\n",
|
|
||||||
encoding="utf-8",
|
|
||||||
)
|
|
||||||
facts_path = tmp / "facts.json"
|
|
||||||
facts_path.write_text(json.dumps([{
|
|
||||||
"text": "container 'brain' unique-fact-token is running and declared in compose.yaml",
|
|
||||||
"source": "docker ps x compose.yaml",
|
|
||||||
"loc": "compose.yaml:brain",
|
|
||||||
"how": "facts/extract",
|
|
||||||
}]), encoding="utf-8")
|
|
||||||
venv_py = ROOT / ".venv" / "bin" / "python"
|
|
||||||
py = str(venv_py) if venv_py.is_file() else sys.executable
|
|
||||||
proc = subprocess.run(
|
|
||||||
[
|
|
||||||
py, str(ROOT / "bin" / "kb" / "index"),
|
|
||||||
"--rebuild", "--db", str(dbpath), "--no-defaults",
|
|
||||||
"--with-chats", str(chats),
|
|
||||||
"--facts-json", str(facts_path),
|
|
||||||
"--json",
|
|
||||||
],
|
|
||||||
cwd=ROOT,
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
env=os.environ.copy(),
|
|
||||||
check=False,
|
|
||||||
)
|
|
||||||
self.assertEqual(proc.returncode, 0, proc.stderr)
|
|
||||||
msg = json.loads(proc.stdout)
|
|
||||||
self.assertGreaterEqual(msg.get("facts_leafs", 0), 1)
|
|
||||||
self.assertGreaterEqual(msg.get("chat_leafs", 0), 1)
|
|
||||||
self.assertTrue(dbpath.exists())
|
|
||||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
|
||||||
import kblib
|
|
||||||
db, conn = kblib.connect(dbpath, read_only=True)
|
|
||||||
try:
|
|
||||||
stats = kblib.stats(conn)
|
|
||||||
self.assertGreaterEqual(stats["by_root"].get("facts", 0), 1)
|
|
||||||
fts = kblib.query_fts(conn, "unique-chat-token", 5)
|
|
||||||
self.assertTrue(fts, "chats markdown must be FTS-searchable")
|
|
||||||
fact_hits = kblib.query_fts(conn, "unique-fact-token", 5)
|
|
||||||
self.assertTrue(any(h.get("root") == "facts" for h in fact_hits))
|
|
||||||
src = conn.execute(
|
|
||||||
"MATCH (l:Leaf {root:'facts'}) RETURN l.source"
|
|
||||||
).get_all()
|
|
||||||
self.assertTrue(any(" x " in str(r[0]) for r in src))
|
|
||||||
finally:
|
|
||||||
conn.close()
|
|
||||||
db.close()
|
|
||||||
|
|||||||
@@ -1,69 +0,0 @@
|
|||||||
"""Incremental add writes leafs without deleting kb.lbug."""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
import os
|
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
import tempfile
|
|
||||||
import unittest
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[2]
|
|
||||||
|
|
||||||
|
|
||||||
class KbAddCLITest(unittest.TestCase):
|
|
||||||
def test_json_add_does_not_delete_db(self) -> None:
|
|
||||||
tmp = Path(tempfile.mkdtemp())
|
|
||||||
dbpath = tmp / "kb.lbug"
|
|
||||||
py = sys.executable
|
|
||||||
venv_py = ROOT / ".venv" / "bin" / "python"
|
|
||||||
if venv_py.is_file():
|
|
||||||
py = str(venv_py)
|
|
||||||
payload = {
|
|
||||||
"text": "cli zebra leaf",
|
|
||||||
"root": "info",
|
|
||||||
"source": "cli-test",
|
|
||||||
"confidence": "confirmed",
|
|
||||||
"how": "test",
|
|
||||||
"loc": str(tmp),
|
|
||||||
"type": "reference",
|
|
||||||
"embedding": [0.0] * 256,
|
|
||||||
}
|
|
||||||
payload["embedding"][0] = 0.3
|
|
||||||
proc = subprocess.run(
|
|
||||||
[py, str(ROOT / "bin" / "kb" / "add"), "--db", str(dbpath), "--json"],
|
|
||||||
cwd=ROOT,
|
|
||||||
input=json.dumps(payload),
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
env=os.environ.copy(),
|
|
||||||
check=False,
|
|
||||||
)
|
|
||||||
self.assertEqual(proc.returncode, 0, proc.stderr)
|
|
||||||
self.assertTrue(dbpath.exists(), "add must create the db, not skip write")
|
|
||||||
out = json.loads(proc.stdout)
|
|
||||||
self.assertEqual(out.get("mode"), "add")
|
|
||||||
self.assertEqual(len(out.get("ids") or []), 1)
|
|
||||||
again = subprocess.run(
|
|
||||||
[py, str(ROOT / "bin" / "kb" / "add"), "--db", str(dbpath), "--json"],
|
|
||||||
cwd=ROOT,
|
|
||||||
input=json.dumps({
|
|
||||||
**payload,
|
|
||||||
"text": "second moose leaf",
|
|
||||||
"source": "cli-test-2",
|
|
||||||
}),
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
env=os.environ.copy(),
|
|
||||||
check=False,
|
|
||||||
)
|
|
||||||
self.assertEqual(again.returncode, 0, again.stderr)
|
|
||||||
self.assertTrue(dbpath.exists())
|
|
||||||
second = json.loads(again.stdout)
|
|
||||||
self.assertEqual(len(second.get("ids") or []), 1)
|
|
||||||
self.assertNotEqual(out["ids"][0], second["ids"][0])
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
unittest.main()
|
|
||||||
@@ -71,69 +71,6 @@ class KblibTest(unittest.TestCase):
|
|||||||
self.assertTrue(hits)
|
self.assertTrue(hits)
|
||||||
self.assertIn("Leaf_vec", kblib.leaf_index_names(self.conn))
|
self.assertIn("Leaf_vec", kblib.leaf_index_names(self.conn))
|
||||||
|
|
||||||
def test_add_after_indexes_keeps_fts_queryable(self):
|
|
||||||
"""Incremental add after FTS+HNSW must find the new leaf on both indexes."""
|
|
||||||
kblib.upsert_leaf(self.conn, text="seed fox leaf", root="info",
|
|
||||||
confidence="confirmed", source="s", source_rev="r1",
|
|
||||||
how="test", loc="/tmp", type_="reference",
|
|
||||||
embedding=make_emb(0.1))
|
|
||||||
kblib.ensure_indexes(self.conn)
|
|
||||||
ids = kblib.add_leafs(self.conn, [{
|
|
||||||
"text": "added zebra after index",
|
|
||||||
"root": "facts",
|
|
||||||
"confidence": "confirmed",
|
|
||||||
"source": "a.md x b.md",
|
|
||||||
"source_rev": "r1",
|
|
||||||
"how": "test",
|
|
||||||
"loc": "/tmp",
|
|
||||||
"type": "fact",
|
|
||||||
"embedding": make_emb(0.9),
|
|
||||||
}])
|
|
||||||
self.assertEqual(len(ids), 1)
|
|
||||||
fts = kblib.query_fts(self.conn, "zebra", 5)
|
|
||||||
self.assertTrue(fts)
|
|
||||||
self.assertIn("zebra", fts[0]["text"])
|
|
||||||
self.assertEqual(fts[0]["root"], "facts")
|
|
||||||
vec = kblib.query_vector(self.conn, make_emb(0.9), 5)
|
|
||||||
self.assertTrue(any("zebra" in h["text"] for h in vec))
|
|
||||||
fox = kblib.query_fts(self.conn, "fox", 5)
|
|
||||||
self.assertTrue(fox)
|
|
||||||
self.assertIn("fox", fox[0]["text"])
|
|
||||||
|
|
||||||
def test_add_facts_and_info_one_transaction(self):
|
|
||||||
"""D12: facts and info land in the same transaction."""
|
|
||||||
kblib.ensure_indexes(self.conn)
|
|
||||||
ids = kblib.add_leafs(self.conn, [
|
|
||||||
{
|
|
||||||
"text": "tx fact leaf two-source",
|
|
||||||
"root": "facts",
|
|
||||||
"confidence": "confirmed",
|
|
||||||
"source": "compose.yml x docker ps",
|
|
||||||
"source_rev": "r1",
|
|
||||||
"how": "test",
|
|
||||||
"loc": "/tmp",
|
|
||||||
"type": "fact",
|
|
||||||
"embedding": make_emb(0.4),
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"text": "tx info narrative",
|
|
||||||
"root": "info",
|
|
||||||
"confidence": "confirmed",
|
|
||||||
"source": "note.md",
|
|
||||||
"source_rev": "r1",
|
|
||||||
"how": "test",
|
|
||||||
"loc": "/tmp",
|
|
||||||
"type": "reference",
|
|
||||||
"embedding": make_emb(0.5),
|
|
||||||
},
|
|
||||||
])
|
|
||||||
self.assertEqual(len(ids), 2)
|
|
||||||
stats = kblib.stats(self.conn)
|
|
||||||
self.assertEqual(stats["by_root"].get("facts"), 1)
|
|
||||||
self.assertEqual(stats["by_root"].get("info"), 1)
|
|
||||||
self.assertTrue(kblib.query_fts(self.conn, "two-source", 5))
|
|
||||||
self.assertTrue(kblib.query_fts(self.conn, "narrative", 5))
|
|
||||||
|
|
||||||
def test_drop_vector_then_create_raises_clear_error(self):
|
def test_drop_vector_then_create_raises_clear_error(self):
|
||||||
"""DROP INDEX leaves ghost catalog; create_fts_and_vector must raise."""
|
"""DROP INDEX leaves ghost catalog; create_fts_and_vector must raise."""
|
||||||
kblib.upsert_leaf(self.conn, text="seed", root="info",
|
kblib.upsert_leaf(self.conn, text="seed", root="info",
|
||||||
@@ -161,39 +98,6 @@ class KblibTest(unittest.TestCase):
|
|||||||
self.assertEqual(stats["total"], 2)
|
self.assertEqual(stats["total"], 2)
|
||||||
self.assertEqual(stats["by_root"], {"facts": 1, "info": 1})
|
self.assertEqual(stats["by_root"], {"facts": 1, "info": 1})
|
||||||
|
|
||||||
def test_hop_1_returns_file_hop_3_reaches_person(self):
|
|
||||||
"""--hop walks FROM_FILE / HAS_VERSION / AUTHORED (Gitea #17)."""
|
|
||||||
import gitimport
|
|
||||||
|
|
||||||
lid = kblib.upsert_leaf(
|
|
||||||
self.conn, text="readme hop fixture", root="info",
|
|
||||||
confidence="confirmed", source="README.md", source_rev="r1",
|
|
||||||
how="test", loc="README.md", type_="reference",
|
|
||||||
embedding=make_emb(0.3),
|
|
||||||
)
|
|
||||||
kblib.link_from_file(self.conn, lid, "README.md", repo="sample-repo")
|
|
||||||
gitimport.index_commits(self.conn, [gitimport.Commit(
|
|
||||||
sha="a1b2c3d",
|
|
||||||
author="Ada Lovelace",
|
|
||||||
email="ada@example.com",
|
|
||||||
date="2026-08-10T12:00:00Z",
|
|
||||||
subject="feat: first commit",
|
|
||||||
files=["README.md"],
|
|
||||||
)], "sample-repo")
|
|
||||||
hop1 = kblib.hop_walk(self.conn, lid, 1)
|
|
||||||
self.assertEqual(len(hop1), 1)
|
|
||||||
self.assertEqual(hop1[0]["label"], "File")
|
|
||||||
self.assertEqual(hop1[0]["name"], "README.md")
|
|
||||||
self.assertEqual(hop1[0]["depth"], 1)
|
|
||||||
hop3 = kblib.hop_walk(self.conn, lid, 3)
|
|
||||||
labels = {n["label"] for n in hop3}
|
|
||||||
self.assertIn("File", labels)
|
|
||||||
self.assertIn("Commit", labels)
|
|
||||||
self.assertIn("Person", labels)
|
|
||||||
person = [n for n in hop3 if n["label"] == "Person"][0]
|
|
||||||
self.assertEqual(person["name"], "Ada Lovelace")
|
|
||||||
self.assertEqual(person["depth"], 3)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
unittest.main()
|
unittest.main()
|
||||||
|
|||||||
@@ -8,13 +8,10 @@ from pathlib import Path
|
|||||||
sys.path.insert(0, os.path.dirname(__file__))
|
sys.path.insert(0, os.path.dirname(__file__))
|
||||||
|
|
||||||
from mailconv import ( # noqa: E402
|
from mailconv import ( # noqa: E402
|
||||||
TESS_LANG,
|
|
||||||
clean_email_address,
|
clean_email_address,
|
||||||
convert_pdf,
|
|
||||||
html_to_markdown,
|
html_to_markdown,
|
||||||
is_convertible,
|
is_convertible,
|
||||||
normalize_markdown,
|
normalize_markdown,
|
||||||
ocr_image,
|
|
||||||
split_zip_members,
|
split_zip_members,
|
||||||
subject_to_filename,
|
subject_to_filename,
|
||||||
zip_extract_safe,
|
zip_extract_safe,
|
||||||
@@ -103,92 +100,6 @@ class TestMailConv(unittest.TestCase):
|
|||||||
self.assertFalse(is_convertible(".exe"))
|
self.assertFalse(is_convertible(".exe"))
|
||||||
self.assertFalse(is_convertible(".unknown"))
|
self.assertFalse(is_convertible(".unknown"))
|
||||||
|
|
||||||
def test_convert_pdf_prefers_pdftotext(self):
|
|
||||||
import mailconv as mc
|
|
||||||
|
|
||||||
calls: list[list[str]] = []
|
|
||||||
|
|
||||||
def fake_run(cmd, **kwargs):
|
|
||||||
calls.append(list(cmd))
|
|
||||||
|
|
||||||
class P:
|
|
||||||
returncode = 0
|
|
||||||
stdout = b"Invoice BM25 layout"
|
|
||||||
stderr = b""
|
|
||||||
|
|
||||||
return P()
|
|
||||||
|
|
||||||
self._patch_run(mc, fake_run)
|
|
||||||
out = convert_pdf(Path(self._tmp("born.pdf")))
|
|
||||||
self.assertIn("BM25", out)
|
|
||||||
self.assertEqual(calls[0][:2], ["pdftotext", "-layout"])
|
|
||||||
self.assertFalse(any(c[0] == "tesseract" for c in calls))
|
|
||||||
self.assertFalse(any(c[0] == "pdftoppm" for c in calls))
|
|
||||||
|
|
||||||
def test_convert_pdf_empty_layer_uses_pdftoppm_tesseract(self):
|
|
||||||
import mailconv as mc
|
|
||||||
|
|
||||||
calls: list[list[str]] = []
|
|
||||||
|
|
||||||
def fake_run(cmd, **kwargs):
|
|
||||||
calls.append(list(cmd))
|
|
||||||
|
|
||||||
class P:
|
|
||||||
returncode = 0
|
|
||||||
stdout = b""
|
|
||||||
stderr = b""
|
|
||||||
|
|
||||||
if cmd[0] == "pdftotext":
|
|
||||||
P.stdout = b" \n"
|
|
||||||
return P()
|
|
||||||
if cmd[0] == "pdftoppm":
|
|
||||||
prefix = Path(cmd[-1])
|
|
||||||
(prefix.parent / "page-1.png").write_bytes(b"fake")
|
|
||||||
return P()
|
|
||||||
if cmd[0] == "tesseract":
|
|
||||||
P.stdout = b"scanned HELLO"
|
|
||||||
return P()
|
|
||||||
return P()
|
|
||||||
|
|
||||||
self._patch_run(mc, fake_run)
|
|
||||||
out = convert_pdf(Path(self._tmp("scan.pdf")))
|
|
||||||
self.assertIn("HELLO", out)
|
|
||||||
bins = [c[0] for c in calls]
|
|
||||||
self.assertIn("pdftotext", bins)
|
|
||||||
self.assertIn("pdftoppm", bins)
|
|
||||||
self.assertIn("tesseract", bins)
|
|
||||||
tess = next(c for c in calls if c[0] == "tesseract")
|
|
||||||
self.assertIn(TESS_LANG, tess)
|
|
||||||
self.assertNotIn("docling", " ".join(bins))
|
|
||||||
|
|
||||||
def test_ocr_image_paddle_engine(self):
|
|
||||||
import mailconv as mc
|
|
||||||
|
|
||||||
calls: list[list[str]] = []
|
|
||||||
|
|
||||||
def fake_run(cmd, **kwargs):
|
|
||||||
calls.append(list(cmd))
|
|
||||||
|
|
||||||
class P:
|
|
||||||
returncode = 0
|
|
||||||
stdout = b"paddle text"
|
|
||||||
stderr = b""
|
|
||||||
|
|
||||||
return P()
|
|
||||||
|
|
||||||
self._patch_run(mc, fake_run)
|
|
||||||
os.environ["OCR_ENGINE"] = "paddle"
|
|
||||||
try:
|
|
||||||
out = ocr_image(Path(self._tmp("x.png")))
|
|
||||||
finally:
|
|
||||||
os.environ.pop("OCR_ENGINE", None)
|
|
||||||
self.assertEqual(out, "paddle text")
|
|
||||||
self.assertEqual(calls[0][:2], ["paddleocr", "ocr"])
|
|
||||||
|
|
||||||
def _patch_run(self, mod, fn) -> None:
|
|
||||||
self.addCleanup(setattr, mod.subprocess, "run", mod.subprocess.run)
|
|
||||||
mod.subprocess.run = fn
|
|
||||||
|
|
||||||
def _mk_zip(self, members):
|
def _mk_zip(self, members):
|
||||||
zpath = Path(self._tmp("arc.zip"))
|
zpath = Path(self._tmp("arc.zip"))
|
||||||
with zipfile.ZipFile(zpath, "w") as zf:
|
with zipfile.ZipFile(zpath, "w") as zf:
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
"""Published docs must match live commands (Gitea SoT, brain/search)."""
|
"""Published docs must match live commands (Gitea SoT, brain/search, no fake --hop)."""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
import unittest
|
import unittest
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -121,7 +122,7 @@ class PublishedDocsTest(unittest.TestCase):
|
|||||||
self.assertIn("D18", plan)
|
self.assertIn("D18", plan)
|
||||||
self.assertIn("Qwen/Qwen3.5-9B", plan)
|
self.assertIn("Qwen/Qwen3.5-9B", plan)
|
||||||
compose = (ROOT / "compose.yaml").read_text()
|
compose = (ROOT / "compose.yaml").read_text()
|
||||||
self.assertIn('"reasoner"', compose)
|
self.assertIn('profiles: ["reasoner"]', compose)
|
||||||
self.assertIn("OLLAMA_NUM_GPU", compose)
|
self.assertIn("OLLAMA_NUM_GPU", compose)
|
||||||
self.assertIn("127.0.0.1:11435", compose)
|
self.assertIn("127.0.0.1:11435", compose)
|
||||||
dockerfile = (ROOT / "Dockerfile").read_text()
|
dockerfile = (ROOT / "Dockerfile").read_text()
|
||||||
@@ -138,21 +139,24 @@ class PublishedDocsTest(unittest.TestCase):
|
|||||||
skill = (ROOT / "skills" / "brain" / "SKILL.md").read_text()
|
skill = (ROOT / "skills" / "brain" / "SKILL.md").read_text()
|
||||||
self.assertIn("`web` block", skill)
|
self.assertIn("`web` block", skill)
|
||||||
|
|
||||||
def test_docs_say_hop_walks_from_file(self) -> None:
|
def test_docs_do_not_claim_hop_walks(self) -> None:
|
||||||
paths = [
|
paths = [
|
||||||
ROOT / "README.md",
|
ROOT / "README.md",
|
||||||
ROOT / "docs" / "design.md",
|
ROOT / "docs" / "design.md",
|
||||||
ROOT / "skills" / "brain" / "SKILL.md",
|
ROOT / "skills" / "brain" / "SKILL.md",
|
||||||
|
ROOT / "skills" / "diataxis-docs" / "SKILL.md",
|
||||||
ROOT / "docs" / "runbook.md",
|
ROOT / "docs" / "runbook.md",
|
||||||
ROOT / "docs" / "README.md",
|
ROOT / "docs" / "README.md",
|
||||||
|
ROOT / "docs" / "roadmap.md",
|
||||||
]
|
]
|
||||||
|
# Command-style `--hop 1` / `--hop N` plus follow/walk = the old lie.
|
||||||
|
# Honest "not implemented" notes must not match.
|
||||||
|
lie = re.compile(r"--hop (?:N|1).*(?:follow|walk)", re.I | re.S)
|
||||||
for path in paths:
|
for path in paths:
|
||||||
text = path.read_text()
|
text = path.read_text()
|
||||||
self.assertIn("--hop", text, f"{path.relative_to(ROOT)} must document --hop")
|
self.assertIsNone(
|
||||||
self.assertNotIn(
|
lie.search(text),
|
||||||
"not implemented",
|
f"{path.relative_to(ROOT)} still claims --hop walks the graph",
|
||||||
text.lower(),
|
|
||||||
f"{path.relative_to(ROOT)} still says hop is not implemented",
|
|
||||||
)
|
)
|
||||||
|
|
||||||
def test_docs_are_portable_diataxis(self) -> None:
|
def test_docs_are_portable_diataxis(self) -> None:
|
||||||
|
|||||||
@@ -47,17 +47,3 @@ class SkillsTest(unittest.TestCase):
|
|||||||
self.assertIn("throttled", skill.lower())
|
self.assertIn("throttled", skill.lower())
|
||||||
self.assertIn("not a negative finding", agents)
|
self.assertIn("not a negative finding", agents)
|
||||||
self.assertIn("Fact-check every", agents)
|
self.assertIn("Fact-check every", agents)
|
||||||
|
|
||||||
def test_yq_is_mikefarah_for_structured_data(self) -> None:
|
|
||||||
skill = (ROOT / "skills" / "yq" / "SKILL.md").read_text()
|
|
||||||
self.assertIn("https://github.com/mikefarah/yq", skill)
|
|
||||||
for fmt in ("YAML", "JSON", "XML", "CSV", "TOML", "HCL"):
|
|
||||||
self.assertIn(fmt, skill)
|
|
||||||
self.assertIn("not kislyuk", skill.lower())
|
|
||||||
plan = (ROOT / "PLAN.md").read_text()
|
|
||||||
self.assertIn("mikefarah/yq", plan)
|
|
||||||
agents = (ROOT / "AGENTS.md").read_text()
|
|
||||||
self.assertIn("mikefarah/yq", agents)
|
|
||||||
web = (ROOT / "skills" / "web-search" / "SKILL.md").read_text()
|
|
||||||
self.assertIn("| yq ", web)
|
|
||||||
self.assertNotIn("| jq ", web)
|
|
||||||
|
|||||||
@@ -1,36 +0,0 @@
|
|||||||
"""qa/system_perf.py is an offline-gated system test (no live brain in CI)."""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import ast
|
|
||||||
import unittest
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[2]
|
|
||||||
|
|
||||||
|
|
||||||
class SystemPerfScriptTest(unittest.TestCase):
|
|
||||||
def test_script_compiles_and_is_read_only(self) -> None:
|
|
||||||
path = ROOT / "qa" / "system_perf.py"
|
|
||||||
src = path.read_text()
|
|
||||||
compile(src, str(path), "exec")
|
|
||||||
self.assertIn("--json", src)
|
|
||||||
self.assertIn("qwen3.5:9b", src)
|
|
||||||
self.assertIn("--picoclaw", src)
|
|
||||||
self.assertIn("BRAIN_URL", src)
|
|
||||||
self.assertIn("tools/list", src)
|
|
||||||
self.assertIn("tools/call", src)
|
|
||||||
self.assertIn("GATE_HEALTH_MS", src)
|
|
||||||
self.assertIn("GATE_GET_P50_MS", src)
|
|
||||||
self.assertNotIn("kb.lbug", src)
|
|
||||||
self.assertNotIn("password", src.lower())
|
|
||||||
self.assertNotIn("token", src.lower())
|
|
||||||
|
|
||||||
def test_script_does_not_write_ladybug(self) -> None:
|
|
||||||
tree = ast.parse((ROOT / "qa" / "system_perf.py").read_text())
|
|
||||||
writes = [
|
|
||||||
n.func.attr
|
|
||||||
for n in ast.walk(tree)
|
|
||||||
if isinstance(n, ast.Call) and isinstance(n.func, ast.Attribute)
|
|
||||||
and n.func.attr in {"write_text", "write_bytes", "dump"}
|
|
||||||
]
|
|
||||||
self.assertEqual(writes, [], f"system_perf must not write files: {writes}")
|
|
||||||
+4
-43
@@ -2,23 +2,14 @@
|
|||||||
#
|
#
|
||||||
# docker compose up -d brain # API (Zig CGO serve)
|
# docker compose up -d brain # API (Zig CGO serve)
|
||||||
# docker compose --profile index run --rm index # Python rebuild
|
# docker compose --profile index run --rm index # Python rebuild
|
||||||
# docker compose --profile picoclaw up -d # brain-mcp + CPU reasoner + PicoClaw gateway
|
# docker compose --profile picoclaw up brain-mcp
|
||||||
# docker compose --profile reasoner up -d reasoner # CPU Ollama :11435
|
# docker compose --profile reasoner up -d reasoner # CPU Ollama :11435
|
||||||
# docker compose --profile searxng up -d
|
# docker compose --profile searxng up -d
|
||||||
# OCR_ENGINE=paddle docker compose --profile ocr-paddle run --rm ocr-paddle
|
|
||||||
#
|
#
|
||||||
# Secrets never baked in: search.env + db-profiles.yml from ~/.config/brain.
|
# Secrets never baked in: search.env + db-profiles.yml from ~/.config/brain.
|
||||||
|
|
||||||
name: 2dph
|
name: 2dph
|
||||||
|
|
||||||
networks:
|
|
||||||
default:
|
|
||||||
name: 2dph_sys
|
|
||||||
driver: bridge
|
|
||||||
ipam:
|
|
||||||
config:
|
|
||||||
- subnet: 10.23.42.0/24
|
|
||||||
|
|
||||||
services:
|
services:
|
||||||
brain:
|
brain:
|
||||||
image: ghcr.io/eslider/2dph:api
|
image: ghcr.io/eslider/2dph:api
|
||||||
@@ -106,8 +97,8 @@ services:
|
|||||||
- ./deploy/searxng/limiter.toml:/etc/searxng/limiter.toml:ro
|
- ./deploy/searxng/limiter.toml:/etc/searxng/limiter.toml:ro
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
|
||||||
# MCP endpoint for PicoClaw (and any MCP client).
|
# MCP endpoint for an external agent (PicoClaw is not shipped here).
|
||||||
# docker compose --profile picoclaw up -d
|
# docker compose --profile picoclaw up brain-mcp
|
||||||
brain-mcp:
|
brain-mcp:
|
||||||
profiles: ["picoclaw"]
|
profiles: ["picoclaw"]
|
||||||
image: ghcr.io/eslider/2dph:api
|
image: ghcr.io/eslider/2dph:api
|
||||||
@@ -129,7 +120,7 @@ services:
|
|||||||
# docker compose --profile reasoner up -d reasoner
|
# docker compose --profile reasoner up -d reasoner
|
||||||
# docker compose --profile reasoner exec reasoner ollama pull qwen3.5:9b
|
# docker compose --profile reasoner exec reasoner ollama pull qwen3.5:9b
|
||||||
reasoner:
|
reasoner:
|
||||||
profiles: ["reasoner", "picoclaw"]
|
profiles: ["reasoner"]
|
||||||
image: docker.io/ollama/ollama:latest
|
image: docker.io/ollama/ollama:latest
|
||||||
environment:
|
environment:
|
||||||
OLLAMA_NUM_GPU: "0"
|
OLLAMA_NUM_GPU: "0"
|
||||||
@@ -140,37 +131,7 @@ services:
|
|||||||
- reasoner-ollama:/root/.ollama
|
- reasoner-ollama:/root/.ollama
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
|
||||||
# Official PicoClaw gateway. Config has no secrets (Ollama + HTTP MCP).
|
|
||||||
# Host network: brain/reasoner bind 127.0.0.1 only, so host.docker.internal
|
|
||||||
# (docker0) cannot reach them. Gateway 127.0.0.1:18790 (not the 18800 launcher).
|
|
||||||
# If :8630/:11435 are already bound, do not start brain-mcp/reasoner:
|
|
||||||
# docker compose --profile picoclaw up -d --no-deps picoclaw
|
|
||||||
picoclaw:
|
|
||||||
profiles: ["picoclaw"]
|
|
||||||
image: docker.io/sipeed/picoclaw:v0.3.1
|
|
||||||
network_mode: host
|
|
||||||
depends_on:
|
|
||||||
- brain-mcp
|
|
||||||
- reasoner
|
|
||||||
environment:
|
|
||||||
PICOCLAW_GATEWAY_HOST: "127.0.0.1"
|
|
||||||
entrypoint: ["picoclaw", "gateway"]
|
|
||||||
volumes:
|
|
||||||
- picoclaw-home:/root/.picoclaw
|
|
||||||
- ./deploy/picoclaw/config.json:/root/.picoclaw/config.json:ro
|
|
||||||
restart: unless-stopped
|
|
||||||
|
|
||||||
# Optional PP-OCRv5 (not default). Default OCR is tesseract eng+deu.
|
|
||||||
# OCR_ENGINE=paddle docker compose --profile ocr-paddle run --rm ocr-paddle
|
|
||||||
ocr-paddle:
|
|
||||||
profiles: ["ocr-paddle"]
|
|
||||||
image: python:3.12-slim
|
|
||||||
environment:
|
|
||||||
OCR_ENGINE: paddle
|
|
||||||
command: ["python", "-c", "print('OCR_ENGINE=paddle; install paddleocr on PATH')"]
|
|
||||||
|
|
||||||
volumes:
|
volumes:
|
||||||
kb-model:
|
kb-model:
|
||||||
kb-var:
|
kb-var:
|
||||||
reasoner-ollama:
|
reasoner-ollama:
|
||||||
picoclaw-home:
|
|
||||||
|
|||||||
@@ -1,33 +0,0 @@
|
|||||||
{
|
|
||||||
"agents": {
|
|
||||||
"defaults": {
|
|
||||||
"model_name": "qwen3.5-9b",
|
|
||||||
"max_tool_iterations": 8,
|
|
||||||
"max_tokens": 512,
|
|
||||||
"context_window": 8192
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"model_list": [
|
|
||||||
{
|
|
||||||
"model_name": "qwen3.5-9b",
|
|
||||||
"model": "ollama/qwen3.5:9b",
|
|
||||||
"api_base": "http://127.0.0.1:11435/v1",
|
|
||||||
"request_timeout": 600
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"tools": {
|
|
||||||
"web": {
|
|
||||||
"enabled": false
|
|
||||||
},
|
|
||||||
"mcp": {
|
|
||||||
"enabled": true,
|
|
||||||
"servers": {
|
|
||||||
"2dph": {
|
|
||||||
"enabled": true,
|
|
||||||
"type": "http",
|
|
||||||
"url": "http://127.0.0.1:8630/mcp"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
+4
-4
@@ -20,16 +20,16 @@ Evidence-first knowledge graph. Facts need proof or they are
|
|||||||
| explanation | [roadmap](roadmap.md) — gap to v1 (epic #16) |
|
| explanation | [roadmap](roadmap.md) — gap to v1 (epic #16) |
|
||||||
| howto | [picoclaw](picoclaw.md) — MCP agent profile |
|
| howto | [picoclaw](picoclaw.md) — MCP agent profile |
|
||||||
| howto | [reasoner](reasoner.md) — CPU bake-off (D18) |
|
| howto | [reasoner](reasoner.md) — CPU bake-off (D18) |
|
||||||
| reference | [PLAN.md](../PLAN.md) — decisions D1–D22 |
|
| reference | [PLAN.md](../PLAN.md) — decisions D1–D21 |
|
||||||
|
|
||||||
Decisions the public face must name: **D3** SearXNG compose, **D6** Go service /
|
Decisions the public face must name: **D3** SearXNG compose, **D6** Go service /
|
||||||
Python write sidecar, **D14** `bin/{subject}/{method}.go`, **D15** Gitea origin,
|
Python write sidecar, **D14** `bin/{subject}/{method}.go`, **D15** Gitea origin,
|
||||||
**D17** assertion gate (facts → info → web), **D18** pluggable reasoner.
|
**D17** assertion gate (facts → info → web), **D18** pluggable reasoner.
|
||||||
|
|
||||||
Search: `bin/brain/search.go "query"` (HTTP: `bin/brain/serve.go` —
|
Search: `bin/brain/search.go "query"` (HTTP: `bin/brain/serve.go` —
|
||||||
`/health` `/search` `/get` `/stats` `/audit` `/ingest`). `--hop N` walks
|
`/health` `/search` `/get` `/stats` `/audit` `/ingest`). `--hop` is
|
||||||
`FROM_FILE` → Commit → Person from each hit (max 3). Rebuild writes
|
not a walk; the flag errors. Schema has `FROM_FILE`; search does not
|
||||||
File edges ([#17](https://git.produktor.io/eSlider/2dph/issues/17)).
|
use it ([#17](https://git.produktor.io/eSlider/2dph/issues/17)).
|
||||||
|
|
||||||
Work board: [Gitea issues](https://git.produktor.io/eSlider/2dph/issues)
|
Work board: [Gitea issues](https://git.produktor.io/eSlider/2dph/issues)
|
||||||
([epic #16](https://git.produktor.io/eSlider/2dph/issues/16)).
|
([epic #16](https://git.produktor.io/eSlider/2dph/issues/16)).
|
||||||
|
|||||||
@@ -21,5 +21,4 @@ OO_CLI (default: $HOME/go/bin/oo)
|
|||||||
./bin/chats/apply.go --dry-run
|
./bin/chats/apply.go --dry-run
|
||||||
```
|
```
|
||||||
|
|
||||||
JSONL → markdown only. Brain ingest is `bin/brain/index.go --with-chats`
|
JSONL → markdown only. Brain ingest is `bin/brain/index.go` (not a `chats index`).
|
||||||
(default `var/chats/md`). WhatsApp sync is out of v1.
|
|
||||||
|
|||||||
+6
-8
@@ -34,9 +34,9 @@ bin/brain/search.go "question"
|
|||||||
is not evidence of absence; `--no-web` / `--root` skip it)
|
is not evidence of absence; `--no-web` / `--root` skip it)
|
||||||
```
|
```
|
||||||
|
|
||||||
`--hop N` walks `Leaf-[:FROM_FILE]->File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person`
|
`--hop` is not implemented. `FROM_FILE` / `HAS_VERSION` exist in schema;
|
||||||
from each hit (1=File, 2=Commit, 3=Person). Rebuild writes FROM_FILE;
|
search does not walk them ([#17](https://git.produktor.io/eSlider/2dph/issues/17)).
|
||||||
git import writes HAS_VERSION/AUTHORED ([#17](https://git.produktor.io/eSlider/2dph/issues/17)).
|
The flag is an error; it is not a graph walk.
|
||||||
|
|
||||||
## Who / What / How / Where / When + evidence
|
## Who / What / How / Where / When + evidence
|
||||||
|
|
||||||
@@ -80,16 +80,14 @@ Conflicting pairings (≥2 yes vs ≥2 no) = hypothesis (OQ1 → v2 resolution).
|
|||||||
They do not exec Python. Control questions for recall@5 live in
|
They do not exec Python. Control questions for recall@5 live in
|
||||||
`internal/brain/rank` so CI can test the table without libladybug.
|
`internal/brain/rank` so CI can test the table without libladybug.
|
||||||
Python `bin/kb/{get,stats,eval}` remain for GitHub Actions until the runner
|
Python `bin/kb/{get,stats,eval}` remain for GitHub Actions until the runner
|
||||||
fetches Zig + libs (`bin/cgo/zig`). Incremental write is `bin/kb/add`
|
fetches Zig + libs (`bin/cgo/zig`). Index/write is still `bin/kb/index`
|
||||||
(`bin/brain/add.go`). Bulk index/write is still `bin/kb/index`
|
|
||||||
(`docker compose --profile index`).
|
(`docker compose --profile index`).
|
||||||
|
|
||||||
## Agent API (D20)
|
## Agent API (D20)
|
||||||
|
|
||||||
`bin/brain/serve.go` exposes the same `internal/httpapi.Ops` table as OpenAPI
|
`bin/brain/serve.go` exposes the same `internal/httpapi.Ops` table as OpenAPI
|
||||||
(`GET /openapi.json`) and MCP (`POST /mcp` JSON-RPC `tools/list` +
|
(`GET /openapi.json`) and MCP (`POST /mcp` JSON-RPC `tools/list` +
|
||||||
`tools/call`). Tool names match paths: `search`, `get`, `stats`, `audit`,
|
`tools/call`). Tool names match paths: `search`, `get`, `stats`, `audit`.
|
||||||
`ingest` (add a leaf; omit body for the CLI hint).
|
|
||||||
Agents should use these endpoints instead of shebang CLIs.
|
Agents should use these endpoints instead of shebang CLIs.
|
||||||
|
|
||||||
## Reasoner (D18)
|
## Reasoner (D18)
|
||||||
@@ -100,5 +98,5 @@ CPU sidecar: compose profile `reasoner` (`OLLAMA_NUM_GPU=0`,
|
|||||||
`127.0.0.1:11435`). Bake-off: `bin/reasoner/bakeoff.go`. Weights stay out
|
`127.0.0.1:11435`). Bake-off: `bin/reasoner/bakeoff.go`. Weights stay out
|
||||||
of the 2dph image. See [docs/reasoner.md](reasoner.md).
|
of the 2dph image. See [docs/reasoner.md](reasoner.md).
|
||||||
|
|
||||||
Gap to v1 (hops, corpus, CI eval): [roadmap](roadmap.md),
|
Gap to v1 (write, hops, corpus, agent loop): [roadmap](roadmap.md),
|
||||||
[epic #16](https://git.produktor.io/eSlider/2dph/issues/16).
|
[epic #16](https://git.produktor.io/eSlider/2dph/issues/16).
|
||||||
+6
-22
@@ -1,34 +1,18 @@
|
|||||||
# PicoClaw profile (reference agent)
|
# PicoClaw profile (reference agent)
|
||||||
|
|
||||||
2dph is the memory/fact gate. Compose profile `picoclaw` runs the official
|
2dph is the memory/fact gate. PicoClaw (or any MCP client) is the agent loop
|
||||||
PicoClaw gateway (`docker.io/sipeed/picoclaw:v0.3.1`) plus `brain-mcp` and the
|
and is **not** shipped in this repo.
|
||||||
CPU reasoner. Default agent model is `qwen3.5:9b` (RAM path, D18). Weights stay
|
|
||||||
in the reasoner volume, not in the 2dph image.
|
|
||||||
No secrets in git: Ollama needs no key; MCP is local HTTP.
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
docker compose --profile picoclaw up -d
|
docker compose --profile picoclaw up brain-mcp
|
||||||
# already serving :8630 / :11435:
|
|
||||||
docker compose --profile picoclaw up -d --no-deps picoclaw
|
|
||||||
```
|
```
|
||||||
|
|
||||||
Gateway: `127.0.0.1:18790`. Brain MCP: `http://127.0.0.1:8630/mcp`.
|
The API listens on `127.0.0.1:8630`. Point the agent at
|
||||||
Cursor-style clients can use [deploy/picoclaw/mcp.json.example](../deploy/picoclaw/mcp.json.example).
|
`http://127.0.0.1:8630/mcp` using [deploy/picoclaw/mcp.json.example](../deploy/picoclaw/mcp.json.example).
|
||||||
PicoClaw itself uses [deploy/picoclaw/config.json](../deploy/picoclaw/config.json)
|
|
||||||
(`127.0.0.1` + host network — loopback publishes are not reachable via docker0).
|
|
||||||
|
|
||||||
OpenAPI: `GET http://127.0.0.1:8630/openapi.json`.
|
OpenAPI: `GET http://127.0.0.1:8630/openapi.json`.
|
||||||
|
|
||||||
Before a factual reply: `search` → `get` → `audit`. `throttled` is not a
|
Before a factual reply: `search` → `get` → `audit`. `throttled` is not a
|
||||||
negative finding. See `skills/picoclaw/SKILL.md`.
|
negative finding. See `skills/picoclaw/SKILL.md`.
|
||||||
|
|
||||||
System performance (MCP gates + qwen3.5:9b tool_call + PicoClaw gateway):
|
No Cursor required. A live PicoClaw binary/image is an operator choice.
|
||||||
|
|
||||||
```bash
|
|
||||||
./qa/system_perf.py --json | yq '.gates'
|
|
||||||
REASONER_MODEL=qwen3.5:9b ./qa/system_perf.py --reasoner --picoclaw --json | yq '.reasoner'
|
|
||||||
```
|
|
||||||
|
|
||||||
The default agent model is `qwen3.5:9b`. PicoClaw `context_window` is 8192
|
|
||||||
(heuristic `max_tokens*4` at 512 is 2048, too small for MCP tool schemas).
|
|
||||||
`request_timeout` is 600s for a CPU turn (tool_call + MCP search + answer).
|
|
||||||
|
|||||||
+2
-4
@@ -1,8 +1,8 @@
|
|||||||
# Reasoner bake-off (D18)
|
# Reasoner bake-off (D18)
|
||||||
|
|
||||||
Pluggable OpenAI-compatible URL. 2dph does not ship weights. PicoClaw is
|
Pluggable OpenAI-compatible URL. 2dph does not ship weights. PicoClaw is
|
||||||
compose profile `picoclaw` (`sipeed/picoclaw`); the bake-off hits the same
|
not in this repo; the bake-off hits the same tool names PicoClaw would
|
||||||
tool names (`search` → `get` → `audit` from `internal/httpapi.Ops`).
|
(`search` → `get` → `audit` from `internal/httpapi.Ops`).
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
docker compose --profile reasoner up -d reasoner
|
docker compose --profile reasoner up -d reasoner
|
||||||
@@ -11,8 +11,6 @@ REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b \
|
|||||||
./bin/reasoner/bakeoff.go --json
|
./bin/reasoner/bakeoff.go --json
|
||||||
```
|
```
|
||||||
|
|
||||||
JSON includes `latency_p50_ms` / `latency_p95_ms` from DuckDB (`internal/duckstats`, D22).
|
|
||||||
|
|
||||||
Host Ollama on `:11434` is left alone. This sidecar binds `127.0.0.1:11435`
|
Host Ollama on `:11434` is left alone. This sidecar binds `127.0.0.1:11435`
|
||||||
with `OLLAMA_NUM_GPU=0` (CPU). Measure RSS (`/api/ps` `size`), not VRAM.
|
with `OLLAMA_NUM_GPU=0` (CPU). Measure RSS (`/api/ps` `size`), not VRAM.
|
||||||
|
|
||||||
|
|||||||
+25
-24
@@ -23,44 +23,45 @@ Decisions: [PLAN.md](../PLAN.md).
|
|||||||
Read path Go + Zig CGO (D21). HTTP + OpenAPI + MCP (D20). PicoClaw compose
|
Read path Go + Zig CGO (D21). HTTP + OpenAPI + MCP (D20). PicoClaw compose
|
||||||
profile + CPU reasoner (D18). Mail sync → import → rebuild. D14 shebangs.
|
profile + CPU reasoner (D18). Mail sync → import → rebuild. D14 shebangs.
|
||||||
Compose `api` (no CPython) / `index` (Python write). Issues #1–#5, #7–#13.
|
Compose `api` (no CPython) / `index` (Python write). Issues #1–#5, #7–#13.
|
||||||
[#15](https://git.produktor.io/eSlider/2dph/issues/15) lever/loop.
|
|
||||||
[#14](https://git.produktor.io/eSlider/2dph/issues/14) `bin/brain/add.go` /
|
|
||||||
`POST /ingest` (Python `kblib.add_leafs`; no Go upsert port).
|
|
||||||
[#17](https://git.produktor.io/eSlider/2dph/issues/17) `--hop N` walks
|
|
||||||
FROM_FILE / HAS_VERSION / AUTHORED.
|
|
||||||
[#18](https://git.produktor.io/eSlider/2dph/issues/18) `--with-facts` /
|
|
||||||
`--with-chats` on rebuild (WhatsApp out of v1).
|
|
||||||
[#19](https://git.produktor.io/eSlider/2dph/issues/19) CI recall SoT =
|
|
||||||
`bin/brain/eval.go` via Zig.
|
|
||||||
Epic [#16](https://git.produktor.io/eSlider/2dph/issues/16) closed.
|
|
||||||
|
|
||||||
## v2
|
`POST /ingest` is a rebuild **hint**. `add` is not implemented.
|
||||||
|
|
||||||
[#6](https://git.produktor.io/eSlider/2dph/issues/6) OCR — **in**.
|
|
||||||
[#30](https://git.produktor.io/eSlider/2dph/issues/30) OQ3 duckdb-go — **in**.
|
|
||||||
[#29](https://git.produktor.io/eSlider/2dph/issues/29) OQ1 contradiction
|
|
||||||
resolution.
|
|
||||||
|
|
||||||
## Blockers
|
## Blockers
|
||||||
|
|
||||||
None for epic #16 (closed). Remaining v2: OQ1, OQ4.
|
|
||||||
|
|
||||||
```
|
```
|
||||||
question
|
question
|
||||||
│
|
│
|
||||||
├─ FTS + HNSW ← in
|
├─ FTS + HNSW ← in
|
||||||
├─ facts / info roots ← in
|
├─ facts / info roots ← in
|
||||||
├─ web (D17) ← in
|
├─ web (D17) ← in
|
||||||
├─ brain/add ACID ← in
|
├─ Cypher hop ← #17 schema yes, search no
|
||||||
├─ Cypher hop ← in
|
├─ brain/add ACID ← #14 rebuild only
|
||||||
└─ facts+chats corpus ← in
|
└─ facts+chats corpus ← #18
|
||||||
```
|
```
|
||||||
|
|
||||||
|
1. **[#14](https://git.produktor.io/eSlider/2dph/issues/14) write** —
|
||||||
|
`bin/brain/index.go --rebuild` (Python `kblib`). No incremental
|
||||||
|
`brain/add`. Watch/mail/git cannot become facts “now”.
|
||||||
|
2. **[#17](https://git.produktor.io/eSlider/2dph/issues/17) hops** —
|
||||||
|
`Leaf-[:FROM_FILE]->File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person`
|
||||||
|
exists; `--hop` still errors. Without a walk, D9/D10 are paper.
|
||||||
|
3. **[#18](https://git.produktor.io/eSlider/2dph/issues/18) corpus** —
|
||||||
|
rebuild loads repo markdown + mail as `info`. `facts/extract` pairing
|
||||||
|
and `bin/chats` are not indexed. WhatsApp is a stub. PII stays in `var/`.
|
||||||
|
4. **[#15](https://git.produktor.io/eSlider/2dph/issues/15) lever/loop** —
|
||||||
|
2dph is the lever (`search` → `get` → `audit`). PicoClaw is the loop.
|
||||||
|
Document the contour in-repo; CPU turns need a large context window.
|
||||||
|
5. **[#19](https://git.produktor.io/eSlider/2dph/issues/19) CI eval** —
|
||||||
|
recall SoT should be `bin/brain/eval.go` via Zig, not Python `bin/kb/eval`.
|
||||||
|
|
||||||
## Not v1
|
## Not v1
|
||||||
|
|
||||||
OQ1 contradiction resolution, OQ4 YAML-first leafs.
|
[#6](https://git.produktor.io/eSlider/2dph/issues/6) OCR (OQ2), OQ1
|
||||||
OCR (OQ2) and duckdb-go (OQ3/D22) are in.
|
contradiction resolution, OQ3 duckdb-md export, OQ4 YAML-first leafs.
|
||||||
|
|
||||||
## Close epic #16 when
|
## Close epic #16 when
|
||||||
|
|
||||||
Children #14, #15, #17, #18, #19 are closed. MCP tool order stays gated by tests.
|
- facts+info can be written without a full rebuild for every leaf
|
||||||
|
- `--hop` stops erroring and runs a Cypher path from search hits
|
||||||
|
- ops pairing + chat import land as leafs on rebuild
|
||||||
|
- MCP tool order is documented and still gated by tests
|
||||||
|
|||||||
+4
-7
@@ -16,7 +16,6 @@ No laptop-absolute paths. Config lives in env files under `$HOME/.config/brain/`
|
|||||||
- Go (see `go.mod`)
|
- Go (see `go.mod`)
|
||||||
- Python 3.12 + [uv](https://docs.astral.sh/uv)
|
- Python 3.12 + [uv](https://docs.astral.sh/uv)
|
||||||
- Optional: Docker, Zig CGO via `bin/cgo/zig` (not gcc)
|
- Optional: Docker, Zig CGO via `bin/cgo/zig` (not gcc)
|
||||||
- Optional: poppler (`pdftotext`/`pdftoppm`) + tesseract `eng+deu` for mail OCR
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv venv .venv
|
uv venv .venv
|
||||||
@@ -44,20 +43,18 @@ That binds `127.0.0.1:8888`. JSON format must stay enabled.
|
|||||||
|
|
||||||
## Index then search
|
## Index then search
|
||||||
|
|
||||||
Write path is `bin/brain/add.go` for a leaf (or `POST /ingest`). Bulk
|
Write path is Compose profile `index` (Python Ladybug rebuild) until
|
||||||
corpus rebuild remains `bin/brain/index.go --rebuild` (Compose profile
|
`brain/add` is v2. The operator command is `bin/brain/index.go`.
|
||||||
`index`). Do not DROP INDEX on Ladybug 0.19.
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
bin/brain/add.go --text "arc-1 runs Matrix" --root facts --source "compose.yml x docker ps"
|
bin/brain/index.go --rebuild
|
||||||
bin/brain/index.go --rebuild --with-facts --with-chats
|
|
||||||
bin/brain/search.go "LadybugDB vector index" # facts → info → web (D17)
|
bin/brain/search.go "LadybugDB vector index" # facts → info → web (D17)
|
||||||
bin/brain/search.go "upstream flag" --no-web
|
bin/brain/search.go "upstream flag" --no-web
|
||||||
bin/brain/get.go <id> --body
|
bin/brain/get.go <id> --body
|
||||||
bin/brain/stats.go
|
bin/brain/stats.go
|
||||||
```
|
```
|
||||||
|
|
||||||
`--hop N` walks File → Commit → Person from each hit. Empty web results are `throttled`, not absence.
|
`--hop` is not implemented. Empty web results are `throttled`, not absence.
|
||||||
Gap to v1: [roadmap](roadmap.md) / [epic #16](https://git.produktor.io/eSlider/2dph/issues/16).
|
Gap to v1: [roadmap](roadmap.md) / [epic #16](https://git.produktor.io/eSlider/2dph/issues/16).
|
||||||
|
|
||||||
Ladybug 0.19: never `DROP INDEX` FTS/VECTOR (ghost catalog). Fresh indexes =
|
Ladybug 0.19: never `DROP INDEX` FTS/VECTOR (ghost catalog). Fresh indexes =
|
||||||
|
|||||||
@@ -7,7 +7,6 @@ require (
|
|||||||
github.com/arran4/golang-ical v0.3.5
|
github.com/arran4/golang-ical v0.3.5
|
||||||
github.com/chewxy/math32 v1.11.2
|
github.com/chewxy/math32 v1.11.2
|
||||||
github.com/daulet/tokenizers v1.27.0
|
github.com/daulet/tokenizers v1.27.0
|
||||||
github.com/duckdb/duckdb-go/v2 v2.10505.0
|
|
||||||
github.com/go-git/go-git/v5 v5.19.2
|
github.com/go-git/go-git/v5 v5.19.2
|
||||||
golang.org/x/sys v0.47.0
|
golang.org/x/sys v0.47.0
|
||||||
golang.org/x/text v0.40.0
|
golang.org/x/text v0.40.0
|
||||||
@@ -21,17 +20,10 @@ require (
|
|||||||
github.com/apache/arrow-go/v18 v18.6.0 // indirect
|
github.com/apache/arrow-go/v18 v18.6.0 // indirect
|
||||||
github.com/cloudflare/circl v1.6.3 // indirect
|
github.com/cloudflare/circl v1.6.3 // indirect
|
||||||
github.com/cyphar/filepath-securejoin v0.6.1 // indirect
|
github.com/cyphar/filepath-securejoin v0.6.1 // indirect
|
||||||
github.com/duckdb/duckdb-go-bindings v0.10505.0 // indirect
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-amd64 v0.10505.0 // indirect
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-arm64 v0.10505.0 // indirect
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/linux-amd64 v0.10505.0 // indirect
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/linux-arm64 v0.10505.0 // indirect
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/windows-amd64 v0.10505.0 // indirect
|
|
||||||
github.com/dustin/go-humanize v1.0.1 // indirect
|
github.com/dustin/go-humanize v1.0.1 // indirect
|
||||||
github.com/emirpasic/gods v1.18.1 // indirect
|
github.com/emirpasic/gods v1.18.1 // indirect
|
||||||
github.com/go-git/gcfg v1.5.1-0.20230307220236-3a3c6141e376 // indirect
|
github.com/go-git/gcfg v1.5.1-0.20230307220236-3a3c6141e376 // indirect
|
||||||
github.com/go-git/go-billy/v5 v5.9.0 // indirect
|
github.com/go-git/go-billy/v5 v5.9.0 // indirect
|
||||||
github.com/go-viper/mapstructure/v2 v2.5.0 // indirect
|
|
||||||
github.com/goccy/go-json v0.10.6 // indirect
|
github.com/goccy/go-json v0.10.6 // indirect
|
||||||
github.com/golang/groupcache v0.0.0-20241129210726-2c02b8208cf8 // indirect
|
github.com/golang/groupcache v0.0.0-20241129210726-2c02b8208cf8 // indirect
|
||||||
github.com/google/flatbuffers v25.12.19+incompatible // indirect
|
github.com/google/flatbuffers v25.12.19+incompatible // indirect
|
||||||
|
|||||||
@@ -31,20 +31,6 @@ github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSs
|
|||||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||||
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM=
|
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM=
|
||||||
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||||
github.com/duckdb/duckdb-go-bindings v0.10505.0 h1:/0pPsTLrcCsTGxT0VrHgJWnOcPe1tQL1vrki1v3jbAI=
|
|
||||||
github.com/duckdb/duckdb-go-bindings v0.10505.0/go.mod h1:HoD5xePkDj3VZbBnVVfxVVYIljZ9khCprWA7FgwIiC4=
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-amd64 v0.10505.0 h1:FrMqquFBQlMsi34h2KZgCku54rqA8xEbXZ0NLVDKwYs=
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-amd64 v0.10505.0/go.mod h1:EnAvZh1kNJHp5yF+M1ZHNEvapnmt6anq1xXHVrAGqMo=
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-arm64 v0.10505.0 h1:lbRbpQwT1MmUhh/VTwukV9K8bxKByV3UghAP3MvsbBo=
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-arm64 v0.10505.0/go.mod h1:IGLSeEcFhNeZF16aVjQCULD7TsFZKG5G7SyKJAXKp5c=
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/linux-amd64 v0.10505.0 h1:nrsaVYj3XYCRbS2FpdOMD/KHE7egRMr+/NR1IHmjT84=
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/linux-amd64 v0.10505.0/go.mod h1:KAIynZ0GHCS7X5fRyuFnQMg/SZBPK/bS9OCOVojClxw=
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/linux-arm64 v0.10505.0 h1:qM6oGDgwXBILJGbTY4fCy6QOczLpucUA6yn6g3ORjh4=
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/linux-arm64 v0.10505.0/go.mod h1:81SGOYoEUs8qaAfSk1wRfM5oobrIJ5KI7AzYhK6/bvQ=
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/windows-amd64 v0.10505.0 h1:DjqZl9rYreHkSOqnqLmkrqH5T8UdQNcxZLJVZzGmXXA=
|
|
||||||
github.com/duckdb/duckdb-go-bindings/lib/windows-amd64 v0.10505.0/go.mod h1:K25pJL26ARblGDeuAkrdblFvUen92+CwksLtPEHRqqQ=
|
|
||||||
github.com/duckdb/duckdb-go/v2 v2.10505.0 h1:SWwvLn2Qx/RQSnQNupwgIF8VbnJ5A6OQU9lYb/mDETI=
|
|
||||||
github.com/duckdb/duckdb-go/v2 v2.10505.0/go.mod h1:m0PW4J4FG9hlFlVdXi6Ds9owpyIDaBdE2jyce00fGcE=
|
|
||||||
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
|
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
|
||||||
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
|
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
|
||||||
github.com/elazarl/goproxy v1.7.2 h1:Y2o6urb7Eule09PjlhQRGNsqRfPmYI3KKQLFpCAV3+o=
|
github.com/elazarl/goproxy v1.7.2 h1:Y2o6urb7Eule09PjlhQRGNsqRfPmYI3KKQLFpCAV3+o=
|
||||||
@@ -61,8 +47,6 @@ github.com/go-git/go-git-fixtures/v4 v4.3.2-0.20231010084843-55a94097c399 h1:eMj
|
|||||||
github.com/go-git/go-git-fixtures/v4 v4.3.2-0.20231010084843-55a94097c399/go.mod h1:1OCfN199q1Jm3HZlxleg+Dw/mwps2Wbk9frAWm+4FII=
|
github.com/go-git/go-git-fixtures/v4 v4.3.2-0.20231010084843-55a94097c399/go.mod h1:1OCfN199q1Jm3HZlxleg+Dw/mwps2Wbk9frAWm+4FII=
|
||||||
github.com/go-git/go-git/v5 v5.19.2 h1:wkfn7vOlUBu8ivAWKBWisTiwJK4jYHzTF8Ndv1LyGqY=
|
github.com/go-git/go-git/v5 v5.19.2 h1:wkfn7vOlUBu8ivAWKBWisTiwJK4jYHzTF8Ndv1LyGqY=
|
||||||
github.com/go-git/go-git/v5 v5.19.2/go.mod h1:QqCBE1EFN5ddFmrliLQ3/ntRCUjZU3EJuwuB/jWEHjk=
|
github.com/go-git/go-git/v5 v5.19.2/go.mod h1:QqCBE1EFN5ddFmrliLQ3/ntRCUjZU3EJuwuB/jWEHjk=
|
||||||
github.com/go-viper/mapstructure/v2 v2.5.0 h1:vM5IJoUAy3d7zRSVtIwQgBj7BiWtMPfmPEgAXnvj1Ro=
|
|
||||||
github.com/go-viper/mapstructure/v2 v2.5.0/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM=
|
|
||||||
github.com/goccy/go-json v0.10.6 h1:p8HrPJzOakx/mn/bQtjgNjdTcN+/S6FcG2CTtQOrHVU=
|
github.com/goccy/go-json v0.10.6 h1:p8HrPJzOakx/mn/bQtjgNjdTcN+/S6FcG2CTtQOrHVU=
|
||||||
github.com/goccy/go-json v0.10.6/go.mod h1:oq7eo15ShAhp70Anwd5lgX2pLfOS3QCiwU/PULtXL6M=
|
github.com/goccy/go-json v0.10.6/go.mod h1:oq7eo15ShAhp70Anwd5lgX2pLfOS3QCiwU/PULtXL6M=
|
||||||
github.com/golang/groupcache v0.0.0-20241129210726-2c02b8208cf8 h1:f+oWsMOmNPc8JmEHVZIycC7hBoQxHH9pNKQORJNozsQ=
|
github.com/golang/groupcache v0.0.0-20241129210726-2c02b8208cf8 h1:f+oWsMOmNPc8JmEHVZIycC7hBoQxHH9pNKQORJNozsQ=
|
||||||
|
|||||||
+4
-16
@@ -7,8 +7,6 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os/exec"
|
|
||||||
"path/filepath"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/brain/rank"
|
"github.com/eSlider/2dph/internal/brain/rank"
|
||||||
)
|
)
|
||||||
@@ -139,22 +137,12 @@ func (HTTP) Audit(context.Context) ([]byte, error) {
|
|||||||
return json.Marshal(map[string]any{"status": "ok", "by_confidence": rows})
|
return json.Marshal(map[string]any{"status": "ok", "by_confidence": rows})
|
||||||
}
|
}
|
||||||
|
|
||||||
func (HTTP) Ingest(ctx context.Context, body []byte) ([]byte, error) {
|
func (HTTP) Ingest(context.Context) ([]byte, error) {
|
||||||
if len(bytes.TrimSpace(body)) == 0 {
|
|
||||||
return json.Marshal(map[string]any{
|
return json.Marshal(map[string]any{
|
||||||
"mode": "add",
|
"mode": "rebuild",
|
||||||
"command": "bin/brain/add.go",
|
"command": "bin/brain/index.go --rebuild",
|
||||||
"rebuild": "bin/brain/index.go --rebuild",
|
"add": "v2",
|
||||||
})
|
})
|
||||||
}
|
|
||||||
cmd := exec.CommandContext(ctx, filepath.Join(repoRoot(), "bin", "kb", "add"), "--json")
|
|
||||||
cmd.Stdin = bytes.NewReader(body)
|
|
||||||
cmd.Dir = repoRoot()
|
|
||||||
out, err := cmd.Output()
|
|
||||||
if err != nil {
|
|
||||||
return nil, fmt.Errorf("add: %w", err)
|
|
||||||
}
|
|
||||||
return out, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func asInt(v any) int64 {
|
func asInt(v any) int64 {
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
const Usage = `usage: bin/brain/search.go "query" [--root facts|info] [--repo REPO] [-n N] [--hop N] [--json] [--no-web]
|
const Usage = `usage: bin/brain/search.go "query" [--root facts|info] [--repo REPO] [-n N] [--json] [--no-web]
|
||||||
bin/brain/search.go serve [port]
|
bin/brain/search.go serve [port]
|
||||||
bin/brain/search.go --list-model`
|
bin/brain/search.go --list-model`
|
||||||
|
|
||||||
@@ -15,7 +15,6 @@ type Options struct {
|
|||||||
Root string
|
Root string
|
||||||
Repo string
|
Repo string
|
||||||
Limit int
|
Limit int
|
||||||
Hop int
|
|
||||||
JSONOut bool
|
JSONOut bool
|
||||||
ListModel bool
|
ListModel bool
|
||||||
NoWeb bool
|
NoWeb bool
|
||||||
@@ -23,6 +22,8 @@ type Options struct {
|
|||||||
|
|
||||||
// ParseArgs reads flags. Unknown flags are an error: silently dropping them
|
// ParseArgs reads flags. Unknown flags are an error: silently dropping them
|
||||||
// meant `--hop 1` vanished and its argument `1` was appended to the query.
|
// meant `--hop 1` vanished and its argument `1` was appended to the query.
|
||||||
|
// --hop is recognised so it cannot be swallowed; it is not implemented until
|
||||||
|
// File/FROM_FILE edges exist.
|
||||||
func ParseArgs(args []string) (Options, error) {
|
func ParseArgs(args []string) (Options, error) {
|
||||||
opt := Options{Limit: 20}
|
opt := Options{Limit: 20}
|
||||||
var queryArgs []string
|
var queryArgs []string
|
||||||
@@ -51,15 +52,7 @@ func ParseArgs(args []string) (Options, error) {
|
|||||||
}
|
}
|
||||||
opt.Limit = n
|
opt.Limit = n
|
||||||
case "--hop":
|
case "--hop":
|
||||||
i++
|
return opt, fmt.Errorf("--hop is not implemented yet (needs File/FROM_FILE edges)")
|
||||||
n, err := strconv.Atoi(args[i])
|
|
||||||
if err != nil || n < 1 {
|
|
||||||
return opt, fmt.Errorf("--hop must be a positive integer, got %q", args[i])
|
|
||||||
}
|
|
||||||
if n > 3 {
|
|
||||||
return opt, fmt.Errorf("--hop max is 3 (File → Commit → Person)")
|
|
||||||
}
|
|
||||||
opt.Hop = n
|
|
||||||
case "--json":
|
case "--json":
|
||||||
opt.JSONOut = true
|
opt.JSONOut = true
|
||||||
case "--no-web":
|
case "--no-web":
|
||||||
|
|||||||
@@ -7,30 +7,3 @@ const FTSStmt = "CALL QUERY_FTS_INDEX('Leaf', 'id', $q) " +
|
|||||||
|
|
||||||
const VecStmt = "CALL QUERY_VECTOR_INDEX('Leaf', 'Leaf_vec', $q, $n) " +
|
const VecStmt = "CALL QUERY_VECTOR_INDEX('Leaf', 'Leaf_vec', $q, $n) " +
|
||||||
"RETURN node.id, node.text, node.root, node.source, distance ORDER BY distance LIMIT $n"
|
"RETURN node.id, node.text, node.root, node.source, distance ORDER BY distance LIMIT $n"
|
||||||
|
|
||||||
// HopStmt is the Cypher walk from a search hit. Depth 1 = File, 2 = Commit, 3 = Person.
|
|
||||||
func HopStmt(depth int) string {
|
|
||||||
switch depth {
|
|
||||||
case 1:
|
|
||||||
return "MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File) RETURN f.id, f.path, 1"
|
|
||||||
case 2:
|
|
||||||
return "MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File)-[:HAS_VERSION]->(c:Commit) RETURN c.id, c.subject, 2"
|
|
||||||
case 3:
|
|
||||||
return "MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File)-[:HAS_VERSION]->(c:Commit)-[:AUTHORED]->(p:Person) RETURN p.id, p.name, 3"
|
|
||||||
default:
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func HopLabel(depth int) string {
|
|
||||||
switch depth {
|
|
||||||
case 1:
|
|
||||||
return "File"
|
|
||||||
case 2:
|
|
||||||
return "Commit"
|
|
||||||
case 3:
|
|
||||||
return "Person"
|
|
||||||
default:
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -7,13 +7,6 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
type HopNode struct {
|
|
||||||
ID string `json:"id"`
|
|
||||||
Label string `json:"label"`
|
|
||||||
Name string `json:"name"`
|
|
||||||
Depth int `json:"depth"`
|
|
||||||
}
|
|
||||||
|
|
||||||
// Hit is one search result, mirroring the python script's dict shape.
|
// Hit is one search result, mirroring the python script's dict shape.
|
||||||
type Hit struct {
|
type Hit struct {
|
||||||
ID string `json:"id"`
|
ID string `json:"id"`
|
||||||
@@ -22,7 +15,6 @@ type Hit struct {
|
|||||||
Source string `json:"-"`
|
Source string `json:"-"`
|
||||||
Score float64 `json:"score"`
|
Score float64 `json:"score"`
|
||||||
Snippet string `json:"snippet,omitempty"`
|
Snippet string `json:"snippet,omitempty"`
|
||||||
Hops []HopNode `json:"hops,omitempty"`
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// rrfK dampens the contribution of low ranks; same constant as kblib.py.
|
// rrfK dampens the contribution of low ranks; same constant as kblib.py.
|
||||||
|
|||||||
@@ -91,41 +91,15 @@ func TestHybridKeepsVectorScoreForSharedHit(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// The old parser dropped unknown flags and appended their arguments to the
|
// The old parser dropped unknown flags and appended their arguments to the
|
||||||
// query, so `search "q" --hop 1` searched for "q 1". --hop must stay a flag.
|
// query, so `search "q" --hop 1` searched for "q 1". --hop is not implemented
|
||||||
|
// here (needs File edges); it must still fail closed instead of changing q.
|
||||||
func TestParseHopIsNotSwallowedIntoTheQuery(t *testing.T) {
|
func TestParseHopIsNotSwallowedIntoTheQuery(t *testing.T) {
|
||||||
opt, err := ParseArgs([]string{"what runs on arc-2", "--hop", "1"})
|
_, err := ParseArgs([]string{"what runs on arc-2", "--hop", "1"})
|
||||||
if err != nil {
|
if err == nil {
|
||||||
t.Fatalf("unexpected error: %v", err)
|
t.Fatal("expected --hop to error (not implemented), not be swallowed")
|
||||||
}
|
}
|
||||||
if opt.Query != "what runs on arc-2" {
|
if !strings.Contains(err.Error(), "--hop") {
|
||||||
t.Fatalf("query swallowed hop arg: %q", opt.Query)
|
t.Fatalf("error should name --hop, got %v", err)
|
||||||
}
|
|
||||||
if opt.Hop != 1 {
|
|
||||||
t.Fatalf("hop = %d, want 1", opt.Hop)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestParseHopMaxIsThree(t *testing.T) {
|
|
||||||
if _, err := ParseArgs([]string{"q", "--hop", "4"}); err == nil {
|
|
||||||
t.Fatal("expected --hop 4 to error")
|
|
||||||
}
|
|
||||||
opt, err := ParseArgs([]string{"q", "--hop", "3"})
|
|
||||||
if err != nil || opt.Hop != 3 {
|
|
||||||
t.Fatalf("hop 3: %+v err=%v", opt, err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestHopStmtWalksFromFile(t *testing.T) {
|
|
||||||
s := HopStmt(1)
|
|
||||||
if !strings.Contains(s, "FROM_FILE") || !strings.Contains(s, "File") {
|
|
||||||
t.Fatalf("hop 1 must walk FROM_FILE, got %q", s)
|
|
||||||
}
|
|
||||||
s3 := HopStmt(3)
|
|
||||||
if !strings.Contains(s3, "HAS_VERSION") || !strings.Contains(s3, "AUTHORED") || !strings.Contains(s3, "Person") {
|
|
||||||
t.Fatalf("hop 3 must reach Person, got %q", s3)
|
|
||||||
}
|
|
||||||
if HopLabel(1) != "File" || HopLabel(3) != "Person" {
|
|
||||||
t.Fatal("hop labels")
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -56,12 +56,6 @@ func runSearch(args []string) int {
|
|||||||
fmt.Fprintf(os.Stderr, "search: %v\n", err)
|
fmt.Fprintf(os.Stderr, "search: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
if opt.Hop > 0 {
|
|
||||||
if err := attachHops(hits, opt.Hop); err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "hop: %v\n", err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
results := hits
|
results := hits
|
||||||
for i := range results {
|
for i := range results {
|
||||||
@@ -114,44 +108,6 @@ func searchHits(query, root, repo string, limit int) ([]Hit, error) {
|
|||||||
return rank.RankAndFilter(fts, vec, root, repo, limit), nil
|
return rank.RankAndFilter(fts, vec, root, repo, limit), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func attachHops(hits []Hit, n int) error {
|
|
||||||
if conn == nil {
|
|
||||||
return fmt.Errorf("brain not open")
|
|
||||||
}
|
|
||||||
for i := range hits {
|
|
||||||
var hops []rank.HopNode
|
|
||||||
for d := 1; d <= n; d++ {
|
|
||||||
stmt, err := conn.Prepare(rank.HopStmt(d))
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
res, err := conn.Execute(stmt, map[string]any{"id": hits[i].ID})
|
|
||||||
stmt.Close()
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
for res.HasNext() {
|
|
||||||
row, err := res.Next()
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
vals, err := row.GetAsSlice()
|
|
||||||
if err != nil || len(vals) < 3 {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
hops = append(hops, rank.HopNode{
|
|
||||||
ID: fmt.Sprint(vals[0]),
|
|
||||||
Label: rank.HopLabel(d),
|
|
||||||
Name: fmt.Sprint(vals[1]),
|
|
||||||
Depth: int(asInt(vals[2])),
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
hits[i].Hops = hops
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func b2i(err error) int {
|
func b2i(err error) int {
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 1
|
return 1
|
||||||
@@ -232,7 +188,6 @@ type jsonHit struct {
|
|||||||
Root string `json:"root"`
|
Root string `json:"root"`
|
||||||
Score float64 `json:"score"`
|
Score float64 `json:"score"`
|
||||||
Snippet string `json:"snippet,omitempty"`
|
Snippet string `json:"snippet,omitempty"`
|
||||||
Hops []rank.HopNode `json:"hops,omitempty"`
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func toJSONOut(hits []Hit, query, rootFilter string, web *rank.SecondSource) *jsonOut {
|
func toJSONOut(hits []Hit, query, rootFilter string, web *rank.SecondSource) *jsonOut {
|
||||||
@@ -244,7 +199,6 @@ func toJSONOut(hits []Hit, query, rootFilter string, web *rank.SecondSource) *js
|
|||||||
Root: h.Root,
|
Root: h.Root,
|
||||||
Score: h.Score,
|
Score: h.Score,
|
||||||
Snippet: h.Snippet,
|
Snippet: h.Snippet,
|
||||||
Hops: h.Hops,
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return &jsonOut{
|
return &jsonOut{
|
||||||
@@ -268,18 +222,6 @@ func resultsToDicts(hits []Hit) []any {
|
|||||||
if h.Snippet != "" {
|
if h.Snippet != "" {
|
||||||
d = append(d, KV{"snippet", h.Snippet})
|
d = append(d, KV{"snippet", h.Snippet})
|
||||||
}
|
}
|
||||||
if len(h.Hops) > 0 {
|
|
||||||
nodes := make([]any, len(h.Hops))
|
|
||||||
for j, n := range h.Hops {
|
|
||||||
nodes[j] = Dict{
|
|
||||||
{"id", n.ID},
|
|
||||||
{"label", n.Label},
|
|
||||||
{"name", n.Name},
|
|
||||||
{"depth", n.Depth},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
d = append(d, KV{"hops", nodes})
|
|
||||||
}
|
|
||||||
out[i] = d
|
out[i] = d
|
||||||
}
|
}
|
||||||
return out
|
return out
|
||||||
|
|||||||
@@ -1,47 +0,0 @@
|
|||||||
// Package duckstats runs in-process DuckDB for columnar aggregates.
|
|
||||||
// Graph facts stay in Ladybug. Web-search KV cache stays modernc sqlite.
|
|
||||||
package duckstats
|
|
||||||
|
|
||||||
import (
|
|
||||||
"database/sql"
|
|
||||||
"fmt"
|
|
||||||
|
|
||||||
_ "github.com/duckdb/duckdb-go/v2"
|
|
||||||
)
|
|
||||||
|
|
||||||
type Stats struct {
|
|
||||||
N int `json:"n"`
|
|
||||||
Min float64 `json:"min"`
|
|
||||||
P50 float64 `json:"p50"`
|
|
||||||
P95 float64 `json:"p95"`
|
|
||||||
Max float64 `json:"max"`
|
|
||||||
Avg float64 `json:"avg"`
|
|
||||||
}
|
|
||||||
|
|
||||||
func Quantiles(samples []float64) (Stats, error) {
|
|
||||||
if len(samples) == 0 {
|
|
||||||
return Stats{}, fmt.Errorf("duckstats: empty samples")
|
|
||||||
}
|
|
||||||
db, err := sql.Open("duckdb", "")
|
|
||||||
if err != nil {
|
|
||||||
return Stats{}, err
|
|
||||||
}
|
|
||||||
defer db.Close()
|
|
||||||
var s Stats
|
|
||||||
err = db.QueryRow(`
|
|
||||||
SELECT count(v), min(v), quantile_cont(v, 0.5), quantile_cont(v, 0.95), max(v), avg(v)
|
|
||||||
FROM (SELECT unnest(?) AS v)`, samples).Scan(
|
|
||||||
&s.N, &s.Min, &s.P50, &s.P95, &s.Max, &s.Avg)
|
|
||||||
return s, err
|
|
||||||
}
|
|
||||||
|
|
||||||
func CountJSONL(path string) (int64, error) {
|
|
||||||
db, err := sql.Open("duckdb", "")
|
|
||||||
if err != nil {
|
|
||||||
return 0, err
|
|
||||||
}
|
|
||||||
defer db.Close()
|
|
||||||
var n int64
|
|
||||||
err = db.QueryRow(`SELECT count(*) FROM read_json_auto(?)`, path).Scan(&n)
|
|
||||||
return n, err
|
|
||||||
}
|
|
||||||
@@ -1,51 +0,0 @@
|
|||||||
package duckstats
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
"testing"
|
|
||||||
)
|
|
||||||
|
|
||||||
func TestQuantilesEmpty(t *testing.T) {
|
|
||||||
_, err := Quantiles(nil)
|
|
||||||
if err == nil {
|
|
||||||
t.Fatal("empty slice must error")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestQuantilesOdd(t *testing.T) {
|
|
||||||
s, err := Quantiles([]float64{1, 2, 3, 4, 5})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if s.N != 5 {
|
|
||||||
t.Fatalf("n=%d", s.N)
|
|
||||||
}
|
|
||||||
if s.Min != 1 || s.Max != 5 {
|
|
||||||
t.Fatalf("min=%v max=%v", s.Min, s.Max)
|
|
||||||
}
|
|
||||||
if s.P50 != 3 {
|
|
||||||
t.Fatalf("p50=%v want 3", s.P50)
|
|
||||||
}
|
|
||||||
if s.Avg != 3 {
|
|
||||||
t.Fatalf("avg=%v want 3", s.Avg)
|
|
||||||
}
|
|
||||||
if s.P95 < 4.5 || s.P95 > 5 {
|
|
||||||
t.Fatalf("p95=%v want in [4.5,5]", s.P95)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCountJSONL(t *testing.T) {
|
|
||||||
dir := t.TempDir()
|
|
||||||
p := dir + "/rows.jsonl"
|
|
||||||
body := "{\"ms\":1}\n{\"ms\":2}\n{\"ms\":3}\n"
|
|
||||||
if err := os.WriteFile(p, []byte(body), 0o600); err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
n, err := CountJSONL(p)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
if n != 3 {
|
|
||||||
t.Fatalf("count=%d want 3", n)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -144,15 +144,7 @@ func (s *Server) mcpCall(r *http.Request, params json.RawMessage) (any, error) {
|
|||||||
return nil, fmt.Errorf("cancelled")
|
return nil, fmt.Errorf("cancelled")
|
||||||
}
|
}
|
||||||
defer s.release()
|
defer s.release()
|
||||||
var payload []byte
|
body, err = s.api.Ingest(r.Context())
|
||||||
text := strings.TrimSpace(fmt.Sprint(p.Arguments["text"]))
|
|
||||||
if text != "" && text != "<nil>" {
|
|
||||||
payload, err = json.Marshal(p.Arguments)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
body, err = s.api.Ingest(r.Context(), payload)
|
|
||||||
default:
|
default:
|
||||||
return nil, fmt.Errorf("unknown tool %s", p.Name)
|
return nil, fmt.Errorf("unknown tool %s", p.Name)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -11,7 +11,6 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"errors"
|
"errors"
|
||||||
"io"
|
|
||||||
"log"
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
"os"
|
"os"
|
||||||
@@ -28,7 +27,7 @@ type API interface {
|
|||||||
Get(ctx context.Context, id string, body bool) ([]byte, error)
|
Get(ctx context.Context, id string, body bool) ([]byte, error)
|
||||||
Stats(ctx context.Context) ([]byte, error)
|
Stats(ctx context.Context) ([]byte, error)
|
||||||
Audit(ctx context.Context) ([]byte, error)
|
Audit(ctx context.Context) ([]byte, error)
|
||||||
Ingest(ctx context.Context, body []byte) ([]byte, error)
|
Ingest(ctx context.Context) ([]byte, error)
|
||||||
}
|
}
|
||||||
|
|
||||||
type Server struct {
|
type Server struct {
|
||||||
@@ -60,7 +59,7 @@ func (s *Server) ServeHTTP(w http.ResponseWriter, r *http.Request) {
|
|||||||
case PathAudit:
|
case PathAudit:
|
||||||
s.handleJSON(w, r, s.api.Audit)
|
s.handleJSON(w, r, s.api.Audit)
|
||||||
case PathIngest:
|
case PathIngest:
|
||||||
s.handleIngest(w, r)
|
s.handleJSON(w, r, s.api.Ingest)
|
||||||
case PathOpenAPI:
|
case PathOpenAPI:
|
||||||
s.handleOpenAPI(w, r)
|
s.handleOpenAPI(w, r)
|
||||||
case PathMCP:
|
case PathMCP:
|
||||||
@@ -117,24 +116,6 @@ func (s *Server) handleJSON(w http.ResponseWriter, r *http.Request, fn func(cont
|
|||||||
writeAPI(w, body, err)
|
writeAPI(w, body, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
func (s *Server) handleIngest(w http.ResponseWriter, r *http.Request) {
|
|
||||||
var raw []byte
|
|
||||||
if r.Method == http.MethodPost {
|
|
||||||
b, err := io.ReadAll(io.LimitReader(r.Body, 1<<20))
|
|
||||||
if err != nil {
|
|
||||||
writeJSON(w, http.StatusBadRequest, map[string]any{"error": "read body"})
|
|
||||||
return
|
|
||||||
}
|
|
||||||
raw = b
|
|
||||||
}
|
|
||||||
if !s.acquire(w, r) {
|
|
||||||
return
|
|
||||||
}
|
|
||||||
defer s.release()
|
|
||||||
body, err := s.api.Ingest(r.Context(), raw)
|
|
||||||
writeAPI(w, body, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (s *Server) tryAcquire(r *http.Request) bool {
|
func (s *Server) tryAcquire(r *http.Request) bool {
|
||||||
return s.acquire(nopWriter{}, r)
|
return s.acquire(nopWriter{}, r)
|
||||||
}
|
}
|
||||||
@@ -210,30 +191,11 @@ func (ExecSearcher) Get(context.Context, string, bool) ([]byte, error) {
|
|||||||
}
|
}
|
||||||
func (ExecSearcher) Stats(context.Context) ([]byte, error) { return nil, errUnimplemented }
|
func (ExecSearcher) Stats(context.Context) ([]byte, error) { return nil, errUnimplemented }
|
||||||
func (ExecSearcher) Audit(context.Context) ([]byte, error) { return nil, errUnimplemented }
|
func (ExecSearcher) Audit(context.Context) ([]byte, error) { return nil, errUnimplemented }
|
||||||
func (b ExecSearcher) Ingest(ctx context.Context, body []byte) ([]byte, error) {
|
func (ExecSearcher) Ingest(context.Context) ([]byte, error) {
|
||||||
if len(strings.TrimSpace(string(body))) == 0 {
|
|
||||||
return json.Marshal(map[string]any{
|
return json.Marshal(map[string]any{
|
||||||
"mode": "add",
|
"mode": "rebuild",
|
||||||
"command": "bin/brain/add.go",
|
"command": "bin/brain/index.go --rebuild",
|
||||||
"rebuild": "bin/brain/index.go --rebuild",
|
|
||||||
})
|
})
|
||||||
}
|
|
||||||
root := os.Getenv("KB_ROOT")
|
|
||||||
if root == "" {
|
|
||||||
root = "."
|
|
||||||
}
|
|
||||||
cmd := exec.CommandContext(ctx, filepath.Join(root, "bin/kb/add"), "--json")
|
|
||||||
cmd.Stdin = strings.NewReader(string(body))
|
|
||||||
cmd.Dir = root
|
|
||||||
out, err := cmd.Output()
|
|
||||||
if err != nil {
|
|
||||||
var exitErr *exec.ExitError
|
|
||||||
if errors.As(err, &exitErr) {
|
|
||||||
return nil, errors.New("add failed: " + strings.TrimSpace(string(exitErr.Stderr)))
|
|
||||||
}
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return out, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func defaultSearchCmd(root string) string {
|
func defaultSearchCmd(root string) string {
|
||||||
|
|||||||
@@ -1,7 +1,6 @@
|
|||||||
package httpapi
|
package httpapi
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
|
||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"net/http"
|
"net/http"
|
||||||
@@ -65,11 +64,8 @@ func (f *fakeSearcher) Audit(context.Context) ([]byte, error) {
|
|||||||
return []byte(`{"status":"ok"}`), nil
|
return []byte(`{"status":"ok"}`), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (f *fakeSearcher) Ingest(_ context.Context, body []byte) ([]byte, error) {
|
func (f *fakeSearcher) Ingest(context.Context) ([]byte, error) {
|
||||||
if len(bytes.TrimSpace(body)) == 0 {
|
return []byte(`{"mode":"rebuild","command":"bin/brain/index.go --rebuild"}`), nil
|
||||||
return []byte(`{"mode":"add","command":"bin/brain/add.go"}`), nil
|
|
||||||
}
|
|
||||||
return []byte(`{"mode":"add","ids":["fake-leaf"]}`), nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func (f *fakeSearcher) count() int {
|
func (f *fakeSearcher) count() int {
|
||||||
@@ -197,27 +193,6 @@ func TestStatsAuditIngest(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestIngestIsAddNotRebuildHint(t *testing.T) {
|
|
||||||
h := NewServer(&fakeSearcher{}, 1)
|
|
||||||
code, body := get(t, h, "/ingest")
|
|
||||||
if code != http.StatusOK {
|
|
||||||
t.Fatalf("GET /ingest code = %d body=%s", code, body)
|
|
||||||
}
|
|
||||||
if strings.Contains(string(body), `"add":"v2"`) || strings.Contains(string(body), "write is v2") {
|
|
||||||
t.Fatalf("GET /ingest still a v2 hint: %s", body)
|
|
||||||
}
|
|
||||||
if !strings.Contains(string(body), "bin/brain/add.go") {
|
|
||||||
t.Fatalf("GET /ingest should name add.go: %s", body)
|
|
||||||
}
|
|
||||||
code, body = postJSON(t, h, "/ingest", `{"text":"hello","root":"info","source":"t"}`)
|
|
||||||
if code != http.StatusOK {
|
|
||||||
t.Fatalf("POST /ingest code = %d body=%s", code, body)
|
|
||||||
}
|
|
||||||
if !strings.Contains(string(body), "fake-leaf") {
|
|
||||||
t.Fatalf("POST /ingest should add: %s", body)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestHTTPPackageDoesNotExecPython(t *testing.T) {
|
func TestHTTPPackageDoesNotExecPython(t *testing.T) {
|
||||||
raw, err := os.ReadFile("server.go")
|
raw, err := os.ReadFile("server.go")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -47,15 +47,7 @@ var Ops = []Op{
|
|||||||
},
|
},
|
||||||
{Path: PathStats, Method: "get", ID: "stats", Summary: "index health", MCP: true},
|
{Path: PathStats, Method: "get", ID: "stats", Summary: "index health", MCP: true},
|
||||||
{Path: PathAudit, Method: "get", ID: "audit", Summary: "facts confidence histogram", MCP: true},
|
{Path: PathAudit, Method: "get", ID: "audit", Summary: "facts confidence histogram", MCP: true},
|
||||||
{
|
{Path: PathIngest, Method: "get", ID: "ingest", Summary: "rebuild hint (write is v2)", MCP: true},
|
||||||
Path: PathIngest, Method: "post", ID: "ingest", Summary: "add a leaf without rebuild",
|
|
||||||
MCP: true,
|
|
||||||
Params: []Param{
|
|
||||||
{Name: "text", In: "query", Type: "string", Description: "leaf text (omit for CLI hint)"},
|
|
||||||
{Name: "root", In: "query", Type: "string", Description: "facts or info (default info)"},
|
|
||||||
{Name: "source", In: "query", Type: "string", Description: "evidence pointer; facts need two sources"},
|
|
||||||
},
|
|
||||||
},
|
|
||||||
{Path: PathOpenAPI, Method: "get", ID: "openapi", Summary: "OpenAPI 3 document for this server"},
|
{Path: PathOpenAPI, Method: "get", ID: "openapi", Summary: "OpenAPI 3 document for this server"},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -38,24 +38,11 @@ func TestMCPToolsMatchOpenAPIPaths(t *testing.T) {
|
|||||||
t.Fatalf("MCP tool %s has no OpenAPI path %s", tool.Name, path)
|
t.Fatalf("MCP tool %s has no OpenAPI path %s", tool.Name, path)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
for _, need := range []string{"search", "get", "stats", "audit", "ingest"} {
|
for _, need := range []string{"search", "get", "stats", "audit"} {
|
||||||
if !names[need] {
|
if !names[need] {
|
||||||
t.Fatalf("MCP tools missing %s: %v", need, names)
|
t.Fatalf("MCP tools missing %s: %v", need, names)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
var ingest MCPTool
|
|
||||||
for _, tool := range tools {
|
|
||||||
if tool.Name == "ingest" {
|
|
||||||
ingest = tool
|
|
||||||
break
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if strings.Contains(ingest.Description, "v2") {
|
|
||||||
t.Fatalf("ingest still a v2 hint: %s", ingest.Description)
|
|
||||||
}
|
|
||||||
if !strings.Contains(ingest.Description, "add") {
|
|
||||||
t.Fatalf("ingest should describe add: %s", ingest.Description)
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestOpenAPIHTTP(t *testing.T) {
|
func TestOpenAPIHTTP(t *testing.T) {
|
||||||
|
|||||||
@@ -1,162 +0,0 @@
|
|||||||
// Package ocr runs Tesseract (eng+deu) on images and scanned PDFs.
|
|
||||||
//
|
|
||||||
// Default engine is the tesseract CLI, not gosseract CGO: Ladybug CGO stays
|
|
||||||
// Zig-only (D21). Same engine, no gocv. OCR_ENGINE=paddle selects paddleocr
|
|
||||||
// when that binary is on PATH (compose profile ocr-paddle).
|
|
||||||
package ocr
|
|
||||||
|
|
||||||
import (
|
|
||||||
"fmt"
|
|
||||||
"image"
|
|
||||||
"image/color"
|
|
||||||
"image/png"
|
|
||||||
"os"
|
|
||||||
"os/exec"
|
|
||||||
"path/filepath"
|
|
||||||
"strings"
|
|
||||||
)
|
|
||||||
|
|
||||||
const TessLang = "eng+deu"
|
|
||||||
|
|
||||||
func ImageFile(path string) (string, error) {
|
|
||||||
engine := os.Getenv("OCR_ENGINE")
|
|
||||||
if engine == "paddle" {
|
|
||||||
return runPaddle(path)
|
|
||||||
}
|
|
||||||
return runTesseract(path)
|
|
||||||
}
|
|
||||||
|
|
||||||
func PDFFile(path string) (string, error) {
|
|
||||||
text, err := pdfToText(path)
|
|
||||||
if err == nil && strings.TrimSpace(text) != "" {
|
|
||||||
return strings.TrimSpace(text), nil
|
|
||||||
}
|
|
||||||
ocr, oerr := pdfPages(path)
|
|
||||||
if oerr != nil {
|
|
||||||
if err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
return "", oerr
|
|
||||||
}
|
|
||||||
if strings.TrimSpace(ocr) != "" {
|
|
||||||
return strings.TrimSpace(ocr), nil
|
|
||||||
}
|
|
||||||
if text != "" {
|
|
||||||
return strings.TrimSpace(text), nil
|
|
||||||
}
|
|
||||||
return "", fmt.Errorf("pdf has no text layer (ocr unavailable)")
|
|
||||||
}
|
|
||||||
|
|
||||||
func pdfToText(path string) (string, error) {
|
|
||||||
cmd := exec.Command("pdftotext", "-layout", path, "-")
|
|
||||||
out, err := cmd.Output()
|
|
||||||
if err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
return string(out), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func pdfPages(path string) (string, error) {
|
|
||||||
dir, err := os.MkdirTemp("", "2dph-ocr-")
|
|
||||||
if err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
defer os.RemoveAll(dir)
|
|
||||||
prefix := filepath.Join(dir, "page")
|
|
||||||
cmd := exec.Command("pdftoppm", "-png", "-r", "200", path, prefix)
|
|
||||||
if err := cmd.Run(); err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
matches, err := filepath.Glob(prefix + "*.png")
|
|
||||||
if err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
var parts []string
|
|
||||||
for _, img := range matches {
|
|
||||||
t, err := ImageFile(img)
|
|
||||||
if err != nil {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
if s := strings.TrimSpace(t); s != "" {
|
|
||||||
parts = append(parts, s)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return strings.Join(parts, "\n\n"), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func runTesseract(path string) (string, error) {
|
|
||||||
pre, err := preprocessFile(path)
|
|
||||||
if err != nil {
|
|
||||||
pre = path
|
|
||||||
} else {
|
|
||||||
defer os.Remove(pre)
|
|
||||||
}
|
|
||||||
cmd := exec.Command("tesseract", pre, "stdout", "-l", TessLang, "--psm", "6")
|
|
||||||
out, err := cmd.Output()
|
|
||||||
if err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
return strings.TrimSpace(string(out)), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func runPaddle(path string) (string, error) {
|
|
||||||
cmd := exec.Command("paddleocr", "ocr", "-i", path)
|
|
||||||
out, err := cmd.Output()
|
|
||||||
if err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
return strings.TrimSpace(string(out)), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func preprocessFile(path string) (string, error) {
|
|
||||||
f, err := os.Open(path)
|
|
||||||
if err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
defer f.Close()
|
|
||||||
img, err := png.Decode(f)
|
|
||||||
if err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
out := filepath.Join(os.TempDir(), filepath.Base(path)+".gray.png")
|
|
||||||
w, err := os.Create(out)
|
|
||||||
if err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
defer w.Close()
|
|
||||||
if err := png.Encode(w, GrayContrast(img)); err != nil {
|
|
||||||
os.Remove(out)
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
return out, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// GrayContrast is a stdlib preprocess (no gocv): grayscale + stretch.
|
|
||||||
func GrayContrast(src image.Image) image.Image {
|
|
||||||
b := src.Bounds()
|
|
||||||
dst := image.NewGray(b)
|
|
||||||
var minL, maxL uint8 = 255, 0
|
|
||||||
for y := b.Min.Y; y < b.Max.Y; y++ {
|
|
||||||
for x := b.Min.X; x < b.Max.X; x++ {
|
|
||||||
g := color.GrayModel.Convert(src.At(x, y)).(color.Gray)
|
|
||||||
if g.Y < minL {
|
|
||||||
minL = g.Y
|
|
||||||
}
|
|
||||||
if g.Y > maxL {
|
|
||||||
maxL = g.Y
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
span := int(maxL) - int(minL)
|
|
||||||
if span < 1 {
|
|
||||||
span = 1
|
|
||||||
}
|
|
||||||
for y := b.Min.Y; y < b.Max.Y; y++ {
|
|
||||||
for x := b.Min.X; x < b.Max.X; x++ {
|
|
||||||
g := color.GrayModel.Convert(src.At(x, y)).(color.Gray)
|
|
||||||
v := uint8((int(g.Y) - int(minL)) * 255 / span)
|
|
||||||
dst.SetGray(x, y, color.Gray{Y: v})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return dst
|
|
||||||
}
|
|
||||||
@@ -1,54 +0,0 @@
|
|||||||
package ocr
|
|
||||||
|
|
||||||
import (
|
|
||||||
"image"
|
|
||||||
"image/color"
|
|
||||||
"os/exec"
|
|
||||||
"path/filepath"
|
|
||||||
"strings"
|
|
||||||
"testing"
|
|
||||||
)
|
|
||||||
|
|
||||||
func TestGrayContrastStretches(t *testing.T) {
|
|
||||||
img := image.NewGray(image.Rect(0, 0, 2, 2))
|
|
||||||
img.SetGray(0, 0, color.Gray{Y: 64})
|
|
||||||
img.SetGray(0, 1, color.Gray{Y: 64})
|
|
||||||
img.SetGray(1, 0, color.Gray{Y: 64})
|
|
||||||
img.SetGray(1, 1, color.Gray{Y: 192})
|
|
||||||
out := GrayContrast(img).(*image.Gray)
|
|
||||||
if out.GrayAt(0, 0).Y != 0 {
|
|
||||||
t.Fatalf("min should map to 0, got %d", out.GrayAt(0, 0).Y)
|
|
||||||
}
|
|
||||||
if out.GrayAt(1, 1).Y != 255 {
|
|
||||||
t.Fatalf("max should map to 255, got %d", out.GrayAt(1, 1).Y)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestHelloPNGFixtureOCR(t *testing.T) {
|
|
||||||
if _, err := exec.LookPath("tesseract"); err != nil {
|
|
||||||
t.Skip("tesseract not installed")
|
|
||||||
}
|
|
||||||
path := filepath.Join("testdata", "hello.png")
|
|
||||||
got, err := ImageFile(path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
up := strings.ToUpper(got)
|
|
||||||
if !strings.Contains(up, "HELLO") {
|
|
||||||
t.Fatalf("ocr %q missing HELLO", got)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestPaddleEngineUsesPaddleocrBinary(t *testing.T) {
|
|
||||||
t.Setenv("OCR_ENGINE", "paddle")
|
|
||||||
_, err := ImageFile(filepath.Join("testdata", "hello.png"))
|
|
||||||
if _, look := exec.LookPath("paddleocr"); look != nil {
|
|
||||||
if err == nil {
|
|
||||||
t.Fatal("expected error when paddleocr is missing")
|
|
||||||
}
|
|
||||||
return
|
|
||||||
}
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Vendored
BIN
Binary file not shown.
|
Before Width: | Height: | Size: 1.7 KiB |
@@ -113,8 +113,6 @@ type Report struct {
|
|||||||
XMLLeak int `json:"xml_leak"`
|
XMLLeak int `json:"xml_leak"`
|
||||||
RSSMB int `json:"rss_mb"`
|
RSSMB int `json:"rss_mb"`
|
||||||
VRAMMB int `json:"vram_mb"`
|
VRAMMB int `json:"vram_mb"`
|
||||||
LatencyP50MS float64 `json:"latency_p50_ms,omitempty"`
|
|
||||||
LatencyP95MS float64 `json:"latency_p95_ms,omitempty"`
|
|
||||||
Prompts []Result `json:"prompts"`
|
Prompts []Result `json:"prompts"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ readme = "README.md"
|
|||||||
requires-python = ">=3.12"
|
requires-python = ">=3.12"
|
||||||
license = { text = "MIT" }
|
license = { text = "MIT" }
|
||||||
dependencies = [
|
dependencies = [
|
||||||
|
"docling>=2.119.0",
|
||||||
"ladybug==0.19.1",
|
"ladybug==0.19.1",
|
||||||
"markitdown[docx,epub,html,image-exif,pdf,pptx,xlsx,zip]>=0.1.7",
|
"markitdown[docx,epub,html,image-exif,pdf,pptx,xlsx,zip]>=0.1.7",
|
||||||
"mistune==3.3.4",
|
"mistune==3.3.4",
|
||||||
|
|||||||
@@ -1,257 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""System performance test: PicoClaw surface (brain MCP) + optional reasoner.
|
|
||||||
|
|
||||||
BRAIN_URL=http://127.0.0.1:8630 ./qa/system_perf.py --json
|
|
||||||
REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b \\
|
|
||||||
./qa/system_perf.py --reasoner --picoclaw --json
|
|
||||||
|
|
||||||
Does not write Ladybug. Search includes web (D17); expect ~10s+ per search.
|
|
||||||
Exit 1 if health/get/audit gates fail. Reasoner is measured, not gated.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
import json
|
|
||||||
import os
|
|
||||||
import statistics
|
|
||||||
import sys
|
|
||||||
import time
|
|
||||||
import urllib.error
|
|
||||||
import urllib.request
|
|
||||||
from concurrent.futures import ThreadPoolExecutor
|
|
||||||
|
|
||||||
DEFAULT_BRAIN = "http://127.0.0.1:8630"
|
|
||||||
DEFAULT_REASONER = "http://127.0.0.1:11435/v1"
|
|
||||||
DEFAULT_MODEL = "qwen3.5:9b"
|
|
||||||
DEFAULT_PICOCLAW = "http://127.0.0.1:18790"
|
|
||||||
|
|
||||||
GATE_HEALTH_MS = 500
|
|
||||||
GATE_GET_P50_MS = 50
|
|
||||||
GATE_AUDIT_P50_MS = 50
|
|
||||||
|
|
||||||
|
|
||||||
def _req(url: str, data: bytes | None = None, timeout: float = 90) -> bytes:
|
|
||||||
headers = {"Content-Type": "application/json"} if data is not None else {}
|
|
||||||
req = urllib.request.Request(url, data=data, headers=headers)
|
|
||||||
with urllib.request.urlopen(req, timeout=timeout) as res:
|
|
||||||
return res.read()
|
|
||||||
|
|
||||||
|
|
||||||
def timed(fn):
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
out = fn()
|
|
||||||
return (time.perf_counter() - t0) * 1000.0, out
|
|
||||||
|
|
||||||
|
|
||||||
def stats(samples: list[float]) -> dict:
|
|
||||||
s = sorted(samples)
|
|
||||||
n = len(s)
|
|
||||||
return {
|
|
||||||
"n": n,
|
|
||||||
"min_ms": round(s[0], 1),
|
|
||||||
"p50_ms": round(s[n // 2], 1),
|
|
||||||
"p95_ms": round(s[min(n - 1, int(n * 0.95))], 1),
|
|
||||||
"max_ms": round(s[-1], 1),
|
|
||||||
"avg_ms": round(statistics.mean(s), 1),
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def mcp(brain: str, method: str, params=None, timeout: float = 90) -> dict:
|
|
||||||
payload: dict = {"jsonrpc": "2.0", "id": 1, "method": method}
|
|
||||||
if params is not None:
|
|
||||||
payload["params"] = params
|
|
||||||
raw = _req(brain.rstrip("/") + "/mcp", json.dumps(payload).encode(), timeout=timeout)
|
|
||||||
return json.loads(raw.decode())
|
|
||||||
|
|
||||||
|
|
||||||
def mcp_call(brain: str, name: str, arguments: dict, timeout: float = 90) -> tuple[bool, str]:
|
|
||||||
d = mcp(brain, "tools/call", {"name": name, "arguments": arguments}, timeout=timeout)
|
|
||||||
res = d.get("result") or {}
|
|
||||||
text = ((res.get("content") or [{}])[0].get("text") or "")
|
|
||||||
return (not res.get("isError")), text
|
|
||||||
|
|
||||||
|
|
||||||
def reasoner_tool_call(base: str, model: str, user: str) -> str:
|
|
||||||
payload = {
|
|
||||||
"model": model,
|
|
||||||
"messages": [
|
|
||||||
{"role": "system", "content": "You are PicoClaw. Always call search before answering."},
|
|
||||||
{"role": "user", "content": user},
|
|
||||||
],
|
|
||||||
"tools": [
|
|
||||||
{
|
|
||||||
"type": "function",
|
|
||||||
"function": {
|
|
||||||
"name": "search",
|
|
||||||
"description": "deduction search",
|
|
||||||
"parameters": {
|
|
||||||
"type": "object",
|
|
||||||
"properties": {"q": {"type": "string"}},
|
|
||||||
"required": ["q"],
|
|
||||||
},
|
|
||||||
},
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"tool_choice": "required",
|
|
||||||
}
|
|
||||||
raw = _req(
|
|
||||||
base.rstrip("/") + "/chat/completions",
|
|
||||||
json.dumps(payload).encode(),
|
|
||||||
timeout=600,
|
|
||||||
)
|
|
||||||
chat = json.loads(raw.decode())
|
|
||||||
tcs = chat["choices"][0]["message"].get("tool_calls") or []
|
|
||||||
if not tcs:
|
|
||||||
return ""
|
|
||||||
return tcs[0]["function"]["name"]
|
|
||||||
|
|
||||||
|
|
||||||
def run(args: argparse.Namespace) -> dict:
|
|
||||||
brain = args.brain.rstrip("/")
|
|
||||||
report: dict = {
|
|
||||||
"brain": brain,
|
|
||||||
"device": "cpu",
|
|
||||||
"ok": True,
|
|
||||||
"gates": {},
|
|
||||||
"mcp": {},
|
|
||||||
}
|
|
||||||
ms, _ = timed(lambda: _req(brain + "/health", timeout=5))
|
|
||||||
report["mcp"]["health"] = {"n": 1, "avg_ms": round(ms, 1)}
|
|
||||||
report["gates"]["health"] = ms <= GATE_HEALTH_MS
|
|
||||||
if ms > GATE_HEALTH_MS:
|
|
||||||
report["ok"] = False
|
|
||||||
|
|
||||||
list_ms = []
|
|
||||||
for _ in range(args.n):
|
|
||||||
ms, d = timed(lambda: mcp(brain, "tools/list", timeout=10))
|
|
||||||
names = [t["name"] for t in ((d.get("result") or {}).get("tools") or [])]
|
|
||||||
if "search" not in names:
|
|
||||||
report["ok"] = False
|
|
||||||
list_ms.append(ms)
|
|
||||||
report["mcp"]["tools_list"] = stats(list_ms)
|
|
||||||
|
|
||||||
audit_ms = []
|
|
||||||
for _ in range(args.n):
|
|
||||||
ms, (ok, _) = timed(lambda: mcp_call(brain, "audit", {}))
|
|
||||||
if not ok:
|
|
||||||
report["ok"] = False
|
|
||||||
audit_ms.append(ms)
|
|
||||||
report["mcp"]["audit"] = stats(audit_ms)
|
|
||||||
report["gates"]["audit_p50"] = report["mcp"]["audit"]["p50_ms"] <= GATE_AUDIT_P50_MS
|
|
||||||
if not report["gates"]["audit_p50"]:
|
|
||||||
report["ok"] = False
|
|
||||||
|
|
||||||
ok, text = mcp_call(brain, "search", {"q": "LadybugDB", "n": 2}, timeout=90)
|
|
||||||
inner = json.loads(text) if ok else {}
|
|
||||||
hits = inner.get("results") or []
|
|
||||||
leaf_id = hits[0]["id"] if hits else ""
|
|
||||||
report["mcp"]["search_seed"] = {
|
|
||||||
"ok": ok,
|
|
||||||
"count": inner.get("count"),
|
|
||||||
"web": (inner.get("web") or {}).get("status"),
|
|
||||||
}
|
|
||||||
|
|
||||||
get_ms = []
|
|
||||||
if leaf_id:
|
|
||||||
for _ in range(args.n):
|
|
||||||
ms, (ok, _) = timed(lambda: mcp_call(brain, "get", {"id": leaf_id, "body": True}))
|
|
||||||
if not ok:
|
|
||||||
report["ok"] = False
|
|
||||||
get_ms.append(ms)
|
|
||||||
report["mcp"]["get"] = stats(get_ms)
|
|
||||||
report["gates"]["get_p50"] = report["mcp"]["get"]["p50_ms"] <= GATE_GET_P50_MS
|
|
||||||
if not report["gates"]["get_p50"]:
|
|
||||||
report["ok"] = False
|
|
||||||
|
|
||||||
def one_get() -> float:
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
mcp_call(brain, "get", {"id": leaf_id, "body": True})
|
|
||||||
return (time.perf_counter() - t0) * 1000.0
|
|
||||||
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
with ThreadPoolExecutor(max_workers=8) as ex:
|
|
||||||
conc = list(ex.map(lambda _: one_get(), range(8)))
|
|
||||||
wall = (time.perf_counter() - t0) * 1000.0
|
|
||||||
report["mcp"]["get_concurrent_8"] = {**stats(conc), "wall_ms": round(wall, 1)}
|
|
||||||
|
|
||||||
search_ms = []
|
|
||||||
for q in ("LadybugDB", "model2vec"):
|
|
||||||
ms, (ok, text) = timed(lambda q=q: mcp_call(brain, "search", {"q": q, "n": 3}, timeout=90))
|
|
||||||
inner = json.loads(text) if ok else {}
|
|
||||||
search_ms.append(ms)
|
|
||||||
report.setdefault("mcp", {}).setdefault("search_samples", []).append(
|
|
||||||
{
|
|
||||||
"q": q,
|
|
||||||
"ms": round(ms, 1),
|
|
||||||
"ok": ok,
|
|
||||||
"count": inner.get("count"),
|
|
||||||
"web": (inner.get("web") or {}).get("status"),
|
|
||||||
}
|
|
||||||
)
|
|
||||||
if search_ms:
|
|
||||||
report["mcp"]["search"] = stats(search_ms)
|
|
||||||
|
|
||||||
if args.reasoner:
|
|
||||||
base = args.reasoner_url
|
|
||||||
model = args.model
|
|
||||||
report["reasoner"] = {"base_url": base, "model": model, "calls": []}
|
|
||||||
for user in (
|
|
||||||
"Use tools. Search the 2dph brain for LadybugDB. Call search.",
|
|
||||||
"Use tools. Search the 2dph brain for model2vec. Call search.",
|
|
||||||
):
|
|
||||||
ms, name = timed(lambda user=user: reasoner_tool_call(base, model, user))
|
|
||||||
report["reasoner"]["calls"].append({"ms": round(ms, 1), "tool": name})
|
|
||||||
tools = [c["tool"] for c in report["reasoner"]["calls"]]
|
|
||||||
report["gates"]["reasoner_tool_call"] = bool(tools) and all(t == "search" for t in tools)
|
|
||||||
if not report["gates"]["reasoner_tool_call"]:
|
|
||||||
report["ok"] = False
|
|
||||||
|
|
||||||
if args.picoclaw:
|
|
||||||
gw = args.picoclaw_url.rstrip("/")
|
|
||||||
ms, raw = timed(lambda: _req(gw + "/health", timeout=5))
|
|
||||||
body = json.loads(raw.decode())
|
|
||||||
report["picoclaw"] = {
|
|
||||||
"url": gw,
|
|
||||||
"health_ms": round(ms, 1),
|
|
||||||
"status": body.get("status"),
|
|
||||||
}
|
|
||||||
report["gates"]["picoclaw_health"] = body.get("status") == "ok" and ms <= GATE_HEALTH_MS
|
|
||||||
if not report["gates"]["picoclaw_health"]:
|
|
||||||
report["ok"] = False
|
|
||||||
return report
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str]) -> int:
|
|
||||||
p = argparse.ArgumentParser(description="2dph system performance (MCP + optional reasoner)")
|
|
||||||
p.add_argument("--brain", default=os.environ.get("BRAIN_URL", DEFAULT_BRAIN))
|
|
||||||
p.add_argument("--n", type=int, default=20)
|
|
||||||
p.add_argument("--json", action="store_true")
|
|
||||||
p.add_argument("--reasoner", action="store_true")
|
|
||||||
p.add_argument("--picoclaw", action="store_true")
|
|
||||||
p.add_argument("--picoclaw-url", default=os.environ.get("PICOCLAW_URL", DEFAULT_PICOCLAW))
|
|
||||||
p.add_argument("--reasoner-url", default=os.environ.get("REASONER_BASE_URL", DEFAULT_REASONER))
|
|
||||||
p.add_argument("--model", default=os.environ.get("REASONER_MODEL", DEFAULT_MODEL))
|
|
||||||
args = p.parse_args(argv)
|
|
||||||
try:
|
|
||||||
report = run(args)
|
|
||||||
except (urllib.error.URLError, TimeoutError, OSError) as e:
|
|
||||||
print(f"system_perf: {e}", file=sys.stderr)
|
|
||||||
return 1
|
|
||||||
if args.json:
|
|
||||||
print(json.dumps(report, indent=2))
|
|
||||||
else:
|
|
||||||
print(f"ok={report['ok']} brain={report['brain']}")
|
|
||||||
for name, block in report.get("mcp", {}).items():
|
|
||||||
if isinstance(block, dict) and "p50_ms" in block:
|
|
||||||
print(f" {name}: p50={block['p50_ms']} p95={block['p95_ms']} n={block['n']}")
|
|
||||||
elif name == "health":
|
|
||||||
print(f" health: {block.get('avg_ms')} ms")
|
|
||||||
for k, v in report.get("gates", {}).items():
|
|
||||||
print(f" gate {k}: {v}")
|
|
||||||
for c in (report.get("reasoner") or {}).get("calls") or []:
|
|
||||||
print(f" reasoner {c['tool']}: {c['ms']} ms")
|
|
||||||
return 0 if report["ok"] else 1
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
raise SystemExit(main(sys.argv[1:]))
|
|
||||||
@@ -26,14 +26,14 @@ second independent source when local roots cannot confirm. An answer is
|
|||||||
bin/brain/search.go "Matrix federation" # pointers + snippets, YAML
|
bin/brain/search.go "Matrix federation" # pointers + snippets, YAML
|
||||||
bin/brain/search.go "onlyoffice postgres" --root facts # restrict to confirmed
|
bin/brain/search.go "onlyoffice postgres" --root facts # restrict to confirmed
|
||||||
bin/brain/search.go "where is cs-lexicon" --json | yq '.[].ref'
|
bin/brain/search.go "where is cs-lexicon" --json | yq '.[].ref'
|
||||||
bin/brain/add.go --text T --root facts --source "a.md x b.md"
|
|
||||||
bin/brain/get.go <id> --body # full chunk only when needed
|
bin/brain/get.go <id> --body # full chunk only when needed
|
||||||
bin/brain/stats.go # index health
|
bin/brain/stats.go # index health
|
||||||
bin/brain/eval.go # recall@5 >= 0.95 gate (Go; Python bin/kb/eval is CI fallback)
|
bin/brain/eval.go # recall@5 >= 0.95 gate (Go; Python bin/kb/eval is CI fallback)
|
||||||
```
|
```
|
||||||
|
|
||||||
`bin/kb/search` is a deprecated wrapper. `--hop N` walks
|
`bin/kb/search` is a deprecated wrapper. `--hop` errors (schema has
|
||||||
`FROM_FILE` / `HAS_VERSION` / `AUTHORED` from each hit (1=File, 3=Person).
|
`FROM_FILE`; search does not walk it yet, [#17](https://git.produktor.io/eSlider/2dph/issues/17));
|
||||||
|
do not treat it as a graph walk.
|
||||||
|
|
||||||
## Rules
|
## Rules
|
||||||
|
|
||||||
|
|||||||
@@ -8,4 +8,4 @@ Serve: `bin/brain/serve.go` (`GET /openapi.json`, `POST /mcp`).
|
|||||||
- `get` — read one leaf by id
|
- `get` — read one leaf by id
|
||||||
- `stats` — index health
|
- `stats` — index health
|
||||||
- `audit` — facts confidence histogram
|
- `audit` — facts confidence histogram
|
||||||
- `ingest` — add a leaf without rebuild
|
- `ingest` — rebuild hint (write is v2)
|
||||||
|
|||||||
@@ -1,38 +0,0 @@
|
|||||||
---
|
|
||||||
name: duckdb
|
|
||||||
description: >-
|
|
||||||
Use https://github.com/duckdb/duckdb-go in-process for columnar analytics
|
|
||||||
(quantiles, GROUP BY, JSON/CSV/Parquet/JSONL scans) when that is faster than
|
|
||||||
nested Go loops. Not Ladybug. Not the web-search sqlite cache. Use when
|
|
||||||
aggregating samples, counting JSONL, or SQL over tabular files.
|
|
||||||
---
|
|
||||||
|
|
||||||
# duckdb-go
|
|
||||||
|
|
||||||
Use https://github.com/duckdb/duckdb-go where it makes sense to get better performance in code.
|
|
||||||
|
|
||||||
In-process DuckDB (`internal/duckstats`, `database/sql` driver `duckdb`).
|
|
||||||
Vectorized SQL over tables, JSONL, CSV, Parquet. CGO with bundled libs
|
|
||||||
(linux/darwin amd64/arm64). Links with **gcc/g++** (libstdc++), not Zig.
|
|
||||||
D21 Zig (`bin/cgo/zcc`) is Ladybug/tokenizers only. After
|
|
||||||
`eval "$(bin/cgo/zig env)"`:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= ./bin/qa/stats.go <<< '[1,2,3,4,5]'
|
|
||||||
CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= go test ./internal/duckstats
|
|
||||||
```
|
|
||||||
|
|
||||||
| Store | Job |
|
|
||||||
|-------|-----|
|
|
||||||
| Ladybug | graph + FTS + HNSW (facts/info) |
|
|
||||||
| modernc sqlite | web-search KV cache + throttle |
|
|
||||||
| duckdb-go | OLAP: quantiles, counts, scans of many rows/files |
|
|
||||||
| mikefarah/yq | small YAML/JSON/XML/CSV/TOML/HCL slice, not bulk |
|
|
||||||
|
|
||||||
```bash
|
|
||||||
./bin/qa/stats.go <<< '[1,2,3,4,5]'
|
|
||||||
./bin/qa/stats.go --jsonl path/to/rows.jsonl
|
|
||||||
```
|
|
||||||
|
|
||||||
Do not open Ladybug through DuckDB. Do not put secrets or client PII into
|
|
||||||
DuckDB files under the repo.
|
|
||||||
@@ -1,15 +1,15 @@
|
|||||||
---
|
---
|
||||||
name: picoclaw
|
name: picoclaw
|
||||||
description: >-
|
description: >-
|
||||||
2dph is the memory/fact gate. Compose runs the official PicoClaw gateway.
|
2dph is the memory/fact gate, not the agent loop. Use when wiring PicoClaw
|
||||||
Use when wiring PicoClaw or any MCP client: call brain search/get/audit
|
or any MCP client: call brain search/get/audit before a factual reply.
|
||||||
before a factual reply. throttled is not a negative finding.
|
throttled is not a negative finding.
|
||||||
---
|
---
|
||||||
|
|
||||||
# PicoClaw — fact-check before assert
|
# PicoClaw — fact-check before assert
|
||||||
|
|
||||||
PicoClaw speaks MCP at `POST /mcp` on `bin/brain/serve.go`. Compose profile
|
PicoClaw (or any agent) speaks MCP at `POST /mcp` on `bin/brain/serve.go`.
|
||||||
`picoclaw` runs the official `sipeed/picoclaw` gateway plus `brain-mcp`
|
2dph does not run the agent loop. Compose: `docker compose --profile picoclaw up brain-mcp`
|
||||||
(see [docs/picoclaw.md](../../docs/picoclaw.md)).
|
(see [docs/picoclaw.md](../../docs/picoclaw.md)).
|
||||||
|
|
||||||
## Tool order (before a factual reply)
|
## Tool order (before a factual reply)
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ description: >-
|
|||||||
# postgres
|
# postgres
|
||||||
|
|
||||||
`bin/postgres/query.go` wraps vendored `bin/db/psql-yq`. Output is YAML
|
`bin/postgres/query.go` wraps vendored `bin/db/psql-yq`. Output is YAML
|
||||||
(cheaper than psql ASCII, easy to slice with mikefarah/yq).
|
(cheaper than psql ASCII, easy to slice with `yq`).
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
bin/postgres/query.go --profile onlyoffice -s document_asset # column list
|
bin/postgres/query.go --profile onlyoffice -s document_asset # column list
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ description: >-
|
|||||||
```bash
|
```bash
|
||||||
bin/web/search.go "LadybugDB vector index"
|
bin/web/search.go "LadybugDB vector index"
|
||||||
bin/web/search.go "model2vec multilingual" --category it
|
bin/web/search.go "model2vec multilingual" --category it
|
||||||
bin/web/search.go "hypervisor" --site example.com --json | yq -r '.results[].url'
|
bin/web/search.go "hypervisor" --site example.com --json | jq -r '.results[].url'
|
||||||
bin/web/search.go "postgres partial index" --lang en --fresh year
|
bin/web/search.go "postgres partial index" --lang en --fresh year
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
@@ -43,7 +43,7 @@ found" unless the client refuses to call it absence.
|
|||||||
|
|
||||||
```bash
|
```bash
|
||||||
for i in $(seq 10); do
|
for i in $(seq 10); do
|
||||||
bin/web/search.go "test $i" -n 1 --refresh --json | yq -r '.status'
|
bin/web/search.go "test $i" -n 1 --refresh --json | jq -r .status
|
||||||
done
|
done
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
@@ -1,33 +0,0 @@
|
|||||||
---
|
|
||||||
name: yq
|
|
||||||
description: >-
|
|
||||||
Use https://github.com/mikefarah/yq to work with YAML, JSON, XML, CSV,
|
|
||||||
TOML, HCL where it's efficient and less code. Use when slicing compose,
|
|
||||||
config, --json tool output, CSV/TOML/HCL/XML, or converting between those
|
|
||||||
formats. Not kislyuk Python yq. Not jq when yq already does the job.
|
|
||||||
---
|
|
||||||
|
|
||||||
# yq (mikefarah)
|
|
||||||
|
|
||||||
Use https://github.com/mikefarah/yq to work with YAML, JSON, XML, CSV, TOML, HCL where it's efficient and less code.
|
|
||||||
|
|
||||||
This is the Go `yq` (`yq --version` contains `mikefarah`). It is not
|
|
||||||
kislyuk/yq (Python, jq-syntax, YAML-only wrapper). `bin/db/psql-yq` already
|
|
||||||
calls this binary.
|
|
||||||
|
|
||||||
Prefer `yq` over `python3 -c`, `jq`, or ad-hoc parsers when one expression
|
|
||||||
reads or converts the file. Keep Python/Go for HTTP, binary protocols, and
|
|
||||||
in-process tests.
|
|
||||||
|
|
||||||
```bash
|
|
||||||
yq '.services.picoclaw.image' compose.yaml
|
|
||||||
yq -P . deploy/picoclaw/config.json # JSON → YAML
|
|
||||||
yq -o=json '.gates' # JSON stdin (qa/system_perf.py --json)
|
|
||||||
yq -p=csv -o=json .
|
|
||||||
yq -p=xml -o=json .
|
|
||||||
yq -p=toml '.package.name' file.toml
|
|
||||||
bin/brain/search.go "LadybugDB" --json | yq '.[].ref'
|
|
||||||
bin/web/search.go "hypervisor" --json | yq -r '.results[].url'
|
|
||||||
```
|
|
||||||
|
|
||||||
Do not print secrets, PII, or `$HOME/.config/brain/` through `yq`.
|
|
||||||
Reference in New Issue
Block a user