Compare commits
21
Commits
v0.4.0
...
feat/duckstats
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
168a0a3b53 | ||
|
|
f99dfea104 | ||
|
|
bae1494258 | ||
|
|
c5be3f19be | ||
|
|
a88dbb490c | ||
|
|
9d1a3f3c70 | ||
|
|
8b5be9b659 | ||
|
|
1edb158f35 | ||
|
|
85caff90b9 | ||
|
|
b317968a4c | ||
|
|
1f9bdb0bf6 | ||
|
|
0a05803f4b | ||
|
|
ad83e2a12f | ||
|
|
d894c6609f | ||
|
|
ff80359684 | ||
|
|
36976d9b53 | ||
|
|
8e6f67cc97 | ||
|
|
aca05626bd | ||
|
|
39ae2abe8d | ||
|
|
ba5cc3a6e2 | ||
|
|
15d59054ff |
@@ -3,6 +3,7 @@
|
||||
var
|
||||
.git
|
||||
.github
|
||||
lib-ladybug
|
||||
__pycache__
|
||||
*.pyc
|
||||
*.lbug
|
||||
|
||||
@@ -37,31 +37,61 @@ jobs:
|
||||
bash -n bin/db/ssh-tunnel
|
||||
bash -n bin/docker-entrypoint
|
||||
bash -n bin/kb/search
|
||||
bash -n bin/cgo/zig
|
||||
sh -n bin/cgo/zcc
|
||||
sh -n bin/cgo/zc++
|
||||
|
||||
- name: Python unit tests (offline, vendored tools)
|
||||
run: |
|
||||
uv run python -m unittest discover -s bin/tools -t .
|
||||
|
||||
- name: Go tests (root module, no ladybug cgo)
|
||||
- name: Go tests (root module; duckdb-go CGO via gcc, no ladybug)
|
||||
run: |
|
||||
go vet ./...
|
||||
go test ./... -count=1
|
||||
CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= go vet ./...
|
||||
CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= go test ./... -count=1
|
||||
|
||||
- name: brain ranking tests (no cgo / no ladybug)
|
||||
run: go test ./internal/brain/rank -count=1
|
||||
|
||||
- name: facts/audit self (lexicon consistency, no network)
|
||||
run: |
|
||||
./bin/facts/audit self 2>/dev/null || echo "audit: not yet implemented; gate skipped"
|
||||
run: ./bin/facts/audit self
|
||||
|
||||
- name: kb/eval recall gate
|
||||
- name: CGO via Zig (compile brain/search + eval)
|
||||
run: |
|
||||
./bin/kb/eval 2>/dev/null || echo "eval: not yet implemented; gate skipped"
|
||||
chmod +x bin/cgo/zig bin/cgo/zcc bin/cgo/zc++
|
||||
bin/cgo/zig go build -tags system_ladybug -o /tmp/brain-search ./bin/brain/search.go
|
||||
bin/cgo/zig go build -tags 'system_ladybug,brain_eval' -o /tmp/brain-eval ./bin/brain/eval.go
|
||||
|
||||
- uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.cache/huggingface
|
||||
key: ${{ runner.os }}-hf-potion-multilingual-128M
|
||||
|
||||
- name: recall@5 SoT (Zig bin/brain/eval.go)
|
||||
run: |
|
||||
uv run python bin/kb/index --rebuild --json
|
||||
KB_ROOT="$PWD" /tmp/brain-eval --json
|
||||
|
||||
ocr:
|
||||
name: OCR (tesseract fixture)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: Install tesseract + poppler
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y --no-install-recommends \
|
||||
tesseract-ocr tesseract-ocr-eng tesseract-ocr-deu poppler-utils
|
||||
- name: Go OCR tests (synthetic HELLO PNG)
|
||||
run: go test ./internal/ocr -count=1
|
||||
|
||||
release:
|
||||
name: Release (semver)
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||
needs: test
|
||||
needs: [test, ocr]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
@@ -11,3 +11,6 @@ __pycache__/
|
||||
.secrets/
|
||||
lib-ladybug/
|
||||
go.work.local
|
||||
models/
|
||||
# Purged from git history. Do not re-add.
|
||||
docs/crm-associations-proof.md
|
||||
|
||||
@@ -3,7 +3,8 @@
|
||||
Evidence-first brain over the ops/eSlider stack. Facts need proof or they are
|
||||
`(not confirmed)`.
|
||||
|
||||
Read first: [PLAN](PLAN.md) → [docs](docs/).
|
||||
Read first: [PLAN](PLAN.md) → [docs](docs/) → [roadmap](docs/roadmap.md)
|
||||
(epic [#16](https://git.produktor.io/eSlider/2dph/issues/16)).
|
||||
|
||||
## Method (detective, no fork)
|
||||
|
||||
@@ -15,6 +16,9 @@ Read first: [PLAN](PLAN.md) → [docs](docs/).
|
||||
- `info` root = descriptive/narrative leafs, searchable, never asserted as fact.
|
||||
- Search is deduction: `facts` → `info` → `web-search` (second independent
|
||||
source). An answer is `confirmed` only if it comes off the facts root.
|
||||
- Fact-check every *claim* (facts → info → live → web), not every edit or
|
||||
syntax tweak. PicoClaw: `search` then `get` then `audit` before a factual
|
||||
reply (`skills/picoclaw/SKILL.md`). `throttled` is not a negative finding.
|
||||
|
||||
## Hard rules
|
||||
|
||||
@@ -36,17 +40,22 @@ PLAN.md decisions + execution + open questions
|
||||
docs/ published docs
|
||||
skills/ in-project agent skills (vendored, no external links)
|
||||
bin/ self-describing tools bin/{subject}/{method}.go (shebang)
|
||||
bin/brain/ search.go serve.go index.go get.go stats.go eval.go watch.go
|
||||
bin/brain/ search.go serve.go index.go add.go get.go stats.go eval.go watch.go
|
||||
bin/chats/ sync.go import.go facts.go apply.go; libs in internal/chats
|
||||
bin/mail/ sync.go import.go (index_mail → brain/index.go)
|
||||
bin/markdown/ import.go (mistune leafs)
|
||||
bin/mail/ sync.go import.go ocr.go (index_mail → brain/index.go)
|
||||
bin/markdown/ import.go (H2 leaf split; Python bin/md/import fallback)
|
||||
bin/postgres/ query.go (read-only YAML)
|
||||
internal/ shared Go (brain/rank is cgo-free; chats parsers too)
|
||||
bin/git/ import.go (go-git history; Python shim execs it)
|
||||
bin/web/ search.go (SearXNG; Python shim execs it)
|
||||
bin/reasoner/ bakeoff.go (D18 CPU OpenAI tool-call bake-off)
|
||||
internal/ shared Go (brain/rank is cgo-free; chats parsers; gitlog; websearch; reasoner; duckstats)
|
||||
bin/qa/ stats.go (DuckDB quantiles / JSONL count; gcc CGO, not Zig)
|
||||
bin/watch/ corpus watcher (used by bin/brain/watch.go)
|
||||
bin/tools/ vendored python libs behind bin/* (kblib, yamlout, websearch)
|
||||
bin/docker-entrypoint container entrypoint (brain index|search|serve|watch)
|
||||
bin/cgo/ zig zcc zc++ (CGO via zig cc, not gcc)
|
||||
bin/docker-entrypoint container entrypoint (api: serve|search|watch; index: python)
|
||||
compose.yaml docker composition (root level, not docker/)
|
||||
Dockerfile multi-stage: python deps + static Go binaries
|
||||
Dockerfile api (Zig CGO, no Python) + index (Python write)
|
||||
var/ kb.lbug, var/mail/*, caches (gitignored)
|
||||
.venv/ ladybug + model2vec + mistune
|
||||
```
|
||||
@@ -57,37 +66,52 @@ var/ kb.lbug, var/mail/*, caches (gitignored)
|
||||
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw message.json + attachments
|
||||
bin/mail/sync.go --source gmail --query 'from:example.com' --out var/mail # Gmail search (default in:inbox)
|
||||
bin/mail/import.go --from-raw var/mail # message.json → message.md (convert only)
|
||||
bin/brain/index.go --rebuild # rebuild brain incl. all mail (fresh DB)
|
||||
bin/brain/index.go --rebuild --with-facts --with-chats
|
||||
```
|
||||
|
||||
- `sync` (Go) downloads messages + attachments; Gmail uses paginated list +
|
||||
`body.attachmentId` (not partId) for attachments.
|
||||
- `import` converts body + attachments to markdown. PDFs use poppler
|
||||
`pdftotext -layout` fast path (~15ms); textless/scanned PDFs fall back to
|
||||
docling (isolated subprocess — its native onnx can segfault the parent).
|
||||
Conversion never touches the brain DB (crash safety).
|
||||
- `index_mail` is a deprecation shim for `bin/brain/index.go --rebuild`. Ladybug
|
||||
corrupts its WAL when brand-new leafs are bulk-inserted while FTS/vector
|
||||
indexes exist; a fresh DB with indexes created last is the only safe path.
|
||||
`pdftotext -layout` fast path (~15ms); textless/scanned PDFs use
|
||||
`pdftoppm` + tesseract `eng+deu` (`bin/mail/ocr.go`). Optional
|
||||
`OCR_ENGINE=paddle`. Conversion never touches the brain DB (crash safety).
|
||||
- `index_mail` is a deprecation shim for `bin/brain/index.go --rebuild`. Bulk
|
||||
rebuild still deletes `var/kb.lbug` and creates FTS/HNSW last. Single-leaf
|
||||
write is `bin/brain/add.go` (safe while indexes exist; do not DROP INDEX).
|
||||
Keep conversion + indexing separate so a conversion crash can't leave the
|
||||
DB mid-transaction.
|
||||
|
||||
## Tools
|
||||
|
||||
```bash
|
||||
bin/facts/audit ["self"|"facts"|"info"|"stale"] # 2-source + staleness gate
|
||||
bin/facts/crm [--dry-run] # proof person↔company/company↔project (ooCRM × corpus SoT)
|
||||
bin/facts/audit.go ["self"|"facts"|"info"|"stale"] # 2-source + staleness gate
|
||||
bin/facts/crm.go [--dry-run] # proof person↔company/company↔project (ooCRM × corpus SoT)
|
||||
bin/kb/search "query" [--repo X] # deprecated wrapper → bin/brain/search.go
|
||||
bin/brain/search.go "query" [--root facts|info] # deduction search → YAML
|
||||
bin/brain/get.go <id> [--body]
|
||||
bin/markdown/import.go [dir] # mistune leaves → YAML
|
||||
bin/brain/search.go "query" --no-web # local graph only
|
||||
eval "$(bin/cgo/zig env)" # Zig cc + liblbug (not gcc)
|
||||
bin/brain/index.go --rebuild [--with-mail] [--with-facts] [--with-chats]
|
||||
bin/brain/add.go --text T --root facts --source "a.md x b.md" # incremental write
|
||||
bin/brain/add.go --json # stdin leaf or {leafs:[...]}
|
||||
bin/brain/get.go <id> [--body] [--json] # Go read; Python bin/kb/get CI fallback
|
||||
bin/brain/stats.go [--json]
|
||||
bin/brain/eval.go [--json] # recall@5; questions in internal/brain/rank
|
||||
bin/brain/serve.go # HTTP :8630; GET /openapi.json POST /mcp
|
||||
bin/markdown/import.go [dir] # H2 leafs → YAML; Python bin/md/import fallback
|
||||
bin/git/import.go [REPO] [--json] [--limit N] # go-git history → commit leafs
|
||||
bin/web/search.go "query" [--json] # SearXNG; throttled ≠ absence
|
||||
bin/reasoner/bakeoff.go [--model ID] [--json] # D18 CPU tool-call bake-off
|
||||
bin/postgres/query.go --profile onlyoffice -c 'SELECT 1'
|
||||
bin/qa/stats.go # D22 DuckDB quantiles / JSONL (gcc CGO)
|
||||
bin/mail/ocr.go <image|pdf> # tesseract eng+deu (scans)
|
||||
bin/md/tables # what the graph holds → YAML
|
||||
bin/brain/deduce "question" # thinking wrapper
|
||||
```
|
||||
|
||||
Never start a shell command with `cd` — use the tool working-directory
|
||||
parameter. Search before reading whole files.
|
||||
parameter. Search before reading whole files. For YAML/JSON/XML/CSV/TOML/HCL
|
||||
prefer mikefarah/yq (`skills/yq/SKILL.md`). For bulk rows and quantiles use
|
||||
duckdb-go (`internal/duckstats`, `skills/duckdb/SKILL.md`), not Ladybug.
|
||||
|
||||
## GitHub safety rules (ABSOLUTE — never violate)
|
||||
|
||||
|
||||
+60
-16
@@ -1,5 +1,13 @@
|
||||
# syntax=docker/dockerfile:1
|
||||
FROM python:3.12-slim AS base
|
||||
#
|
||||
# docker build --target api -t 2dph:api .
|
||||
# docker build --target index -t 2dph:index .
|
||||
#
|
||||
# API: Go + ladybug via Zig CGO (no CPython).
|
||||
# Index: Python write path (profile `index` until brain/add is v2).
|
||||
|
||||
# --- Python sidecar (Ladybug write / rebuild) ---
|
||||
FROM python:3.12-slim AS index
|
||||
|
||||
ENV PYTHONUNBUFFERED=1 \
|
||||
PYTHONDONTWRITEBYTECODE=1 \
|
||||
@@ -8,26 +16,16 @@ ENV PYTHONUNBUFFERED=1 \
|
||||
|
||||
WORKDIR /app
|
||||
RUN id -u 2dph 2>/dev/null || useradd --create-home --uid 1001 2dph
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
poppler-utils tesseract-ocr tesseract-ocr-eng tesseract-ocr-deu \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# deps layer-first: rebuild only on dependency change
|
||||
COPY requirements.lock.txt /tmp/requirements.lock.txt
|
||||
RUN python -m pip install --no-cache-dir -r /tmp/requirements.lock.txt \
|
||||
&& rm /tmp/requirements.lock.txt
|
||||
|
||||
# Go services: static binaries, no interpreter at runtime
|
||||
FROM golang:1.25 AS go-build
|
||||
WORKDIR /src
|
||||
COPY go.mod ./
|
||||
COPY bin/server ./bin/server
|
||||
COPY bin/watch ./bin/watch
|
||||
RUN CGO_ENABLED=0 go build -o /serve ./bin/server \
|
||||
&& CGO_ENABLED=0 go build -o /watch ./bin/watch
|
||||
|
||||
# runtime: python toolchain + Go services
|
||||
FROM base
|
||||
COPY . .
|
||||
COPY --from=go-build /serve /app/bin/serve
|
||||
COPY --from=go-build /watch /app/bin/watch
|
||||
RUN chmod +x /app/bin/docker-entrypoint \
|
||||
&& chown -R 2dph:2dph /app
|
||||
USER 2dph
|
||||
@@ -37,5 +35,51 @@ ENV PATH="/app/bin:${PATH}" \
|
||||
KB_ROOT=/app
|
||||
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \
|
||||
CMD python -c "import model2vec, ladybug, mistune; print('ok')" || exit 1
|
||||
|
||||
ENTRYPOINT ["/app/bin/docker-entrypoint"]
|
||||
|
||||
# --- Go API: CGO with Zig, not gcc ---
|
||||
FROM golang:1.26-bookworm AS api-build
|
||||
WORKDIR /src
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends curl xz-utils ca-certificates \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY bin/cgo ./bin/cgo
|
||||
RUN chmod +x bin/cgo/zig bin/cgo/zcc bin/cgo/zc++ \
|
||||
&& ./bin/cgo/zig env >/dev/null
|
||||
|
||||
COPY go.mod go.sum ./
|
||||
RUN go mod download
|
||||
|
||||
COPY . .
|
||||
ENV CGO_RPATH=/usr/local/lib
|
||||
RUN eval "$(./bin/cgo/zig env)" \
|
||||
&& go build -tags brain_serve,system_ladybug -o /out/brain-serve ./bin/brain/serve.go \
|
||||
&& go build -tags system_ladybug -o /out/brain-search ./bin/brain/search.go \
|
||||
&& CGO_ENABLED=0 go build -tags brain_watch -o /out/brain-watch ./bin/brain/watch.go
|
||||
|
||||
FROM debian:bookworm-slim AS api
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends libssl3 ca-certificates wget \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& useradd --create-home --uid 1001 2dph
|
||||
COPY --from=api-build /out/brain-serve /usr/local/bin/brain-serve
|
||||
COPY --from=api-build /out/brain-search /usr/local/bin/brain-search
|
||||
COPY --from=api-build /out/brain-watch /usr/local/bin/brain-watch
|
||||
COPY --from=api-build /src/lib-ladybug/liblbug.so.0.19.1 /usr/local/lib/liblbug.so.0.19.1
|
||||
COPY bin/docker-entrypoint /usr/local/bin/docker-entrypoint
|
||||
RUN chmod +x /usr/local/bin/docker-entrypoint \
|
||||
&& ln -s liblbug.so.0.19.1 /usr/local/lib/liblbug.so.0 \
|
||||
&& ln -s liblbug.so.0 /usr/local/lib/liblbug.so \
|
||||
&& ldconfig
|
||||
USER 2dph
|
||||
ENV KB_ROOT=/data \
|
||||
KB_PORT=8630 \
|
||||
LD_LIBRARY_PATH=/usr/local/lib \
|
||||
HF_HOME=/data/hf
|
||||
WORKDIR /data
|
||||
EXPOSE 8630
|
||||
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \
|
||||
CMD wget -qO- http://127.0.0.1:8630/health || exit 1
|
||||
ENTRYPOINT ["/usr/local/bin/docker-entrypoint"]
|
||||
CMD ["serve"]
|
||||
|
||||
@@ -4,7 +4,10 @@ A brain that loves facts and deduction. Evidence-first knowledge graph + hybrid
|
||||
RAG over the operational Brain/ops/eSlider stack. Built like Sherlock
|
||||
Holmes: nothing is asserted unless it has proof.
|
||||
|
||||
Status: **in progress** — this file is the plan and the record of decisions.
|
||||
Status: **v1 in** (epic [#16](https://git.produktor.io/eSlider/2dph/issues/16) closed).
|
||||
v2 board: milestone [v2](https://git.produktor.io/eSlider/2dph/milestone/13) — OCR [#6](https://git.produktor.io/eSlider/2dph/issues/6),
|
||||
[#29](https://git.produktor.io/eSlider/2dph/issues/29) OQ1, [#30](https://git.produktor.io/eSlider/2dph/issues/30) OQ3.
|
||||
Gap: [docs/roadmap.md](docs/roadmap.md).
|
||||
|
||||
## What
|
||||
|
||||
@@ -26,12 +29,12 @@ detective method: **a fact needs ≥2 independent sources or it is
|
||||
|---|----------|--------|
|
||||
| D1 | RAG corpus | ops stack (chat, onlyoffice, gitea/NPM, searchxng, observability, ai-bot, mcp-servers, `~/.ssh/config`) + portfolio. Exclude `office.dev` + jobs/applications. |
|
||||
| D2 | skill merging | integrate skills **in this project** `skills/`; skip gitea / brain-dependent skills. |
|
||||
| D3 | web search | Vendored client; SearXNG URL is config. Optional Compose instance (sanitized settings). Do not run a second copy on a host that already has one. Empty/`throttled` ≠ “nothing exists”. |
|
||||
| D3 | web search | Go client `bin/web/search.go` (`internal/websearch`). SearXNG URL is config (`BRAIN_SEARCH_URL`). Optional Compose profile `searxng` (sanitized settings). Do not run a second copy on a host that already has one. Empty/`throttled` ≠ “nothing exists”. |
|
||||
| D4 | embeddings | **model2vec** `minishlab/potion-multilingual-128M` instead of embeddinggemma. |
|
||||
| D5 | parser | **mistune** for MD → leaf extraction (duckdb-md documented as future optional SQL/export layer, not v1). |
|
||||
| D6 | graph engine | **LadybugDB**. Go is the service (`bin/brain/search.go`, `bin/brain/serve.go` in-process, `internal/brain`); Python remains for index/write until the Go write path is safe. |
|
||||
| D6 | graph engine | **LadybugDB**. Go is the service (`bin/brain/search.go`, `bin/brain/serve.go` in-process, `internal/brain`). Read path is Go + Zig CGO (D21). Python `bin/kb/{get,stats,eval}` is the CI fallback when Zig/libs are not fetched. Incremental write is Python `bin/kb/add` (`bin/brain/add.go`). Bulk rebuild stays `compose --profile index` until the Go write path is safe. |
|
||||
| D7 | db access | `db-yaml`/`psql-yq`-style, read-only, YAML out. OnlyOffice Postgres via SSH tunnel (`127.0.0.1:5433`). |
|
||||
| D8 | evidence | detective method: ≥2 independent sources or `(not confirmed)`. Auto-pair docker ps × compose × ssh-config × docs. |
|
||||
| D8 | evidence | detective method: ≥2 independent sources or `(not confirmed)`. 2-source auto-pair docker ps × compose × ssh-config × docs. |
|
||||
| D9 | facts/goal model | Who / What / How / Where / When + evidence + confidence on every edge. |
|
||||
| D10 | versioning | everything is a leaf with `sha256 + observed_at + source_rev`; `File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person`. Stale = `source_rev` < git HEAD. |
|
||||
| D11 | strong/weak | `root` column: `facts` (strong) vs `info` (weak). Answer is `confirmed` only from facts root. |
|
||||
@@ -40,8 +43,12 @@ detective method: **a fact needs ≥2 independent sources or it is
|
||||
| D14 | tooling style | `bin/{subject}/{method}.go` shebang (e.g. `bin/brain/search.go`). Shared code in `internal/`. One root `go.mod` + `go.work`. No `bin/*/main.go`, no nested modules. |
|
||||
| D15 | repo | Gitea [`eSlider/2dph`](https://git.produktor.io/eSlider/2dph) is origin + [issues](https://git.produktor.io/eSlider/2dph/issues). GitHub `eSlider/2dph` is the public clone (PRs + Actions CI). No direct `main` pushes. TDD → PR → CI green → merge. |
|
||||
| D16 | contradictions | ≥2 yes vs ≥2 no → unrelated sources conflict → hypothesis → `(not confirmed)`. Resolution (authority, staleness adjudication) = **v2**, tracked as open question. |
|
||||
| D17 | assertion gate | Fact-check every *claim* (facts → info → live sources → web), not every edit. Missing graph ≠ “does not exist”. |
|
||||
| D18 | reasoner | Pluggable OpenAI-compatible URL. RAM: Qwen3.5-9B. Quality: Bonsai-27B or Qwen3.6-27B. No official Qwen3.6-9B. |
|
||||
| D17 | assertion gate | Fact-check every *claim* (facts → info → live → web), not every edit. `bin/brain/search.go` adds a `web` block when there is no facts hit (`throttled`/`skipped`/`refused` ≠ absence). `--root` and `--no-web` stay local. Missing graph ≠ “does not exist”. |
|
||||
| D18 | reasoner | Pluggable OpenAI-compatible URL (`REASONER_BASE_URL`). RAM: `Qwen/Qwen3.5-9B`. Quality: `prism-ml/Bonsai-27B-gguf` or `Qwen/Qwen3.6-27B`. No official Qwen3.6-9B. CPU bake-off: `bin/reasoner/bakeoff.go` + compose profile `reasoner` (`OLLAMA_NUM_GPU=0`, `:11435`). PicoClaw is compose profile `picoclaw`; tools are `search`/`get`/`audit`. Weights are not copied into the 2dph image. Agent lever/loop: [#15](https://git.produktor.io/eSlider/2dph/issues/15). |
|
||||
| D19 | git history | [go-git](https://github.com/go-git/go-git) via `bin/git/import.go`. No subprocess of the git binary. Conversion prints commit leafs; brain write is `bin/brain/index.go`. |
|
||||
| D20 | agent API | OpenAPI + MCP are generated from the same `internal/httpapi.Ops` table as `bin/brain/serve.go` handlers. `GET /openapi.json`, `POST /mcp` (JSON-RPC tools/list + tools/call). Tool names match OpenAPI paths (`search`/`get`/`stats`/`audit`/`ingest`). |
|
||||
| D21 | CGO | Ladybug/tokenizers CGO is compiled with **Zig** (`bin/cgo/zcc` → `zig cc -target …-linux-gnu`), not gcc. `bin/cgo/zig` pins Zig 0.14.1 + liblbug 0.19.1 + libtokenizers 1.27.0. Compose `target: api` has no CPython; write/rebuild is profile `index`. |
|
||||
| D22 | analytics | **duckdb-go** in-process (`internal/duckstats`, `bin/qa/stats.go`) for quantiles/JSONL. Links with **gcc/g++**, not Zig. Ladybug stays the graph; web-search cache stays modernc sqlite. Slice small structured docs with **mikefarah/yq**, not kislyuk/jq. [#30](https://git.produktor.io/eSlider/2dph/issues/30). |
|
||||
|
||||
## Architecture
|
||||
|
||||
@@ -49,23 +56,30 @@ detective method: **a fact needs ≥2 independent sources or it is
|
||||
2dph/
|
||||
PLAN.md / AGENTS.md
|
||||
docs/ published docs (this conversation → docs/ as md)
|
||||
skills/ in-project skills (web-search, db-yaml, kb-search, agent-cost, diataxis-docs, …)
|
||||
skills/ in-project skills (web-search, postgres, brain, picoclaw, diataxis-docs)
|
||||
bin/
|
||||
facts/extract auto-pair 2 sources → lexicon yaml + graph
|
||||
facts/audit ["self"|"facts"|"info"|"stale"] 2-source + staleness gate
|
||||
kb/index Python write path (called by bin/brain/index.go)
|
||||
facts/extract.go audit.go crm.go # D14 shebang; Python implementation
|
||||
kb/index Python bulk write (called by bin/brain/index.go)
|
||||
kb/add Python incremental write (called by bin/brain/add.go)
|
||||
brain/index.go rebuild FTS + HNSW (incl. --with-mail)
|
||||
brain/get.go stats.go eval.go watch.go
|
||||
brain/add.go incremental leaf write (no rebuild)
|
||||
brain/get.go stats.go eval.go # Go read (cgo); Python bin/kb/* CI fallback
|
||||
brain/watch.go
|
||||
brain/search.go deduction: facts → info → web-search
|
||||
brain/serve.go HTTP API in-process (internal/httpapi + internal/brain)
|
||||
brain/serve.go HTTP API in-process + OpenAPI/MCP (D20); Zig CGO (D21)
|
||||
cgo/zig zcc zc++ CGO toolchain (zig cc, not gcc)
|
||||
mail/import.go JSON → markdown (no brain write)
|
||||
markdown/import.go mistune leaves
|
||||
markdown/import.go H2 leaf split (Go); Python bin/md/import fallback
|
||||
postgres/query.go read-only YAML (wraps bin/db/psql-yq)
|
||||
git/import.go go-git history (no git binary; conversion only)
|
||||
web/search.go SearXNG client (throttled ≠ absence)
|
||||
reasoner/bakeoff.go CPU tool-call bake-off (D18; OpenAI tools)
|
||||
chats/sync.go import.go facts.go apply.go
|
||||
(libs in internal/chats; no chats index)
|
||||
mail/ocr.go tesseract eng+deu (pdftoppm scans)
|
||||
md/import (deprecated; bin/markdown/import.go)
|
||||
brain/extract brain/audit brain/deduce (thinking wrapper)
|
||||
web/search (vendored)
|
||||
web/search (deprecated shim → web/search.go)
|
||||
db/psql-yq (vendored)
|
||||
ssh-tunnel onlyoffice pg tunnel 5433
|
||||
var/kb.lbug single embedded store (gitignored)
|
||||
@@ -76,7 +90,10 @@ detective method: **a fact needs ≥2 independent sources or it is
|
||||
|
||||
Node tables: `Person, Service, Host, Container, Repo, File, Commit, Leaf`.
|
||||
`Leaf(embedding FLOAT[N])` — FTS on `text`, HNSW vector index on `embedding`.
|
||||
Edges: `RUNS / USES / HAS_VERSION / AUTHORED / ABOUT / ASSOCIATED / SIMILAR_0.85`.
|
||||
Edges: `RUNS / USES / FROM_FILE / HAS_VERSION / AUTHORED / ABOUT / ASSOCIATED / SIMILAR_0.85`.
|
||||
`FROM_FILE` / `HAS_VERSION` / `AUTHORED`: `bin/brain/search.go --hop N` walks
|
||||
them from each hit (1=File, 2=Commit, 3=Person). Rebuild writes
|
||||
`Leaf-[:FROM_FILE]->File`; git import writes the rest.
|
||||
|
||||
Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
||||
`where`, `when`, `source_rev`.
|
||||
@@ -94,17 +111,20 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
||||
|
||||
- `bin/{subject}/{method}` — line 2 is a usage comment (mirrors `psql-yq`).
|
||||
- bash + python primary; golang via Go shebang when a compiled helper is right.
|
||||
- YAML default output, `--json` for machines. Slice with `yq`.
|
||||
- YAML default output, `--json` for machines. Slice with mikefarah/yq.
|
||||
- Everything that touches the network / DB is read-only, throttled, cached.
|
||||
- Tests (TDD) gate every commit; `gh` + CI/CD on every push.
|
||||
|
||||
## Open questions (v2)
|
||||
|
||||
- OQ1: mutually-contradicting evidence — how to resolve (authority weighting,
|
||||
temporal freshness, audit adjudication).
|
||||
- OQ2: OCR pipeline for pdfs/images/docs — mostly solved: poppler pdftotext
|
||||
fast-path for born-digital PDFs, docling fallback for the ~5% textless ones.
|
||||
- OQ3: optional duckdb-md layer for `SELECT … FORMAT MARKDOWN` export/write-back.
|
||||
temporal freshness, audit adjudication). **v2**; [#29](https://git.produktor.io/eSlider/2dph/issues/29).
|
||||
- OQ2: OCR — **in**. `pdftotext -layout` first; scans `pdftoppm` + tesseract
|
||||
`eng+deu` (`bin/mail/ocr.go`, `internal/ocr`). No gocv, no gosseract CGO
|
||||
(D21 Zig owns Ladybug CGO). Optional `OCR_ENGINE=paddle` / compose profile
|
||||
`ocr-paddle`. Docling left the default path. [#6](https://git.produktor.io/eSlider/2dph/issues/6).
|
||||
- OQ3: **in** — duckdb-go (`internal/duckstats`, `bin/qa/stats.go`) for
|
||||
quantiles / JSONL count. Not a second graph. [#30](https://git.produktor.io/eSlider/2dph/issues/30).
|
||||
- OQ4: YAML-first storage for leafs — deferred: JSON is ~10x faster to
|
||||
serialize and unambiguous; YAML only where humans edit files.
|
||||
|
||||
@@ -113,7 +133,8 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
||||
1. `bin/mail/sync.go` (Go, 8 workers) — paginated Gmail/OnlyOffice download.
|
||||
Gmail attachments key off `body.attachmentId`, not MIME `partId`.
|
||||
2. `bin/mail/import.go --from-raw` — message.json → message.md; PDFs via
|
||||
`pdftotext -layout` (~15ms) with docling subprocess fallback; ICS sidecars
|
||||
`pdftotext -layout` (~15ms); textless/scanned PDFs `pdftoppm` + tesseract
|
||||
`eng+deu`. ICS sidecars
|
||||
Latin-1→UTF-8 normalized.
|
||||
3. `bin/brain/index.go --rebuild` — fresh rebuild (repo corpus + mail) because ladybug
|
||||
corrupts its WAL on bulk-insert into an already-indexed DB. Conversion and
|
||||
@@ -129,9 +150,10 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
||||
1. go vet + go test ./... (root module; packages without ladybug cgo)
|
||||
2. `go test ./internal/brain/rank` (cgo-free ranking + flag parser)
|
||||
3. python -m unittest discover -s bin/tools (includes published-docs SoT)
|
||||
4. bin/facts/audit self (lexicon internal consistency)
|
||||
5. bin/brain/eval.go (recall@5 ≥ 0.95, gates index regressions)
|
||||
6. md-docs build/lint if docs tooling arrives.
|
||||
4. `bin/facts/audit self` (lexicon internal consistency; `bin/facts/audit.go` is the D14 wrapper)
|
||||
5. `bin/brain/eval.go` via Zig (recall@5 ≥ 0.95). Python `bin/kb/eval` is an
|
||||
explicit fallback, not the CI SoT.
|
||||
6. `bin/cgo/zig go build -tags system_ladybug` (compile search with zig cc; fetches pinned zig+libs).
|
||||
|
||||
Feedback loop: every commit → PR → CI → green/gate → merge. Same discipline as
|
||||
`db/tech-poc`: contract first where there is an OpenAPI/message shape.
|
||||
@@ -140,9 +162,26 @@ Feedback loop: every commit → PR → CI → green/gate → merge. Same discipl
|
||||
|
||||
1. scaffold repo (:done after this file + AGENTS.md + .gitignore + ci)
|
||||
2. gh repo create eSlider/2dph --private + initial commit + CI
|
||||
3. vendored skill integration (web-search, db-yaml, kb-search, agent-cost, diataxis-docs) — no remote links
|
||||
3. vendored skill integration (web-search, postgres, brain, diataxis-docs) — no remote links
|
||||
4. .venv: ladybug + model2vec + mistune
|
||||
5. schema + tools with TDD (kb + md + facts + brain)
|
||||
6. ~/.config/brain config
|
||||
7. corpus extraction (facts/info)
|
||||
7. corpus extraction (facts/info) — **in**: [#18](https://git.produktor.io/eSlider/2dph/issues/18)
|
||||
8. verify: web-search smoke, onlyoffice pg, md-db round-trip, eval, audit
|
||||
|
||||
## Gap to v1 (epic #16)
|
||||
|
||||
Remaining: none for epic #16 (v1). Board:
|
||||
[epic #16](https://git.produktor.io/eSlider/2dph/issues/16),
|
||||
milestone [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12).
|
||||
Narrative: [docs/roadmap.md](docs/roadmap.md).
|
||||
|
||||
| Order | Issue | Gap |
|
||||
|-------|-------|-----|
|
||||
| 1 | [#14](https://git.produktor.io/eSlider/2dph/issues/14) | **in** — `bin/brain/add.go` / `POST /ingest` write facts+info without deleting `kb.lbug`. Bulk corpus still `--rebuild`. Leftover Python (mail/facts) is not the living-graph blocker. |
|
||||
| 2 | [#17](https://git.produktor.io/eSlider/2dph/issues/17) | **in** — `--hop N` walks `FROM_FILE` → `HAS_VERSION` → `AUTHORED` (max 3). |
|
||||
| 3 | [#18](https://git.produktor.io/eSlider/2dph/issues/18) | **in** — `--with-facts` / `--facts-json` land `root=facts`; `--with-chats` indexes `var/chats/md`. WhatsApp sync is out of v1. |
|
||||
| 4 | [#15](https://git.produktor.io/eSlider/2dph/issues/15) | **in** — lever/loop documented (`search` → `get` → `audit`). |
|
||||
| 5 | [#19](https://git.produktor.io/eSlider/2dph/issues/19) | **in** — CI recall SoT is `bin/brain/eval.go` via Zig. Python `bin/kb/eval` stays as an explicit fallback. |
|
||||
|
||||
Does **not** block epic close: OQ1 [#29](https://git.produktor.io/eSlider/2dph/issues/29), OQ3 [#30](https://git.produktor.io/eSlider/2dph/issues/30), OQ4. OCR [#6](https://git.produktor.io/eSlider/2dph/issues/6) is **in**.
|
||||
@@ -7,14 +7,16 @@
|
||||
[](https://github.com/eSlider/2dph/releases)
|
||||
[](https://github.com/eSlider/2dph/stargazers)
|
||||
|
||||
An evidence-first brain over the operational eSlider stack. **Facts need two
|
||||
independent sources, or they are `(not confirmed)`.**
|
||||
An evidence-first brain. **Facts need two independent sources, or they are
|
||||
`(not confirmed)`.** Cursor is not the runtime.
|
||||
|
||||
`2dph` is a single embedded knowledge graph (LadybugDB = Kuzu successor) with
|
||||
native **HNSW vector** + **BM25 full-text** indexes, built from markdown,
|
||||
compose files, ssh config, docker state, and git history. Search is
|
||||
*deduction*: confirmed facts first, supporting info second, `web-search` as
|
||||
the independent second source when the local graph cannot confirm.
|
||||
`2dph` is a single embedded knowledge graph (LadybugDB) with native **HNSW
|
||||
vector** + **BM25 full-text** indexes. Search is *deduction*: confirmed facts
|
||||
first, supporting info second, `web-search` as the independent second source
|
||||
when the local graph cannot confirm.
|
||||
|
||||
Run it: [docs/runbook.md](docs/runbook.md). Design: [docs/design.md](docs/design.md).
|
||||
Docs index: [docs/README.md](docs/README.md).
|
||||
|
||||
## Architecture
|
||||
|
||||
@@ -28,10 +30,10 @@ graph TB
|
||||
end
|
||||
|
||||
subgraph dph["2dph tools"]
|
||||
EX["bin/facts/extract<br/>2-source pairing"]
|
||||
AU["bin/facts/audit<br/>confidence + staleness"]
|
||||
EX["bin/facts/extract.go<br/>2-source pairing"]
|
||||
AU["bin/facts/audit.go<br/>confidence + staleness"]
|
||||
IDX["bin/brain/index.go<br/>chunk + embed"]
|
||||
MD["bin/markdown/import.go<br/>mistune leaves"]
|
||||
MD["bin/markdown/import.go<br/>H2 leaf split"]
|
||||
SR["bin/brain/search.go<br/>deduction"]
|
||||
end
|
||||
|
||||
@@ -74,7 +76,7 @@ graph TB
|
||||
## The method
|
||||
|
||||
Every assertion is `Who / What / How / Where / When + evidence + confidence`,
|
||||
mirroring the detective detective skill: **≥2 independent sources confirm a
|
||||
mirroring the detective method: **≥2 independent sources confirm a
|
||||
fact; conflicting sources or a single source → `hypothesis` → `(not confirmed)`.**
|
||||
|
||||
| root | meaning | used for answers |
|
||||
@@ -85,72 +87,100 @@ fact; conflicting sources or a single source → `hypothesis` → `(not confirme
|
||||
## Deduction search
|
||||
|
||||
```bash
|
||||
bin/brain/search.go "Matrix federation over HTTPS" # facts → info → web-search
|
||||
bin/brain/search.go "Matrix federation over HTTPS" # facts → info → web
|
||||
bin/brain/search.go "onlyoffice postgres" --root facts
|
||||
bin/brain/search.go "where is cs-lexicon" --json | yq '.'
|
||||
bin/brain/search.go "upstream flag" --no-web # local graph only
|
||||
bin/brain/get.go <id> --body # full chunk on demand
|
||||
bin/brain/stats.go # index health
|
||||
bin/brain/eval.go # recall@5 gate
|
||||
```
|
||||
|
||||
`--hop` is not implemented (needs File/FROM_FILE edges); the flag errors instead of walking. `bin/kb/search` is a deprecated wrapper around `bin/brain/search.go`.
|
||||
`--hop N` walks File/Commit/Person from each hit (max 3). `bin/kb/search` is a deprecated wrapper around `bin/brain/search.go`.
|
||||
|
||||
Git history is read with [go-git](https://github.com/go-git/go-git) (no git binary):
|
||||
|
||||
```bash
|
||||
bin/git/import.go --json --limit 100 # commit leafs for this repo
|
||||
bin/git/import.go --root "$PROJECTS_ROOT" --json # one pass per .git under root
|
||||
```
|
||||
|
||||
Conversion only. Graph write (`File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person`) stays with `bin/brain/index.go`.
|
||||
|
||||
Web search (second independent source) goes through SearXNG. Empty results mean **throttled**, not “nothing exists”:
|
||||
|
||||
```bash
|
||||
bin/web/search.go "LadybugDB vector index" --json
|
||||
# Optional local instance (skip if BRAIN_SEARCH_URL already points at one):
|
||||
# SEARXNG_SECRET=$(openssl rand -hex 32) docker compose --profile searxng up -d
|
||||
```
|
||||
|
||||
Mail is a first-class corpus (retrievable through the same search):
|
||||
|
||||
```bash
|
||||
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw sync (Go)
|
||||
bin/mail/import.go --from-raw var/mail # JSON → markdown
|
||||
bin/brain/add.go --text T --root facts --source "a.md x b.md"
|
||||
bin/brain/index.go --rebuild --with-facts --with-chats # facts extract + chats md
|
||||
bin/brain/index.go --rebuild # rebuild brain (incl. mail)
|
||||
bin/brain/search.go "invoice from last week" # same search over mail leafs
|
||||
```
|
||||
|
||||
## Storage
|
||||
|
||||
- **LadybugDB** — single `var/kb.lbug`, Cypher property graph, HNSW + BM25
|
||||
in one engine, embedded (no server), ACID, read-only-safe for concurrent
|
||||
readers. **Never `DROP INDEX` FTS/VECTOR** on Ladybug 0.19: DROP leaves
|
||||
ghost catalog tables (`_0_Leaf_vec_UPPER`) so recreate fails while
|
||||
`SHOW_INDEXES` omits HNSW. Fresh indexes = delete `var/kb.lbug` +
|
||||
`bin/brain/index.go --rebuild`. Use `ensure_indexes()` after upserts.
|
||||
- **model2vec** — `potion-multilingual-128M` static embeddings (256-dim),
|
||||
CPU-fast, deterministic, no Ollama runtime dependency.
|
||||
- facts and info split semantically by `root` column but written inside the
|
||||
same transaction.
|
||||
- **LadybugDB** — single `var/kb.lbug`, Cypher + HNSW + BM25, embedded.
|
||||
Read tools (`get` / `stats` / `eval`) are Go + Zig CGO (`bin/cgo/zcc`).
|
||||
Python fallbacks stay for CI until the runner fetches Zig. Incremental
|
||||
write is `bin/brain/add.go` (Python `kblib.add_leafs`). Bulk rebuild is
|
||||
Compose profile `index` (`bin/brain/index.go --rebuild`).
|
||||
- **model2vec** — `potion-multilingual-128M` (256-dim), CPU, no Ollama
|
||||
runtime dependency.
|
||||
- facts and info split by `root` but written in the same transaction.
|
||||
|
||||
Ladybug 0.19 DROP INDEX warning: [docs/runbook.md](docs/runbook.md).
|
||||
|
||||
## Tooling conventions
|
||||
|
||||
`bin/{subject}/{method}.go` — self-describing: shebang on line 1, usage comment
|
||||
from line 2. Shared code in `internal/`. YAML default output, `--json` for
|
||||
machines. Tests gate every commit. HTTP: `bin/brain/serve.go` calls
|
||||
`internal/brain` in-process (`/health` `/search` `/get` `/stats` `/audit` `/ingest`).
|
||||
`internal/brain` in-process (`/health` `/search` `/get` `/stats` `/audit` `/ingest` `/openapi.json` `/mcp`).
|
||||
|
||||
## Development
|
||||
|
||||
See the portable runbook: [docs/runbook.md](docs/runbook.md).
|
||||
|
||||
```bash
|
||||
uv venv .venv # Python 3.12, uv-managed
|
||||
uv pip install -r requirements.lock.txt # pinned toolchain
|
||||
bin/facts/audit self # lexicon consistency gate
|
||||
go test ./... && python -m unittest discover -s bin/tools -t .
|
||||
uv venv .venv
|
||||
uv pip install -r requirements.lock.txt
|
||||
bin/facts/audit.go self
|
||||
go test ./... && uv run python -m unittest discover -s bin/tools -t .
|
||||
```
|
||||
|
||||
Docker (optional, cached model + var volumes):
|
||||
|
||||
```bash
|
||||
docker compose run --rm brain index # (re)index corpus
|
||||
docker compose run --rm brain search "query" # one-shot query
|
||||
docker compose run --rm brain serve # bin/brain/serve.go
|
||||
docker compose up -d brain # API (Zig CGO serve :8630)
|
||||
docker compose --profile index run --rm index # Python Ladybug rebuild
|
||||
docker compose --profile picoclaw up brain-mcp # MCP on 127.0.0.1:8630
|
||||
docker compose --profile reasoner up -d reasoner # CPU Ollama 127.0.0.1:11435
|
||||
docker compose up brain-watch # auto re-index on change
|
||||
```
|
||||
|
||||
## Related
|
||||
|
||||
eSlider DevOps engineer practice: ops, OnlyOffice, and mail feed the facts
|
||||
root through `bin/facts/extract` (two-source pairing).
|
||||
|
||||
- [go-second-brain](https://github.com/eSlider/go-second-brain) — the earlier
|
||||
Neo4j + Qdrant + Matrix RAG brain
|
||||
- [agent-skills](https://github.com/eSlider/agent-skills) — upstream
|
||||
skills (`web-search`, `db-yaml`, …) that 2dph integrates
|
||||
skills (`web-search`, `postgres`, …) that 2dph integrates
|
||||
- detective method — the two-source method
|
||||
|
||||
Work board (issues): [git.produktor.io/eSlider/2dph/issues](https://git.produktor.io/eSlider/2dph/issues).
|
||||
Work board (issues): [epic #16](https://git.produktor.io/eSlider/2dph/issues/16)
|
||||
on [git.produktor.io/eSlider/2dph/issues](https://git.produktor.io/eSlider/2dph/issues).
|
||||
PRs and CI: GitHub [`eSlider/2dph`](https://github.com/eSlider/2dph).
|
||||
|
||||
See [PLAN.md](PLAN.md) for decisions, execution status, and v2 open questions.
|
||||
See [PLAN.md](PLAN.md) for decisions, [docs/roadmap.md](docs/roadmap.md) for
|
||||
the gap to v1, and v2 open questions.
|
||||
|
||||
Executable
+21
@@ -0,0 +1,21 @@
|
||||
//usr/bin/env go run -tags=brain_add "$0" "$@"; exit
|
||||
//go:build brain_add
|
||||
//
|
||||
// bin/brain/add.go - incremental leaf write (Python kblib, no rebuild).
|
||||
//
|
||||
// ./bin/brain/add.go --text T --root facts --source "a.md x b.md"
|
||||
// ./bin/brain/add.go --json
|
||||
//
|
||||
// D6: write stays Python. Does not delete var/kb.lbug.
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
|
||||
"github.com/eSlider/2dph/internal/cmdbin"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(cmdbin.ExecFile("bin/kb/add", os.Args[1:]))
|
||||
}
|
||||
+6
-4
@@ -1,20 +1,22 @@
|
||||
//usr/bin/env go run -tags=brain_eval "$0" "$@"; exit
|
||||
//go:build brain_eval
|
||||
//usr/bin/env go run -tags=system_ladybug,brain_eval "$0" "$@"; exit
|
||||
//go:build cgo && system_ladybug && brain_eval
|
||||
//
|
||||
// bin/brain/eval.go - recall@5 gate.
|
||||
//
|
||||
// ./bin/brain/eval.go
|
||||
// ./bin/brain/eval.go --json
|
||||
//
|
||||
// Needs CGO + libladybug. Python bin/kb/eval is the CI fallback (no cgo).
|
||||
// Control questions live in internal/brain/rank (cgo-free).
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
|
||||
"github.com/eSlider/2dph/internal/cmdbin"
|
||||
"github.com/eSlider/2dph/internal/brain"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(cmdbin.ExecFile("bin/kb/eval", os.Args[1:]))
|
||||
os.Exit(brain.MainEval(os.Args[1:]))
|
||||
}
|
||||
|
||||
+7
-4
@@ -1,20 +1,23 @@
|
||||
//usr/bin/env go run -tags=brain_get "$0" "$@"; exit
|
||||
//go:build brain_get
|
||||
//usr/bin/env go run -tags=system_ladybug,brain_get "$0" "$@"; exit
|
||||
//go:build cgo && system_ladybug && brain_get
|
||||
//
|
||||
// bin/brain/get.go - read one leaf by id.
|
||||
//
|
||||
// ./bin/brain/get.go <id>
|
||||
// ./bin/brain/get.go <id> --body
|
||||
// ./bin/brain/get.go <id> --json
|
||||
//
|
||||
// Needs CGO + libladybug. Python bin/kb/get is the CI fallback (no cgo).
|
||||
// CGO compiler is Zig (`eval "$(bin/cgo/zig env)"`), not gcc.
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
|
||||
"github.com/eSlider/2dph/internal/cmdbin"
|
||||
"github.com/eSlider/2dph/internal/brain"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(cmdbin.ExecFile("bin/kb/get", os.Args[1:]))
|
||||
os.Exit(brain.MainGet(os.Args[1:]))
|
||||
}
|
||||
|
||||
+3
-3
@@ -3,12 +3,12 @@
|
||||
//
|
||||
// bin/brain/index.go - rebuild the Ladybug graph (Python write path).
|
||||
//
|
||||
// ./bin/brain/index.go --rebuild
|
||||
// ./bin/brain/index.go --rebuild --with-facts --with-chats
|
||||
// ./bin/brain/index.go --rebuild --with-mail
|
||||
// ./bin/brain/index.go --dry-run --with-mail
|
||||
//
|
||||
// v1 write is always a rebuild when mail is included (live FTS/HNSW + bulk
|
||||
// insert corrupts Ladybug 0.19 WAL). `add` is v2.
|
||||
// v1 write: bin/brain/add.go for one/few leafs (indexes may already exist).
|
||||
// Bulk mail/corpus still --rebuild (fresh file, indexes last).
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
|
||||
+2
-2
@@ -3,11 +3,11 @@
|
||||
//
|
||||
// bin/brain/search.go - deduction search over the 2dph brain.
|
||||
//
|
||||
// ./bin/brain/search.go "query" [--root facts|info] [--repo P] [-n N] [--json]
|
||||
// ./bin/brain/search.go "query" [--root facts|info] [--repo P] [-n N] [--hop N] [--json] [--no-web]
|
||||
// ./bin/brain/search.go serve [port]
|
||||
// ./bin/brain/search.go --list-model
|
||||
//
|
||||
// Needs CGO + libladybug (CGO_CFLAGS/CGO_LDFLAGS). Prefer the wrapper
|
||||
// Needs CGO + libladybug via Zig (`eval "$(bin/cgo/zig env)"`), not gcc.
|
||||
// bin/kb/search which sets those and builds a binary for the embed daemon.
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
// KB_ROOT=/path/to/2dph ./bin/brain/serve.go
|
||||
// KB_WORKERS=4 KB_PORT=8630 ./bin/brain/serve.go
|
||||
//
|
||||
// GET /openapi.json same Ops table as the handlers
|
||||
// POST /mcp JSON-RPC tools/list + tools/call
|
||||
//
|
||||
// Needs CGO + libladybug (same as bin/brain/search.go).
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
+5
-4
@@ -1,20 +1,21 @@
|
||||
//usr/bin/env go run -tags=brain_stats "$0" "$@"; exit
|
||||
//go:build brain_stats
|
||||
//usr/bin/env go run -tags=system_ladybug,brain_stats "$0" "$@"; exit
|
||||
//go:build cgo && system_ladybug && brain_stats
|
||||
//
|
||||
// bin/brain/stats.go - index health.
|
||||
//
|
||||
// ./bin/brain/stats.go
|
||||
// ./bin/brain/stats.go --json
|
||||
//
|
||||
// Needs CGO + libladybug. Python bin/kb/stats is the CI fallback (no cgo).
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
|
||||
"github.com/eSlider/2dph/internal/cmdbin"
|
||||
"github.com/eSlider/2dph/internal/brain"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(cmdbin.ExecFile("bin/kb/stats", os.Args[1:]))
|
||||
os.Exit(brain.MainStats(os.Args[1:]))
|
||||
}
|
||||
|
||||
Executable
+23
@@ -0,0 +1,23 @@
|
||||
#!/bin/sh
|
||||
# bin/cgo/zc++ — CGO CXX. Zig, not g++.
|
||||
set -eu
|
||||
ROOT="$(CDPATH= cd -- "$(dirname "$0")/../.." && pwd)"
|
||||
case "$(uname -m)" in
|
||||
x86_64|amd64) TARGET=x86_64-linux-gnu ;;
|
||||
aarch64|arm64) TARGET=aarch64-linux-gnu ;;
|
||||
*)
|
||||
echo "zc++: unsupported arch $(uname -m)" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
if [ -n "${ZIG:-}" ] && [ -x "$ZIG" ]; then
|
||||
:
|
||||
elif [ -x "$ROOT/var/zig/zig" ]; then
|
||||
ZIG="$ROOT/var/zig/zig"
|
||||
elif command -v zig >/dev/null 2>&1; then
|
||||
ZIG="$(command -v zig)"
|
||||
else
|
||||
echo "zc++: zig missing; run bin/cgo/zig first" >&2
|
||||
exit 127
|
||||
fi
|
||||
exec "$ZIG" c++ -target "$TARGET" "$@"
|
||||
Executable
+24
@@ -0,0 +1,24 @@
|
||||
#!/bin/sh
|
||||
# bin/cgo/zcc — CGO CC. Zig, not gcc.
|
||||
# Go invokes CC with many args; a wrapper avoids spaces in $CC.
|
||||
set -eu
|
||||
ROOT="$(CDPATH= cd -- "$(dirname "$0")/../.." && pwd)"
|
||||
case "$(uname -m)" in
|
||||
x86_64|amd64) TARGET=x86_64-linux-gnu ;;
|
||||
aarch64|arm64) TARGET=aarch64-linux-gnu ;;
|
||||
*)
|
||||
echo "zcc: unsupported arch $(uname -m)" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
if [ -n "${ZIG:-}" ] && [ -x "$ZIG" ]; then
|
||||
:
|
||||
elif [ -x "$ROOT/var/zig/zig" ]; then
|
||||
ZIG="$ROOT/var/zig/zig"
|
||||
elif command -v zig >/dev/null 2>&1; then
|
||||
ZIG="$(command -v zig)"
|
||||
else
|
||||
echo "zcc: zig missing; run bin/cgo/zig first" >&2
|
||||
exit 127
|
||||
fi
|
||||
exec "$ZIG" cc -target "$TARGET" "$@"
|
||||
Executable
+130
@@ -0,0 +1,130 @@
|
||||
#!/usr/bin/env bash
|
||||
# bin/cgo/zig — CGO toolchain: zig cc (not gcc) + pinned liblbug + libtokenizers.
|
||||
#
|
||||
# eval "$(bin/cgo/zig env)" # export CC/CXX/CGO_*
|
||||
# bin/cgo/zig go build ... # ensure, then exec with env
|
||||
# bin/cgo/zig ./bin/brain/search.go "query"
|
||||
#
|
||||
# Pins live in this file. Downloads land in var/ (gitignored).
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(CDPATH= cd -- "$(dirname "$0")/../.." && pwd)"
|
||||
ZIG_VERSION=0.14.1
|
||||
LBUG_VERSION=0.19.1
|
||||
TOKENIZERS_VERSION=1.27.0
|
||||
|
||||
arch="$(uname -m)"
|
||||
case "$arch" in
|
||||
x86_64|amd64)
|
||||
ZIG_ARCH=x86_64
|
||||
LBUG_ARCH=x86_64
|
||||
TOK_ARCH=x86_64
|
||||
ZIG_SHA=24aeeec8af16c381934a6cd7d95c807a8cb2cf7df9fa40d359aa884195c4716c
|
||||
LBUG_SHA=ed263ae913f68cb0ddba0b98548b58edaac49929766d03bdaaa83be46c68847d
|
||||
TOK_SHA=72556cdca798dd4ea7cdaba308e5f0d68a8cb93b67c96edf485b7a0edd7b07f4
|
||||
;;
|
||||
aarch64|arm64)
|
||||
ZIG_ARCH=aarch64
|
||||
LBUG_ARCH=aarch64
|
||||
TOK_ARCH=aarch64
|
||||
ZIG_SHA=f7a654acc967864f7a050ddacfaa778c7504a0eca8d2b678839c21eea47c992b
|
||||
LBUG_SHA=b07df2cd533c3976a2a3025866d6420a5f35514d0a822ecc4b2902d55b4725b7
|
||||
TOK_SHA=e96545ad05930c26f51f63d932ee6d3bbd32bbed149e102c5290d587a2293067
|
||||
;;
|
||||
*)
|
||||
echo "bin/cgo/zig: unsupported arch $arch" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
CACHE="$ROOT/var/cache"
|
||||
LIB="$ROOT/lib-ladybug"
|
||||
ZIG_DIR="$ROOT/var/zig-dist"
|
||||
ZIG_BIN="$ROOT/var/zig/zig"
|
||||
|
||||
sha256of() {
|
||||
if command -v sha256sum >/dev/null 2>&1; then
|
||||
sha256sum "$1" | awk '{print $1}'
|
||||
else
|
||||
shasum -a 256 "$1" | awk '{print $1}'
|
||||
fi
|
||||
}
|
||||
|
||||
fetch() {
|
||||
local url="$1" dest="$2" expect="$3"
|
||||
if [ -f "$dest" ] && [ "$(sha256of "$dest")" = "$expect" ]; then
|
||||
return 0
|
||||
fi
|
||||
mkdir -p "$(dirname "$dest")"
|
||||
echo "fetch $url" >&2
|
||||
curl -fsSL "$url" -o "$dest"
|
||||
local got
|
||||
got="$(sha256of "$dest")"
|
||||
if [ "$got" != "$expect" ]; then
|
||||
echo "checksum mismatch $dest: got $got want $expect" >&2
|
||||
rm -f "$dest"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
ensure_zig() {
|
||||
if [ -n "${ZIG:-}" ] && [ -x "$ZIG" ]; then
|
||||
return 0
|
||||
fi
|
||||
if [ -x "$ZIG_BIN" ]; then
|
||||
export ZIG="$ZIG_BIN"
|
||||
return 0
|
||||
fi
|
||||
if command -v zig >/dev/null 2>&1; then
|
||||
export ZIG
|
||||
ZIG="$(command -v zig)"
|
||||
return 0
|
||||
fi
|
||||
local tar="$CACHE/zig-${ZIG_ARCH}-linux-${ZIG_VERSION}.tar.xz"
|
||||
fetch "https://ziglang.org/download/${ZIG_VERSION}/zig-${ZIG_ARCH}-linux-${ZIG_VERSION}.tar.xz" \
|
||||
"$tar" "$ZIG_SHA"
|
||||
mkdir -p "$CACHE"
|
||||
rm -rf "$ZIG_DIR"
|
||||
tar -xJf "$tar" -C "$CACHE"
|
||||
mv "$CACHE/zig-${ZIG_ARCH}-linux-${ZIG_VERSION}" "$ZIG_DIR"
|
||||
mkdir -p "$ROOT/var/zig"
|
||||
ln -sfn "$ZIG_DIR/zig" "$ZIG_BIN"
|
||||
export ZIG="$ZIG_BIN"
|
||||
}
|
||||
|
||||
ensure_libs() {
|
||||
mkdir -p "$LIB"
|
||||
if [ ! -f "$LIB/liblbug.so" ]; then
|
||||
local tar="$CACHE/liblbug-linux-${LBUG_ARCH}.tar.gz"
|
||||
fetch "https://github.com/LadybugDB/ladybug/releases/download/v${LBUG_VERSION}/liblbug-linux-${LBUG_ARCH}.tar.gz" \
|
||||
"$tar" "$LBUG_SHA"
|
||||
tar -xzf "$tar" -C "$LIB"
|
||||
fi
|
||||
if [ ! -f "$LIB/libtokenizers.a" ]; then
|
||||
local tar="$CACHE/libtokenizers.linux-${TOK_ARCH}.tar.gz"
|
||||
fetch "https://github.com/daulet/tokenizers/releases/download/v${TOKENIZERS_VERSION}/libtokenizers.linux-${TOK_ARCH}.tar.gz" \
|
||||
"$tar" "$TOK_SHA"
|
||||
tar -xzf "$tar" -C "$LIB"
|
||||
fi
|
||||
}
|
||||
|
||||
print_env() {
|
||||
printf 'export ZIG=%q\n' "$ZIG"
|
||||
printf 'export CC=%q\n' "$ROOT/bin/cgo/zcc"
|
||||
printf 'export CXX=%q\n' "$ROOT/bin/cgo/zc++"
|
||||
printf 'export CGO_ENABLED=1\n'
|
||||
printf 'export CGO_CFLAGS=%q\n' "-I$LIB"
|
||||
printf 'export CGO_LDFLAGS=%q\n' "-L$LIB -Wl,-rpath,${CGO_RPATH:-$LIB}"
|
||||
}
|
||||
|
||||
ensure_zig
|
||||
ensure_libs
|
||||
|
||||
cmd="${1:-env}"
|
||||
if [ "$cmd" = "env" ]; then
|
||||
print_env
|
||||
exit 0
|
||||
fi
|
||||
|
||||
eval "$(print_env)"
|
||||
exec "$@"
|
||||
+3
-2
@@ -29,10 +29,11 @@ func main() {
|
||||
case "linkedin":
|
||||
os.Exit(chats.RunSyncLinkedIn(args))
|
||||
case "whatsapp":
|
||||
fmt.Fprintln(os.Stderr, "chats: WhatsApp not implemented yet")
|
||||
fmt.Fprintln(os.Stderr, "chats: WhatsApp sync is out of v1")
|
||||
os.Exit(1)
|
||||
case "help", "-h", "--help":
|
||||
fmt.Fprintln(os.Stderr, `usage: bin/chats/sync.go telegram|linkedin [flags]`)
|
||||
fmt.Fprintln(os.Stderr, `usage: bin/chats/sync.go telegram|linkedin [flags]
|
||||
WhatsApp sync is out of v1.`)
|
||||
return
|
||||
default:
|
||||
fmt.Fprintf(os.Stderr, "chats: unknown platform %q\n", platform)
|
||||
|
||||
+26
-15
@@ -1,13 +1,10 @@
|
||||
#!/usr/bin/env bash
|
||||
# bin/docker-entrypoint - run 2dph tools inside the container.
|
||||
#
|
||||
# brain shell (default)
|
||||
# brain search <q> bin/brain/search.go
|
||||
# brain index bin/kb/index --with-mail
|
||||
# brain watch <dir> compiled /app/bin/watch (bin/brain/watch.go)
|
||||
# brain serve compiled /app/bin/serve (bin/brain/serve.go)
|
||||
# brain extract bin/facts/extract (docker×compose pairing)
|
||||
# brain audit bin/facts/audit
|
||||
# API image (Zig CGO binaries):
|
||||
# serve | search | watch
|
||||
# Index image (Python write path, compose profile `index`):
|
||||
# index | extract | audit | search (deprecated python wrapper)
|
||||
#
|
||||
# Usage comment starts at line 2 (self-describing convention).
|
||||
set -euo pipefail
|
||||
@@ -15,13 +12,27 @@ set -euo pipefail
|
||||
CMD="${1:-shell}"
|
||||
shift || true
|
||||
|
||||
if [ -x /usr/local/bin/brain-serve ]; then
|
||||
case "$CMD" in
|
||||
shell) exec bash ;;
|
||||
serve) exec /usr/local/bin/brain-serve "$@" ;;
|
||||
search) exec /usr/local/bin/brain-search "$@" ;;
|
||||
watch) exec /usr/local/bin/brain-watch "$@" ;;
|
||||
index)
|
||||
echo "index is the Python sidecar: docker compose --profile index run --rm index" >&2
|
||||
exit 2
|
||||
;;
|
||||
*) echo "unknown command: $CMD (api: serve|search|watch)" >&2; exit 2 ;;
|
||||
esac
|
||||
fi
|
||||
|
||||
case "$CMD" in
|
||||
shell) exec bash ;;
|
||||
search) exec "$KB_PY" /app/bin/kb/search "$@" ;;
|
||||
index) exec "$KB_PY" /app/bin/kb/index --with-mail "$@" ;;
|
||||
watch) exec /app/bin/watch "$@" ;;
|
||||
serve) exec /app/bin/serve "$@" ;;
|
||||
extract) exec "$KB_PY" /app/bin/facts/extract "$@" ;;
|
||||
audit) exec "$KB_PY" /app/bin/facts/audit "$@" ;;
|
||||
*) echo "unknown command: $CMD" >&2; exit 2 ;;
|
||||
shell) exec bash ;;
|
||||
search) exec "$KB_PY" /app/bin/kb/search "$@" ;;
|
||||
index) exec "$KB_PY" /app/bin/kb/index --with-mail "$@" ;;
|
||||
watch) exec /app/bin/watch "$@" ;;
|
||||
serve) exec /app/bin/serve "$@" ;;
|
||||
extract) exec "$KB_PY" /app/bin/facts/extract "$@" ;;
|
||||
audit) exec "$KB_PY" /app/bin/facts/audit "$@" ;;
|
||||
*) echo "unknown command: $CMD" >&2; exit 2 ;;
|
||||
esac
|
||||
|
||||
Executable
+21
@@ -0,0 +1,21 @@
|
||||
//usr/bin/env go run -tags=facts_audit "$0" "$@"; exit
|
||||
//go:build facts_audit
|
||||
//
|
||||
// bin/facts/audit.go - 2-source + lexicon checks.
|
||||
//
|
||||
// ./bin/facts/audit.go self
|
||||
// ./bin/facts/audit.go db
|
||||
//
|
||||
// Python bin/facts/audit is the implementation (CI runs it directly).
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
|
||||
"github.com/eSlider/2dph/internal/cmdbin"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(cmdbin.ExecFile("bin/facts/audit", os.Args[1:]))
|
||||
}
|
||||
Executable
+20
@@ -0,0 +1,20 @@
|
||||
//usr/bin/env go run -tags=facts_crm "$0" "$@"; exit
|
||||
//go:build facts_crm
|
||||
//
|
||||
// bin/facts/crm.go - prove person↔company / company↔project (ooCRM × corpus).
|
||||
//
|
||||
// ./bin/facts/crm.go [--dry-run] [--mismatches]
|
||||
//
|
||||
// Python bin/facts/crm is the implementation. Graph write stays Python.
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
|
||||
"github.com/eSlider/2dph/internal/cmdbin"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(cmdbin.ExecFile("bin/facts/crm", os.Args[1:]))
|
||||
}
|
||||
Executable
+20
@@ -0,0 +1,20 @@
|
||||
//usr/bin/env go run -tags=facts_extract "$0" "$@"; exit
|
||||
//go:build facts_extract
|
||||
//
|
||||
// bin/facts/extract.go - acquire confirmed facts (2-source each).
|
||||
//
|
||||
// ./bin/facts/extract.go [--json] [--dry-run]
|
||||
//
|
||||
// Python bin/facts/extract is the implementation. Graph write stays Python.
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
|
||||
"github.com/eSlider/2dph/internal/cmdbin"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(cmdbin.ExecFile("bin/facts/extract", os.Args[1:]))
|
||||
}
|
||||
+10
-138
@@ -1,153 +1,25 @@
|
||||
#!/usr/bin/env python3
|
||||
"""git/import - import git history (commits, authors, files) into the brain.
|
||||
"""git/import — deprecated. Use bin/git/import.go (go-git, no git binary).
|
||||
|
||||
bin/git/import [REPO] import all commits -> leafs + graph
|
||||
bin/git/import --json emit import leafs as JSON, no write
|
||||
bin/git/import --limit 100 cap commits processed
|
||||
bin/git/import --since 2026-01-01 only recent commits
|
||||
bin/git/import --root DIR run per repo dir under DIR
|
||||
bin/git/import --no-env never read .env anywhere (default: true)
|
||||
|
||||
Reads `git log --no-merges --name-only` from the repo, maps commits to
|
||||
`info` leafs (root=info, type=commit) and writes the version graph
|
||||
`File -[:HAS_VERSION]-> Commit -[:AUTHORED]-> Person` into var/kb.lbug.
|
||||
Idempotent: leaf MERGE by (source,text via leaf_id), graph MERGE by sha.
|
||||
bin/git/import.go [REPO] [--json] [--limit N] [--since DATE]
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
||||
|
||||
from kblib import ( # noqa: E402
|
||||
connect, ensure_indexes, init_schema, upsert_leaf,
|
||||
)
|
||||
from gitimport import commits_to_leafs, ensure_git_schema, index_commits, parse_log # noqa: E402
|
||||
|
||||
LOG_FMT = "--format=%x1e%H%x1f%an%x1f%ae%x1f%aI%x1f%s"
|
||||
|
||||
|
||||
def git_log(repo: Path, limit: int = 0, since: str = "") -> str:
|
||||
cmd = ["git", "-C", str(repo), "log", "--no-merges", "--name-only", LOG_FMT]
|
||||
if since:
|
||||
cmd += ["--since", since]
|
||||
if limit:
|
||||
cmd += ["-n", str(limit)]
|
||||
try:
|
||||
out = subprocess.run(cmd, capture_output=True, text=True, timeout=120)
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
return ""
|
||||
if out.returncode != 0:
|
||||
print(f"git/import: {repo}: {out.stderr.strip()}", file=sys.stderr)
|
||||
return ""
|
||||
return out.stdout
|
||||
|
||||
|
||||
def repo_name(repo: Path) -> str:
|
||||
try:
|
||||
out = subprocess.run(
|
||||
["git", "-C", str(repo), "remote", "get-url", "origin"],
|
||||
capture_output=True, text=True, timeout=20)
|
||||
url = out.stdout.strip()
|
||||
return url.rstrip("/").split("/")[-1].removesuffix(".git") if url else repo.name
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
return repo.name
|
||||
|
||||
|
||||
def embedder():
|
||||
from model2vec import StaticModel
|
||||
model = StaticModel.from_pretrained("minishlab/potion-multilingual-128M")
|
||||
return lambda text: model.encode([text])[0].astype(float).tolist()
|
||||
|
||||
|
||||
def import_repo(conn, repo: Path, embed, limit: int, since: str,
|
||||
no_write: bool = False) -> tuple[int, int]:
|
||||
raw = git_log(repo, limit, since)
|
||||
commits = parse_log(raw)
|
||||
leafs = commits_to_leafs(commits, repo_name(repo))
|
||||
if no_write:
|
||||
return len(commits), 0
|
||||
written = 0
|
||||
for lf in leafs:
|
||||
query = f"{lf['heading']}\n\n{lf['text']}"
|
||||
emb = embed(lf["text"]) if lf["text"] else None
|
||||
upsert_leaf(conn, text=query, root="info", confidence="confirmed",
|
||||
source=lf["source"], source_rev="git", how="git/import",
|
||||
loc=lf["source"], type_=lf.get("type", "commit"),
|
||||
embedding=emb)
|
||||
written += 1
|
||||
index_commits(conn, commits, repo_name(repo))
|
||||
return len(commits), written
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
import argparse
|
||||
p = argparse.ArgumentParser(description="import git history into the brain")
|
||||
p.add_argument("repo", nargs="?", default=None)
|
||||
p.add_argument("--root", default=None, help="directory of repos to import (each git dir separately)")
|
||||
p.add_argument("--limit", type=int, default=0)
|
||||
p.add_argument("--since", default="")
|
||||
p.add_argument("--json", action="store_true")
|
||||
p.add_argument("--dry-run", action="store_true", help="parse + report, no db write")
|
||||
a = p.parse_args(argv)
|
||||
|
||||
repos: list[Path] = []
|
||||
if a.repo:
|
||||
repos = [Path(a.repo)]
|
||||
elif a.root:
|
||||
root = Path(a.root)
|
||||
if root.is_file():
|
||||
repos = [root]
|
||||
else:
|
||||
repos = [dp for dp in sorted(root.iterdir()) if (dp / ".git").exists() or dp.is_file()]
|
||||
else:
|
||||
repos = [ROOT]
|
||||
|
||||
total_commits = 0
|
||||
results: list[dict] = []
|
||||
if a.dry_run:
|
||||
for repo in repos:
|
||||
if not repo.exists():
|
||||
continue
|
||||
commits = parse_log(git_log(repo, a.limit, a.since))
|
||||
name = repo_name(repo)
|
||||
total_commits += len(commits)
|
||||
results.append({"repo": name, "commits": len(commits),
|
||||
"leafs": len(commits_to_leafs(commits, name)), "path": str(repo)})
|
||||
if a.json:
|
||||
print(json.dumps(results, indent=2))
|
||||
else:
|
||||
for r in results:
|
||||
print(f"{r['repo']:<24} {r['commits']:>5} commits -> {r['leafs']} leafs {r['path']}")
|
||||
return 0
|
||||
|
||||
# Never DROP FTS/VECTOR (ghost catalog). Upsert while indexes exist is OK;
|
||||
# ensure_indexes only CREATEs when missing.
|
||||
db, conn = connect(ROOT / "var" / "kb.lbug", read_only=False)
|
||||
init_schema(conn)
|
||||
embed = embedder()
|
||||
rows: list[dict] = []
|
||||
for repo in repos:
|
||||
if not repo.exists():
|
||||
continue
|
||||
reached, written = import_repo(conn, repo, embed, a.limit, a.since)
|
||||
total_commits += reached
|
||||
rows.append({"repo": repo_name(repo), "commits": reached, "written": written})
|
||||
ensure_indexes(conn)
|
||||
conn.close()
|
||||
db.close()
|
||||
|
||||
if a.json:
|
||||
print(json.dumps(rows, indent=2))
|
||||
else:
|
||||
for r in rows:
|
||||
print(f"imported {r['commits']:>5} commits -> {r['written']} leafs {r['repo']}")
|
||||
print(f"total: {total_commits} commits")
|
||||
return 0
|
||||
print(
|
||||
"bin/git/import is deprecated; use bin/git/import.go (go-git)",
|
||||
file=sys.stderr,
|
||||
)
|
||||
target = ROOT / "bin" / "git" / "import.go"
|
||||
os.execvp("go", ["go", "run", str(target), *argv])
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
Executable
+143
@@ -0,0 +1,143 @@
|
||||
//usr/bin/env go run "$0" "$@"; exit
|
||||
//
|
||||
// bin/git/import.go - read git history with go-git (no git binary).
|
||||
//
|
||||
// ./bin/git/import.go [REPO]
|
||||
// ./bin/git/import.go --json
|
||||
// ./bin/git/import.go --limit 100 --since 2026-01-01
|
||||
// ./bin/git/import.go --root DIR
|
||||
//
|
||||
// Conversion only: prints commit leafs. Brain write is bin/brain/index.go.
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"github.com/eSlider/2dph/internal/cmdbin"
|
||||
"github.com/eSlider/2dph/internal/gitlog"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(run(os.Args[1:]))
|
||||
}
|
||||
|
||||
func run(args []string) int {
|
||||
var repo, root, since string
|
||||
limit := 0
|
||||
jsonOut := false
|
||||
i := 0
|
||||
for i < len(args) {
|
||||
a := args[i]
|
||||
switch {
|
||||
case a == "--json":
|
||||
jsonOut = true
|
||||
case a == "--limit" && i+1 < len(args):
|
||||
i++
|
||||
n, err := strconv.Atoi(args[i])
|
||||
if err != nil || n < 0 {
|
||||
fmt.Fprintf(os.Stderr, "git/import: --limit must be a non-negative integer\n")
|
||||
return 2
|
||||
}
|
||||
limit = n
|
||||
case a == "--since" && i+1 < len(args):
|
||||
i++
|
||||
since = args[i]
|
||||
case a == "--root" && i+1 < len(args):
|
||||
i++
|
||||
root = args[i]
|
||||
case a == "-h" || a == "--help":
|
||||
fmt.Fprintln(os.Stderr, `usage: bin/git/import.go [REPO] [--json] [--limit N] [--since DATE] [--root DIR]`)
|
||||
return 0
|
||||
case len(a) > 0 && a[0] != '-':
|
||||
repo = a
|
||||
default:
|
||||
fmt.Fprintf(os.Stderr, "git/import: unknown flag %s\n", a)
|
||||
return 2
|
||||
}
|
||||
i++
|
||||
}
|
||||
|
||||
var sinceT time.Time
|
||||
if since != "" {
|
||||
var err error
|
||||
sinceT, err = parseSince(since)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "git/import: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
}
|
||||
|
||||
repos := []string{}
|
||||
if repo != "" {
|
||||
repos = []string{repo}
|
||||
} else if root != "" {
|
||||
entries, err := os.ReadDir(root)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "git/import: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
for _, e := range entries {
|
||||
p := filepath.Join(root, e.Name())
|
||||
if _, err := os.Stat(filepath.Join(p, ".git")); err == nil {
|
||||
repos = append(repos, p)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
repos = []string{cmdbin.Root()}
|
||||
}
|
||||
|
||||
opt := gitlog.Options{Limit: limit, Since: sinceT}
|
||||
type row struct {
|
||||
Repo string `json:"repo"`
|
||||
Path string `json:"path"`
|
||||
Commits int `json:"commits"`
|
||||
Leafs []gitlog.Leaf `json:"leafs,omitempty"`
|
||||
}
|
||||
var rows []row
|
||||
for _, p := range repos {
|
||||
name, err := gitlog.RepoName(p)
|
||||
if err != nil && name == "" {
|
||||
fmt.Fprintf(os.Stderr, "git/import: %s: %v\n", p, err)
|
||||
continue
|
||||
}
|
||||
cs, err := gitlog.Log(p, opt)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "git/import: %s: %v\n", p, err)
|
||||
return 1
|
||||
}
|
||||
leafs := make([]gitlog.Leaf, 0, len(cs))
|
||||
for _, c := range cs {
|
||||
leafs = append(leafs, gitlog.ToLeaf(c, name))
|
||||
}
|
||||
rows = append(rows, row{Repo: name, Path: p, Commits: len(cs), Leafs: leafs})
|
||||
}
|
||||
|
||||
if jsonOut {
|
||||
enc := json.NewEncoder(os.Stdout)
|
||||
enc.SetIndent("", " ")
|
||||
enc.SetEscapeHTML(false)
|
||||
if err := enc.Encode(rows); err != nil {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
for _, r := range rows {
|
||||
fmt.Printf("%-24s %5d commits %s\n", r.Repo, r.Commits, r.Path)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func parseSince(s string) (time.Time, error) {
|
||||
for _, layout := range []string{time.RFC3339, "2006-01-02"} {
|
||||
if t, err := time.Parse(layout, s); err == nil {
|
||||
return t, nil
|
||||
}
|
||||
}
|
||||
return time.Time{}, fmt.Errorf("cannot parse --since %q", s)
|
||||
}
|
||||
Executable
+114
@@ -0,0 +1,114 @@
|
||||
#!/usr/bin/env python3
|
||||
"""kb/add - incremental leaf write (no rebuild).
|
||||
|
||||
bin/kb/add --text T --root facts|info --source S
|
||||
bin/kb/add --json # stdin: one object or {"leafs":[...]}
|
||||
bin/kb/add --db PATH --json
|
||||
|
||||
Writes facts+info in one Ladybug transaction. Does not delete kb.lbug.
|
||||
Embedding is used when provided; otherwise model2vec encodes the text.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
||||
|
||||
from kblib import ( # noqa: E402
|
||||
EMBED_DIM,
|
||||
add_leafs,
|
||||
connect,
|
||||
ensure_indexes,
|
||||
init_schema,
|
||||
)
|
||||
|
||||
|
||||
def _as_leafs(payload: object) -> list[dict]:
|
||||
if isinstance(payload, list):
|
||||
return [dict(x) for x in payload]
|
||||
if isinstance(payload, dict):
|
||||
if "leafs" in payload:
|
||||
return [dict(x) for x in payload["leafs"]]
|
||||
return [dict(payload)]
|
||||
raise ValueError("json must be an object, a list, or {leafs:[...]}")
|
||||
|
||||
|
||||
def _embed_missing(leafs: list[dict]) -> None:
|
||||
missing = [lf for lf in leafs if not lf.get("embedding")]
|
||||
if not missing:
|
||||
return
|
||||
from model2vec import StaticModel
|
||||
|
||||
model = StaticModel.from_pretrained("minishlab/potion-multilingual-128M")
|
||||
for lf in missing:
|
||||
text = str(lf.get("text") or "")
|
||||
vec = model.encode([text])[0].astype(float).tolist()
|
||||
if len(vec) != EMBED_DIM:
|
||||
vec = (vec + [0.0] * EMBED_DIM)[:EMBED_DIM]
|
||||
lf["embedding"] = vec
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
import argparse
|
||||
|
||||
p = argparse.ArgumentParser(description="add leafs without rebuilding the brain")
|
||||
p.add_argument("--db", default="", help="path to kb.lbug (default var/kb.lbug)")
|
||||
p.add_argument("--json", action="store_true", help="read leaf JSON from stdin")
|
||||
p.add_argument("--text", default="", help="leaf text")
|
||||
p.add_argument("--root", default="info", choices=("facts", "info"))
|
||||
p.add_argument("--source", default="")
|
||||
p.add_argument("--confidence", default="confirmed")
|
||||
p.add_argument("--source-rev", default="working-tree")
|
||||
p.add_argument("--how", default="brain/add")
|
||||
p.add_argument("--loc", default="")
|
||||
p.add_argument("--type", default="reference", dest="type_")
|
||||
args = p.parse_args(argv)
|
||||
|
||||
if args.json:
|
||||
raw = sys.stdin.read()
|
||||
if not raw.strip():
|
||||
print("kb/add: empty stdin", file=sys.stderr)
|
||||
return 2
|
||||
leafs = _as_leafs(json.loads(raw))
|
||||
else:
|
||||
if not args.text or not args.source:
|
||||
print("kb/add: --text and --source are required (or --json)", file=sys.stderr)
|
||||
return 2
|
||||
leafs = [{
|
||||
"text": args.text,
|
||||
"root": args.root,
|
||||
"source": args.source,
|
||||
"confidence": args.confidence,
|
||||
"source_rev": args.source_rev,
|
||||
"how": args.how,
|
||||
"loc": args.loc or args.source,
|
||||
"type": args.type_,
|
||||
}]
|
||||
|
||||
for lf in leafs:
|
||||
if not lf.get("text") or not lf.get("source"):
|
||||
print("kb/add: each leaf needs text and source", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
_embed_missing(leafs)
|
||||
|
||||
from kblib import DB_PATH, VAR
|
||||
|
||||
dbpath = Path(args.db) if args.db else DB_PATH
|
||||
dbpath.parent.mkdir(parents=True, exist_ok=True)
|
||||
VAR.mkdir(exist_ok=True)
|
||||
db, conn = connect(dbpath, read_only=False)
|
||||
init_schema(conn)
|
||||
ids = add_leafs(conn, leafs)
|
||||
ensure_indexes(conn)
|
||||
conn.close()
|
||||
db.close()
|
||||
print(json.dumps({"mode": "add", "ids": ids, "db": str(dbpath)}))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv[1:]))
|
||||
+98
-14
@@ -2,12 +2,15 @@
|
||||
"""kb/index - build the 2dph brain from markdown + factual leafs.
|
||||
|
||||
bin/kb/index [--corpus DIR] [--rebuild] [--limit N]
|
||||
bin/kb/index --rebuild --with-facts --with-chats
|
||||
bin/kb/index --json # emit stats as JSON
|
||||
|
||||
Reads every .md under the corpus (default: repo root docs, skills, READMEs)
|
||||
as `info` leafs, embeds them with model2vec (potion-multilingual-128M), and
|
||||
writes them into var/kb.lbug with FTS + HNSW indexes. `facts` leafs come
|
||||
from bin/facts/extract (docker x compose x ssh-config pairing).
|
||||
from bin/facts/extract (docker × compose × ssh-config pairing) when
|
||||
`--with-facts` is set. `--with-chats` indexes markdown under var/chats/md
|
||||
(or a given dir) as info. WhatsApp sync stays out of v1.
|
||||
|
||||
--rebuild drops the database file and indexes from scratch. Without it a run
|
||||
is idempotent (MERGE by (source,text) id).
|
||||
@@ -22,7 +25,7 @@ ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
||||
|
||||
from kblib import ( # noqa: E402
|
||||
connect, ensure_indexes, init_schema, upsert_leaf,
|
||||
add_leafs, connect, ensure_indexes, init_schema, upsert_leaf, link_from_file,
|
||||
open_readonly, stats,
|
||||
)
|
||||
from mdleaves import read_markdown, to_all, walk_markdown # noqa: E402
|
||||
@@ -81,10 +84,11 @@ def index_leafs(conn, leafs: list[dict], embed_fn, limit: int) -> tuple[int, int
|
||||
for lf in leafs[:limit] if limit else leafs:
|
||||
query = f"{lf['heading']}\n\n{lf['text']}"
|
||||
emb = embed_fn(lf["text"]) if lf["text"] else None
|
||||
upsert_leaf(conn, text=query, root="info", confidence="confirmed",
|
||||
lid = upsert_leaf(conn, text=query, root="info", confidence="confirmed",
|
||||
source=lf["source"], source_rev="working-tree",
|
||||
how="kb/index", loc=lf["source"], type_=lf.get("type", "reference"),
|
||||
embedding=emb)
|
||||
link_from_file(conn, lid, lf["source"], repo=str(lf.get("repo") or ""))
|
||||
count += 1
|
||||
return count, len(leafs)
|
||||
|
||||
@@ -95,12 +99,65 @@ def embedder():
|
||||
return lambda text: model.encode([text])[0].astype(float).tolist()
|
||||
|
||||
|
||||
def index_fact_dicts(conn, facts: list[dict], embed_fn) -> int:
|
||||
"""Write extract-shaped dicts as root=facts leafs (2-source source field)."""
|
||||
leafs = []
|
||||
for f in facts:
|
||||
text = str(f.get("text") or "")
|
||||
source = str(f.get("source") or "")
|
||||
if not text or not source:
|
||||
continue
|
||||
leafs.append({
|
||||
"text": text,
|
||||
"root": "facts",
|
||||
"confidence": "confirmed",
|
||||
"source": source,
|
||||
"source_rev": f.get("source_rev") or "working-tree",
|
||||
"how": f.get("how") or "facts/extract",
|
||||
"loc": f.get("loc") or source,
|
||||
"type": "fact",
|
||||
"embedding": embed_fn(text) if text else None,
|
||||
})
|
||||
return len(add_leafs(conn, leafs))
|
||||
|
||||
|
||||
def facts_from_extract() -> list[dict]:
|
||||
import subprocess
|
||||
proc = subprocess.run(
|
||||
[sys.executable, str(ROOT / "bin" / "facts" / "extract"), "--json", "--dry-run"],
|
||||
cwd=ROOT,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
if proc.returncode != 0:
|
||||
print(f"kb/index: facts/extract failed: {proc.stderr}", file=sys.stderr)
|
||||
return []
|
||||
try:
|
||||
payload = json.loads(proc.stdout)
|
||||
except json.JSONDecodeError:
|
||||
print("kb/index: facts/extract produced non-JSON", file=sys.stderr)
|
||||
return []
|
||||
return list(payload.get("facts") or [])
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
import argparse
|
||||
p = argparse.ArgumentParser(description="build the 2dph brain index")
|
||||
p.add_argument("--corpus", action="append", help="extra markdown dir/file to index (may repeat)")
|
||||
p.add_argument("--rebuild", action="store_true", help="fresh db + indexes")
|
||||
p.add_argument("--db", default="", help="path to kb.lbug (default var/kb.lbug)")
|
||||
p.add_argument("--no-defaults", action="store_true", help="do not index repo README/docs/skills")
|
||||
p.add_argument("--with-mail", action="store_true", help="include var/mail message.md leafs")
|
||||
p.add_argument("--with-facts", action="store_true", help="run facts/extract into root=facts")
|
||||
p.add_argument("--facts-json", default="", help="JSON list (or {facts:[...]}) of fact dicts")
|
||||
p.add_argument(
|
||||
"--with-chats",
|
||||
nargs="?",
|
||||
const=str(ROOT / "var" / "chats" / "md"),
|
||||
default="",
|
||||
help="index chat markdown as info (default var/chats/md)",
|
||||
)
|
||||
p.add_argument("--since", default="", help="with --with-mail, only messages dated >= YYYY-MM-DD")
|
||||
p.add_argument("--dry-run", action="store_true", help="count leafs, write nothing")
|
||||
p.add_argument(
|
||||
@@ -114,44 +171,71 @@ def main(argv: list[str]) -> int:
|
||||
|
||||
from kblib import DB_PATH, VAR
|
||||
|
||||
leafs = load_corpus(ROOT)
|
||||
dbpath = Path(a.db) if a.db else DB_PATH
|
||||
leafs: list[dict] = [] if a.no_defaults else load_corpus(ROOT)
|
||||
if a.corpus:
|
||||
for source in a.corpus:
|
||||
leafs.extend(load_corpus_glob(source))
|
||||
chat_n = 0
|
||||
if a.with_chats:
|
||||
chats = load_corpus_glob(a.with_chats)
|
||||
chat_n = len(chats)
|
||||
leafs.extend(chats)
|
||||
mail_n = 0
|
||||
if a.with_mail:
|
||||
mail = from_mail_root(ROOT / "var" / "mail", since=a.since)
|
||||
mail_n = len(mail)
|
||||
leafs.extend(mail)
|
||||
|
||||
facts: list[dict] = []
|
||||
if a.facts_json:
|
||||
raw = Path(a.facts_json).read_text(encoding="utf-8")
|
||||
payload = json.loads(raw)
|
||||
facts = list(payload.get("facts") if isinstance(payload, dict) else payload)
|
||||
if a.with_facts:
|
||||
facts.extend(facts_from_extract())
|
||||
|
||||
if a.dry_run:
|
||||
msg = {"indexed": 0, "corpus_total": len(leafs), "mail_leafs": mail_n, "dry_run": True}
|
||||
msg = {
|
||||
"indexed": 0,
|
||||
"corpus_total": len(leafs),
|
||||
"mail_leafs": mail_n,
|
||||
"chat_leafs": chat_n,
|
||||
"facts_leafs": len(facts),
|
||||
"dry_run": True,
|
||||
}
|
||||
print(json.dumps(msg, indent=2) if a.json else
|
||||
f"brain/index: {len(leafs)} leafs would be indexed (mail={mail_n})")
|
||||
f"brain/index: {len(leafs)} info + {len(facts)} facts would be indexed")
|
||||
return 0
|
||||
|
||||
VAR.mkdir(exist_ok=True)
|
||||
if a.rebuild and DB_PATH.exists():
|
||||
DB_PATH.unlink()
|
||||
dbpath.parent.mkdir(parents=True, exist_ok=True)
|
||||
if a.rebuild and dbpath.exists():
|
||||
dbpath.unlink()
|
||||
|
||||
db, conn = connect(DB_PATH, read_only=False)
|
||||
db, conn = connect(dbpath, read_only=False)
|
||||
init_schema(conn)
|
||||
|
||||
# Never DROP FTS/VECTOR (ghost catalog). Write leafs, then ensure indexes
|
||||
# unless --skip-indexes (seed facts first — MERGE under live FTS corrupts it).
|
||||
# --rebuild already deleted kb.lbug above, so CREATE runs on a clean DB.
|
||||
embed = embedder()
|
||||
done, total = index_leafs(conn, leafs, embed, a.limit)
|
||||
fact_n = index_fact_dicts(conn, facts, embed) if facts else 0
|
||||
if not a.skip_indexes:
|
||||
ensure_indexes(conn)
|
||||
s = stats(conn)
|
||||
conn.close()
|
||||
db.close()
|
||||
|
||||
result = {"indexed": done, "corpus_total": total, **{k: v for k, v in s.items() if k in ("total", "by_root")}}
|
||||
result = {
|
||||
"indexed": done,
|
||||
"corpus_total": total,
|
||||
"facts_leafs": fact_n,
|
||||
"chat_leafs": chat_n,
|
||||
**{k: v for k, v in s.items() if k in ("total", "by_root")},
|
||||
}
|
||||
if a.skip_indexes:
|
||||
result["indexes"] = "skipped"
|
||||
print(json.dumps(result, indent=2) if a.json else f"indexed {done}/{total} leafs; db total {s['total']}")
|
||||
print(json.dumps(result, indent=2) if a.json else
|
||||
f"indexed {done}/{total} info + {fact_n} facts; db total {s['total']}")
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
+4
-6
@@ -1,10 +1,9 @@
|
||||
#!/usr/bin/env bash
|
||||
# bin/kb/search — deprecated wrapper. Use bin/brain/search.go.
|
||||
# Sets CGO for ladybug, builds a binary (embed daemon needs a real executable),
|
||||
# then execs it. Prints one deprecation line.
|
||||
# CGO via Zig (bin/cgo/zig), not gcc. Builds a binary then execs it.
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||
ROOT="$(CDPATH= cd -- "$(dirname "$0")/../.." && pwd)"
|
||||
BIN="$ROOT/var/bin/brain-search"
|
||||
SRC="$ROOT/internal/brain"
|
||||
CMD="$ROOT/bin/brain"
|
||||
@@ -24,11 +23,10 @@ else
|
||||
fi
|
||||
|
||||
if [ "$need_build" -eq 1 ]; then
|
||||
echo "Building brain/search..." >&2
|
||||
echo "Building brain/search (zig cc)..." >&2
|
||||
(
|
||||
cd "$ROOT" &&
|
||||
CGO_CFLAGS="-I$ROOT/lib-ladybug" \
|
||||
CGO_LDFLAGS="-L$ROOT/lib-ladybug -Wl,-rpath,$ROOT/lib-ladybug" \
|
||||
eval "$("$ROOT/bin/cgo/zig" env)" &&
|
||||
go build -tags system_ladybug -o "$BIN" ./bin/brain
|
||||
) || exit 1
|
||||
fi
|
||||
|
||||
+11
-77
@@ -7,7 +7,7 @@
|
||||
bin/mail/import --since 2026-01-01 only messages after a date
|
||||
bin/mail/import --limit 50 cap messages per run
|
||||
bin/mail/import --no-attachments body only, skip attachment conversion
|
||||
bin/mail/import --ocr OCR scanned PDFs/images via docling
|
||||
bin/mail/import --ocr OCR images (PDFs OCR when textless)
|
||||
bin/mail/import --dry-run list messages without writing anything
|
||||
|
||||
Writes one directory per message: var/mail/{folder}/{message_id}/
|
||||
@@ -16,10 +16,11 @@ Writes one directory per message: var/mail/{folder}/{message_id}/
|
||||
attachments/*.md converted attachment content
|
||||
|
||||
Indexing is a separate step (`bin/brain/index.go --rebuild`): conversion can
|
||||
crash in native docling and must not leave the brain DB mid-transaction.
|
||||
crash and must not leave the brain DB mid-transaction.
|
||||
|
||||
Requires ONLYOFFICE_URL/USER/PASS in .env (or env). Idempotent: a message
|
||||
already present (message.md exists) is skipped unless --force.
|
||||
Requires ONLYOFFICE_URL/USER/PASS in .env (or env) except `--from-raw`.
|
||||
Idempotent: a message already present (message.md exists) is skipped unless
|
||||
--force.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -41,10 +42,10 @@ from mailconv import ( # noqa: E402
|
||||
IMAGE_SUFFIXES,
|
||||
LEGACY_OFFICE_SUFFIXES,
|
||||
TEXT_SUFFIXES,
|
||||
convert_pdf,
|
||||
html_to_markdown,
|
||||
is_convertible,
|
||||
normalize_markdown,
|
||||
subject_to_filename,
|
||||
ocr_image,
|
||||
zip_extract_safe,
|
||||
)
|
||||
|
||||
@@ -147,9 +148,9 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
|
||||
except Exception as e:
|
||||
return f"\n<!-- conversion failed: {e} -->\n"
|
||||
if suffix == ".pdf":
|
||||
return _convert_pdf(path, ocr)
|
||||
return convert_pdf(path, ocr)
|
||||
if suffix in IMAGE_SUFFIXES and ocr:
|
||||
return _convert_pdf(path, ocr)
|
||||
return ocr_image(path) or "\n<!-- ocr unavailable -->\n"
|
||||
if suffix in LEGACY_OFFICE_SUFFIXES:
|
||||
return _convert_legacy(path)
|
||||
if suffix in ARCHIVE_SUFFIXES:
|
||||
@@ -157,67 +158,6 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
|
||||
return None
|
||||
|
||||
|
||||
def _convert_pdf(path: Path, ocr: bool) -> str:
|
||||
"""Convert one PDF to markdown.
|
||||
|
||||
Fast path: poppler's pdftotext (-layout) extracts exact text from
|
||||
born-digital PDFs in ~15ms vs docling's 1-3s. Only textless PDFs (scanned
|
||||
pages, layout-heavy) fall back to docling, which runs isolated in a
|
||||
subprocess because its native onnx/RT-DETR has segfaulted the main process.
|
||||
"""
|
||||
text = _pdf_fast_text(path)
|
||||
if ocr or text is None or not text.strip():
|
||||
return _convert_pdf_docling(path, ocr)
|
||||
return normalize_markdown(text)
|
||||
|
||||
|
||||
def _pdf_fast_text(path: Path) -> str | None:
|
||||
"""pdftotext -layout; None when poppler is unavailable (or the PDF has no text layer)."""
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["pdftotext", "-layout", str(path), "-"],
|
||||
capture_output=True, timeout=60)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return None
|
||||
if proc.returncode != 0:
|
||||
return None
|
||||
return proc.stdout.decode("utf-8", errors="replace")
|
||||
|
||||
|
||||
def _convert_pdf_docling(path: Path, ocr: bool) -> str:
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
[sys.executable, os.path.abspath(__file__), "--pdf-worker", str(path),
|
||||
"--ocr" if ocr else "--no-ocr"],
|
||||
capture_output=True, text=True, timeout=600)
|
||||
except subprocess.TimeoutExpired:
|
||||
return "\n<!-- pdf conversion timed out -->\n"
|
||||
if proc.returncode != 0:
|
||||
tail = proc.stderr.strip().splitlines()[-3:]
|
||||
return f"\n<!-- pdf conversion failed: {proc.returncode}: {' | '.join(tail)} -->\n"
|
||||
return proc.stdout
|
||||
|
||||
|
||||
def _pdf_worker(path: Path, ocr: bool) -> None:
|
||||
"""docling worker entry: prints converted markdown on stdout, exits non-zero on error."""
|
||||
try:
|
||||
from docling.document_converter import DocumentConverter, PdfFormatOption
|
||||
from docling.datamodel.pipeline_options import PdfPipelineOptions
|
||||
opts = PdfPipelineOptions()
|
||||
opts.do_ocr = bool(ocr)
|
||||
opts.do_table_structure = True
|
||||
conv = DocumentConverter(format_options={"pdf": PdfFormatOption(pipeline_options=opts)})
|
||||
res = conv.convert(str(path))
|
||||
sys.stdout.write(normalize_markdown(res.document.export_to_markdown()))
|
||||
sys.exit(0)
|
||||
except Exception as e:
|
||||
# errors/stacktraces to stderr; the caller only reports a one-liner
|
||||
print(f"pdf-worker: {e}", file=sys.stderr)
|
||||
import traceback
|
||||
traceback.print_exc(file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def _convert_legacy(path: Path) -> str:
|
||||
"""Legacy .doc/.xls/.ppt -> md via pandoc (installed) or a stub."""
|
||||
try:
|
||||
@@ -356,19 +296,12 @@ def main(argv: list[str]) -> int:
|
||||
p.add_argument("--from-raw", default="",
|
||||
help="convert Go-synced dirs (var/mail/<folder>/<id>/message.json) to markdown")
|
||||
p.add_argument("--no-attachments", action="store_true", help="skip attachment download+convert")
|
||||
p.add_argument("--ocr", action="store_true", help="OCR scanned PDFs/images via docling")
|
||||
p.add_argument("--ocr", action="store_true", help="OCR images (PDFs OCR when textless)")
|
||||
p.add_argument("--force", action="store_true", help="re-import even if message.md exists")
|
||||
p.add_argument("--dry-run", action="store_true", help="list messages, write nothing")
|
||||
p.add_argument("--json", action="store_true")
|
||||
p.add_argument("--pdf-worker", default="", help=argparse.SUPPRESS)
|
||||
p.add_argument("--no-ocr", action="store_true", help=argparse.SUPPRESS)
|
||||
a = p.parse_args(argv)
|
||||
|
||||
if a.pdf_worker:
|
||||
_pdf_worker(Path(a.pdf_worker), ocr=not a.no_ocr)
|
||||
return 0
|
||||
|
||||
conf = load_env()
|
||||
fid = folder_id(a.folder)
|
||||
out_root = ROOT / "var" / "mail"
|
||||
summary: list[dict] = []
|
||||
@@ -394,6 +327,7 @@ def main(argv: list[str]) -> int:
|
||||
target_dir=msg_dir.parent))
|
||||
summary.append(entry)
|
||||
else:
|
||||
conf = load_env()
|
||||
OOCLIENT = OOClient(conf)
|
||||
if a.id:
|
||||
messages = [{"id": i} for i in a.id]
|
||||
|
||||
Executable
+48
@@ -0,0 +1,48 @@
|
||||
//usr/bin/env go run -tags=mail_ocr "$0" "$@"; exit
|
||||
//go:build mail_ocr
|
||||
//
|
||||
// bin/mail/ocr.go - OCR an image or scanned PDF (tesseract eng+deu).
|
||||
//
|
||||
// ./bin/mail/ocr.go scan.png
|
||||
// ./bin/mail/ocr.go scan.pdf
|
||||
// OCR_ENGINE=paddle ./bin/mail/ocr.go scan.png
|
||||
//
|
||||
// PDFs try pdftotext -layout first; empty text layer uses pdftoppm + tesseract.
|
||||
// No gocv. Tesseract CGO bindings are not used (D21 Zig owns Ladybug CGO).
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"github.com/eSlider/2dph/internal/ocr"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(run(os.Args[1:]))
|
||||
}
|
||||
|
||||
func run(args []string) int {
|
||||
if len(args) != 1 || strings.HasPrefix(args[0], "-") {
|
||||
fmt.Fprintln(os.Stderr, `usage: bin/mail/ocr.go <image|pdf>`)
|
||||
return 2
|
||||
}
|
||||
path := args[0]
|
||||
var (
|
||||
text string
|
||||
err error
|
||||
)
|
||||
if strings.HasSuffix(strings.ToLower(path), ".pdf") {
|
||||
text, err = ocr.PDFFile(path)
|
||||
} else {
|
||||
text, err = ocr.ImageFile(path)
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "mail/ocr: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Println(text)
|
||||
return 0
|
||||
}
|
||||
+85
-5
@@ -1,20 +1,100 @@
|
||||
//usr/bin/env go run -tags=markdown_import "$0" "$@"; exit
|
||||
//go:build markdown_import
|
||||
//usr/bin/env go run "$0" "$@"; exit
|
||||
//
|
||||
// bin/markdown/import.go - split markdown into leafs (mistune).
|
||||
// bin/markdown/import.go - split markdown into leafs (H2 boundaries).
|
||||
//
|
||||
// ./bin/markdown/import.go [dir]
|
||||
// ./bin/markdown/import.go --files a.md,b.md --json
|
||||
//
|
||||
// Conversion only. Brain write is bin/brain/index.go.
|
||||
// Python bin/md/import remains as a fallback.
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"github.com/eSlider/2dph/internal/cmdbin"
|
||||
"github.com/eSlider/2dph/internal/mdleaves"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(cmdbin.ExecFile("bin/md/import", os.Args[1:]))
|
||||
os.Exit(run(os.Args[1:]))
|
||||
}
|
||||
|
||||
func run(args []string) int {
|
||||
jsonOut := false
|
||||
files := ""
|
||||
root := "."
|
||||
for i := 0; i < len(args); i++ {
|
||||
a := args[i]
|
||||
switch {
|
||||
case a == "--json":
|
||||
jsonOut = true
|
||||
case a == "--files" && i+1 < len(args):
|
||||
i++
|
||||
files = args[i]
|
||||
case strings.HasPrefix(a, "--files="):
|
||||
files = strings.TrimPrefix(a, "--files=")
|
||||
case a == "-h" || a == "--help":
|
||||
fmt.Fprintln(os.Stderr, "bin/markdown/import.go [dir] [--files a.md,b.md] [--json]")
|
||||
return 0
|
||||
case strings.HasPrefix(a, "-"):
|
||||
fmt.Fprintln(os.Stderr, "unknown arg:", a)
|
||||
return 2
|
||||
default:
|
||||
root = a
|
||||
}
|
||||
}
|
||||
|
||||
var paths []string
|
||||
if files != "" {
|
||||
for _, f := range strings.Split(files, ",") {
|
||||
f = strings.TrimSpace(f)
|
||||
if f != "" {
|
||||
paths = append(paths, f)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
st, err := os.Stat(root)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "md/import: no such path %s\n", root)
|
||||
return 2
|
||||
}
|
||||
if !st.IsDir() {
|
||||
paths = []string{root}
|
||||
} else {
|
||||
var err error
|
||||
paths, err = mdleaves.WalkMarkdown(root)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "md/import: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(paths) == 0 {
|
||||
fmt.Fprintln(os.Stderr, "md/import: no markdown files")
|
||||
return 1
|
||||
}
|
||||
|
||||
var all []mdleaves.Leaf
|
||||
for _, p := range paths {
|
||||
raw, err := os.ReadFile(p)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "md/import: %s: %v\n", p, err)
|
||||
continue
|
||||
}
|
||||
all = append(all, mdleaves.ToAll(string(raw), p, "")...)
|
||||
}
|
||||
if jsonOut {
|
||||
s, err := mdleaves.EncodeJSON(all)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
return 1
|
||||
}
|
||||
fmt.Print(s)
|
||||
return 0
|
||||
}
|
||||
fmt.Print(mdleaves.EncodeYAML(all))
|
||||
return 0
|
||||
}
|
||||
|
||||
Executable
+73
@@ -0,0 +1,73 @@
|
||||
//usr/bin/env go run -tags=qa_stats "$0" "$@"; exit
|
||||
//go:build qa_stats
|
||||
//
|
||||
// bin/qa/stats.go - DuckDB quantiles over a JSON number array or JSONL count.
|
||||
//
|
||||
// ./bin/qa/stats.go <<< '[1,2,3,4,5]'
|
||||
// ./bin/qa/stats.go --jsonl rows.jsonl
|
||||
//
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
// DuckDB CGO needs gcc/g++ (not Zig). After eval "$(bin/cgo/zig env)":
|
||||
// CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= ./bin/qa/stats.go
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"github.com/eSlider/2dph/internal/duckstats"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(run(os.Args[1:]))
|
||||
}
|
||||
|
||||
func run(args []string) int {
|
||||
jsonl := ""
|
||||
for i := 0; i < len(args); i++ {
|
||||
a := args[i]
|
||||
switch {
|
||||
case a == "--jsonl" && i+1 < len(args):
|
||||
i++
|
||||
jsonl = args[i]
|
||||
case strings.HasPrefix(a, "--jsonl="):
|
||||
jsonl = strings.TrimPrefix(a, "--jsonl=")
|
||||
case a == "-h" || a == "--help":
|
||||
fmt.Fprintln(os.Stderr, "bin/qa/stats.go [--jsonl FILE] # stdin = JSON [float,…]")
|
||||
return 0
|
||||
default:
|
||||
fmt.Fprintln(os.Stderr, "unknown arg:", a)
|
||||
return 2
|
||||
}
|
||||
}
|
||||
if jsonl != "" {
|
||||
n, err := duckstats.CountJSONL(jsonl)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
return 1
|
||||
}
|
||||
fmt.Printf("n: %d\n", n)
|
||||
return 0
|
||||
}
|
||||
raw, err := io.ReadAll(os.Stdin)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
return 1
|
||||
}
|
||||
var samples []float64
|
||||
if err := json.Unmarshal(raw, &samples); err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
return 1
|
||||
}
|
||||
s, err := duckstats.Quantiles(samples)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
return 1
|
||||
}
|
||||
fmt.Printf("n: %d\nmin: %g\np50: %g\np95: %g\nmax: %g\navg: %g\n",
|
||||
s.N, s.Min, s.P50, s.P95, s.Max, s.Avg)
|
||||
return 0
|
||||
}
|
||||
Executable
+102
@@ -0,0 +1,102 @@
|
||||
//usr/bin/env go run -tags=reasoner_bakeoff "$0" "$@"; exit
|
||||
//go:build reasoner_bakeoff
|
||||
//
|
||||
// bin/reasoner/bakeoff.go - CPU tool-call bake-off against an OpenAI-compatible URL (D18).
|
||||
//
|
||||
// REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b ./bin/reasoner/bakeoff.go
|
||||
// ./bin/reasoner/bakeoff.go --model MichelRosselli/bonsai-27b:Q1_0 --json
|
||||
//
|
||||
// Measures OpenAI tool_calls (search/get/audit) and RSS from Ollama /api/ps, not VRAM.
|
||||
// PicoClaw is compose profile picoclaw; tool names match internal/httpapi MCP ops.
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"github.com/eSlider/2dph/internal/duckstats"
|
||||
"github.com/eSlider/2dph/internal/reasoner"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(run(os.Args[1:]))
|
||||
}
|
||||
|
||||
func run(args []string) int {
|
||||
base := os.Getenv("REASONER_BASE_URL")
|
||||
if base == "" {
|
||||
base = "http://127.0.0.1:11435/v1"
|
||||
}
|
||||
model := os.Getenv("REASONER_MODEL")
|
||||
if model == "" {
|
||||
model = reasoner.OllamaRAM
|
||||
}
|
||||
jsonOut := false
|
||||
device := "cpu"
|
||||
for i := 0; i < len(args); i++ {
|
||||
a := args[i]
|
||||
switch {
|
||||
case a == "--json":
|
||||
jsonOut = true
|
||||
case a == "--model" && i+1 < len(args):
|
||||
i++
|
||||
model = args[i]
|
||||
case strings.HasPrefix(a, "--model="):
|
||||
model = strings.TrimPrefix(a, "--model=")
|
||||
case a == "--base-url" && i+1 < len(args):
|
||||
i++
|
||||
base = args[i]
|
||||
case a == "--device" && i+1 < len(args):
|
||||
i++
|
||||
device = args[i]
|
||||
case a == "-h" || a == "--help":
|
||||
fmt.Fprintln(os.Stderr, "bin/reasoner/bakeoff.go [--model ID] [--base-url URL] [--device cpu] [--json]")
|
||||
return 0
|
||||
default:
|
||||
fmt.Fprintln(os.Stderr, "unknown arg:", a)
|
||||
return 2
|
||||
}
|
||||
}
|
||||
c := reasoner.Client{BaseURL: base, Model: model, Device: device}
|
||||
rep := reasoner.Run(c)
|
||||
lat := make([]float64, 0, len(rep.Prompts))
|
||||
for _, p := range rep.Prompts {
|
||||
lat = append(lat, float64(p.LatencyMS))
|
||||
}
|
||||
if st, err := duckstats.Quantiles(lat); err == nil {
|
||||
rep.LatencyP50MS = st.P50
|
||||
rep.LatencyP95MS = st.P95
|
||||
}
|
||||
raw, err := json.MarshalIndent(rep, "", " ")
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
return 1
|
||||
}
|
||||
if jsonOut {
|
||||
fmt.Println(string(raw))
|
||||
} else {
|
||||
fmt.Printf("model: %s\n", rep.Model)
|
||||
fmt.Printf("hf_id: %s\n", rep.HF)
|
||||
fmt.Printf("device: %s\n", rep.Device)
|
||||
fmt.Printf("tool_call: %d/%d\n", rep.ToolCallOK, rep.ToolCallN)
|
||||
fmt.Printf("xml_leak: %d\n", rep.XMLLeak)
|
||||
fmt.Printf("rss_mb: %d\n", rep.RSSMB)
|
||||
fmt.Printf("vram_mb: %d\n", rep.VRAMMB)
|
||||
fmt.Printf("latency_p50_ms: %g\n", rep.LatencyP50MS)
|
||||
fmt.Printf("latency_p95_ms: %g\n", rep.LatencyP95MS)
|
||||
for _, p := range rep.Prompts {
|
||||
status := "fail"
|
||||
if p.OK {
|
||||
status = "ok"
|
||||
}
|
||||
fmt.Printf(" %s: %s wanted=%s got=%s xml=%v %dms %s\n", p.WantedTool, status, p.WantedTool, p.ToolName, p.XMLLeak, p.LatencyMS, p.Err)
|
||||
}
|
||||
}
|
||||
if rep.ToolCallN == 0 {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
// Commands in this directory are shebang mains (bakeoff.go).
|
||||
package main
|
||||
+3
-60
@@ -1,21 +1,12 @@
|
||||
"""gitimport - parse `git log` output and turn commits into brain leafs.
|
||||
"""gitimport - Ladybug graph writes for Commit/File/Person (no git binary).
|
||||
|
||||
Pure, testable functions. Field grammar (see bin/git/import):
|
||||
|
||||
git log --no-merges --name-only \
|
||||
--format='%x1e%H%x1f%an%x1f%ae%x1f%aI%x1f%s'
|
||||
|
||||
0x1e = record separator, 0x1f = field separator.
|
||||
Files: newline-separated lines following each record's subject.
|
||||
Commit records come from bin/git/import.go (go-git). This module only MERGEs
|
||||
the version graph File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
REC_SEP = "\x1e"
|
||||
FIELD_SEP = "\x1f"
|
||||
|
||||
|
||||
@dataclass
|
||||
class Commit:
|
||||
@@ -26,54 +17,6 @@ class Commit:
|
||||
subject: str
|
||||
files: list[str] = field(default_factory=list)
|
||||
|
||||
def leaf_text(self, repo: str) -> str:
|
||||
head = f"commit {self.sha[:12]} in {repo} — {self.subject}"
|
||||
body = [head, f"Author: {self.author} <{self.email}>", f"Date: {self.date}"]
|
||||
if self.files:
|
||||
body.append("Changing: " + ", ".join(self.files))
|
||||
return "\n".join(body)
|
||||
|
||||
|
||||
def parse_log(text: str) -> list[Commit]:
|
||||
"""Parse `git log` output into Commit records.
|
||||
|
||||
Records are separated by 0x1e. A record is fields joined by 0x1f,
|
||||
followed by optional newline-separated file paths inside the next
|
||||
segment (git emits blank line + files after each record).
|
||||
"""
|
||||
commits: list[Commit] = []
|
||||
# field records and file lists alternate; simpler: split on REC_SEP,
|
||||
# each chunk = header line, possibly followed by newline + files.
|
||||
for chunk in text.split(REC_SEP):
|
||||
chunk = chunk.strip("\n")
|
||||
if not chunk:
|
||||
continue
|
||||
lines = chunk.split("\n", 1)
|
||||
header = lines[0].split(FIELD_SEP)
|
||||
if len(header) < 5:
|
||||
continue
|
||||
sha, author, email, date, subject = header[:5]
|
||||
files = [ln.strip() for ln in lines[1].splitlines() if ln.strip()] if len(lines) > 1 else []
|
||||
commits.append(Commit(sha=sha, author=author, email=email,
|
||||
date=date, subject=subject, files=files))
|
||||
return commits
|
||||
|
||||
|
||||
def commits_to_leafs(commits: list[Commit], repo: str) -> list[dict]:
|
||||
"""Map commits to the leaf shape bin/kb/index expects (source/repo/...)."""
|
||||
out: list[dict] = []
|
||||
for c in commits:
|
||||
out.append({
|
||||
"source": f"{repo}@{c.sha}",
|
||||
"repo": repo,
|
||||
"heading": f"commit {c.sha[:12]} — {c.subject}",
|
||||
"text": c.leaf_text(repo),
|
||||
"type": "commit",
|
||||
"status": "current",
|
||||
"related": ",".join(c.files),
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
GIT_SCHEMA = (
|
||||
"CREATE NODE TABLE IF NOT EXISTS Commit (id STRING, repo STRING, subject STRING, "
|
||||
|
||||
+92
-1
@@ -4,7 +4,7 @@ Single embedded graph `var/kb.lbug`. Two roots: facts (assertions backed by
|
||||
>=2 independent sources) and info (narrative leafs). Hybrid retrieval: BM25
|
||||
(FTS extension) + HNSW cosine (VECTOR extension) + Cypher graph hops.
|
||||
|
||||
All access is read-only unless `--rebuild` is passed to kb/index.
|
||||
All access is read-only unless `--rebuild` (kb/index) or `kb/add`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -118,6 +118,97 @@ def upsert_leaf(conn: ladybug.Connection, *, text: str, root: str, confidence: s
|
||||
return lid
|
||||
|
||||
|
||||
def add_leafs(conn: ladybug.Connection, leafs: list[dict]) -> list[str]:
|
||||
"""Write facts+info leafs in one transaction. Safe while FTS/HNSW exist.
|
||||
|
||||
Each leaf dict: text, source, optional root/confidence/source_rev/how/loc/type/embedding.
|
||||
Does not delete the database file. Measured on Ladybug 0.19: MERGE of new
|
||||
ids (and updates) stays FTS+HNSW queryable; DROP INDEX is the fatal path.
|
||||
"""
|
||||
if not leafs:
|
||||
return []
|
||||
started = False
|
||||
try:
|
||||
conn.execute("BEGIN TRANSACTION")
|
||||
started = True
|
||||
except Exception:
|
||||
started = False
|
||||
ids: list[str] = []
|
||||
try:
|
||||
for lf in leafs:
|
||||
ids.append(
|
||||
upsert_leaf(
|
||||
conn,
|
||||
text=str(lf["text"]),
|
||||
root=str(lf.get("root") or ROOT_INFO),
|
||||
confidence=str(lf.get("confidence") or CONF_CONFIRMED),
|
||||
source=str(lf["source"]),
|
||||
source_rev=str(lf.get("source_rev") or "working-tree"),
|
||||
how=str(lf.get("how") or "brain/add"),
|
||||
loc=str(lf.get("loc") or lf.get("source") or ""),
|
||||
type_=str(lf.get("type") or lf.get("type_") or "reference"),
|
||||
embedding=lf.get("embedding"),
|
||||
)
|
||||
)
|
||||
if started:
|
||||
conn.execute("COMMIT")
|
||||
except Exception:
|
||||
if started:
|
||||
try:
|
||||
conn.execute("ROLLBACK")
|
||||
except Exception:
|
||||
pass
|
||||
raise
|
||||
return ids
|
||||
|
||||
|
||||
def file_id(repo: str, path: str) -> str:
|
||||
"""Stable File.id matching gitimport (`repo:path`)."""
|
||||
return f"{repo}:{path}" if repo else path
|
||||
|
||||
|
||||
def link_from_file(conn: ladybug.Connection, leaf_id: str, path: str,
|
||||
repo: str = "", mtime: str = "") -> str:
|
||||
"""MERGE File and Leaf-[:FROM_FILE]->File so --hop 1 can walk."""
|
||||
fid = file_id(repo, path)
|
||||
conn.execute(
|
||||
"MERGE (f:File {id:$id}) SET f.path=$path, f.repo=$repo, f.mtime=$mtime",
|
||||
parameters={"id": fid, "path": path, "repo": repo, "mtime": mtime},
|
||||
)
|
||||
conn.execute(
|
||||
"MATCH (l:Leaf {id:$lid}), (f:File {id:$fid}) "
|
||||
"MERGE (l)-[:FROM_FILE]->(f)",
|
||||
parameters={"lid": leaf_id, "fid": fid},
|
||||
)
|
||||
return fid
|
||||
|
||||
|
||||
HOP_STMTS = {
|
||||
1: "MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File) RETURN f.id, f.path, 1",
|
||||
2: ("MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File)-[:HAS_VERSION]->(c:Commit) "
|
||||
"RETURN c.id, c.subject, 2"),
|
||||
3: ("MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File)-[:HAS_VERSION]->(c:Commit)"
|
||||
"-[:AUTHORED]->(p:Person) RETURN p.id, p.name, 3"),
|
||||
}
|
||||
HOP_LABELS = {1: "File", 2: "Commit", 3: "Person"}
|
||||
|
||||
|
||||
def hop_walk(conn: ladybug.Connection, leaf_id: str, n: int) -> list[dict]:
|
||||
"""Walk Leaf → File → Commit → Person up to n hops (max 3)."""
|
||||
depth = min(max(int(n), 0), 3)
|
||||
out: list[dict] = []
|
||||
for d in range(1, depth + 1):
|
||||
rows = conn.execute(HOP_STMTS[d], parameters={"id": leaf_id}).get_all()
|
||||
for row in rows:
|
||||
out.append({
|
||||
"id": row[0],
|
||||
"label": HOP_LABELS[d],
|
||||
"name": row[1],
|
||||
"depth": int(row[2]),
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
def leaf_index_names(conn: ladybug.Connection) -> set[str]:
|
||||
"""Return index names on the Leaf table (e.g. {'id', 'Leaf_vec', '_PK'})."""
|
||||
rows = conn.execute("CALL SHOW_INDEXES() RETURN *").get_all()
|
||||
|
||||
+84
-1
@@ -7,7 +7,10 @@ offline against fixtures.
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import tempfile
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
@@ -18,9 +21,10 @@ OFFICE_SUFFIXES = {".docx", ".pptx", ".xlsx", ".html", ".htm", ".epub", ".eml",
|
||||
PDF_SUFFIXES = {".pdf"}
|
||||
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".tiff", ".tif", ".webp"}
|
||||
ARCHIVE_SUFFIXES = {".zip"}
|
||||
# Legacy binary Office (doc/xls/ppt) — markitdown/docling skip them; we try
|
||||
# Legacy binary Office (doc/xls/ppt) — markitdown skip them; we try
|
||||
# pandoc first, else leave a stub.
|
||||
LEGACY_OFFICE_SUFFIXES = {".doc", ".xls", ".ppt"}
|
||||
TESS_LANG = "eng+deu"
|
||||
|
||||
CONVERTIBLE_SUFFIXES = (
|
||||
TEXT_SUFFIXES | OFFICE_SUFFIXES | PDF_SUFFIXES | IMAGE_SUFFIXES | ARCHIVE_SUFFIXES | LEGACY_OFFICE_SUFFIXES
|
||||
@@ -146,3 +150,82 @@ def zip_extract_safe(zip_path: Path, dest: Path) -> list[Path]:
|
||||
|
||||
def is_convertible(suffix: str) -> bool:
|
||||
return suffix.lower() in CONVERTIBLE_SUFFIXES
|
||||
|
||||
|
||||
def convert_pdf(path: Path, ocr: bool = False) -> str:
|
||||
"""pdftotext -layout first; empty text layer → pdftoppm + tesseract.
|
||||
|
||||
`ocr` is unused for born-digital PDFs (text layer wins). Scans OCR
|
||||
automatically. This path never execs an ONNX document converter.
|
||||
"""
|
||||
del ocr # scans OCR when the text layer is empty; flag is for images
|
||||
text = pdf_fast_text(path)
|
||||
if text and text.strip():
|
||||
return normalize_markdown(text)
|
||||
scanned = ocr_pdf(path)
|
||||
if scanned and scanned.strip():
|
||||
return normalize_markdown(scanned)
|
||||
if text:
|
||||
return normalize_markdown(text)
|
||||
return "\n<!-- pdf has no text layer (ocr unavailable) -->\n"
|
||||
|
||||
|
||||
def pdf_fast_text(path: Path) -> str | None:
|
||||
"""pdftotext -layout; None when poppler is missing or the command fails."""
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["pdftotext", "-layout", str(path), "-"],
|
||||
capture_output=True, timeout=60)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return None
|
||||
if proc.returncode != 0:
|
||||
return None
|
||||
return proc.stdout.decode("utf-8", errors="replace")
|
||||
|
||||
|
||||
def ocr_pdf(path: Path) -> str:
|
||||
"""Rasterize with pdftoppm and OCR each page (tesseract or paddle)."""
|
||||
try:
|
||||
with tempfile.TemporaryDirectory(prefix="2dph-ocr-") as tmp:
|
||||
prefix = str(Path(tmp) / "page")
|
||||
proc = subprocess.run(
|
||||
["pdftoppm", "-png", "-r", "200", str(path), prefix],
|
||||
capture_output=True, timeout=120)
|
||||
if proc.returncode != 0:
|
||||
return ""
|
||||
pages = sorted(Path(tmp).glob("page*.png"))
|
||||
parts = [ocr_image(p) for p in pages]
|
||||
return "\n\n".join(p for p in parts if p and p.strip())
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return ""
|
||||
|
||||
|
||||
def ocr_image(path: Path) -> str:
|
||||
engine = os.environ.get("OCR_ENGINE", "tesseract")
|
||||
if engine == "paddle":
|
||||
return _ocr_paddle(path)
|
||||
return _ocr_tesseract(path)
|
||||
|
||||
|
||||
def _ocr_tesseract(path: Path) -> str:
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["tesseract", str(path), "stdout", "-l", TESS_LANG, "--psm", "6"],
|
||||
capture_output=True, timeout=120)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return ""
|
||||
if proc.returncode != 0:
|
||||
return ""
|
||||
return proc.stdout.decode("utf-8", errors="replace").strip()
|
||||
|
||||
|
||||
def _ocr_paddle(path: Path) -> str:
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["paddleocr", "ocr", "-i", str(path)],
|
||||
capture_output=True, timeout=180)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return ""
|
||||
if proc.returncode != 0:
|
||||
return ""
|
||||
return proc.stdout.decode("utf-8", errors="replace").strip()
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
"""D14 layout: bin/{subject}/{method}.go, libs in internal/, one go.mod."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
@@ -74,9 +75,58 @@ class BinLayoutTest(unittest.TestCase):
|
||||
)
|
||||
|
||||
def test_brain_methods_are_shebangs(self) -> None:
|
||||
for method in ("index.go", "get.go", "stats.go", "eval.go", "watch.go"):
|
||||
for method in ("index.go", "add.go", "get.go", "stats.go", "eval.go", "watch.go"):
|
||||
self._assert_shebang(f"bin/brain/{method}")
|
||||
|
||||
def test_brain_add_is_python_write_not_rebuild(self) -> None:
|
||||
self._assert_shebang("bin/brain/add.go")
|
||||
text = (ROOT / "bin" / "brain" / "add.go").read_text()
|
||||
self.assertIn("cmdbin.ExecFile", text)
|
||||
self.assertIn("bin/kb/add", text)
|
||||
self.assertNotIn("--rebuild", text)
|
||||
py = (ROOT / "bin" / "kb" / "add").read_text()
|
||||
self.assertIn("add_leafs", py)
|
||||
self.assertIn("--json", py)
|
||||
self.assertNotIn("unlink", py.lower())
|
||||
|
||||
def test_brain_get_stats_eval_are_not_python_exec(self) -> None:
|
||||
for method in ("get.go", "stats.go", "eval.go"):
|
||||
text = (ROOT / "bin" / "brain" / method).read_text()
|
||||
self.assertNotIn(
|
||||
"ExecFile",
|
||||
text,
|
||||
f"bin/brain/{method} must call internal/brain, not ExecFile Python",
|
||||
)
|
||||
self.assertNotIn(
|
||||
"cmdbin",
|
||||
text,
|
||||
f"bin/brain/{method} must not import internal/cmdbin",
|
||||
)
|
||||
self.assertIn(
|
||||
"system_ladybug",
|
||||
text.splitlines()[0],
|
||||
f"bin/brain/{method} shebang must pass -tags=system_ladybug",
|
||||
)
|
||||
self.assertIn(
|
||||
"github.com/eSlider/2dph/internal/brain",
|
||||
text,
|
||||
)
|
||||
|
||||
def test_eval_control_questions_live_in_rank(self) -> None:
|
||||
rank = (ROOT / "internal" / "brain" / "rank" / "evalq.go").read_text()
|
||||
py = (ROOT / "bin" / "kb" / "eval").read_text()
|
||||
for frag in ("BM25", "DevOps", "LadybugDB"):
|
||||
self.assertIn(frag, rank)
|
||||
self.assertIn(frag, py)
|
||||
self.assertIn("0.95", rank)
|
||||
|
||||
def test_facts_methods_are_shebangs(self) -> None:
|
||||
for method in ("audit.go", "extract.go", "crm.go"):
|
||||
self._assert_shebang(f"bin/facts/{method}")
|
||||
text = (ROOT / "bin" / "facts" / method).read_text()
|
||||
self.assertIn("cmdbin.ExecFile", text)
|
||||
self.assertIn(f"bin/facts/{method.removesuffix('.go')}", text)
|
||||
|
||||
def test_mail_import_is_shebang_not_brain_write(self) -> None:
|
||||
self._assert_shebang("bin/mail/import.go")
|
||||
index_mail = (ROOT / "bin" / "mail" / "index_mail").read_text()
|
||||
@@ -86,8 +136,149 @@ class BinLayoutTest(unittest.TestCase):
|
||||
"index_mail must point at bin/brain/index.go",
|
||||
)
|
||||
|
||||
def test_markdown_import_is_shebang(self) -> None:
|
||||
def test_mail_ocr_is_tesseract_not_docling(self) -> None:
|
||||
self._assert_shebang("bin/mail/ocr.go")
|
||||
ocr = (ROOT / "bin" / "mail" / "ocr.go").read_text()
|
||||
self.assertIn("internal/ocr", ocr)
|
||||
self.assertIn("mail_ocr", ocr)
|
||||
self.assertNotIn("github.com/otiai10/gosseract", ocr)
|
||||
py = (ROOT / "bin" / "mail" / "import").read_text()
|
||||
self.assertNotIn("from docling", py)
|
||||
self.assertNotIn("import docling", py)
|
||||
self.assertIn("convert_pdf", py)
|
||||
conv = (ROOT / "bin" / "tools" / "mailconv.py").read_text()
|
||||
self.assertIn("pdftotext", conv)
|
||||
self.assertIn("pdftoppm", conv)
|
||||
self.assertIn("tesseract", conv)
|
||||
self.assertIn("eng+deu", conv)
|
||||
self.assertNotIn("from docling", conv)
|
||||
self.assertNotIn("import docling", conv)
|
||||
self.assertNotIn("gocv", conv.lower())
|
||||
proj = (ROOT / "pyproject.toml").read_text()
|
||||
self.assertNotIn("docling", proj)
|
||||
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
|
||||
self.assertIn("tesseract-ocr", ci)
|
||||
self.assertIn("./internal/ocr", ci)
|
||||
compose = (ROOT / "compose.yaml").read_text()
|
||||
self.assertIn("ocr-paddle", compose)
|
||||
self.assertIn("OCR_ENGINE", compose)
|
||||
|
||||
def test_markdown_import_is_go_not_python_exec(self) -> None:
|
||||
self._assert_shebang("bin/markdown/import.go")
|
||||
text = (ROOT / "bin" / "markdown" / "import.go").read_text()
|
||||
self.assertNotIn("ExecFile", text)
|
||||
self.assertNotIn("cmdbin", text)
|
||||
self.assertIn("internal/mdleaves", text)
|
||||
self.assertNotIn("kb.lbug", text)
|
||||
|
||||
def test_import_adapters_do_not_write_ladybug(self) -> None:
|
||||
for rel in (
|
||||
"bin/mail/import.go",
|
||||
"bin/mail/import",
|
||||
"bin/markdown/import.go",
|
||||
"bin/chats/import.go",
|
||||
"bin/git/import.go",
|
||||
):
|
||||
text = (ROOT / rel).read_text()
|
||||
self.assertNotIn("upsert_leaf", text, rel)
|
||||
self.assertNotIn("kb.lbug", text, rel)
|
||||
self.assertNotIn("var/brain.lbug", text, rel)
|
||||
index = (ROOT / "bin" / "brain" / "index.go").read_text()
|
||||
self.assertIn("bin/kb/index", index)
|
||||
|
||||
def test_postgres_query_is_shebang(self) -> None:
|
||||
self._assert_shebang("bin/postgres/query.go")
|
||||
|
||||
def test_git_import_is_gogit_shebang(self) -> None:
|
||||
self._assert_shebang("bin/git/import.go")
|
||||
py = (ROOT / "bin" / "git" / "import").read_text()
|
||||
self.assertNotIn(
|
||||
'["git"',
|
||||
py,
|
||||
"Python git/import must not subprocess the git binary",
|
||||
)
|
||||
self.assertIn("bin/git/import.go", py)
|
||||
|
||||
def test_web_search_is_shebang(self) -> None:
|
||||
self._assert_shebang("bin/web/search.go")
|
||||
py = (ROOT / "bin" / "web" / "search").read_text()
|
||||
self.assertIn("bin/web/search.go", py)
|
||||
|
||||
def test_gitimport_py_has_no_git_binary(self) -> None:
|
||||
py = (ROOT / "bin" / "tools" / "gitimport.py").read_text()
|
||||
self.assertNotIn("subprocess", py)
|
||||
self.assertNotIn("git log", py)
|
||||
|
||||
def test_gogit_is_direct_go_mod_require(self) -> None:
|
||||
text = (ROOT / "go.mod").read_text()
|
||||
first = text.split("require (")[1].split(")")[0]
|
||||
self.assertRegex(first, r"github.com/go-git/go-git/v5\s+v")
|
||||
for line in first.splitlines():
|
||||
if "go-git/go-git" in line:
|
||||
self.assertNotIn("indirect", line)
|
||||
|
||||
def test_duckdb_go_is_direct_require(self) -> None:
|
||||
text = (ROOT / "go.mod").read_text()
|
||||
first = text.split("require (")[1].split(")")[0]
|
||||
self.assertRegex(first, r"github.com/duckdb/duckdb-go/v2\s+v")
|
||||
for line in first.splitlines():
|
||||
if "duckdb/duckdb-go" in line:
|
||||
self.assertNotIn("indirect", line)
|
||||
skill = (ROOT / "skills" / "duckdb" / "SKILL.md").read_text()
|
||||
self.assertIn("github.com/duckdb/duckdb-go", skill)
|
||||
self.assertIn("Ladybug", skill)
|
||||
self.assertIn("sqlite", skill.lower())
|
||||
self.assertIn("gcc", skill.lower())
|
||||
self.assertIn("Zig", skill)
|
||||
plan = (ROOT / "PLAN.md").read_text()
|
||||
self.assertIn("D22", plan)
|
||||
self.assertIn("duckdb-go", plan)
|
||||
self._assert_shebang("bin/qa/stats.go")
|
||||
reasoner = (ROOT / "internal" / "reasoner" / "client.go").read_text()
|
||||
self.assertNotIn("duckdb", reasoner)
|
||||
self.assertNotIn("duckstats", reasoner)
|
||||
bakeoff = (ROOT / "bin" / "reasoner" / "bakeoff.go").read_text()
|
||||
self.assertIn("internal/duckstats", bakeoff)
|
||||
webcache = (ROOT / "internal" / "websearch" / "cache.go").read_text()
|
||||
self.assertNotIn("duckdb", webcache)
|
||||
self.assertIn("modernc.org/sqlite", webcache)
|
||||
|
||||
def test_cgo_uses_zig_not_gcc(self) -> None:
|
||||
for rel in ("bin/cgo/zig", "bin/cgo/zcc", "bin/cgo/zc++"):
|
||||
p = ROOT / rel
|
||||
self.assertTrue(p.is_file(), f"missing {rel}")
|
||||
self.assertTrue(
|
||||
os.access(p, os.X_OK),
|
||||
f"{rel} must be executable",
|
||||
)
|
||||
zig = (ROOT / "bin" / "cgo" / "zig").read_text()
|
||||
self.assertIn("zig cc", zig)
|
||||
self.assertIn("0.14.1", zig)
|
||||
zcc = (ROOT / "bin" / "cgo" / "zcc").read_text()
|
||||
self.assertIn('exec "$ZIG" cc', zcc)
|
||||
self.assertNotIn("command -v gcc", zcc)
|
||||
search = (ROOT / "bin" / "kb" / "search").read_text()
|
||||
self.assertIn("bin/cgo/zig", search)
|
||||
self.assertNotIn("command -v gcc", search)
|
||||
|
||||
def test_ci_recall_sot_is_zig_brain_eval(self) -> None:
|
||||
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
|
||||
self.assertIn("bin/brain/eval.go", ci)
|
||||
self.assertIn("system_ladybug,brain_eval", ci)
|
||||
self.assertIn("/tmp/brain-eval", ci)
|
||||
self.assertIn("KB_ROOT", ci)
|
||||
self.assertNotIn("bin/kb/eval", ci)
|
||||
self.assertNotIn("gate skipped", ci)
|
||||
self.assertIn("./bin/facts/audit self", ci)
|
||||
|
||||
def test_eval_fragments_live_in_default_corpus(self) -> None:
|
||||
"""CI --rebuild indexes README/PLAN/docs/skills; fragments must be there."""
|
||||
corpus = []
|
||||
for rel in ("README.md", "PLAN.md", "AGENTS.md"):
|
||||
corpus.append((ROOT / rel).read_text())
|
||||
for d in ("docs", "skills"):
|
||||
for p in (ROOT / d).rglob("*.md"):
|
||||
corpus.append(p.read_text())
|
||||
blob = "\n".join(corpus)
|
||||
for frag in ("BM25", "DevOps", "LadybugDB"):
|
||||
self.assertIn(frag, blob, f"{frag} must appear in default index corpus")
|
||||
|
||||
+13
-10
@@ -9,12 +9,6 @@ sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
import kblib # noqa: E402
|
||||
import gitimport # noqa: E402
|
||||
|
||||
SAMPLE = (
|
||||
"\x1e" + "a1b2c3d" + "\x1f" + "Ada Lovelace" + "\x1f" + "ada@example.com"
|
||||
+ "\x1f" + "2026-08-10T12:00:00+01:00" + "\x1f" + "feat: first commit"
|
||||
+ "\n\nREADME.md\nsrc/main.c\n"
|
||||
)
|
||||
|
||||
COMMIT_PERSON_SCHEMA = (
|
||||
"CREATE NODE TABLE IF NOT EXISTS Commit (id STRING, repo STRING, subject STRING, "
|
||||
"author STRING, email STRING, date STRING, PRIMARY KEY(id))"
|
||||
@@ -26,6 +20,17 @@ HAS_VERSION_SCHEMA = "CREATE REL TABLE IF NOT EXISTS HAS_VERSION (FROM File TO C
|
||||
AUTHORED_SCHEMA = "CREATE REL TABLE IF NOT EXISTS AUTHORED (FROM Commit TO Person)"
|
||||
|
||||
|
||||
def sample_commit() -> gitimport.Commit:
|
||||
return gitimport.Commit(
|
||||
sha="a1b2c3d",
|
||||
author="Ada Lovelace",
|
||||
email="ada@example.com",
|
||||
date="2026-08-10T12:00:00+01:00",
|
||||
subject="feat: first commit",
|
||||
files=["README.md", "src/main.c"],
|
||||
)
|
||||
|
||||
|
||||
class GitGraphTest(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.dir = tempfile.mkdtemp()
|
||||
@@ -42,14 +47,12 @@ class GitGraphTest(unittest.TestCase):
|
||||
self.db.close()
|
||||
|
||||
def test_index_commits_creates_nodes_and_edges(self):
|
||||
cs = gitimport.parse_log(SAMPLE)
|
||||
gitimport.index_commits(self.conn, cs, "sample-repo")
|
||||
gitimport.index_commits(self.conn, [sample_commit()], "sample-repo")
|
||||
rp = self.conn.execute("MATCH (p:Person) RETURN p.name, p.email").get_all()
|
||||
self.assertEqual([tuple(r) for r in rp], [("Ada Lovelace", "ada@example.com")])
|
||||
rc = self.conn.execute("MATCH (c:Commit) RETURN c.id, c.repo").get_all()
|
||||
self.assertEqual(len(rc), 1)
|
||||
self.assertEqual(rc[0][1], "sample-repo")
|
||||
# File -[:HAS_VERSION]-> Commit -[:AUTHORED]-> Person
|
||||
rf = self.conn.execute(
|
||||
"MATCH (f:File)-[:HAS_VERSION]->(c:Commit)-[:AUTHORED]->(p:Person) "
|
||||
"RETURN f.path, c.id, p.email").get_all()
|
||||
@@ -58,7 +61,7 @@ class GitGraphTest(unittest.TestCase):
|
||||
self.assertTrue(all(r[2] == "ada@example.com" for r in rf))
|
||||
|
||||
def test_index_commits_idempotent(self):
|
||||
cs = gitimport.parse_log(SAMPLE)
|
||||
cs = [sample_commit()]
|
||||
gitimport.index_commits(self.conn, cs, "sample-repo")
|
||||
gitimport.index_commits(self.conn, cs, "sample-repo")
|
||||
n = self.conn.execute("MATCH (c:Commit) RETURN count(*)").get_all()[0][0]
|
||||
|
||||
@@ -1,57 +0,0 @@
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
|
||||
import gitimport # noqa: E402
|
||||
|
||||
SAMPLE = (
|
||||
"\x1e" + "a1b2c3d" + "\x1f" + "Ada Lovelace" + "\x1f" + "ada@example.com"
|
||||
+ "\x1f" + "2026-08-10T12:00:00+01:00" + "\x1f" + "feat: first commit"
|
||||
+ "\n\nREADME.md\nsrc/main.c\n"
|
||||
+ "\x1e" + "e4f5a6b" + "\x1f" + "Bob Babbage" + "\x1f" + "bob@example.com"
|
||||
+ "\x1f" + "2026-08-11T09:30:00+01:00" + "\x1f" + "fix: typo"
|
||||
+ "\n\ndocs/notes.md"
|
||||
)
|
||||
|
||||
|
||||
class GitparseTest(unittest.TestCase):
|
||||
def test_parses_records(self):
|
||||
cs = gitimport.parse_log(SAMPLE)
|
||||
self.assertEqual(len(cs), 2)
|
||||
|
||||
def test_parses_commit_fields(self):
|
||||
cs = gitimport.parse_log(SAMPLE)
|
||||
c = cs[0]
|
||||
self.assertEqual(c.sha, "a1b2c3d")
|
||||
self.assertEqual(c.author, "Ada Lovelace")
|
||||
self.assertEqual(c.email, "ada@example.com")
|
||||
self.assertEqual(c.date, "2026-08-10T12:00:00+01:00")
|
||||
self.assertEqual(c.subject, "feat: first commit")
|
||||
|
||||
def test_parses_changed_files(self):
|
||||
cs = gitimport.parse_log(SAMPLE)
|
||||
self.assertEqual(cs[0].files, ["README.md", "src/main.c"])
|
||||
self.assertEqual(cs[1].files, ["docs/notes.md"])
|
||||
|
||||
def test_ignores_empty(self):
|
||||
self.assertEqual(gitimport.parse_log(""), [])
|
||||
|
||||
def test_skip_malformed_record(self):
|
||||
self.assertEqual(gitimport.parse_log("\x1eweird\x1e"), [])
|
||||
|
||||
def test_commit_leaf_shape(self):
|
||||
leafs = gitimport.commits_to_leafs(gitimport.parse_log(SAMPLE), "sample-repo")
|
||||
self.assertEqual(len(leafs), 2)
|
||||
lf = leafs[0]
|
||||
self.assertEqual(lf["type"], "commit")
|
||||
self.assertEqual(lf["repo"], "sample-repo")
|
||||
self.assertEqual(lf["source"], "sample-repo@a1b2c3d")
|
||||
self.assertIn("Ada Lovelace", lf["text"])
|
||||
self.assertIn("README.md", lf["related"])
|
||||
self.assertIn("feat: first commit", lf["heading"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,97 @@
|
||||
"""Import adapters write files only. Index rebuild is brain/index (D14 / Gitea #7)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
class IndexAdapterTest(unittest.TestCase):
|
||||
def test_dry_run_fixture_corpus_does_not_write_lbug(self) -> None:
|
||||
tmp = Path(tempfile.mkdtemp())
|
||||
(tmp / "note.md").write_text("# Fixture\n\n## Leaf\n\nhello corpus\n", encoding="utf-8")
|
||||
lbug = tmp / "kb.lbug"
|
||||
try:
|
||||
import ladybug # noqa: F401
|
||||
except ImportError:
|
||||
venv_py = ROOT / ".venv" / "bin" / "python"
|
||||
if not venv_py.is_file():
|
||||
self.skipTest("ladybug missing")
|
||||
py = str(venv_py)
|
||||
else:
|
||||
py = sys.executable
|
||||
proc = subprocess.run(
|
||||
[py, str(ROOT / "bin" / "kb" / "index"), "--dry-run", "--json", "--corpus", str(tmp)],
|
||||
cwd=ROOT,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
env=os.environ.copy(),
|
||||
check=False,
|
||||
)
|
||||
self.assertEqual(proc.returncode, 0, proc.stderr)
|
||||
msg = json.loads(proc.stdout)
|
||||
self.assertTrue(msg.get("dry_run"))
|
||||
self.assertGreaterEqual(msg.get("corpus_total", 0), 1)
|
||||
self.assertFalse(lbug.exists(), "dry-run must not create a Ladybug file")
|
||||
|
||||
def test_facts_json_and_chats_land_on_rebuild(self) -> None:
|
||||
"""Gitea #18: facts (2-source) + chats markdown become leafs on rebuild."""
|
||||
tmp = Path(tempfile.mkdtemp())
|
||||
dbpath = tmp / "kb.lbug"
|
||||
chats = tmp / "chats"
|
||||
chats.mkdir()
|
||||
(chats / "alice.md").write_text(
|
||||
"# Chat\n\n## Alice and Bob\n\nhello from chats fixture unique-chat-token\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
facts_path = tmp / "facts.json"
|
||||
facts_path.write_text(json.dumps([{
|
||||
"text": "container 'brain' unique-fact-token is running and declared in compose.yaml",
|
||||
"source": "docker ps x compose.yaml",
|
||||
"loc": "compose.yaml:brain",
|
||||
"how": "facts/extract",
|
||||
}]), encoding="utf-8")
|
||||
venv_py = ROOT / ".venv" / "bin" / "python"
|
||||
py = str(venv_py) if venv_py.is_file() else sys.executable
|
||||
proc = subprocess.run(
|
||||
[
|
||||
py, str(ROOT / "bin" / "kb" / "index"),
|
||||
"--rebuild", "--db", str(dbpath), "--no-defaults",
|
||||
"--with-chats", str(chats),
|
||||
"--facts-json", str(facts_path),
|
||||
"--json",
|
||||
],
|
||||
cwd=ROOT,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
env=os.environ.copy(),
|
||||
check=False,
|
||||
)
|
||||
self.assertEqual(proc.returncode, 0, proc.stderr)
|
||||
msg = json.loads(proc.stdout)
|
||||
self.assertGreaterEqual(msg.get("facts_leafs", 0), 1)
|
||||
self.assertGreaterEqual(msg.get("chat_leafs", 0), 1)
|
||||
self.assertTrue(dbpath.exists())
|
||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
||||
import kblib
|
||||
db, conn = kblib.connect(dbpath, read_only=True)
|
||||
try:
|
||||
stats = kblib.stats(conn)
|
||||
self.assertGreaterEqual(stats["by_root"].get("facts", 0), 1)
|
||||
fts = kblib.query_fts(conn, "unique-chat-token", 5)
|
||||
self.assertTrue(fts, "chats markdown must be FTS-searchable")
|
||||
fact_hits = kblib.query_fts(conn, "unique-fact-token", 5)
|
||||
self.assertTrue(any(h.get("root") == "facts" for h in fact_hits))
|
||||
src = conn.execute(
|
||||
"MATCH (l:Leaf {root:'facts'}) RETURN l.source"
|
||||
).get_all()
|
||||
self.assertTrue(any(" x " in str(r[0]) for r in src))
|
||||
finally:
|
||||
conn.close()
|
||||
db.close()
|
||||
@@ -0,0 +1,69 @@
|
||||
"""Incremental add writes leafs without deleting kb.lbug."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
class KbAddCLITest(unittest.TestCase):
|
||||
def test_json_add_does_not_delete_db(self) -> None:
|
||||
tmp = Path(tempfile.mkdtemp())
|
||||
dbpath = tmp / "kb.lbug"
|
||||
py = sys.executable
|
||||
venv_py = ROOT / ".venv" / "bin" / "python"
|
||||
if venv_py.is_file():
|
||||
py = str(venv_py)
|
||||
payload = {
|
||||
"text": "cli zebra leaf",
|
||||
"root": "info",
|
||||
"source": "cli-test",
|
||||
"confidence": "confirmed",
|
||||
"how": "test",
|
||||
"loc": str(tmp),
|
||||
"type": "reference",
|
||||
"embedding": [0.0] * 256,
|
||||
}
|
||||
payload["embedding"][0] = 0.3
|
||||
proc = subprocess.run(
|
||||
[py, str(ROOT / "bin" / "kb" / "add"), "--db", str(dbpath), "--json"],
|
||||
cwd=ROOT,
|
||||
input=json.dumps(payload),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
env=os.environ.copy(),
|
||||
check=False,
|
||||
)
|
||||
self.assertEqual(proc.returncode, 0, proc.stderr)
|
||||
self.assertTrue(dbpath.exists(), "add must create the db, not skip write")
|
||||
out = json.loads(proc.stdout)
|
||||
self.assertEqual(out.get("mode"), "add")
|
||||
self.assertEqual(len(out.get("ids") or []), 1)
|
||||
again = subprocess.run(
|
||||
[py, str(ROOT / "bin" / "kb" / "add"), "--db", str(dbpath), "--json"],
|
||||
cwd=ROOT,
|
||||
input=json.dumps({
|
||||
**payload,
|
||||
"text": "second moose leaf",
|
||||
"source": "cli-test-2",
|
||||
}),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
env=os.environ.copy(),
|
||||
check=False,
|
||||
)
|
||||
self.assertEqual(again.returncode, 0, again.stderr)
|
||||
self.assertTrue(dbpath.exists())
|
||||
second = json.loads(again.stdout)
|
||||
self.assertEqual(len(second.get("ids") or []), 1)
|
||||
self.assertNotEqual(out["ids"][0], second["ids"][0])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -71,6 +71,69 @@ class KblibTest(unittest.TestCase):
|
||||
self.assertTrue(hits)
|
||||
self.assertIn("Leaf_vec", kblib.leaf_index_names(self.conn))
|
||||
|
||||
def test_add_after_indexes_keeps_fts_queryable(self):
|
||||
"""Incremental add after FTS+HNSW must find the new leaf on both indexes."""
|
||||
kblib.upsert_leaf(self.conn, text="seed fox leaf", root="info",
|
||||
confidence="confirmed", source="s", source_rev="r1",
|
||||
how="test", loc="/tmp", type_="reference",
|
||||
embedding=make_emb(0.1))
|
||||
kblib.ensure_indexes(self.conn)
|
||||
ids = kblib.add_leafs(self.conn, [{
|
||||
"text": "added zebra after index",
|
||||
"root": "facts",
|
||||
"confidence": "confirmed",
|
||||
"source": "a.md x b.md",
|
||||
"source_rev": "r1",
|
||||
"how": "test",
|
||||
"loc": "/tmp",
|
||||
"type": "fact",
|
||||
"embedding": make_emb(0.9),
|
||||
}])
|
||||
self.assertEqual(len(ids), 1)
|
||||
fts = kblib.query_fts(self.conn, "zebra", 5)
|
||||
self.assertTrue(fts)
|
||||
self.assertIn("zebra", fts[0]["text"])
|
||||
self.assertEqual(fts[0]["root"], "facts")
|
||||
vec = kblib.query_vector(self.conn, make_emb(0.9), 5)
|
||||
self.assertTrue(any("zebra" in h["text"] for h in vec))
|
||||
fox = kblib.query_fts(self.conn, "fox", 5)
|
||||
self.assertTrue(fox)
|
||||
self.assertIn("fox", fox[0]["text"])
|
||||
|
||||
def test_add_facts_and_info_one_transaction(self):
|
||||
"""D12: facts and info land in the same transaction."""
|
||||
kblib.ensure_indexes(self.conn)
|
||||
ids = kblib.add_leafs(self.conn, [
|
||||
{
|
||||
"text": "tx fact leaf two-source",
|
||||
"root": "facts",
|
||||
"confidence": "confirmed",
|
||||
"source": "compose.yml x docker ps",
|
||||
"source_rev": "r1",
|
||||
"how": "test",
|
||||
"loc": "/tmp",
|
||||
"type": "fact",
|
||||
"embedding": make_emb(0.4),
|
||||
},
|
||||
{
|
||||
"text": "tx info narrative",
|
||||
"root": "info",
|
||||
"confidence": "confirmed",
|
||||
"source": "note.md",
|
||||
"source_rev": "r1",
|
||||
"how": "test",
|
||||
"loc": "/tmp",
|
||||
"type": "reference",
|
||||
"embedding": make_emb(0.5),
|
||||
},
|
||||
])
|
||||
self.assertEqual(len(ids), 2)
|
||||
stats = kblib.stats(self.conn)
|
||||
self.assertEqual(stats["by_root"].get("facts"), 1)
|
||||
self.assertEqual(stats["by_root"].get("info"), 1)
|
||||
self.assertTrue(kblib.query_fts(self.conn, "two-source", 5))
|
||||
self.assertTrue(kblib.query_fts(self.conn, "narrative", 5))
|
||||
|
||||
def test_drop_vector_then_create_raises_clear_error(self):
|
||||
"""DROP INDEX leaves ghost catalog; create_fts_and_vector must raise."""
|
||||
kblib.upsert_leaf(self.conn, text="seed", root="info",
|
||||
@@ -98,6 +161,39 @@ class KblibTest(unittest.TestCase):
|
||||
self.assertEqual(stats["total"], 2)
|
||||
self.assertEqual(stats["by_root"], {"facts": 1, "info": 1})
|
||||
|
||||
def test_hop_1_returns_file_hop_3_reaches_person(self):
|
||||
"""--hop walks FROM_FILE / HAS_VERSION / AUTHORED (Gitea #17)."""
|
||||
import gitimport
|
||||
|
||||
lid = kblib.upsert_leaf(
|
||||
self.conn, text="readme hop fixture", root="info",
|
||||
confidence="confirmed", source="README.md", source_rev="r1",
|
||||
how="test", loc="README.md", type_="reference",
|
||||
embedding=make_emb(0.3),
|
||||
)
|
||||
kblib.link_from_file(self.conn, lid, "README.md", repo="sample-repo")
|
||||
gitimport.index_commits(self.conn, [gitimport.Commit(
|
||||
sha="a1b2c3d",
|
||||
author="Ada Lovelace",
|
||||
email="ada@example.com",
|
||||
date="2026-08-10T12:00:00Z",
|
||||
subject="feat: first commit",
|
||||
files=["README.md"],
|
||||
)], "sample-repo")
|
||||
hop1 = kblib.hop_walk(self.conn, lid, 1)
|
||||
self.assertEqual(len(hop1), 1)
|
||||
self.assertEqual(hop1[0]["label"], "File")
|
||||
self.assertEqual(hop1[0]["name"], "README.md")
|
||||
self.assertEqual(hop1[0]["depth"], 1)
|
||||
hop3 = kblib.hop_walk(self.conn, lid, 3)
|
||||
labels = {n["label"] for n in hop3}
|
||||
self.assertIn("File", labels)
|
||||
self.assertIn("Commit", labels)
|
||||
self.assertIn("Person", labels)
|
||||
person = [n for n in hop3 if n["label"] == "Person"][0]
|
||||
self.assertEqual(person["name"], "Ada Lovelace")
|
||||
self.assertEqual(person["depth"], 3)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -8,10 +8,13 @@ from pathlib import Path
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
|
||||
from mailconv import ( # noqa: E402
|
||||
TESS_LANG,
|
||||
clean_email_address,
|
||||
convert_pdf,
|
||||
html_to_markdown,
|
||||
is_convertible,
|
||||
normalize_markdown,
|
||||
ocr_image,
|
||||
split_zip_members,
|
||||
subject_to_filename,
|
||||
zip_extract_safe,
|
||||
@@ -100,6 +103,92 @@ class TestMailConv(unittest.TestCase):
|
||||
self.assertFalse(is_convertible(".exe"))
|
||||
self.assertFalse(is_convertible(".unknown"))
|
||||
|
||||
def test_convert_pdf_prefers_pdftotext(self):
|
||||
import mailconv as mc
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
calls.append(list(cmd))
|
||||
|
||||
class P:
|
||||
returncode = 0
|
||||
stdout = b"Invoice BM25 layout"
|
||||
stderr = b""
|
||||
|
||||
return P()
|
||||
|
||||
self._patch_run(mc, fake_run)
|
||||
out = convert_pdf(Path(self._tmp("born.pdf")))
|
||||
self.assertIn("BM25", out)
|
||||
self.assertEqual(calls[0][:2], ["pdftotext", "-layout"])
|
||||
self.assertFalse(any(c[0] == "tesseract" for c in calls))
|
||||
self.assertFalse(any(c[0] == "pdftoppm" for c in calls))
|
||||
|
||||
def test_convert_pdf_empty_layer_uses_pdftoppm_tesseract(self):
|
||||
import mailconv as mc
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
calls.append(list(cmd))
|
||||
|
||||
class P:
|
||||
returncode = 0
|
||||
stdout = b""
|
||||
stderr = b""
|
||||
|
||||
if cmd[0] == "pdftotext":
|
||||
P.stdout = b" \n"
|
||||
return P()
|
||||
if cmd[0] == "pdftoppm":
|
||||
prefix = Path(cmd[-1])
|
||||
(prefix.parent / "page-1.png").write_bytes(b"fake")
|
||||
return P()
|
||||
if cmd[0] == "tesseract":
|
||||
P.stdout = b"scanned HELLO"
|
||||
return P()
|
||||
return P()
|
||||
|
||||
self._patch_run(mc, fake_run)
|
||||
out = convert_pdf(Path(self._tmp("scan.pdf")))
|
||||
self.assertIn("HELLO", out)
|
||||
bins = [c[0] for c in calls]
|
||||
self.assertIn("pdftotext", bins)
|
||||
self.assertIn("pdftoppm", bins)
|
||||
self.assertIn("tesseract", bins)
|
||||
tess = next(c for c in calls if c[0] == "tesseract")
|
||||
self.assertIn(TESS_LANG, tess)
|
||||
self.assertNotIn("docling", " ".join(bins))
|
||||
|
||||
def test_ocr_image_paddle_engine(self):
|
||||
import mailconv as mc
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
calls.append(list(cmd))
|
||||
|
||||
class P:
|
||||
returncode = 0
|
||||
stdout = b"paddle text"
|
||||
stderr = b""
|
||||
|
||||
return P()
|
||||
|
||||
self._patch_run(mc, fake_run)
|
||||
os.environ["OCR_ENGINE"] = "paddle"
|
||||
try:
|
||||
out = ocr_image(Path(self._tmp("x.png")))
|
||||
finally:
|
||||
os.environ.pop("OCR_ENGINE", None)
|
||||
self.assertEqual(out, "paddle text")
|
||||
self.assertEqual(calls[0][:2], ["paddleocr", "ocr"])
|
||||
|
||||
def _patch_run(self, mod, fn) -> None:
|
||||
self.addCleanup(setattr, mod.subprocess, "run", mod.subprocess.run)
|
||||
mod.subprocess.run = fn
|
||||
|
||||
def _mk_zip(self, members):
|
||||
zpath = Path(self._tmp("arc.zip"))
|
||||
with zipfile.ZipFile(zpath, "w") as zf:
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
"""Published docs must match live commands (Gitea SoT, brain/search, no fake --hop)."""
|
||||
"""Published docs must match live commands (Gitea SoT, brain/search)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
@@ -39,19 +38,151 @@ class PublishedDocsTest(unittest.TestCase):
|
||||
"mail index is a brain write; README must name bin/brain/index.go",
|
||||
)
|
||||
|
||||
def test_docs_do_not_claim_hop_walks(self) -> None:
|
||||
def test_readme_git_import_is_gogit(self) -> None:
|
||||
text = (ROOT / "README.md").read_text()
|
||||
self.assertIn("bin/git/import.go", text)
|
||||
self.assertIn("go-git", text)
|
||||
self.assertIn("D19", (ROOT / "PLAN.md").read_text())
|
||||
|
||||
def test_web_search_is_go_not_ops_host(self) -> None:
|
||||
readme = (ROOT / "README.md").read_text()
|
||||
self.assertIn("bin/web/search.go", readme)
|
||||
skill = (ROOT / "skills" / "web-search" / "SKILL.md").read_text()
|
||||
self.assertIn("bin/web/search.go", skill)
|
||||
self.assertNotIn("search.ops.io", skill)
|
||||
self.assertNotIn("search.ops.io", readme)
|
||||
compose = (ROOT / "compose.yaml").read_text()
|
||||
self.assertIn("searxng", compose)
|
||||
self.assertNotIn("search.ops.io", compose)
|
||||
settings = (ROOT / "deploy" / "searxng" / "settings.yml").read_text()
|
||||
self.assertNotIn("password", settings.lower())
|
||||
self.assertIn("json", settings)
|
||||
|
||||
def test_picoclaw_compose_profile_has_mcp_example(self) -> None:
|
||||
compose = (ROOT / "compose.yaml").read_text()
|
||||
self.assertIn('profiles: ["picoclaw"]', compose)
|
||||
self.assertIn("127.0.0.1:8630", compose)
|
||||
example = (ROOT / "deploy" / "picoclaw" / "mcp.json.example").read_text()
|
||||
self.assertIn("127.0.0.1:8630/mcp", example)
|
||||
self.assertNotIn("password", example.lower())
|
||||
self.assertNotIn("token", example.lower())
|
||||
docs = (ROOT / "docs" / "picoclaw.md").read_text()
|
||||
self.assertIn("search", docs)
|
||||
self.assertIn("throttled", docs)
|
||||
|
||||
def test_readme_read_path_is_go(self) -> None:
|
||||
plan = (ROOT / "PLAN.md").read_text()
|
||||
self.assertIn("get.go", plan)
|
||||
self.assertIn("CI fallback", plan)
|
||||
design = (ROOT / "docs" / "design.md").read_text()
|
||||
self.assertIn("internal/brain/rank", design)
|
||||
self.assertIn("They do not exec Python", design)
|
||||
|
||||
def test_openapi_mcp_from_same_handlers(self) -> None:
|
||||
plan = (ROOT / "PLAN.md").read_text()
|
||||
self.assertIn("D20", plan)
|
||||
self.assertIn("/openapi.json", (ROOT / "README.md").read_text())
|
||||
self.assertIn("/mcp", (ROOT / "README.md").read_text())
|
||||
skill = (ROOT / "skills" / "brain" / "SKILL.md").read_text()
|
||||
self.assertIn("/mcp", skill)
|
||||
self.assertFalse((ROOT / "skills" / "db-yaml").exists())
|
||||
self.assertTrue((ROOT / "skills" / "postgres" / "SKILL.md").is_file())
|
||||
|
||||
def test_cgo_zig_and_index_profile(self) -> None:
|
||||
plan = (ROOT / "PLAN.md").read_text()
|
||||
self.assertIn("D21", plan)
|
||||
self.assertIn("zig cc", plan)
|
||||
dockerfile = (ROOT / "Dockerfile").read_text()
|
||||
self.assertIn("bin/cgo/zcc", dockerfile)
|
||||
self.assertIn("FROM debian:bookworm-slim AS api", dockerfile)
|
||||
self.assertIn("FROM python:3.12-slim AS index", dockerfile)
|
||||
api = dockerfile[dockerfile.index("FROM debian:bookworm-slim AS api") :]
|
||||
self.assertNotIn("pip install", api)
|
||||
compose = (ROOT / "compose.yaml").read_text()
|
||||
self.assertIn('profiles: ["index"]', compose)
|
||||
self.assertIn("target: api", compose)
|
||||
|
||||
def test_reasoner_docs_name_real_hf_ids_cpu_sidecar(self) -> None:
|
||||
docs = (ROOT / "docs" / "reasoner.md").read_text()
|
||||
for hf in (
|
||||
"Qwen/Qwen3.5-9B",
|
||||
"Qwen/Qwen3.6-27B",
|
||||
"prism-ml/Bonsai-27B-gguf",
|
||||
):
|
||||
self.assertIn(hf, docs)
|
||||
self.assertIn("no official qwen3.6-9b", docs.lower())
|
||||
self.assertIn("OLLAMA_NUM_GPU", docs)
|
||||
self.assertIn("rss_mb", docs)
|
||||
self.assertIn("vram_mb", docs)
|
||||
self.assertIn("3/3", docs)
|
||||
self.assertIn("Do not claim 9B is better at tools", docs)
|
||||
self.assertNotIn("Qwen/Qwen3.6-9B", docs)
|
||||
plan = (ROOT / "PLAN.md").read_text()
|
||||
self.assertIn("D18", plan)
|
||||
self.assertIn("Qwen/Qwen3.5-9B", plan)
|
||||
compose = (ROOT / "compose.yaml").read_text()
|
||||
self.assertIn('"reasoner"', compose)
|
||||
self.assertIn("OLLAMA_NUM_GPU", compose)
|
||||
self.assertIn("127.0.0.1:11435", compose)
|
||||
dockerfile = (ROOT / "Dockerfile").read_text()
|
||||
self.assertNotIn(".gguf", dockerfile.lower())
|
||||
self.assertNotIn(".safetensors", dockerfile.lower())
|
||||
api = dockerfile[dockerfile.index("FROM debian:bookworm-slim AS api") :]
|
||||
self.assertNotIn("COPY models", api)
|
||||
self.assertNotIn("qwen", api.lower())
|
||||
|
||||
def test_readme_search_escalates_web(self) -> None:
|
||||
text = (ROOT / "README.md").read_text()
|
||||
self.assertIn("--no-web", text)
|
||||
self.assertIn("D17", (ROOT / "PLAN.md").read_text())
|
||||
skill = (ROOT / "skills" / "brain" / "SKILL.md").read_text()
|
||||
self.assertIn("`web` block", skill)
|
||||
|
||||
def test_docs_say_hop_walks_from_file(self) -> None:
|
||||
paths = [
|
||||
ROOT / "README.md",
|
||||
ROOT / "docs" / "design.md",
|
||||
ROOT / "skills" / "kb-search" / "SKILL.md",
|
||||
ROOT / "skills" / "diataxis-docs" / "SKILL.md",
|
||||
ROOT / "skills" / "brain" / "SKILL.md",
|
||||
ROOT / "docs" / "runbook.md",
|
||||
ROOT / "docs" / "README.md",
|
||||
]
|
||||
# Command-style `--hop 1` / `--hop N` plus follow/walk = the old lie.
|
||||
# Honest "not implemented" notes must not match.
|
||||
lie = re.compile(r"--hop (?:N|1).*(?:follow|walk)", re.I | re.S)
|
||||
for path in paths:
|
||||
text = path.read_text()
|
||||
self.assertIsNone(
|
||||
lie.search(text),
|
||||
f"{path.relative_to(ROOT)} still claims --hop walks the graph",
|
||||
self.assertIn("--hop", text, f"{path.relative_to(ROOT)} must document --hop")
|
||||
self.assertNotIn(
|
||||
"not implemented",
|
||||
text.lower(),
|
||||
f"{path.relative_to(ROOT)} still says hop is not implemented",
|
||||
)
|
||||
|
||||
def test_docs_are_portable_diataxis(self) -> None:
|
||||
index = (ROOT / "docs" / "README.md").read_text()
|
||||
self.assertIn("type: reference", index)
|
||||
for d in ("D3", "D6", "D14", "D15", "D17", "D18"):
|
||||
self.assertIn(d, index)
|
||||
runbook = (ROOT / "docs" / "runbook.md").read_text()
|
||||
self.assertIn("type: howto", runbook)
|
||||
self.assertIn("bin/brain/search.go", runbook)
|
||||
self.assertIn("bin/brain/index.go", runbook)
|
||||
self.assertNotIn("search.ops.io", runbook)
|
||||
self.assertNotIn("/mnt/", runbook)
|
||||
self.assertNotIn("/home/", runbook)
|
||||
readme = (ROOT / "README.md").read_text()
|
||||
self.assertIn("docs/runbook.md", readme)
|
||||
self.assertNotIn("search.ops.io", readme)
|
||||
|
||||
def test_v1_epic_is_named_in_docs(self) -> None:
|
||||
plan = (ROOT / "PLAN.md").read_text()
|
||||
self.assertIn("Gap to v1", plan)
|
||||
self.assertIn("eSlider/2dph/issues/16", plan)
|
||||
self.assertIn("eSlider/2dph/issues/17", plan)
|
||||
self.assertIn("eSlider/2dph/milestone/12", plan)
|
||||
road = (ROOT / "docs" / "roadmap.md").read_text()
|
||||
self.assertIn("type: explanation", road)
|
||||
self.assertIn("issues/16", road)
|
||||
self.assertIn("issues/14", road)
|
||||
index = (ROOT / "docs" / "README.md").read_text()
|
||||
self.assertIn("roadmap.md", index)
|
||||
self.assertIn("epic #16", index)
|
||||
agents = (ROOT / "AGENTS.md").read_text()
|
||||
self.assertIn("roadmap.md", agents)
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
"""Skills must name live commands; every bin/ path in SKILL.md must exist."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
BIN_PATH = re.compile(r"(bin/[A-Za-z0-9_./-]+)")
|
||||
|
||||
|
||||
class SkillsTest(unittest.TestCase):
|
||||
def test_db_yaml_renamed_to_postgres(self) -> None:
|
||||
self.assertFalse(
|
||||
(ROOT / "skills" / "db-yaml").exists(),
|
||||
"skills/db-yaml must be skills/postgres",
|
||||
)
|
||||
self.assertTrue((ROOT / "skills" / "postgres" / "SKILL.md").is_file())
|
||||
text = (ROOT / "skills" / "postgres" / "SKILL.md").read_text()
|
||||
self.assertIn("bin/postgres/query.go", text)
|
||||
self.assertNotIn("search.ops.io", text)
|
||||
|
||||
def test_every_bin_path_in_skills_exists(self) -> None:
|
||||
missing: list[str] = []
|
||||
for path in (ROOT / "skills").rglob("SKILL.md"):
|
||||
text = path.read_text()
|
||||
for m in BIN_PATH.finditer(text):
|
||||
rel = m.group(1).rstrip(")`.,;")
|
||||
candidate = ROOT / rel
|
||||
if not candidate.exists():
|
||||
missing.append(f"{path.relative_to(ROOT)}: {rel}")
|
||||
self.assertEqual(missing, [], "skill bin paths must exist")
|
||||
|
||||
def test_brain_skill_lists_generated_tools(self) -> None:
|
||||
tools = (ROOT / "skills" / "brain" / "tools.md").read_text()
|
||||
skill = (ROOT / "skills" / "brain" / "SKILL.md").read_text()
|
||||
self.assertIn("tools.md", skill)
|
||||
for name in ("search", "get", "stats", "audit"):
|
||||
self.assertIn(f"`{name}`", tools)
|
||||
|
||||
def test_picoclaw_lists_tool_order(self) -> None:
|
||||
skill = (ROOT / "skills" / "picoclaw" / "SKILL.md").read_text()
|
||||
agents = (ROOT / "AGENTS.md").read_text()
|
||||
self.assertIn("**`search`**", skill)
|
||||
self.assertIn("**`get`**", skill)
|
||||
self.assertIn("**`audit`**", skill)
|
||||
self.assertIn("throttled", skill.lower())
|
||||
self.assertIn("not a negative finding", agents)
|
||||
self.assertIn("Fact-check every", agents)
|
||||
|
||||
def test_yq_is_mikefarah_for_structured_data(self) -> None:
|
||||
skill = (ROOT / "skills" / "yq" / "SKILL.md").read_text()
|
||||
self.assertIn("https://github.com/mikefarah/yq", skill)
|
||||
for fmt in ("YAML", "JSON", "XML", "CSV", "TOML", "HCL"):
|
||||
self.assertIn(fmt, skill)
|
||||
self.assertIn("not kislyuk", skill.lower())
|
||||
plan = (ROOT / "PLAN.md").read_text()
|
||||
self.assertIn("mikefarah/yq", plan)
|
||||
agents = (ROOT / "AGENTS.md").read_text()
|
||||
self.assertIn("mikefarah/yq", agents)
|
||||
web = (ROOT / "skills" / "web-search" / "SKILL.md").read_text()
|
||||
self.assertIn("| yq ", web)
|
||||
self.assertNotIn("| jq ", web)
|
||||
@@ -0,0 +1,33 @@
|
||||
"""Every bin/ path named in skills/ must exist on disk."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
BIN_PATH = re.compile(r"\b(bin/[A-Za-z0-9_./-]+)")
|
||||
|
||||
|
||||
class SkillsBinPathsTest(unittest.TestCase):
|
||||
def test_agent_cost_skill_is_gone(self) -> None:
|
||||
self.assertFalse(
|
||||
(ROOT / "skills" / "agent-cost").exists(),
|
||||
"skills/agent-cost documents bin/agents/cost which does not exist",
|
||||
)
|
||||
|
||||
def test_brain_skill_replaces_kb_search(self) -> None:
|
||||
self.assertTrue((ROOT / "skills" / "brain" / "SKILL.md").is_file())
|
||||
self.assertFalse((ROOT / "skills" / "kb-search").exists())
|
||||
|
||||
def test_skill_bin_paths_exist(self) -> None:
|
||||
missing: list[str] = []
|
||||
for skill in sorted((ROOT / "skills").rglob("SKILL.md")):
|
||||
text = skill.read_text()
|
||||
for match in BIN_PATH.findall(text):
|
||||
rel = match.rstrip("`'.,")
|
||||
if rel.endswith(".go") or Path(rel).suffix == "" or Path(rel).suffix in {".go", ".py"}:
|
||||
p = ROOT / rel
|
||||
if not p.exists():
|
||||
missing.append(f"{skill.relative_to(ROOT)}: {rel}")
|
||||
self.assertEqual(missing, [], "SKILL.md names bin/ paths that do not exist")
|
||||
@@ -0,0 +1,36 @@
|
||||
"""qa/system_perf.py is an offline-gated system test (no live brain in CI)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
class SystemPerfScriptTest(unittest.TestCase):
|
||||
def test_script_compiles_and_is_read_only(self) -> None:
|
||||
path = ROOT / "qa" / "system_perf.py"
|
||||
src = path.read_text()
|
||||
compile(src, str(path), "exec")
|
||||
self.assertIn("--json", src)
|
||||
self.assertIn("qwen3.5:9b", src)
|
||||
self.assertIn("--picoclaw", src)
|
||||
self.assertIn("BRAIN_URL", src)
|
||||
self.assertIn("tools/list", src)
|
||||
self.assertIn("tools/call", src)
|
||||
self.assertIn("GATE_HEALTH_MS", src)
|
||||
self.assertIn("GATE_GET_P50_MS", src)
|
||||
self.assertNotIn("kb.lbug", src)
|
||||
self.assertNotIn("password", src.lower())
|
||||
self.assertNotIn("token", src.lower())
|
||||
|
||||
def test_script_does_not_write_ladybug(self) -> None:
|
||||
tree = ast.parse((ROOT / "qa" / "system_perf.py").read_text())
|
||||
writes = [
|
||||
n.func.attr
|
||||
for n in ast.walk(tree)
|
||||
if isinstance(n, ast.Call) and isinstance(n.func, ast.Attribute)
|
||||
and n.func.attr in {"write_text", "write_bytes", "dump"}
|
||||
]
|
||||
self.assertEqual(writes, [], f"system_perf must not write files: {writes}")
|
||||
+12
-136
@@ -1,150 +1,26 @@
|
||||
#!/usr/bin/env python3
|
||||
"""web/search - web search through the self-hosted SearXNG at search.ops.io.
|
||||
"""web/search — deprecated. Use bin/web/search.go (SearXNG, no Python client).
|
||||
|
||||
bin/web/search "LadybugDB vector search"
|
||||
bin/web/search "model2vec multilingual" --site github.com
|
||||
bin/web/search "uclancy" --category it -n 3 --json | jq -r '.results[].url'
|
||||
bin/web/search "sqlite-vec" --refresh # ignore the cached answer
|
||||
|
||||
This complements bin/kb/search: the knowledge base holds our own facts, this
|
||||
reaches the public web. Use it as the second, independent source that the
|
||||
detective method asks for.
|
||||
|
||||
Exit codes: 0 results, 2 refused as possible PII, 3 throttled (not "nothing
|
||||
found" - the instance answers 200 with an empty list when it throttles).
|
||||
bin/web/search.go QUERY [--json] [-n N] [--site HOST]
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import fcntl
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
TOOLS = Path(__file__).resolve().parents[1] / "tools"
|
||||
sys.path.insert(0, str(TOOLS))
|
||||
sys.path.insert(0, str(TOOLS / "web-search"))
|
||||
|
||||
import websearch as ws # noqa: E402
|
||||
from yamlout import to_yaml # noqa: E402
|
||||
|
||||
CONFIG = Path(os.environ.get("BRAIN_SEARCH_ENV", Path.home() / ".config/brain/search.env"))
|
||||
CACHE = Path(os.environ.get("BRAIN_SEARCH_CACHE", Path.home() / ".cache/brain/web-search.sqlite"))
|
||||
LOCK = CACHE.with_suffix(".lock")
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
def load_config() -> dict:
|
||||
if not CONFIG.exists():
|
||||
sys.exit(f"no credentials at {CONFIG} (mode 600, BRAIN_SEARCH_URL/USER/PASS)")
|
||||
conf = {}
|
||||
for line in CONFIG.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#") or "=" not in line:
|
||||
continue
|
||||
key, _, value = line.partition("=")
|
||||
conf[key.strip()] = value.strip().strip("\"'")
|
||||
missing = {"BRAIN_SEARCH_URL", "BRAIN_SEARCH_USER", "BRAIN_SEARCH_PASS"} - conf.keys()
|
||||
if missing:
|
||||
sys.exit(f"{CONFIG} is missing {', '.join(sorted(missing))}")
|
||||
return conf
|
||||
|
||||
|
||||
def fetch(conf: dict, query: str, params: dict, timeout: int) -> dict:
|
||||
args = {"q": query, "format": "json", **params}
|
||||
url = f"{conf['BRAIN_SEARCH_URL'].rstrip('/')}/search?{urllib.parse.urlencode(args)}"
|
||||
request = urllib.request.Request(url)
|
||||
token = f"{conf['BRAIN_SEARCH_USER']}:{conf['BRAIN_SEARCH_PASS']}".encode()
|
||||
import base64
|
||||
request.add_header("Authorization", "Basic " + base64.b64encode(token).decode())
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
return json.loads(response.read().decode())
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="web search via SearXNG")
|
||||
parser.add_argument("query")
|
||||
parser.add_argument("-n", "--limit", type=int, default=ws.DEFAULT_LIMIT)
|
||||
parser.add_argument("--site", help="restrict to one domain")
|
||||
parser.add_argument("--lang", help="language code, e.g. de")
|
||||
parser.add_argument("--fresh", choices=["day", "week", "month", "year"],
|
||||
help="time range")
|
||||
parser.add_argument("--category", help="SearXNG category, e.g. it, science, news")
|
||||
parser.add_argument("--engines", help="comma separated engine list")
|
||||
parser.add_argument("--json", action="store_true")
|
||||
parser.add_argument("--refresh", action="store_true", help="bypass the cache")
|
||||
parser.add_argument("--ttl", type=float, default=ws.CACHE_TTL)
|
||||
parser.add_argument("--timeout", type=int, default=25)
|
||||
parser.add_argument("--force", action="store_true",
|
||||
help="send even if the query looks like PII")
|
||||
args = parser.parse_args()
|
||||
|
||||
query = f"site:{args.site} {args.query}" if args.site else args.query
|
||||
|
||||
reason = ws.phi_reason(query)
|
||||
if reason and not args.force:
|
||||
print(f"refused: {reason}. This query would leave the host.", file=sys.stderr)
|
||||
print("Rephrase without identifiers, or pass --force if it is genuinely public.",
|
||||
file=sys.stderr)
|
||||
return 2
|
||||
|
||||
params = {}
|
||||
if args.lang:
|
||||
params["language"] = args.lang
|
||||
if args.fresh:
|
||||
params["time_range"] = args.fresh
|
||||
if args.category:
|
||||
params["categories"] = args.category
|
||||
if args.engines:
|
||||
params["engines"] = args.engines
|
||||
|
||||
key = ws.cache_key(query, params)
|
||||
conn = ws.open_cache(CACHE)
|
||||
|
||||
if not args.refresh:
|
||||
cached = ws.cache_get(conn, key, ttl=args.ttl)
|
||||
if cached is not None:
|
||||
out = ws.project(cached, limit=args.limit)
|
||||
out["cached"] = True
|
||||
sys.stdout.write(json.dumps(out, indent=2, ensure_ascii=False) + "\n"
|
||||
if args.json else to_yaml(out))
|
||||
return 0
|
||||
|
||||
conf = load_config()
|
||||
LOCK.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# One request at a time across every agent on this host: the instance
|
||||
# suspends engines for minutes when several of us ask at once.
|
||||
with open(LOCK, "w") as lock:
|
||||
fcntl.flock(lock, fcntl.LOCK_EX)
|
||||
|
||||
payload = None
|
||||
for attempt in range(1 + len(ws.RETRY_BACKOFF)):
|
||||
delay = ws.wait_for(ws.last_call(conn), time.time())
|
||||
if delay:
|
||||
time.sleep(delay)
|
||||
ws.mark_call(conn)
|
||||
try:
|
||||
payload = fetch(conf, query, params, args.timeout)
|
||||
except Exception as error: # noqa: BLE001 - report, do not crash
|
||||
print(f"request failed: {error}", file=sys.stderr)
|
||||
return 3
|
||||
if ws.classify(payload) == "ok":
|
||||
break
|
||||
if attempt < len(ws.RETRY_BACKOFF):
|
||||
time.sleep(ws.RETRY_BACKOFF[attempt])
|
||||
|
||||
if ws.classify(payload) == "ok":
|
||||
ws.cache_put(conn, key, payload)
|
||||
|
||||
out = ws.project(payload, limit=args.limit)
|
||||
sys.stdout.write(json.dumps(out, indent=2, ensure_ascii=False) + "\n"
|
||||
if args.json else to_yaml(out))
|
||||
return 0 if out["status"] == "ok" else 3
|
||||
def main(argv: list[str]) -> int:
|
||||
print(
|
||||
"bin/web/search is deprecated; use bin/web/search.go",
|
||||
file=sys.stderr,
|
||||
)
|
||||
target = ROOT / "bin" / "web" / "search.go"
|
||||
os.execvp("go", ["go", "run", str(target), *argv])
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
sys.exit(main(sys.argv[1:]))
|
||||
|
||||
Executable
+232
@@ -0,0 +1,232 @@
|
||||
//usr/bin/env go run "$0" "$@"; exit
|
||||
//
|
||||
// bin/web/search.go - SearXNG as the second independent source (D3).
|
||||
//
|
||||
// ./bin/web/search.go "LadybugDB vector search"
|
||||
// ./bin/web/search.go "model2vec" --category it --json
|
||||
// ./bin/web/search.go "postgres" --site github.com --fresh year
|
||||
//
|
||||
// Empty results mean throttled, not "nothing exists". Exit 2 = PII refuse, 3 = throttled.
|
||||
// Config: $BRAIN_SEARCH_ENV (default $HOME/.config/brain/search.env).
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"os"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"github.com/eSlider/2dph/internal/websearch"
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(run(os.Args[1:]))
|
||||
}
|
||||
|
||||
func run(args []string) int {
|
||||
var (
|
||||
query, site, lang, fresh, category, engines string
|
||||
limit = websearch.DefaultLimit
|
||||
jsonOut, refresh, force bool
|
||||
ttl = float64(websearch.CacheTTL)
|
||||
timeout = 25
|
||||
)
|
||||
i := 0
|
||||
for i < len(args) {
|
||||
a := args[i]
|
||||
switch {
|
||||
case a == "--json":
|
||||
jsonOut = true
|
||||
case a == "--refresh":
|
||||
refresh = true
|
||||
case a == "--force":
|
||||
force = true
|
||||
case (a == "-n" || a == "--limit") && i+1 < len(args):
|
||||
i++
|
||||
n, err := strconv.Atoi(args[i])
|
||||
if err != nil || n < 0 {
|
||||
fmt.Fprintln(os.Stderr, "web/search: --limit must be a non-negative integer")
|
||||
return 2
|
||||
}
|
||||
limit = n
|
||||
case a == "--site" && i+1 < len(args):
|
||||
i++
|
||||
site = args[i]
|
||||
case a == "--lang" && i+1 < len(args):
|
||||
i++
|
||||
lang = args[i]
|
||||
case a == "--fresh" && i+1 < len(args):
|
||||
i++
|
||||
fresh = args[i]
|
||||
case a == "--category" && i+1 < len(args):
|
||||
i++
|
||||
category = args[i]
|
||||
case a == "--engines" && i+1 < len(args):
|
||||
i++
|
||||
engines = args[i]
|
||||
case a == "--ttl" && i+1 < len(args):
|
||||
i++
|
||||
v, err := strconv.ParseFloat(args[i], 64)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, "web/search: --ttl must be a number")
|
||||
return 2
|
||||
}
|
||||
ttl = v
|
||||
case a == "--timeout" && i+1 < len(args):
|
||||
i++
|
||||
n, err := strconv.Atoi(args[i])
|
||||
if err != nil || n <= 0 {
|
||||
fmt.Fprintln(os.Stderr, "web/search: --timeout must be a positive integer")
|
||||
return 2
|
||||
}
|
||||
timeout = n
|
||||
case a == "-h" || a == "--help":
|
||||
fmt.Fprintln(os.Stderr, `usage: bin/web/search.go QUERY [--json] [-n N] [--site HOST] [--lang LANG] [--fresh day|week|month|year] [--category CAT] [--engines LIST] [--refresh] [--force]`)
|
||||
return 0
|
||||
case len(a) > 0 && a[0] != '-' && query == "":
|
||||
query = a
|
||||
default:
|
||||
fmt.Fprintf(os.Stderr, "web/search: unknown flag %s\n", a)
|
||||
return 2
|
||||
}
|
||||
i++
|
||||
}
|
||||
if query == "" {
|
||||
fmt.Fprintln(os.Stderr, "web/search: query required")
|
||||
return 2
|
||||
}
|
||||
if site != "" {
|
||||
query = "site:" + site + " " + query
|
||||
}
|
||||
if reason := websearch.PHIReason(query); reason != "" && !force {
|
||||
fmt.Fprintf(os.Stderr, "refused: %s. This query would leave the host.\n", reason)
|
||||
fmt.Fprintln(os.Stderr, "Rephrase without identifiers, or pass --force if it is genuinely public.")
|
||||
return 2
|
||||
}
|
||||
|
||||
params := map[string]string{}
|
||||
if lang != "" {
|
||||
params["language"] = lang
|
||||
}
|
||||
if fresh != "" {
|
||||
params["time_range"] = fresh
|
||||
}
|
||||
if category != "" {
|
||||
params["categories"] = category
|
||||
}
|
||||
if engines != "" {
|
||||
params["engines"] = engines
|
||||
}
|
||||
|
||||
cachePath := os.Getenv("BRAIN_SEARCH_CACHE")
|
||||
if cachePath == "" {
|
||||
cachePath = os.Getenv("HOME") + "/.cache/brain/web-search.sqlite"
|
||||
}
|
||||
cache, err := websearch.OpenCache(cachePath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "web/search: cache: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer cache.Close()
|
||||
|
||||
key := websearch.CacheKey(query, params)
|
||||
now := float64(time.Now().Unix())
|
||||
if !refresh {
|
||||
if cached, err := cache.Get(key, ttl, now); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "web/search: cache: %v\n", err)
|
||||
return 1
|
||||
} else if cached != nil {
|
||||
out := websearch.Project(*cached, limit, websearch.DefaultSnippetChars)
|
||||
out.Cached = true
|
||||
return writeOut(out, jsonOut)
|
||||
}
|
||||
}
|
||||
|
||||
envPath := os.Getenv("BRAIN_SEARCH_ENV")
|
||||
if envPath == "" {
|
||||
envPath = os.Getenv("HOME") + "/.config/brain/search.env"
|
||||
}
|
||||
conf, err := websearch.LoadConfig(envPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "web/search: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
lockPath := cachePath + ".lock"
|
||||
lock, err := os.OpenFile(lockPath, os.O_CREATE|os.O_RDWR, 0o600)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "web/search: lock: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer lock.Close()
|
||||
if err := unix.Flock(int(lock.Fd()), unix.LOCK_EX); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "web/search: lock: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer unix.Flock(int(lock.Fd()), unix.LOCK_UN)
|
||||
|
||||
var payload websearch.Payload
|
||||
attempts := 1 + len(websearch.RetryBackoff)
|
||||
client := &http.Client{}
|
||||
for attempt := 0; attempt < attempts; attempt++ {
|
||||
last, err := cache.LastCall()
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "web/search: cache: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
if delay := websearch.WaitFor(last, float64(time.Now().Unix()), websearch.MinInterval); delay > 0 {
|
||||
time.Sleep(time.Duration(delay * float64(time.Second)))
|
||||
}
|
||||
if err := cache.MarkCall(float64(time.Now().Unix())); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "web/search: cache: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
payload, err = websearch.Fetch(client, conf, query, params, time.Duration(timeout)*time.Second)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "request failed: %v\n", err)
|
||||
return 3
|
||||
}
|
||||
if websearch.Classify(payload) == websearch.StatusOK {
|
||||
break
|
||||
}
|
||||
if attempt < len(websearch.RetryBackoff) {
|
||||
time.Sleep(time.Duration(websearch.RetryBackoff[attempt] * float64(time.Second)))
|
||||
}
|
||||
}
|
||||
|
||||
if websearch.Classify(payload) == websearch.StatusOK {
|
||||
if err := cache.Put(key, payload, float64(time.Now().Unix())); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "web/search: cache: %v\n", err)
|
||||
}
|
||||
}
|
||||
out := websearch.Project(payload, limit, websearch.DefaultSnippetChars)
|
||||
code := writeOut(out, jsonOut)
|
||||
if out.Status != websearch.StatusOK && code == 0 {
|
||||
return 3
|
||||
}
|
||||
return code
|
||||
}
|
||||
|
||||
func writeOut(out websearch.Output, jsonOut bool) int {
|
||||
if jsonOut {
|
||||
enc := json.NewEncoder(os.Stdout)
|
||||
enc.SetIndent("", " ")
|
||||
enc.SetEscapeHTML(false)
|
||||
if err := enc.Encode(out); err != nil {
|
||||
return 1
|
||||
}
|
||||
if out.Status != websearch.StatusOK {
|
||||
return 3
|
||||
}
|
||||
return 0
|
||||
}
|
||||
fmt.Print(out.YAML())
|
||||
if out.Status != websearch.StatusOK {
|
||||
return 3
|
||||
}
|
||||
return 0
|
||||
}
|
||||
+127
-17
@@ -1,66 +1,176 @@
|
||||
# 2dph — docker composition
|
||||
#
|
||||
# docker compose run --rm brain index # rebuild graph
|
||||
# docker compose run --rm brain search "Matrix fed" # one-shot query
|
||||
# docker compose run --rm brain serve # async Go server
|
||||
# docker compose up brain-watch # auto re-index
|
||||
# docker compose up -d brain # API (Zig CGO serve)
|
||||
# docker compose --profile index run --rm index # Python rebuild
|
||||
# docker compose --profile picoclaw up -d # brain-mcp + CPU reasoner + PicoClaw gateway
|
||||
# docker compose --profile reasoner up -d reasoner # CPU Ollama :11435
|
||||
# docker compose --profile searxng up -d
|
||||
# OCR_ENGINE=paddle docker compose --profile ocr-paddle run --rm ocr-paddle
|
||||
#
|
||||
# Caching: the 128M model (HF_HOME) and kb.lbug (VAR_DIR) live in named
|
||||
# volumes, so rebuilds never redownload the model or re-derive the graph.
|
||||
# Secrets are never baked into the image: search.env + db-profiles.yml mount
|
||||
# read-only from ~/.config/brain.
|
||||
# Secrets never baked in: search.env + db-profiles.yml from ~/.config/brain.
|
||||
|
||||
name: 2dph
|
||||
|
||||
networks:
|
||||
default:
|
||||
name: 2dph_sys
|
||||
driver: bridge
|
||||
ipam:
|
||||
config:
|
||||
- subnet: 10.23.42.0/24
|
||||
|
||||
services:
|
||||
brain:
|
||||
image: ghcr.io/eslider/2dph:latest
|
||||
image: ghcr.io/eslider/2dph:api
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
target: api
|
||||
cache_from:
|
||||
- ghcr.io/eslider/2dph:cache
|
||||
command: ["brain", "search", "help"]
|
||||
command: ["serve"]
|
||||
environment: &env
|
||||
HF_HOME: /data/hf
|
||||
BRAIN_SEARCH_CACHE: /data/cache/web-search.sqlite
|
||||
BRAIN_DB_PROFILES: /secret/db-profiles.yml
|
||||
BRAIN_SEARCH_ENV: /secret/search.env
|
||||
KB_SEARCH_CMD: /app/bin/kb/search
|
||||
KB_ROOT: /data
|
||||
KB_WORKERS: "4"
|
||||
KB_PORT: "8630"
|
||||
volumes:
|
||||
- kb-model:/data/hf
|
||||
- kb-var:/data
|
||||
# corpus is read-only on the host, never written from the container
|
||||
- ..:/corpus:ro
|
||||
- ~/.config/brain:/secret:ro
|
||||
ports:
|
||||
- "127.0.0.1:8630:8630"
|
||||
read_only: true
|
||||
tmpfs:
|
||||
- /tmp
|
||||
healthcheck:
|
||||
test: ["CMD", "python3", "-c", "import ladybug, model2vec, mistune; print('ok')"]
|
||||
test: ["CMD", "wget", "-qO-", "http://127.0.0.1:8630/health"]
|
||||
interval: 30s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
restart: unless-stopped
|
||||
stop_grace_period: 20s
|
||||
|
||||
# watcher: re-index on corpus file change (watchdog script)
|
||||
brain-watch:
|
||||
image: ghcr.io/eslider/2dph:latest
|
||||
image: ghcr.io/eslider/2dph:api
|
||||
environment: *env
|
||||
volumes:
|
||||
- kb-model:/data/hf
|
||||
- kb-var:/data
|
||||
- ..:/corpus:ro
|
||||
- ~/.config/brain:/secret:ro
|
||||
command: ["brain", "watch", "/corpus"]
|
||||
command: ["watch", "/corpus"]
|
||||
read_only: true
|
||||
tmpfs:
|
||||
- /tmp
|
||||
restart: unless-stopped
|
||||
stop_grace_period: 20s
|
||||
|
||||
# Python write path (Ladybug rebuild). Not in the API image.
|
||||
# docker compose --profile index run --rm index
|
||||
index:
|
||||
profiles: ["index"]
|
||||
image: ghcr.io/eslider/2dph:index
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
target: index
|
||||
environment:
|
||||
HF_HOME: /data/hf
|
||||
KB_PY: python3
|
||||
volumes:
|
||||
- kb-model:/data/hf
|
||||
- kb-var:/app/var
|
||||
- ..:/corpus:ro
|
||||
- ~/.config/brain:/secret:ro
|
||||
command: ["index"]
|
||||
read_only: true
|
||||
tmpfs:
|
||||
- /tmp
|
||||
|
||||
# Optional local SearXNG (D3). Skip if BRAIN_SEARCH_URL already points at a
|
||||
# live instance — do not run a second copy on that host.
|
||||
# SEARXNG_SECRET=$(openssl rand -hex 32) docker compose --profile searxng up -d
|
||||
searxng:
|
||||
profiles: ["searxng"]
|
||||
image: docker.io/searxng/searxng:2026.8.10-0a118066d
|
||||
ports:
|
||||
- "127.0.0.1:8888:8080"
|
||||
environment:
|
||||
SEARXNG_SECRET: ${SEARXNG_SECRET:-}
|
||||
volumes:
|
||||
- ./deploy/searxng/settings.yml:/etc/searxng/settings.yml:ro
|
||||
- ./deploy/searxng/limiter.toml:/etc/searxng/limiter.toml:ro
|
||||
restart: unless-stopped
|
||||
|
||||
# MCP endpoint for PicoClaw (and any MCP client).
|
||||
# docker compose --profile picoclaw up -d
|
||||
brain-mcp:
|
||||
profiles: ["picoclaw"]
|
||||
image: ghcr.io/eslider/2dph:api
|
||||
environment: *env
|
||||
volumes:
|
||||
- kb-model:/data/hf
|
||||
- kb-var:/data
|
||||
- ~/.config/brain:/secret:ro
|
||||
command: ["serve"]
|
||||
ports:
|
||||
- "127.0.0.1:8630:8630"
|
||||
read_only: true
|
||||
tmpfs:
|
||||
- /tmp
|
||||
restart: unless-stopped
|
||||
|
||||
# CPU OpenAI-compatible sidecar (D18). Weights are pulled at runtime, not
|
||||
# baked into the 2dph image. Does not touch host Ollama on :11434.
|
||||
# docker compose --profile reasoner up -d reasoner
|
||||
# docker compose --profile reasoner exec reasoner ollama pull qwen3.5:9b
|
||||
reasoner:
|
||||
profiles: ["reasoner", "picoclaw"]
|
||||
image: docker.io/ollama/ollama:latest
|
||||
environment:
|
||||
OLLAMA_NUM_GPU: "0"
|
||||
OLLAMA_HOST: "0.0.0.0:11434"
|
||||
ports:
|
||||
- "127.0.0.1:11435:11434"
|
||||
volumes:
|
||||
- reasoner-ollama:/root/.ollama
|
||||
restart: unless-stopped
|
||||
|
||||
# Official PicoClaw gateway. Config has no secrets (Ollama + HTTP MCP).
|
||||
# Host network: brain/reasoner bind 127.0.0.1 only, so host.docker.internal
|
||||
# (docker0) cannot reach them. Gateway 127.0.0.1:18790 (not the 18800 launcher).
|
||||
# If :8630/:11435 are already bound, do not start brain-mcp/reasoner:
|
||||
# docker compose --profile picoclaw up -d --no-deps picoclaw
|
||||
picoclaw:
|
||||
profiles: ["picoclaw"]
|
||||
image: docker.io/sipeed/picoclaw:v0.3.1
|
||||
network_mode: host
|
||||
depends_on:
|
||||
- brain-mcp
|
||||
- reasoner
|
||||
environment:
|
||||
PICOCLAW_GATEWAY_HOST: "127.0.0.1"
|
||||
entrypoint: ["picoclaw", "gateway"]
|
||||
volumes:
|
||||
- picoclaw-home:/root/.picoclaw
|
||||
- ./deploy/picoclaw/config.json:/root/.picoclaw/config.json:ro
|
||||
restart: unless-stopped
|
||||
|
||||
# Optional PP-OCRv5 (not default). Default OCR is tesseract eng+deu.
|
||||
# OCR_ENGINE=paddle docker compose --profile ocr-paddle run --rm ocr-paddle
|
||||
ocr-paddle:
|
||||
profiles: ["ocr-paddle"]
|
||||
image: python:3.12-slim
|
||||
environment:
|
||||
OCR_ENGINE: paddle
|
||||
command: ["python", "-c", "print('OCR_ENGINE=paddle; install paddleocr on PATH')"]
|
||||
|
||||
volumes:
|
||||
kb-model:
|
||||
kb-var:
|
||||
reasoner-ollama:
|
||||
picoclaw-home:
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
{
|
||||
"agents": {
|
||||
"defaults": {
|
||||
"model_name": "qwen3.5-9b",
|
||||
"max_tool_iterations": 8,
|
||||
"max_tokens": 512,
|
||||
"context_window": 8192
|
||||
}
|
||||
},
|
||||
"model_list": [
|
||||
{
|
||||
"model_name": "qwen3.5-9b",
|
||||
"model": "ollama/qwen3.5:9b",
|
||||
"api_base": "http://127.0.0.1:11435/v1",
|
||||
"request_timeout": 600
|
||||
}
|
||||
],
|
||||
"tools": {
|
||||
"web": {
|
||||
"enabled": false
|
||||
},
|
||||
"mcp": {
|
||||
"enabled": true,
|
||||
"servers": {
|
||||
"2dph": {
|
||||
"enabled": true,
|
||||
"type": "http",
|
||||
"url": "http://127.0.0.1:8630/mcp"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
{
|
||||
"mcpServers": {
|
||||
"2dph": {
|
||||
"url": "http://127.0.0.1:8630/mcp",
|
||||
"description": "2dph fact gate. Tool order: search → get → audit. throttled is not absence."
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
[botdetection.ip_lists]
|
||||
# RFC1918 only. Do not copy a live instance egress IP into git.
|
||||
pass_ip = [
|
||||
"10.0.0.0/8",
|
||||
"172.16.0.0/12",
|
||||
"192.168.0.0/16",
|
||||
]
|
||||
@@ -0,0 +1,28 @@
|
||||
use_default_settings: true
|
||||
|
||||
general:
|
||||
instance_name: "2dph"
|
||||
|
||||
search:
|
||||
formats:
|
||||
- html
|
||||
- json
|
||||
suspended_times:
|
||||
SearxEngineCaptcha: 300
|
||||
SearxEngineTooManyRequests: 120
|
||||
SearxEngineAccessDenied: 300
|
||||
|
||||
server:
|
||||
limiter: true
|
||||
image_proxy: false
|
||||
# secret_key comes from SEARXNG_SECRET (never commit a real secret)
|
||||
|
||||
engines:
|
||||
- name: bing
|
||||
disabled: false
|
||||
- name: google
|
||||
disabled: false
|
||||
- name: duckduckgo
|
||||
disabled: false
|
||||
- name: wikipedia
|
||||
disabled: false
|
||||
+32
-9
@@ -1,15 +1,38 @@
|
||||
# 2dph (deductionphile)
|
||||
---
|
||||
type: reference
|
||||
status: current
|
||||
related:
|
||||
- docs/runbook.md
|
||||
- docs/design.md
|
||||
- PLAN.md
|
||||
- docs/roadmap.md
|
||||
---
|
||||
|
||||
Evidence-first knowledge graph + hybrid RAG over the operational
|
||||
Brain/ops/eSlider stack. Facts need proof or they are
|
||||
# 2dph docs (Diataxis)
|
||||
|
||||
Evidence-first knowledge graph. Facts need proof or they are
|
||||
`(not confirmed)`.
|
||||
|
||||
- [PLAN.md](../PLAN.md) — decisions, execution order, open questions (v2)
|
||||
- [design](design.md) — schema, deduction model, sources
|
||||
- [Gitea issues](https://git.produktor.io/eSlider/2dph/issues) — work board (origin)
|
||||
| Type | Doc |
|
||||
|------|-----|
|
||||
| tutorial / howto | [runbook](runbook.md) — run anywhere (uv, Go, Docker) |
|
||||
| explanation | [design](design.md) — two roots, deduction, D17/D20/D18 |
|
||||
| explanation | [roadmap](roadmap.md) — gap to v1 (epic #16) |
|
||||
| howto | [picoclaw](picoclaw.md) — MCP agent profile |
|
||||
| howto | [reasoner](reasoner.md) — CPU bake-off (D18) |
|
||||
| reference | [PLAN.md](../PLAN.md) — decisions D1–D22 |
|
||||
|
||||
Decisions the public face must name: **D3** SearXNG compose, **D6** Go service /
|
||||
Python write sidecar, **D14** `bin/{subject}/{method}.go`, **D15** Gitea origin,
|
||||
**D17** assertion gate (facts → info → web), **D18** pluggable reasoner.
|
||||
|
||||
Search: `bin/brain/search.go "query"` (HTTP: `bin/brain/serve.go` —
|
||||
`/health` `/search` `/get` `/stats` `/audit` `/ingest`). `--hop` is
|
||||
not a walk; the flag errors until File/FROM_FILE edges exist.
|
||||
`/health` `/search` `/get` `/stats` `/audit` `/ingest`). `--hop N` walks
|
||||
`FROM_FILE` → Commit → Person from each hit (max 3). Rebuild writes
|
||||
File edges ([#17](https://git.produktor.io/eSlider/2dph/issues/17)).
|
||||
|
||||
Published docs live here and mirror the project state.
|
||||
Work board: [Gitea issues](https://git.produktor.io/eSlider/2dph/issues)
|
||||
([epic #16](https://git.produktor.io/eSlider/2dph/issues/16)).
|
||||
PRs and CI: GitHub [`eSlider/2dph`](https://github.com/eSlider/2dph).
|
||||
|
||||
Published docs live here and match live commands.
|
||||
|
||||
@@ -21,4 +21,5 @@ OO_CLI (default: $HOME/go/bin/oo)
|
||||
./bin/chats/apply.go --dry-run
|
||||
```
|
||||
|
||||
JSONL → markdown only. Brain ingest is `bin/brain/index.go` (not a `chats index`).
|
||||
JSONL → markdown only. Brain ingest is `bin/brain/index.go --with-chats`
|
||||
(default `var/chats/md`). WhatsApp sync is out of v1.
|
||||
|
||||
+46
-3
@@ -1,3 +1,12 @@
|
||||
---
|
||||
type: explanation
|
||||
status: current
|
||||
related:
|
||||
- docs/README.md
|
||||
- docs/runbook.md
|
||||
- docs/roadmap.md
|
||||
---
|
||||
|
||||
# Design — facts, info, deduction
|
||||
|
||||
## Two roots, one transaction
|
||||
@@ -21,10 +30,13 @@ bin/brain/search.go "question"
|
||||
1. facts root — confirmed answers only → return with evidence links
|
||||
2. info root — supporting narrative → snippets, marked (not confirmed)
|
||||
3. web-search — second independent source → upgrade hypothesis to confirmed
|
||||
(`web` block from `bin/web/search.go` when no facts hit; status `throttled`
|
||||
is not evidence of absence; `--no-web` / `--root` skip it)
|
||||
```
|
||||
|
||||
`--hop` is not implemented yet (needs File/FROM_FILE edges). The flag is an
|
||||
error; it is not a graph walk.
|
||||
`--hop N` walks `Leaf-[:FROM_FILE]->File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person`
|
||||
from each hit (1=File, 2=Commit, 3=Person). Rebuild writes FROM_FILE;
|
||||
git import writes HAS_VERSION/AUTHORED ([#17](https://git.produktor.io/eSlider/2dph/issues/17)).
|
||||
|
||||
## Who / What / How / Where / When + evidence
|
||||
|
||||
@@ -46,7 +58,8 @@ Every assertion edge carries:
|
||||
Content leafs: `sha256`, `observed_at`, `source_rev`, `confidence`. Stale = a
|
||||
file changed on disk (git HEAD/mtime) after its last observed `source_rev`.
|
||||
`File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person` records the history of every
|
||||
content leaf.
|
||||
content leaf. Commit records come from `bin/git/import.go` (go-git, no git
|
||||
binary); conversion prints leafs, brain write is `bin/brain/index.go`.
|
||||
|
||||
`bin/facts/audit stale` flags leafs whose observed revision is behind the
|
||||
corpus HEAD.
|
||||
@@ -59,3 +72,33 @@ corpus HEAD.
|
||||
|
||||
Confirmed = A×B or B×C agreement. Single source = hypothesis + `(not confirmed)`.
|
||||
Conflicting pairings (≥2 yes vs ≥2 no) = hypothesis (OQ1 → v2 resolution).
|
||||
|
||||
## Read path
|
||||
|
||||
`bin/brain/get.go`, `stats.go`, and `eval.go` call `internal/brain` with cgo
|
||||
(`system_ladybug`), compiled by **Zig** (`bin/cgo/zcc`, D21), not gcc.
|
||||
They do not exec Python. Control questions for recall@5 live in
|
||||
`internal/brain/rank` so CI can test the table without libladybug.
|
||||
Python `bin/kb/{get,stats,eval}` remain for GitHub Actions until the runner
|
||||
fetches Zig + libs (`bin/cgo/zig`). Incremental write is `bin/kb/add`
|
||||
(`bin/brain/add.go`). Bulk index/write is still `bin/kb/index`
|
||||
(`docker compose --profile index`).
|
||||
|
||||
## Agent API (D20)
|
||||
|
||||
`bin/brain/serve.go` exposes the same `internal/httpapi.Ops` table as OpenAPI
|
||||
(`GET /openapi.json`) and MCP (`POST /mcp` JSON-RPC `tools/list` +
|
||||
`tools/call`). Tool names match paths: `search`, `get`, `stats`, `audit`,
|
||||
`ingest` (add a leaf; omit body for the CLI hint).
|
||||
Agents should use these endpoints instead of shebang CLIs.
|
||||
|
||||
## Reasoner (D18)
|
||||
|
||||
Pluggable OpenAI-compatible URL. RAM: `Qwen/Qwen3.5-9B`. Quality:
|
||||
`prism-ml/Bonsai-27B-gguf` or `Qwen/Qwen3.6-27B`. No official Qwen3.6-9B.
|
||||
CPU sidecar: compose profile `reasoner` (`OLLAMA_NUM_GPU=0`,
|
||||
`127.0.0.1:11435`). Bake-off: `bin/reasoner/bakeoff.go`. Weights stay out
|
||||
of the 2dph image. See [docs/reasoner.md](reasoner.md).
|
||||
|
||||
Gap to v1 (hops, corpus, CI eval): [roadmap](roadmap.md),
|
||||
[epic #16](https://git.produktor.io/eSlider/2dph/issues/16).
|
||||
@@ -0,0 +1,34 @@
|
||||
# PicoClaw profile (reference agent)
|
||||
|
||||
2dph is the memory/fact gate. Compose profile `picoclaw` runs the official
|
||||
PicoClaw gateway (`docker.io/sipeed/picoclaw:v0.3.1`) plus `brain-mcp` and the
|
||||
CPU reasoner. Default agent model is `qwen3.5:9b` (RAM path, D18). Weights stay
|
||||
in the reasoner volume, not in the 2dph image.
|
||||
No secrets in git: Ollama needs no key; MCP is local HTTP.
|
||||
|
||||
```bash
|
||||
docker compose --profile picoclaw up -d
|
||||
# already serving :8630 / :11435:
|
||||
docker compose --profile picoclaw up -d --no-deps picoclaw
|
||||
```
|
||||
|
||||
Gateway: `127.0.0.1:18790`. Brain MCP: `http://127.0.0.1:8630/mcp`.
|
||||
Cursor-style clients can use [deploy/picoclaw/mcp.json.example](../deploy/picoclaw/mcp.json.example).
|
||||
PicoClaw itself uses [deploy/picoclaw/config.json](../deploy/picoclaw/config.json)
|
||||
(`127.0.0.1` + host network — loopback publishes are not reachable via docker0).
|
||||
|
||||
OpenAPI: `GET http://127.0.0.1:8630/openapi.json`.
|
||||
|
||||
Before a factual reply: `search` → `get` → `audit`. `throttled` is not a
|
||||
negative finding. See `skills/picoclaw/SKILL.md`.
|
||||
|
||||
System performance (MCP gates + qwen3.5:9b tool_call + PicoClaw gateway):
|
||||
|
||||
```bash
|
||||
./qa/system_perf.py --json | yq '.gates'
|
||||
REASONER_MODEL=qwen3.5:9b ./qa/system_perf.py --reasoner --picoclaw --json | yq '.reasoner'
|
||||
```
|
||||
|
||||
The default agent model is `qwen3.5:9b`. PicoClaw `context_window` is 8192
|
||||
(heuristic `max_tokens*4` at 512 is 2048, too small for MCP tool schemas).
|
||||
`request_timeout` is 600s for a CPU turn (tool_call + MCP search + answer).
|
||||
@@ -0,0 +1,71 @@
|
||||
# Reasoner bake-off (D18)
|
||||
|
||||
Pluggable OpenAI-compatible URL. 2dph does not ship weights. PicoClaw is
|
||||
compose profile `picoclaw` (`sipeed/picoclaw`); the bake-off hits the same
|
||||
tool names (`search` → `get` → `audit` from `internal/httpapi.Ops`).
|
||||
|
||||
```bash
|
||||
docker compose --profile reasoner up -d reasoner
|
||||
docker compose --profile reasoner exec reasoner ollama pull qwen3.5:9b
|
||||
REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b \
|
||||
./bin/reasoner/bakeoff.go --json
|
||||
```
|
||||
|
||||
JSON includes `latency_p50_ms` / `latency_p95_ms` from DuckDB (`internal/duckstats`, D22).
|
||||
|
||||
Host Ollama on `:11434` is left alone. This sidecar binds `127.0.0.1:11435`
|
||||
with `OLLAMA_NUM_GPU=0` (CPU). Measure RSS (`/api/ps` `size`), not VRAM.
|
||||
|
||||
If Compose cannot allocate a project network (Docker IPAM pool exhausted),
|
||||
the same sidecar is:
|
||||
|
||||
```bash
|
||||
docker run -d --name 2dph-reasoner \
|
||||
-e OLLAMA_NUM_GPU=0 \
|
||||
-p 127.0.0.1:11435:11434 \
|
||||
-v 2dph-reasoner-ollama:/root/.ollama \
|
||||
ollama/ollama:latest
|
||||
```
|
||||
|
||||
## Real Hugging Face ids
|
||||
|
||||
| Role | HF id | Ollama tag (this bake-off) |
|
||||
|------|-------|----------------------------|
|
||||
| RAM / 9B | `Qwen/Qwen3.5-9B` | `qwen3.5:9b` |
|
||||
| Quality 27B (CPU) | `prism-ml/Bonsai-27B-gguf` (derived from Qwen3.6-27B) | `MichelRosselli/bonsai-27b:Q1_0` |
|
||||
| Quality 27B (full) | `Qwen/Qwen3.6-27B` | not pulled on this CPU box |
|
||||
|
||||
There is **no official Qwen3.6-9B**. Do not invent that id.
|
||||
|
||||
Qwen3.5-9B has documented upstream tool-call XML bugs. A 9B win on tools
|
||||
is only claimed if this bake-off records OpenAI `tool_calls` (not
|
||||
`<tool_call>` XML in `content`).
|
||||
|
||||
The 2dph API image does not `COPY` GGUF/safetensors. Pull at runtime into
|
||||
the `reasoner-ollama` volume.
|
||||
|
||||
## Live CPU run
|
||||
|
||||
Host sidecar: Ollama **0.32.9**, `OLLAMA_NUM_GPU=0`, `127.0.0.1:11435`,
|
||||
`device: cpu`, `vram_mb: 0`. Date: 2026-08-13. Same three prompts
|
||||
(`search` / `get` / `audit`). PicoClaw binary was not used; the OpenAI
|
||||
tools payload is the surface it would send.
|
||||
|
||||
| Model | HF id | tool_call | xml_leak | rss_mb | latency_ms (search/get/audit) |
|
||||
|-------|-------|-----------|----------|--------|-------------------------------|
|
||||
| `qwen3.5:9b` | `Qwen/Qwen3.5-9B` | 3/3 | 0 | 5790 | 50059 / 51678 / 31980 |
|
||||
| `MichelRosselli/bonsai-27b:Q1_0` | `prism-ml/Bonsai-27B-gguf` | 3/3 | 0 | 21951 | 327702 / 166830 / 119808 |
|
||||
| `Qwen/Qwen3.6-27B` | `Qwen/Qwen3.6-27B` | not loaded | — | — | too heavy for this CPU box |
|
||||
|
||||
Both loaded models emitted OpenAI `tool_calls` (not `<tool_call>` XML) on
|
||||
this runtime. **Do not claim 9B is better at tools** — the score is tied
|
||||
at 3/3. 9B is smaller and faster. Bonsai RSS includes weights + KV
|
||||
(`size` from `/api/ps`); first Bonsai prompt includes cold load.
|
||||
|
||||
Re-run:
|
||||
|
||||
```bash
|
||||
REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b \
|
||||
./bin/reasoner/bakeoff.go --json
|
||||
REASONER_MODEL=MichelRosselli/bonsai-27b:Q1_0 ./bin/reasoner/bakeoff.go --json
|
||||
```
|
||||
@@ -0,0 +1,66 @@
|
||||
---
|
||||
type: explanation
|
||||
status: current
|
||||
related:
|
||||
- PLAN.md
|
||||
- docs/design.md
|
||||
- docs/runbook.md
|
||||
---
|
||||
|
||||
# Gap to v1 — detective brain
|
||||
|
||||
Goal: a brain that does not assert without proof. Search is deduction
|
||||
(`facts` ≥2 sources → `info` → `web`). `confirmed` only from the facts root.
|
||||
|
||||
**v1 is a living graph the agent can write and walk**, not “more RAG”.
|
||||
|
||||
Epic: [Gitea #16](https://git.produktor.io/eSlider/2dph/issues/16).
|
||||
Milestone: [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12).
|
||||
Decisions: [PLAN.md](../PLAN.md).
|
||||
|
||||
## In (do not reopen)
|
||||
|
||||
Read path Go + Zig CGO (D21). HTTP + OpenAPI + MCP (D20). PicoClaw compose
|
||||
profile + CPU reasoner (D18). Mail sync → import → rebuild. D14 shebangs.
|
||||
Compose `api` (no CPython) / `index` (Python write). Issues #1–#5, #7–#13.
|
||||
[#15](https://git.produktor.io/eSlider/2dph/issues/15) lever/loop.
|
||||
[#14](https://git.produktor.io/eSlider/2dph/issues/14) `bin/brain/add.go` /
|
||||
`POST /ingest` (Python `kblib.add_leafs`; no Go upsert port).
|
||||
[#17](https://git.produktor.io/eSlider/2dph/issues/17) `--hop N` walks
|
||||
FROM_FILE / HAS_VERSION / AUTHORED.
|
||||
[#18](https://git.produktor.io/eSlider/2dph/issues/18) `--with-facts` /
|
||||
`--with-chats` on rebuild (WhatsApp out of v1).
|
||||
[#19](https://git.produktor.io/eSlider/2dph/issues/19) CI recall SoT =
|
||||
`bin/brain/eval.go` via Zig.
|
||||
Epic [#16](https://git.produktor.io/eSlider/2dph/issues/16) closed.
|
||||
|
||||
## v2
|
||||
|
||||
[#6](https://git.produktor.io/eSlider/2dph/issues/6) OCR — **in**.
|
||||
[#30](https://git.produktor.io/eSlider/2dph/issues/30) OQ3 duckdb-go — **in**.
|
||||
[#29](https://git.produktor.io/eSlider/2dph/issues/29) OQ1 contradiction
|
||||
resolution.
|
||||
|
||||
## Blockers
|
||||
|
||||
None for epic #16 (closed). Remaining v2: OQ1, OQ4.
|
||||
|
||||
```
|
||||
question
|
||||
│
|
||||
├─ FTS + HNSW ← in
|
||||
├─ facts / info roots ← in
|
||||
├─ web (D17) ← in
|
||||
├─ brain/add ACID ← in
|
||||
├─ Cypher hop ← in
|
||||
└─ facts+chats corpus ← in
|
||||
```
|
||||
|
||||
## Not v1
|
||||
|
||||
OQ1 contradiction resolution, OQ4 YAML-first leafs.
|
||||
OCR (OQ2) and duckdb-go (OQ3/D22) are in.
|
||||
|
||||
## Close epic #16 when
|
||||
|
||||
Children #14, #15, #17, #18, #19 are closed. MCP tool order stays gated by tests.
|
||||
@@ -0,0 +1,85 @@
|
||||
---
|
||||
type: howto
|
||||
status: current
|
||||
related:
|
||||
- docs/README.md
|
||||
- PLAN.md
|
||||
---
|
||||
|
||||
# Run 2dph (portable)
|
||||
|
||||
No laptop-absolute paths. Config lives in env files under `$HOME/.config/brain/`
|
||||
(mode 0600), not in git.
|
||||
|
||||
## Toolchain
|
||||
|
||||
- Go (see `go.mod`)
|
||||
- Python 3.12 + [uv](https://docs.astral.sh/uv)
|
||||
- Optional: Docker, Zig CGO via `bin/cgo/zig` (not gcc)
|
||||
- Optional: poppler (`pdftotext`/`pdftoppm`) + tesseract `eng+deu` for mail OCR
|
||||
|
||||
```bash
|
||||
uv venv .venv
|
||||
uv pip install -r requirements.lock.txt
|
||||
eval "$(bin/cgo/zig env)" # when compiling Ladybug read tools
|
||||
go test ./...
|
||||
uv run python -m unittest discover -s bin/tools -t .
|
||||
```
|
||||
|
||||
## Config
|
||||
|
||||
| File / env | Purpose |
|
||||
|------------|---------|
|
||||
| `$BRAIN_SEARCH_ENV` (default `$HOME/.config/brain/search.env`) | `BRAIN_SEARCH_URL` (SearXNG). Optional Basic Auth. |
|
||||
| `$HOME/.config/brain/db-profiles.yml` | read-only Postgres profiles (OnlyOffice via tunnel) |
|
||||
|
||||
If the host already runs SearXNG, point `BRAIN_SEARCH_URL` at it. Do not start
|
||||
a second copy (D3). Optional Compose instance:
|
||||
|
||||
```bash
|
||||
SEARXNG_SECRET=$(openssl rand -hex 32) docker compose --profile searxng up -d
|
||||
```
|
||||
|
||||
That binds `127.0.0.1:8888`. JSON format must stay enabled.
|
||||
|
||||
## Index then search
|
||||
|
||||
Write path is `bin/brain/add.go` for a leaf (or `POST /ingest`). Bulk
|
||||
corpus rebuild remains `bin/brain/index.go --rebuild` (Compose profile
|
||||
`index`). Do not DROP INDEX on Ladybug 0.19.
|
||||
|
||||
```bash
|
||||
bin/brain/add.go --text "arc-1 runs Matrix" --root facts --source "compose.yml x docker ps"
|
||||
bin/brain/index.go --rebuild --with-facts --with-chats
|
||||
bin/brain/search.go "LadybugDB vector index" # facts → info → web (D17)
|
||||
bin/brain/search.go "upstream flag" --no-web
|
||||
bin/brain/get.go <id> --body
|
||||
bin/brain/stats.go
|
||||
```
|
||||
|
||||
`--hop N` walks File → Commit → Person from each hit. Empty web results are `throttled`, not absence.
|
||||
Gap to v1: [roadmap](roadmap.md) / [epic #16](https://git.produktor.io/eSlider/2dph/issues/16).
|
||||
|
||||
Ladybug 0.19: never `DROP INDEX` FTS/VECTOR (ghost catalog). Fresh indexes =
|
||||
delete `var/kb.lbug` then `--rebuild`.
|
||||
|
||||
## HTTP / MCP
|
||||
|
||||
```bash
|
||||
docker compose up -d brain # :8630 Zig CGO serve
|
||||
docker compose --profile index run --rm index # rebuild
|
||||
docker compose --profile picoclaw up brain-mcp # MCP 127.0.0.1:8630
|
||||
```
|
||||
|
||||
`GET /openapi.json`, `POST /mcp`. Agent tool order: `search` → `get` → `audit`.
|
||||
|
||||
## Reasoner (optional, D18)
|
||||
|
||||
CPU sidecar on `127.0.0.1:11435`. Weights are not in the 2dph image.
|
||||
|
||||
```bash
|
||||
docker compose --profile reasoner up -d reasoner
|
||||
REASONER_BASE_URL=http://127.0.0.1:11435/v1 ./bin/reasoner/bakeoff.go --json
|
||||
```
|
||||
|
||||
See [reasoner.md](reasoner.md).
|
||||
@@ -7,19 +7,54 @@ require (
|
||||
github.com/arran4/golang-ical v0.3.5
|
||||
github.com/chewxy/math32 v1.11.2
|
||||
github.com/daulet/tokenizers v1.27.0
|
||||
github.com/duckdb/duckdb-go/v2 v2.10505.0
|
||||
github.com/go-git/go-git/v5 v5.19.2
|
||||
golang.org/x/sys v0.47.0
|
||||
golang.org/x/text v0.40.0
|
||||
modernc.org/sqlite v1.56.0
|
||||
)
|
||||
|
||||
require (
|
||||
dario.cat/mergo v1.0.0 // indirect
|
||||
github.com/Microsoft/go-winio v0.6.2 // indirect
|
||||
github.com/ProtonMail/go-crypto v1.1.6 // indirect
|
||||
github.com/apache/arrow-go/v18 v18.6.0 // indirect
|
||||
github.com/cloudflare/circl v1.6.3 // indirect
|
||||
github.com/cyphar/filepath-securejoin v0.6.1 // indirect
|
||||
github.com/duckdb/duckdb-go-bindings v0.10505.0 // indirect
|
||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-amd64 v0.10505.0 // indirect
|
||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-arm64 v0.10505.0 // indirect
|
||||
github.com/duckdb/duckdb-go-bindings/lib/linux-amd64 v0.10505.0 // indirect
|
||||
github.com/duckdb/duckdb-go-bindings/lib/linux-arm64 v0.10505.0 // indirect
|
||||
github.com/duckdb/duckdb-go-bindings/lib/windows-amd64 v0.10505.0 // indirect
|
||||
github.com/dustin/go-humanize v1.0.1 // indirect
|
||||
github.com/emirpasic/gods v1.18.1 // indirect
|
||||
github.com/go-git/gcfg v1.5.1-0.20230307220236-3a3c6141e376 // indirect
|
||||
github.com/go-git/go-billy/v5 v5.9.0 // indirect
|
||||
github.com/go-viper/mapstructure/v2 v2.5.0 // indirect
|
||||
github.com/goccy/go-json v0.10.6 // indirect
|
||||
github.com/golang/groupcache v0.0.0-20241129210726-2c02b8208cf8 // indirect
|
||||
github.com/google/flatbuffers v25.12.19+incompatible // indirect
|
||||
github.com/google/uuid v1.6.0 // indirect
|
||||
github.com/jbenet/go-context v0.0.0-20150711004518-d14ea06fba99 // indirect
|
||||
github.com/kevinburke/ssh_config v1.2.0 // indirect
|
||||
github.com/klauspost/compress v1.18.5 // indirect
|
||||
github.com/klauspost/cpuid/v2 v2.3.0 // indirect
|
||||
github.com/mattn/go-isatty v0.0.24 // indirect
|
||||
github.com/ncruces/go-strftime v1.0.0 // indirect
|
||||
github.com/pierrec/lz4/v4 v4.1.26 // indirect
|
||||
github.com/pjbgf/sha1cd v0.6.0 // indirect
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
github.com/sergi/go-diff v1.3.2-0.20230802210424-5b0b94c5c0d3 // indirect
|
||||
github.com/shopspring/decimal v1.4.0 // indirect
|
||||
github.com/skeema/knownhosts v1.3.1 // indirect
|
||||
github.com/xanzy/ssh-agent v0.3.3 // indirect
|
||||
github.com/zeebo/xxh3 v1.1.0 // indirect
|
||||
golang.org/x/exp v0.0.0-20260112195511-716be5621a96 // indirect
|
||||
golang.org/x/sys v0.43.0 // indirect
|
||||
golang.org/x/crypto v0.53.0 // indirect
|
||||
golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f // indirect
|
||||
golang.org/x/net v0.56.0 // indirect
|
||||
gopkg.in/warnings.v0 v0.1.2 // indirect
|
||||
modernc.org/libc v1.74.4 // indirect
|
||||
modernc.org/mathutil v1.7.1 // indirect
|
||||
modernc.org/memory v1.11.0 // indirect
|
||||
)
|
||||
|
||||
@@ -1,50 +1,200 @@
|
||||
dario.cat/mergo v1.0.0 h1:AGCNq9Evsj31mOgNPcLyXc+4PNABt905YmuqPYYpBWk=
|
||||
dario.cat/mergo v1.0.0/go.mod h1:uNxQE+84aUszobStD9th8a29P2fMDhsBdgRYvZOxGmk=
|
||||
github.com/LadybugDB/go-ladybug v0.17.0 h1:RXDbkBjrbRmLdEbhGl4CLOIEzSt09gbP0n9UbKDEfwI=
|
||||
github.com/LadybugDB/go-ladybug v0.17.0/go.mod h1:GeIXmE8XyF5TFS94NAuTag7vgCC+no/HTBMRA6Rd5Cs=
|
||||
github.com/Microsoft/go-winio v0.5.2/go.mod h1:WpS1mjBmmwHBEWmogvA2mj8546UReBk4v8QkMxJ6pZY=
|
||||
github.com/Microsoft/go-winio v0.6.2 h1:F2VQgta7ecxGYO8k3ZZz3RS8fVIXVxONVUPlNERoyfY=
|
||||
github.com/Microsoft/go-winio v0.6.2/go.mod h1:yd8OoFMLzJbo9gZq8j5qaps8bJ9aShtEA8Ipt1oGCvU=
|
||||
github.com/ProtonMail/go-crypto v1.1.6 h1:ZcV+Ropw6Qn0AX9brlQLAUXfqLBc7Bl+f/DmNxpLfdw=
|
||||
github.com/ProtonMail/go-crypto v1.1.6/go.mod h1:rA3QumHc/FZ8pAHreoekgiAbzpNsfQAosU5td4SnOrE=
|
||||
github.com/andybalholm/brotli v1.2.1 h1:R+f5xP285VArJDRgowrfb9DqL18yVK0gKAW/F+eTWro=
|
||||
github.com/andybalholm/brotli v1.2.1/go.mod h1:rzTDkvFWvIrjDXZHkuS16NPggd91W3kUSvPlQ1pLaKY=
|
||||
github.com/anmitsu/go-shlex v0.0.0-20200514113438-38f4b401e2be h1:9AeTilPcZAjCFIImctFaOjnTIavg87rW78vTPkQqLI8=
|
||||
github.com/anmitsu/go-shlex v0.0.0-20200514113438-38f4b401e2be/go.mod h1:ySMOLuWl6zY27l47sB3qLNK6tF2fkHG55UZxx8oIVo4=
|
||||
github.com/apache/arrow-go/v18 v18.6.0 h1:GX/Jyd3R7mCLiECAwY9FWbbaYblie2WXBSz4Sw8fNpM=
|
||||
github.com/apache/arrow-go/v18 v18.6.0/go.mod h1:gm3MiPpY82fLYK5VKPB3WoJbsiLVDfT7flD5/vHReKw=
|
||||
github.com/apache/thrift v0.22.0 h1:r7mTJdj51TMDe6RtcmNdQxgn9XcyfGDOzegMDRg47uc=
|
||||
github.com/apache/thrift v0.22.0/go.mod h1:1e7J/O1Ae6ZQMTYdy9xa3w9k+XHWPfRvdPyJeynQ+/g=
|
||||
github.com/armon/go-socks5 v0.0.0-20160902184237-e75332964ef5 h1:0CwZNZbxp69SHPdPJAN/hZIm0C4OItdklCFmMRWYpio=
|
||||
github.com/armon/go-socks5 v0.0.0-20160902184237-e75332964ef5/go.mod h1:wHh0iHkYZB8zMSxRWpUBQtwG5a7fFgvEO+odwuTv2gs=
|
||||
github.com/arran4/golang-ical v0.3.5 h1:bbz6ld4dC+MmCKiFfOd6SkmIGnhNMBACZ485ULh7p9A=
|
||||
github.com/arran4/golang-ical v0.3.5/go.mod h1:OnguFgjN0Hmx8jzpmWcC+AkHio94ujmLHKoaef7xQh8=
|
||||
github.com/chewxy/math32 v1.11.2 h1:IufN08Zwr1NKuWfY+4Tz55BcwKmyKKNdOP7KtumehnM=
|
||||
github.com/chewxy/math32 v1.11.2/go.mod h1:dOB2rcuFrCn6UHrze36WSLVPKtzPMRAQvBvUwkSsLqs=
|
||||
github.com/cloudflare/circl v1.6.3 h1:9GPOhQGF9MCYUeXyMYlqTR6a5gTrgR/fBLXvUgtVcg8=
|
||||
github.com/cloudflare/circl v1.6.3/go.mod h1:2eXP6Qfat4O/Yhh8BznvKnJ+uzEoTQ6jVKJRn81BiS4=
|
||||
github.com/cyphar/filepath-securejoin v0.6.1 h1:5CeZ1jPXEiYt3+Z6zqprSAgSWiggmpVyciv8syjIpVE=
|
||||
github.com/cyphar/filepath-securejoin v0.6.1/go.mod h1:A8hd4EnAeyujCJRrICiOWqjS1AX0a9kM5XL+NwKoYSc=
|
||||
github.com/daulet/tokenizers v1.27.0 h1:MmFYAEDFz69s/nNQfHg59DWqHz3v94m99kEZ/JbL+s4=
|
||||
github.com/daulet/tokenizers v1.27.0/go.mod h1:YjFY1o1HGMyWkQgbXJDghhvke/yFDp2vGdIO2hYs4MQ=
|
||||
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM=
|
||||
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/duckdb/duckdb-go-bindings v0.10505.0 h1:/0pPsTLrcCsTGxT0VrHgJWnOcPe1tQL1vrki1v3jbAI=
|
||||
github.com/duckdb/duckdb-go-bindings v0.10505.0/go.mod h1:HoD5xePkDj3VZbBnVVfxVVYIljZ9khCprWA7FgwIiC4=
|
||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-amd64 v0.10505.0 h1:FrMqquFBQlMsi34h2KZgCku54rqA8xEbXZ0NLVDKwYs=
|
||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-amd64 v0.10505.0/go.mod h1:EnAvZh1kNJHp5yF+M1ZHNEvapnmt6anq1xXHVrAGqMo=
|
||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-arm64 v0.10505.0 h1:lbRbpQwT1MmUhh/VTwukV9K8bxKByV3UghAP3MvsbBo=
|
||||
github.com/duckdb/duckdb-go-bindings/lib/darwin-arm64 v0.10505.0/go.mod h1:IGLSeEcFhNeZF16aVjQCULD7TsFZKG5G7SyKJAXKp5c=
|
||||
github.com/duckdb/duckdb-go-bindings/lib/linux-amd64 v0.10505.0 h1:nrsaVYj3XYCRbS2FpdOMD/KHE7egRMr+/NR1IHmjT84=
|
||||
github.com/duckdb/duckdb-go-bindings/lib/linux-amd64 v0.10505.0/go.mod h1:KAIynZ0GHCS7X5fRyuFnQMg/SZBPK/bS9OCOVojClxw=
|
||||
github.com/duckdb/duckdb-go-bindings/lib/linux-arm64 v0.10505.0 h1:qM6oGDgwXBILJGbTY4fCy6QOczLpucUA6yn6g3ORjh4=
|
||||
github.com/duckdb/duckdb-go-bindings/lib/linux-arm64 v0.10505.0/go.mod h1:81SGOYoEUs8qaAfSk1wRfM5oobrIJ5KI7AzYhK6/bvQ=
|
||||
github.com/duckdb/duckdb-go-bindings/lib/windows-amd64 v0.10505.0 h1:DjqZl9rYreHkSOqnqLmkrqH5T8UdQNcxZLJVZzGmXXA=
|
||||
github.com/duckdb/duckdb-go-bindings/lib/windows-amd64 v0.10505.0/go.mod h1:K25pJL26ARblGDeuAkrdblFvUen92+CwksLtPEHRqqQ=
|
||||
github.com/duckdb/duckdb-go/v2 v2.10505.0 h1:SWwvLn2Qx/RQSnQNupwgIF8VbnJ5A6OQU9lYb/mDETI=
|
||||
github.com/duckdb/duckdb-go/v2 v2.10505.0/go.mod h1:m0PW4J4FG9hlFlVdXi6Ds9owpyIDaBdE2jyce00fGcE=
|
||||
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
|
||||
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
|
||||
github.com/elazarl/goproxy v1.7.2 h1:Y2o6urb7Eule09PjlhQRGNsqRfPmYI3KKQLFpCAV3+o=
|
||||
github.com/elazarl/goproxy v1.7.2/go.mod h1:82vkLNir0ALaW14Rc399OTTjyNREgmdL2cVoIbS6XaE=
|
||||
github.com/emirpasic/gods v1.18.1 h1:FXtiHYKDGKCW2KzwZKx0iC0PQmdlorYgdFG9jPXJ1Bc=
|
||||
github.com/emirpasic/gods v1.18.1/go.mod h1:8tpGGwCnJ5H4r6BWwaV6OrWmMoPhUl5jm/FMNAnJvWQ=
|
||||
github.com/gliderlabs/ssh v0.3.8 h1:a4YXD1V7xMF9g5nTkdfnja3Sxy1PVDCj1Zg4Wb8vY6c=
|
||||
github.com/gliderlabs/ssh v0.3.8/go.mod h1:xYoytBv1sV0aL3CavoDuJIQNURXkkfPA/wxQ1pL1fAU=
|
||||
github.com/go-git/gcfg v1.5.1-0.20230307220236-3a3c6141e376 h1:+zs/tPmkDkHx3U66DAb0lQFJrpS6731Oaa12ikc+DiI=
|
||||
github.com/go-git/gcfg v1.5.1-0.20230307220236-3a3c6141e376/go.mod h1:an3vInlBmSxCcxctByoQdvwPiA7DTK7jaaFDBTtu0ic=
|
||||
github.com/go-git/go-billy/v5 v5.9.0 h1:jItGXszUDRtR/AlferWPTMN4j38BQ88XnXKbilmmBPA=
|
||||
github.com/go-git/go-billy/v5 v5.9.0/go.mod h1:jCnQMLj9eUgGU7+ludSTYoZL/GGmii14RxKFj7ROgHw=
|
||||
github.com/go-git/go-git-fixtures/v4 v4.3.2-0.20231010084843-55a94097c399 h1:eMje31YglSBqCdIqdhKBW8lokaMrL3uTkpGYlE2OOT4=
|
||||
github.com/go-git/go-git-fixtures/v4 v4.3.2-0.20231010084843-55a94097c399/go.mod h1:1OCfN199q1Jm3HZlxleg+Dw/mwps2Wbk9frAWm+4FII=
|
||||
github.com/go-git/go-git/v5 v5.19.2 h1:wkfn7vOlUBu8ivAWKBWisTiwJK4jYHzTF8Ndv1LyGqY=
|
||||
github.com/go-git/go-git/v5 v5.19.2/go.mod h1:QqCBE1EFN5ddFmrliLQ3/ntRCUjZU3EJuwuB/jWEHjk=
|
||||
github.com/go-viper/mapstructure/v2 v2.5.0 h1:vM5IJoUAy3d7zRSVtIwQgBj7BiWtMPfmPEgAXnvj1Ro=
|
||||
github.com/go-viper/mapstructure/v2 v2.5.0/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM=
|
||||
github.com/goccy/go-json v0.10.6 h1:p8HrPJzOakx/mn/bQtjgNjdTcN+/S6FcG2CTtQOrHVU=
|
||||
github.com/goccy/go-json v0.10.6/go.mod h1:oq7eo15ShAhp70Anwd5lgX2pLfOS3QCiwU/PULtXL6M=
|
||||
github.com/golang/groupcache v0.0.0-20241129210726-2c02b8208cf8 h1:f+oWsMOmNPc8JmEHVZIycC7hBoQxHH9pNKQORJNozsQ=
|
||||
github.com/golang/groupcache v0.0.0-20241129210726-2c02b8208cf8/go.mod h1:wcDNUvekVysuuOpQKo3191zZyTpiI6se1N1ULghS0sw=
|
||||
github.com/google/flatbuffers v25.12.19+incompatible h1:haMV2JRRJCe1998HeW/p0X9UaMTK6SDo0ffLn2+DbLs=
|
||||
github.com/google/flatbuffers v25.12.19+incompatible/go.mod h1:1AeVuKshWv4vARoZatz6mlQ0JxURH0Kv5+zNeJKJCa8=
|
||||
github.com/google/go-cmp v0.6.0 h1:ofyhxvXcZhMsU5ulbFiLKl/XBFqE1GSq7atu8tAmTRI=
|
||||
github.com/google/go-cmp v0.6.0/go.mod h1:17dUlkBOakJ0+DkrSSNjCkIjxS6bF9zb3elmeNGIjoY=
|
||||
github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8=
|
||||
github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU=
|
||||
github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3 h1:LMLX+LgTNWpfvCBdFebv6EsYotImrt/Ppc5cXIriCSo=
|
||||
github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3/go.mod h1:jl5iWTm0/hd5PjEYEOuwAJ57L/CibdZfrqZ5XA5GrCk=
|
||||
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
|
||||
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7 h1:a+bsQ5rvGLjzHuww6tVxozPZFVghXaHOwFs4luLUK2k=
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7/go.mod h1:QeFd9opnmA6QUJc5vARoKUSoFhyfM2/ZepoAG6RGpeM=
|
||||
github.com/jbenet/go-context v0.0.0-20150711004518-d14ea06fba99 h1:BQSFePA1RWJOlocH6Fxy8MmwDt+yVQYULKfN0RoTN8A=
|
||||
github.com/jbenet/go-context v0.0.0-20150711004518-d14ea06fba99/go.mod h1:1lJo3i6rXxKeerYnT8Nvf0QmHCRC1n8sfWVwXF2Frvo=
|
||||
github.com/kevinburke/ssh_config v1.2.0 h1:x584FjTGwHzMwvHx18PXxbBVzfnxogHaAReU4gf13a4=
|
||||
github.com/kevinburke/ssh_config v1.2.0/go.mod h1:CT57kijsi8u/K/BOFA39wgDQJ9CxiF4nAY/ojJ6r6mM=
|
||||
github.com/klauspost/compress v1.18.5 h1:/h1gH5Ce+VWNLSWqPzOVn6XBO+vJbCNGvjoaGBFW2IE=
|
||||
github.com/klauspost/compress v1.18.5/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
|
||||
github.com/klauspost/cpuid/v2 v2.3.0 h1:S4CRMLnYUhGeDFDqkGriYKdfoFlDnMtqTiI/sFzhA9Y=
|
||||
github.com/klauspost/cpuid/v2 v2.3.0/go.mod h1:hqwkgyIinND0mEev00jJYCxPNVRVXFQeu1XKlok6oO0=
|
||||
github.com/kr/pretty v0.1.0/go.mod h1:dAy3ld7l9f0ibDNOQOHHMYYIIbhfbHSm3C4ZsoJORNo=
|
||||
github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE=
|
||||
github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk=
|
||||
github.com/kr/pty v1.1.1/go.mod h1:pFQYn66WHrOpPYNljwOMqo10TkYh1fy3cYio2l3bCsQ=
|
||||
github.com/kr/text v0.1.0/go.mod h1:4Jbv+DJW3UT/LiOwJeYQe1efqtUx/iVham/4vfdArNI=
|
||||
github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY=
|
||||
github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE=
|
||||
github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI=
|
||||
github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A=
|
||||
github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOFAw7w=
|
||||
github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls=
|
||||
github.com/onsi/gomega v1.34.1 h1:EUMJIKUjM8sKjYbtxQI9A4z2o+rruxnzNvpknOXie6k=
|
||||
github.com/onsi/gomega v1.34.1/go.mod h1:kU1QgUvBDLXBJq618Xvm2LUX6rSAfRaFRTcdOeDLwwY=
|
||||
github.com/pierrec/lz4/v4 v4.1.26 h1:GrpZw1gZttORinvzBdXPUXATeqlJjqUG/D87TKMnhjY=
|
||||
github.com/pierrec/lz4/v4 v4.1.26/go.mod h1:EoQMVJgeeEOMsCqCzqFm2O0cJvljX2nGZjcRIPL34O4=
|
||||
github.com/pjbgf/sha1cd v0.6.0 h1:3WJ8Wz8gvDz29quX1OcEmkAlUg9diU4GxJHqs0/XiwU=
|
||||
github.com/pjbgf/sha1cd v0.6.0/go.mod h1:lhpGlyHLpQZoxMv8HcgXvZEhcGs0PG/vsZnEJ7H0iCM=
|
||||
github.com/pkg/errors v0.9.1 h1:FEBLx1zS214owpjy7qsBeixbURkuhQAwrK5UwLGTwt4=
|
||||
github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINEl0=
|
||||
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 h1:Jamvg5psRIccs7FGNTlIRMkT8wgtp5eCXdBlqhYGL6U=
|
||||
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
|
||||
github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ=
|
||||
github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc=
|
||||
github.com/sergi/go-diff v1.3.2-0.20230802210424-5b0b94c5c0d3 h1:n661drycOFuPLCN3Uc8sB6B/s6Z4t2xvBgU1htSHuq8=
|
||||
github.com/sergi/go-diff v1.3.2-0.20230802210424-5b0b94c5c0d3/go.mod h1:A0bzQcvG0E7Rwjx0REVgAGH58e96+X0MeOfepqsbeW4=
|
||||
github.com/shopspring/decimal v1.4.0 h1:bxl37RwXBklmTi0C79JfXCEBD1cqqHt0bbgBAGFp81k=
|
||||
github.com/shopspring/decimal v1.4.0/go.mod h1:gawqmDU56v4yIKSwfBSFip1HdCCXN8/+DMd9qYNcwME=
|
||||
github.com/sirupsen/logrus v1.7.0/go.mod h1:yWOB1SBYBC5VeMP7gHvWumXLIWorT60ONWic61uBYv0=
|
||||
github.com/skeema/knownhosts v1.3.1 h1:X2osQ+RAjK76shCbvhHHHVl3ZlgDm8apHEHFqRjnBY8=
|
||||
github.com/skeema/knownhosts v1.3.1/go.mod h1:r7KTdC8l4uxWRyK2TpQZ/1o5HaSzh06ePQNxPwTcfiY=
|
||||
github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME=
|
||||
github.com/stretchr/testify v1.2.2/go.mod h1:a8OnRcib4nhh0OaRAV+Yts87kKdq0PP7pXfy6kDkUVs=
|
||||
github.com/stretchr/testify v1.4.0/go.mod h1:j7eGeouHqKxXV5pUuKE4zz7dFj8WfuZ+81PSLYec5m4=
|
||||
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
|
||||
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
|
||||
github.com/xanzy/ssh-agent v0.3.3 h1:+/15pJfg/RsTxqYcX6fHqOXZwwMP+2VyYWJeWM2qQFM=
|
||||
github.com/xanzy/ssh-agent v0.3.3/go.mod h1:6dzNDKs0J9rVPHPhaGCukekBHKqfl+L3KghI1Bc68Uw=
|
||||
github.com/zeebo/assert v1.3.0 h1:g7C04CbJuIDKNPFHmsk4hwZDO5O+kntRxzaUoNXj+IQ=
|
||||
github.com/zeebo/assert v1.3.0/go.mod h1:Pq9JiuJQpG8JLJdtkwrJESF0Foym2/D9XMU5ciN/wJ0=
|
||||
github.com/zeebo/xxh3 v1.1.0 h1:s7DLGDK45Dyfg7++yxI0khrfwq9661w9EN78eP/UZVs=
|
||||
github.com/zeebo/xxh3 v1.1.0/go.mod h1:IisAie1LELR4xhVinxWS5+zf1lA4p0MW4T+w+W07F5s=
|
||||
golang.org/x/exp v0.0.0-20260112195511-716be5621a96 h1:Z/6YuSHTLOHfNFdb8zVZomZr7cqNgTJvA8+Qz75D8gU=
|
||||
golang.org/x/exp v0.0.0-20260112195511-716be5621a96/go.mod h1:nzimsREAkjBCIEFtHiYkrJyT+2uy9YZJB7H1k68CXZU=
|
||||
golang.org/x/sys v0.43.0 h1:Rlag2XtaFTxp19wS8MXlJwTvoh8ArU6ezoyFsMyCTNI=
|
||||
golang.org/x/sys v0.43.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/crypto v0.0.0-20220622213112-05595931fe9d/go.mod h1:IxCIyHEi3zRg3s0A5j5BB6A9Jmi73HwBIUl50j+osU4=
|
||||
golang.org/x/crypto v0.53.0 h1:QZ4Muo8THX6CizN2vPPd5fBGHyogrdK9fG4wLPFUsto=
|
||||
golang.org/x/crypto v0.53.0/go.mod h1:DNLU434OwVakk9PzuwV8w62mAJpRJL3vsgcfp4Qnsio=
|
||||
golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f h1:W3F4c+6OLc6H2lb//N1q4WpJkhzJCK5J6kUi1NTVXfM=
|
||||
golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f/go.mod h1:J1xhfL/vlindoeF/aINzNzt2Bket5bjo9sdOYzOsU80=
|
||||
golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ=
|
||||
golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0=
|
||||
golang.org/x/net v0.0.0-20211112202133-69e39bad7dc2/go.mod h1:9nx3DQGgdP8bBQD5qxJ1jj9UTztislL4KSBs9R2vV5Y=
|
||||
golang.org/x/net v0.56.0 h1:Rw8j/hFzGvJUZwNBXnAtf5sVDVt+65SK2C7IxCxZt5o=
|
||||
golang.org/x/net v0.56.0/go.mod h1:D3Ku6r+V6JROoZK144D2XfMHFcMq/0zSfLelVTCFKec=
|
||||
golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
|
||||
golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sys v0.0.0-20191026070338-33540a1f6037/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20210124154548-22da62e12c0c/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20210423082822-04245dca01da/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20220715151400-c0bba94af5f8/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
|
||||
golang.org/x/term v0.44.0 h1:0rLvDRCtNj0gZkyIXhCyOb2OAzEhLVqc4B+hrsBhrmc=
|
||||
golang.org/x/term v0.44.0/go.mod h1:7ze4MdzUzLXpSAoFP1H0bOI9aXDqveSvatT5vKcFh2Y=
|
||||
golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
|
||||
golang.org/x/text v0.40.0 h1:Ub2Z6/xjgF1WrYQz2nuITOEegKFtiIy+rieRJ5lHZKs=
|
||||
golang.org/x/text v0.40.0/go.mod h1:hpnzDAfGV753zIKo+wk3u1bVKCGPbrnF7+7LBF/UHVY=
|
||||
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
|
||||
golang.org/x/tools v0.47.0 h1:7Kn5x/d1svx/PzryTsqeoZN4TZwqeH5pGWjefhLi/1Q=
|
||||
golang.org/x/tools v0.47.0/go.mod h1:dFHnyTvFWY212G+h7ZY4Vsp/K3U4/7W9TyVaAul8uCA=
|
||||
gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4=
|
||||
gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||
gopkg.in/check.v1 v1.0.0-20190902080502-41f04d3bba15/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk=
|
||||
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q=
|
||||
gopkg.in/warnings.v0 v0.1.2 h1:wFXVbFY8DY5/xOe1ECiWdKCzZlxgshcYVNkBHstARME=
|
||||
gopkg.in/warnings.v0 v0.1.2/go.mod h1:jksf8JmL6Qr/oQM2OXTHunEvvTAsrWBLb6OOjuVWRNI=
|
||||
gopkg.in/yaml.v2 v2.2.2/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||
gopkg.in/yaml.v2 v2.4.0/go.mod h1:RDklbk79AGWmwhnvt/jBztapEOGDOx6ZbXqjP6csGnQ=
|
||||
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
||||
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||
modernc.org/cc/v4 v4.29.1 h1:MKgdCV3WykTSPqpVrnxdEDS0HEd2FHpKZDzxzU5LyeI=
|
||||
modernc.org/cc/v4 v4.29.1/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI=
|
||||
modernc.org/ccgo/v4 v4.34.6 h1:sBgfIwyN0TQ9C5hwIeuqyeAKyMWnbvj2fvpF4L11uzU=
|
||||
modernc.org/ccgo/v4 v4.34.6/go.mod h1:SZ8YcN9NG7XVsQYdm6jYBvi8PQP1qi+kqB6OhjqI3Fk=
|
||||
modernc.org/fileutil v1.4.0 h1:j6ZzNTftVS054gi281TyLjHPp6CPHr2KCxEXjEbD6SM=
|
||||
modernc.org/fileutil v1.4.0/go.mod h1:EqdKFDxiByqxLk8ozOxObDSfcVOv/54xDs/DUHdvCUU=
|
||||
modernc.org/gc/v2 v2.6.5 h1:nyqdV8q46KvTpZlsw66kWqwXRHdjIlJOhG6kxiV/9xI=
|
||||
modernc.org/gc/v2 v2.6.5/go.mod h1:YgIahr1ypgfe7chRuJi2gD7DBQiKSLMPgBQe9oIiito=
|
||||
modernc.org/gc/v3 v3.1.4 h1:2g65LGVSmFQrXeITAw97x7hCRvZFcyE1uDP+7Vng7JI=
|
||||
modernc.org/gc/v3 v3.1.4/go.mod h1:HFK/6AGESC7Ex+EZJhJ2Gni6cTaYpSMmU/cT9RmlfYY=
|
||||
modernc.org/goabi0 v0.2.0 h1:HvEowk7LxcPd0eq6mVOAEMai46V+i7Jrj13t4AzuNks=
|
||||
modernc.org/goabi0 v0.2.0/go.mod h1:CEFRnnJhKvWT1c1JTI3Avm+tgOWbkOu5oPA8eH8LnMI=
|
||||
modernc.org/libc v1.74.4 h1:fX1Omw4o2/1C2iRkkIsrQTasJQldLhRmuPreXLoWs9k=
|
||||
modernc.org/libc v1.74.4/go.mod h1:eeQAS9W3sZeKYMFubydxJpII9ybHWshk+7or7bLG9co=
|
||||
modernc.org/mathutil v1.7.1 h1:GCZVGXdaN8gTqB1Mf/usp1Y/hSqgI2vAGGP4jZMCxOU=
|
||||
modernc.org/mathutil v1.7.1/go.mod h1:4p5IwJITfppl0G4sUEDtCr4DthTaT47/N3aT6MhfgJg=
|
||||
modernc.org/memory v1.11.0 h1:o4QC8aMQzmcwCK3t3Ux/ZHmwFPzE6hf2Y5LbkRs+hbI=
|
||||
modernc.org/memory v1.11.0/go.mod h1:/JP4VbVC+K5sU2wZi9bHoq2MAkCnrt2r98UGeSK7Mjw=
|
||||
modernc.org/opt v0.2.0 h1:tGyef5ApycA7FSEOMraay9SaTk5zmbx7Tu+cJs4QKZg=
|
||||
modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns=
|
||||
modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w=
|
||||
modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE=
|
||||
modernc.org/sqlite v1.56.0 h1:/D8e2RfFqoy/Zc6PuC76U28zFwmI/sYx1Kjm4yEn9e0=
|
||||
modernc.org/sqlite v1.56.0/go.mod h1:yCJ2cmAaIkHQ25oXWrF8H4O1lIfPYPR26yCEDj2P3pQ=
|
||||
modernc.org/strutil v1.2.1 h1:UneZBkQA+DX2Rp35KcM69cSsNES9ly8mQWD71HKlOA0=
|
||||
modernc.org/strutil v1.2.1/go.mod h1:EHkiggD70koQxjVdSBM3JKM7k6L0FbGE5eymy9i3B9A=
|
||||
modernc.org/token v1.1.0 h1:Xl7Ap9dKaEs5kLoOQeQmPWevfnk/DM5qcLcYlA8ys6Y=
|
||||
modernc.org/token v1.1.0/go.mod h1:UGzOrNV1mAFSEB63lOFHIpNRUVMvYTc6yu1SMY/XTDM=
|
||||
|
||||
+25
-8
@@ -7,6 +7,10 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
|
||||
"github.com/eSlider/2dph/internal/brain/rank"
|
||||
)
|
||||
|
||||
// Ready opens the Ladybug file for the life of the serve process.
|
||||
@@ -17,7 +21,7 @@ func Ready() error {
|
||||
// HTTP is the in-process API used by bin/brain/serve.go.
|
||||
type HTTP struct{}
|
||||
|
||||
func (HTTP) Search(_ context.Context, query string, limit int) ([]byte, error) {
|
||||
func (HTTP) Search(ctx context.Context, query string, limit int) ([]byte, error) {
|
||||
hits, err := searchHits(query, "", "", limit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -31,10 +35,13 @@ func (HTTP) Search(_ context.Context, query string, limit int) ([]byte, error) {
|
||||
hits[i].Snippet = string(runes)
|
||||
}
|
||||
}
|
||||
webOut := rank.Deduce(hits, query, "", false, func(q string) rank.SecondSource {
|
||||
return lookupWeb(ctx, q)
|
||||
})
|
||||
var buf bytes.Buffer
|
||||
enc := json.NewEncoder(&buf)
|
||||
enc.SetEscapeHTML(false)
|
||||
if err := enc.Encode(toJSONOut(hits, query, "")); err != nil {
|
||||
if err := enc.Encode(toJSONOut(hits, query, "", webOut)); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return buf.Bytes(), nil
|
||||
@@ -132,12 +139,22 @@ func (HTTP) Audit(context.Context) ([]byte, error) {
|
||||
return json.Marshal(map[string]any{"status": "ok", "by_confidence": rows})
|
||||
}
|
||||
|
||||
func (HTTP) Ingest(context.Context) ([]byte, error) {
|
||||
return json.Marshal(map[string]any{
|
||||
"mode": "rebuild",
|
||||
"command": "bin/brain/index.go --rebuild",
|
||||
"add": "v2",
|
||||
})
|
||||
func (HTTP) Ingest(ctx context.Context, body []byte) ([]byte, error) {
|
||||
if len(bytes.TrimSpace(body)) == 0 {
|
||||
return json.Marshal(map[string]any{
|
||||
"mode": "add",
|
||||
"command": "bin/brain/add.go",
|
||||
"rebuild": "bin/brain/index.go --rebuild",
|
||||
})
|
||||
}
|
||||
cmd := exec.CommandContext(ctx, filepath.Join(repoRoot(), "bin", "kb", "add"), "--json")
|
||||
cmd.Stdin = bytes.NewReader(body)
|
||||
cmd.Dir = repoRoot()
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("add: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func asInt(v any) int64 {
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
package brain
|
||||
|
||||
const ModelID = "minishlab/potion-multilingual-128M"
|
||||
@@ -6,7 +6,7 @@ import (
|
||||
"strings"
|
||||
)
|
||||
|
||||
const Usage = `usage: bin/brain/search.go "query" [--root facts|info] [--repo REPO] [-n N] [--json]
|
||||
const Usage = `usage: bin/brain/search.go "query" [--root facts|info] [--repo REPO] [-n N] [--hop N] [--json] [--no-web]
|
||||
bin/brain/search.go serve [port]
|
||||
bin/brain/search.go --list-model`
|
||||
|
||||
@@ -15,14 +15,14 @@ type Options struct {
|
||||
Root string
|
||||
Repo string
|
||||
Limit int
|
||||
Hop int
|
||||
JSONOut bool
|
||||
ListModel bool
|
||||
NoWeb bool
|
||||
}
|
||||
|
||||
// ParseArgs reads flags. Unknown flags are an error: silently dropping them
|
||||
// meant `--hop 1` vanished and its argument `1` was appended to the query.
|
||||
// --hop is recognised so it cannot be swallowed; it is not implemented until
|
||||
// File/FROM_FILE edges exist.
|
||||
func ParseArgs(args []string) (Options, error) {
|
||||
opt := Options{Limit: 20}
|
||||
var queryArgs []string
|
||||
@@ -51,9 +51,19 @@ func ParseArgs(args []string) (Options, error) {
|
||||
}
|
||||
opt.Limit = n
|
||||
case "--hop":
|
||||
return opt, fmt.Errorf("--hop is not implemented yet (needs File/FROM_FILE edges)")
|
||||
i++
|
||||
n, err := strconv.Atoi(args[i])
|
||||
if err != nil || n < 1 {
|
||||
return opt, fmt.Errorf("--hop must be a positive integer, got %q", args[i])
|
||||
}
|
||||
if n > 3 {
|
||||
return opt, fmt.Errorf("--hop max is 3 (File → Commit → Person)")
|
||||
}
|
||||
opt.Hop = n
|
||||
case "--json":
|
||||
opt.JSONOut = true
|
||||
case "--no-web":
|
||||
opt.NoWeb = true
|
||||
case "--list-model":
|
||||
opt.ListModel = true
|
||||
default:
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
package rank
|
||||
|
||||
// SecondSource is the web-search block on a deduction answer.
|
||||
// Kept apart from graph hits so "ours" and "not ours" stay visible.
|
||||
type SecondSource struct {
|
||||
Status string `json:"status"`
|
||||
Note string `json:"note,omitempty"`
|
||||
Cached bool `json:"cached,omitempty"`
|
||||
Results []SecondSourceHit `json:"results,omitempty"`
|
||||
}
|
||||
|
||||
type SecondSourceHit struct {
|
||||
Rank int `json:"rank"`
|
||||
Title string `json:"title"`
|
||||
URL string `json:"url"`
|
||||
Snippet string `json:"snippet"`
|
||||
Engine string `json:"engine"`
|
||||
}
|
||||
|
||||
type WebFn func(query string) SecondSource
|
||||
|
||||
// ShouldEscalate is true when the default deduction path has no facts hit.
|
||||
// `--root facts|info` is a single-root ask: do not mix in the web.
|
||||
func ShouldEscalate(hits []Hit, rootFilter string) bool {
|
||||
if rootFilter != "" {
|
||||
return false
|
||||
}
|
||||
for _, h := range hits {
|
||||
if h.Root == "facts" {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// Deduce returns the second-source block, or nil when web must not run.
|
||||
func Deduce(hits []Hit, query, rootFilter string, noWeb bool, web WebFn) *SecondSource {
|
||||
if noWeb || web == nil || !ShouldEscalate(hits, rootFilter) {
|
||||
return nil
|
||||
}
|
||||
out := web(query)
|
||||
return &out
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
package rank
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestShouldEscalateWhenNoFacts(t *testing.T) {
|
||||
if !ShouldEscalate(nil, "") {
|
||||
t.Fatal("empty local graph must escalate")
|
||||
}
|
||||
if !ShouldEscalate([]Hit{h("i", "info", "docs/a.md")}, "") {
|
||||
t.Fatal("info-only must escalate (not confirmed)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestShouldNotEscalateWhenFactsConfirm(t *testing.T) {
|
||||
hits := []Hit{h("f", "facts", "docker ps x compose"), h("i", "info", "docs/a.md")}
|
||||
if ShouldEscalate(hits, "") {
|
||||
t.Fatal("facts hit is already confirmed; do not mix web")
|
||||
}
|
||||
}
|
||||
|
||||
func TestShouldNotEscalateWhenRootFilterSet(t *testing.T) {
|
||||
if ShouldEscalate(nil, "facts") {
|
||||
t.Fatal("--root facts must stay local")
|
||||
}
|
||||
if ShouldEscalate([]Hit{h("i", "info", "x")}, "info") {
|
||||
t.Fatal("--root info must stay local")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDeduceCallsWebOnlyWhenEscalating(t *testing.T) {
|
||||
called := 0
|
||||
web := func(q string) SecondSource {
|
||||
called++
|
||||
if q != "LadybugDB" {
|
||||
t.Fatalf("query = %q", q)
|
||||
}
|
||||
return SecondSource{Status: "ok", Results: []SecondSourceHit{{Title: "t", URL: "http://example.com"}}}
|
||||
}
|
||||
got := Deduce([]Hit{h("i", "info", "x")}, "LadybugDB", "", false, web)
|
||||
if called != 1 || got == nil || got.Status != "ok" {
|
||||
t.Fatalf("got %+v called=%d", got, called)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDeduceNilWhenFactsOrNoWeb(t *testing.T) {
|
||||
web := func(string) SecondSource {
|
||||
t.Fatal("web must not run")
|
||||
return SecondSource{}
|
||||
}
|
||||
if Deduce([]Hit{h("f", "facts", "x")}, "q", "", false, web) != nil {
|
||||
t.Fatal("facts")
|
||||
}
|
||||
if Deduce([]Hit{h("i", "info", "x")}, "q", "", true, web) != nil {
|
||||
t.Fatal("--no-web")
|
||||
}
|
||||
if Deduce(nil, "q", "facts", false, web) != nil {
|
||||
t.Fatal("--root facts")
|
||||
}
|
||||
if Deduce(nil, "q", "", false, nil) != nil {
|
||||
t.Fatal("nil web fn")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseNoWeb(t *testing.T) {
|
||||
opt, err := ParseArgs([]string{"query", "--no-web", "--json"})
|
||||
if err != nil || !opt.NoWeb || !opt.JSONOut || opt.Query != "query" {
|
||||
t.Fatalf("got %+v err=%v", opt, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUsageNamesNoWeb(t *testing.T) {
|
||||
if !strings.Contains(Usage, "--no-web") {
|
||||
t.Fatalf("usage must name --no-web, got:\n%s", Usage)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
package rank
|
||||
|
||||
// Eval control questions (recall@5). Kept here so CI can test the gate
|
||||
// table without ladybug cgo. The runner lives in internal/brain (cgo).
|
||||
const EvalRecallThreshold = 0.95
|
||||
|
||||
type EvalQuestion struct {
|
||||
Query string
|
||||
Fragment string
|
||||
}
|
||||
|
||||
var EvalQuestions = []EvalQuestion{
|
||||
{"hybrid search fts and vector", "BM25"},
|
||||
{"eslider devops engineer", "DevOps"},
|
||||
{"ladybugdb graph engine storage", "LadybugDB"},
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
package rank
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestEvalQuestionsAreThreeAndThreshold(t *testing.T) {
|
||||
if EvalRecallThreshold != 0.95 {
|
||||
t.Fatalf("threshold = %v", EvalRecallThreshold)
|
||||
}
|
||||
if len(EvalQuestions) != 3 {
|
||||
t.Fatalf("questions = %d, want 3", len(EvalQuestions))
|
||||
}
|
||||
for _, q := range EvalQuestions {
|
||||
if q.Query == "" || q.Fragment == "" {
|
||||
t.Fatalf("empty control: %+v", q)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -7,3 +7,30 @@ const FTSStmt = "CALL QUERY_FTS_INDEX('Leaf', 'id', $q) " +
|
||||
|
||||
const VecStmt = "CALL QUERY_VECTOR_INDEX('Leaf', 'Leaf_vec', $q, $n) " +
|
||||
"RETURN node.id, node.text, node.root, node.source, distance ORDER BY distance LIMIT $n"
|
||||
|
||||
// HopStmt is the Cypher walk from a search hit. Depth 1 = File, 2 = Commit, 3 = Person.
|
||||
func HopStmt(depth int) string {
|
||||
switch depth {
|
||||
case 1:
|
||||
return "MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File) RETURN f.id, f.path, 1"
|
||||
case 2:
|
||||
return "MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File)-[:HAS_VERSION]->(c:Commit) RETURN c.id, c.subject, 2"
|
||||
case 3:
|
||||
return "MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File)-[:HAS_VERSION]->(c:Commit)-[:AUTHORED]->(p:Person) RETURN p.id, p.name, 3"
|
||||
default:
|
||||
return ""
|
||||
}
|
||||
}
|
||||
|
||||
func HopLabel(depth int) string {
|
||||
switch depth {
|
||||
case 1:
|
||||
return "File"
|
||||
case 2:
|
||||
return "Commit"
|
||||
case 3:
|
||||
return "Person"
|
||||
default:
|
||||
return ""
|
||||
}
|
||||
}
|
||||
|
||||
@@ -7,14 +7,22 @@ import (
|
||||
"strings"
|
||||
)
|
||||
|
||||
type HopNode struct {
|
||||
ID string `json:"id"`
|
||||
Label string `json:"label"`
|
||||
Name string `json:"name"`
|
||||
Depth int `json:"depth"`
|
||||
}
|
||||
|
||||
// Hit is one search result, mirroring the python script's dict shape.
|
||||
type Hit struct {
|
||||
ID string `json:"id"`
|
||||
Text string `json:"text"`
|
||||
Root string `json:"root"`
|
||||
Source string `json:"-"`
|
||||
Score float64 `json:"score"`
|
||||
Snippet string `json:"snippet,omitempty"`
|
||||
ID string `json:"id"`
|
||||
Text string `json:"text"`
|
||||
Root string `json:"root"`
|
||||
Source string `json:"-"`
|
||||
Score float64 `json:"score"`
|
||||
Snippet string `json:"snippet,omitempty"`
|
||||
Hops []HopNode `json:"hops,omitempty"`
|
||||
}
|
||||
|
||||
// rrfK dampens the contribution of low ranks; same constant as kblib.py.
|
||||
|
||||
@@ -91,15 +91,41 @@ func TestHybridKeepsVectorScoreForSharedHit(t *testing.T) {
|
||||
}
|
||||
|
||||
// The old parser dropped unknown flags and appended their arguments to the
|
||||
// query, so `search "q" --hop 1` searched for "q 1". --hop is not implemented
|
||||
// here (needs File edges); it must still fail closed instead of changing q.
|
||||
// query, so `search "q" --hop 1` searched for "q 1". --hop must stay a flag.
|
||||
func TestParseHopIsNotSwallowedIntoTheQuery(t *testing.T) {
|
||||
_, err := ParseArgs([]string{"what runs on arc-2", "--hop", "1"})
|
||||
if err == nil {
|
||||
t.Fatal("expected --hop to error (not implemented), not be swallowed")
|
||||
opt, err := ParseArgs([]string{"what runs on arc-2", "--hop", "1"})
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "--hop") {
|
||||
t.Fatalf("error should name --hop, got %v", err)
|
||||
if opt.Query != "what runs on arc-2" {
|
||||
t.Fatalf("query swallowed hop arg: %q", opt.Query)
|
||||
}
|
||||
if opt.Hop != 1 {
|
||||
t.Fatalf("hop = %d, want 1", opt.Hop)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseHopMaxIsThree(t *testing.T) {
|
||||
if _, err := ParseArgs([]string{"q", "--hop", "4"}); err == nil {
|
||||
t.Fatal("expected --hop 4 to error")
|
||||
}
|
||||
opt, err := ParseArgs([]string{"q", "--hop", "3"})
|
||||
if err != nil || opt.Hop != 3 {
|
||||
t.Fatalf("hop 3: %+v err=%v", opt, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHopStmtWalksFromFile(t *testing.T) {
|
||||
s := HopStmt(1)
|
||||
if !strings.Contains(s, "FROM_FILE") || !strings.Contains(s, "File") {
|
||||
t.Fatalf("hop 1 must walk FROM_FILE, got %q", s)
|
||||
}
|
||||
s3 := HopStmt(3)
|
||||
if !strings.Contains(s3, "HAS_VERSION") || !strings.Contains(s3, "AUTHORED") || !strings.Contains(s3, "Person") {
|
||||
t.Fatalf("hop 3 must reach Person, got %q", s3)
|
||||
}
|
||||
if HopLabel(1) != "File" || HopLabel(3) != "Person" {
|
||||
t.Fatal("hop labels")
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,281 @@
|
||||
//go:build cgo && system_ladybug
|
||||
|
||||
package brain
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
"unicode/utf8"
|
||||
|
||||
"github.com/eSlider/2dph/internal/brain/rank"
|
||||
)
|
||||
|
||||
func MainGet(args []string) int {
|
||||
id, body, jsonOut := "", false, false
|
||||
for _, a := range args {
|
||||
switch {
|
||||
case a == "--body":
|
||||
body = true
|
||||
case a == "--json":
|
||||
jsonOut = true
|
||||
case a == "-h" || a == "--help":
|
||||
fmt.Fprintln(os.Stderr, `usage: bin/brain/get.go <id> [--body] [--json]`)
|
||||
return 0
|
||||
case strings.HasPrefix(a, "-"):
|
||||
fmt.Fprintf(os.Stderr, "brain/get: unknown flag %s\n", a)
|
||||
return 2
|
||||
default:
|
||||
id = a
|
||||
}
|
||||
}
|
||||
if id == "" {
|
||||
fmt.Fprintln(os.Stderr, "brain/get: id required")
|
||||
return 2
|
||||
}
|
||||
if err := openBrain(); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "open brain: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer closeBrain()
|
||||
meta, text, err := lookupLeaf(id)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "brain/get: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
out := Dict{
|
||||
{"id", meta["id"]},
|
||||
{"root", meta["root"]},
|
||||
{"confidence", meta["confidence"]},
|
||||
{"source", meta["source"]},
|
||||
{"type", meta["type"]},
|
||||
}
|
||||
if body {
|
||||
out = append(out, KV{"text", text})
|
||||
} else {
|
||||
out = append(out, KV{"snippet", clip(text, 280)})
|
||||
}
|
||||
if jsonOut {
|
||||
m := map[string]any{}
|
||||
for _, kv := range out {
|
||||
m[kv.K] = kv.V
|
||||
}
|
||||
enc := json.NewEncoder(os.Stdout)
|
||||
enc.SetIndent("", " ")
|
||||
enc.SetEscapeHTML(false)
|
||||
return b2i(enc.Encode(m))
|
||||
}
|
||||
fmt.Print(toYAML(out, 0))
|
||||
return 0
|
||||
}
|
||||
|
||||
func MainStats(args []string) int {
|
||||
jsonOut := false
|
||||
for _, a := range args {
|
||||
switch a {
|
||||
case "--json":
|
||||
jsonOut = true
|
||||
case "-h", "--help":
|
||||
fmt.Fprintln(os.Stderr, `usage: bin/brain/stats.go [--json]`)
|
||||
return 0
|
||||
default:
|
||||
if strings.HasPrefix(a, "-") {
|
||||
fmt.Fprintf(os.Stderr, "brain/stats: unknown flag %s\n", a)
|
||||
return 2
|
||||
}
|
||||
}
|
||||
}
|
||||
if err := openBrain(); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "open brain: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer closeBrain()
|
||||
s, err := leafStats()
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "brain/stats: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
if jsonOut {
|
||||
enc := json.NewEncoder(os.Stdout)
|
||||
enc.SetIndent("", " ")
|
||||
enc.SetEscapeHTML(false)
|
||||
return b2i(enc.Encode(s))
|
||||
}
|
||||
by := s["by_root"].(map[string]int)
|
||||
keys := make([]string, 0, len(by))
|
||||
for k := range by {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
byRoot := make(Dict, 0, len(keys))
|
||||
for _, k := range keys {
|
||||
byRoot = append(byRoot, KV{k, by[k]})
|
||||
}
|
||||
out := Dict{
|
||||
{"total", s["total"]},
|
||||
{"by_root", byRoot},
|
||||
{"db", s["db"]},
|
||||
{"model", s["model"]},
|
||||
}
|
||||
fmt.Print(toYAML(out, 0))
|
||||
return 0
|
||||
}
|
||||
|
||||
func MainEval(args []string) int {
|
||||
jsonOut := false
|
||||
for _, a := range args {
|
||||
switch a {
|
||||
case "--json":
|
||||
jsonOut = true
|
||||
case "-h", "--help":
|
||||
fmt.Fprintln(os.Stderr, `usage: bin/brain/eval.go [--json]`)
|
||||
return 0
|
||||
default:
|
||||
if strings.HasPrefix(a, "-") {
|
||||
fmt.Fprintf(os.Stderr, "brain/eval: unknown flag %s\n", a)
|
||||
return 2
|
||||
}
|
||||
}
|
||||
}
|
||||
if err := openBrain(); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "open brain: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer closeBrain()
|
||||
recalled := 0
|
||||
details := make([]any, 0, len(rank.EvalQuestions))
|
||||
jsDetails := make([]map[string]any, 0, len(rank.EvalQuestions))
|
||||
for _, q := range rank.EvalQuestions {
|
||||
hits, err := queryFTS(q.Query, 5)
|
||||
ok := false
|
||||
if err == nil {
|
||||
frag := strings.ToLower(q.Fragment)
|
||||
for _, h := range hits {
|
||||
if strings.Contains(strings.ToLower(h.Text), frag) {
|
||||
ok = true
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if ok {
|
||||
recalled++
|
||||
}
|
||||
details = append(details, Dict{
|
||||
{"q", q.Query},
|
||||
{"fragment", q.Fragment},
|
||||
{"in_top5", ok},
|
||||
})
|
||||
jsDetails = append(jsDetails, map[string]any{
|
||||
"q": q.Query, "fragment": q.Fragment, "in_top5": ok,
|
||||
})
|
||||
}
|
||||
n := len(rank.EvalQuestions)
|
||||
recall := 0.0
|
||||
if n > 0 {
|
||||
recall = float64(recalled) / float64(n)
|
||||
}
|
||||
passed := recall >= rank.EvalRecallThreshold
|
||||
if jsonOut {
|
||||
enc := json.NewEncoder(os.Stdout)
|
||||
enc.SetIndent("", " ")
|
||||
enc.SetEscapeHTML(false)
|
||||
_ = enc.Encode(map[string]any{
|
||||
"recall@5": round3(recall),
|
||||
"passed": passed,
|
||||
"gate": n,
|
||||
"details": jsDetails,
|
||||
})
|
||||
} else {
|
||||
out := Dict{
|
||||
{"recall@5", round3(recall)},
|
||||
{"passed", passed},
|
||||
{"gate", n},
|
||||
{"details", details},
|
||||
}
|
||||
fmt.Print(toYAML(out, 0))
|
||||
}
|
||||
if !passed {
|
||||
return 2
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func lookupLeaf(id string) (map[string]string, string, error) {
|
||||
if conn == nil {
|
||||
return nil, "", fmt.Errorf("brain not open")
|
||||
}
|
||||
stmt, err := conn.Prepare(
|
||||
"MATCH (l:Leaf {id:$id}) RETURN l.id, l.text, l.root, l.confidence, l.source, l.type",
|
||||
)
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
defer stmt.Close()
|
||||
res, err := conn.Execute(stmt, map[string]any{"id": id})
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
if !res.HasNext() {
|
||||
return nil, "", fmt.Errorf("no leaf %s", id)
|
||||
}
|
||||
row, err := res.Next()
|
||||
if err != nil {
|
||||
return nil, "", err
|
||||
}
|
||||
vals, err := row.GetAsSlice()
|
||||
if err != nil || len(vals) < 6 {
|
||||
return nil, "", fmt.Errorf("leaf row")
|
||||
}
|
||||
meta := map[string]string{
|
||||
"id": fmt.Sprint(vals[0]),
|
||||
"root": fmt.Sprint(vals[2]),
|
||||
"confidence": fmt.Sprint(vals[3]),
|
||||
"source": fmt.Sprint(vals[4]),
|
||||
"type": fmt.Sprint(vals[5]),
|
||||
}
|
||||
return meta, fmt.Sprint(vals[1]), nil
|
||||
}
|
||||
|
||||
func leafStats() (map[string]any, error) {
|
||||
if conn == nil {
|
||||
return nil, fmt.Errorf("brain not open")
|
||||
}
|
||||
res, err := conn.Query("MATCH (l:Leaf) RETURN l.root, count(*)")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
byRoot := map[string]int{}
|
||||
total := 0
|
||||
for res.HasNext() {
|
||||
row, err := res.Next()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
vals, err := row.GetAsSlice()
|
||||
if err != nil || len(vals) < 2 {
|
||||
continue
|
||||
}
|
||||
n := int(asInt(vals[1]))
|
||||
byRoot[fmt.Sprint(vals[0])] = n
|
||||
total += n
|
||||
}
|
||||
return map[string]any{
|
||||
"total": total,
|
||||
"by_root": byRoot,
|
||||
"db": dbPath(),
|
||||
"model": ModelID,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func clip(s string, n int) string {
|
||||
if utf8.RuneCountInString(s) <= n {
|
||||
return s
|
||||
}
|
||||
return string([]rune(s)[:n])
|
||||
}
|
||||
|
||||
func round3(f float64) float64 {
|
||||
return float64(int(f*1000+0.5)) / 1000
|
||||
}
|
||||
+78
-11
@@ -56,6 +56,12 @@ func runSearch(args []string) int {
|
||||
fmt.Fprintf(os.Stderr, "search: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
if opt.Hop > 0 {
|
||||
if err := attachHops(hits, opt.Hop); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "hop: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
}
|
||||
|
||||
results := hits
|
||||
for i := range results {
|
||||
@@ -68,18 +74,25 @@ func runSearch(args []string) int {
|
||||
}
|
||||
}
|
||||
|
||||
webOut := rank.Deduce(results, query, root, opt.NoWeb, func(q string) rank.SecondSource {
|
||||
return lookupWeb(context.Background(), q)
|
||||
})
|
||||
|
||||
out := Dict{
|
||||
{"query", query},
|
||||
{"root_filter", root},
|
||||
{"count", len(results)},
|
||||
{"results", resultsToDicts(results)},
|
||||
}
|
||||
if webOut != nil {
|
||||
out = append(out, KV{"web", secondToDict(*webOut)})
|
||||
}
|
||||
|
||||
if jsonOut {
|
||||
enc := json.NewEncoder(os.Stdout)
|
||||
enc.SetIndent("", " ")
|
||||
enc.SetEscapeHTML(false)
|
||||
return b2i(enc.Encode(toJSONOut(results, query, root)))
|
||||
return b2i(enc.Encode(toJSONOut(results, query, root, webOut)))
|
||||
}
|
||||
fmt.Print(toYAML(out, 0))
|
||||
return 0
|
||||
@@ -101,6 +114,44 @@ func searchHits(query, root, repo string, limit int) ([]Hit, error) {
|
||||
return rank.RankAndFilter(fts, vec, root, repo, limit), nil
|
||||
}
|
||||
|
||||
func attachHops(hits []Hit, n int) error {
|
||||
if conn == nil {
|
||||
return fmt.Errorf("brain not open")
|
||||
}
|
||||
for i := range hits {
|
||||
var hops []rank.HopNode
|
||||
for d := 1; d <= n; d++ {
|
||||
stmt, err := conn.Prepare(rank.HopStmt(d))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
res, err := conn.Execute(stmt, map[string]any{"id": hits[i].ID})
|
||||
stmt.Close()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for res.HasNext() {
|
||||
row, err := res.Next()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
vals, err := row.GetAsSlice()
|
||||
if err != nil || len(vals) < 3 {
|
||||
continue
|
||||
}
|
||||
hops = append(hops, rank.HopNode{
|
||||
ID: fmt.Sprint(vals[0]),
|
||||
Label: rank.HopLabel(d),
|
||||
Name: fmt.Sprint(vals[1]),
|
||||
Depth: int(asInt(vals[2])),
|
||||
})
|
||||
}
|
||||
}
|
||||
hits[i].Hops = hops
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func b2i(err error) int {
|
||||
if err != nil {
|
||||
return 1
|
||||
@@ -168,21 +219,23 @@ func rowsToHits(res *lbug.QueryResult) ([]Hit, error) {
|
||||
|
||||
// JSON output types
|
||||
type jsonOut struct {
|
||||
Query string `json:"query"`
|
||||
RootFilter string `json:"root_filter"`
|
||||
Count int `json:"count"`
|
||||
Results []jsonHit `json:"results"`
|
||||
Query string `json:"query"`
|
||||
RootFilter string `json:"root_filter"`
|
||||
Count int `json:"count"`
|
||||
Results []jsonHit `json:"results"`
|
||||
Web *rank.SecondSource `json:"web,omitempty"`
|
||||
}
|
||||
|
||||
type jsonHit struct {
|
||||
ID string `json:"id"`
|
||||
Text string `json:"text"`
|
||||
Root string `json:"root"`
|
||||
Score float64 `json:"score"`
|
||||
Snippet string `json:"snippet,omitempty"`
|
||||
ID string `json:"id"`
|
||||
Text string `json:"text"`
|
||||
Root string `json:"root"`
|
||||
Score float64 `json:"score"`
|
||||
Snippet string `json:"snippet,omitempty"`
|
||||
Hops []rank.HopNode `json:"hops,omitempty"`
|
||||
}
|
||||
|
||||
func toJSONOut(hits []Hit, query, rootFilter string) *jsonOut {
|
||||
func toJSONOut(hits []Hit, query, rootFilter string, web *rank.SecondSource) *jsonOut {
|
||||
out := make([]jsonHit, len(hits))
|
||||
for i, h := range hits {
|
||||
out[i] = jsonHit{
|
||||
@@ -191,6 +244,7 @@ func toJSONOut(hits []Hit, query, rootFilter string) *jsonOut {
|
||||
Root: h.Root,
|
||||
Score: h.Score,
|
||||
Snippet: h.Snippet,
|
||||
Hops: h.Hops,
|
||||
}
|
||||
}
|
||||
return &jsonOut{
|
||||
@@ -198,6 +252,7 @@ func toJSONOut(hits []Hit, query, rootFilter string) *jsonOut {
|
||||
RootFilter: rootFilter,
|
||||
Count: len(hits),
|
||||
Results: out,
|
||||
Web: web,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -213,6 +268,18 @@ func resultsToDicts(hits []Hit) []any {
|
||||
if h.Snippet != "" {
|
||||
d = append(d, KV{"snippet", h.Snippet})
|
||||
}
|
||||
if len(h.Hops) > 0 {
|
||||
nodes := make([]any, len(h.Hops))
|
||||
for j, n := range h.Hops {
|
||||
nodes[j] = Dict{
|
||||
{"id", n.ID},
|
||||
{"label", n.Label},
|
||||
{"name", n.Name},
|
||||
{"depth", n.Depth},
|
||||
}
|
||||
}
|
||||
d = append(d, KV{"hops", nodes})
|
||||
}
|
||||
out[i] = d
|
||||
}
|
||||
return out
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
package brain
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
"github.com/eSlider/2dph/internal/brain/rank"
|
||||
"github.com/eSlider/2dph/internal/websearch"
|
||||
)
|
||||
|
||||
func lookupWeb(ctx context.Context, query string) rank.SecondSource {
|
||||
o := websearch.Lookup(ctx, query, websearch.LookupOpt{Limit: 5})
|
||||
return toSecond(o)
|
||||
}
|
||||
|
||||
func toSecond(o websearch.Output) rank.SecondSource {
|
||||
hits := make([]rank.SecondSourceHit, 0, len(o.Results))
|
||||
for _, h := range o.Results {
|
||||
hits = append(hits, rank.SecondSourceHit{
|
||||
Rank: h.Rank,
|
||||
Title: h.Title,
|
||||
URL: h.URL,
|
||||
Snippet: h.Snippet,
|
||||
Engine: h.Engine,
|
||||
})
|
||||
}
|
||||
return rank.SecondSource{
|
||||
Status: o.Status,
|
||||
Note: o.Note,
|
||||
Cached: o.Cached,
|
||||
Results: hits,
|
||||
}
|
||||
}
|
||||
|
||||
func secondToDict(w rank.SecondSource) Dict {
|
||||
d := Dict{
|
||||
{"status", w.Status},
|
||||
}
|
||||
if w.Note != "" {
|
||||
d = append(d, KV{"note", w.Note})
|
||||
}
|
||||
if w.Cached {
|
||||
d = append(d, KV{"cached", true})
|
||||
}
|
||||
rows := make([]any, 0, len(w.Results))
|
||||
for _, h := range w.Results {
|
||||
rows = append(rows, Dict{
|
||||
{"rank", h.Rank},
|
||||
{"title", h.Title},
|
||||
{"url", h.URL},
|
||||
{"snippet", h.Snippet},
|
||||
{"engine", h.Engine},
|
||||
})
|
||||
}
|
||||
d = append(d, KV{"results", rows})
|
||||
return d
|
||||
}
|
||||
@@ -0,0 +1,47 @@
|
||||
// Package duckstats runs in-process DuckDB for columnar aggregates.
|
||||
// Graph facts stay in Ladybug. Web-search KV cache stays modernc sqlite.
|
||||
package duckstats
|
||||
|
||||
import (
|
||||
"database/sql"
|
||||
"fmt"
|
||||
|
||||
_ "github.com/duckdb/duckdb-go/v2"
|
||||
)
|
||||
|
||||
type Stats struct {
|
||||
N int `json:"n"`
|
||||
Min float64 `json:"min"`
|
||||
P50 float64 `json:"p50"`
|
||||
P95 float64 `json:"p95"`
|
||||
Max float64 `json:"max"`
|
||||
Avg float64 `json:"avg"`
|
||||
}
|
||||
|
||||
func Quantiles(samples []float64) (Stats, error) {
|
||||
if len(samples) == 0 {
|
||||
return Stats{}, fmt.Errorf("duckstats: empty samples")
|
||||
}
|
||||
db, err := sql.Open("duckdb", "")
|
||||
if err != nil {
|
||||
return Stats{}, err
|
||||
}
|
||||
defer db.Close()
|
||||
var s Stats
|
||||
err = db.QueryRow(`
|
||||
SELECT count(v), min(v), quantile_cont(v, 0.5), quantile_cont(v, 0.95), max(v), avg(v)
|
||||
FROM (SELECT unnest(?) AS v)`, samples).Scan(
|
||||
&s.N, &s.Min, &s.P50, &s.P95, &s.Max, &s.Avg)
|
||||
return s, err
|
||||
}
|
||||
|
||||
func CountJSONL(path string) (int64, error) {
|
||||
db, err := sql.Open("duckdb", "")
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer db.Close()
|
||||
var n int64
|
||||
err = db.QueryRow(`SELECT count(*) FROM read_json_auto(?)`, path).Scan(&n)
|
||||
return n, err
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
package duckstats
|
||||
|
||||
import (
|
||||
"os"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestQuantilesEmpty(t *testing.T) {
|
||||
_, err := Quantiles(nil)
|
||||
if err == nil {
|
||||
t.Fatal("empty slice must error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuantilesOdd(t *testing.T) {
|
||||
s, err := Quantiles([]float64{1, 2, 3, 4, 5})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if s.N != 5 {
|
||||
t.Fatalf("n=%d", s.N)
|
||||
}
|
||||
if s.Min != 1 || s.Max != 5 {
|
||||
t.Fatalf("min=%v max=%v", s.Min, s.Max)
|
||||
}
|
||||
if s.P50 != 3 {
|
||||
t.Fatalf("p50=%v want 3", s.P50)
|
||||
}
|
||||
if s.Avg != 3 {
|
||||
t.Fatalf("avg=%v want 3", s.Avg)
|
||||
}
|
||||
if s.P95 < 4.5 || s.P95 > 5 {
|
||||
t.Fatalf("p95=%v want in [4.5,5]", s.P95)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCountJSONL(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
p := dir + "/rows.jsonl"
|
||||
body := "{\"ms\":1}\n{\"ms\":2}\n{\"ms\":3}\n"
|
||||
if err := os.WriteFile(p, []byte(body), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
n, err := CountJSONL(p)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if n != 3 {
|
||||
t.Fatalf("count=%d want 3", n)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,180 @@
|
||||
// Package gitlog reads commit history with go-git (no git binary).
|
||||
package gitlog
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"path"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/go-git/go-git/v5"
|
||||
"github.com/go-git/go-git/v5/plumbing/object"
|
||||
)
|
||||
|
||||
type Options struct {
|
||||
Limit int
|
||||
Since time.Time
|
||||
}
|
||||
|
||||
type Commit struct {
|
||||
SHA string `json:"sha"`
|
||||
Author string `json:"author"`
|
||||
Email string `json:"email"`
|
||||
Date string `json:"date"`
|
||||
Subject string `json:"subject"`
|
||||
Files []string `json:"files"`
|
||||
}
|
||||
|
||||
type Leaf struct {
|
||||
Source string `json:"source"`
|
||||
Repo string `json:"repo"`
|
||||
Heading string `json:"heading"`
|
||||
Text string `json:"text"`
|
||||
Type string `json:"type"`
|
||||
Status string `json:"status"`
|
||||
Related string `json:"related"`
|
||||
}
|
||||
|
||||
// Log walks commits from HEAD, newest first, skipping merges.
|
||||
func Log(repo string, opt Options) ([]Commit, error) {
|
||||
r, err := git.PlainOpen(repo)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
logOpt := &git.LogOptions{Order: git.LogOrderCommitterTime}
|
||||
if !opt.Since.IsZero() {
|
||||
t := opt.Since
|
||||
logOpt.Since = &t
|
||||
}
|
||||
iter, err := r.Log(logOpt)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer iter.Close()
|
||||
|
||||
var out []Commit
|
||||
err = iter.ForEach(func(c *object.Commit) error {
|
||||
if c.NumParents() > 1 {
|
||||
return nil
|
||||
}
|
||||
if opt.Limit > 0 && len(out) >= opt.Limit {
|
||||
return Stop
|
||||
}
|
||||
files, ferr := changedFiles(c)
|
||||
if ferr != nil {
|
||||
return ferr
|
||||
}
|
||||
out = append(out, Commit{
|
||||
SHA: c.Hash.String(),
|
||||
Author: c.Author.Name,
|
||||
Email: c.Author.Email,
|
||||
Date: c.Author.When.Format(time.RFC3339),
|
||||
Subject: firstLine(c.Message),
|
||||
Files: files,
|
||||
})
|
||||
return nil
|
||||
})
|
||||
if errors.Is(err, Stop) {
|
||||
err = nil
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
|
||||
// Stop ends a log walk early (limit reached).
|
||||
var Stop = fmt.Errorf("gitlog: stop")
|
||||
|
||||
func changedFiles(c *object.Commit) ([]string, error) {
|
||||
var names []string
|
||||
if c.NumParents() == 0 {
|
||||
t, err := c.Tree()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
err = t.Files().ForEach(func(f *object.File) error {
|
||||
names = append(names, f.Name)
|
||||
return nil
|
||||
})
|
||||
sort.Strings(names)
|
||||
return names, err
|
||||
}
|
||||
parent, err := c.Parent(0)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
from, err := parent.Tree()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
to, err := c.Tree()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
changes, err := object.DiffTree(from, to)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
for _, ch := range changes {
|
||||
name := ch.To.Name
|
||||
if name == "" {
|
||||
name = ch.From.Name
|
||||
}
|
||||
if name != "" {
|
||||
names = append(names, name)
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
return names, nil
|
||||
}
|
||||
|
||||
func firstLine(msg string) string {
|
||||
msg = strings.ReplaceAll(msg, "\r\n", "\n")
|
||||
if i := strings.IndexByte(msg, '\n'); i >= 0 {
|
||||
return strings.TrimSpace(msg[:i])
|
||||
}
|
||||
return strings.TrimSpace(msg)
|
||||
}
|
||||
|
||||
func ToLeaf(c Commit, repo string) Leaf {
|
||||
short := c.SHA
|
||||
if len(short) > 12 {
|
||||
short = short[:12]
|
||||
}
|
||||
head := fmt.Sprintf("commit %s — %s", short, c.Subject)
|
||||
body := []string{
|
||||
fmt.Sprintf("commit %s in %s — %s", short, repo, c.Subject),
|
||||
fmt.Sprintf("Author: %s <%s>", c.Author, c.Email),
|
||||
fmt.Sprintf("Date: %s", c.Date),
|
||||
}
|
||||
if len(c.Files) > 0 {
|
||||
body = append(body, "Changing: "+strings.Join(c.Files, ", "))
|
||||
}
|
||||
return Leaf{
|
||||
Source: repo + "@" + c.SHA,
|
||||
Repo: repo,
|
||||
Heading: head,
|
||||
Text: strings.Join(body, "\n"),
|
||||
Type: "commit",
|
||||
Status: "current",
|
||||
Related: strings.Join(c.Files, ","),
|
||||
}
|
||||
}
|
||||
|
||||
func RepoName(repo string) (string, error) {
|
||||
r, err := git.PlainOpen(repo)
|
||||
if err != nil {
|
||||
return filepath.Base(repo), err
|
||||
}
|
||||
rem, err := r.Remote("origin")
|
||||
if err != nil {
|
||||
return filepath.Base(repo), nil
|
||||
}
|
||||
urls := rem.Config().URLs
|
||||
if len(urls) == 0 {
|
||||
return filepath.Base(repo), nil
|
||||
}
|
||||
u := strings.TrimSuffix(strings.TrimSuffix(urls[0], "/"), ".git")
|
||||
return path.Base(strings.ReplaceAll(u, "\\", "/")), nil
|
||||
}
|
||||
@@ -0,0 +1,243 @@
|
||||
package gitlog
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/go-git/go-git/v5"
|
||||
"github.com/go-git/go-git/v5/config"
|
||||
"github.com/go-git/go-git/v5/plumbing"
|
||||
"github.com/go-git/go-git/v5/plumbing/object"
|
||||
)
|
||||
|
||||
func TestLogReadsCommitsWithoutGitBinary(t *testing.T) {
|
||||
dir := initRepo(t, []commitSpec{
|
||||
{
|
||||
when: time.Date(2026, 8, 10, 12, 0, 0, 0, time.FixedZone("CEST", 3600)),
|
||||
name: "Ada Lovelace",
|
||||
email: "ada@example.com",
|
||||
subject: "feat: first commit",
|
||||
files: map[string]string{"README.md": "hi\n", "src/main.c": "int main(){}\n"},
|
||||
},
|
||||
{
|
||||
when: time.Date(2026, 8, 11, 9, 30, 0, 0, time.FixedZone("CEST", 3600)),
|
||||
name: "Bob Babbage",
|
||||
email: "bob@example.com",
|
||||
subject: "fix: typo",
|
||||
files: map[string]string{"docs/notes.md": "note\n"},
|
||||
},
|
||||
})
|
||||
|
||||
cs, err := Log(dir, Options{})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(cs) != 2 {
|
||||
t.Fatalf("commits = %d, want 2", len(cs))
|
||||
}
|
||||
if cs[0].Subject != "fix: typo" {
|
||||
t.Fatalf("head subject = %q, want fix: typo", cs[0].Subject)
|
||||
}
|
||||
if cs[1].Author != "Ada Lovelace" || cs[1].Email != "ada@example.com" {
|
||||
t.Fatalf("author = %s <%s>", cs[1].Author, cs[1].Email)
|
||||
}
|
||||
sort.Strings(cs[1].Files)
|
||||
if got := cs[1].Files; len(got) != 2 || got[0] != "README.md" || got[1] != "src/main.c" {
|
||||
t.Fatalf("first commit files = %v", got)
|
||||
}
|
||||
if cs[0].Files[0] != "docs/notes.md" {
|
||||
t.Fatalf("second commit files = %v", cs[0].Files)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLogSkipsMerges(t *testing.T) {
|
||||
dir := initRepo(t, []commitSpec{{
|
||||
when: time.Now(), name: "Ada Lovelace", email: "ada@example.com",
|
||||
subject: "base", files: map[string]string{"a.txt": "a\n"},
|
||||
}})
|
||||
r, err := git.PlainOpen(dir)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
head, err := r.Head()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
c, err := r.CommitObject(head.Hash())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Second parent: duplicate the same tree so we do not need a real branch.
|
||||
merge := &object.Commit{
|
||||
Author: object.Signature{Name: "Ada Lovelace", Email: "ada@example.com", When: time.Now()},
|
||||
Committer: object.Signature{Name: "Ada Lovelace", Email: "ada@example.com", When: time.Now()},
|
||||
Message: "merge",
|
||||
TreeHash: c.TreeHash,
|
||||
ParentHashes: []plumbing.Hash{c.Hash, c.Hash},
|
||||
}
|
||||
obj := r.Storer.NewEncodedObject()
|
||||
if err := merge.Encode(obj); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
h, err := r.Storer.SetEncodedObject(obj)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := r.Storer.SetReference(plumbing.NewHashReference(head.Name(), h)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
cs, err := Log(dir, Options{})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, x := range cs {
|
||||
if x.Subject == "merge" {
|
||||
t.Fatal("merge commit was not skipped")
|
||||
}
|
||||
}
|
||||
if len(cs) != 1 || cs[0].Subject != "base" {
|
||||
t.Fatalf("after skip merges: %+v", subjects(cs))
|
||||
}
|
||||
}
|
||||
|
||||
func TestLogSinceAndLimit(t *testing.T) {
|
||||
old := time.Date(2025, 1, 1, 0, 0, 0, 0, time.UTC)
|
||||
neu := time.Date(2026, 6, 1, 0, 0, 0, 0, time.UTC)
|
||||
dir := initRepo(t, []commitSpec{
|
||||
{when: old, name: "Ada Lovelace", email: "ada@example.com", subject: "old", files: map[string]string{"old.md": "x"}},
|
||||
{when: neu, name: "Ada Lovelace", email: "ada@example.com", subject: "new", files: map[string]string{"new.md": "y"}},
|
||||
})
|
||||
cs, err := Log(dir, Options{Since: neu.Add(-time.Hour)})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(cs) != 1 || cs[0].Subject != "new" {
|
||||
t.Fatalf("since filter: %v", subjects(cs))
|
||||
}
|
||||
cs, err = Log(dir, Options{Limit: 1})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(cs) != 1 {
|
||||
t.Fatalf("limit=1 got %d", len(cs))
|
||||
}
|
||||
}
|
||||
|
||||
func TestLeafShape(t *testing.T) {
|
||||
c := Commit{
|
||||
SHA: "a1b2c3d4e5f6aaaa",
|
||||
Author: "Ada Lovelace",
|
||||
Email: "ada@example.com",
|
||||
Date: "2026-08-10T12:00:00+01:00",
|
||||
Subject: "feat: first commit",
|
||||
Files: []string{"README.md", "src/main.c"},
|
||||
}
|
||||
lf := ToLeaf(c, "sample-repo")
|
||||
if lf.Type != "commit" || lf.Repo != "sample-repo" {
|
||||
t.Fatalf("leaf meta = %+v", lf)
|
||||
}
|
||||
if lf.Source != "sample-repo@a1b2c3d4e5f6aaaa" {
|
||||
t.Fatalf("source = %s", lf.Source)
|
||||
}
|
||||
if lf.Related != "README.md,src/main.c" {
|
||||
t.Fatalf("related = %s", lf.Related)
|
||||
}
|
||||
if lf.Heading != "commit a1b2c3d4e5f6 — feat: first commit" {
|
||||
t.Fatalf("heading = %q", lf.Heading)
|
||||
}
|
||||
if !strings.Contains(lf.Text, "Ada Lovelace") || !strings.Contains(lf.Text, "README.md") {
|
||||
t.Fatalf("text = %s", lf.Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRepoNameFromOrigin(t *testing.T) {
|
||||
dir := initRepo(t, []commitSpec{{
|
||||
when: time.Now(), name: "Ada Lovelace", email: "ada@example.com",
|
||||
subject: "init", files: map[string]string{"README.md": "x"},
|
||||
}})
|
||||
r, err := git.PlainOpen(dir)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := r.CreateRemote(&config.RemoteConfig{
|
||||
Name: "origin",
|
||||
URLs: []string{"https://git.example.com/eSlider/sample-repo.git"},
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
name, err := RepoName(dir)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if name != "sample-repo" {
|
||||
t.Fatalf("RepoName = %q, want sample-repo", name)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRepoNameFallsBackToDir(t *testing.T) {
|
||||
dir := initRepo(t, []commitSpec{{
|
||||
when: time.Now(), name: "Ada Lovelace", email: "ada@example.com",
|
||||
subject: "init", files: map[string]string{"README.md": "x"},
|
||||
}})
|
||||
name, err := RepoName(dir)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if name != filepath.Base(dir) {
|
||||
t.Fatalf("RepoName = %q, want %s", name, filepath.Base(dir))
|
||||
}
|
||||
}
|
||||
|
||||
type commitSpec struct {
|
||||
when time.Time
|
||||
name string
|
||||
email string
|
||||
subject string
|
||||
files map[string]string
|
||||
}
|
||||
|
||||
func initRepo(t *testing.T, specs []commitSpec) string {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
r, err := git.PlainInit(dir, false)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
w, err := r.Worktree()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, s := range specs {
|
||||
for path, body := range s.files {
|
||||
full := filepath.Join(dir, path)
|
||||
if err := os.MkdirAll(filepath.Dir(full), 0o755); err != nil && !os.IsExist(err) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(full, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := w.Add(path); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if _, err := w.Commit(s.subject, &git.CommitOptions{
|
||||
Author: &object.Signature{Name: s.name, Email: s.email, When: s.when},
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
return dir
|
||||
}
|
||||
|
||||
func subjects(cs []Commit) []string {
|
||||
out := make([]string, len(cs))
|
||||
for i, c := range cs {
|
||||
out[i] = c.Subject
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,193 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
type rpcReq struct {
|
||||
JSONRPC string `json:"jsonrpc"`
|
||||
ID json.RawMessage `json:"id"`
|
||||
Method string `json:"method"`
|
||||
Params json.RawMessage `json:"params"`
|
||||
}
|
||||
|
||||
type rpcErr struct {
|
||||
Code int `json:"code"`
|
||||
Message string `json:"message"`
|
||||
}
|
||||
|
||||
func (s *Server) handleOpenAPI(w http.ResponseWriter, _ *http.Request) {
|
||||
writeJSON(w, http.StatusOK, OpenAPI())
|
||||
}
|
||||
|
||||
func (s *Server) handleMCP(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != http.MethodPost {
|
||||
writeJSON(w, http.StatusMethodNotAllowed, map[string]any{"error": "POST JSON-RPC"})
|
||||
return
|
||||
}
|
||||
raw, err := io.ReadAll(io.LimitReader(r.Body, 1<<20))
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusBadRequest, map[string]any{"error": "read body"})
|
||||
return
|
||||
}
|
||||
var req rpcReq
|
||||
if err := json.Unmarshal(raw, &req); err != nil {
|
||||
writeJSON(w, http.StatusOK, rpcResult(nil, nil, &rpcErr{-32700, "parse error"}))
|
||||
return
|
||||
}
|
||||
result, rpcErrv, callErr := s.mcpDispatch(r, req)
|
||||
if callErr != nil {
|
||||
writeJSON(w, http.StatusOK, rpcResult(req.ID, nil, &rpcErr{-32603, callErr.Error()}))
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, rpcResult(req.ID, result, rpcErrv))
|
||||
}
|
||||
|
||||
func (s *Server) mcpDispatch(r *http.Request, req rpcReq) (any, *rpcErr, error) {
|
||||
switch req.Method {
|
||||
case "initialize":
|
||||
return map[string]any{
|
||||
"protocolVersion": "2024-11-05",
|
||||
"capabilities": map[string]any{"tools": map[string]any{}},
|
||||
"serverInfo": map[string]any{"name": "2dph", "version": "1"},
|
||||
}, nil, nil
|
||||
case "notifications/initialized", "notifications/cancelled":
|
||||
return map[string]any{}, nil, nil
|
||||
case "tools/list":
|
||||
return map[string]any{"tools": MCPTools()}, nil, nil
|
||||
case "tools/call":
|
||||
out, err := s.mcpCall(r, req.Params)
|
||||
return out, nil, err
|
||||
case "ping":
|
||||
return map[string]any{}, nil, nil
|
||||
default:
|
||||
return nil, &rpcErr{-32601, "method not found"}, nil
|
||||
}
|
||||
}
|
||||
|
||||
func (s *Server) mcpCall(r *http.Request, params json.RawMessage) (any, error) {
|
||||
var p struct {
|
||||
Name string `json:"name"`
|
||||
Arguments map[string]any `json:"arguments"`
|
||||
}
|
||||
if err := json.Unmarshal(params, &p); err != nil {
|
||||
return nil, fmt.Errorf("params")
|
||||
}
|
||||
if p.Arguments == nil {
|
||||
p.Arguments = map[string]any{}
|
||||
}
|
||||
var (
|
||||
body []byte
|
||||
err error
|
||||
)
|
||||
switch p.Name {
|
||||
case "search":
|
||||
q := strings.TrimSpace(fmt.Sprint(p.Arguments["q"]))
|
||||
if q == "" || q == "<nil>" {
|
||||
return mcpText(`{"error":"q required"}`, true), nil
|
||||
}
|
||||
limit := 10
|
||||
if raw, ok := p.Arguments["n"]; ok {
|
||||
switch n := raw.(type) {
|
||||
case float64:
|
||||
limit = int(n)
|
||||
case string:
|
||||
if v, e := strconv.Atoi(n); e == nil {
|
||||
limit = v
|
||||
}
|
||||
}
|
||||
}
|
||||
if limit < 1 || limit > 100 {
|
||||
return mcpText(`{"error":"n must be int 1..100"}`, true), nil
|
||||
}
|
||||
if !s.tryAcquire(r) {
|
||||
return nil, fmt.Errorf("cancelled")
|
||||
}
|
||||
defer s.release()
|
||||
body, err = s.api.Search(r.Context(), q, limit)
|
||||
case "get":
|
||||
id := strings.TrimSpace(fmt.Sprint(p.Arguments["id"]))
|
||||
if id == "" || id == "<nil>" {
|
||||
return mcpText(`{"error":"id required"}`, true), nil
|
||||
}
|
||||
full := false
|
||||
switch v := p.Arguments["body"].(type) {
|
||||
case bool:
|
||||
full = v
|
||||
case string:
|
||||
full = v == "1" || v == "true"
|
||||
}
|
||||
if !s.tryAcquire(r) {
|
||||
return nil, fmt.Errorf("cancelled")
|
||||
}
|
||||
defer s.release()
|
||||
body, err = s.api.Get(r.Context(), id, full)
|
||||
case "stats":
|
||||
if !s.tryAcquire(r) {
|
||||
return nil, fmt.Errorf("cancelled")
|
||||
}
|
||||
defer s.release()
|
||||
body, err = s.api.Stats(r.Context())
|
||||
case "audit":
|
||||
if !s.tryAcquire(r) {
|
||||
return nil, fmt.Errorf("cancelled")
|
||||
}
|
||||
defer s.release()
|
||||
body, err = s.api.Audit(r.Context())
|
||||
case "ingest":
|
||||
if !s.tryAcquire(r) {
|
||||
return nil, fmt.Errorf("cancelled")
|
||||
}
|
||||
defer s.release()
|
||||
var payload []byte
|
||||
text := strings.TrimSpace(fmt.Sprint(p.Arguments["text"]))
|
||||
if text != "" && text != "<nil>" {
|
||||
payload, err = json.Marshal(p.Arguments)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
body, err = s.api.Ingest(r.Context(), payload)
|
||||
default:
|
||||
return nil, fmt.Errorf("unknown tool %s", p.Name)
|
||||
}
|
||||
if err != nil {
|
||||
return mcpText(err.Error(), true), nil
|
||||
}
|
||||
return mcpText(string(body), false), nil
|
||||
}
|
||||
|
||||
func mcpText(text string, isError bool) map[string]any {
|
||||
return map[string]any{
|
||||
"content": []any{map[string]any{"type": "text", "text": text}},
|
||||
"isError": isError,
|
||||
}
|
||||
}
|
||||
|
||||
type rpcResp struct {
|
||||
JSONRPC string `json:"jsonrpc"`
|
||||
ID json.RawMessage `json:"id"`
|
||||
Result any `json:"result,omitempty"`
|
||||
Error *rpcErr `json:"error,omitempty"`
|
||||
}
|
||||
|
||||
func rpcResult(id json.RawMessage, result any, err *rpcErr) rpcResp {
|
||||
out := rpcResp{JSONRPC: "2.0", ID: id}
|
||||
if len(id) == 0 {
|
||||
out.ID = []byte("null")
|
||||
}
|
||||
if err != nil {
|
||||
out.Error = err
|
||||
return out
|
||||
}
|
||||
if result == nil {
|
||||
result = map[string]any{}
|
||||
}
|
||||
out.Result = result
|
||||
return out
|
||||
}
|
||||
+65
-13
@@ -11,6 +11,7 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"io"
|
||||
"log"
|
||||
"net/http"
|
||||
"os"
|
||||
@@ -27,7 +28,7 @@ type API interface {
|
||||
Get(ctx context.Context, id string, body bool) ([]byte, error)
|
||||
Stats(ctx context.Context) ([]byte, error)
|
||||
Audit(ctx context.Context) ([]byte, error)
|
||||
Ingest(ctx context.Context) ([]byte, error)
|
||||
Ingest(ctx context.Context, body []byte) ([]byte, error)
|
||||
}
|
||||
|
||||
type Server struct {
|
||||
@@ -48,18 +49,22 @@ func NewServer(api API, workers int) http.Handler {
|
||||
|
||||
func (s *Server) ServeHTTP(w http.ResponseWriter, r *http.Request) {
|
||||
switch r.URL.Path {
|
||||
case "/health":
|
||||
case PathHealth:
|
||||
writeJSON(w, http.StatusOK, map[string]any{"status": "ok"})
|
||||
case "/search":
|
||||
case PathSearch:
|
||||
s.handleSearch(w, r)
|
||||
case "/get":
|
||||
case PathGet:
|
||||
s.handleGet(w, r)
|
||||
case "/stats":
|
||||
case PathStats:
|
||||
s.handleJSON(w, r, s.api.Stats)
|
||||
case "/audit":
|
||||
case PathAudit:
|
||||
s.handleJSON(w, r, s.api.Audit)
|
||||
case "/ingest":
|
||||
s.handleJSON(w, r, s.api.Ingest)
|
||||
case PathIngest:
|
||||
s.handleIngest(w, r)
|
||||
case PathOpenAPI:
|
||||
s.handleOpenAPI(w, r)
|
||||
case PathMCP:
|
||||
s.handleMCP(w, r)
|
||||
default:
|
||||
writeJSON(w, http.StatusNotFound, map[string]any{"error": "not found"})
|
||||
}
|
||||
@@ -112,6 +117,34 @@ func (s *Server) handleJSON(w http.ResponseWriter, r *http.Request, fn func(cont
|
||||
writeAPI(w, body, err)
|
||||
}
|
||||
|
||||
func (s *Server) handleIngest(w http.ResponseWriter, r *http.Request) {
|
||||
var raw []byte
|
||||
if r.Method == http.MethodPost {
|
||||
b, err := io.ReadAll(io.LimitReader(r.Body, 1<<20))
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusBadRequest, map[string]any{"error": "read body"})
|
||||
return
|
||||
}
|
||||
raw = b
|
||||
}
|
||||
if !s.acquire(w, r) {
|
||||
return
|
||||
}
|
||||
defer s.release()
|
||||
body, err := s.api.Ingest(r.Context(), raw)
|
||||
writeAPI(w, body, err)
|
||||
}
|
||||
|
||||
func (s *Server) tryAcquire(r *http.Request) bool {
|
||||
return s.acquire(nopWriter{}, r)
|
||||
}
|
||||
|
||||
type nopWriter struct{}
|
||||
|
||||
func (nopWriter) Header() http.Header { return http.Header{} }
|
||||
func (nopWriter) Write([]byte) (int, error) { return 0, nil }
|
||||
func (nopWriter) WriteHeader(int) {}
|
||||
|
||||
func (s *Server) acquire(w http.ResponseWriter, r *http.Request) bool {
|
||||
select {
|
||||
case s.semaphore <- struct{}{}:
|
||||
@@ -177,11 +210,30 @@ func (ExecSearcher) Get(context.Context, string, bool) ([]byte, error) {
|
||||
}
|
||||
func (ExecSearcher) Stats(context.Context) ([]byte, error) { return nil, errUnimplemented }
|
||||
func (ExecSearcher) Audit(context.Context) ([]byte, error) { return nil, errUnimplemented }
|
||||
func (ExecSearcher) Ingest(context.Context) ([]byte, error) {
|
||||
return json.Marshal(map[string]any{
|
||||
"mode": "rebuild",
|
||||
"command": "bin/brain/index.go --rebuild",
|
||||
})
|
||||
func (b ExecSearcher) Ingest(ctx context.Context, body []byte) ([]byte, error) {
|
||||
if len(strings.TrimSpace(string(body))) == 0 {
|
||||
return json.Marshal(map[string]any{
|
||||
"mode": "add",
|
||||
"command": "bin/brain/add.go",
|
||||
"rebuild": "bin/brain/index.go --rebuild",
|
||||
})
|
||||
}
|
||||
root := os.Getenv("KB_ROOT")
|
||||
if root == "" {
|
||||
root = "."
|
||||
}
|
||||
cmd := exec.CommandContext(ctx, filepath.Join(root, "bin/kb/add"), "--json")
|
||||
cmd.Stdin = strings.NewReader(string(body))
|
||||
cmd.Dir = root
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
var exitErr *exec.ExitError
|
||||
if errors.As(err, &exitErr) {
|
||||
return nil, errors.New("add failed: " + strings.TrimSpace(string(exitErr.Stderr)))
|
||||
}
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func defaultSearchCmd(root string) string {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
@@ -64,8 +65,11 @@ func (f *fakeSearcher) Audit(context.Context) ([]byte, error) {
|
||||
return []byte(`{"status":"ok"}`), nil
|
||||
}
|
||||
|
||||
func (f *fakeSearcher) Ingest(context.Context) ([]byte, error) {
|
||||
return []byte(`{"mode":"rebuild","command":"bin/brain/index.go --rebuild"}`), nil
|
||||
func (f *fakeSearcher) Ingest(_ context.Context, body []byte) ([]byte, error) {
|
||||
if len(bytes.TrimSpace(body)) == 0 {
|
||||
return []byte(`{"mode":"add","command":"bin/brain/add.go"}`), nil
|
||||
}
|
||||
return []byte(`{"mode":"add","ids":["fake-leaf"]}`), nil
|
||||
}
|
||||
|
||||
func (f *fakeSearcher) count() int {
|
||||
@@ -193,6 +197,27 @@ func TestStatsAuditIngest(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestIngestIsAddNotRebuildHint(t *testing.T) {
|
||||
h := NewServer(&fakeSearcher{}, 1)
|
||||
code, body := get(t, h, "/ingest")
|
||||
if code != http.StatusOK {
|
||||
t.Fatalf("GET /ingest code = %d body=%s", code, body)
|
||||
}
|
||||
if strings.Contains(string(body), `"add":"v2"`) || strings.Contains(string(body), "write is v2") {
|
||||
t.Fatalf("GET /ingest still a v2 hint: %s", body)
|
||||
}
|
||||
if !strings.Contains(string(body), "bin/brain/add.go") {
|
||||
t.Fatalf("GET /ingest should name add.go: %s", body)
|
||||
}
|
||||
code, body = postJSON(t, h, "/ingest", `{"text":"hello","root":"info","source":"t"}`)
|
||||
if code != http.StatusOK {
|
||||
t.Fatalf("POST /ingest code = %d body=%s", code, body)
|
||||
}
|
||||
if !strings.Contains(string(body), "fake-leaf") {
|
||||
t.Fatalf("POST /ingest should add: %s", body)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPPackageDoesNotExecPython(t *testing.T) {
|
||||
raw, err := os.ReadFile("server.go")
|
||||
if err != nil {
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
package httpapi
|
||||
|
||||
import "strings"
|
||||
|
||||
// Shared HTTP surface: OpenAPI paths and MCP tools are generated from Ops.
|
||||
// ServeHTTP must keep the same path strings.
|
||||
|
||||
type Param struct {
|
||||
Name, In, Type, Description string
|
||||
Required bool
|
||||
}
|
||||
|
||||
type Op struct {
|
||||
Path, Method, ID, Summary string
|
||||
Params []Param
|
||||
MCP bool
|
||||
}
|
||||
|
||||
const (
|
||||
PathHealth = "/health"
|
||||
PathSearch = "/search"
|
||||
PathGet = "/get"
|
||||
PathStats = "/stats"
|
||||
PathAudit = "/audit"
|
||||
PathIngest = "/ingest"
|
||||
PathOpenAPI = "/openapi.json"
|
||||
PathMCP = "/mcp"
|
||||
)
|
||||
|
||||
var Ops = []Op{
|
||||
{Path: PathHealth, Method: "get", ID: "health", Summary: "liveness"},
|
||||
{
|
||||
Path: PathSearch, Method: "get", ID: "search", Summary: "deduction search (facts → info → web)",
|
||||
MCP: true,
|
||||
Params: []Param{
|
||||
{Name: "q", In: "query", Type: "string", Description: "search query", Required: true},
|
||||
{Name: "n", In: "query", Type: "integer", Description: "hit limit 1..100 (default 10)"},
|
||||
},
|
||||
},
|
||||
{
|
||||
Path: PathGet, Method: "get", ID: "get", Summary: "read one leaf by id",
|
||||
MCP: true,
|
||||
Params: []Param{
|
||||
{Name: "id", In: "query", Type: "string", Description: "leaf id", Required: true},
|
||||
{Name: "body", In: "query", Type: "boolean", Description: "include full text"},
|
||||
},
|
||||
},
|
||||
{Path: PathStats, Method: "get", ID: "stats", Summary: "index health", MCP: true},
|
||||
{Path: PathAudit, Method: "get", ID: "audit", Summary: "facts confidence histogram", MCP: true},
|
||||
{
|
||||
Path: PathIngest, Method: "post", ID: "ingest", Summary: "add a leaf without rebuild",
|
||||
MCP: true,
|
||||
Params: []Param{
|
||||
{Name: "text", In: "query", Type: "string", Description: "leaf text (omit for CLI hint)"},
|
||||
{Name: "root", In: "query", Type: "string", Description: "facts or info (default info)"},
|
||||
{Name: "source", In: "query", Type: "string", Description: "evidence pointer; facts need two sources"},
|
||||
},
|
||||
},
|
||||
{Path: PathOpenAPI, Method: "get", ID: "openapi", Summary: "OpenAPI 3 document for this server"},
|
||||
}
|
||||
|
||||
func OpenAPI() map[string]any {
|
||||
paths := map[string]any{}
|
||||
for _, op := range Ops {
|
||||
params := make([]any, 0, len(op.Params))
|
||||
for _, p := range op.Params {
|
||||
params = append(params, map[string]any{
|
||||
"name": p.Name,
|
||||
"in": p.In,
|
||||
"required": p.Required,
|
||||
"description": p.Description,
|
||||
"schema": map[string]any{"type": p.Type},
|
||||
})
|
||||
}
|
||||
item := map[string]any{
|
||||
"operationId": op.ID,
|
||||
"summary": op.Summary,
|
||||
"responses": map[string]any{
|
||||
"200": map[string]any{
|
||||
"description": "JSON",
|
||||
"content": map[string]any{
|
||||
"application/json": map[string]any{
|
||||
"schema": map[string]any{"type": "object"},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
if len(params) > 0 {
|
||||
item["parameters"] = params
|
||||
}
|
||||
paths[op.Path] = map[string]any{op.Method: item}
|
||||
}
|
||||
return map[string]any{
|
||||
"openapi": "3.0.3",
|
||||
"info": map[string]any{
|
||||
"title": "2dph brain",
|
||||
"version": "1",
|
||||
"description": "Same handlers as bin/brain/serve.go. MCP tools at POST /mcp match these paths.",
|
||||
},
|
||||
"paths": paths,
|
||||
}
|
||||
}
|
||||
|
||||
type MCPTool struct {
|
||||
Name string `json:"name"`
|
||||
Description string `json:"description"`
|
||||
InputSchema map[string]any `json:"inputSchema"`
|
||||
}
|
||||
|
||||
func MCPTools() []MCPTool {
|
||||
out := make([]MCPTool, 0, len(Ops))
|
||||
for _, op := range Ops {
|
||||
if !op.MCP {
|
||||
continue
|
||||
}
|
||||
props := map[string]any{}
|
||||
var required []string
|
||||
for _, p := range op.Params {
|
||||
props[p.Name] = map[string]any{"type": p.Type, "description": p.Description}
|
||||
if p.Required {
|
||||
required = append(required, p.Name)
|
||||
}
|
||||
}
|
||||
schema := map[string]any{"type": "object", "properties": props}
|
||||
if len(required) > 0 {
|
||||
schema["required"] = required
|
||||
}
|
||||
out = append(out, MCPTool{
|
||||
Name: op.ID,
|
||||
Description: op.Summary,
|
||||
InputSchema: schema,
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// SkillMarkdown is the Cursor skill fragment generated from Ops/MCPTools.
|
||||
func SkillMarkdown() string {
|
||||
var b strings.Builder
|
||||
b.WriteString("# brain HTTP / MCP tools\n\n")
|
||||
b.WriteString("Generated from `internal/httpapi.Ops`. Do not edit by hand.\n\n")
|
||||
b.WriteString("Serve: `bin/brain/serve.go` (`GET /openapi.json`, `POST /mcp`).\n\n")
|
||||
for _, t := range MCPTools() {
|
||||
b.WriteString("- `")
|
||||
b.WriteString(t.Name)
|
||||
b.WriteString("` — ")
|
||||
b.WriteString(t.Description)
|
||||
b.WriteString("\n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
@@ -0,0 +1,112 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestOpenAPIIncludesCorePaths(t *testing.T) {
|
||||
doc := OpenAPI()
|
||||
raw, err := json.Marshal(doc)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
paths, _ := doc["paths"].(map[string]any)
|
||||
for _, p := range []string{"/search", "/get", "/stats", "/audit"} {
|
||||
if _, ok := paths[p]; !ok {
|
||||
t.Fatalf("openapi missing path %s (%s)", p, raw)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestMCPToolsMatchOpenAPIPaths(t *testing.T) {
|
||||
paths, _ := OpenAPI()["paths"].(map[string]any)
|
||||
tools := MCPTools()
|
||||
if len(tools) == 0 {
|
||||
t.Fatal("no MCP tools")
|
||||
}
|
||||
names := map[string]bool{}
|
||||
for _, tool := range tools {
|
||||
names[tool.Name] = true
|
||||
path := "/" + tool.Name
|
||||
if _, ok := paths[path]; !ok {
|
||||
t.Fatalf("MCP tool %s has no OpenAPI path %s", tool.Name, path)
|
||||
}
|
||||
}
|
||||
for _, need := range []string{"search", "get", "stats", "audit", "ingest"} {
|
||||
if !names[need] {
|
||||
t.Fatalf("MCP tools missing %s: %v", need, names)
|
||||
}
|
||||
}
|
||||
var ingest MCPTool
|
||||
for _, tool := range tools {
|
||||
if tool.Name == "ingest" {
|
||||
ingest = tool
|
||||
break
|
||||
}
|
||||
}
|
||||
if strings.Contains(ingest.Description, "v2") {
|
||||
t.Fatalf("ingest still a v2 hint: %s", ingest.Description)
|
||||
}
|
||||
if !strings.Contains(ingest.Description, "add") {
|
||||
t.Fatalf("ingest should describe add: %s", ingest.Description)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenAPIHTTP(t *testing.T) {
|
||||
h := NewServer(&fakeSearcher{}, 1)
|
||||
code, body := get(t, h, "/openapi.json")
|
||||
if code != http.StatusOK {
|
||||
t.Fatalf("code = %d body=%s", code, body)
|
||||
}
|
||||
var doc map[string]any
|
||||
if err := json.Unmarshal(body, &doc); err != nil {
|
||||
t.Fatalf("not json: %v", err)
|
||||
}
|
||||
if doc["openapi"] == nil {
|
||||
t.Fatalf("missing openapi version: %s", body)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMCPToolsListAndCall(t *testing.T) {
|
||||
h := NewServer(&fakeSearcher{}, 1)
|
||||
code, body := postJSON(t, h, "/mcp", `{"jsonrpc":"2.0","id":1,"method":"tools/list"}`)
|
||||
if code != http.StatusOK {
|
||||
t.Fatalf("list code = %d body=%s", code, body)
|
||||
}
|
||||
if !strings.Contains(string(body), `"search"`) {
|
||||
t.Fatalf("tools/list missing search: %s", body)
|
||||
}
|
||||
code, body = postJSON(t, h, "/mcp", `{"jsonrpc":"2.0","id":2,"method":"tools/call","params":{"name":"search","arguments":{"q":"matrix","n":3}}}`)
|
||||
if code != http.StatusOK {
|
||||
t.Fatalf("call code = %d body=%s", code, body)
|
||||
}
|
||||
if !strings.Contains(string(body), "matrix") {
|
||||
t.Fatalf("search call body %s", body)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSkillMarkdownMatchesCommittedFile(t *testing.T) {
|
||||
want, err := os.ReadFile(filepath.Join("..", "..", "skills", "brain", "tools.md"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got := SkillMarkdown()
|
||||
if got != string(want) {
|
||||
t.Fatalf("skills/brain/tools.md stale; regenerate from SkillMarkdown()\n--- got ---\n%s\n--- want ---\n%s", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func postJSON(t *testing.T, h http.Handler, path, raw string) (int, []byte) {
|
||||
t.Helper()
|
||||
req := httptest.NewRequest(http.MethodPost, path, strings.NewReader(raw))
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
rec := httptest.NewRecorder()
|
||||
h.ServeHTTP(rec, req)
|
||||
return rec.Code, rec.Body.Bytes()
|
||||
}
|
||||
@@ -0,0 +1,146 @@
|
||||
package mdleaves
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strings"
|
||||
)
|
||||
|
||||
type Leaf struct {
|
||||
Source string `json:"source"`
|
||||
Repo string `json:"repo"`
|
||||
Heading string `json:"heading"`
|
||||
Text string `json:"text"`
|
||||
Type string `json:"type"`
|
||||
Status string `json:"status"`
|
||||
Related string `json:"related"`
|
||||
}
|
||||
|
||||
type chunk struct {
|
||||
Heading string
|
||||
Text string
|
||||
}
|
||||
|
||||
var (
|
||||
h1 = regexp.MustCompile(`^# \S`)
|
||||
h2 = regexp.MustCompile(`^## \S`)
|
||||
)
|
||||
|
||||
func ExtractFrontmatter(text string) (map[string]string, string) {
|
||||
if !strings.HasPrefix(text, "---") {
|
||||
return map[string]string{}, text
|
||||
}
|
||||
end := strings.Index(text[3:], "\n---")
|
||||
if end == -1 {
|
||||
return map[string]string{}, text
|
||||
}
|
||||
fm := strings.TrimSpace(text[3 : 3+end])
|
||||
body := text[3+end+4:]
|
||||
meta := map[string]string{}
|
||||
for _, line := range strings.Split(fm, "\n") {
|
||||
key, value, ok := strings.Cut(line, ":")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
meta[strings.TrimSpace(key)] = strings.Trim(strings.TrimSpace(value), `"'`)
|
||||
}
|
||||
return meta, body
|
||||
}
|
||||
|
||||
func SplitLeafs(meta map[string]string, body string) []chunk {
|
||||
lines := strings.Split(body, "\n")
|
||||
title := ""
|
||||
type hdr struct {
|
||||
heading string
|
||||
start int
|
||||
}
|
||||
var headers []hdr
|
||||
for i, line := range lines {
|
||||
switch {
|
||||
case h1.MatchString(line):
|
||||
title = strings.TrimSpace(strings.TrimLeft(line, "#"))
|
||||
case h2.MatchString(line):
|
||||
headers = append(headers, hdr{strings.TrimSpace(strings.TrimLeft(line, "#")), i})
|
||||
}
|
||||
}
|
||||
if len(headers) == 0 {
|
||||
var kept []string
|
||||
for _, l := range lines {
|
||||
if strings.TrimSpace(l) != "" {
|
||||
kept = append(kept, l)
|
||||
}
|
||||
}
|
||||
return []chunk{{Heading: title, Text: strings.TrimSpace(strings.Join(kept, "\n"))}}
|
||||
}
|
||||
out := make([]chunk, 0, len(headers))
|
||||
for idx, h := range headers {
|
||||
end := len(lines)
|
||||
if idx+1 < len(headers) {
|
||||
end = headers[idx+1].start
|
||||
}
|
||||
var kept []string
|
||||
for _, l := range lines[h.start:end] {
|
||||
if strings.TrimSpace(l) != "" {
|
||||
kept = append(kept, l)
|
||||
}
|
||||
}
|
||||
text := strings.Join(kept, "\n")
|
||||
if idx == 0 && title != "" {
|
||||
text = title + "\n\n" + text
|
||||
}
|
||||
out = append(out, chunk{Heading: h.heading, Text: strings.TrimSpace(text)})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func ToAll(text, path, repo string) []Leaf {
|
||||
meta, body := ExtractFrontmatter(text)
|
||||
if meta["type"] == "" {
|
||||
meta["type"] = "reference"
|
||||
}
|
||||
if meta["status"] == "" {
|
||||
meta["status"] = "current"
|
||||
}
|
||||
chunks := SplitLeafs(meta, body)
|
||||
out := make([]Leaf, 0, len(chunks))
|
||||
for _, c := range chunks {
|
||||
out = append(out, Leaf{
|
||||
Source: path,
|
||||
Repo: repo,
|
||||
Heading: c.Heading,
|
||||
Text: c.Text,
|
||||
Type: meta["type"],
|
||||
Status: meta["status"],
|
||||
Related: meta["related"],
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func WalkMarkdown(root string) ([]string, error) {
|
||||
var out []string
|
||||
err := filepath.Walk(root, func(p string, info os.FileInfo, err error) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if info.IsDir() {
|
||||
return nil
|
||||
}
|
||||
ext := strings.ToLower(filepath.Ext(p))
|
||||
if ext == ".md" || ext == ".markdown" {
|
||||
out = append(out, p)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
return out, err
|
||||
}
|
||||
|
||||
func EncodeJSON(leafs []Leaf) (string, error) {
|
||||
raw, err := json.MarshalIndent(leafs, "", " ")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return string(raw) + "\n", nil
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
package mdleaves
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestSplitLeafsOnH2(t *testing.T) {
|
||||
body := "# Title\n\n## One\n\nalpha\n\n## Two\n\nbeta\n"
|
||||
leafs := SplitLeafs(map[string]string{}, body)
|
||||
if len(leafs) != 2 {
|
||||
t.Fatalf("n=%d", len(leafs))
|
||||
}
|
||||
if leafs[0].Heading != "One" || !strings.Contains(leafs[0].Text, "Title") {
|
||||
t.Fatalf("first=%+v", leafs[0])
|
||||
}
|
||||
if leafs[1].Heading != "Two" || strings.Contains(leafs[1].Text, "Title") {
|
||||
t.Fatalf("second=%+v", leafs[1])
|
||||
}
|
||||
}
|
||||
|
||||
func TestSplitLeafsNoH2IsWholeDoc(t *testing.T) {
|
||||
body := "# Title\n\njust a paragraph\n"
|
||||
leafs := SplitLeafs(nil, body)
|
||||
if len(leafs) != 1 {
|
||||
t.Fatalf("n=%d", len(leafs))
|
||||
}
|
||||
if leafs[0].Heading != "Title" {
|
||||
t.Fatalf("heading=%q", leafs[0].Heading)
|
||||
}
|
||||
if !strings.Contains(leafs[0].Text, "just a paragraph") {
|
||||
t.Fatalf("text=%q", leafs[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFrontmatterAndToAll(t *testing.T) {
|
||||
raw := "---\ntype: howto\nstatus: current\nrelated: docs/design.md\n---\n# Doc\n\n## Step\n\ndo it\n"
|
||||
got := ToAll(raw, "docs/x.md", "eSlider/2dph")
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("n=%d", len(got))
|
||||
}
|
||||
if got[0].Type != "howto" || got[0].Status != "current" {
|
||||
t.Fatalf("%+v", got[0])
|
||||
}
|
||||
if got[0].Source != "docs/x.md" || got[0].Repo != "eSlider/2dph" {
|
||||
t.Fatalf("%+v", got[0])
|
||||
}
|
||||
if got[0].Related != "docs/design.md" {
|
||||
t.Fatalf("related=%q", got[0].Related)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEncodeJSONAndYAML(t *testing.T) {
|
||||
leafs := ToAll("# Hi\n\nbody\n", "a.md", "")
|
||||
js, err := EncodeJSON(leafs)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(js, `"heading": "Hi"`) {
|
||||
t.Fatalf("json=%s", js)
|
||||
}
|
||||
y := EncodeYAML(leafs)
|
||||
if !strings.Contains(y, "heading: Hi") {
|
||||
t.Fatalf("yaml=%s", y)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefaultsTypeAndStatus(t *testing.T) {
|
||||
got := ToAll("# Hi\n\nbody\n", "a.md", "")
|
||||
if got[0].Type != "reference" || got[0].Status != "current" {
|
||||
t.Fatalf("%+v", got[0])
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
package mdleaves
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
func EncodeYAML(leafs []Leaf) string {
|
||||
if len(leafs) == 0 {
|
||||
return "[]\n"
|
||||
}
|
||||
var b strings.Builder
|
||||
for _, lf := range leafs {
|
||||
b.WriteString("-\n")
|
||||
writeKV(&b, "source", lf.Source)
|
||||
writeKV(&b, "repo", lf.Repo)
|
||||
writeKV(&b, "heading", lf.Heading)
|
||||
writeKV(&b, "text", lf.Text)
|
||||
writeKV(&b, "type", lf.Type)
|
||||
writeKV(&b, "status", lf.Status)
|
||||
writeKV(&b, "related", lf.Related)
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func writeKV(b *strings.Builder, k, v string) {
|
||||
fmt.Fprintf(b, " %s: %s\n", k, yamlScalar(v))
|
||||
}
|
||||
|
||||
func yamlScalar(s string) string {
|
||||
if strings.Contains(s, "\n") || s == "" || strings.ContainsAny(s, ":#'\"[]{}&*!|>%@`") || s != strings.TrimSpace(s) {
|
||||
return strconv.Quote(s)
|
||||
}
|
||||
return s
|
||||
}
|
||||
@@ -0,0 +1,162 @@
|
||||
// Package ocr runs Tesseract (eng+deu) on images and scanned PDFs.
|
||||
//
|
||||
// Default engine is the tesseract CLI, not gosseract CGO: Ladybug CGO stays
|
||||
// Zig-only (D21). Same engine, no gocv. OCR_ENGINE=paddle selects paddleocr
|
||||
// when that binary is on PATH (compose profile ocr-paddle).
|
||||
package ocr
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"image"
|
||||
"image/color"
|
||||
"image/png"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
const TessLang = "eng+deu"
|
||||
|
||||
func ImageFile(path string) (string, error) {
|
||||
engine := os.Getenv("OCR_ENGINE")
|
||||
if engine == "paddle" {
|
||||
return runPaddle(path)
|
||||
}
|
||||
return runTesseract(path)
|
||||
}
|
||||
|
||||
func PDFFile(path string) (string, error) {
|
||||
text, err := pdfToText(path)
|
||||
if err == nil && strings.TrimSpace(text) != "" {
|
||||
return strings.TrimSpace(text), nil
|
||||
}
|
||||
ocr, oerr := pdfPages(path)
|
||||
if oerr != nil {
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return "", oerr
|
||||
}
|
||||
if strings.TrimSpace(ocr) != "" {
|
||||
return strings.TrimSpace(ocr), nil
|
||||
}
|
||||
if text != "" {
|
||||
return strings.TrimSpace(text), nil
|
||||
}
|
||||
return "", fmt.Errorf("pdf has no text layer (ocr unavailable)")
|
||||
}
|
||||
|
||||
func pdfToText(path string) (string, error) {
|
||||
cmd := exec.Command("pdftotext", "-layout", path, "-")
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return string(out), nil
|
||||
}
|
||||
|
||||
func pdfPages(path string) (string, error) {
|
||||
dir, err := os.MkdirTemp("", "2dph-ocr-")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
prefix := filepath.Join(dir, "page")
|
||||
cmd := exec.Command("pdftoppm", "-png", "-r", "200", path, prefix)
|
||||
if err := cmd.Run(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
matches, err := filepath.Glob(prefix + "*.png")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
var parts []string
|
||||
for _, img := range matches {
|
||||
t, err := ImageFile(img)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if s := strings.TrimSpace(t); s != "" {
|
||||
parts = append(parts, s)
|
||||
}
|
||||
}
|
||||
return strings.Join(parts, "\n\n"), nil
|
||||
}
|
||||
|
||||
func runTesseract(path string) (string, error) {
|
||||
pre, err := preprocessFile(path)
|
||||
if err != nil {
|
||||
pre = path
|
||||
} else {
|
||||
defer os.Remove(pre)
|
||||
}
|
||||
cmd := exec.Command("tesseract", pre, "stdout", "-l", TessLang, "--psm", "6")
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return strings.TrimSpace(string(out)), nil
|
||||
}
|
||||
|
||||
func runPaddle(path string) (string, error) {
|
||||
cmd := exec.Command("paddleocr", "ocr", "-i", path)
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return strings.TrimSpace(string(out)), nil
|
||||
}
|
||||
|
||||
func preprocessFile(path string) (string, error) {
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer f.Close()
|
||||
img, err := png.Decode(f)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
out := filepath.Join(os.TempDir(), filepath.Base(path)+".gray.png")
|
||||
w, err := os.Create(out)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer w.Close()
|
||||
if err := png.Encode(w, GrayContrast(img)); err != nil {
|
||||
os.Remove(out)
|
||||
return "", err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// GrayContrast is a stdlib preprocess (no gocv): grayscale + stretch.
|
||||
func GrayContrast(src image.Image) image.Image {
|
||||
b := src.Bounds()
|
||||
dst := image.NewGray(b)
|
||||
var minL, maxL uint8 = 255, 0
|
||||
for y := b.Min.Y; y < b.Max.Y; y++ {
|
||||
for x := b.Min.X; x < b.Max.X; x++ {
|
||||
g := color.GrayModel.Convert(src.At(x, y)).(color.Gray)
|
||||
if g.Y < minL {
|
||||
minL = g.Y
|
||||
}
|
||||
if g.Y > maxL {
|
||||
maxL = g.Y
|
||||
}
|
||||
}
|
||||
}
|
||||
span := int(maxL) - int(minL)
|
||||
if span < 1 {
|
||||
span = 1
|
||||
}
|
||||
for y := b.Min.Y; y < b.Max.Y; y++ {
|
||||
for x := b.Min.X; x < b.Max.X; x++ {
|
||||
g := color.GrayModel.Convert(src.At(x, y)).(color.Gray)
|
||||
v := uint8((int(g.Y) - int(minL)) * 255 / span)
|
||||
dst.SetGray(x, y, color.Gray{Y: v})
|
||||
}
|
||||
}
|
||||
return dst
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
package ocr
|
||||
|
||||
import (
|
||||
"image"
|
||||
"image/color"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestGrayContrastStretches(t *testing.T) {
|
||||
img := image.NewGray(image.Rect(0, 0, 2, 2))
|
||||
img.SetGray(0, 0, color.Gray{Y: 64})
|
||||
img.SetGray(0, 1, color.Gray{Y: 64})
|
||||
img.SetGray(1, 0, color.Gray{Y: 64})
|
||||
img.SetGray(1, 1, color.Gray{Y: 192})
|
||||
out := GrayContrast(img).(*image.Gray)
|
||||
if out.GrayAt(0, 0).Y != 0 {
|
||||
t.Fatalf("min should map to 0, got %d", out.GrayAt(0, 0).Y)
|
||||
}
|
||||
if out.GrayAt(1, 1).Y != 255 {
|
||||
t.Fatalf("max should map to 255, got %d", out.GrayAt(1, 1).Y)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHelloPNGFixtureOCR(t *testing.T) {
|
||||
if _, err := exec.LookPath("tesseract"); err != nil {
|
||||
t.Skip("tesseract not installed")
|
||||
}
|
||||
path := filepath.Join("testdata", "hello.png")
|
||||
got, err := ImageFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
up := strings.ToUpper(got)
|
||||
if !strings.Contains(up, "HELLO") {
|
||||
t.Fatalf("ocr %q missing HELLO", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPaddleEngineUsesPaddleocrBinary(t *testing.T) {
|
||||
t.Setenv("OCR_ENGINE", "paddle")
|
||||
_, err := ImageFile(filepath.Join("testdata", "hello.png"))
|
||||
if _, look := exec.LookPath("paddleocr"); look != nil {
|
||||
if err == nil {
|
||||
t.Fatal("expected error when paddleocr is missing")
|
||||
}
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
Vendored
BIN
Binary file not shown.
|
After Width: | Height: | Size: 1.7 KiB |
@@ -0,0 +1,327 @@
|
||||
package reasoner
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// HF IDs named in docs. No Qwen3.6-9B exists.
|
||||
const (
|
||||
HFQwen35_9B = "Qwen/Qwen3.5-9B"
|
||||
HFQwen36_27B = "Qwen/Qwen3.6-27B"
|
||||
HFBonsai27B = "prism-ml/Bonsai-27B-gguf"
|
||||
OllamaRAM = "qwen3.5:9b"
|
||||
OllamaQuality = "MichelRosselli/bonsai-27b:Q1_0"
|
||||
)
|
||||
|
||||
type ToolCall struct {
|
||||
Name string
|
||||
Arguments string
|
||||
}
|
||||
|
||||
type Result struct {
|
||||
Model string `json:"model"`
|
||||
OK bool `json:"ok"`
|
||||
ToolName string `json:"tool_name,omitempty"`
|
||||
XMLLeak bool `json:"xml_leak"`
|
||||
Err string `json:"error,omitempty"`
|
||||
LatencyMS int64 `json:"latency_ms"`
|
||||
RSSMB int `json:"rss_mb,omitempty"`
|
||||
Device string `json:"device"`
|
||||
WantedTool string `json:"wanted_tool"`
|
||||
}
|
||||
|
||||
type Prompt struct {
|
||||
Name string
|
||||
Want string
|
||||
User string
|
||||
}
|
||||
|
||||
var BakePrompts = []Prompt{
|
||||
{
|
||||
Name: "search-before-claim",
|
||||
Want: "search",
|
||||
User: "Use tools. Search the 2dph brain for LadybugDB before you answer. Call search.",
|
||||
},
|
||||
{
|
||||
Name: "get-leaf",
|
||||
Want: "get",
|
||||
User: "Use tools. Fetch leaf id leaf-demo with get. Do not invent the body.",
|
||||
},
|
||||
{
|
||||
Name: "audit-index",
|
||||
Want: "audit",
|
||||
User: "Use tools. Call audit on the brain index health.",
|
||||
},
|
||||
}
|
||||
|
||||
func MCPTools() []map[string]any {
|
||||
return []map[string]any{
|
||||
openaiTool("search", "deduction search (facts → info → web)", map[string]any{
|
||||
"type": "object",
|
||||
"properties": map[string]any{
|
||||
"q": map[string]any{"type": "string", "description": "search query"},
|
||||
"n": map[string]any{"type": "integer"},
|
||||
},
|
||||
"required": []string{"q"},
|
||||
}),
|
||||
openaiTool("get", "read one leaf by id", map[string]any{
|
||||
"type": "object",
|
||||
"properties": map[string]any{
|
||||
"id": map[string]any{"type": "string"},
|
||||
"body": map[string]any{"type": "boolean"},
|
||||
},
|
||||
"required": []string{"id"},
|
||||
}),
|
||||
openaiTool("audit", "facts confidence histogram", map[string]any{
|
||||
"type": "object",
|
||||
"properties": map[string]any{},
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
func openaiTool(name, desc string, schema map[string]any) map[string]any {
|
||||
return map[string]any{
|
||||
"type": "function",
|
||||
"function": map[string]any{
|
||||
"name": name,
|
||||
"description": desc,
|
||||
"parameters": schema,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
type Client struct {
|
||||
BaseURL string
|
||||
Model string
|
||||
HTTP *http.Client
|
||||
Device string
|
||||
ToolChoice string
|
||||
}
|
||||
|
||||
type Report struct {
|
||||
Model string `json:"model"`
|
||||
HF string `json:"hf_id"`
|
||||
Device string `json:"device"`
|
||||
ToolCallOK int `json:"tool_call_ok"`
|
||||
ToolCallN int `json:"tool_call_n"`
|
||||
XMLLeak int `json:"xml_leak"`
|
||||
RSSMB int `json:"rss_mb"`
|
||||
VRAMMB int `json:"vram_mb"`
|
||||
LatencyP50MS float64 `json:"latency_p50_ms,omitempty"`
|
||||
LatencyP95MS float64 `json:"latency_p95_ms,omitempty"`
|
||||
Prompts []Result `json:"prompts"`
|
||||
}
|
||||
|
||||
func HFFor(model string) string {
|
||||
switch model {
|
||||
case OllamaRAM:
|
||||
return HFQwen35_9B
|
||||
case OllamaQuality:
|
||||
return HFBonsai27B
|
||||
default:
|
||||
if strings.Contains(model, "qwen3.6") || strings.Contains(model, "Qwen3.6") {
|
||||
return HFQwen36_27B
|
||||
}
|
||||
return ""
|
||||
}
|
||||
}
|
||||
|
||||
func (c *Client) httpc() *http.Client {
|
||||
if c.HTTP == nil {
|
||||
c.HTTP = &http.Client{Timeout: 10 * time.Minute}
|
||||
}
|
||||
return c.HTTP
|
||||
}
|
||||
|
||||
func Origin(base string) string {
|
||||
s := strings.TrimRight(base, "/")
|
||||
return strings.TrimSuffix(s, "/v1")
|
||||
}
|
||||
|
||||
func (c Client) ChatTools(user string) (ToolCall, string, error) {
|
||||
choice := c.ToolChoice
|
||||
if choice == "" {
|
||||
choice = "required"
|
||||
}
|
||||
body, _ := json.Marshal(map[string]any{
|
||||
"model": c.Model,
|
||||
"messages": []map[string]string{
|
||||
{"role": "system", "content": "You are PicoClaw talking to 2dph MCP. Always call a tool before a factual claim. search then get then audit."},
|
||||
{"role": "user", "content": user},
|
||||
},
|
||||
"tools": MCPTools(),
|
||||
"tool_choice": choice,
|
||||
})
|
||||
base := strings.TrimRight(c.BaseURL, "/")
|
||||
if !strings.HasSuffix(base, "/v1") {
|
||||
base += "/v1"
|
||||
}
|
||||
url := base + "/chat/completions"
|
||||
req, err := http.NewRequest(http.MethodPost, url, bytes.NewReader(body))
|
||||
if err != nil {
|
||||
return ToolCall{}, "", err
|
||||
}
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
res, err := c.httpc().Do(req)
|
||||
if err != nil {
|
||||
return ToolCall{}, "", err
|
||||
}
|
||||
defer res.Body.Close()
|
||||
raw, _ := io.ReadAll(res.Body)
|
||||
if res.StatusCode >= 300 {
|
||||
return ToolCall{}, string(raw), fmt.Errorf("http %d", res.StatusCode)
|
||||
}
|
||||
return ParseToolResponse(raw)
|
||||
}
|
||||
|
||||
func ParseToolResponse(raw []byte) (ToolCall, string, error) {
|
||||
var wrap struct {
|
||||
Choices []struct {
|
||||
Message struct {
|
||||
Content string `json:"content"`
|
||||
ToolCalls []struct {
|
||||
Function struct {
|
||||
Name string `json:"name"`
|
||||
Arguments json.RawMessage `json:"arguments"`
|
||||
} `json:"function"`
|
||||
} `json:"tool_calls"`
|
||||
} `json:"message"`
|
||||
} `json:"choices"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &wrap); err != nil {
|
||||
return ToolCall{}, "", err
|
||||
}
|
||||
content := ""
|
||||
if len(wrap.Choices) > 0 {
|
||||
content = wrap.Choices[0].Message.Content
|
||||
if n := len(wrap.Choices[0].Message.ToolCalls); n > 0 {
|
||||
fn := wrap.Choices[0].Message.ToolCalls[0].Function
|
||||
return ToolCall{Name: fn.Name, Arguments: rawArgs(fn.Arguments)}, content, nil
|
||||
}
|
||||
}
|
||||
return ToolCall{}, content, fmt.Errorf("no tool_calls")
|
||||
}
|
||||
|
||||
func rawArgs(raw json.RawMessage) string {
|
||||
if len(raw) == 0 {
|
||||
return ""
|
||||
}
|
||||
var s string
|
||||
if err := json.Unmarshal(raw, &s); err == nil {
|
||||
return s
|
||||
}
|
||||
return string(raw)
|
||||
}
|
||||
|
||||
func XMLLeak(content string) bool {
|
||||
s := strings.ToLower(content)
|
||||
return strings.Contains(s, "<tool_call>") ||
|
||||
strings.Contains(s, "<function=") ||
|
||||
strings.Contains(s, "<parameter")
|
||||
}
|
||||
|
||||
func RunPrompt(c Client, p Prompt) Result {
|
||||
start := time.Now()
|
||||
tc, content, err := c.ChatTools(p.User)
|
||||
out := Result{
|
||||
Model: c.Model,
|
||||
WantedTool: p.Want,
|
||||
LatencyMS: time.Since(start).Milliseconds(),
|
||||
Device: c.Device,
|
||||
XMLLeak: XMLLeak(content),
|
||||
}
|
||||
if err != nil {
|
||||
out.Err = err.Error()
|
||||
if content != "" && out.XMLLeak {
|
||||
out.Err = "xml tool call instead of openai tool_calls"
|
||||
}
|
||||
return out
|
||||
}
|
||||
out.ToolName = tc.Name
|
||||
out.OK = tc.Name == p.Want
|
||||
if !out.OK {
|
||||
out.Err = "wanted " + p.Want + " got " + tc.Name
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
type ProcMem struct {
|
||||
Name string
|
||||
SizeMB int
|
||||
VRAMMB int
|
||||
}
|
||||
|
||||
func ParsePS(raw []byte) []ProcMem {
|
||||
var wrap struct {
|
||||
Models []struct {
|
||||
Name string `json:"name"`
|
||||
Size int64 `json:"size"`
|
||||
SizeVRAM int64 `json:"size_vram"`
|
||||
} `json:"models"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &wrap); err != nil {
|
||||
return nil
|
||||
}
|
||||
out := make([]ProcMem, 0, len(wrap.Models))
|
||||
for _, m := range wrap.Models {
|
||||
out = append(out, ProcMem{
|
||||
Name: m.Name,
|
||||
SizeMB: int(m.Size / (1024 * 1024)),
|
||||
VRAMMB: int(m.SizeVRAM / (1024 * 1024)),
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func (c Client) FetchPS() []ProcMem {
|
||||
url := Origin(c.BaseURL) + "/api/ps"
|
||||
res, err := c.httpc().Get(url)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
defer res.Body.Close()
|
||||
raw, _ := io.ReadAll(res.Body)
|
||||
if res.StatusCode >= 300 {
|
||||
return nil
|
||||
}
|
||||
return ParsePS(raw)
|
||||
}
|
||||
|
||||
func Run(c Client) Report {
|
||||
if c.Device == "" {
|
||||
c.Device = "cpu"
|
||||
}
|
||||
rep := Report{
|
||||
Model: c.Model,
|
||||
HF: HFFor(c.Model),
|
||||
Device: c.Device,
|
||||
}
|
||||
for _, p := range BakePrompts {
|
||||
r := RunPrompt(c, p)
|
||||
rep.Prompts = append(rep.Prompts, r)
|
||||
rep.ToolCallN++
|
||||
if r.OK {
|
||||
rep.ToolCallOK++
|
||||
}
|
||||
if r.XMLLeak {
|
||||
rep.XMLLeak++
|
||||
}
|
||||
}
|
||||
if mems := c.FetchPS(); len(mems) > 0 {
|
||||
rep.RSSMB = mems[0].SizeMB
|
||||
rep.VRAMMB = mems[0].VRAMMB
|
||||
if c.Device == "cpu" && mems[0].VRAMMB > 0 {
|
||||
rep.Device = "gpu"
|
||||
}
|
||||
for i := range rep.Prompts {
|
||||
rep.Prompts[i].RSSMB = rep.RSSMB
|
||||
}
|
||||
}
|
||||
return rep
|
||||
}
|
||||
@@ -0,0 +1,166 @@
|
||||
package reasoner
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/eSlider/2dph/internal/httpapi"
|
||||
)
|
||||
|
||||
func TestHFIdsAreRealAndNoQwen36Nine(t *testing.T) {
|
||||
if HFQwen35_9B != "Qwen/Qwen3.5-9B" {
|
||||
t.Fatalf("9B id = %s", HFQwen35_9B)
|
||||
}
|
||||
if HFQwen36_27B != "Qwen/Qwen3.6-27B" {
|
||||
t.Fatalf("27B id = %s", HFQwen36_27B)
|
||||
}
|
||||
if HFBonsai27B != "prism-ml/Bonsai-27B-gguf" {
|
||||
t.Fatalf("bonsai id = %s", HFBonsai27B)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseToolResponseOpenAI(t *testing.T) {
|
||||
raw := []byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"search","arguments":"{\"q\":\"LadybugDB\"}"}}]}}]}`)
|
||||
tc, _, err := ParseToolResponse(raw)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if tc.Name != "search" {
|
||||
t.Fatalf("name=%s", tc.Name)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseToolResponseXMLIsFailure(t *testing.T) {
|
||||
raw := []byte(`{"choices":[{"message":{"content":"<tool_call>search</tool_call>"}}]}`)
|
||||
_, content, err := ParseToolResponse(raw)
|
||||
if err == nil {
|
||||
t.Fatal("expected no tool_calls")
|
||||
}
|
||||
if !XMLLeak(content) {
|
||||
t.Fatal("xml leak not detected")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBakePromptsWantMCPTools(t *testing.T) {
|
||||
if len(BakePrompts) != 3 {
|
||||
t.Fatalf("prompts=%d", len(BakePrompts))
|
||||
}
|
||||
names := map[string]bool{}
|
||||
for _, p := range BakePrompts {
|
||||
names[p.Want] = true
|
||||
}
|
||||
for _, n := range []string{"search", "get", "audit"} {
|
||||
if !names[n] {
|
||||
t.Fatalf("missing want %s", n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseToolResponseObjectArgs(t *testing.T) {
|
||||
raw := []byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"get","arguments":{"id":"leaf-demo"}}}]}}]}`)
|
||||
tc, _, err := ParseToolResponse(raw)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if tc.Name != "get" {
|
||||
t.Fatalf("name=%s", tc.Name)
|
||||
}
|
||||
if !strings.Contains(tc.Arguments, "leaf-demo") {
|
||||
t.Fatalf("args=%s", tc.Arguments)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParsePSCPUNotVRAM(t *testing.T) {
|
||||
raw := []byte(`{"models":[{"name":"qwen3.5:9b","size":6900000000,"size_vram":0}]}`)
|
||||
got := ParsePS(raw)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("n=%d", len(got))
|
||||
}
|
||||
if got[0].VRAMMB != 0 {
|
||||
t.Fatalf("vram=%d", got[0].VRAMMB)
|
||||
}
|
||||
if got[0].SizeMB < 6000 {
|
||||
t.Fatalf("rss=%d", got[0].SizeMB)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMCPToolsArePicoClawSubset(t *testing.T) {
|
||||
mcp := map[string]bool{}
|
||||
for _, op := range httpapi.Ops {
|
||||
if op.MCP {
|
||||
mcp[op.ID] = true
|
||||
}
|
||||
}
|
||||
seen := map[string]bool{}
|
||||
for _, tool := range MCPTools() {
|
||||
fn, _ := tool["function"].(map[string]any)
|
||||
name, _ := fn["name"].(string)
|
||||
if !mcp[name] {
|
||||
t.Fatalf("%s is not an MCP op", name)
|
||||
}
|
||||
seen[name] = true
|
||||
}
|
||||
for _, n := range []string{"search", "get", "audit"} {
|
||||
if !seen[n] {
|
||||
t.Fatalf("bake-off missing %s", n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestChatToolsHitsOpenAIPath(t *testing.T) {
|
||||
var gotTools bool
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.URL.Path != "/v1/chat/completions" {
|
||||
t.Errorf("path=%s", r.URL.Path)
|
||||
}
|
||||
body, _ := io.ReadAll(r.Body)
|
||||
var payload map[string]any
|
||||
_ = json.Unmarshal(body, &payload)
|
||||
if _, ok := payload["tools"]; ok {
|
||||
gotTools = true
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"search","arguments":"{\"q\":\"LadybugDB\"}"}}]}}]}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
c := Client{BaseURL: srv.URL + "/v1", Model: "stub", HTTP: srv.Client(), Device: "cpu"}
|
||||
r := RunPrompt(c, BakePrompts[0])
|
||||
if !gotTools {
|
||||
t.Fatal("tools not sent")
|
||||
}
|
||||
if !r.OK {
|
||||
t.Fatalf("result=%+v", r)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunSamplesPSAfterPrompts(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
if r.URL.Path == "/api/ps" {
|
||||
_, _ = w.Write([]byte(`{"models":[{"name":"stub","size":6500000000,"size_vram":0}]}`))
|
||||
return
|
||||
}
|
||||
_, _ = w.Write([]byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"search","arguments":"{}"}}]}}]}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
c := Client{BaseURL: srv.URL + "/v1", Model: "stub", HTTP: srv.Client(), Device: "cpu"}
|
||||
if mems := c.FetchPS(); len(mems) != 1 || mems[0].VRAMMB != 0 {
|
||||
t.Fatalf("%+v", mems)
|
||||
}
|
||||
if mems := c.FetchPS(); mems[0].SizeMB < 6000 {
|
||||
t.Fatalf("rss=%d", mems[0].SizeMB)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHFForKnownOllamaTags(t *testing.T) {
|
||||
if HFFor(OllamaRAM) != HFQwen35_9B {
|
||||
t.Fatal(HFFor(OllamaRAM))
|
||||
}
|
||||
if HFFor(OllamaQuality) != HFBonsai27B {
|
||||
t.Fatal(HFFor(OllamaQuality))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
package websearch
|
||||
|
||||
import (
|
||||
"database/sql"
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
const cacheSchema = `
|
||||
CREATE TABLE IF NOT EXISTS responses (
|
||||
key TEXT PRIMARY KEY,
|
||||
fetched REAL NOT NULL,
|
||||
payload TEXT NOT NULL
|
||||
);
|
||||
CREATE TABLE IF NOT EXISTS meta (
|
||||
key TEXT PRIMARY KEY,
|
||||
value REAL NOT NULL
|
||||
);
|
||||
`
|
||||
|
||||
type Cache struct {
|
||||
db *sql.DB
|
||||
}
|
||||
|
||||
func OpenCache(path string) (*Cache, error) {
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o700); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
db, err := sql.Open("sqlite", path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if _, err := db.Exec(cacheSchema); err != nil {
|
||||
db.Close()
|
||||
return nil, err
|
||||
}
|
||||
return &Cache{db: db}, nil
|
||||
}
|
||||
|
||||
func (c *Cache) Close() error {
|
||||
if c == nil || c.db == nil {
|
||||
return nil
|
||||
}
|
||||
return c.db.Close()
|
||||
}
|
||||
|
||||
func (c *Cache) Get(key string, ttl, now float64) (*Payload, error) {
|
||||
var fetched float64
|
||||
var raw string
|
||||
err := c.db.QueryRow("SELECT fetched, payload FROM responses WHERE key = ?", key).Scan(&fetched, &raw)
|
||||
if err == sql.ErrNoRows {
|
||||
return nil, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if now-fetched > ttl {
|
||||
return nil, nil
|
||||
}
|
||||
var p Payload
|
||||
if err := json.Unmarshal([]byte(raw), &p); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &p, nil
|
||||
}
|
||||
|
||||
func (c *Cache) Put(key string, p Payload, now float64) error {
|
||||
raw, err := json.Marshal(p)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
_, err = c.db.Exec(
|
||||
"INSERT OR REPLACE INTO responses (key, fetched, payload) VALUES (?, ?, ?)",
|
||||
key, now, string(raw),
|
||||
)
|
||||
return err
|
||||
}
|
||||
|
||||
func (c *Cache) LastCall() (*float64, error) {
|
||||
var v float64
|
||||
err := c.db.QueryRow("SELECT value FROM meta WHERE key = 'last_call'").Scan(&v)
|
||||
if err == sql.ErrNoRows {
|
||||
return nil, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &v, nil
|
||||
}
|
||||
|
||||
func (c *Cache) MarkCall(now float64) error {
|
||||
_, err := c.db.Exec("INSERT OR REPLACE INTO meta (key, value) VALUES ('last_call', ?)", now)
|
||||
return err
|
||||
}
|
||||
@@ -0,0 +1,90 @@
|
||||
package websearch
|
||||
|
||||
import (
|
||||
"encoding/base64"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
type Config struct {
|
||||
URL string
|
||||
User string
|
||||
Pass string
|
||||
}
|
||||
|
||||
func LoadConfig(path string) (Config, error) {
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return Config{}, fmt.Errorf("no credentials at %s (mode 600, BRAIN_SEARCH_URL)", path)
|
||||
}
|
||||
conf := map[string]string{}
|
||||
for _, line := range strings.Split(string(raw), "\n") {
|
||||
line = strings.TrimSpace(line)
|
||||
if line == "" || strings.HasPrefix(line, "#") || !strings.Contains(line, "=") {
|
||||
continue
|
||||
}
|
||||
k, v, _ := strings.Cut(line, "=")
|
||||
v = strings.TrimSpace(v)
|
||||
v = strings.Trim(v, `"'`)
|
||||
conf[strings.TrimSpace(k)] = v
|
||||
}
|
||||
out := Config{
|
||||
URL: conf["BRAIN_SEARCH_URL"],
|
||||
User: conf["BRAIN_SEARCH_USER"],
|
||||
Pass: conf["BRAIN_SEARCH_PASS"],
|
||||
}
|
||||
if out.URL == "" {
|
||||
return Config{}, fmt.Errorf("%s is missing BRAIN_SEARCH_URL", path)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func Fetch(client *http.Client, conf Config, query string, params map[string]string, timeout time.Duration) (Payload, error) {
|
||||
if client == nil {
|
||||
client = &http.Client{Timeout: timeout}
|
||||
} else if timeout > 0 {
|
||||
c := *client
|
||||
c.Timeout = timeout
|
||||
client = &c
|
||||
}
|
||||
q := url.Values{}
|
||||
q.Set("q", query)
|
||||
q.Set("format", "json")
|
||||
for k, v := range params {
|
||||
if v != "" {
|
||||
q.Set(k, v)
|
||||
}
|
||||
}
|
||||
u := strings.TrimRight(conf.URL, "/") + "/search?" + q.Encode()
|
||||
req, err := http.NewRequest(http.MethodGet, u, nil)
|
||||
if err != nil {
|
||||
return Payload{}, err
|
||||
}
|
||||
if conf.User != "" || conf.Pass != "" {
|
||||
token := base64.StdEncoding.EncodeToString([]byte(conf.User + ":" + conf.Pass))
|
||||
req.Header.Set("Authorization", "Basic "+token)
|
||||
}
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return Payload{}, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
body, err := io.ReadAll(io.LimitReader(resp.Body, 8<<20))
|
||||
if err != nil {
|
||||
return Payload{}, err
|
||||
}
|
||||
if resp.StatusCode >= 400 {
|
||||
return Payload{}, fmt.Errorf("HTTP %d", resp.StatusCode)
|
||||
}
|
||||
var p Payload
|
||||
if err := json.Unmarshal(body, &p); err != nil {
|
||||
return Payload{}, err
|
||||
}
|
||||
return p, nil
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
package websearch
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestLoadConfigRequiresURL(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
p := filepath.Join(dir, "search.env")
|
||||
if err := os.WriteFile(p, []byte("BRAIN_SEARCH_USER=x\n"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := LoadConfig(p); err == nil {
|
||||
t.Fatal("expected missing URL error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLoadConfigOptionalAuth(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
p := filepath.Join(dir, "search.env")
|
||||
if err := os.WriteFile(p, []byte("BRAIN_SEARCH_URL=http://127.0.0.1:8080\n"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
c, err := LoadConfig(p)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if c.URL != "http://127.0.0.1:8080" || c.User != "" || c.Pass != "" {
|
||||
t.Fatalf("%+v", c)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFetchJSONNoBasicAuth(t *testing.T) {
|
||||
payload := Payload{Query: "x", Results: []RawHit{{Title: "t", URL: "http://example.com", Content: "c", Engine: "bing"}}}
|
||||
var sawAuth string
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
sawAuth = r.Header.Get("Authorization")
|
||||
if r.URL.Query().Get("format") != "json" || r.URL.Query().Get("q") != "x" {
|
||||
t.Errorf("query = %s", r.URL.RawQuery)
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
json.NewEncoder(w).Encode(payload)
|
||||
}))
|
||||
defer srv.Close()
|
||||
got, err := Fetch(srv.Client(), Config{URL: srv.URL}, "x", nil, 2*time.Second)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if sawAuth != "" {
|
||||
t.Fatalf("Authorization = %q, want empty for local instance", sawAuth)
|
||||
}
|
||||
if Classify(got) != StatusOK {
|
||||
t.Fatalf("classify = %s", Classify(got))
|
||||
}
|
||||
}
|
||||
|
||||
func TestFetchSendsBasicAuthWhenConfigured(t *testing.T) {
|
||||
var sawAuth string
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
sawAuth = r.Header.Get("Authorization")
|
||||
w.Write([]byte(`{"query":"x","results":[]}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
_, err := Fetch(srv.Client(), Config{URL: srv.URL, User: "u", Pass: "p"}, "x", nil, 2*time.Second)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if sawAuth == "" {
|
||||
t.Fatal("expected Basic auth")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
package websearch
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
const (
|
||||
StatusSkipped = "skipped"
|
||||
StatusRefused = "refused"
|
||||
)
|
||||
|
||||
type LookupOpt struct {
|
||||
Limit int
|
||||
Timeout time.Duration
|
||||
EnvPath string
|
||||
CachePath string
|
||||
Client *http.Client
|
||||
Now func() float64
|
||||
Sleep func(context.Context, time.Duration) error
|
||||
}
|
||||
|
||||
func Lookup(ctx context.Context, query string, opt LookupOpt) Output {
|
||||
if ctx == nil {
|
||||
ctx = context.Background()
|
||||
}
|
||||
if opt.Limit <= 0 {
|
||||
opt.Limit = DefaultLimit
|
||||
}
|
||||
if opt.Timeout <= 0 {
|
||||
opt.Timeout = 25 * time.Second
|
||||
}
|
||||
nowFn := opt.Now
|
||||
if nowFn == nil {
|
||||
nowFn = func() float64 { return float64(time.Now().Unix()) }
|
||||
}
|
||||
sleepFn := opt.Sleep
|
||||
if sleepFn == nil {
|
||||
sleepFn = func(ctx context.Context, d time.Duration) error {
|
||||
t := time.NewTimer(d)
|
||||
defer t.Stop()
|
||||
select {
|
||||
case <-t.C:
|
||||
return nil
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if reason := PHIReason(query); reason != "" {
|
||||
return Output{Query: query, Status: StatusRefused, Note: reason}
|
||||
}
|
||||
|
||||
cachePath := opt.CachePath
|
||||
if cachePath == "" {
|
||||
cachePath = os.Getenv("BRAIN_SEARCH_CACHE")
|
||||
}
|
||||
if cachePath == "" {
|
||||
cachePath = os.Getenv("HOME") + "/.cache/brain/web-search.sqlite"
|
||||
}
|
||||
cache, err := OpenCache(cachePath)
|
||||
if err != nil {
|
||||
return Output{Query: query, Status: StatusSkipped, Note: "cache: " + err.Error()}
|
||||
}
|
||||
defer cache.Close()
|
||||
|
||||
key := CacheKey(query, nil)
|
||||
now := nowFn()
|
||||
if cached, err := cache.Get(key, CacheTTL, now); err == nil && cached != nil {
|
||||
out := Project(*cached, opt.Limit, DefaultSnippetChars)
|
||||
out.Cached = true
|
||||
return out
|
||||
}
|
||||
|
||||
envPath := opt.EnvPath
|
||||
if envPath == "" {
|
||||
envPath = os.Getenv("BRAIN_SEARCH_ENV")
|
||||
}
|
||||
if envPath == "" {
|
||||
envPath = os.Getenv("HOME") + "/.config/brain/search.env"
|
||||
}
|
||||
conf, err := LoadConfig(envPath)
|
||||
if err != nil {
|
||||
return Output{Query: query, Status: StatusSkipped, Note: "no BRAIN_SEARCH_URL; second source not consulted"}
|
||||
}
|
||||
|
||||
lock, err := os.OpenFile(cachePath+".lock", os.O_CREATE|os.O_RDWR, 0o600)
|
||||
if err != nil {
|
||||
return Output{Query: query, Status: StatusSkipped, Note: "lock: " + err.Error()}
|
||||
}
|
||||
defer lock.Close()
|
||||
if err := unix.Flock(int(lock.Fd()), unix.LOCK_EX); err != nil {
|
||||
return Output{Query: query, Status: StatusSkipped, Note: "lock: " + err.Error()}
|
||||
}
|
||||
defer unix.Flock(int(lock.Fd()), unix.LOCK_UN)
|
||||
|
||||
last, err := cache.LastCall()
|
||||
if err != nil {
|
||||
return Output{Query: query, Status: StatusSkipped, Note: "cache: " + err.Error()}
|
||||
}
|
||||
if delay := WaitFor(last, nowFn(), MinInterval); delay > 0 {
|
||||
if err := sleepFn(ctx, time.Duration(delay*float64(time.Second))); err != nil {
|
||||
return Output{Query: query, Status: StatusSkipped, Note: "cancelled"}
|
||||
}
|
||||
}
|
||||
_ = cache.MarkCall(nowFn())
|
||||
|
||||
payload, err := Fetch(opt.Client, conf, query, nil, opt.Timeout)
|
||||
if err != nil {
|
||||
return Output{Query: query, Status: StatusThrottled, Note: fmt.Sprintf("request failed: %v", err)}
|
||||
}
|
||||
if Classify(payload) == StatusOK {
|
||||
_ = cache.Put(key, payload, nowFn())
|
||||
}
|
||||
return Project(payload, opt.Limit, DefaultSnippetChars)
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
package websearch
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestLookupRefusesPIIWithoutFetch(t *testing.T) {
|
||||
hits := 0
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) {
|
||||
hits++
|
||||
}))
|
||||
defer srv.Close()
|
||||
out := Lookup(context.Background(), "Personalnummer 12", LookupOpt{
|
||||
EnvPath: writeEnv(t, srv.URL),
|
||||
CachePath: filepath.Join(t.TempDir(), "c.sqlite"),
|
||||
Client: srv.Client(),
|
||||
Sleep: func(context.Context, time.Duration) error { return nil },
|
||||
})
|
||||
if out.Status != StatusRefused {
|
||||
t.Fatalf("status = %s", out.Status)
|
||||
}
|
||||
if hits != 0 {
|
||||
t.Fatal("PII query left the host")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLookupSkipsWhenNoConfig(t *testing.T) {
|
||||
out := Lookup(context.Background(), "LadybugDB", LookupOpt{
|
||||
EnvPath: filepath.Join(t.TempDir(), "missing.env"),
|
||||
CachePath: filepath.Join(t.TempDir(), "c.sqlite"),
|
||||
Sleep: func(context.Context, time.Duration) error { return nil },
|
||||
})
|
||||
if out.Status != StatusSkipped {
|
||||
t.Fatalf("status = %s", out.Status)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLookupFetchesOnceAndCaches(t *testing.T) {
|
||||
hits := 0
|
||||
payload := Payload{Query: "x", Results: []RawHit{{Title: "t", URL: "http://example.com", Content: "c", Engine: "bing"}}}
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
hits++
|
||||
json.NewEncoder(w).Encode(payload)
|
||||
}))
|
||||
defer srv.Close()
|
||||
opt := LookupOpt{
|
||||
EnvPath: writeEnv(t, srv.URL),
|
||||
CachePath: filepath.Join(t.TempDir(), "c.sqlite"),
|
||||
Client: srv.Client(),
|
||||
Now: func() float64 { return 1_000 },
|
||||
Sleep: func(context.Context, time.Duration) error { return nil },
|
||||
}
|
||||
a := Lookup(context.Background(), "LadybugDB", opt)
|
||||
b := Lookup(context.Background(), "LadybugDB", opt)
|
||||
if a.Status != StatusOK || b.Status != StatusOK {
|
||||
t.Fatalf("a=%s b=%s", a.Status, b.Status)
|
||||
}
|
||||
if hits != 1 {
|
||||
t.Fatalf("hits = %d, want 1 (second from cache)", hits)
|
||||
}
|
||||
if !b.Cached {
|
||||
t.Fatal("second lookup not cached")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLookupEmptyIsThrottled(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Write([]byte(`{"query":"x","results":[]}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
out := Lookup(context.Background(), "LadybugDB", LookupOpt{
|
||||
EnvPath: writeEnv(t, srv.URL),
|
||||
CachePath: filepath.Join(t.TempDir(), "c.sqlite"),
|
||||
Client: srv.Client(),
|
||||
Now: func() float64 { return 1_000 },
|
||||
Sleep: func(context.Context, time.Duration) error { return nil },
|
||||
})
|
||||
if out.Status != StatusThrottled {
|
||||
t.Fatalf("status = %s", out.Status)
|
||||
}
|
||||
}
|
||||
|
||||
func writeEnv(t *testing.T, url string) string {
|
||||
t.Helper()
|
||||
p := filepath.Join(t.TempDir(), "search.env")
|
||||
if err := os.WriteFile(p, []byte("BRAIN_SEARCH_URL="+url+"\n"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return p
|
||||
}
|
||||
@@ -0,0 +1,205 @@
|
||||
// Package websearch is the SearXNG client used as the second independent source.
|
||||
//
|
||||
// An empty result list from this instance is throttling, not evidence of absence.
|
||||
package websearch
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"regexp"
|
||||
"strings"
|
||||
"unicode"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
const (
|
||||
StatusOK = "ok"
|
||||
StatusThrottled = "throttled"
|
||||
|
||||
DefaultLimit = 5
|
||||
DefaultSnippetChars = 150
|
||||
MinInterval = 10.0
|
||||
CacheTTL = 7 * 24 * 3600
|
||||
)
|
||||
|
||||
var RetryBackoff = []float64{20, 60}
|
||||
|
||||
type Payload struct {
|
||||
Query string `json:"query"`
|
||||
Results []RawHit `json:"results"`
|
||||
UnresponsiveEngines [][]string `json:"unresponsive_engines"`
|
||||
}
|
||||
|
||||
type RawHit struct {
|
||||
Title string `json:"title"`
|
||||
URL string `json:"url"`
|
||||
Content string `json:"content"`
|
||||
Engine string `json:"engine"`
|
||||
}
|
||||
|
||||
type Hit struct {
|
||||
Rank int `json:"rank"`
|
||||
Title string `json:"title"`
|
||||
URL string `json:"url"`
|
||||
Snippet string `json:"snippet"`
|
||||
Engine string `json:"engine"`
|
||||
}
|
||||
|
||||
type Output struct {
|
||||
Query string `json:"query"`
|
||||
Status string `json:"status"`
|
||||
Results []Hit `json:"results"`
|
||||
Unresponsive []string `json:"unresponsive,omitempty"`
|
||||
Note string `json:"note,omitempty"`
|
||||
Cached bool `json:"cached,omitempty"`
|
||||
}
|
||||
|
||||
func Classify(p Payload) string {
|
||||
if len(p.Results) > 0 {
|
||||
return StatusOK
|
||||
}
|
||||
return StatusThrottled
|
||||
}
|
||||
|
||||
func Project(p Payload, limit, snippetChars int) Output {
|
||||
if limit <= 0 {
|
||||
limit = DefaultLimit
|
||||
}
|
||||
if snippetChars <= 0 {
|
||||
snippetChars = DefaultSnippetChars
|
||||
}
|
||||
status := Classify(p)
|
||||
n := limit
|
||||
if n > len(p.Results) {
|
||||
n = len(p.Results)
|
||||
}
|
||||
hits := make([]Hit, 0, n)
|
||||
for i := 0; i < n; i++ {
|
||||
item := p.Results[i]
|
||||
hits = append(hits, Hit{
|
||||
Rank: i + 1,
|
||||
Title: item.Title,
|
||||
URL: item.URL,
|
||||
Snippet: trimSnippet(item.Content, snippetChars),
|
||||
Engine: item.Engine,
|
||||
})
|
||||
}
|
||||
out := Output{
|
||||
Query: p.Query,
|
||||
Status: status,
|
||||
Results: hits,
|
||||
}
|
||||
for _, pair := range p.UnresponsiveEngines {
|
||||
if len(pair) >= 2 {
|
||||
out.Unresponsive = append(out.Unresponsive, pair[0]+": "+pair[1])
|
||||
} else if len(pair) == 1 {
|
||||
out.Unresponsive = append(out.Unresponsive, pair[0])
|
||||
}
|
||||
}
|
||||
if status == StatusThrottled {
|
||||
out.Note = "no engine answered - this is a throttled instance, not evidence that nothing exists"
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
var spaceRE = regexp.MustCompile(`\s+`)
|
||||
|
||||
func trimSnippet(s string, max int) string {
|
||||
s = strings.TrimSpace(spaceRE.ReplaceAllString(s, " "))
|
||||
if utf8.RuneCountInString(s) <= max {
|
||||
return s
|
||||
}
|
||||
runes := []rune(s)
|
||||
cut := strings.TrimRightFunc(string(runes[:max]), unicode.IsSpace)
|
||||
return cut + "..."
|
||||
}
|
||||
|
||||
func CacheKey(query string, params map[string]string) string {
|
||||
norm := strings.Join(strings.Fields(strings.ToLower(query)), " ")
|
||||
if params == nil {
|
||||
params = map[string]string{}
|
||||
}
|
||||
stable, _ := json.Marshal(params)
|
||||
sum := sha256.Sum256([]byte(norm + "\x00" + string(stable)))
|
||||
return hex.EncodeToString(sum[:])
|
||||
}
|
||||
|
||||
func WaitFor(last *float64, now, interval float64) float64 {
|
||||
if last == nil {
|
||||
return 0
|
||||
}
|
||||
d := interval - (now - *last)
|
||||
if d < 0 {
|
||||
return 0
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
func PHIReason(query string) string {
|
||||
for _, p := range phiPatterns {
|
||||
if p.re.MatchString(query) {
|
||||
return p.reason
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
type phiPat struct {
|
||||
re *regexp.Regexp
|
||||
reason string
|
||||
}
|
||||
|
||||
var phiPatterns = []phiPat{
|
||||
{regexp.MustCompile(`\d{6,}`), "a run of six or more digits looks like an ID"},
|
||||
{regexp.MustCompile(`(?i)\bpersonalnummer\b`), "Personalnummer is staff data"},
|
||||
{regexp.MustCompile(`(?i)\bkv[-\s]?nr\b`), "KV-Nr is an insurance number"},
|
||||
{regexp.MustCompile(`(?i)\bversichertennummer\b`), "insurance number"},
|
||||
{regexp.MustCompile(`(?i)\b[A-Za-zÄÖÜäöüß]+(?:stra(?:ss|ß)e|str\.)\s*\d+`), "a street with a house number looks like an address"},
|
||||
{regexp.MustCompile(`(?i)\bgeb(?:urtsdatum)?\.?\s*\d{1,2}[./]\d{1,2}[./]\d{2,4}`), "a date of birth"},
|
||||
}
|
||||
|
||||
func (o Output) YAML() string {
|
||||
var b strings.Builder
|
||||
fmt.Fprintf(&b, "query: %s\n", yamlScalar(o.Query))
|
||||
fmt.Fprintf(&b, "status: %s\n", yamlScalar(o.Status))
|
||||
if len(o.Results) == 0 {
|
||||
b.WriteString("results: []\n")
|
||||
} else {
|
||||
b.WriteString("results:\n")
|
||||
for _, r := range o.Results {
|
||||
b.WriteString("-\n")
|
||||
fmt.Fprintf(&b, " rank: %d\n", r.Rank)
|
||||
fmt.Fprintf(&b, " title: %s\n", yamlScalar(r.Title))
|
||||
fmt.Fprintf(&b, " url: %s\n", yamlScalar(r.URL))
|
||||
fmt.Fprintf(&b, " snippet: %s\n", yamlScalar(r.Snippet))
|
||||
fmt.Fprintf(&b, " engine: %s\n", yamlScalar(r.Engine))
|
||||
}
|
||||
}
|
||||
if len(o.Unresponsive) > 0 {
|
||||
b.WriteString("unresponsive:\n")
|
||||
for _, u := range o.Unresponsive {
|
||||
fmt.Fprintf(&b, "- %s\n", yamlScalar(u))
|
||||
}
|
||||
}
|
||||
if o.Note != "" {
|
||||
fmt.Fprintf(&b, "note: %s\n", yamlScalar(o.Note))
|
||||
}
|
||||
if o.Cached {
|
||||
b.WriteString("cached: true\n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func yamlScalar(s string) string {
|
||||
if strings.Contains(s, "\n") {
|
||||
b, _ := json.Marshal(s)
|
||||
return string(b)
|
||||
}
|
||||
if s == "" || strings.ContainsAny(s, ":#'\"[]{}&*!|>%@`") || s != strings.TrimSpace(s) {
|
||||
b, _ := json.Marshal(s)
|
||||
return string(b)
|
||||
}
|
||||
return s
|
||||
}
|
||||
@@ -0,0 +1,203 @@
|
||||
package websearch
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strings"
|
||||
"testing"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
func loadFixture(t *testing.T, name string) Payload {
|
||||
t.Helper()
|
||||
_, file, _, ok := runtime.Caller(0)
|
||||
if !ok {
|
||||
t.Fatal("runtime.Caller")
|
||||
}
|
||||
path := filepath.Join(filepath.Dir(file), "..", "..", "bin", "tools", "web-search", "fixtures", name)
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var p Payload
|
||||
if err := json.Unmarshal(raw, &p); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
func TestClassifyHealthyIsOK(t *testing.T) {
|
||||
if got := Classify(loadFixture(t, "healthy.json")); got != StatusOK {
|
||||
t.Fatalf("classify healthy = %q, want ok", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestClassifyEmptyIsThrottledNotEmpty(t *testing.T) {
|
||||
got := Classify(loadFixture(t, "throttled.json"))
|
||||
if got != StatusThrottled {
|
||||
t.Fatalf("classify empty = %q, want throttled", got)
|
||||
}
|
||||
if got == "empty" || got == "no_results" {
|
||||
t.Fatal("status must never sound like absence")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProjectKeepsContextFields(t *testing.T) {
|
||||
out := Project(loadFixture(t, "healthy.json"), 3, DefaultSnippetChars)
|
||||
if out.Status != StatusOK {
|
||||
t.Fatalf("status = %q", out.Status)
|
||||
}
|
||||
if len(out.Results) != 3 {
|
||||
t.Fatalf("len = %d, want 3", len(out.Results))
|
||||
}
|
||||
r := out.Results[0]
|
||||
if r.Rank != 1 || r.Title == "" || r.URL == "" {
|
||||
t.Fatalf("hit = %+v", r)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProjectTrimsSnippet(t *testing.T) {
|
||||
out := Project(loadFixture(t, "healthy.json"), 5, 40)
|
||||
for _, r := range out.Results {
|
||||
n := utf8.RuneCountInString(r.Snippet)
|
||||
if n > 43 {
|
||||
t.Fatalf("snippet len %d > 43: %q", n, r.Snippet)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestProjectIsCheaperThanRaw(t *testing.T) {
|
||||
raw, err := os.ReadFile(filepath.Join(fixtureDir(t), "healthy.json"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
out, err := json.Marshal(Project(loadFixture(t, "healthy.json"), 5, DefaultSnippetChars))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(out)*3 >= len(raw) {
|
||||
t.Fatalf("projected %d not cheaper than raw %d", len(out), len(raw))
|
||||
}
|
||||
}
|
||||
|
||||
func TestThrottledProjectionCarriesEngineReasons(t *testing.T) {
|
||||
out := Project(loadFixture(t, "throttled.json"), 5, DefaultSnippetChars)
|
||||
if out.Status != StatusThrottled {
|
||||
t.Fatalf("status = %q", out.Status)
|
||||
}
|
||||
if len(out.Results) != 0 {
|
||||
t.Fatalf("results = %v", out.Results)
|
||||
}
|
||||
if len(out.Unresponsive) == 0 {
|
||||
t.Fatal("unresponsive empty")
|
||||
}
|
||||
if !strings.Contains(out.Note, "not evidence that nothing exists") {
|
||||
t.Fatalf("note = %q", out.Note)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCacheKeyStable(t *testing.T) {
|
||||
if CacheKey("Pflegegrad", nil) != CacheKey("Pflegegrad", map[string]string{}) {
|
||||
t.Fatal("nil vs empty params")
|
||||
}
|
||||
if CacheKey(" Pflegegrad ", nil) != CacheKey("pflegegrad", nil) {
|
||||
t.Fatal("case/padding")
|
||||
}
|
||||
if CacheKey("x", map[string]string{"lang": "de"}) == CacheKey("x", nil) {
|
||||
t.Fatal("params must change key")
|
||||
}
|
||||
a := CacheKey("x", map[string]string{"a": "1", "b": "2"})
|
||||
b := CacheKey("x", map[string]string{"b": "2", "a": "1"})
|
||||
if a != b {
|
||||
t.Fatal("param order must not change key")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPHIGuard(t *testing.T) {
|
||||
if PHIReason("Pflegegrad SGB XI Einstufung") != "" {
|
||||
t.Fatal("technical query refused")
|
||||
}
|
||||
if PHIReason("site:example.com technical query") != "" {
|
||||
t.Fatal("site query refused")
|
||||
}
|
||||
if PHIReason("SGB XI Paragraph 45b") != "" {
|
||||
t.Fatal("short numbers refused")
|
||||
}
|
||||
if PHIReason("Kunde 4711220385 Adresse") == "" {
|
||||
t.Fatal("long digit run allowed")
|
||||
}
|
||||
if PHIReason("KV-Nr A123456789") == "" {
|
||||
t.Fatal("KV-Nr allowed")
|
||||
}
|
||||
if PHIReason("Hauptstraße 14 Berlin") == "" {
|
||||
t.Fatal("street allowed")
|
||||
}
|
||||
if PHIReason("Lindenstr. 7") == "" {
|
||||
t.Fatal("str. allowed")
|
||||
}
|
||||
if PHIReason("Personalnummer 12") == "" {
|
||||
t.Fatal("Personalnummer allowed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWaitFor(t *testing.T) {
|
||||
last := 100.0
|
||||
if got := WaitFor(&last, 104.0, 10); got != 6 {
|
||||
t.Fatalf("wait = %v, want 6", got)
|
||||
}
|
||||
if got := WaitFor(&last, 130.0, 10); got != 0 {
|
||||
t.Fatalf("wait = %v, want 0", got)
|
||||
}
|
||||
if got := WaitFor(nil, 130.0, 10); got != 0 {
|
||||
t.Fatalf("first call wait = %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSQLiteCacheRoundTrip(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
c, err := OpenCache(filepath.Join(dir, "web-search.sqlite"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer c.Close()
|
||||
p := loadFixture(t, "healthy.json")
|
||||
key := CacheKey("pflegegrad", nil)
|
||||
if got, err := c.Get(key, CacheTTL, 1_000); err != nil || got != nil {
|
||||
t.Fatalf("empty get = %v %v", got, err)
|
||||
}
|
||||
if err := c.Put(key, p, 1_000); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := c.Get(key, CacheTTL, 1_001)
|
||||
if err != nil || got == nil {
|
||||
t.Fatalf("get = %v %v", got, err)
|
||||
}
|
||||
if Classify(*got) != StatusOK {
|
||||
t.Fatalf("cached classify = %s", Classify(*got))
|
||||
}
|
||||
expired, err := c.Get(key, 10, 2_000)
|
||||
if err != nil || expired != nil {
|
||||
t.Fatalf("expired = %v %v", expired, err)
|
||||
}
|
||||
if v, err := c.LastCall(); err != nil || v != nil {
|
||||
t.Fatalf("last = %v %v", v, err)
|
||||
}
|
||||
if err := c.MarkCall(50); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
v, err := c.LastCall()
|
||||
if err != nil || v == nil || *v != 50 {
|
||||
t.Fatalf("last after mark = %v %v", v, err)
|
||||
}
|
||||
}
|
||||
|
||||
func fixtureDir(t *testing.T) string {
|
||||
t.Helper()
|
||||
_, file, _, ok := runtime.Caller(0)
|
||||
if !ok {
|
||||
t.Fatal("runtime.Caller")
|
||||
}
|
||||
return filepath.Join(filepath.Dir(file), "..", "..", "bin", "tools", "web-search", "fixtures")
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user