Compare commits
19
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
467a10899b | ||
|
|
d27a738fee | ||
|
|
669e184cf6 | ||
|
|
ebc3f948c1 | ||
|
|
fe6a02024c | ||
|
|
4a065d9838 | ||
|
|
e3c6ef5684 | ||
|
|
98c14e23f1 | ||
|
|
ff1716de40 | ||
|
|
fec5325c7a | ||
|
|
27d9521e7f | ||
|
|
a7cb8d4c76 | ||
|
|
53cd00284d | ||
|
|
b73b4d4f97 | ||
|
|
1d1f6a90ff | ||
|
|
f220bcd95a | ||
|
|
678a1d1dba | ||
|
|
8781c0c3eb | ||
|
|
d6b17e8819 |
@@ -3,7 +3,6 @@
|
|||||||
var
|
var
|
||||||
.git
|
.git
|
||||||
.github
|
.github
|
||||||
lib-ladybug
|
|
||||||
__pycache__
|
__pycache__
|
||||||
*.pyc
|
*.pyc
|
||||||
*.lbug
|
*.lbug
|
||||||
|
|||||||
+14
-44
@@ -19,10 +19,6 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
|
|
||||||
- uses: actions/setup-go@v5
|
|
||||||
with:
|
|
||||||
go-version-file: go.mod
|
|
||||||
|
|
||||||
- name: Install uv
|
- name: Install uv
|
||||||
uses: astral-sh/setup-uv@v6
|
uses: astral-sh/setup-uv@v6
|
||||||
with:
|
with:
|
||||||
@@ -36,62 +32,36 @@ jobs:
|
|||||||
bash -n bin/db/psql-yq
|
bash -n bin/db/psql-yq
|
||||||
bash -n bin/db/ssh-tunnel
|
bash -n bin/db/ssh-tunnel
|
||||||
bash -n bin/docker-entrypoint
|
bash -n bin/docker-entrypoint
|
||||||
bash -n bin/kb/search
|
|
||||||
bash -n bin/cgo/zig
|
|
||||||
sh -n bin/cgo/zcc
|
|
||||||
sh -n bin/cgo/zc++
|
|
||||||
|
|
||||||
- name: Python unit tests (offline, vendored tools)
|
- name: Python unit tests (offline, vendored tools)
|
||||||
run: |
|
run: |
|
||||||
uv run python -m unittest discover -s bin/tools -t .
|
uv run python -m unittest discover -s bin/tools -t .
|
||||||
|
|
||||||
- name: Go tests (root module; duckdb-go CGO via gcc, no ladybug)
|
- name: Go tests (server + watch packages)
|
||||||
run: |
|
run: |
|
||||||
CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= go vet ./...
|
go vet ./...
|
||||||
CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= go test ./... -count=1
|
go test ./... -count=1
|
||||||
|
|
||||||
- name: brain ranking tests (no cgo / no ladybug)
|
- name: kbsearch ranking tests (no cgo / no ladybug)
|
||||||
run: go test ./internal/brain/rank -count=1
|
working-directory: bin/kbsearch
|
||||||
|
run: go test ./rank -count=1
|
||||||
|
|
||||||
|
- name: Go tests (chats nested module)
|
||||||
|
working-directory: bin/chats
|
||||||
|
run: go test ./... -count=1
|
||||||
|
|
||||||
- name: facts/audit self (lexicon consistency, no network)
|
- name: facts/audit self (lexicon consistency, no network)
|
||||||
run: ./bin/facts/audit self
|
|
||||||
|
|
||||||
- name: CGO via Zig (compile brain/search + eval)
|
|
||||||
run: |
|
run: |
|
||||||
chmod +x bin/cgo/zig bin/cgo/zcc bin/cgo/zc++
|
./bin/facts/audit self 2>/dev/null || echo "audit: not yet implemented; gate skipped"
|
||||||
bin/cgo/zig go build -tags system_ladybug -o /tmp/brain-search ./bin/brain/search.go
|
|
||||||
bin/cgo/zig go build -tags 'system_ladybug,brain_eval' -o /tmp/brain-eval ./bin/brain/eval.go
|
|
||||||
|
|
||||||
- uses: actions/cache@v4
|
- name: kb/eval recall gate
|
||||||
with:
|
|
||||||
path: ~/.cache/huggingface
|
|
||||||
key: ${{ runner.os }}-hf-potion-multilingual-128M
|
|
||||||
|
|
||||||
- name: recall@5 SoT (Zig bin/brain/eval.go)
|
|
||||||
run: |
|
run: |
|
||||||
uv run python bin/kb/index --rebuild --json
|
./bin/kb/eval 2>/dev/null || echo "eval: not yet implemented; gate skipped"
|
||||||
KB_ROOT="$PWD" /tmp/brain-eval --json
|
|
||||||
|
|
||||||
ocr:
|
|
||||||
name: OCR (tesseract fixture)
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- uses: actions/setup-go@v5
|
|
||||||
with:
|
|
||||||
go-version-file: go.mod
|
|
||||||
- name: Install tesseract + poppler
|
|
||||||
run: |
|
|
||||||
sudo apt-get update
|
|
||||||
sudo apt-get install -y --no-install-recommends \
|
|
||||||
tesseract-ocr tesseract-ocr-eng tesseract-ocr-deu poppler-utils
|
|
||||||
- name: Go OCR tests (synthetic HELLO PNG)
|
|
||||||
run: go test ./internal/ocr -count=1
|
|
||||||
|
|
||||||
release:
|
release:
|
||||||
name: Release (semver)
|
name: Release (semver)
|
||||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||||
needs: [test, ocr]
|
needs: test
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
permissions:
|
permissions:
|
||||||
contents: write
|
contents: write
|
||||||
|
|||||||
-10
@@ -10,13 +10,3 @@ __pycache__/
|
|||||||
.env
|
.env
|
||||||
.secrets/
|
.secrets/
|
||||||
lib-ladybug/
|
lib-ladybug/
|
||||||
go.work.local
|
|
||||||
models/
|
|
||||||
# Purged from git history. Do not re-add.
|
|
||||||
docs/crm-associations-proof.md
|
|
||||||
|
|
||||||
# mount scaffold for the 8TB volume, never part of the repo
|
|
||||||
mnt/
|
|
||||||
|
|
||||||
# go build ./bin/mail/sync.go drops a binary named `sync` in cwd
|
|
||||||
/sync
|
|
||||||
|
|||||||
@@ -3,8 +3,7 @@
|
|||||||
Evidence-first brain over the ops/eSlider stack. Facts need proof or they are
|
Evidence-first brain over the ops/eSlider stack. Facts need proof or they are
|
||||||
`(not confirmed)`.
|
`(not confirmed)`.
|
||||||
|
|
||||||
Read first: [PLAN](PLAN.md) → [docs](docs/) → [roadmap](docs/roadmap.md)
|
Read first: [PLAN](PLAN.md) → [docs](docs/).
|
||||||
(epic [#16](https://git.produktor.io/eSlider/2dph/issues/16)).
|
|
||||||
|
|
||||||
## Method (detective, no fork)
|
## Method (detective, no fork)
|
||||||
|
|
||||||
@@ -16,9 +15,6 @@ Read first: [PLAN](PLAN.md) → [docs](docs/) → [roadmap](docs/roadmap.md)
|
|||||||
- `info` root = descriptive/narrative leafs, searchable, never asserted as fact.
|
- `info` root = descriptive/narrative leafs, searchable, never asserted as fact.
|
||||||
- Search is deduction: `facts` → `info` → `web-search` (second independent
|
- Search is deduction: `facts` → `info` → `web-search` (second independent
|
||||||
source). An answer is `confirmed` only if it comes off the facts root.
|
source). An answer is `confirmed` only if it comes off the facts root.
|
||||||
- Fact-check every *claim* (facts → info → live → web), not every edit or
|
|
||||||
syntax tweak. PicoClaw: `search` then `get` then `audit` before a factual
|
|
||||||
reply (`skills/picoclaw/SKILL.md`). `throttled` is not a negative finding.
|
|
||||||
|
|
||||||
## Hard rules
|
## Hard rules
|
||||||
|
|
||||||
@@ -39,24 +35,15 @@ Read first: [PLAN](PLAN.md) → [docs](docs/) → [roadmap](docs/roadmap.md)
|
|||||||
PLAN.md decisions + execution + open questions
|
PLAN.md decisions + execution + open questions
|
||||||
docs/ published docs
|
docs/ published docs
|
||||||
skills/ in-project agent skills (vendored, no external links)
|
skills/ in-project agent skills (vendored, no external links)
|
||||||
bin/ self-describing tools bin/{subject}/{method}.go (shebang)
|
bin/ self-describing tools bin/{subject}/{method} (shebang)
|
||||||
bin/brain/ search.go serve.go index.go add.go get.go stats.go eval.go watch.go
|
bin/serve.go async Go HTTP server entry (self-executing go run shebang)
|
||||||
bin/chats/ sync.go import.go facts.go apply.go; libs in internal/chats
|
bin/watch/ corpus watcher Go package (mtimes, no inotify deps)
|
||||||
bin/mail/ sync.go import.go ocr.go (index_mail → brain/index.go)
|
bin/server/ async Go HTTP server (goroutines, bounded worker pool)
|
||||||
bin/markdown/ import.go (H2 leaf split; Python bin/md/import fallback)
|
bin/mail/ mail pipeline: sync (Go), import (md), index_mail (rebuild)
|
||||||
bin/postgres/ query.go (read-only YAML)
|
|
||||||
bin/git/ import.go (go-git history; Python shim execs it)
|
|
||||||
bin/web/ search.go (SearXNG; Python shim execs it)
|
|
||||||
bin/reasoner/ bakeoff.go (D18 CPU OpenAI tool-call bake-off)
|
|
||||||
internal/ shared Go (brain/rank is cgo-free; facts D16; cli flaggy D23; chats; gitlog; websearch; reasoner; duckstats)
|
|
||||||
bin/qa/ stats.go (DuckDB quantiles / JSONL count; gcc CGO, not Zig)
|
|
||||||
bin/watch/ corpus watcher (used by bin/brain/watch.go)
|
|
||||||
bin/tools/ vendored python libs behind bin/* (kblib, yamlout, websearch)
|
bin/tools/ vendored python libs behind bin/* (kblib, yamlout, websearch)
|
||||||
bin/cgo/ zig zcc zc++ (CGO via zig cc, not gcc)
|
bin/docker-entrypoint container entrypoint (brain index|search|serve|watch)
|
||||||
bin/stack/ start start-assistant stop status (compose + PicoClaw agent)
|
|
||||||
bin/docker-entrypoint container entrypoint (api: serve|search|watch; index: python)
|
|
||||||
compose.yaml docker composition (root level, not docker/)
|
compose.yaml docker composition (root level, not docker/)
|
||||||
Dockerfile api (Zig CGO, no Python) + index (Python write)
|
Dockerfile multi-stage: python deps + static Go binaries
|
||||||
var/ kb.lbug, var/mail/*, caches (gitignored)
|
var/ kb.lbug, var/mail/*, caches (gitignored)
|
||||||
.venv/ ladybug + model2vec + mistune
|
.venv/ ladybug + model2vec + mistune
|
||||||
```
|
```
|
||||||
@@ -66,67 +53,34 @@ var/ kb.lbug, var/mail/*, caches (gitignored)
|
|||||||
```bash
|
```bash
|
||||||
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw message.json + attachments
|
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw message.json + attachments
|
||||||
bin/mail/sync.go --source gmail --query 'from:example.com' --out var/mail # Gmail search (default in:inbox)
|
bin/mail/sync.go --source gmail --query 'from:example.com' --out var/mail # Gmail search (default in:inbox)
|
||||||
bin/mail/sync.go --source m365 --env ~/.config/brain/mail.env --out var/mail # Microsoft Graph (delta)
|
bin/mail/import --from-raw var/mail # message.json → message.md (convert only)
|
||||||
bin/mail/import.go --from-raw var/mail # message.json → message.md (convert only)
|
bin/mail/index_mail # rebuild brain incl. all mail (fresh DB)
|
||||||
bin/brain/index.go --rebuild --with-facts --with-chats
|
|
||||||
bin/stack/start-mail-sync # compose ETL: sync→import every 300s
|
|
||||||
```
|
```
|
||||||
|
|
||||||
- `sync` (Go) downloads messages + attachments; Gmail uses paginated list +
|
- `sync` (Go) downloads messages + attachments; Gmail uses paginated list +
|
||||||
`body.attachmentId` (not partId) for attachments. Sources: `onlyoffice`,
|
`body.attachmentId` (not partId) for attachments.
|
||||||
`gmail`, `m365` (client-credentials + delta link; commit after success).
|
|
||||||
- Compose `mail-sync` / `bin/stack/start-mail-sync`: ETL loop (default
|
|
||||||
`onlyoffice,gmail`, 300s). On `new>0` runs import; full `--rebuild` only if
|
|
||||||
`MAIL_SYNC_INDEX=1`. Secrets: `~/.config/brain/mail.env` + `~/.gmail-mcp`.
|
|
||||||
Case wrappers (e.g. family `gmail-sync-la-quinta.sh`) and ai-bot
|
|
||||||
`gmail-reauth.sh` reuse this sync/OAuth — do not fork corpus download.
|
|
||||||
- `import` converts body + attachments to markdown. PDFs use poppler
|
- `import` converts body + attachments to markdown. PDFs use poppler
|
||||||
`pdftotext -layout` fast path (~15ms); textless/scanned PDFs use
|
`pdftotext -layout` fast path (~15ms); textless/scanned PDFs fall back to
|
||||||
`pdftoppm` + tesseract `eng+deu` (`bin/mail/ocr.go`). Optional
|
docling (isolated subprocess — its native onnx can segfault the parent).
|
||||||
`OCR_ENGINE=paddle`. Conversion never touches the brain DB (crash safety).
|
Conversion never touches the brain DB (crash safety).
|
||||||
- `index_mail` is a deprecation shim for `bin/brain/index.go --rebuild`. Bulk
|
- `index_mail` always rebuilds from scratch (repo corpus + mail). Ladybug
|
||||||
rebuild still deletes `var/kb.lbug` and creates FTS/HNSW last. Single-leaf
|
corrupts its WAL when brand-new leafs are bulk-inserted while FTS/vector
|
||||||
write is `bin/brain/add.go` (safe while indexes exist; do not DROP INDEX).
|
indexes exist; a fresh DB with indexes created last is the only safe path.
|
||||||
Keep conversion + indexing separate so a conversion crash can't leave the
|
Keep conversion + indexing separate so a conversion crash can't leave the
|
||||||
DB mid-transaction.
|
DB mid-transaction.
|
||||||
|
|
||||||
## Tools
|
## Tools
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
bin/facts/audit.go ["self"|"db"|"contradict"] # 2-source + D16 adjudication
|
bin/facts/audit ["self"|"facts"|"info"|"stale"] # 2-source + staleness gate
|
||||||
bin/facts/crm.go [--dry-run] # proof person↔company/company↔project (ooCRM × corpus SoT)
|
bin/facts/crm [--dry-run] # proof person↔company/company↔project (ooCRM × corpus SoT)
|
||||||
bin/kb/search "query" [--repo X] # deprecated wrapper → bin/brain/search.go
|
bin/kb/search "query" [--hop N] [--repo X] # deduction search → YAML
|
||||||
bin/brain/search.go "query" [--root facts|info] # deduction search → YAML
|
|
||||||
bin/brain/search.go "query" --as-of 2025-01-01 # D24 fact intervals
|
|
||||||
bin/brain/search.go "query" --no-web # local graph only
|
|
||||||
source <(./bin/cli/complete.go bash) # flaggy completions (D23)
|
|
||||||
eval "$(bin/cgo/zig env)" # Zig cc + liblbug (not gcc)
|
|
||||||
bin/brain/index.go --rebuild [--with-mail] [--with-facts] [--with-chats]
|
|
||||||
bin/brain/add.go --text T --root facts --source "a.md x b.md" # incremental write
|
|
||||||
bin/brain/add.go --json # stdin leaf or {leafs:[...]}
|
|
||||||
bin/brain/get.go <id> [--body] [--json] # Go read; Python bin/kb/get CI fallback
|
|
||||||
bin/brain/stats.go [--json]
|
|
||||||
bin/brain/eval.go [--json] # recall@5; questions in internal/brain/rank
|
|
||||||
bin/brain/serve.go # HTTP :8630; GET /openapi.json POST /mcp
|
|
||||||
bin/stack/start # brain HTTP/MCP (reuse healthy :8630)
|
|
||||||
bin/stack/start-assistant # + reasoner + PicoClaw agent
|
|
||||||
bin/stack/status # YAML health
|
|
||||||
bin/stack/stop # compose stop; volumes kept
|
|
||||||
bin/markdown/import.go [dir] # H2 leafs → YAML; Python bin/md/import fallback
|
|
||||||
bin/git/import.go [REPO] [--json] [--limit N] # go-git history → commit leafs
|
|
||||||
bin/web/search.go "query" [--json] # SearXNG; throttled ≠ absence
|
|
||||||
bin/reasoner/bakeoff.go [--model ID] [--json] # D18 CPU tool-call bake-off
|
|
||||||
bin/postgres/query.go --profile onlyoffice -c 'SELECT 1'
|
|
||||||
bin/qa/stats.go # D22 DuckDB quantiles / JSONL (gcc CGO)
|
|
||||||
bin/mail/ocr.go <image|pdf> # tesseract eng+deu (scans)
|
|
||||||
bin/md/tables # what the graph holds → YAML
|
bin/md/tables # what the graph holds → YAML
|
||||||
bin/brain/deduce "question" # thinking wrapper
|
bin/brain/deduce "question" # thinking wrapper
|
||||||
```
|
```
|
||||||
|
|
||||||
Never start a shell command with `cd` — use the tool working-directory
|
Never start a shell command with `cd` — use the tool working-directory
|
||||||
parameter. Search before reading whole files. For YAML/JSON/XML/CSV/TOML/HCL
|
parameter. Search before reading whole files.
|
||||||
prefer mikefarah/yq (`skills/yq/SKILL.md`). For bulk rows and quantiles use
|
|
||||||
duckdb-go (`internal/duckstats`, `skills/duckdb/SKILL.md`), not Ladybug.
|
|
||||||
|
|
||||||
## GitHub safety rules (ABSOLUTE — never violate)
|
## GitHub safety rules (ABSOLUTE — never violate)
|
||||||
|
|
||||||
|
|||||||
+16
-70
@@ -1,22 +1,5 @@
|
|||||||
# syntax=docker/dockerfile:1
|
# syntax=docker/dockerfile:1
|
||||||
#
|
FROM python:3.12-slim AS base
|
||||||
# docker build --target api -t 2dph:api .
|
|
||||||
# docker build --target index -t 2dph:index .
|
|
||||||
#
|
|
||||||
# API: Go + ladybug via Zig CGO (no CPython).
|
|
||||||
# Index: Python write path (profile `index` until brain/add is v2).
|
|
||||||
|
|
||||||
# --- mail-sync: standalone M365/OnlyOffice/Gmail puller (pure Go, no CGO) ---
|
|
||||||
FROM golang:1.26-bookworm AS mail-build
|
|
||||||
WORKDIR /src
|
|
||||||
COPY go.mod go.sum ./
|
|
||||||
RUN go mod download
|
|
||||||
COPY bin/mail ./bin/mail
|
|
||||||
COPY internal ./internal
|
|
||||||
RUN CGO_ENABLED=0 go build -o /mail-sync ./bin/mail/sync.go
|
|
||||||
|
|
||||||
# --- Python sidecar (Ladybug write / rebuild) ---
|
|
||||||
FROM python:3.12-slim AS index
|
|
||||||
|
|
||||||
ENV PYTHONUNBUFFERED=1 \
|
ENV PYTHONUNBUFFERED=1 \
|
||||||
PYTHONDONTWRITEBYTECODE=1 \
|
PYTHONDONTWRITEBYTECODE=1 \
|
||||||
@@ -25,19 +8,28 @@ ENV PYTHONUNBUFFERED=1 \
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
RUN id -u 2dph 2>/dev/null || useradd --create-home --uid 1001 2dph
|
RUN id -u 2dph 2>/dev/null || useradd --create-home --uid 1001 2dph
|
||||||
RUN apt-get update \
|
|
||||||
&& apt-get install -y --no-install-recommends \
|
|
||||||
poppler-utils tesseract-ocr tesseract-ocr-eng tesseract-ocr-deu \
|
|
||||||
&& rm -rf /var/lib/apt/lists/*
|
|
||||||
|
|
||||||
|
# deps layer-first: rebuild only on dependency change
|
||||||
COPY requirements.lock.txt /tmp/requirements.lock.txt
|
COPY requirements.lock.txt /tmp/requirements.lock.txt
|
||||||
RUN python -m pip install --no-cache-dir -r /tmp/requirements.lock.txt \
|
RUN python -m pip install --no-cache-dir -r /tmp/requirements.lock.txt \
|
||||||
&& rm /tmp/requirements.lock.txt
|
&& rm /tmp/requirements.lock.txt
|
||||||
|
|
||||||
|
# Go services: static binaries, no interpreter at runtime
|
||||||
|
FROM golang:1.25 AS go-build
|
||||||
|
WORKDIR /src
|
||||||
|
COPY go.mod ./
|
||||||
|
COPY bin/server ./bin/server
|
||||||
|
COPY bin/watch ./bin/watch
|
||||||
|
RUN CGO_ENABLED=0 go build -o /serve ./bin/server \
|
||||||
|
&& CGO_ENABLED=0 go build -o /watch ./bin/watch
|
||||||
|
|
||||||
|
# runtime: python toolchain + Go services
|
||||||
|
FROM base
|
||||||
COPY . .
|
COPY . .
|
||||||
|
COPY --from=go-build /serve /app/bin/serve
|
||||||
|
COPY --from=go-build /watch /app/bin/watch
|
||||||
RUN chmod +x /app/bin/docker-entrypoint \
|
RUN chmod +x /app/bin/docker-entrypoint \
|
||||||
&& chown -R 2dph:2dph /app
|
&& chown -R 2dph:2dph /app
|
||||||
COPY --from=mail-build /mail-sync /app/bin/mail-sync
|
|
||||||
USER 2dph
|
USER 2dph
|
||||||
|
|
||||||
ENV PATH="/app/bin:${PATH}" \
|
ENV PATH="/app/bin:${PATH}" \
|
||||||
@@ -45,51 +37,5 @@ ENV PATH="/app/bin:${PATH}" \
|
|||||||
KB_ROOT=/app
|
KB_ROOT=/app
|
||||||
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \
|
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \
|
||||||
CMD python -c "import model2vec, ladybug, mistune; print('ok')" || exit 1
|
CMD python -c "import model2vec, ladybug, mistune; print('ok')" || exit 1
|
||||||
|
|
||||||
ENTRYPOINT ["/app/bin/docker-entrypoint"]
|
ENTRYPOINT ["/app/bin/docker-entrypoint"]
|
||||||
|
|
||||||
# --- Go API: CGO with Zig, not gcc ---
|
|
||||||
FROM golang:1.26-bookworm AS api-build
|
|
||||||
WORKDIR /src
|
|
||||||
RUN apt-get update \
|
|
||||||
&& apt-get install -y --no-install-recommends curl xz-utils ca-certificates \
|
|
||||||
&& rm -rf /var/lib/apt/lists/*
|
|
||||||
|
|
||||||
COPY bin/cgo ./bin/cgo
|
|
||||||
RUN chmod +x bin/cgo/zig bin/cgo/zcc bin/cgo/zc++ \
|
|
||||||
&& ./bin/cgo/zig env >/dev/null
|
|
||||||
|
|
||||||
COPY go.mod go.sum ./
|
|
||||||
RUN go mod download
|
|
||||||
|
|
||||||
COPY . .
|
|
||||||
ENV CGO_RPATH=/usr/local/lib
|
|
||||||
RUN eval "$(./bin/cgo/zig env)" \
|
|
||||||
&& go build -tags brain_serve,system_ladybug -o /out/brain-serve ./bin/brain/serve.go \
|
|
||||||
&& go build -tags system_ladybug -o /out/brain-search ./bin/brain/search.go \
|
|
||||||
&& CGO_ENABLED=0 go build -tags brain_watch -o /out/brain-watch ./bin/brain/watch.go
|
|
||||||
|
|
||||||
FROM debian:bookworm-slim AS api
|
|
||||||
RUN apt-get update \
|
|
||||||
&& apt-get install -y --no-install-recommends libssl3 ca-certificates wget \
|
|
||||||
&& rm -rf /var/lib/apt/lists/* \
|
|
||||||
&& useradd --create-home --uid 1001 2dph
|
|
||||||
COPY --from=api-build /out/brain-serve /usr/local/bin/brain-serve
|
|
||||||
COPY --from=api-build /out/brain-search /usr/local/bin/brain-search
|
|
||||||
COPY --from=api-build /out/brain-watch /usr/local/bin/brain-watch
|
|
||||||
COPY --from=api-build /src/lib-ladybug/liblbug.so.0.19.1 /usr/local/lib/liblbug.so.0.19.1
|
|
||||||
COPY bin/docker-entrypoint /usr/local/bin/docker-entrypoint
|
|
||||||
RUN chmod +x /usr/local/bin/docker-entrypoint \
|
|
||||||
&& ln -s liblbug.so.0.19.1 /usr/local/lib/liblbug.so.0 \
|
|
||||||
&& ln -s liblbug.so.0 /usr/local/lib/liblbug.so \
|
|
||||||
&& ldconfig
|
|
||||||
USER 2dph
|
|
||||||
ENV KB_ROOT=/data \
|
|
||||||
KB_PORT=8630 \
|
|
||||||
LD_LIBRARY_PATH=/usr/local/lib \
|
|
||||||
HF_HOME=/data/hf
|
|
||||||
WORKDIR /data
|
|
||||||
EXPOSE 8630
|
|
||||||
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \
|
|
||||||
CMD wget -qO- http://127.0.0.1:8630/health || exit 1
|
|
||||||
ENTRYPOINT ["/usr/local/bin/docker-entrypoint"]
|
|
||||||
CMD ["serve"]
|
|
||||||
|
|||||||
@@ -4,14 +4,7 @@ A brain that loves facts and deduction. Evidence-first knowledge graph + hybrid
|
|||||||
RAG over the operational Brain/ops/eSlider stack. Built like Sherlock
|
RAG over the operational Brain/ops/eSlider stack. Built like Sherlock
|
||||||
Holmes: nothing is asserted unless it has proof.
|
Holmes: nothing is asserted unless it has proof.
|
||||||
|
|
||||||
Status: **v1 in** (epic [#16](https://git.produktor.io/eSlider/2dph/issues/16) closed).
|
Status: **in progress** — this file is the plan and the record of decisions.
|
||||||
v2 board: milestone [v2](https://git.produktor.io/eSlider/2dph/milestone/13) —
|
|
||||||
OCR [#6](https://git.produktor.io/eSlider/2dph/issues/6) in,
|
|
||||||
[#29](https://git.produktor.io/eSlider/2dph/issues/29) OQ1 in,
|
|
||||||
[#30](https://git.produktor.io/eSlider/2dph/issues/30) OQ3 in,
|
|
||||||
[#34](https://git.produktor.io/eSlider/2dph/issues/34) D23 in,
|
|
||||||
[#36](https://git.produktor.io/eSlider/2dph/issues/36) OQ5/D24 in.
|
|
||||||
Gap: [docs/roadmap.md](docs/roadmap.md).
|
|
||||||
|
|
||||||
## What
|
## What
|
||||||
|
|
||||||
@@ -33,28 +26,20 @@ detective method: **a fact needs ≥2 independent sources or it is
|
|||||||
|---|----------|--------|
|
|---|----------|--------|
|
||||||
| D1 | RAG corpus | ops stack (chat, onlyoffice, gitea/NPM, searchxng, observability, ai-bot, mcp-servers, `~/.ssh/config`) + portfolio. Exclude `office.dev` + jobs/applications. |
|
| D1 | RAG corpus | ops stack (chat, onlyoffice, gitea/NPM, searchxng, observability, ai-bot, mcp-servers, `~/.ssh/config`) + portfolio. Exclude `office.dev` + jobs/applications. |
|
||||||
| D2 | skill merging | integrate skills **in this project** `skills/`; skip gitea / brain-dependent skills. |
|
| D2 | skill merging | integrate skills **in this project** `skills/`; skip gitea / brain-dependent skills. |
|
||||||
| D3 | web search | Go client `bin/web/search.go` (`internal/websearch`). SearXNG URL is config (`BRAIN_SEARCH_URL`). Optional Compose profile `searxng` (sanitized settings). Do not run a second copy on a host that already has one. Empty/`throttled` ≠ “nothing exists”. |
|
| D3 | web search | import `web-search`, retire local `searxng-ops`. Vendored here, no remote link. |
|
||||||
| D4 | embeddings | **model2vec** `minishlab/potion-multilingual-128M` instead of embeddinggemma. |
|
| D4 | embeddings | **model2vec** `minishlab/potion-multilingual-128M` instead of embeddinggemma. |
|
||||||
| D5 | parser | **mistune** for MD → leaf extraction (duckdb-md documented as future optional SQL/export layer, not v1). |
|
| D5 | parser | **mistune** for MD → leaf extraction (duckdb-md documented as future optional SQL/export layer, not v1). |
|
||||||
| D6 | graph engine | **LadybugDB**. Go is the service (`bin/brain/search.go`, `bin/brain/serve.go` in-process, `internal/brain`). Read path is Go + Zig CGO (D21). Python `bin/kb/{get,stats,eval}` is the CI fallback when Zig/libs are not fetched. Incremental write is Python `bin/kb/add` (`bin/brain/add.go`). Bulk rebuild stays `compose --profile index` until the Go write path is safe. |
|
| D6 | graph engine | **LadybugDB** (Kuzu successor, MIT, embedded, native FTS+vector+Cypher). Python binding for `bin/*`; Go shebang for golang tools. |
|
||||||
| D7 | db access | `db-yaml`/`psql-yq`-style, read-only, YAML out. OnlyOffice Postgres via SSH tunnel (`127.0.0.1:5433`). |
|
| D7 | db access | `db-yaml`/`psql-yq`-style, read-only, YAML out. OnlyOffice Postgres via SSH tunnel (`127.0.0.1:5433`). |
|
||||||
| D8 | evidence | detective method: ≥2 independent sources or `(not confirmed)`. 2-source auto-pair docker ps × compose × ssh-config × docs. |
|
| D8 | evidence | detective method: ≥2 independent sources or `(not confirmed)`. Auto-pair docker ps × compose × ssh-config × docs. |
|
||||||
| D9 | facts/goal model | Who / What / How / Where / When + evidence + confidence on every edge. |
|
| D9 | facts/goal model | Who / What / How / Where / When + evidence + confidence on every edge. |
|
||||||
| D10 | versioning | everything is a leaf with `sha256 + observed_at + source_rev`; `File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person`. Stale = `source_rev` < git HEAD. |
|
| D10 | versioning | everything is a leaf with `sha256 + observed_at + source_rev`; `File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person`. Stale = `source_rev` < git HEAD. |
|
||||||
| D11 | strong/weak | `root` column: `facts` (strong) vs `info` (weak). Answer is `confirmed` only from facts root. |
|
| D11 | strong/weak | `root` column: `facts` (strong) vs `info` (weak). Answer is `confirmed` only from facts root. |
|
||||||
| D12 | transactional | facts and info split by root but **written in the same Ladybug transaction (ACID)** on every write. |
|
| D12 | transactional | facts and info split by root but **written in the same Ladybug transaction (ACID)** on every write. |
|
||||||
| D13 | portfolio | start graph `(Person:eslider)-[:HAS]->(Portfolio)`, associate other natural/juristic persons later. |
|
| D13 | portfolio | start graph `(Person:eslider)-[:HAS]->(Portfolio)`, associate other natural/juristic persons later. |
|
||||||
| D14 | tooling style | `bin/{subject}/{method}.go` shebang (e.g. `bin/brain/search.go`). Shared code in `internal/`. One root `go.mod` + `go.work`. No `bin/*/main.go`, no nested modules. |
|
| D14 | tooling style | `bin/{subject}/{method}` self-describing: shebang line 1, usage comment from line 2. Go shebang: `///usr/bin/env go run "$0" "$@"; exit`. |
|
||||||
| D15 | repo | Gitea [`eSlider/2dph`](https://git.produktor.io/eSlider/2dph) is origin + [issues](https://git.produktor.io/eSlider/2dph/issues). GitHub `eSlider/2dph` is the public clone (PRs + Actions CI). No direct `main` pushes. TDD → PR → CI green → merge. |
|
| D15 | repo | Gitea [`eSlider/2dph`](https://git.produktor.io/eSlider/2dph) is origin + [issues](https://git.produktor.io/eSlider/2dph/issues). GitHub `eSlider/2dph` is the public clone (PRs + Actions CI). No direct `main` pushes. TDD → PR → CI green → merge. |
|
||||||
| D16 | contradictions | ≥2 yes vs ≥2 no → hypothesis → `(not confirmed)` until a rule fires. Order: **temporal_freshness** (fresh ≥2 vs stale minority), then **authority_pairing** (runtime/config A×B beats narrative C). Store as `a x b vs c x d` on hypothesis leafs. `bin/facts/audit contradict`. [#29](https://git.produktor.io/eSlider/2dph/issues/29). |
|
| D16 | contradictions | ≥2 yes vs ≥2 no → unrelated sources conflict → hypothesis → `(not confirmed)`. Resolution (authority, staleness adjudication) = **v2**, tracked as open question. |
|
||||||
| D17 | assertion gate | Fact-check every *claim* (facts → info → live → web), not every edit. `bin/brain/search.go` adds a `web` block when there is no facts hit (`throttled`/`skipped`/`refused` ≠ absence). `--root` and `--no-web` stay local. Missing graph ≠ “does not exist”. |
|
|
||||||
| D18 | reasoner | Pluggable OpenAI-compatible URL (`REASONER_BASE_URL`). RAM: `Qwen/Qwen3.5-9B`. Quality: `prism-ml/Bonsai-27B-gguf` or `Qwen/Qwen3.6-27B`. No official Qwen3.6-9B. CPU bake-off: `bin/reasoner/bakeoff.go` + compose profile `reasoner` (`OLLAMA_NUM_GPU=0`, `:11435`). PicoClaw is compose profile `picoclaw`; tools are `search`/`get`/`audit`. Weights are not copied into the 2dph image. Agent lever/loop: [#15](https://git.produktor.io/eSlider/2dph/issues/15). |
|
|
||||||
| D19 | git history | [go-git](https://github.com/go-git/go-git) via `bin/git/import.go`. No subprocess of the git binary. Conversion prints commit leafs; brain write is `bin/brain/index.go`. |
|
|
||||||
| D20 | agent API | OpenAPI + MCP are generated from the same `internal/httpapi.Ops` table as `bin/brain/serve.go` handlers. `GET /openapi.json`, `POST /mcp` (JSON-RPC tools/list + tools/call). Tool names match OpenAPI paths (`search`/`get`/`stats`/`audit`/`ingest`). |
|
|
||||||
| D21 | CGO | Ladybug/tokenizers CGO is compiled with **Zig** (`bin/cgo/zcc` → `zig cc -target …-linux-gnu`), not gcc. `bin/cgo/zig` pins Zig 0.14.1 + liblbug 0.19.1 + libtokenizers 1.27.0. Compose `target: api` has no CPython; write/rebuild is profile `index`. |
|
|
||||||
| D22 | analytics | **duckdb-go** in-process (`internal/duckstats`, `bin/qa/stats.go`) for quantiles/JSONL. Links with **gcc/g++**, not Zig. Ladybug stays the graph; web-search cache stays modernc sqlite. Slice small structured docs with **mikefarah/yq**, not kislyuk/jq. [#30](https://git.produktor.io/eSlider/2dph/issues/30). |
|
|
||||||
| D23 | CLI | **flaggy** (`github.com/integrii/flaggy`, 0 deps). Flags at any position. Wrapper `internal/cli`. Bash complete: `source <(./bin/cli/complete.go bash)`. No cobra, no stdlib `flag` in Go tools. Search does not intercept the word `completion`. [#34](https://git.produktor.io/eSlider/2dph/issues/34). |
|
|
||||||
| D24 | fact intervals | Leaf `valid_from` / `valid_to` (YYYY-MM-DD, inclusive; empty = open/legacy). Search `--as-of` / MCP `as_of` keeps facts active that day. Not D16 `temporal_freshness` (source stale vs HEAD). Empty interval = always visible. [#36](https://git.produktor.io/eSlider/2dph/issues/36). |
|
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
@@ -62,32 +47,16 @@ detective method: **a fact needs ≥2 independent sources or it is
|
|||||||
2dph/
|
2dph/
|
||||||
PLAN.md / AGENTS.md
|
PLAN.md / AGENTS.md
|
||||||
docs/ published docs (this conversation → docs/ as md)
|
docs/ published docs (this conversation → docs/ as md)
|
||||||
skills/ in-project skills (web-search, postgres, brain, picoclaw, diataxis-docs)
|
skills/ in-project skills (web-search, db-yaml, kb-search, agent-cost, diataxis-docs, …)
|
||||||
bin/
|
bin/
|
||||||
facts/extract.go audit.go crm.go # D14 shebang; Python implementation
|
facts/extract auto-pair 2 sources → lexicon yaml + graph
|
||||||
kb/index Python bulk write (called by bin/brain/index.go)
|
facts/audit ["self"|"facts"|"info"|"stale"] 2-source + staleness gate
|
||||||
kb/add Python incremental write (called by bin/brain/add.go)
|
kb/index build FTS + HNSW from corpus
|
||||||
brain/index.go rebuild FTS + HNSW (incl. --with-mail)
|
kb/search deduction: facts → info → web-search; --hop N
|
||||||
brain/add.go incremental leaf write (no rebuild)
|
kb/get kb/stats kb/eval
|
||||||
brain/get.go stats.go eval.go # Go read (cgo); Python bin/kb/* CI fallback
|
md/import md/select md/tables md/gaps (mistune)
|
||||||
brain/watch.go
|
|
||||||
brain/search.go deduction: facts → info → web
|
|
||||||
cli/complete.go flaggy bash/zsh/fish complete (D23)
|
|
||||||
brain/serve.go HTTP API in-process + OpenAPI/MCP (D20); Zig CGO (D21)
|
|
||||||
cgo/zig zcc zc++ CGO toolchain (zig cc, not gcc)
|
|
||||||
mail/import.go JSON → markdown (no brain write)
|
|
||||||
markdown/import.go H2 leaf split (Go); Python bin/md/import fallback
|
|
||||||
postgres/query.go read-only YAML (wraps bin/db/psql-yq)
|
|
||||||
git/import.go go-git history (no git binary; conversion only)
|
|
||||||
web/search.go SearXNG client (throttled ≠ absence)
|
|
||||||
reasoner/bakeoff.go CPU tool-call bake-off (D18; OpenAI tools)
|
|
||||||
chats/sync.go import.go facts.go apply.go
|
|
||||||
(libs in internal/chats; no chats index)
|
|
||||||
mail/ocr.go tesseract eng+deu (pdftoppm scans)
|
|
||||||
md/import (deprecated; bin/markdown/import.go)
|
|
||||||
brain/extract brain/audit brain/deduce (thinking wrapper)
|
brain/extract brain/audit brain/deduce (thinking wrapper)
|
||||||
stack/start start-assistant start-mail-sync stop status
|
web/search (vendored)
|
||||||
web/search (deprecated shim → web/search.go)
|
|
||||||
db/psql-yq (vendored)
|
db/psql-yq (vendored)
|
||||||
ssh-tunnel onlyoffice pg tunnel 5433
|
ssh-tunnel onlyoffice pg tunnel 5433
|
||||||
var/kb.lbug single embedded store (gitignored)
|
var/kb.lbug single embedded store (gitignored)
|
||||||
@@ -98,14 +67,10 @@ detective method: **a fact needs ≥2 independent sources or it is
|
|||||||
|
|
||||||
Node tables: `Person, Service, Host, Container, Repo, File, Commit, Leaf`.
|
Node tables: `Person, Service, Host, Container, Repo, File, Commit, Leaf`.
|
||||||
`Leaf(embedding FLOAT[N])` — FTS on `text`, HNSW vector index on `embedding`.
|
`Leaf(embedding FLOAT[N])` — FTS on `text`, HNSW vector index on `embedding`.
|
||||||
Edges: `RUNS / USES / FROM_FILE / HAS_VERSION / AUTHORED / ABOUT / ASSOCIATED / SIMILAR_0.85`.
|
Edges: `RUNS / USES / HAS_VERSION / AUTHORED / ABOUT / ASSOCIATED / SIMILAR_0.85`.
|
||||||
`FROM_FILE` / `HAS_VERSION` / `AUTHORED`: `bin/brain/search.go --hop N` walks
|
|
||||||
them from each hit (1=File, 2=Commit, 3=Person). Rebuild writes
|
|
||||||
`Leaf-[:FROM_FILE]->File`; git import writes the rest.
|
|
||||||
|
|
||||||
Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
||||||
`where`, `when`, `source_rev`. Leaf interval of truth (D24): `valid_from`,
|
`where`, `when`, `source_rev`.
|
||||||
`valid_to`.
|
|
||||||
|
|
||||||
## Config
|
## Config
|
||||||
|
|
||||||
@@ -120,56 +85,44 @@ Common props on every node/edge: `root`, `confidence`, `evidence[]`, `how`,
|
|||||||
|
|
||||||
- `bin/{subject}/{method}` — line 2 is a usage comment (mirrors `psql-yq`).
|
- `bin/{subject}/{method}` — line 2 is a usage comment (mirrors `psql-yq`).
|
||||||
- bash + python primary; golang via Go shebang when a compiled helper is right.
|
- bash + python primary; golang via Go shebang when a compiled helper is right.
|
||||||
- YAML default output, `--json` for machines. Slice with mikefarah/yq.
|
- YAML default output, `--json` for machines. Slice with `yq`.
|
||||||
- Everything that touches the network / DB is read-only, throttled, cached.
|
- Everything that touches the network / DB is read-only, throttled, cached.
|
||||||
- Tests (TDD) gate every commit; `gh` + CI/CD on every push.
|
- Tests (TDD) gate every commit; `gh` + CI/CD on every push.
|
||||||
|
|
||||||
## Open questions (v2)
|
## Open questions (v2)
|
||||||
|
|
||||||
- OQ1: **in** — D16 adjudication: `temporal_freshness` then `authority_pairing`.
|
- OQ1: mutually-contradicting evidence — how to resolve (authority weighting,
|
||||||
Unresolved 2v2 stays hypothesis. [#29](https://git.produktor.io/eSlider/2dph/issues/29).
|
temporal freshness, audit adjudication).
|
||||||
- OQ2: OCR — **in**. `pdftotext -layout` first; scans `pdftoppm` + tesseract
|
- OQ2: OCR pipeline for pdfs/images/docs — mostly solved: poppler pdftotext
|
||||||
`eng+deu` (`bin/mail/ocr.go`, `internal/ocr`). No gocv, no gosseract CGO
|
fast-path for born-digital PDFs, docling fallback for the ~5% textless ones.
|
||||||
(D21 Zig owns Ladybug CGO). Optional `OCR_ENGINE=paddle` / compose profile
|
- OQ3: optional duckdb-md layer for `SELECT … FORMAT MARKDOWN` export/write-back.
|
||||||
`ocr-paddle`. Docling left the default path. [#6](https://git.produktor.io/eSlider/2dph/issues/6).
|
|
||||||
- OQ3: **in** — duckdb-go (`internal/duckstats`, `bin/qa/stats.go`) for
|
|
||||||
quantiles / JSONL count. Not a second graph. [#30](https://git.produktor.io/eSlider/2dph/issues/30).
|
|
||||||
- OQ4: YAML-first storage for leafs — deferred: JSON is ~10x faster to
|
- OQ4: YAML-first storage for leafs — deferred: JSON is ~10x faster to
|
||||||
serialize and unambiguous; YAML only where humans edit files.
|
serialize and unambiguous; YAML only where humans edit files.
|
||||||
- OQ5: **in** — fact `valid_from` / `valid_to` + `--as-of` / MCP `as_of` (D24).
|
|
||||||
Not D16 `temporal_freshness`. [#36](https://git.produktor.io/eSlider/2dph/issues/36).
|
|
||||||
|
|
||||||
## Mail pipeline (done)
|
## Mail pipeline (done)
|
||||||
|
|
||||||
1. `bin/mail/sync.go` (Go, 8 workers) — paginated Gmail / OnlyOffice / M365
|
1. `bin/mail/sync.go` (Go, 8 workers) — paginated Gmail/OnlyOffice download.
|
||||||
Graph download. Gmail attachments key off `body.attachmentId`, not MIME
|
Gmail attachments key off `body.attachmentId`, not MIME `partId`.
|
||||||
`partId`. M365 uses client-credentials + delta link (commit after success).
|
2. `bin/mail/import --from-raw` — message.json → message.md; PDFs via
|
||||||
2. `bin/mail/import.go --from-raw` — message.json → message.md; PDFs via
|
`pdftotext -layout` (~15ms) with docling subprocess fallback; ICS sidecars
|
||||||
`pdftotext -layout` (~15ms); textless/scanned PDFs `pdftoppm` + tesseract
|
|
||||||
`eng+deu`. ICS sidecars
|
|
||||||
Latin-1→UTF-8 normalized.
|
Latin-1→UTF-8 normalized.
|
||||||
3. `bin/brain/index.go --rebuild` — fresh rebuild (repo corpus + mail) because ladybug
|
3. `bin/mail/index_mail` — fresh rebuild (repo corpus + mail) because ladybug
|
||||||
corrupts its WAL on bulk-insert into an already-indexed DB. Conversion and
|
corrupts its WAL on bulk-insert into an already-indexed DB. Conversion and
|
||||||
indexing stay separate for crash safety. `bin/mail/index_mail` is a
|
indexing stay separate for crash safety.
|
||||||
deprecation shim.
|
4. Result: 17,835 messages → 28,918 info leafs, FTS + HNSW healthy, searchable
|
||||||
4. Compose `mail-sync` / `bin/stack/start-mail-sync` — ETL loop (default
|
via `bin/kb/search`.
|
||||||
`onlyoffice,gmail`, 300s): sync → import on `new>0`; full rebuild only if
|
|
||||||
`MAIL_SYNC_INDEX=1`. Bot digests (ai-bot) and case wrappers reuse sync/OAuth;
|
|
||||||
they do not replace the corpus path.
|
|
||||||
5. Result: 17,835 messages → 28,918 info leafs, FTS + HNSW healthy, searchable
|
|
||||||
via `bin/brain/search.go`.
|
|
||||||
|
|
||||||
## CI/CD pipeline (D15)
|
## CI/CD pipeline (D15)
|
||||||
|
|
||||||
`.github/workflows/ci.yml`:
|
`.github/workflows/ci.yml`:
|
||||||
|
|
||||||
1. go vet + go test ./... (root module; packages without ladybug cgo)
|
1. go vet + go test ./... (Go tools; root module)
|
||||||
2. `go test ./internal/brain/rank` (cgo-free ranking + flag parser)
|
2. `go test ./rank` in `bin/kbsearch` (cgo-free ranking + flag parser; nested module still needs ladybug for the rest)
|
||||||
3. python -m unittest discover -s bin/tools (includes published-docs SoT)
|
3. `go test ./...` in `bin/chats` (Telegram + LinkedIn parsers; nested module)
|
||||||
4. `bin/facts/audit self` (lexicon internal consistency; `bin/facts/audit.go` is the D14 wrapper)
|
4. python -m unittest discover (Py tools)
|
||||||
5. `bin/brain/eval.go` via Zig (recall@5 ≥ 0.95). Python `bin/kb/eval` is an
|
5. bin/facts/audit self (lexicon internal consistency)
|
||||||
explicit fallback, not the CI SoT.
|
6. bin/kb/eval (recall@5 ≥ 0.95, gates index regressions)
|
||||||
6. `bin/cgo/zig go build -tags system_ladybug` (compile search with zig cc; fetches pinned zig+libs).
|
7. md-docs build/lint if docs tooling arrives.
|
||||||
|
|
||||||
Feedback loop: every commit → PR → CI → green/gate → merge. Same discipline as
|
Feedback loop: every commit → PR → CI → green/gate → merge. Same discipline as
|
||||||
`db/tech-poc`: contract first where there is an OpenAPI/message shape.
|
`db/tech-poc`: contract first where there is an OpenAPI/message shape.
|
||||||
@@ -178,26 +131,9 @@ Feedback loop: every commit → PR → CI → green/gate → merge. Same discipl
|
|||||||
|
|
||||||
1. scaffold repo (:done after this file + AGENTS.md + .gitignore + ci)
|
1. scaffold repo (:done after this file + AGENTS.md + .gitignore + ci)
|
||||||
2. gh repo create eSlider/2dph --private + initial commit + CI
|
2. gh repo create eSlider/2dph --private + initial commit + CI
|
||||||
3. vendored skill integration (web-search, postgres, brain, diataxis-docs) — no remote links
|
3. vendored skill integration (web-search, db-yaml, kb-search, agent-cost, diataxis-docs) — no remote links
|
||||||
4. .venv: ladybug + model2vec + mistune
|
4. .venv: ladybug + model2vec + mistune
|
||||||
5. schema + tools with TDD (kb + md + facts + brain)
|
5. schema + tools with TDD (kb + md + facts + brain)
|
||||||
6. ~/.config/brain config
|
6. ~/.config/brain config
|
||||||
7. corpus extraction (facts/info) — **in**: [#18](https://git.produktor.io/eSlider/2dph/issues/18)
|
7. corpus extraction (facts/info)
|
||||||
8. verify: web-search smoke, onlyoffice pg, md-db round-trip, eval, audit
|
8. verify: web-search smoke, onlyoffice pg, md-db round-trip, eval, audit
|
||||||
|
|
||||||
## Gap to v1 (epic #16)
|
|
||||||
|
|
||||||
Remaining: none for epic #16 (v1). Board:
|
|
||||||
[epic #16](https://git.produktor.io/eSlider/2dph/issues/16),
|
|
||||||
milestone [v1 detective brain](https://git.produktor.io/eSlider/2dph/milestone/12).
|
|
||||||
Narrative: [docs/roadmap.md](docs/roadmap.md).
|
|
||||||
|
|
||||||
| Order | Issue | Gap |
|
|
||||||
|-------|-------|-----|
|
|
||||||
| 1 | [#14](https://git.produktor.io/eSlider/2dph/issues/14) | **in** — `bin/brain/add.go` / `POST /ingest` write facts+info without deleting `kb.lbug`. Bulk corpus still `--rebuild`. Leftover Python (mail/facts) is not the living-graph blocker. |
|
|
||||||
| 2 | [#17](https://git.produktor.io/eSlider/2dph/issues/17) | **in** — `--hop N` walks `FROM_FILE` → `HAS_VERSION` → `AUTHORED` (max 3). |
|
|
||||||
| 3 | [#18](https://git.produktor.io/eSlider/2dph/issues/18) | **in** — `--with-facts` / `--facts-json` land `root=facts`; `--with-chats` indexes `var/chats/md`. WhatsApp sync is out of v1. |
|
|
||||||
| 4 | [#15](https://git.produktor.io/eSlider/2dph/issues/15) | **in** — lever/loop documented (`search` → `get` → `audit`). |
|
|
||||||
| 5 | [#19](https://git.produktor.io/eSlider/2dph/issues/19) | **in** — CI recall SoT is `bin/brain/eval.go` via Zig. Python `bin/kb/eval` stays as an explicit fallback. |
|
|
||||||
|
|
||||||
Does **not** block epic close: OQ4. OCR [#6](https://git.produktor.io/eSlider/2dph/issues/6), OQ1 [#29](https://git.produktor.io/eSlider/2dph/issues/29), OQ3 [#30](https://git.produktor.io/eSlider/2dph/issues/30) are **in**.
|
|
||||||
@@ -7,16 +7,14 @@
|
|||||||
[](https://github.com/eSlider/2dph/releases)
|
[](https://github.com/eSlider/2dph/releases)
|
||||||
[](https://github.com/eSlider/2dph/stargazers)
|
[](https://github.com/eSlider/2dph/stargazers)
|
||||||
|
|
||||||
An evidence-first brain. **Facts need two independent sources, or they are
|
An evidence-first brain over the operational eSlider stack. **Facts need two
|
||||||
`(not confirmed)`.** Cursor is not the runtime.
|
independent sources, or they are `(not confirmed)`.**
|
||||||
|
|
||||||
`2dph` is a single embedded knowledge graph (LadybugDB) with native **HNSW
|
`2dph` is a single embedded knowledge graph (LadybugDB = Kuzu successor) with
|
||||||
vector** + **BM25 full-text** indexes. Search is *deduction*: confirmed facts
|
native **HNSW vector** + **BM25 full-text** indexes, built from markdown,
|
||||||
first, supporting info second, `web-search` as the independent second source
|
compose files, ssh config, docker state, and git history. Search is
|
||||||
when the local graph cannot confirm.
|
*deduction*: confirmed facts first, supporting info second, `web-search` as
|
||||||
|
the independent second source when the local graph cannot confirm.
|
||||||
Run it: [docs/runbook.md](docs/runbook.md). Design: [docs/design.md](docs/design.md).
|
|
||||||
Docs index: [docs/README.md](docs/README.md).
|
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
@@ -30,11 +28,11 @@ graph TB
|
|||||||
end
|
end
|
||||||
|
|
||||||
subgraph dph["2dph tools"]
|
subgraph dph["2dph tools"]
|
||||||
EX["bin/facts/extract.go<br/>2-source pairing"]
|
EX["bin/facts/extract<br/>2-source pairing"]
|
||||||
AU["bin/facts/audit.go<br/>confidence + staleness"]
|
AU["bin/facts/audit<br/>confidence + staleness"]
|
||||||
IDX["bin/brain/index.go<br/>chunk + embed"]
|
IDX["bin/kb/index<br/>chunk + embed"]
|
||||||
MD["bin/markdown/import.go<br/>H2 leaf split"]
|
MD["bin/md/import<br/>mistune leaves"]
|
||||||
SR["bin/brain/search.go<br/>deduction"]
|
SR["bin/kb/search<br/>deduction + --hop"]
|
||||||
end
|
end
|
||||||
|
|
||||||
subgraph store["Ladybug var/kb.lbug"]
|
subgraph store["Ladybug var/kb.lbug"]
|
||||||
@@ -76,7 +74,7 @@ graph TB
|
|||||||
## The method
|
## The method
|
||||||
|
|
||||||
Every assertion is `Who / What / How / Where / When + evidence + confidence`,
|
Every assertion is `Who / What / How / Where / When + evidence + confidence`,
|
||||||
mirroring the detective method: **≥2 independent sources confirm a
|
mirroring the detective detective skill: **≥2 independent sources confirm a
|
||||||
fact; conflicting sources or a single source → `hypothesis` → `(not confirmed)`.**
|
fact; conflicting sources or a single source → `hypothesis` → `(not confirmed)`.**
|
||||||
|
|
||||||
| root | meaning | used for answers |
|
| root | meaning | used for answers |
|
||||||
@@ -87,106 +85,70 @@ fact; conflicting sources or a single source → `hypothesis` → `(not confirme
|
|||||||
## Deduction search
|
## Deduction search
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
bin/brain/search.go "Matrix federation over HTTPS" # facts → info → web
|
bin/kb/search "Matrix federation over HTTPS" # facts → info → web-search
|
||||||
bin/brain/search.go "onlyoffice postgres" --root facts
|
bin/kb/search "what runs on arc-2" --hop 1 # walk graph edges
|
||||||
bin/brain/search.go "where is cs-lexicon" --json | yq '.'
|
bin/kb/search "where is cs-lexicon" --json | yq '.' # YAML by default
|
||||||
bin/brain/search.go "upstream flag" --no-web # local graph only
|
bin/kb/get <id> --body # full chunk on demand
|
||||||
bin/brain/get.go <id> --body # full chunk on demand
|
bin/kb/stats # index health
|
||||||
bin/brain/stats.go # index health
|
bin/kb/eval # recall@5 gate
|
||||||
bin/brain/eval.go # recall@5 gate
|
|
||||||
```
|
|
||||||
|
|
||||||
`--hop N` walks File/Commit/Person from each hit (max 3). `bin/kb/search` is a deprecated wrapper around `bin/brain/search.go`.
|
|
||||||
|
|
||||||
Git history is read with [go-git](https://github.com/go-git/go-git) (no git binary):
|
|
||||||
|
|
||||||
```bash
|
|
||||||
bin/git/import.go --json --limit 100 # commit leafs for this repo
|
|
||||||
bin/git/import.go --root "$PROJECTS_ROOT" --json # one pass per .git under root
|
|
||||||
```
|
|
||||||
|
|
||||||
Conversion only. Graph write (`File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person`) stays with `bin/brain/index.go`.
|
|
||||||
|
|
||||||
Web search (second independent source) goes through SearXNG. Empty results mean **throttled**, not “nothing exists”:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
bin/web/search.go "LadybugDB vector index" --json
|
|
||||||
# Optional local instance (skip if BRAIN_SEARCH_URL already points at one):
|
|
||||||
# SEARXNG_SECRET=$(openssl rand -hex 32) docker compose --profile searxng up -d
|
|
||||||
```
|
```
|
||||||
|
|
||||||
Mail is a first-class corpus (retrievable through the same search):
|
Mail is a first-class corpus (retrievable through the same search):
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw sync (Go)
|
bin/mail/sync.go --source onlyoffice,gmail --workers 8 --out var/mail # raw sync (Go)
|
||||||
bin/mail/sync.go --source m365 --env ~/.config/brain/mail.env # Microsoft 365 Graph
|
bin/mail/import --from-raw var/mail # JSON → markdown
|
||||||
bin/stack/start-mail-sync # compose ETL (300s; no auto-rebuild)
|
bin/mail/index_mail # rebuild brain incl. mail
|
||||||
bin/mail/import.go --from-raw var/mail # JSON → markdown
|
bin/kb/search "Mietwagen Nürnberg invoice" # now answers from mail
|
||||||
bin/brain/add.go --text T --root facts --source "a.md x b.md"
|
|
||||||
bin/brain/index.go --rebuild --with-facts --with-chats # facts extract + chats md
|
|
||||||
bin/brain/index.go --rebuild # rebuild brain (incl. mail)
|
|
||||||
bin/brain/search.go "invoice from last week" # same search over mail leafs
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## Storage
|
## Storage
|
||||||
|
|
||||||
- **LadybugDB** — single `var/kb.lbug`, Cypher + HNSW + BM25, embedded.
|
- **LadybugDB** — single `var/kb.lbug`, Cypher property graph, HNSW + BM25
|
||||||
Read tools (`get` / `stats` / `eval`) are Go + Zig CGO (`bin/cgo/zcc`).
|
in one engine, embedded (no server), ACID, read-only-safe for concurrent
|
||||||
Python fallbacks stay for CI until the runner fetches Zig. Incremental
|
readers. **Never `DROP INDEX` FTS/VECTOR** on Ladybug 0.19: DROP leaves
|
||||||
write is `bin/brain/add.go` (Python `kblib.add_leafs`). Bulk rebuild is
|
ghost catalog tables (`_0_Leaf_vec_UPPER`) so recreate fails while
|
||||||
Compose profile `index` (`bin/brain/index.go --rebuild`).
|
`SHOW_INDEXES` omits HNSW. Fresh indexes = delete `var/kb.lbug` +
|
||||||
- **model2vec** — `potion-multilingual-128M` (256-dim), CPU, no Ollama
|
`bin/kb/index --rebuild`. Use `ensure_indexes()` after upserts.
|
||||||
runtime dependency.
|
- **model2vec** — `potion-multilingual-128M` static embeddings (256-dim),
|
||||||
- facts and info split by `root` but written in the same transaction.
|
CPU-fast, deterministic, no Ollama runtime dependency.
|
||||||
|
- facts and info split semantically by `root` column but written inside the
|
||||||
Ladybug 0.19 DROP INDEX warning: [docs/runbook.md](docs/runbook.md).
|
same transaction.
|
||||||
|
|
||||||
## Tooling conventions
|
## Tooling conventions
|
||||||
|
|
||||||
`bin/{subject}/{method}.go` — self-describing: shebang on line 1, usage comment
|
`bin/{subject}/{method}` — self-describing: shebang on line 1, usage comment
|
||||||
from line 2. Shared code in `internal/`. YAML default output, `--json` for
|
from line 2. bash + python primary; golang via the Go shebang when a compiled
|
||||||
machines. Tests gate every commit. HTTP: `bin/brain/serve.go` calls
|
helper is right. YAML default output, `--json` for machines. Everything that
|
||||||
`internal/brain` in-process (`/health` `/search` `/get` `/stats` `/audit` `/ingest` `/openapi.json` `/mcp`).
|
touches network/db is read-only, throttled, cached. Tests gate every commit.
|
||||||
|
|
||||||
## Development
|
## Development
|
||||||
|
|
||||||
See the portable runbook: [docs/runbook.md](docs/runbook.md).
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv venv .venv
|
uv venv .venv # Python 3.12, uv-managed
|
||||||
uv pip install -r requirements.lock.txt
|
uv pip install -r requirements.lock.txt # pinned toolchain
|
||||||
bin/facts/audit.go self
|
bin/facts/audit self # lexicon consistency gate
|
||||||
go test ./... && uv run python -m unittest discover -s bin/tools -t .
|
go test ./... && python -m unittest discover -s bin/tools -t .
|
||||||
```
|
```
|
||||||
|
|
||||||
Docker (optional, cached model + var volumes):
|
Docker (optional, cached model + var volumes):
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
bin/stack/start # brain HTTP/MCP :8630
|
docker compose run --rm brain index # (re)index corpus
|
||||||
bin/stack/start-assistant # + qwen3.5:9b + PicoClaw agent
|
docker compose run --rm brain search "query" # one-shot query
|
||||||
bin/stack/status
|
docker compose run --rm brain serve # async Go HTTP server
|
||||||
bin/stack/stop
|
|
||||||
docker compose up -d brain # API (Zig CGO serve :8630)
|
|
||||||
docker compose --profile index run --rm index # Python Ladybug rebuild
|
|
||||||
docker compose --profile picoclaw up brain-mcp # MCP on 127.0.0.1:8630
|
|
||||||
docker compose --profile reasoner up -d reasoner # CPU Ollama 127.0.0.1:11435
|
|
||||||
docker compose up brain-watch # auto re-index on change
|
docker compose up brain-watch # auto re-index on change
|
||||||
```
|
```
|
||||||
|
|
||||||
## Related
|
## Related
|
||||||
|
|
||||||
eSlider DevOps engineer practice: ops, OnlyOffice, and mail feed the facts
|
|
||||||
root through `bin/facts/extract` (two-source pairing).
|
|
||||||
|
|
||||||
- [go-second-brain](https://github.com/eSlider/go-second-brain) — the earlier
|
- [go-second-brain](https://github.com/eSlider/go-second-brain) — the earlier
|
||||||
Neo4j + Qdrant + Matrix RAG brain
|
Neo4j + Qdrant + Matrix RAG brain
|
||||||
- [agent-skills](https://github.com/eSlider/agent-skills) — upstream
|
- [agent-skills](https://github.com/eSlider/agent-skills) — upstream
|
||||||
skills (`web-search`, `postgres`, …) that 2dph integrates
|
skills (`web-search`, `db-yaml`, …) that 2dph integrates
|
||||||
- detective method — the two-source method
|
- detective method — the two-source method
|
||||||
|
|
||||||
Work board (issues): [epic #16](https://git.produktor.io/eSlider/2dph/issues/16)
|
Work board (issues): [git.produktor.io/eSlider/2dph/issues](https://git.produktor.io/eSlider/2dph/issues).
|
||||||
on [git.produktor.io/eSlider/2dph/issues](https://git.produktor.io/eSlider/2dph/issues).
|
|
||||||
PRs and CI: GitHub [`eSlider/2dph`](https://github.com/eSlider/2dph).
|
PRs and CI: GitHub [`eSlider/2dph`](https://github.com/eSlider/2dph).
|
||||||
|
|
||||||
See [PLAN.md](PLAN.md) for decisions, [docs/roadmap.md](docs/roadmap.md) for
|
See [PLAN.md](PLAN.md) for decisions, execution status, and v2 open questions.
|
||||||
the gap to v1, and v2 open questions.
|
|
||||||
|
|||||||
@@ -1,21 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=brain_add "$0" "$@"; exit
|
|
||||||
//go:build brain_add
|
|
||||||
//
|
|
||||||
// bin/brain/add.go - incremental leaf write (Python kblib, no rebuild).
|
|
||||||
//
|
|
||||||
// ./bin/brain/add.go --text T --root facts --source "a.md x b.md"
|
|
||||||
// ./bin/brain/add.go --json
|
|
||||||
//
|
|
||||||
// D6: write stays Python. Does not delete var/kb.lbug.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/cmdbin"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(cmdbin.ExecFile("bin/kb/add", os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,3 +0,0 @@
|
|||||||
// Commands in this directory are shebang mains (search.go, serve.go, index.go,
|
|
||||||
// get.go, stats.go, eval.go, watch.go), each behind an exclusive build tag.
|
|
||||||
package main
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=system_ladybug,brain_eval "$0" "$@"; exit
|
|
||||||
//go:build cgo && system_ladybug && brain_eval
|
|
||||||
//
|
|
||||||
// bin/brain/eval.go - recall@5 gate.
|
|
||||||
//
|
|
||||||
// ./bin/brain/eval.go
|
|
||||||
// ./bin/brain/eval.go --json
|
|
||||||
//
|
|
||||||
// Needs CGO + libladybug. Python bin/kb/eval is the CI fallback (no cgo).
|
|
||||||
// Control questions live in internal/brain/rank (cgo-free).
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/brain"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(brain.MainEval(os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,23 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=system_ladybug,brain_get "$0" "$@"; exit
|
|
||||||
//go:build cgo && system_ladybug && brain_get
|
|
||||||
//
|
|
||||||
// bin/brain/get.go - read one leaf by id.
|
|
||||||
//
|
|
||||||
// ./bin/brain/get.go <id>
|
|
||||||
// ./bin/brain/get.go <id> --body
|
|
||||||
// ./bin/brain/get.go <id> --json
|
|
||||||
//
|
|
||||||
// Needs CGO + libladybug. Python bin/kb/get is the CI fallback (no cgo).
|
|
||||||
// CGO compiler is Zig (`eval "$(bin/cgo/zig env)"`), not gcc.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/brain"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(brain.MainGet(os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,24 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=brain_index "$0" "$@"; exit
|
|
||||||
//go:build brain_index
|
|
||||||
//
|
|
||||||
// bin/brain/index.go - rebuild the Ladybug graph (Python write path).
|
|
||||||
//
|
|
||||||
// ./bin/brain/index.go --rebuild --with-facts --with-chats
|
|
||||||
// ./bin/brain/index.go --rebuild --with-mail
|
|
||||||
// ./bin/brain/index.go --dry-run --with-mail
|
|
||||||
//
|
|
||||||
// v1 write: bin/brain/add.go for one/few leafs (indexes may already exist).
|
|
||||||
// Bulk mail/corpus still --rebuild (fresh file, indexes last).
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/cmdbin"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
args := append([]string{"--with-mail"}, os.Args[1:]...)
|
|
||||||
os.Exit(cmdbin.ExecFile("bin/kb/index", args))
|
|
||||||
}
|
|
||||||
@@ -1,23 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=system_ladybug "$0" "$@"; exit
|
|
||||||
//go:build cgo && system_ladybug
|
|
||||||
//
|
|
||||||
// bin/brain/search.go - deduction search over the 2dph brain.
|
|
||||||
//
|
|
||||||
// ./bin/brain/search.go "query" [--root facts|info] [--repo P] [-n N] [--hop N] [--json] [--no-web]
|
|
||||||
// ./bin/brain/search.go serve [port]
|
|
||||||
// ./bin/brain/search.go --list-model
|
|
||||||
//
|
|
||||||
// Needs CGO + libladybug via Zig (`eval "$(bin/cgo/zig env)"`), not gcc.
|
|
||||||
// bin/kb/search which sets those and builds a binary for the embed daemon.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/brain"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(brain.Main(os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,34 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=brain_serve,system_ladybug "$0" "$@"; exit
|
|
||||||
//go:build brain_serve && cgo && system_ladybug
|
|
||||||
//
|
|
||||||
// bin/brain/serve.go - HTTP API (in-process ladybug search).
|
|
||||||
//
|
|
||||||
// KB_ROOT=/path/to/2dph ./bin/brain/serve.go
|
|
||||||
// KB_WORKERS=4 KB_PORT=8630 ./bin/brain/serve.go
|
|
||||||
//
|
|
||||||
// GET /openapi.json same Ops table as the handlers
|
|
||||||
// POST /mcp JSON-RPC tools/list + tools/call
|
|
||||||
//
|
|
||||||
// Needs CGO + libladybug (same as bin/brain/search.go).
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"log"
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/brain"
|
|
||||||
"github.com/eSlider/2dph/internal/httpapi"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
if os.Getenv("KB_ROOT") == "" {
|
|
||||||
if wd, err := os.Getwd(); err == nil {
|
|
||||||
os.Setenv("KB_ROOT", wd)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if err := brain.Ready(); err != nil {
|
|
||||||
log.Fatal(err)
|
|
||||||
}
|
|
||||||
httpapi.Run(brain.HTTP{})
|
|
||||||
}
|
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
//go:build brain_serve && !system_ladybug
|
|
||||||
//
|
|
||||||
// Fallback serve when ladybug cgo is not in the build (CI / tags=brain_serve).
|
|
||||||
// Production shebang is serve.go (in-process).
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/httpapi"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
if os.Getenv("KB_ROOT") == "" {
|
|
||||||
if wd, err := os.Getwd(); err == nil {
|
|
||||||
os.Setenv("KB_ROOT", wd)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
httpapi.Run(nil)
|
|
||||||
}
|
|
||||||
@@ -1,21 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=system_ladybug,brain_stats "$0" "$@"; exit
|
|
||||||
//go:build cgo && system_ladybug && brain_stats
|
|
||||||
//
|
|
||||||
// bin/brain/stats.go - index health.
|
|
||||||
//
|
|
||||||
// ./bin/brain/stats.go
|
|
||||||
// ./bin/brain/stats.go --json
|
|
||||||
//
|
|
||||||
// Needs CGO + libladybug. Python bin/kb/stats is the CI fallback (no cgo).
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/brain"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(brain.MainStats(os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=brain_watch "$0" "$@"; exit
|
|
||||||
//go:build brain_watch
|
|
||||||
//
|
|
||||||
// bin/brain/watch.go - re-index when corpus files change.
|
|
||||||
//
|
|
||||||
// ./bin/brain/watch.go [dir...]
|
|
||||||
// KB_WATCH_INTERVAL=15 ./bin/brain/watch.go
|
|
||||||
//
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/bin/watch"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
watch.Run(os.Args[1:])
|
|
||||||
}
|
|
||||||
@@ -1,23 +0,0 @@
|
|||||||
#!/bin/sh
|
|
||||||
# bin/cgo/zc++ — CGO CXX. Zig, not g++.
|
|
||||||
set -eu
|
|
||||||
ROOT="$(CDPATH= cd -- "$(dirname "$0")/../.." && pwd)"
|
|
||||||
case "$(uname -m)" in
|
|
||||||
x86_64|amd64) TARGET=x86_64-linux-gnu ;;
|
|
||||||
aarch64|arm64) TARGET=aarch64-linux-gnu ;;
|
|
||||||
*)
|
|
||||||
echo "zc++: unsupported arch $(uname -m)" >&2
|
|
||||||
exit 2
|
|
||||||
;;
|
|
||||||
esac
|
|
||||||
if [ -n "${ZIG:-}" ] && [ -x "$ZIG" ]; then
|
|
||||||
:
|
|
||||||
elif [ -x "$ROOT/var/zig/zig" ]; then
|
|
||||||
ZIG="$ROOT/var/zig/zig"
|
|
||||||
elif command -v zig >/dev/null 2>&1; then
|
|
||||||
ZIG="$(command -v zig)"
|
|
||||||
else
|
|
||||||
echo "zc++: zig missing; run bin/cgo/zig first" >&2
|
|
||||||
exit 127
|
|
||||||
fi
|
|
||||||
exec "$ZIG" c++ -target "$TARGET" "$@"
|
|
||||||
-24
@@ -1,24 +0,0 @@
|
|||||||
#!/bin/sh
|
|
||||||
# bin/cgo/zcc — CGO CC. Zig, not gcc.
|
|
||||||
# Go invokes CC with many args; a wrapper avoids spaces in $CC.
|
|
||||||
set -eu
|
|
||||||
ROOT="$(CDPATH= cd -- "$(dirname "$0")/../.." && pwd)"
|
|
||||||
case "$(uname -m)" in
|
|
||||||
x86_64|amd64) TARGET=x86_64-linux-gnu ;;
|
|
||||||
aarch64|arm64) TARGET=aarch64-linux-gnu ;;
|
|
||||||
*)
|
|
||||||
echo "zcc: unsupported arch $(uname -m)" >&2
|
|
||||||
exit 2
|
|
||||||
;;
|
|
||||||
esac
|
|
||||||
if [ -n "${ZIG:-}" ] && [ -x "$ZIG" ]; then
|
|
||||||
:
|
|
||||||
elif [ -x "$ROOT/var/zig/zig" ]; then
|
|
||||||
ZIG="$ROOT/var/zig/zig"
|
|
||||||
elif command -v zig >/dev/null 2>&1; then
|
|
||||||
ZIG="$(command -v zig)"
|
|
||||||
else
|
|
||||||
echo "zcc: zig missing; run bin/cgo/zig first" >&2
|
|
||||||
exit 127
|
|
||||||
fi
|
|
||||||
exec "$ZIG" cc -target "$TARGET" "$@"
|
|
||||||
-130
@@ -1,130 +0,0 @@
|
|||||||
#!/usr/bin/env bash
|
|
||||||
# bin/cgo/zig — CGO toolchain: zig cc (not gcc) + pinned liblbug + libtokenizers.
|
|
||||||
#
|
|
||||||
# eval "$(bin/cgo/zig env)" # export CC/CXX/CGO_*
|
|
||||||
# bin/cgo/zig go build ... # ensure, then exec with env
|
|
||||||
# bin/cgo/zig ./bin/brain/search.go "query"
|
|
||||||
#
|
|
||||||
# Pins live in this file. Downloads land in var/ (gitignored).
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
ROOT="$(CDPATH= cd -- "$(dirname "$0")/../.." && pwd)"
|
|
||||||
ZIG_VERSION=0.14.1
|
|
||||||
LBUG_VERSION=0.19.1
|
|
||||||
TOKENIZERS_VERSION=1.27.0
|
|
||||||
|
|
||||||
arch="$(uname -m)"
|
|
||||||
case "$arch" in
|
|
||||||
x86_64|amd64)
|
|
||||||
ZIG_ARCH=x86_64
|
|
||||||
LBUG_ARCH=x86_64
|
|
||||||
TOK_ARCH=x86_64
|
|
||||||
ZIG_SHA=24aeeec8af16c381934a6cd7d95c807a8cb2cf7df9fa40d359aa884195c4716c
|
|
||||||
LBUG_SHA=ed263ae913f68cb0ddba0b98548b58edaac49929766d03bdaaa83be46c68847d
|
|
||||||
TOK_SHA=72556cdca798dd4ea7cdaba308e5f0d68a8cb93b67c96edf485b7a0edd7b07f4
|
|
||||||
;;
|
|
||||||
aarch64|arm64)
|
|
||||||
ZIG_ARCH=aarch64
|
|
||||||
LBUG_ARCH=aarch64
|
|
||||||
TOK_ARCH=aarch64
|
|
||||||
ZIG_SHA=f7a654acc967864f7a050ddacfaa778c7504a0eca8d2b678839c21eea47c992b
|
|
||||||
LBUG_SHA=b07df2cd533c3976a2a3025866d6420a5f35514d0a822ecc4b2902d55b4725b7
|
|
||||||
TOK_SHA=e96545ad05930c26f51f63d932ee6d3bbd32bbed149e102c5290d587a2293067
|
|
||||||
;;
|
|
||||||
*)
|
|
||||||
echo "bin/cgo/zig: unsupported arch $arch" >&2
|
|
||||||
exit 2
|
|
||||||
;;
|
|
||||||
esac
|
|
||||||
|
|
||||||
CACHE="$ROOT/var/cache"
|
|
||||||
LIB="$ROOT/lib-ladybug"
|
|
||||||
ZIG_DIR="$ROOT/var/zig-dist"
|
|
||||||
ZIG_BIN="$ROOT/var/zig/zig"
|
|
||||||
|
|
||||||
sha256of() {
|
|
||||||
if command -v sha256sum >/dev/null 2>&1; then
|
|
||||||
sha256sum "$1" | awk '{print $1}'
|
|
||||||
else
|
|
||||||
shasum -a 256 "$1" | awk '{print $1}'
|
|
||||||
fi
|
|
||||||
}
|
|
||||||
|
|
||||||
fetch() {
|
|
||||||
local url="$1" dest="$2" expect="$3"
|
|
||||||
if [ -f "$dest" ] && [ "$(sha256of "$dest")" = "$expect" ]; then
|
|
||||||
return 0
|
|
||||||
fi
|
|
||||||
mkdir -p "$(dirname "$dest")"
|
|
||||||
echo "fetch $url" >&2
|
|
||||||
curl -fsSL "$url" -o "$dest"
|
|
||||||
local got
|
|
||||||
got="$(sha256of "$dest")"
|
|
||||||
if [ "$got" != "$expect" ]; then
|
|
||||||
echo "checksum mismatch $dest: got $got want $expect" >&2
|
|
||||||
rm -f "$dest"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_zig() {
|
|
||||||
if [ -n "${ZIG:-}" ] && [ -x "$ZIG" ]; then
|
|
||||||
return 0
|
|
||||||
fi
|
|
||||||
if [ -x "$ZIG_BIN" ]; then
|
|
||||||
export ZIG="$ZIG_BIN"
|
|
||||||
return 0
|
|
||||||
fi
|
|
||||||
if command -v zig >/dev/null 2>&1; then
|
|
||||||
export ZIG
|
|
||||||
ZIG="$(command -v zig)"
|
|
||||||
return 0
|
|
||||||
fi
|
|
||||||
local tar="$CACHE/zig-${ZIG_ARCH}-linux-${ZIG_VERSION}.tar.xz"
|
|
||||||
fetch "https://ziglang.org/download/${ZIG_VERSION}/zig-${ZIG_ARCH}-linux-${ZIG_VERSION}.tar.xz" \
|
|
||||||
"$tar" "$ZIG_SHA"
|
|
||||||
mkdir -p "$CACHE"
|
|
||||||
rm -rf "$ZIG_DIR"
|
|
||||||
tar -xJf "$tar" -C "$CACHE"
|
|
||||||
mv "$CACHE/zig-${ZIG_ARCH}-linux-${ZIG_VERSION}" "$ZIG_DIR"
|
|
||||||
mkdir -p "$ROOT/var/zig"
|
|
||||||
ln -sfn "$ZIG_DIR/zig" "$ZIG_BIN"
|
|
||||||
export ZIG="$ZIG_BIN"
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_libs() {
|
|
||||||
mkdir -p "$LIB"
|
|
||||||
if [ ! -f "$LIB/liblbug.so" ]; then
|
|
||||||
local tar="$CACHE/liblbug-linux-${LBUG_ARCH}.tar.gz"
|
|
||||||
fetch "https://github.com/LadybugDB/ladybug/releases/download/v${LBUG_VERSION}/liblbug-linux-${LBUG_ARCH}.tar.gz" \
|
|
||||||
"$tar" "$LBUG_SHA"
|
|
||||||
tar -xzf "$tar" -C "$LIB"
|
|
||||||
fi
|
|
||||||
if [ ! -f "$LIB/libtokenizers.a" ]; then
|
|
||||||
local tar="$CACHE/libtokenizers.linux-${TOK_ARCH}.tar.gz"
|
|
||||||
fetch "https://github.com/daulet/tokenizers/releases/download/v${TOKENIZERS_VERSION}/libtokenizers.linux-${TOK_ARCH}.tar.gz" \
|
|
||||||
"$tar" "$TOK_SHA"
|
|
||||||
tar -xzf "$tar" -C "$LIB"
|
|
||||||
fi
|
|
||||||
}
|
|
||||||
|
|
||||||
print_env() {
|
|
||||||
printf 'export ZIG=%q\n' "$ZIG"
|
|
||||||
printf 'export CC=%q\n' "$ROOT/bin/cgo/zcc"
|
|
||||||
printf 'export CXX=%q\n' "$ROOT/bin/cgo/zc++"
|
|
||||||
printf 'export CGO_ENABLED=1\n'
|
|
||||||
printf 'export CGO_CFLAGS=%q\n' "-I$LIB"
|
|
||||||
printf 'export CGO_LDFLAGS=%q\n' "-L$LIB -Wl,-rpath,${CGO_RPATH:-$LIB}"
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_zig
|
|
||||||
ensure_libs
|
|
||||||
|
|
||||||
cmd="${1:-env}"
|
|
||||||
if [ "$cmd" = "env" ]; then
|
|
||||||
print_env
|
|
||||||
exit 0
|
|
||||||
fi
|
|
||||||
|
|
||||||
eval "$(print_env)"
|
|
||||||
exec "$@"
|
|
||||||
@@ -1,19 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=chats_apply "$0" "$@"; exit
|
|
||||||
//go:build chats_apply
|
|
||||||
//
|
|
||||||
// bin/chats/apply.go - push extracted chat facts to OnlyOffice CRM.
|
|
||||||
//
|
|
||||||
// ./bin/chats/apply.go [--dry-run]
|
|
||||||
//
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/chats"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(chats.RunApply(os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,15 +1,14 @@
|
|||||||
package chats
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
"os"
|
||||||
"os/exec"
|
"os/exec"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
type ooContact struct {
|
type ooContact struct {
|
||||||
@@ -25,10 +24,17 @@ type ooContact struct {
|
|||||||
} `json:"commonData"`
|
} `json:"commonData"`
|
||||||
}
|
}
|
||||||
|
|
||||||
func RunApply(args []string) int {
|
func runApply(args []string) int {
|
||||||
dryRun, err := parseApplyFlags(args)
|
fs := flag.NewFlagSet("chats apply", flag.ContinueOnError)
|
||||||
if err != nil {
|
dryRun := fs.Bool("dry-run", false, "show what would be done without writing")
|
||||||
return cliparse.Fail(err)
|
help := fs.Bool("help", false, "")
|
||||||
|
fs.SetOutput(os.Stderr)
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *help {
|
||||||
|
fmt.Fprintln(os.Stderr, "usage: chats apply [--dry-run]")
|
||||||
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
ooCLI := findOO()
|
ooCLI := findOO()
|
||||||
@@ -122,7 +128,7 @@ func RunApply(args []string) int {
|
|||||||
|
|
||||||
fmt.Printf("\nchats apply: %d actions to apply\n", len(resolved))
|
fmt.Printf("\nchats apply: %d actions to apply\n", len(resolved))
|
||||||
|
|
||||||
if dryRun {
|
if *dryRun {
|
||||||
for _, r := range resolved {
|
for _, r := range resolved {
|
||||||
switch r.Action {
|
switch r.Action {
|
||||||
case "info-add":
|
case "info-add":
|
||||||
@@ -170,7 +176,7 @@ func RunApply(args []string) int {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func loadFacts() ([]ExtractedFact, error) {
|
func loadFacts() ([]ExtractedFact, error) {
|
||||||
factsPath := filepath.Join(Dir(), "facts", "chat-facts.json")
|
factsPath := filepath.Join(chatsDir(), "facts", "chat-facts.json")
|
||||||
data, err := os.ReadFile(factsPath)
|
data, err := os.ReadFile(factsPath)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if os.IsNotExist(err) {
|
if os.IsNotExist(err) {
|
||||||
@@ -3,7 +3,7 @@
|
|||||||
// These are integration tests using real data and real Telegram API (when
|
// These are integration tests using real data and real Telegram API (when
|
||||||
// credentials are available). They follow the TDD workflow pattern:
|
// credentials are available). They follow the TDD workflow pattern:
|
||||||
// sync → import → facts → verify.
|
// sync → import → facts → verify.
|
||||||
package chats
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
@@ -48,7 +48,7 @@ func TestChatsImport(t *testing.T) {
|
|||||||
t.Cleanup(func() { os.Chdir(cwd) })
|
t.Cleanup(func() { os.Chdir(cwd) })
|
||||||
t.Setenv("KB_ROOT", dir)
|
t.Setenv("KB_ROOT", dir)
|
||||||
|
|
||||||
exitCode := RunImport([]string{})
|
exitCode := runImport([]string{})
|
||||||
if exitCode != 0 {
|
if exitCode != 0 {
|
||||||
t.Fatalf("import exit code %d", exitCode)
|
t.Fatalf("import exit code %d", exitCode)
|
||||||
}
|
}
|
||||||
@@ -140,7 +140,7 @@ func TestChatsImportEmpty(t *testing.T) {
|
|||||||
t.Cleanup(func() { os.Chdir(cwd) })
|
t.Cleanup(func() { os.Chdir(cwd) })
|
||||||
t.Setenv("KB_ROOT", dir)
|
t.Setenv("KB_ROOT", dir)
|
||||||
|
|
||||||
exitCode := RunImport([]string{})
|
exitCode := runImport([]string{})
|
||||||
if exitCode == 0 {
|
if exitCode == 0 {
|
||||||
t.Fatal("expected non-zero exit for empty data dir")
|
t.Fatal("expected non-zero exit for empty data dir")
|
||||||
}
|
}
|
||||||
@@ -171,7 +171,7 @@ func TestChatsRoundTrip(t *testing.T) {
|
|||||||
t.Cleanup(func() { os.Chdir(cwd) })
|
t.Cleanup(func() { os.Chdir(cwd) })
|
||||||
t.Setenv("KB_ROOT", dir)
|
t.Setenv("KB_ROOT", dir)
|
||||||
|
|
||||||
if code := RunImport([]string{}); code != 0 {
|
if code := runImport([]string{}); code != 0 {
|
||||||
t.Fatalf("import exit %d", code)
|
t.Fatalf("import exit %d", code)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1,4 +0,0 @@
|
|||||||
// Commands in this directory are shebang mains (sync.go, import.go, facts.go,
|
|
||||||
// apply.go), each behind an exclusive build tag so `go build ./bin/chats`
|
|
||||||
// does not see two mains. Shared code lives in internal/chats.
|
|
||||||
package main
|
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=chats_facts "$0" "$@"; exit
|
|
||||||
//go:build chats_facts
|
|
||||||
//
|
|
||||||
// bin/chats/facts.go - extract phone/email/linkedin facts from JSONL.
|
|
||||||
//
|
|
||||||
// ./bin/chats/facts.go
|
|
||||||
//
|
|
||||||
// Writes var/chats/facts/. Does not index the brain.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/chats"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(chats.RunFacts(os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,15 +1,16 @@
|
|||||||
package chats
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bufio"
|
"bufio"
|
||||||
|
"bytes"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
"os"
|
||||||
|
"os/exec"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"regexp"
|
"regexp"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
var (
|
var (
|
||||||
@@ -76,12 +77,19 @@ type ExtractedFact struct {
|
|||||||
MessageID string `json:"message_id"`
|
MessageID string `json:"message_id"`
|
||||||
}
|
}
|
||||||
|
|
||||||
func RunFacts(args []string) int {
|
func runFacts(args []string) int {
|
||||||
if err := parseNoFlags("chats-facts", args); err != nil {
|
fs := flag.NewFlagSet("chats facts", flag.ContinueOnError)
|
||||||
return cliparse.Fail(err)
|
help := fs.Bool("help", false, "")
|
||||||
|
fs.SetOutput(os.Stderr)
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *help {
|
||||||
|
fmt.Fprintln(os.Stderr, "usage: chats facts")
|
||||||
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
root := Dir()
|
root := chatsDir()
|
||||||
telegramDir := filepath.Join(root, "telegram")
|
telegramDir := filepath.Join(root, "telegram")
|
||||||
|
|
||||||
entries, err := os.ReadDir(telegramDir)
|
entries, err := os.ReadDir(telegramDir)
|
||||||
@@ -141,7 +149,7 @@ func RunFacts(args []string) int {
|
|||||||
}
|
}
|
||||||
fmt.Printf("chats facts: saved to %s\n", factsPath)
|
fmt.Printf("chats facts: saved to %s\n", factsPath)
|
||||||
|
|
||||||
writeFactsMarkdown(allFacts)
|
writeFactsToBrain(root, allFacts)
|
||||||
|
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
@@ -265,10 +273,14 @@ func filterFacts(facts []ExtractedFact, factType string) []ExtractedFact {
|
|||||||
return result
|
return result
|
||||||
}
|
}
|
||||||
|
|
||||||
// writeFactsMarkdown stores a sidecar for humans. Brain ingest is
|
func writeFactsToBrain(root string, facts []ExtractedFact) {
|
||||||
// bin/brain/index.go (not this subject).
|
indexScript := filepath.Join(root, "bin", "kb", "index")
|
||||||
func writeFactsMarkdown(facts []ExtractedFact) {
|
if _, err := os.Stat(indexScript); os.IsNotExist(err) {
|
||||||
mdDir := filepath.Join(Dir(), "facts")
|
fmt.Fprintf(os.Stderr, "chats facts: kb/index not found, skipping brain write\n")
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
mdDir := filepath.Join(chatsDir(), "facts")
|
||||||
if err := os.MkdirAll(mdDir, 0755); err != nil {
|
if err := os.MkdirAll(mdDir, 0755); err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "chats facts: mkdir %s: %v\n", mdDir, err)
|
fmt.Fprintf(os.Stderr, "chats facts: mkdir %s: %v\n", mdDir, err)
|
||||||
return
|
return
|
||||||
@@ -276,7 +288,7 @@ func writeFactsMarkdown(facts []ExtractedFact) {
|
|||||||
|
|
||||||
var sb strings.Builder
|
var sb strings.Builder
|
||||||
sb.WriteString("---\n")
|
sb.WriteString("---\n")
|
||||||
sb.WriteString("root: info\n")
|
sb.WriteString("root: facts\n")
|
||||||
sb.WriteString("---\n\n")
|
sb.WriteString("---\n\n")
|
||||||
sb.WriteString("# Chat-Derived Facts\n\n")
|
sb.WriteString("# Chat-Derived Facts\n\n")
|
||||||
for _, f := range facts {
|
for _, f := range facts {
|
||||||
@@ -290,5 +302,15 @@ func writeFactsMarkdown(facts []ExtractedFact) {
|
|||||||
fmt.Fprintf(os.Stderr, "chats facts: write %s: %v\n", factsMD, err)
|
fmt.Fprintf(os.Stderr, "chats facts: write %s: %v\n", factsMD, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
fmt.Printf("chats facts: markdown %s (index via brain, not chats)\n", factsMD)
|
|
||||||
|
cmd := exec.Command(indexScript, "--corpus", mdDir, "--skip-indexes")
|
||||||
|
var outBuf, errBuf bytes.Buffer
|
||||||
|
cmd.Stdout = &outBuf
|
||||||
|
cmd.Stderr = &errBuf
|
||||||
|
cmd.Dir = root
|
||||||
|
if err := cmd.Run(); err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "chats facts: brain index: %v\n%s", err, errBuf.String())
|
||||||
|
return
|
||||||
|
}
|
||||||
|
fmt.Printf("chats facts: written to brain (%s)\n", strings.TrimSpace(outBuf.String()))
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
module github.com/eSlider/2dph/bin/chats
|
||||||
|
|
||||||
|
go 1.25.0
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=chats_import "$0" "$@"; exit
|
|
||||||
//go:build chats_import
|
|
||||||
//
|
|
||||||
// bin/chats/import.go - JSONL → markdown under var/chats/md/.
|
|
||||||
//
|
|
||||||
// ./bin/chats/import.go
|
|
||||||
//
|
|
||||||
// Conversion only. Brain ingest is bin/brain/index.go, not this command.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/chats"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(chats.RunImport(os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,25 +1,31 @@
|
|||||||
package chats
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bufio"
|
"bufio"
|
||||||
"bytes"
|
"bytes"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
"html"
|
"html"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"sort"
|
"sort"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
func RunImport(args []string) int {
|
func runImport(args []string) int {
|
||||||
if err := parseNoFlags("chats-import", args); err != nil {
|
fs := flag.NewFlagSet("chats import", flag.ContinueOnError)
|
||||||
return cliparse.Fail(err)
|
help := fs.Bool("help", false, "")
|
||||||
|
fs.SetOutput(os.Stderr)
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *help {
|
||||||
|
fmt.Fprintln(os.Stderr, "usage: chats import")
|
||||||
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
root := Dir()
|
root := chatsDir()
|
||||||
mdRoot := filepath.Join(root, "md")
|
mdRoot := filepath.Join(root, "md")
|
||||||
glob := filepath.Join(root, "telegram", "*", "messages.jsonl")
|
glob := filepath.Join(root, "telegram", "*", "messages.jsonl")
|
||||||
|
|
||||||
@@ -0,0 +1,56 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
func runIndex(args []string) int {
|
||||||
|
fs := flag.NewFlagSet("chats index", flag.ContinueOnError)
|
||||||
|
help := fs.Bool("help", false, "")
|
||||||
|
fs.SetOutput(os.Stderr)
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *help {
|
||||||
|
fmt.Fprintln(os.Stderr, "usage: chats index")
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
root := repoRoot()
|
||||||
|
mdDir := filepath.Join(chatsDir(), "md")
|
||||||
|
|
||||||
|
_, err := os.Stat(mdDir)
|
||||||
|
if os.IsNotExist(err) {
|
||||||
|
fmt.Fprintf(os.Stderr, "chats index: no chat markdown at %s; run 'chats import' first\n", mdDir)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
indexScript := filepath.Join(root, "bin", "kb", "index")
|
||||||
|
if _, err := os.Stat(indexScript); os.IsNotExist(err) {
|
||||||
|
fmt.Fprintf(os.Stderr, "chats index: %s not found\n", indexScript)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
cmd := exec.Command(indexScript, "--corpus", mdDir)
|
||||||
|
var outBuf, errBuf bytes.Buffer
|
||||||
|
cmd.Stdout = &outBuf
|
||||||
|
cmd.Stderr = &errBuf
|
||||||
|
cmd.Dir = root
|
||||||
|
|
||||||
|
if err := cmd.Run(); err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "chats index: %v\n%s", err, errBuf.String())
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
result := strings.TrimSpace(outBuf.String())
|
||||||
|
if result == "" {
|
||||||
|
result = strings.TrimSpace(errBuf.String())
|
||||||
|
}
|
||||||
|
fmt.Printf("chats index: %s\n", result)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
package chats
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bufio"
|
"bufio"
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
package chats
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"errors"
|
"errors"
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
// bin/chats - sync, import, index, extract facts, and apply chat data
|
||||||
|
// from Telegram, WhatsApp, LinkedIn into the brain and OnlyOffice CRM.
|
||||||
|
//
|
||||||
|
// Usage:
|
||||||
|
//
|
||||||
|
// chats sync telegram [--limit N] [--since DATE] [--phone PHONE]
|
||||||
|
// chats sync whatsapp [--qr] [--limit N]
|
||||||
|
// chats sync linkedin [--limit N]
|
||||||
|
// chats import # JSONL → MD (all sources)
|
||||||
|
// chats index # rebuild var/kb.lbug with chats
|
||||||
|
// chats facts # extract + cross-check
|
||||||
|
// chats apply [--dry-run] # push to OnlyOffice CRM
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
if len(os.Args) < 2 {
|
||||||
|
usage()
|
||||||
|
os.Exit(2)
|
||||||
|
}
|
||||||
|
cmd := os.Args[1]
|
||||||
|
args := os.Args[2:]
|
||||||
|
switch cmd {
|
||||||
|
case "sync":
|
||||||
|
if len(args) < 1 {
|
||||||
|
usage()
|
||||||
|
os.Exit(2)
|
||||||
|
}
|
||||||
|
platform := args[0]
|
||||||
|
platformArgs := args[1:]
|
||||||
|
switch platform {
|
||||||
|
case "telegram":
|
||||||
|
os.Exit(runSyncTelegram(platformArgs))
|
||||||
|
case "whatsapp":
|
||||||
|
fmt.Fprintf(os.Stderr, "chats: WhatsApp not implemented yet\n")
|
||||||
|
os.Exit(1)
|
||||||
|
case "linkedin":
|
||||||
|
os.Exit(runSyncLinkedIn(platformArgs))
|
||||||
|
default:
|
||||||
|
fmt.Fprintf(os.Stderr, "chats: unknown platform %q\n", platform)
|
||||||
|
os.Exit(2)
|
||||||
|
}
|
||||||
|
case "import":
|
||||||
|
os.Exit(runImport(args))
|
||||||
|
case "index":
|
||||||
|
os.Exit(runIndex(args))
|
||||||
|
case "facts":
|
||||||
|
os.Exit(runFacts(args))
|
||||||
|
case "apply":
|
||||||
|
os.Exit(runApply(args))
|
||||||
|
case "help", "-h", "--help":
|
||||||
|
usage()
|
||||||
|
return
|
||||||
|
default:
|
||||||
|
fmt.Fprintf(os.Stderr, "chats: unknown command %q\n", cmd)
|
||||||
|
usage()
|
||||||
|
os.Exit(2)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func usage() {
|
||||||
|
w := os.Stderr
|
||||||
|
fmt.Fprintln(w, `Usage: chats <command> [args]
|
||||||
|
|
||||||
|
Commands:
|
||||||
|
sync telegram [--limit N] [--since DATE] [--phone PHONE]
|
||||||
|
sync whatsapp [--qr] [--limit N]
|
||||||
|
sync linkedin [--limit N]
|
||||||
|
import JSONL → MD (all sources)
|
||||||
|
index rebuild var/kb.lbug with chats
|
||||||
|
facts extract + cross-check facts
|
||||||
|
apply [--dry-run] push to OnlyOffice CRM
|
||||||
|
|
||||||
|
Output layout:
|
||||||
|
var/chats/<platform>/<chat_id>/messages.jsonl
|
||||||
|
var/chats/md/<platform>/<chat_name>/messages.md`)
|
||||||
|
}
|
||||||
|
|
||||||
|
// repoRoot locates the 2dph project root by walking up from the binary.
|
||||||
|
func repoRoot() string {
|
||||||
|
if v := os.Getenv("KB_ROOT"); v != "" {
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
wd, err := os.Getwd()
|
||||||
|
if err != nil {
|
||||||
|
return "."
|
||||||
|
}
|
||||||
|
for i := 0; i < 10; i++ {
|
||||||
|
if _, err := os.Stat(wd + "/var"); err == nil {
|
||||||
|
return wd
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(wd + "/.git"); err == nil {
|
||||||
|
return wd
|
||||||
|
}
|
||||||
|
parent := wd
|
||||||
|
if idx := strings.LastIndex(wd, "/"); idx >= 0 {
|
||||||
|
parent = wd[:idx]
|
||||||
|
}
|
||||||
|
if parent == wd {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
wd = parent
|
||||||
|
}
|
||||||
|
return "."
|
||||||
|
}
|
||||||
|
|
||||||
|
// chatsDir returns var/chats under the repo root.
|
||||||
|
func chatsDir() string {
|
||||||
|
return repoRoot() + "/var/chats"
|
||||||
|
}
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
package chats
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bufio"
|
"bufio"
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
package chats
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
@@ -1,42 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=chats_sync "$0" "$@"; exit
|
|
||||||
//go:build chats_sync
|
|
||||||
//
|
|
||||||
// bin/chats/sync.go - download chat messages to var/chats/<platform>/.
|
|
||||||
//
|
|
||||||
// ./bin/chats/sync.go telegram [--limit N] [--phone PHONE]
|
|
||||||
// ./bin/chats/sync.go linkedin [--limit N] [--refresh]
|
|
||||||
//
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"fmt"
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/chats"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
if len(os.Args) < 2 {
|
|
||||||
fmt.Fprintln(os.Stderr, `usage: bin/chats/sync.go telegram|linkedin [flags]`)
|
|
||||||
os.Exit(2)
|
|
||||||
}
|
|
||||||
platform := os.Args[1]
|
|
||||||
args := os.Args[2:]
|
|
||||||
switch platform {
|
|
||||||
case "telegram":
|
|
||||||
os.Exit(chats.RunSyncTelegram(args))
|
|
||||||
case "linkedin":
|
|
||||||
os.Exit(chats.RunSyncLinkedIn(args))
|
|
||||||
case "whatsapp":
|
|
||||||
fmt.Fprintln(os.Stderr, "chats: WhatsApp sync is out of v1")
|
|
||||||
os.Exit(1)
|
|
||||||
case "help", "-h", "--help":
|
|
||||||
fmt.Fprintln(os.Stderr, `usage: bin/chats/sync.go telegram|linkedin [flags]
|
|
||||||
WhatsApp sync is out of v1.`)
|
|
||||||
return
|
|
||||||
default:
|
|
||||||
fmt.Fprintf(os.Stderr, "chats: unknown platform %q\n", platform)
|
|
||||||
os.Exit(2)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,29 +1,34 @@
|
|||||||
package chats
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
func RunSyncTelegram(args []string) int {
|
func runSyncTelegram(args []string) int {
|
||||||
f, err := parseTelegramFlags(args)
|
fs := flag.NewFlagSet("chats sync telegram", flag.ContinueOnError)
|
||||||
if err != nil {
|
limit := fs.Int("limit", 0, "max messages per chat (0 = all)")
|
||||||
return cliparse.Fail(err)
|
phone := fs.String("phone", "", "phone number (default env TELEGRAM_PHONE)")
|
||||||
|
help := fs.Bool("help", false, "")
|
||||||
|
fs.SetOutput(os.Stderr)
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *help {
|
||||||
|
fmt.Fprintln(os.Stderr, "usage: chats sync telegram [--limit N] [--phone PHONE]")
|
||||||
|
return 0
|
||||||
}
|
}
|
||||||
limit := f.Limit
|
|
||||||
phone := f.Phone
|
|
||||||
|
|
||||||
apiIDStr := envVar("TELEGRAM_API_ID", "")
|
apiIDStr := envVar("TELEGRAM_API_ID", "")
|
||||||
apiHash := envVar("TELEGRAM_API_HASH", "")
|
apiHash := envVar("TELEGRAM_API_HASH", "")
|
||||||
sessionStr := envVar("TELEGRAM_SESSION_STRING", "")
|
sessionStr := envVar("TELEGRAM_SESSION_STRING", "")
|
||||||
phoneNum := phone
|
phoneNum := *phone
|
||||||
if phoneNum == "" {
|
if phoneNum == "" {
|
||||||
phoneNum = envVar("TELEGRAM_PHONE", "")
|
phoneNum = envVar("TELEGRAM_PHONE", "")
|
||||||
}
|
}
|
||||||
@@ -73,7 +78,7 @@ func RunSyncTelegram(args []string) int {
|
|||||||
defer cancel()
|
defer cancel()
|
||||||
|
|
||||||
start := time.Now()
|
start := time.Now()
|
||||||
if err := src.Sync(ctx, Dir(), limit); err != nil {
|
if err := src.Sync(ctx, chatsDir(), *limit); err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "chats sync telegram: %v\n", err)
|
fmt.Fprintf(os.Stderr, "chats sync telegram: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
@@ -1,14 +1,13 @@
|
|||||||
package chats
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
"os"
|
||||||
"os/exec"
|
"os/exec"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
func checkLinkedInSession(userDataDir string) (bool, error) {
|
func checkLinkedInSession(userDataDir string) (bool, error) {
|
||||||
@@ -29,13 +28,19 @@ func checkLinkedInSession(userDataDir string) (bool, error) {
|
|||||||
return false, nil
|
return false, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func RunSyncLinkedIn(args []string) int {
|
func runSyncLinkedIn(args []string) int {
|
||||||
f, err := parseLinkedInFlags(args)
|
fs := flag.NewFlagSet("chats sync linkedin", flag.ContinueOnError)
|
||||||
if err != nil {
|
limit := fs.Int("limit", 0, "max messages per conversation (0 = all)")
|
||||||
return cliparse.Fail(err)
|
refresh := fs.Bool("refresh", false, "refresh session from live webtop browser before sync")
|
||||||
|
help := fs.Bool("help", false, "")
|
||||||
|
fs.SetOutput(os.Stderr)
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *help {
|
||||||
|
fmt.Fprintln(os.Stderr, "usage: chats sync linkedin [--limit N] [--refresh]")
|
||||||
|
return 0
|
||||||
}
|
}
|
||||||
limit := f.Limit
|
|
||||||
refresh := f.Refresh
|
|
||||||
|
|
||||||
userDataDir := envVar("LINKEDIN_USER_DATA_DIR", "")
|
userDataDir := envVar("LINKEDIN_USER_DATA_DIR", "")
|
||||||
if userDataDir == "" {
|
if userDataDir == "" {
|
||||||
@@ -43,7 +48,7 @@ func RunSyncLinkedIn(args []string) int {
|
|||||||
userDataDir = home + "/.linkedin-mcp/profile"
|
userDataDir = home + "/.linkedin-mcp/profile"
|
||||||
}
|
}
|
||||||
|
|
||||||
if refresh {
|
if *refresh {
|
||||||
if code := refreshLinkedInSession(userDataDir); code != 0 {
|
if code := refreshLinkedInSession(userDataDir); code != 0 {
|
||||||
return code
|
return code
|
||||||
}
|
}
|
||||||
@@ -67,7 +72,7 @@ func RunSyncLinkedIn(args []string) int {
|
|||||||
defer cancel()
|
defer cancel()
|
||||||
|
|
||||||
start := time.Now()
|
start := time.Now()
|
||||||
if err := src.Sync(ctx, Dir(), limit); err != nil {
|
if err := src.Sync(ctx, chatsDir(), *limit); err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "chats sync linkedin: %v\n", err)
|
fmt.Fprintf(os.Stderr, "chats sync linkedin: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
Vendored
@@ -1,91 +0,0 @@
|
|||||||
//usr/bin/env go run "$0" "$@"; exit
|
|
||||||
//
|
|
||||||
// bin/cli/complete.go - dump flaggy shell completions for all Go shebang tools (D23).
|
|
||||||
//
|
|
||||||
// source <(./bin/cli/complete.go bash)
|
|
||||||
// ./bin/cli/complete.go zsh|fish|powershell|nushell
|
|
||||||
//
|
|
||||||
// Search does not steal the word "completion"; this binary dumps scripts.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"fmt"
|
|
||||||
"os"
|
|
||||||
"strings"
|
|
||||||
|
|
||||||
mailsync "github.com/eSlider/2dph/bin/mail/sync"
|
|
||||||
"github.com/eSlider/2dph/internal/brain/rank"
|
|
||||||
"github.com/eSlider/2dph/internal/chats"
|
|
||||||
"github.com/eSlider/2dph/internal/cli"
|
|
||||||
"github.com/eSlider/2dph/internal/gitlog"
|
|
||||||
"github.com/eSlider/2dph/internal/mdleaves"
|
|
||||||
"github.com/eSlider/2dph/internal/ocr"
|
|
||||||
"github.com/eSlider/2dph/internal/reasoner"
|
|
||||||
"github.com/eSlider/2dph/internal/websearch"
|
|
||||||
"github.com/integrii/flaggy"
|
|
||||||
)
|
|
||||||
|
|
||||||
func tools() []cli.Tool {
|
|
||||||
return []cli.Tool{
|
|
||||||
{Path: "bin/brain/search.go", Name: "brain-search", New: rank.Parser},
|
|
||||||
{Path: "bin/brain/get.go", Name: "brain-get", New: func() *flaggy.Parser {
|
|
||||||
o := rank.GetOptions{}
|
|
||||||
return rank.GetParser(&o)
|
|
||||||
}},
|
|
||||||
{Path: "bin/brain/stats.go", Name: "brain-stats", New: rank.StatsParser},
|
|
||||||
{Path: "bin/brain/eval.go", Name: "brain-eval", New: rank.EvalParser},
|
|
||||||
{Path: "bin/web/search.go", Name: "web-search", New: websearch.Parser},
|
|
||||||
{Path: "bin/git/import.go", Name: "git-import", New: gitlog.Parser},
|
|
||||||
{Path: "bin/markdown/import.go", Name: "markdown-import", New: mdleaves.Parser},
|
|
||||||
{Path: "bin/qa/stats.go", Name: "qa-stats", New: cli.QAParser},
|
|
||||||
{Path: "bin/reasoner/bakeoff.go", Name: "reasoner-bakeoff", New: reasoner.Parser},
|
|
||||||
{Path: "bin/mail/ocr.go", Name: "mail-ocr", New: ocr.Parser},
|
|
||||||
{Path: "bin/mail/sync.go", Name: "mail-sync", New: mailsync.Parser},
|
|
||||||
{Path: "bin/chats/sync.go", Name: "chats-sync", New: chats.SyncParser},
|
|
||||||
{Path: "bin/chats/import.go", Name: "chats-import", New: chats.ImportParser},
|
|
||||||
{Path: "bin/chats/facts.go", Name: "chats-facts", New: chats.FactsParser},
|
|
||||||
{Path: "bin/chats/apply.go", Name: "chats-apply", New: chats.ApplyParser},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(run(os.Args[1:]))
|
|
||||||
}
|
|
||||||
|
|
||||||
func run(args []string) int {
|
|
||||||
shell := "bash"
|
|
||||||
if len(args) > 0 {
|
|
||||||
switch args[0] {
|
|
||||||
case "bash", "zsh", "fish", "powershell", "nushell":
|
|
||||||
shell = args[0]
|
|
||||||
case "-h", "--help", "help":
|
|
||||||
fmt.Fprintln(os.Stderr, "usage: bin/cli/complete.go [bash|zsh|fish|powershell|nushell]")
|
|
||||||
return 0
|
|
||||||
default:
|
|
||||||
fmt.Fprintf(os.Stderr, "cli/complete: unknown shell %q\n", args[0])
|
|
||||||
return 2
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if shell == "bash" {
|
|
||||||
fmt.Print(cli.BashScript(tools()))
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
var b strings.Builder
|
|
||||||
for _, t := range tools() {
|
|
||||||
p := t.New()
|
|
||||||
p.Name = t.Name
|
|
||||||
switch shell {
|
|
||||||
case "zsh":
|
|
||||||
b.WriteString(flaggy.GenerateZshCompletion(p))
|
|
||||||
case "fish":
|
|
||||||
b.WriteString(flaggy.GenerateFishCompletion(p))
|
|
||||||
case "powershell":
|
|
||||||
b.WriteString(flaggy.GeneratePowerShellCompletion(p))
|
|
||||||
case "nushell":
|
|
||||||
b.WriteString(flaggy.GenerateNushellCompletion(p))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
fmt.Print(b.String())
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
@@ -1,2 +0,0 @@
|
|||||||
// Deprecated shebang mains at bin root (serve.go is tagged brain_serve).
|
|
||||||
package main
|
|
||||||
+8
-42
@@ -1,11 +1,13 @@
|
|||||||
#!/usr/bin/env bash
|
#!/usr/bin/env bash
|
||||||
# bin/docker-entrypoint - run 2dph tools inside the container.
|
# bin/docker-entrypoint - run 2dph tools inside the container.
|
||||||
#
|
#
|
||||||
# API image (Zig CGO binaries):
|
# brain shell (default)
|
||||||
# serve | search | watch
|
# brain search <q> bin/kb/search
|
||||||
# Index image (Python write path, compose profile `index`):
|
# brain index bin/kb/index
|
||||||
# index | extract | audit | search (deprecated python wrapper)
|
# brain watch <dir> watchdog re-indexer (bin/kb/watch)
|
||||||
# mail-sync [N] ETL loop: sync -> import; optional index (default 300s)
|
# brain serve async Go HTTP server (bin/serve)
|
||||||
|
# brain extract bin/facts/extract (docker×compose pairing)
|
||||||
|
# brain audit bin/facts/audit
|
||||||
#
|
#
|
||||||
# Usage comment starts at line 2 (self-describing convention).
|
# Usage comment starts at line 2 (self-describing convention).
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
@@ -13,49 +15,13 @@ set -euo pipefail
|
|||||||
CMD="${1:-shell}"
|
CMD="${1:-shell}"
|
||||||
shift || true
|
shift || true
|
||||||
|
|
||||||
if [ -x /usr/local/bin/brain-serve ]; then
|
|
||||||
case "$CMD" in
|
|
||||||
shell) exec bash ;;
|
|
||||||
serve) exec /usr/local/bin/brain-serve "$@" ;;
|
|
||||||
search) exec /usr/local/bin/brain-search "$@" ;;
|
|
||||||
watch) exec /usr/local/bin/brain-watch "$@" ;;
|
|
||||||
index)
|
|
||||||
echo "index is the Python sidecar: docker compose --profile index run --rm index" >&2
|
|
||||||
exit 2
|
|
||||||
;;
|
|
||||||
*) echo "unknown command: $CMD (api: serve|search|watch)" >&2; exit 2 ;;
|
|
||||||
esac
|
|
||||||
fi
|
|
||||||
|
|
||||||
case "$CMD" in
|
case "$CMD" in
|
||||||
shell) exec bash ;;
|
shell) exec bash ;;
|
||||||
search) exec "$KB_PY" /app/bin/kb/search "$@" ;;
|
search) exec "$KB_PY" /app/bin/kb/search "$@" ;;
|
||||||
index) exec "$KB_PY" /app/bin/kb/index --with-mail "$@" ;;
|
index) exec "$KB_PY" /app/bin/kb/index "$@" ;;
|
||||||
watch) exec /app/bin/watch "$@" ;;
|
watch) exec /app/bin/watch "$@" ;;
|
||||||
serve) exec /app/bin/serve "$@" ;;
|
serve) exec /app/bin/serve "$@" ;;
|
||||||
extract) exec "$KB_PY" /app/bin/facts/extract "$@" ;;
|
extract) exec "$KB_PY" /app/bin/facts/extract "$@" ;;
|
||||||
audit) exec "$KB_PY" /app/bin/facts/audit "$@" ;;
|
audit) exec "$KB_PY" /app/bin/facts/audit "$@" ;;
|
||||||
mail-sync)
|
|
||||||
# ETL: pull mail, convert to md when new>0. Full --rebuild is opt-in
|
|
||||||
# (MAIL_SYNC_INDEX=1) — ~29k leaf rebuild is minutes, not a 10s loop.
|
|
||||||
interval="${1:-300}"
|
|
||||||
[ "$interval" -gt 0 ] 2>/dev/null || interval=300
|
|
||||||
: "${MAIL_SYNC_ENV:=/secret/mail.env}"
|
|
||||||
: "${MAIL_SYNC_SRC:=onlyoffice,gmail}"
|
|
||||||
: "${MAIL_SYNC_OUT:=/app/var/mail}"
|
|
||||||
: "${MAIL_SYNC_INDEX:=0}"
|
|
||||||
while true; do
|
|
||||||
out="$("/app/bin/mail-sync" --source "$MAIL_SYNC_SRC" --env "$MAIL_SYNC_ENV" --out "$MAIL_SYNC_OUT" 2>&1)" || true
|
|
||||||
echo "$out"
|
|
||||||
new="$(printf '%s\n' "$out" | sed -n 's/.*new=\([0-9]*\).*/\1/p' | tail -1)"
|
|
||||||
if [ -n "$new" ] && [ "$new" -gt 0 ] 2>/dev/null; then
|
|
||||||
"$KB_PY" /app/bin/mail/import --from-raw "$MAIL_SYNC_OUT" 2>&1 | tail -1
|
|
||||||
if [ "$MAIL_SYNC_INDEX" = "1" ]; then
|
|
||||||
"$KB_PY" /app/bin/kb/index --rebuild --with-mail 2>&1 | tail -1
|
|
||||||
fi
|
|
||||||
fi
|
|
||||||
sleep "$interval"
|
|
||||||
done
|
|
||||||
;;
|
|
||||||
*) echo "unknown command: $CMD" >&2; exit 2 ;;
|
*) echo "unknown command: $CMD" >&2; exit 2 ;;
|
||||||
esac
|
esac
|
||||||
|
|||||||
+18
-45
@@ -1,15 +1,14 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""facts/audit - evidence & lexicon checks for the 2dph brain.
|
"""facts/audit - evidence & lexicon checks for the 2dph brain.
|
||||||
|
|
||||||
bin/facts/audit self # lexicon: docs + two-source rule
|
bin/facts/audit self # lexicon: every fact in db has >=2 sources
|
||||||
bin/facts/audit db # evidence gate against var/kb.lbug
|
bin/facts/audit db # evidence gate: run against var/kb.lbug
|
||||||
bin/facts/audit contradict # D16 adjudication (JSON claim(s) on stdin)
|
|
||||||
|
|
||||||
`self` mode checks the repo itself (no network, no runtime deps).
|
`self` mode checks the repo itself (no network, no runtime deps). It greps
|
||||||
`db` mode loads every Leaf with root=facts. Confirmed facts need ` x `;
|
for known-good two-source pairings and confirms the docs are consistent.
|
||||||
hypothesis contradictions need `a x b vs c x d` (both sides ≥2).
|
`db` mode loads every Leaf with root=facts and asserts each has source_rev
|
||||||
`contradict` applies temporal_freshness then authority_pairing; ≥2 vs ≥2
|
and a non-empty `loc` (the "where did you see it" evidence pointer) and that
|
||||||
with no rule stays hypothesis / `(not confirmed)`.
|
'confirmed' facts carry a two-source `source` field.
|
||||||
|
|
||||||
Exit 0 = all checks pass, 1 = audit failures, 2 = could not evaluate.
|
Exit 0 = all checks pass, 1 = audit failures, 2 = could not evaluate.
|
||||||
"""
|
"""
|
||||||
@@ -23,8 +22,6 @@ from pathlib import Path
|
|||||||
ROOT = Path(__file__).resolve().parents[2]
|
ROOT = Path(__file__).resolve().parents[2]
|
||||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
||||||
|
|
||||||
from contradict import adjudicate, check_fact_row # noqa: E402
|
|
||||||
|
|
||||||
|
|
||||||
def audit_db() -> list[str]:
|
def audit_db() -> list[str]:
|
||||||
from kblib import connect
|
from kblib import connect
|
||||||
@@ -36,8 +33,14 @@ def audit_db() -> list[str]:
|
|||||||
r = conn.execute("MATCH (l:Leaf {root:'facts'}) RETURN l.id, l.source, l.loc, l.how, l.confidence")
|
r = conn.execute("MATCH (l:Leaf {root:'facts'}) RETURN l.id, l.source, l.loc, l.how, l.confidence")
|
||||||
problems: list[str] = []
|
problems: list[str] = []
|
||||||
for lid, source, loc, how, conf in r.get_all():
|
for lid, source, loc, how, conf in r.get_all():
|
||||||
problems.extend(check_fact_row(str(lid), str(source or ""), str(loc or ""),
|
if conf != "confirmed":
|
||||||
str(how or ""), str(conf or "")))
|
problems.append(f"{lid}: facts require confidence='confirmed', got '{conf}'")
|
||||||
|
if not source or " x " not in source:
|
||||||
|
problems.append(f"{lid}: needs 2-source evidence in source, got '{source}'")
|
||||||
|
if not loc:
|
||||||
|
problems.append(f"{lid}: missing loc (evidence pointer)")
|
||||||
|
if not how:
|
||||||
|
problems.append(f"{lid}: missing how")
|
||||||
conn.close()
|
conn.close()
|
||||||
db.close()
|
db.close()
|
||||||
return problems
|
return problems
|
||||||
@@ -51,50 +54,20 @@ def audit_self() -> list[str]:
|
|||||||
problems.append("PLAN.md missing recall@5 gate")
|
problems.append("PLAN.md missing recall@5 gate")
|
||||||
if re.search(r"(?i)facts must have.*2 sources|2.source", plan) is None:
|
if re.search(r"(?i)facts must have.*2 sources|2.source", plan) is None:
|
||||||
problems.append("PLAN.md missing the two-source evidence rule for facts")
|
problems.append("PLAN.md missing the two-source evidence rule for facts")
|
||||||
if "temporal_freshness" not in plan or "authority_pairing" not in plan:
|
|
||||||
problems.append("PLAN.md missing D16 adjudication rules")
|
|
||||||
if re.search(r"(?i)HNSW|BM25|deduction", (ROOT / "README.md").read_text()) is None:
|
if re.search(r"(?i)HNSW|BM25|deduction", (ROOT / "README.md").read_text()) is None:
|
||||||
problems.append("README.md missing search/retrieval description")
|
problems.append("README.md missing search/retrieval description")
|
||||||
return problems
|
return problems
|
||||||
|
|
||||||
|
|
||||||
def audit_contradict(raw: str) -> tuple[list[str], list[dict]]:
|
|
||||||
raw = raw.strip()
|
|
||||||
if not raw:
|
|
||||||
return ["contradict: empty stdin (JSON claim or {claims:[...]})"], []
|
|
||||||
try:
|
|
||||||
payload = json.loads(raw)
|
|
||||||
except json.JSONDecodeError as e:
|
|
||||||
return [f"contradict: invalid JSON: {e}"], []
|
|
||||||
if isinstance(payload, dict) and "claims" in payload:
|
|
||||||
claims = list(payload.get("claims") or [])
|
|
||||||
elif isinstance(payload, dict):
|
|
||||||
claims = [payload]
|
|
||||||
elif isinstance(payload, list):
|
|
||||||
claims = payload
|
|
||||||
else:
|
|
||||||
return ["contradict: expected object or list"], []
|
|
||||||
details = [adjudicate(c) for c in claims]
|
|
||||||
return [], details
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str]) -> int:
|
def main(argv: list[str]) -> int:
|
||||||
import argparse
|
import argparse
|
||||||
p = argparse.ArgumentParser(description="evidence & lexicon audit")
|
p = argparse.ArgumentParser(description="evidence & lexicon audit")
|
||||||
p.add_argument("mode", choices=("self", "db", "contradict"))
|
p.add_argument("mode", choices=("self", "db"))
|
||||||
p.add_argument("--json", action="store_true")
|
p.add_argument("--json", action="store_true")
|
||||||
a = p.parse_args(argv)
|
a = p.parse_args(argv)
|
||||||
|
|
||||||
details: list[dict] = []
|
problems = audit_self() if a.mode == "self" else audit_db()
|
||||||
if a.mode == "self":
|
out = {"mode": a.mode, "ok": not problems, "problems": problems}
|
||||||
problems = audit_self()
|
|
||||||
elif a.mode == "db":
|
|
||||||
problems = audit_db()
|
|
||||||
else:
|
|
||||||
problems, details = audit_contradict(sys.stdin.read())
|
|
||||||
out: dict = {"mode": a.mode, "ok": not problems, "problems": problems}
|
|
||||||
if details:
|
|
||||||
out["contradictions"] = details
|
|
||||||
if a.json:
|
if a.json:
|
||||||
print(json.dumps(out, indent=2))
|
print(json.dumps(out, indent=2))
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -1,22 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=facts_audit "$0" "$@"; exit
|
|
||||||
//go:build facts_audit
|
|
||||||
//
|
|
||||||
// bin/facts/audit.go - 2-source + lexicon checks.
|
|
||||||
//
|
|
||||||
// ./bin/facts/audit.go self
|
|
||||||
// ./bin/facts/audit.go db
|
|
||||||
// ./bin/facts/audit.go contradict --json < claim.json
|
|
||||||
//
|
|
||||||
// Python bin/facts/audit is the implementation (CI runs it directly).
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/cmdbin"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(cmdbin.ExecFile("bin/facts/audit", os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=facts_crm "$0" "$@"; exit
|
|
||||||
//go:build facts_crm
|
|
||||||
//
|
|
||||||
// bin/facts/crm.go - prove person↔company / company↔project (ooCRM × corpus).
|
|
||||||
//
|
|
||||||
// ./bin/facts/crm.go [--dry-run] [--mismatches]
|
|
||||||
//
|
|
||||||
// Python bin/facts/crm is the implementation. Graph write stays Python.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/cmdbin"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(cmdbin.ExecFile("bin/facts/crm", os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=facts_extract "$0" "$@"; exit
|
|
||||||
//go:build facts_extract
|
|
||||||
//
|
|
||||||
// bin/facts/extract.go - acquire confirmed facts (2-source each).
|
|
||||||
//
|
|
||||||
// ./bin/facts/extract.go [--json] [--dry-run]
|
|
||||||
//
|
|
||||||
// Python bin/facts/extract is the implementation. Graph write stays Python.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/cmdbin"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(cmdbin.ExecFile("bin/facts/extract", os.Args[1:]))
|
|
||||||
}
|
|
||||||
+138
-10
@@ -1,25 +1,153 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""git/import — deprecated. Use bin/git/import.go (go-git, no git binary).
|
"""git/import - import git history (commits, authors, files) into the brain.
|
||||||
|
|
||||||
bin/git/import.go [REPO] [--json] [--limit N] [--since DATE]
|
bin/git/import [REPO] import all commits -> leafs + graph
|
||||||
|
bin/git/import --json emit import leafs as JSON, no write
|
||||||
|
bin/git/import --limit 100 cap commits processed
|
||||||
|
bin/git/import --since 2026-01-01 only recent commits
|
||||||
|
bin/git/import --root DIR run per repo dir under DIR
|
||||||
|
bin/git/import --no-env never read .env anywhere (default: true)
|
||||||
|
|
||||||
|
Reads `git log --no-merges --name-only` from the repo, maps commits to
|
||||||
|
`info` leafs (root=info, type=commit) and writes the version graph
|
||||||
|
`File -[:HAS_VERSION]-> Commit -[:AUTHORED]-> Person` into var/kb.lbug.
|
||||||
|
Idempotent: leaf MERGE by (source,text via leaf_id), graph MERGE by sha.
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import os
|
import json
|
||||||
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[2]
|
ROOT = Path(__file__).resolve().parents[2]
|
||||||
|
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
||||||
|
|
||||||
|
from kblib import ( # noqa: E402
|
||||||
|
connect, ensure_indexes, init_schema, upsert_leaf,
|
||||||
|
)
|
||||||
|
from gitimport import commits_to_leafs, ensure_git_schema, index_commits, parse_log # noqa: E402
|
||||||
|
|
||||||
|
LOG_FMT = "--format=%x1e%H%x1f%an%x1f%ae%x1f%aI%x1f%s"
|
||||||
|
|
||||||
|
|
||||||
|
def git_log(repo: Path, limit: int = 0, since: str = "") -> str:
|
||||||
|
cmd = ["git", "-C", str(repo), "log", "--no-merges", "--name-only", LOG_FMT]
|
||||||
|
if since:
|
||||||
|
cmd += ["--since", since]
|
||||||
|
if limit:
|
||||||
|
cmd += ["-n", str(limit)]
|
||||||
|
try:
|
||||||
|
out = subprocess.run(cmd, capture_output=True, text=True, timeout=120)
|
||||||
|
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||||
|
return ""
|
||||||
|
if out.returncode != 0:
|
||||||
|
print(f"git/import: {repo}: {out.stderr.strip()}", file=sys.stderr)
|
||||||
|
return ""
|
||||||
|
return out.stdout
|
||||||
|
|
||||||
|
|
||||||
|
def repo_name(repo: Path) -> str:
|
||||||
|
try:
|
||||||
|
out = subprocess.run(
|
||||||
|
["git", "-C", str(repo), "remote", "get-url", "origin"],
|
||||||
|
capture_output=True, text=True, timeout=20)
|
||||||
|
url = out.stdout.strip()
|
||||||
|
return url.rstrip("/").split("/")[-1].removesuffix(".git") if url else repo.name
|
||||||
|
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||||
|
return repo.name
|
||||||
|
|
||||||
|
|
||||||
|
def embedder():
|
||||||
|
from model2vec import StaticModel
|
||||||
|
model = StaticModel.from_pretrained("minishlab/potion-multilingual-128M")
|
||||||
|
return lambda text: model.encode([text])[0].astype(float).tolist()
|
||||||
|
|
||||||
|
|
||||||
|
def import_repo(conn, repo: Path, embed, limit: int, since: str,
|
||||||
|
no_write: bool = False) -> tuple[int, int]:
|
||||||
|
raw = git_log(repo, limit, since)
|
||||||
|
commits = parse_log(raw)
|
||||||
|
leafs = commits_to_leafs(commits, repo_name(repo))
|
||||||
|
if no_write:
|
||||||
|
return len(commits), 0
|
||||||
|
written = 0
|
||||||
|
for lf in leafs:
|
||||||
|
query = f"{lf['heading']}\n\n{lf['text']}"
|
||||||
|
emb = embed(lf["text"]) if lf["text"] else None
|
||||||
|
upsert_leaf(conn, text=query, root="info", confidence="confirmed",
|
||||||
|
source=lf["source"], source_rev="git", how="git/import",
|
||||||
|
loc=lf["source"], type_=lf.get("type", "commit"),
|
||||||
|
embedding=emb)
|
||||||
|
written += 1
|
||||||
|
index_commits(conn, commits, repo_name(repo))
|
||||||
|
return len(commits), written
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str]) -> int:
|
def main(argv: list[str]) -> int:
|
||||||
print(
|
import argparse
|
||||||
"bin/git/import is deprecated; use bin/git/import.go (go-git)",
|
p = argparse.ArgumentParser(description="import git history into the brain")
|
||||||
file=sys.stderr,
|
p.add_argument("repo", nargs="?", default=None)
|
||||||
)
|
p.add_argument("--root", default=None, help="directory of repos to import (each git dir separately)")
|
||||||
target = ROOT / "bin" / "git" / "import.go"
|
p.add_argument("--limit", type=int, default=0)
|
||||||
os.execvp("go", ["go", "run", str(target), *argv])
|
p.add_argument("--since", default="")
|
||||||
return 1
|
p.add_argument("--json", action="store_true")
|
||||||
|
p.add_argument("--dry-run", action="store_true", help="parse + report, no db write")
|
||||||
|
a = p.parse_args(argv)
|
||||||
|
|
||||||
|
repos: list[Path] = []
|
||||||
|
if a.repo:
|
||||||
|
repos = [Path(a.repo)]
|
||||||
|
elif a.root:
|
||||||
|
root = Path(a.root)
|
||||||
|
if root.is_file():
|
||||||
|
repos = [root]
|
||||||
|
else:
|
||||||
|
repos = [dp for dp in sorted(root.iterdir()) if (dp / ".git").exists() or dp.is_file()]
|
||||||
|
else:
|
||||||
|
repos = [ROOT]
|
||||||
|
|
||||||
|
total_commits = 0
|
||||||
|
results: list[dict] = []
|
||||||
|
if a.dry_run:
|
||||||
|
for repo in repos:
|
||||||
|
if not repo.exists():
|
||||||
|
continue
|
||||||
|
commits = parse_log(git_log(repo, a.limit, a.since))
|
||||||
|
name = repo_name(repo)
|
||||||
|
total_commits += len(commits)
|
||||||
|
results.append({"repo": name, "commits": len(commits),
|
||||||
|
"leafs": len(commits_to_leafs(commits, name)), "path": str(repo)})
|
||||||
|
if a.json:
|
||||||
|
print(json.dumps(results, indent=2))
|
||||||
|
else:
|
||||||
|
for r in results:
|
||||||
|
print(f"{r['repo']:<24} {r['commits']:>5} commits -> {r['leafs']} leafs {r['path']}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# Never DROP FTS/VECTOR (ghost catalog). Upsert while indexes exist is OK;
|
||||||
|
# ensure_indexes only CREATEs when missing.
|
||||||
|
db, conn = connect(ROOT / "var" / "kb.lbug", read_only=False)
|
||||||
|
init_schema(conn)
|
||||||
|
embed = embedder()
|
||||||
|
rows: list[dict] = []
|
||||||
|
for repo in repos:
|
||||||
|
if not repo.exists():
|
||||||
|
continue
|
||||||
|
reached, written = import_repo(conn, repo, embed, a.limit, a.since)
|
||||||
|
total_commits += reached
|
||||||
|
rows.append({"repo": repo_name(repo), "commits": reached, "written": written})
|
||||||
|
ensure_indexes(conn)
|
||||||
|
conn.close()
|
||||||
|
db.close()
|
||||||
|
|
||||||
|
if a.json:
|
||||||
|
print(json.dumps(rows, indent=2))
|
||||||
|
else:
|
||||||
|
for r in rows:
|
||||||
|
print(f"imported {r['commits']:>5} commits -> {r['written']} leafs {r['repo']}")
|
||||||
|
print(f"total: {total_commits} commits")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
@@ -1,100 +0,0 @@
|
|||||||
//usr/bin/env go run "$0" "$@"; exit
|
|
||||||
//
|
|
||||||
// bin/git/import.go - read git history with go-git (no git binary).
|
|
||||||
//
|
|
||||||
// ./bin/git/import.go [REPO]
|
|
||||||
// ./bin/git/import.go --json
|
|
||||||
// ./bin/git/import.go --limit 100 --since 2026-01-01
|
|
||||||
// ./bin/git/import.go --root DIR
|
|
||||||
//
|
|
||||||
// Conversion only: prints commit leafs. Brain write is bin/brain/index.go.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"encoding/json"
|
|
||||||
"fmt"
|
|
||||||
"os"
|
|
||||||
"path/filepath"
|
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
"github.com/eSlider/2dph/internal/cmdbin"
|
|
||||||
"github.com/eSlider/2dph/internal/gitlog"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(run(os.Args[1:]))
|
|
||||||
}
|
|
||||||
|
|
||||||
func run(args []string) int {
|
|
||||||
c, err := gitlog.ParseArgs(args)
|
|
||||||
if err != nil {
|
|
||||||
return cliparse.Fail(err)
|
|
||||||
}
|
|
||||||
repo, root, since, limit, jsonOut := c.Repo, c.Root, c.Since, c.Limit, c.JSONOut
|
|
||||||
|
|
||||||
sinceT, err := gitlog.ParseSince(since)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "git/import: %v\n", err)
|
|
||||||
return 2
|
|
||||||
}
|
|
||||||
|
|
||||||
repos := []string{}
|
|
||||||
if repo != "" {
|
|
||||||
repos = []string{repo}
|
|
||||||
} else if root != "" {
|
|
||||||
entries, err := os.ReadDir(root)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "git/import: %v\n", err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
for _, e := range entries {
|
|
||||||
p := filepath.Join(root, e.Name())
|
|
||||||
if _, err := os.Stat(filepath.Join(p, ".git")); err == nil {
|
|
||||||
repos = append(repos, p)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
repos = []string{cmdbin.Root()}
|
|
||||||
}
|
|
||||||
|
|
||||||
opt := gitlog.Options{Limit: limit, Since: sinceT}
|
|
||||||
type row struct {
|
|
||||||
Repo string `json:"repo"`
|
|
||||||
Path string `json:"path"`
|
|
||||||
Commits int `json:"commits"`
|
|
||||||
Leafs []gitlog.Leaf `json:"leafs,omitempty"`
|
|
||||||
}
|
|
||||||
var rows []row
|
|
||||||
for _, p := range repos {
|
|
||||||
name, err := gitlog.RepoName(p)
|
|
||||||
if err != nil && name == "" {
|
|
||||||
fmt.Fprintf(os.Stderr, "git/import: %s: %v\n", p, err)
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
cs, err := gitlog.Log(p, opt)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "git/import: %s: %v\n", p, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
leafs := make([]gitlog.Leaf, 0, len(cs))
|
|
||||||
for _, c := range cs {
|
|
||||||
leafs = append(leafs, gitlog.ToLeaf(c, name))
|
|
||||||
}
|
|
||||||
rows = append(rows, row{Repo: name, Path: p, Commits: len(cs), Leafs: leafs})
|
|
||||||
}
|
|
||||||
|
|
||||||
if jsonOut {
|
|
||||||
enc := json.NewEncoder(os.Stdout)
|
|
||||||
enc.SetIndent("", " ")
|
|
||||||
enc.SetEscapeHTML(false)
|
|
||||||
if err := enc.Encode(rows); err != nil {
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
for _, r := range rows {
|
|
||||||
fmt.Printf("%-24s %5d commits %s\n", r.Repo, r.Commits, r.Path)
|
|
||||||
}
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
-120
@@ -1,120 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""kb/add - incremental leaf write (no rebuild).
|
|
||||||
|
|
||||||
bin/kb/add --text T --root facts|info --source S
|
|
||||||
bin/kb/add --json # stdin: one object or {"leafs":[...]}
|
|
||||||
bin/kb/add --db PATH --json
|
|
||||||
|
|
||||||
Writes facts+info in one Ladybug transaction. Does not delete kb.lbug.
|
|
||||||
Embedding is used when provided; otherwise model2vec encodes the text.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[2]
|
|
||||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
|
||||||
|
|
||||||
from kblib import ( # noqa: E402
|
|
||||||
EMBED_DIM,
|
|
||||||
add_leafs,
|
|
||||||
connect,
|
|
||||||
ensure_indexes,
|
|
||||||
init_schema,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def _as_leafs(payload: object) -> list[dict]:
|
|
||||||
if isinstance(payload, list):
|
|
||||||
return [dict(x) for x in payload]
|
|
||||||
if isinstance(payload, dict):
|
|
||||||
if "leafs" in payload:
|
|
||||||
return [dict(x) for x in payload["leafs"]]
|
|
||||||
return [dict(payload)]
|
|
||||||
raise ValueError("json must be an object, a list, or {leafs:[...]}")
|
|
||||||
|
|
||||||
|
|
||||||
def _embed_missing(leafs: list[dict]) -> None:
|
|
||||||
missing = [lf for lf in leafs if not lf.get("embedding")]
|
|
||||||
if not missing:
|
|
||||||
return
|
|
||||||
from model2vec import StaticModel
|
|
||||||
|
|
||||||
model = StaticModel.from_pretrained("minishlab/potion-multilingual-128M")
|
|
||||||
for lf in missing:
|
|
||||||
text = str(lf.get("text") or "")
|
|
||||||
vec = model.encode([text])[0].astype(float).tolist()
|
|
||||||
if len(vec) != EMBED_DIM:
|
|
||||||
vec = (vec + [0.0] * EMBED_DIM)[:EMBED_DIM]
|
|
||||||
lf["embedding"] = vec
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str]) -> int:
|
|
||||||
import argparse
|
|
||||||
|
|
||||||
p = argparse.ArgumentParser(description="add leafs without rebuilding the brain")
|
|
||||||
p.add_argument("--db", default="", help="path to kb.lbug (default var/kb.lbug)")
|
|
||||||
p.add_argument("--json", action="store_true", help="read leaf JSON from stdin")
|
|
||||||
p.add_argument("--text", default="", help="leaf text")
|
|
||||||
p.add_argument("--root", default="info", choices=("facts", "info"))
|
|
||||||
p.add_argument("--source", default="")
|
|
||||||
p.add_argument("--confidence", default="confirmed")
|
|
||||||
p.add_argument("--source-rev", default="working-tree")
|
|
||||||
p.add_argument("--how", default="brain/add")
|
|
||||||
p.add_argument("--loc", default="")
|
|
||||||
p.add_argument("--type", default="reference", dest="type_")
|
|
||||||
p.add_argument("--valid-from", default="", dest="valid_from",
|
|
||||||
help="fact interval start YYYY-MM-DD (D24)")
|
|
||||||
p.add_argument("--valid-to", default="", dest="valid_to",
|
|
||||||
help="fact interval end YYYY-MM-DD inclusive; empty=open (D24)")
|
|
||||||
args = p.parse_args(argv)
|
|
||||||
|
|
||||||
if args.json:
|
|
||||||
raw = sys.stdin.read()
|
|
||||||
if not raw.strip():
|
|
||||||
print("kb/add: empty stdin", file=sys.stderr)
|
|
||||||
return 2
|
|
||||||
leafs = _as_leafs(json.loads(raw))
|
|
||||||
else:
|
|
||||||
if not args.text or not args.source:
|
|
||||||
print("kb/add: --text and --source are required (or --json)", file=sys.stderr)
|
|
||||||
return 2
|
|
||||||
leafs = [{
|
|
||||||
"text": args.text,
|
|
||||||
"root": args.root,
|
|
||||||
"source": args.source,
|
|
||||||
"confidence": args.confidence,
|
|
||||||
"source_rev": args.source_rev,
|
|
||||||
"how": args.how,
|
|
||||||
"loc": args.loc or args.source,
|
|
||||||
"type": args.type_,
|
|
||||||
"valid_from": args.valid_from,
|
|
||||||
"valid_to": args.valid_to,
|
|
||||||
}]
|
|
||||||
|
|
||||||
for lf in leafs:
|
|
||||||
if not lf.get("text") or not lf.get("source"):
|
|
||||||
print("kb/add: each leaf needs text and source", file=sys.stderr)
|
|
||||||
return 2
|
|
||||||
|
|
||||||
_embed_missing(leafs)
|
|
||||||
|
|
||||||
from kblib import DB_PATH, VAR
|
|
||||||
|
|
||||||
dbpath = Path(args.db) if args.db else DB_PATH
|
|
||||||
dbpath.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
VAR.mkdir(exist_ok=True)
|
|
||||||
db, conn = connect(dbpath, read_only=False)
|
|
||||||
init_schema(conn)
|
|
||||||
ids = add_leafs(conn, leafs)
|
|
||||||
ensure_indexes(conn)
|
|
||||||
conn.close()
|
|
||||||
db.close()
|
|
||||||
print(json.dumps({"mode": "add", "ids": ids, "db": str(dbpath)}))
|
|
||||||
return 0
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
sys.exit(main(sys.argv[1:]))
|
|
||||||
+13
-113
@@ -2,15 +2,12 @@
|
|||||||
"""kb/index - build the 2dph brain from markdown + factual leafs.
|
"""kb/index - build the 2dph brain from markdown + factual leafs.
|
||||||
|
|
||||||
bin/kb/index [--corpus DIR] [--rebuild] [--limit N]
|
bin/kb/index [--corpus DIR] [--rebuild] [--limit N]
|
||||||
bin/kb/index --rebuild --with-facts --with-chats
|
|
||||||
bin/kb/index --json # emit stats as JSON
|
bin/kb/index --json # emit stats as JSON
|
||||||
|
|
||||||
Reads every .md under the corpus (default: repo root docs, skills, READMEs)
|
Reads every .md under the corpus (default: repo root docs, skills, READMEs)
|
||||||
as `info` leafs, embeds them with model2vec (potion-multilingual-128M), and
|
as `info` leafs, embeds them with model2vec (potion-multilingual-128M), and
|
||||||
writes them into var/kb.lbug with FTS + HNSW indexes. `facts` leafs come
|
writes them into var/kb.lbug with FTS + HNSW indexes. `facts` leafs come
|
||||||
from bin/facts/extract (docker × compose × ssh-config pairing) when
|
from bin/facts/extract (docker x compose x ssh-config pairing).
|
||||||
`--with-facts` is set. `--with-chats` indexes markdown under var/chats/md
|
|
||||||
(or a given dir) as info. WhatsApp sync stays out of v1.
|
|
||||||
|
|
||||||
--rebuild drops the database file and indexes from scratch. Without it a run
|
--rebuild drops the database file and indexes from scratch. Without it a run
|
||||||
is idempotent (MERGE by (source,text) id).
|
is idempotent (MERGE by (source,text) id).
|
||||||
@@ -25,11 +22,10 @@ ROOT = Path(__file__).resolve().parents[2]
|
|||||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
||||||
|
|
||||||
from kblib import ( # noqa: E402
|
from kblib import ( # noqa: E402
|
||||||
add_leafs, connect, ensure_indexes, init_schema, upsert_leaf, link_from_file,
|
connect, ensure_indexes, init_schema, upsert_leaf,
|
||||||
open_readonly, stats,
|
open_readonly, stats,
|
||||||
)
|
)
|
||||||
from mdleaves import read_markdown, to_all, walk_markdown # noqa: E402
|
from mdleaves import read_markdown, to_all, walk_markdown # noqa: E402
|
||||||
from mailleafs import from_mail_root # noqa: E402
|
|
||||||
|
|
||||||
CORPUS_DEFAULTS = ["README.md", "PLAN.md", "AGENTS.md", "docs", "skills"]
|
CORPUS_DEFAULTS = ["README.md", "PLAN.md", "AGENTS.md", "docs", "skills"]
|
||||||
|
|
||||||
@@ -84,11 +80,10 @@ def index_leafs(conn, leafs: list[dict], embed_fn, limit: int) -> tuple[int, int
|
|||||||
for lf in leafs[:limit] if limit else leafs:
|
for lf in leafs[:limit] if limit else leafs:
|
||||||
query = f"{lf['heading']}\n\n{lf['text']}"
|
query = f"{lf['heading']}\n\n{lf['text']}"
|
||||||
emb = embed_fn(lf["text"]) if lf["text"] else None
|
emb = embed_fn(lf["text"]) if lf["text"] else None
|
||||||
lid = upsert_leaf(conn, text=query, root="info", confidence="confirmed",
|
upsert_leaf(conn, text=query, root="info", confidence="confirmed",
|
||||||
source=lf["source"], source_rev="working-tree",
|
source=lf["source"], source_rev="working-tree",
|
||||||
how="kb/index", loc=lf["source"], type_=lf.get("type", "reference"),
|
how="kb/index", loc=lf["source"], type_=lf.get("type", "reference"),
|
||||||
embedding=emb)
|
embedding=emb)
|
||||||
link_from_file(conn, lid, lf["source"], repo=str(lf.get("repo") or ""))
|
|
||||||
count += 1
|
count += 1
|
||||||
return count, len(leafs)
|
return count, len(leafs)
|
||||||
|
|
||||||
@@ -99,67 +94,11 @@ def embedder():
|
|||||||
return lambda text: model.encode([text])[0].astype(float).tolist()
|
return lambda text: model.encode([text])[0].astype(float).tolist()
|
||||||
|
|
||||||
|
|
||||||
def index_fact_dicts(conn, facts: list[dict], embed_fn) -> int:
|
|
||||||
"""Write extract-shaped dicts as root=facts leafs (2-source source field)."""
|
|
||||||
leafs = []
|
|
||||||
for f in facts:
|
|
||||||
text = str(f.get("text") or "")
|
|
||||||
source = str(f.get("source") or "")
|
|
||||||
if not text or not source:
|
|
||||||
continue
|
|
||||||
leafs.append({
|
|
||||||
"text": text,
|
|
||||||
"root": "facts",
|
|
||||||
"confidence": "confirmed",
|
|
||||||
"source": source,
|
|
||||||
"source_rev": f.get("source_rev") or "working-tree",
|
|
||||||
"how": f.get("how") or "facts/extract",
|
|
||||||
"loc": f.get("loc") or source,
|
|
||||||
"type": "fact",
|
|
||||||
"embedding": embed_fn(text) if text else None,
|
|
||||||
})
|
|
||||||
return len(add_leafs(conn, leafs))
|
|
||||||
|
|
||||||
|
|
||||||
def facts_from_extract() -> list[dict]:
|
|
||||||
import subprocess
|
|
||||||
proc = subprocess.run(
|
|
||||||
[sys.executable, str(ROOT / "bin" / "facts" / "extract"), "--json", "--dry-run"],
|
|
||||||
cwd=ROOT,
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
check=False,
|
|
||||||
)
|
|
||||||
if proc.returncode != 0:
|
|
||||||
print(f"kb/index: facts/extract failed: {proc.stderr}", file=sys.stderr)
|
|
||||||
return []
|
|
||||||
try:
|
|
||||||
payload = json.loads(proc.stdout)
|
|
||||||
except json.JSONDecodeError:
|
|
||||||
print("kb/index: facts/extract produced non-JSON", file=sys.stderr)
|
|
||||||
return []
|
|
||||||
return list(payload.get("facts") or [])
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str]) -> int:
|
def main(argv: list[str]) -> int:
|
||||||
import argparse
|
import argparse
|
||||||
p = argparse.ArgumentParser(description="build the 2dph brain index")
|
p = argparse.ArgumentParser(description="build the 2dph brain index")
|
||||||
p.add_argument("--corpus", action="append", help="extra markdown dir/file to index (may repeat)")
|
p.add_argument("--corpus", action="append", help="extra markdown dir/file to index (may repeat)")
|
||||||
p.add_argument("--rebuild", action="store_true", help="fresh db + indexes")
|
p.add_argument("--rebuild", action="store_true", help="fresh db + indexes")
|
||||||
p.add_argument("--db", default="", help="path to kb.lbug (default var/kb.lbug)")
|
|
||||||
p.add_argument("--no-defaults", action="store_true", help="do not index repo README/docs/skills")
|
|
||||||
p.add_argument("--with-mail", action="store_true", help="include var/mail message.md leafs")
|
|
||||||
p.add_argument("--with-facts", action="store_true", help="run facts/extract into root=facts")
|
|
||||||
p.add_argument("--facts-json", default="", help="JSON list (or {facts:[...]}) of fact dicts")
|
|
||||||
p.add_argument(
|
|
||||||
"--with-chats",
|
|
||||||
nargs="?",
|
|
||||||
const=str(ROOT / "var" / "chats" / "md"),
|
|
||||||
default="",
|
|
||||||
help="index chat markdown as info (default var/chats/md)",
|
|
||||||
)
|
|
||||||
p.add_argument("--since", default="", help="with --with-mail, only messages dated >= YYYY-MM-DD")
|
|
||||||
p.add_argument("--dry-run", action="store_true", help="count leafs, write nothing")
|
|
||||||
p.add_argument(
|
p.add_argument(
|
||||||
"--skip-indexes",
|
"--skip-indexes",
|
||||||
action="store_true",
|
action="store_true",
|
||||||
@@ -170,72 +109,33 @@ def main(argv: list[str]) -> int:
|
|||||||
a = p.parse_args(argv)
|
a = p.parse_args(argv)
|
||||||
|
|
||||||
from kblib import DB_PATH, VAR
|
from kblib import DB_PATH, VAR
|
||||||
|
VAR.mkdir(exist_ok=True)
|
||||||
|
if a.rebuild and DB_PATH.exists():
|
||||||
|
DB_PATH.unlink()
|
||||||
|
|
||||||
dbpath = Path(a.db) if a.db else DB_PATH
|
leafs = load_corpus(ROOT)
|
||||||
leafs: list[dict] = [] if a.no_defaults else load_corpus(ROOT)
|
|
||||||
if a.corpus:
|
if a.corpus:
|
||||||
for source in a.corpus:
|
for source in a.corpus:
|
||||||
leafs.extend(load_corpus_glob(source))
|
leafs.extend(load_corpus_glob(source))
|
||||||
chat_n = 0
|
|
||||||
if a.with_chats:
|
|
||||||
chats = load_corpus_glob(a.with_chats)
|
|
||||||
chat_n = len(chats)
|
|
||||||
leafs.extend(chats)
|
|
||||||
mail_n = 0
|
|
||||||
if a.with_mail:
|
|
||||||
mail = from_mail_root(ROOT / "var" / "mail", since=a.since)
|
|
||||||
mail_n = len(mail)
|
|
||||||
leafs.extend(mail)
|
|
||||||
|
|
||||||
facts: list[dict] = []
|
db, conn = connect(DB_PATH, read_only=False)
|
||||||
if a.facts_json:
|
|
||||||
raw = Path(a.facts_json).read_text(encoding="utf-8")
|
|
||||||
payload = json.loads(raw)
|
|
||||||
facts = list(payload.get("facts") if isinstance(payload, dict) else payload)
|
|
||||||
if a.with_facts:
|
|
||||||
facts.extend(facts_from_extract())
|
|
||||||
|
|
||||||
if a.dry_run:
|
|
||||||
msg = {
|
|
||||||
"indexed": 0,
|
|
||||||
"corpus_total": len(leafs),
|
|
||||||
"mail_leafs": mail_n,
|
|
||||||
"chat_leafs": chat_n,
|
|
||||||
"facts_leafs": len(facts),
|
|
||||||
"dry_run": True,
|
|
||||||
}
|
|
||||||
print(json.dumps(msg, indent=2) if a.json else
|
|
||||||
f"brain/index: {len(leafs)} info + {len(facts)} facts would be indexed")
|
|
||||||
return 0
|
|
||||||
|
|
||||||
VAR.mkdir(exist_ok=True)
|
|
||||||
dbpath.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
if a.rebuild and dbpath.exists():
|
|
||||||
dbpath.unlink()
|
|
||||||
|
|
||||||
db, conn = connect(dbpath, read_only=False)
|
|
||||||
init_schema(conn)
|
init_schema(conn)
|
||||||
|
|
||||||
|
# Never DROP FTS/VECTOR (ghost catalog). Write leafs, then ensure indexes
|
||||||
|
# unless --skip-indexes (seed facts first — MERGE under live FTS corrupts it).
|
||||||
|
# --rebuild already deleted kb.lbug above, so CREATE runs on a clean DB.
|
||||||
embed = embedder()
|
embed = embedder()
|
||||||
done, total = index_leafs(conn, leafs, embed, a.limit)
|
done, total = index_leafs(conn, leafs, embed, a.limit)
|
||||||
fact_n = index_fact_dicts(conn, facts, embed) if facts else 0
|
|
||||||
if not a.skip_indexes:
|
if not a.skip_indexes:
|
||||||
ensure_indexes(conn)
|
ensure_indexes(conn)
|
||||||
s = stats(conn)
|
s = stats(conn)
|
||||||
conn.close()
|
conn.close()
|
||||||
db.close()
|
db.close()
|
||||||
|
|
||||||
result = {
|
result = {"indexed": done, "corpus_total": total, **{k: v for k, v in s.items() if k in ("total", "by_root")}}
|
||||||
"indexed": done,
|
|
||||||
"corpus_total": total,
|
|
||||||
"facts_leafs": fact_n,
|
|
||||||
"chat_leafs": chat_n,
|
|
||||||
**{k: v for k, v in s.items() if k in ("total", "by_root")},
|
|
||||||
}
|
|
||||||
if a.skip_indexes:
|
if a.skip_indexes:
|
||||||
result["indexes"] = "skipped"
|
result["indexes"] = "skipped"
|
||||||
print(json.dumps(result, indent=2) if a.json else
|
print(json.dumps(result, indent=2) if a.json else f"indexed {done}/{total} leafs; db total {s['total']}")
|
||||||
f"indexed {done}/{total} info + {fact_n} facts; db total {s['total']}")
|
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+14
-15
@@ -1,35 +1,34 @@
|
|||||||
#!/usr/bin/env bash
|
#!/usr/bin/env bash
|
||||||
# bin/kb/search — deprecated wrapper. Use bin/brain/search.go.
|
# bin/kb/search - Go deduction search over the brain (model served by daemon).
|
||||||
# CGO via Zig (bin/cgo/zig), not gcc. Builds a binary then execs it.
|
# Builds the kbsearch binary on first run / when source changes, then execs it.
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
ROOT="$(CDPATH= cd -- "$(dirname "$0")/../.." && pwd)"
|
KB="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||||
BIN="$ROOT/var/bin/brain-search"
|
BIN="$KB/var/bin/kbsearch"
|
||||||
SRC="$ROOT/internal/brain"
|
SRC="$KB/bin/kbsearch"
|
||||||
CMD="$ROOT/bin/brain"
|
|
||||||
|
|
||||||
mkdir -p "$ROOT/var/bin"
|
mkdir -p "$KB/var/bin"
|
||||||
|
|
||||||
|
# Rebuild if binary missing or any .go source newer
|
||||||
need_build=0
|
need_build=0
|
||||||
if [ ! -x "$BIN" ]; then
|
if [ ! -x "$BIN" ]; then
|
||||||
need_build=1
|
need_build=1
|
||||||
else
|
else
|
||||||
|
# Check if any .go in kbsearch is newer than binary
|
||||||
while IFS= read -r -d '' f; do
|
while IFS= read -r -d '' f; do
|
||||||
if [ "$f" -nt "$BIN" ]; then
|
if [ "$f" -nt "$BIN" ]; then
|
||||||
need_build=1
|
need_build=1
|
||||||
break
|
break
|
||||||
fi
|
fi
|
||||||
done < <(find "$SRC" "$CMD" -name '*.go' -print0 2>/dev/null)
|
done < <(find "$SRC" -name '*.go' -print0 2>/dev/null)
|
||||||
fi
|
fi
|
||||||
|
|
||||||
if [ "$need_build" -eq 1 ]; then
|
if [ "$need_build" -eq 1 ]; then
|
||||||
echo "Building brain/search (zig cc)..." >&2
|
echo "Building kbsearch..." >&2
|
||||||
(
|
(cd "$SRC" && \
|
||||||
cd "$ROOT" &&
|
CGO_CFLAGS="-I$KB/lib-ladybug" \
|
||||||
eval "$("$ROOT/bin/cgo/zig" env)" &&
|
CGO_LDFLAGS="-L$KB/lib-ladybug -Wl,-rpath,$KB/lib-ladybug" \
|
||||||
go build -tags system_ladybug -o "$BIN" ./bin/brain
|
go build -tags system_ladybug -o "$BIN" .) || exit 1
|
||||||
) || exit 1
|
|
||||||
fi
|
fi
|
||||||
|
|
||||||
echo "bin/kb/search is deprecated; use bin/brain/search.go" >&2
|
|
||||||
exec "$BIN" "$@"
|
exec "$BIN" "$@"
|
||||||
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
//usr/bin/env go run "$0" "$@"; exit
|
//usr/bin/env go run "$0" "$@"; exit
|
||||||
// bin/kb/watch.go — deprecated. Use bin/brain/watch.go.
|
// bin/kb/watch.go - re-index the 2dph brain when corpus files change.
|
||||||
//
|
//
|
||||||
// Usage:
|
// Usage:
|
||||||
//
|
//
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
//go:build cgo && system_ladybug
|
// Brain connection management using go-ladybug.
|
||||||
|
package main
|
||||||
package brain
|
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
@@ -85,18 +84,9 @@ func openWithSandbox(epsv string) error {
|
|||||||
closeBrain()
|
closeBrain()
|
||||||
return fmt.Errorf("LOAD EXTENSION VECTOR: %w", err)
|
return fmt.Errorf("LOAD EXTENSION VECTOR: %w", err)
|
||||||
}
|
}
|
||||||
migrateIntervalColumns()
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// migrateIntervalColumns adds D24 valid_from/valid_to on existing Leaf tables.
|
|
||||||
// Fresh CREATE already has them; ALTER is a no-op when the column exists.
|
|
||||||
func migrateIntervalColumns() {
|
|
||||||
for _, col := range []string{"valid_from", "valid_to"} {
|
|
||||||
_, _ = conn.Query("ALTER TABLE Leaf ADD " + col + " STRING")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func closeBrain() {
|
func closeBrain() {
|
||||||
if conn != nil {
|
if conn != nil {
|
||||||
conn.Close()
|
conn.Close()
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
module github.com/eSlider/2dph/bin/kbsearch
|
||||||
|
|
||||||
|
go 1.26.0
|
||||||
|
|
||||||
|
require (
|
||||||
|
github.com/LadybugDB/go-ladybug v0.17.0
|
||||||
|
github.com/chewxy/math32 v1.11.2
|
||||||
|
github.com/daulet/tokenizers v1.27.0
|
||||||
|
)
|
||||||
|
|
||||||
|
require (
|
||||||
|
github.com/apache/arrow-go/v18 v18.6.0 // indirect
|
||||||
|
github.com/goccy/go-json v0.10.6 // indirect
|
||||||
|
github.com/google/flatbuffers v25.12.19+incompatible // indirect
|
||||||
|
github.com/google/uuid v1.6.0 // indirect
|
||||||
|
github.com/klauspost/compress v1.18.5 // indirect
|
||||||
|
github.com/klauspost/cpuid/v2 v2.3.0 // indirect
|
||||||
|
github.com/pierrec/lz4/v4 v4.1.26 // indirect
|
||||||
|
github.com/shopspring/decimal v1.4.0 // indirect
|
||||||
|
github.com/zeebo/xxh3 v1.1.0 // indirect
|
||||||
|
golang.org/x/exp v0.0.0-20260112195511-716be5621a96 // indirect
|
||||||
|
golang.org/x/sys v0.43.0 // indirect
|
||||||
|
)
|
||||||
@@ -0,0 +1,44 @@
|
|||||||
|
github.com/LadybugDB/go-ladybug v0.17.0 h1:RXDbkBjrbRmLdEbhGl4CLOIEzSt09gbP0n9UbKDEfwI=
|
||||||
|
github.com/LadybugDB/go-ladybug v0.17.0/go.mod h1:GeIXmE8XyF5TFS94NAuTag7vgCC+no/HTBMRA6Rd5Cs=
|
||||||
|
github.com/andybalholm/brotli v1.2.1 h1:R+f5xP285VArJDRgowrfb9DqL18yVK0gKAW/F+eTWro=
|
||||||
|
github.com/andybalholm/brotli v1.2.1/go.mod h1:rzTDkvFWvIrjDXZHkuS16NPggd91W3kUSvPlQ1pLaKY=
|
||||||
|
github.com/apache/arrow-go/v18 v18.6.0 h1:GX/Jyd3R7mCLiECAwY9FWbbaYblie2WXBSz4Sw8fNpM=
|
||||||
|
github.com/apache/arrow-go/v18 v18.6.0/go.mod h1:gm3MiPpY82fLYK5VKPB3WoJbsiLVDfT7flD5/vHReKw=
|
||||||
|
github.com/apache/thrift v0.22.0 h1:r7mTJdj51TMDe6RtcmNdQxgn9XcyfGDOzegMDRg47uc=
|
||||||
|
github.com/apache/thrift v0.22.0/go.mod h1:1e7J/O1Ae6ZQMTYdy9xa3w9k+XHWPfRvdPyJeynQ+/g=
|
||||||
|
github.com/chewxy/math32 v1.11.2 h1:IufN08Zwr1NKuWfY+4Tz55BcwKmyKKNdOP7KtumehnM=
|
||||||
|
github.com/chewxy/math32 v1.11.2/go.mod h1:dOB2rcuFrCn6UHrze36WSLVPKtzPMRAQvBvUwkSsLqs=
|
||||||
|
github.com/daulet/tokenizers v1.27.0 h1:MmFYAEDFz69s/nNQfHg59DWqHz3v94m99kEZ/JbL+s4=
|
||||||
|
github.com/daulet/tokenizers v1.27.0/go.mod h1:YjFY1o1HGMyWkQgbXJDghhvke/yFDp2vGdIO2hYs4MQ=
|
||||||
|
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM=
|
||||||
|
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||||
|
github.com/goccy/go-json v0.10.6 h1:p8HrPJzOakx/mn/bQtjgNjdTcN+/S6FcG2CTtQOrHVU=
|
||||||
|
github.com/goccy/go-json v0.10.6/go.mod h1:oq7eo15ShAhp70Anwd5lgX2pLfOS3QCiwU/PULtXL6M=
|
||||||
|
github.com/google/flatbuffers v25.12.19+incompatible h1:haMV2JRRJCe1998HeW/p0X9UaMTK6SDo0ffLn2+DbLs=
|
||||||
|
github.com/google/flatbuffers v25.12.19+incompatible/go.mod h1:1AeVuKshWv4vARoZatz6mlQ0JxURH0Kv5+zNeJKJCa8=
|
||||||
|
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
|
||||||
|
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
|
||||||
|
github.com/klauspost/compress v1.18.5 h1:/h1gH5Ce+VWNLSWqPzOVn6XBO+vJbCNGvjoaGBFW2IE=
|
||||||
|
github.com/klauspost/compress v1.18.5/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
|
||||||
|
github.com/klauspost/cpuid/v2 v2.3.0 h1:S4CRMLnYUhGeDFDqkGriYKdfoFlDnMtqTiI/sFzhA9Y=
|
||||||
|
github.com/klauspost/cpuid/v2 v2.3.0/go.mod h1:hqwkgyIinND0mEev00jJYCxPNVRVXFQeu1XKlok6oO0=
|
||||||
|
github.com/pierrec/lz4/v4 v4.1.26 h1:GrpZw1gZttORinvzBdXPUXATeqlJjqUG/D87TKMnhjY=
|
||||||
|
github.com/pierrec/lz4/v4 v4.1.26/go.mod h1:EoQMVJgeeEOMsCqCzqFm2O0cJvljX2nGZjcRIPL34O4=
|
||||||
|
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 h1:Jamvg5psRIccs7FGNTlIRMkT8wgtp5eCXdBlqhYGL6U=
|
||||||
|
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||||
|
github.com/shopspring/decimal v1.4.0 h1:bxl37RwXBklmTi0C79JfXCEBD1cqqHt0bbgBAGFp81k=
|
||||||
|
github.com/shopspring/decimal v1.4.0/go.mod h1:gawqmDU56v4yIKSwfBSFip1HdCCXN8/+DMd9qYNcwME=
|
||||||
|
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
|
||||||
|
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
|
||||||
|
github.com/zeebo/assert v1.3.0 h1:g7C04CbJuIDKNPFHmsk4hwZDO5O+kntRxzaUoNXj+IQ=
|
||||||
|
github.com/zeebo/assert v1.3.0/go.mod h1:Pq9JiuJQpG8JLJdtkwrJESF0Foym2/D9XMU5ciN/wJ0=
|
||||||
|
github.com/zeebo/xxh3 v1.1.0 h1:s7DLGDK45Dyfg7++yxI0khrfwq9661w9EN78eP/UZVs=
|
||||||
|
github.com/zeebo/xxh3 v1.1.0/go.mod h1:IisAie1LELR4xhVinxWS5+zf1lA4p0MW4T+w+W07F5s=
|
||||||
|
golang.org/x/exp v0.0.0-20260112195511-716be5621a96 h1:Z/6YuSHTLOHfNFdb8zVZomZr7cqNgTJvA8+Qz75D8gU=
|
||||||
|
golang.org/x/exp v0.0.0-20260112195511-716be5621a96/go.mod h1:nzimsREAkjBCIEFtHiYkrJyT+2uy9YZJB7H1k68CXZU=
|
||||||
|
golang.org/x/sys v0.43.0 h1:Rlag2XtaFTxp19wS8MXlJwTvoh8ArU6ezoyFsMyCTNI=
|
||||||
|
golang.org/x/sys v0.43.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||||
|
gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4=
|
||||||
|
gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E=
|
||||||
|
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
||||||
|
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||||
@@ -0,0 +1,46 @@
|
|||||||
|
// bin/kbsearch - the Go implementation of bin/kb/search (nested module so the
|
||||||
|
// root `go test ./...` and CI never compile it against native ladyships).
|
||||||
|
//
|
||||||
|
// Usage (built/run by ./bin/kb/search):
|
||||||
|
//
|
||||||
|
// kbsearch "query" [--root facts|info] [--repo P] [-n N] [--json]
|
||||||
|
// kbsearch serve [port] start the embedding daemon
|
||||||
|
// kbsearch --list-model print the resolved model dir
|
||||||
|
//
|
||||||
|
// The potion-multilingual model is loaded only in `serve`; a CLI reuses the
|
||||||
|
// daemon over localhost HTTP (KBSEARCH_PORT, default 17830) and starts one in
|
||||||
|
// the background when none answers. KBSEARCH_NO_DAEMON=1 skips that and embeds
|
||||||
|
// in-process instead.
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"log"
|
||||||
|
"os"
|
||||||
|
"strconv"
|
||||||
|
)
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
if len(os.Args) > 1 && os.Args[1] == "serve" {
|
||||||
|
port := 17830
|
||||||
|
if len(os.Args) > 2 {
|
||||||
|
if p, err := strconv.Atoi(os.Args[2]); err == nil {
|
||||||
|
port = p
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := serve(port); err != nil {
|
||||||
|
log.Fatalf("kbsearch serve: %v", err)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if len(os.Args) > 1 && os.Args[1] == "--list-model" {
|
||||||
|
dir, err := modelDir()
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintln(os.Stderr, err)
|
||||||
|
os.Exit(1)
|
||||||
|
}
|
||||||
|
fmt.Println(dir)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
os.Exit(runSearch(os.Args[1:]))
|
||||||
|
}
|
||||||
@@ -1,11 +1,9 @@
|
|||||||
//go:build cgo && system_ladybug
|
|
||||||
|
|
||||||
// StaticModel wraps the potion-multilingual-128m embedding model.
|
// StaticModel wraps the potion-multilingual-128m embedding model.
|
||||||
//
|
//
|
||||||
// Mirrors model2vec.StaticModel: tokenizer (daulet Unigram) + safetensors matrix.
|
// Mirrors model2vec.StaticModel: tokenizer (daulet Unigram) + safetensors matrix.
|
||||||
// Embed(text) applies the same preprocessing: median_token_length pre-truncation,
|
// Embed(text) applies the same preprocessing: median_token_length pre-truncation,
|
||||||
// add_special_tokens=false, drop unk (id=1), truncate to 512, mean pool, L2 normalize +1e-32.
|
// add_special_tokens=false, drop unk (id=1), truncate to 512, mean pool, L2 normalize +1e-32.
|
||||||
package brain
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
@@ -1,4 +1,5 @@
|
|||||||
package brain
|
// modelDir returns the resolved potion-multilingual-128m model directory.
|
||||||
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
@@ -0,0 +1,72 @@
|
|||||||
|
package rank
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
const Usage = `usage: kbsearch "query" [--root facts|info] [--repo REPO] [-n N] [--json]
|
||||||
|
kbsearch serve [port]
|
||||||
|
kbsearch --list-model`
|
||||||
|
|
||||||
|
type Options struct {
|
||||||
|
Query string
|
||||||
|
Root string
|
||||||
|
Repo string
|
||||||
|
Limit int
|
||||||
|
JSONOut bool
|
||||||
|
ListModel bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// ParseArgs reads flags. Unknown flags are an error: silently dropping them
|
||||||
|
// meant `--hop 1` vanished and its argument `1` was appended to the query.
|
||||||
|
// --hop is recognised so it cannot be swallowed; it is not implemented until
|
||||||
|
// File/FROM_FILE edges exist.
|
||||||
|
func ParseArgs(args []string) (Options, error) {
|
||||||
|
opt := Options{Limit: 20}
|
||||||
|
var queryArgs []string
|
||||||
|
|
||||||
|
for i := 0; i < len(args); i++ {
|
||||||
|
arg := args[i]
|
||||||
|
wantsValue := arg == "--root" || arg == "--repo" || arg == "-n" || arg == "--hop"
|
||||||
|
if wantsValue && i+1 >= len(args) {
|
||||||
|
return opt, fmt.Errorf("%s needs a value", arg)
|
||||||
|
}
|
||||||
|
switch arg {
|
||||||
|
case "--root":
|
||||||
|
i++
|
||||||
|
opt.Root = args[i]
|
||||||
|
if opt.Root != "facts" && opt.Root != "info" {
|
||||||
|
return opt, fmt.Errorf("--root must be facts or info, got %q", opt.Root)
|
||||||
|
}
|
||||||
|
case "--repo":
|
||||||
|
i++
|
||||||
|
opt.Repo = args[i]
|
||||||
|
case "-n":
|
||||||
|
i++
|
||||||
|
n, err := strconv.Atoi(args[i])
|
||||||
|
if err != nil || n < 1 {
|
||||||
|
return opt, fmt.Errorf("-n must be a positive integer, got %q", args[i])
|
||||||
|
}
|
||||||
|
opt.Limit = n
|
||||||
|
case "--hop":
|
||||||
|
return opt, fmt.Errorf("--hop is not implemented yet (needs File/FROM_FILE edges)")
|
||||||
|
case "--json":
|
||||||
|
opt.JSONOut = true
|
||||||
|
case "--list-model":
|
||||||
|
opt.ListModel = true
|
||||||
|
default:
|
||||||
|
if strings.HasPrefix(arg, "-") {
|
||||||
|
return opt, fmt.Errorf("unknown flag %q", arg)
|
||||||
|
}
|
||||||
|
queryArgs = append(queryArgs, arg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
opt.Query = strings.TrimSpace(strings.Join(queryArgs, " "))
|
||||||
|
if opt.Query == "" && !opt.ListModel {
|
||||||
|
return opt, fmt.Errorf("no query given")
|
||||||
|
}
|
||||||
|
return opt, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
package rank
|
||||||
|
|
||||||
|
// BM25 ranks best-first, so the top hits are the *highest* scores; cosine
|
||||||
|
// distance ranks best-first ascending. Both mirror kblib.py.
|
||||||
|
const FTSStmt = "CALL QUERY_FTS_INDEX('Leaf', 'id', $q) " +
|
||||||
|
"RETURN node.id, node.text, node.root, node.source, score ORDER BY score DESC LIMIT $n"
|
||||||
|
|
||||||
|
const VecStmt = "CALL QUERY_VECTOR_INDEX('Leaf', 'Leaf_vec', $q, $n) " +
|
||||||
|
"RETURN node.id, node.text, node.root, node.source, distance ORDER BY distance LIMIT $n"
|
||||||
@@ -1,48 +1,30 @@
|
|||||||
// Package rank is the cgo-free ranking and CLI parsing for brain search.
|
// Package rank is the cgo-free ranking and CLI parsing for kbsearch.
|
||||||
// CI can `go test ./rank` without the native ladybug library.
|
// CI can `go test ./rank` without the native ladybug library.
|
||||||
package rank
|
package rank
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"sort"
|
"sort"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/facts"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
type HopNode struct {
|
|
||||||
ID string `json:"id"`
|
|
||||||
Label string `json:"label"`
|
|
||||||
Name string `json:"name"`
|
|
||||||
Depth int `json:"depth"`
|
|
||||||
}
|
|
||||||
|
|
||||||
// Hit is one search result, mirroring the python script's dict shape.
|
// Hit is one search result, mirroring the python script's dict shape.
|
||||||
type Hit struct {
|
type Hit struct {
|
||||||
ID string `json:"id"`
|
ID string `json:"id"`
|
||||||
Text string `json:"text"`
|
Text string `json:"text"`
|
||||||
Root string `json:"root"`
|
Root string `json:"root"`
|
||||||
Confidence string `json:"confidence,omitempty"`
|
|
||||||
Source string `json:"-"`
|
Source string `json:"-"`
|
||||||
Score float64 `json:"score"`
|
Score float64 `json:"score"`
|
||||||
Snippet string `json:"snippet,omitempty"`
|
Snippet string `json:"snippet,omitempty"`
|
||||||
ValidFrom string `json:"valid_from,omitempty"`
|
|
||||||
ValidTo string `json:"valid_to,omitempty"`
|
|
||||||
Hops []HopNode `json:"hops,omitempty"`
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// rrfK dampens the contribution of low ranks; same constant as kblib.py.
|
// rrfK dampens the contribution of low ranks; same constant as kblib.py.
|
||||||
const rrfK = 60
|
const rrfK = 60
|
||||||
|
|
||||||
// RankAndFilter fuses the two hit lists, applies --root/--repo/--as-of, then
|
// RankAndFilter fuses the two hit lists, applies --root/--repo, then cuts to
|
||||||
// cuts to limit. Cutting first dropped every matching leaf ranked below the
|
// limit. Cutting first dropped every matching leaf ranked below the cut, so
|
||||||
// cut, so `--root facts` came back empty whenever info leafs filled the top N.
|
// `--root facts` came back empty whenever info leafs filled the top N.
|
||||||
// limit <= 0 keeps everything. asOf empty skips interval filter (D24).
|
// limit <= 0 keeps everything.
|
||||||
func RankAndFilter(fts, vec []Hit, root, repo string, limit int) []Hit {
|
func RankAndFilter(fts, vec []Hit, root, repo string, limit int) []Hit {
|
||||||
return RankAndFilterAsOf(fts, vec, root, repo, "", limit)
|
|
||||||
}
|
|
||||||
|
|
||||||
// RankAndFilterAsOf is RankAndFilter with D24 fact-interval filter.
|
|
||||||
func RankAndFilterAsOf(fts, vec []Hit, root, repo, asOf string, limit int) []Hit {
|
|
||||||
out := Hybrid(fts, vec, 0)
|
out := Hybrid(fts, vec, 0)
|
||||||
if root != "" {
|
if root != "" {
|
||||||
out = FilterRoot(out, root)
|
out = FilterRoot(out, root)
|
||||||
@@ -50,30 +32,12 @@ func RankAndFilterAsOf(fts, vec []Hit, root, repo, asOf string, limit int) []Hit
|
|||||||
if repo != "" {
|
if repo != "" {
|
||||||
out = FilterRepo(out, repo)
|
out = FilterRepo(out, repo)
|
||||||
}
|
}
|
||||||
if asOf != "" {
|
|
||||||
out = FilterAsOf(out, asOf)
|
|
||||||
}
|
|
||||||
if limit > 0 && len(out) > limit {
|
if limit > 0 && len(out) > limit {
|
||||||
out = out[:limit]
|
out = out[:limit]
|
||||||
}
|
}
|
||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
// FilterAsOf keeps hits whose [valid_from, valid_to] covers asOf (D24).
|
|
||||||
// Empty intervals stay (legacy leafs). Empty asOf keeps all.
|
|
||||||
func FilterAsOf(hits []Hit, asOf string) []Hit {
|
|
||||||
if asOf == "" {
|
|
||||||
return hits
|
|
||||||
}
|
|
||||||
var out []Hit
|
|
||||||
for _, h := range hits {
|
|
||||||
if facts.ActiveAt(h.ValidFrom, h.ValidTo, asOf) {
|
|
||||||
out = append(out, h)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return out
|
|
||||||
}
|
|
||||||
|
|
||||||
// Hybrid merges FTS and vector hits by reciprocal rank fusion.
|
// Hybrid merges FTS and vector hits by reciprocal rank fusion.
|
||||||
// limit <= 0 returns the full fused list.
|
// limit <= 0 returns the full fused list.
|
||||||
func Hybrid(fts, vec []Hit, limit int) []Hit {
|
func Hybrid(fts, vec []Hit, limit int) []Hit {
|
||||||
@@ -91,41 +91,15 @@ func TestHybridKeepsVectorScoreForSharedHit(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// The old parser dropped unknown flags and appended their arguments to the
|
// The old parser dropped unknown flags and appended their arguments to the
|
||||||
// query, so `search "q" --hop 1` searched for "q 1". --hop must stay a flag.
|
// query, so `search "q" --hop 1` searched for "q 1". --hop is not implemented
|
||||||
|
// here (needs File edges); it must still fail closed instead of changing q.
|
||||||
func TestParseHopIsNotSwallowedIntoTheQuery(t *testing.T) {
|
func TestParseHopIsNotSwallowedIntoTheQuery(t *testing.T) {
|
||||||
opt, err := ParseArgs([]string{"what runs on arc-2", "--hop", "1"})
|
_, err := ParseArgs([]string{"what runs on arc-2", "--hop", "1"})
|
||||||
if err != nil {
|
if err == nil {
|
||||||
t.Fatalf("unexpected error: %v", err)
|
t.Fatal("expected --hop to error (not implemented), not be swallowed")
|
||||||
}
|
}
|
||||||
if opt.Query != "what runs on arc-2" {
|
if !strings.Contains(err.Error(), "--hop") {
|
||||||
t.Fatalf("query swallowed hop arg: %q", opt.Query)
|
t.Fatalf("error should name --hop, got %v", err)
|
||||||
}
|
|
||||||
if opt.Hop != 1 {
|
|
||||||
t.Fatalf("hop = %d, want 1", opt.Hop)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestParseHopMaxIsThree(t *testing.T) {
|
|
||||||
if _, err := ParseArgs([]string{"q", "--hop", "4"}); err == nil {
|
|
||||||
t.Fatal("expected --hop 4 to error")
|
|
||||||
}
|
|
||||||
opt, err := ParseArgs([]string{"q", "--hop", "3"})
|
|
||||||
if err != nil || opt.Hop != 3 {
|
|
||||||
t.Fatalf("hop 3: %+v err=%v", opt, err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestHopStmtWalksFromFile(t *testing.T) {
|
|
||||||
s := HopStmt(1)
|
|
||||||
if !strings.Contains(s, "FROM_FILE") || !strings.Contains(s, "File") {
|
|
||||||
t.Fatalf("hop 1 must walk FROM_FILE, got %q", s)
|
|
||||||
}
|
|
||||||
s3 := HopStmt(3)
|
|
||||||
if !strings.Contains(s3, "HAS_VERSION") || !strings.Contains(s3, "AUTHORED") || !strings.Contains(s3, "Person") {
|
|
||||||
t.Fatalf("hop 3 must reach Person, got %q", s3)
|
|
||||||
}
|
|
||||||
if HopLabel(1) != "File" || HopLabel(3) != "Person" {
|
|
||||||
t.Fatal("hop labels")
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -162,17 +136,8 @@ func TestListModelNeedsNoQuery(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestUsageNamesBrainSearch(t *testing.T) {
|
|
||||||
if !strings.Contains(Usage, "bin/brain/search.go") {
|
|
||||||
t.Fatalf("usage must name bin/brain/search.go, got:\n%s", Usage)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestFTSQueryOrdersByScoreDescending(t *testing.T) {
|
func TestFTSQueryOrdersByScoreDescending(t *testing.T) {
|
||||||
if !strings.Contains(FTSStmt, "ORDER BY score DESC") {
|
if !strings.Contains(FTSStmt, "ORDER BY score DESC") {
|
||||||
t.Fatalf("FTS query must order by score DESC, got:\n%s", FTSStmt)
|
t.Fatalf("FTS query must order by score DESC, got:\n%s", FTSStmt)
|
||||||
}
|
}
|
||||||
if !strings.Contains(FTSStmt, "node.confidence") {
|
|
||||||
t.Fatal("FTS must return confidence for D16")
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
@@ -1,6 +1,5 @@
|
|||||||
//go:build cgo && system_ladybug
|
// Hybrid FTS + vector search implementation, plus daemon client/server.
|
||||||
|
package main
|
||||||
package brain
|
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
@@ -19,8 +18,7 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
|
|
||||||
lbug "github.com/LadybugDB/go-ladybug"
|
lbug "github.com/LadybugDB/go-ladybug"
|
||||||
"github.com/eSlider/2dph/internal/brain/rank"
|
"github.com/eSlider/2dph/bin/kbsearch/rank"
|
||||||
"github.com/eSlider/2dph/internal/cli"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
const defaultPort = 17830
|
const defaultPort = 17830
|
||||||
@@ -30,10 +28,7 @@ const healthPath = "/health"
|
|||||||
func runSearch(args []string) int {
|
func runSearch(args []string) int {
|
||||||
opt, err := rank.ParseArgs(args)
|
opt, err := rank.ParseArgs(args)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if errors.Is(err, cli.ErrHelp) {
|
fmt.Fprintf(os.Stderr, "kbsearch: %v\n%s\n", err, rank.Usage)
|
||||||
return 0
|
|
||||||
}
|
|
||||||
fmt.Fprintf(os.Stderr, "brain/search: %v\n%s\n", err, rank.Usage)
|
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
root, repo, limit, query := opt.Root, opt.Repo, opt.Limit, opt.Query
|
root, repo, limit, query := opt.Root, opt.Repo, opt.Limit, opt.Query
|
||||||
@@ -55,19 +50,25 @@ func runSearch(args []string) int {
|
|||||||
}
|
}
|
||||||
defer closeBrain()
|
defer closeBrain()
|
||||||
|
|
||||||
hits, err := searchHits(query, root, repo, limit, opt.AsOf)
|
emb, err := embedQuery(query)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "search: %v\n", err)
|
fmt.Fprintf(os.Stderr, "embed: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
if opt.Hop > 0 {
|
|
||||||
if err := attachHops(hits, opt.Hop); err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "hop: %v\n", err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
results := hits
|
fts, err := queryFTS(query, limit*3)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "fts: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
var vec []Hit
|
||||||
|
if vec, err = queryVector(emb, limit*3); err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "vec: %v\n", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
results := rank.RankAndFilter(fts, vec, root, repo, limit)
|
||||||
|
|
||||||
for i := range results {
|
for i := range results {
|
||||||
if results[i].Text != "" {
|
if results[i].Text != "" {
|
||||||
runes := []rune(results[i].Text)
|
runes := []rune(results[i].Text)
|
||||||
@@ -78,85 +79,23 @@ func runSearch(args []string) int {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
webOut := rank.Deduce(results, query, root, opt.NoWeb, func(q string) rank.SecondSource {
|
|
||||||
return lookupWeb(context.Background(), q)
|
|
||||||
})
|
|
||||||
|
|
||||||
out := Dict{
|
out := Dict{
|
||||||
{"query", query},
|
{"query", query},
|
||||||
{"root_filter", root},
|
{"root_filter", root},
|
||||||
{"as_of", opt.AsOf},
|
|
||||||
{"count", len(results)},
|
{"count", len(results)},
|
||||||
{"results", resultsToDicts(results)},
|
{"results", resultsToDicts(results)},
|
||||||
}
|
}
|
||||||
if webOut != nil {
|
|
||||||
out = append(out, KV{"web", secondToDict(*webOut)})
|
|
||||||
}
|
|
||||||
|
|
||||||
if jsonOut {
|
if jsonOut {
|
||||||
enc := json.NewEncoder(os.Stdout)
|
enc := json.NewEncoder(os.Stdout)
|
||||||
enc.SetIndent("", " ")
|
enc.SetIndent("", " ")
|
||||||
enc.SetEscapeHTML(false)
|
enc.SetEscapeHTML(false)
|
||||||
return b2i(enc.Encode(toJSONOut(results, query, root, opt.AsOf, webOut)))
|
return b2i(enc.Encode(toJSONOut(results, query, root)))
|
||||||
}
|
}
|
||||||
fmt.Print(toYAML(out, 0))
|
fmt.Print(toYAML(out, 0))
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
func searchHits(query, root, repo string, limit int, asOf string) ([]Hit, error) {
|
|
||||||
emb, err := embedQuery(query)
|
|
||||||
if err != nil {
|
|
||||||
return nil, fmt.Errorf("embed: %w", err)
|
|
||||||
}
|
|
||||||
fts, err := queryFTS(query, limit*3)
|
|
||||||
if err != nil {
|
|
||||||
return nil, fmt.Errorf("fts: %w", err)
|
|
||||||
}
|
|
||||||
var vec []Hit
|
|
||||||
if vec, err = queryVector(emb, limit*3); err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "vec: %v\n", err)
|
|
||||||
}
|
|
||||||
return rank.RankAndFilterAsOf(fts, vec, root, repo, asOf, limit), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func attachHops(hits []Hit, n int) error {
|
|
||||||
if conn == nil {
|
|
||||||
return fmt.Errorf("brain not open")
|
|
||||||
}
|
|
||||||
for i := range hits {
|
|
||||||
var hops []rank.HopNode
|
|
||||||
for d := 1; d <= n; d++ {
|
|
||||||
stmt, err := conn.Prepare(rank.HopStmt(d))
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
res, err := conn.Execute(stmt, map[string]any{"id": hits[i].ID})
|
|
||||||
stmt.Close()
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
for res.HasNext() {
|
|
||||||
row, err := res.Next()
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
vals, err := row.GetAsSlice()
|
|
||||||
if err != nil || len(vals) < 3 {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
hops = append(hops, rank.HopNode{
|
|
||||||
ID: fmt.Sprint(vals[0]),
|
|
||||||
Label: rank.HopLabel(d),
|
|
||||||
Name: fmt.Sprint(vals[1]),
|
|
||||||
Depth: int(asInt(vals[2])),
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
hits[i].Hops = hops
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func b2i(err error) int {
|
func b2i(err error) int {
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 1
|
return 1
|
||||||
@@ -217,78 +156,43 @@ func rowsToHits(res *lbug.QueryResult) ([]Hit, error) {
|
|||||||
root := fmt.Sprint(vals[2])
|
root := fmt.Sprint(vals[2])
|
||||||
source := fmt.Sprint(vals[3])
|
source := fmt.Sprint(vals[3])
|
||||||
score := float64(vals[4].(float64))
|
score := float64(vals[4].(float64))
|
||||||
conf := ""
|
hits = append(hits, Hit{ID: id, Text: text, Root: root, Source: source, Score: score})
|
||||||
if len(vals) >= 6 {
|
|
||||||
conf = fmt.Sprint(vals[5])
|
|
||||||
}
|
|
||||||
vf, vt := "", ""
|
|
||||||
if len(vals) >= 8 {
|
|
||||||
vf = nullStr(vals[6])
|
|
||||||
vt = nullStr(vals[7])
|
|
||||||
}
|
|
||||||
hits = append(hits, Hit{
|
|
||||||
ID: id, Text: text, Root: root, Source: source, Score: score,
|
|
||||||
Confidence: conf, ValidFrom: vf, ValidTo: vt,
|
|
||||||
})
|
|
||||||
}
|
}
|
||||||
return hits, nil
|
return hits, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func nullStr(v any) string {
|
|
||||||
if v == nil {
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
s := fmt.Sprint(v)
|
|
||||||
if s == "<nil>" {
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
return s
|
|
||||||
}
|
|
||||||
|
|
||||||
// JSON output types
|
// JSON output types
|
||||||
type jsonOut struct {
|
type jsonOut struct {
|
||||||
Query string `json:"query"`
|
Query string `json:"query"`
|
||||||
RootFilter string `json:"root_filter"`
|
RootFilter string `json:"root_filter"`
|
||||||
AsOf string `json:"as_of,omitempty"`
|
|
||||||
Count int `json:"count"`
|
Count int `json:"count"`
|
||||||
Results []jsonHit `json:"results"`
|
Results []jsonHit `json:"results"`
|
||||||
Web *rank.SecondSource `json:"web,omitempty"`
|
|
||||||
}
|
}
|
||||||
|
|
||||||
type jsonHit struct {
|
type jsonHit struct {
|
||||||
ID string `json:"id"`
|
ID string `json:"id"`
|
||||||
Text string `json:"text"`
|
Text string `json:"text"`
|
||||||
Root string `json:"root"`
|
Root string `json:"root"`
|
||||||
Confidence string `json:"confidence,omitempty"`
|
|
||||||
Score float64 `json:"score"`
|
Score float64 `json:"score"`
|
||||||
Snippet string `json:"snippet,omitempty"`
|
Snippet string `json:"snippet,omitempty"`
|
||||||
ValidFrom string `json:"valid_from,omitempty"`
|
|
||||||
ValidTo string `json:"valid_to,omitempty"`
|
|
||||||
Hops []rank.HopNode `json:"hops,omitempty"`
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func toJSONOut(hits []Hit, query, rootFilter, asOf string, web *rank.SecondSource) *jsonOut {
|
func toJSONOut(hits []Hit, query, rootFilter string) *jsonOut {
|
||||||
out := make([]jsonHit, len(hits))
|
out := make([]jsonHit, len(hits))
|
||||||
for i, h := range hits {
|
for i, h := range hits {
|
||||||
out[i] = jsonHit{
|
out[i] = jsonHit{
|
||||||
ID: h.ID,
|
ID: h.ID,
|
||||||
Text: h.Text,
|
Text: h.Text,
|
||||||
Root: h.Root,
|
Root: h.Root,
|
||||||
Confidence: h.Confidence,
|
|
||||||
Score: h.Score,
|
Score: h.Score,
|
||||||
Snippet: h.Snippet,
|
Snippet: h.Snippet,
|
||||||
ValidFrom: h.ValidFrom,
|
|
||||||
ValidTo: h.ValidTo,
|
|
||||||
Hops: h.Hops,
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return &jsonOut{
|
return &jsonOut{
|
||||||
Query: query,
|
Query: query,
|
||||||
RootFilter: rootFilter,
|
RootFilter: rootFilter,
|
||||||
AsOf: asOf,
|
|
||||||
Count: len(hits),
|
Count: len(hits),
|
||||||
Results: out,
|
Results: out,
|
||||||
Web: web,
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -301,30 +205,9 @@ func resultsToDicts(hits []Hit) []any {
|
|||||||
{"root", h.Root},
|
{"root", h.Root},
|
||||||
{"score", h.Score},
|
{"score", h.Score},
|
||||||
}
|
}
|
||||||
if h.Confidence != "" {
|
|
||||||
d = append(d, KV{"confidence", h.Confidence})
|
|
||||||
}
|
|
||||||
if h.ValidFrom != "" {
|
|
||||||
d = append(d, KV{"valid_from", h.ValidFrom})
|
|
||||||
}
|
|
||||||
if h.ValidTo != "" {
|
|
||||||
d = append(d, KV{"valid_to", h.ValidTo})
|
|
||||||
}
|
|
||||||
if h.Snippet != "" {
|
if h.Snippet != "" {
|
||||||
d = append(d, KV{"snippet", h.Snippet})
|
d = append(d, KV{"snippet", h.Snippet})
|
||||||
}
|
}
|
||||||
if len(h.Hops) > 0 {
|
|
||||||
nodes := make([]any, len(h.Hops))
|
|
||||||
for j, n := range h.Hops {
|
|
||||||
nodes[j] = Dict{
|
|
||||||
{"id", n.ID},
|
|
||||||
{"label", n.Label},
|
|
||||||
{"name", n.Name},
|
|
||||||
{"depth", n.Depth},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
d = append(d, KV{"hops", nodes})
|
|
||||||
}
|
|
||||||
out[i] = d
|
out[i] = d
|
||||||
}
|
}
|
||||||
return out
|
return out
|
||||||
@@ -364,7 +247,7 @@ func serve(port int) error {
|
|||||||
})
|
})
|
||||||
|
|
||||||
addr := fmt.Sprintf("127.0.0.1:%d", port)
|
addr := fmt.Sprintf("127.0.0.1:%d", port)
|
||||||
log.Printf("brain search daemon listening on %s", addr)
|
log.Printf("kbsearch daemon listening on %s", addr)
|
||||||
return http.ListenAndServe(addr, mux)
|
return http.ListenAndServe(addr, mux)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -474,21 +357,3 @@ func ensureDaemon(port int) error {
|
|||||||
}
|
}
|
||||||
return fmt.Errorf("daemon failed to start on port %d", port)
|
return fmt.Errorf("daemon failed to start on port %d", port)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Main is the bin/brain/search.go entry: search, serve, or --list-model.
|
|
||||||
func Main(args []string) int {
|
|
||||||
if len(args) > 0 && args[0] == "serve" {
|
|
||||||
port := defaultPort
|
|
||||||
if len(args) > 1 {
|
|
||||||
if p, err := strconv.Atoi(args[1]); err == nil {
|
|
||||||
port = p
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if err := serve(port); err != nil {
|
|
||||||
log.Printf("brain/search serve: %v", err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
return runSearch(args)
|
|
||||||
}
|
|
||||||
@@ -1,9 +1,10 @@
|
|||||||
package brain
|
// Common types and helpers for kbsearch.
|
||||||
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"os"
|
"os"
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/brain/rank"
|
"github.com/eSlider/2dph/bin/kbsearch/rank"
|
||||||
)
|
)
|
||||||
|
|
||||||
func eps() string { return os.Getenv("KBTEST_EPS") }
|
func eps() string { return os.Getenv("KBTEST_EPS") }
|
||||||
@@ -1,5 +1,5 @@
|
|||||||
// YAML emitter ported from bin/kb/yamlout.py — preserves insertion order.
|
// YAML emitter ported from bin/kb/yamlout.py — preserves insertion order.
|
||||||
package brain
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
+78
-12
@@ -7,7 +7,7 @@
|
|||||||
bin/mail/import --since 2026-01-01 only messages after a date
|
bin/mail/import --since 2026-01-01 only messages after a date
|
||||||
bin/mail/import --limit 50 cap messages per run
|
bin/mail/import --limit 50 cap messages per run
|
||||||
bin/mail/import --no-attachments body only, skip attachment conversion
|
bin/mail/import --no-attachments body only, skip attachment conversion
|
||||||
bin/mail/import --ocr OCR images (PDFs OCR when textless)
|
bin/mail/import --ocr OCR scanned PDFs/images via docling
|
||||||
bin/mail/import --dry-run list messages without writing anything
|
bin/mail/import --dry-run list messages without writing anything
|
||||||
|
|
||||||
Writes one directory per message: var/mail/{folder}/{message_id}/
|
Writes one directory per message: var/mail/{folder}/{message_id}/
|
||||||
@@ -15,12 +15,11 @@ Writes one directory per message: var/mail/{folder}/{message_id}/
|
|||||||
attachments/ raw attachment files (zips unpacked to _unpacked/)
|
attachments/ raw attachment files (zips unpacked to _unpacked/)
|
||||||
attachments/*.md converted attachment content
|
attachments/*.md converted attachment content
|
||||||
|
|
||||||
Indexing is a separate step (`bin/brain/index.go --rebuild`): conversion can
|
Indexing is a separate step (bin/mail/index_mail): conversion can crash in
|
||||||
crash and must not leave the brain DB mid-transaction.
|
native docling and must not leave the brain DB mid-transaction.
|
||||||
|
|
||||||
Requires ONLYOFFICE_URL/USER/PASS in .env (or env) except `--from-raw`.
|
Requires ONLYOFFICE_URL/USER/PASS in .env (or env). Idempotent: a message
|
||||||
Idempotent: a message already present (message.md exists) is skipped unless
|
already present (message.md exists) is skipped unless --force.
|
||||||
--force.
|
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -42,10 +41,10 @@ from mailconv import ( # noqa: E402
|
|||||||
IMAGE_SUFFIXES,
|
IMAGE_SUFFIXES,
|
||||||
LEGACY_OFFICE_SUFFIXES,
|
LEGACY_OFFICE_SUFFIXES,
|
||||||
TEXT_SUFFIXES,
|
TEXT_SUFFIXES,
|
||||||
convert_pdf,
|
|
||||||
html_to_markdown,
|
html_to_markdown,
|
||||||
|
is_convertible,
|
||||||
normalize_markdown,
|
normalize_markdown,
|
||||||
ocr_image,
|
subject_to_filename,
|
||||||
zip_extract_safe,
|
zip_extract_safe,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -148,9 +147,9 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
return f"\n<!-- conversion failed: {e} -->\n"
|
return f"\n<!-- conversion failed: {e} -->\n"
|
||||||
if suffix == ".pdf":
|
if suffix == ".pdf":
|
||||||
return convert_pdf(path, ocr)
|
return _convert_pdf(path, ocr)
|
||||||
if suffix in IMAGE_SUFFIXES and ocr:
|
if suffix in IMAGE_SUFFIXES and ocr:
|
||||||
return ocr_image(path) or "\n<!-- ocr unavailable -->\n"
|
return _convert_pdf(path, ocr)
|
||||||
if suffix in LEGACY_OFFICE_SUFFIXES:
|
if suffix in LEGACY_OFFICE_SUFFIXES:
|
||||||
return _convert_legacy(path)
|
return _convert_legacy(path)
|
||||||
if suffix in ARCHIVE_SUFFIXES:
|
if suffix in ARCHIVE_SUFFIXES:
|
||||||
@@ -158,6 +157,67 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _convert_pdf(path: Path, ocr: bool) -> str:
|
||||||
|
"""Convert one PDF to markdown.
|
||||||
|
|
||||||
|
Fast path: poppler's pdftotext (-layout) extracts exact text from
|
||||||
|
born-digital PDFs in ~15ms vs docling's 1-3s. Only textless PDFs (scanned
|
||||||
|
pages, layout-heavy) fall back to docling, which runs isolated in a
|
||||||
|
subprocess because its native onnx/RT-DETR has segfaulted the main process.
|
||||||
|
"""
|
||||||
|
text = _pdf_fast_text(path)
|
||||||
|
if ocr or text is None or not text.strip():
|
||||||
|
return _convert_pdf_docling(path, ocr)
|
||||||
|
return normalize_markdown(text)
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_fast_text(path: Path) -> str | None:
|
||||||
|
"""pdftotext -layout; None when poppler is unavailable (or the PDF has no text layer)."""
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
["pdftotext", "-layout", str(path), "-"],
|
||||||
|
capture_output=True, timeout=60)
|
||||||
|
except (OSError, subprocess.TimeoutExpired):
|
||||||
|
return None
|
||||||
|
if proc.returncode != 0:
|
||||||
|
return None
|
||||||
|
return proc.stdout.decode("utf-8", errors="replace")
|
||||||
|
|
||||||
|
|
||||||
|
def _convert_pdf_docling(path: Path, ocr: bool) -> str:
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
[sys.executable, os.path.abspath(__file__), "--pdf-worker", str(path),
|
||||||
|
"--ocr" if ocr else "--no-ocr"],
|
||||||
|
capture_output=True, text=True, timeout=600)
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
return "\n<!-- pdf conversion timed out -->\n"
|
||||||
|
if proc.returncode != 0:
|
||||||
|
tail = proc.stderr.strip().splitlines()[-3:]
|
||||||
|
return f"\n<!-- pdf conversion failed: {proc.returncode}: {' | '.join(tail)} -->\n"
|
||||||
|
return proc.stdout
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_worker(path: Path, ocr: bool) -> None:
|
||||||
|
"""docling worker entry: prints converted markdown on stdout, exits non-zero on error."""
|
||||||
|
try:
|
||||||
|
from docling.document_converter import DocumentConverter, PdfFormatOption
|
||||||
|
from docling.datamodel.pipeline_options import PdfPipelineOptions
|
||||||
|
opts = PdfPipelineOptions()
|
||||||
|
opts.do_ocr = bool(ocr)
|
||||||
|
opts.do_table_structure = True
|
||||||
|
conv = DocumentConverter(format_options={"pdf": PdfFormatOption(pipeline_options=opts)})
|
||||||
|
res = conv.convert(str(path))
|
||||||
|
sys.stdout.write(normalize_markdown(res.document.export_to_markdown()))
|
||||||
|
sys.exit(0)
|
||||||
|
except Exception as e:
|
||||||
|
# errors/stacktraces to stderr; the caller only reports a one-liner
|
||||||
|
print(f"pdf-worker: {e}", file=sys.stderr)
|
||||||
|
import traceback
|
||||||
|
traceback.print_exc(file=sys.stderr)
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
|
||||||
def _convert_legacy(path: Path) -> str:
|
def _convert_legacy(path: Path) -> str:
|
||||||
"""Legacy .doc/.xls/.ppt -> md via pandoc (installed) or a stub."""
|
"""Legacy .doc/.xls/.ppt -> md via pandoc (installed) or a stub."""
|
||||||
try:
|
try:
|
||||||
@@ -296,12 +356,19 @@ def main(argv: list[str]) -> int:
|
|||||||
p.add_argument("--from-raw", default="",
|
p.add_argument("--from-raw", default="",
|
||||||
help="convert Go-synced dirs (var/mail/<folder>/<id>/message.json) to markdown")
|
help="convert Go-synced dirs (var/mail/<folder>/<id>/message.json) to markdown")
|
||||||
p.add_argument("--no-attachments", action="store_true", help="skip attachment download+convert")
|
p.add_argument("--no-attachments", action="store_true", help="skip attachment download+convert")
|
||||||
p.add_argument("--ocr", action="store_true", help="OCR images (PDFs OCR when textless)")
|
p.add_argument("--ocr", action="store_true", help="OCR scanned PDFs/images via docling")
|
||||||
p.add_argument("--force", action="store_true", help="re-import even if message.md exists")
|
p.add_argument("--force", action="store_true", help="re-import even if message.md exists")
|
||||||
p.add_argument("--dry-run", action="store_true", help="list messages, write nothing")
|
p.add_argument("--dry-run", action="store_true", help="list messages, write nothing")
|
||||||
p.add_argument("--json", action="store_true")
|
p.add_argument("--json", action="store_true")
|
||||||
|
p.add_argument("--pdf-worker", default="", help=argparse.SUPPRESS)
|
||||||
|
p.add_argument("--no-ocr", action="store_true", help=argparse.SUPPRESS)
|
||||||
a = p.parse_args(argv)
|
a = p.parse_args(argv)
|
||||||
|
|
||||||
|
if a.pdf_worker:
|
||||||
|
_pdf_worker(Path(a.pdf_worker), ocr=not a.no_ocr)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
conf = load_env()
|
||||||
fid = folder_id(a.folder)
|
fid = folder_id(a.folder)
|
||||||
out_root = ROOT / "var" / "mail"
|
out_root = ROOT / "var" / "mail"
|
||||||
summary: list[dict] = []
|
summary: list[dict] = []
|
||||||
@@ -327,7 +394,6 @@ def main(argv: list[str]) -> int:
|
|||||||
target_dir=msg_dir.parent))
|
target_dir=msg_dir.parent))
|
||||||
summary.append(entry)
|
summary.append(entry)
|
||||||
else:
|
else:
|
||||||
conf = load_env()
|
|
||||||
OOCLIENT = OOClient(conf)
|
OOCLIENT = OOClient(conf)
|
||||||
if a.id:
|
if a.id:
|
||||||
messages = [{"id": i} for i in a.id]
|
messages = [{"id": i} for i in a.id]
|
||||||
|
|||||||
@@ -1,20 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=mail_import "$0" "$@"; exit
|
|
||||||
//go:build mail_import
|
|
||||||
//
|
|
||||||
// bin/mail/import.go - message.json → markdown (no brain write).
|
|
||||||
//
|
|
||||||
// ./bin/mail/import.go --from-raw var/mail
|
|
||||||
//
|
|
||||||
// Indexing is bin/brain/index.go --rebuild, not this command.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/cmdbin"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(cmdbin.ExecFile("bin/mail/import", os.Args[1:]))
|
|
||||||
}
|
|
||||||
+120
-11
@@ -1,26 +1,135 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""mail/index_mail — deprecated. Use bin/brain/index.go --rebuild --with-mail.
|
"""mail/index_mail - rebuild the brain with every markdown under var/mail.
|
||||||
|
|
||||||
Ladybug corrupts its WAL on bulk-insert into an already-indexed DB, so this
|
Ladybug corrupts its WAL when brand-new leafs are bulk-inserted while the
|
||||||
shim always rebuilds (repo corpus + var/mail). Conversion stays in mail/import.
|
FTS/VECTOR indexes already exist, so indexing ALWAYS runs as a fresh rebuild
|
||||||
|
(repo corpus + var/mail), matching the proven-safe `kb/index --rebuild` path.
|
||||||
|
Conversion and indexing stay separate: conversion can crash in native docling
|
||||||
|
and must not leave the brain DB mid-transaction.
|
||||||
|
|
||||||
|
bin/mail/index_mail rebuild the index incl. all mail
|
||||||
|
bin/mail/index_mail --dry-run count without writing
|
||||||
|
bin/mail/index_mail --limit N cap messages included
|
||||||
|
bin/mail/index_mail --since D only messages dated >= D (YYYY-MM-DD)
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import os
|
import argparse
|
||||||
|
import json
|
||||||
import sys
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[2]
|
ROOT = Path(__file__).resolve().parents[2]
|
||||||
|
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
||||||
|
|
||||||
|
from kblib import DB_PATH, VAR, connect, ensure_indexes, init_schema, stats, upsert_leaf # noqa: E402
|
||||||
|
from mdleaves import read_markdown, to_all, walk_markdown # noqa: E402
|
||||||
|
|
||||||
|
|
||||||
|
def msg_date(md: Path) -> str:
|
||||||
|
j = md.parent / "message.json"
|
||||||
|
try:
|
||||||
|
d = json.loads(j.read_text(encoding="utf-8"))
|
||||||
|
return (d.get("receivedDate") or d.get("receivedAt") or "")[:10]
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def mail_leafs(limit: int, since: str, repo: str = "ooMail") -> list[dict]:
|
||||||
|
root = ROOT / "var" / "mail"
|
||||||
|
mds = sorted(root.rglob("message.md"))
|
||||||
|
if since:
|
||||||
|
mds = [m for m in mds if msg_date(m) >= since]
|
||||||
|
if limit:
|
||||||
|
mds = mds[:limit]
|
||||||
|
leafs: list[dict] = []
|
||||||
|
for md in mds:
|
||||||
|
files = [md] + sorted((md.parent / "attachments").glob("*.md"))
|
||||||
|
for f in files:
|
||||||
|
if not f.exists():
|
||||||
|
continue
|
||||||
|
for lf in to_all(read_markdown(f), f, repo=repo):
|
||||||
|
lf["source"] = f"ooMail:{md.parent.name}:{f.name}"
|
||||||
|
lf["how"] = "mail/import"
|
||||||
|
leafs.append(lf)
|
||||||
|
return leafs
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str]) -> int:
|
def main(argv: list[str]) -> int:
|
||||||
print(
|
p = argparse.ArgumentParser(description="rebuild the brain incl. all mail")
|
||||||
"bin/mail/index_mail is deprecated; use bin/brain/index.go --rebuild --with-mail",
|
p.add_argument("--dry-run", action="store_true", help="count only, write nothing")
|
||||||
file=sys.stderr,
|
p.add_argument("--limit", type=int, default=0, help="cap messages included")
|
||||||
)
|
p.add_argument("--since", default="", help="only messages dated >= YYYY-MM-DD")
|
||||||
index = ROOT / "bin" / "kb" / "index"
|
p.add_argument("--json", action="store_true")
|
||||||
os.execv(sys.executable, [sys.executable, str(index), "--rebuild", "--with-mail", *argv])
|
a = p.parse_args(argv)
|
||||||
return 1
|
|
||||||
|
mail = mail_leafs(a.limit, a.since)
|
||||||
|
if a.dry_run:
|
||||||
|
print(f"mail/index_mail: {len(mail)} mail leafs would be indexed")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# Fresh rebuild: delete DB, index repo corpus + mail, create indexes once
|
||||||
|
# at the end. Never insert into an already-indexed DB (WAL corruption).
|
||||||
|
VAR.mkdir(exist_ok=True)
|
||||||
|
if DB_PATH.exists():
|
||||||
|
DB_PATH.unlink()
|
||||||
|
|
||||||
|
corpus = _load_corpus()
|
||||||
|
leafs = corpus + mail
|
||||||
|
|
||||||
|
db, conn = connect(DB_PATH, read_only=False)
|
||||||
|
init_schema(conn)
|
||||||
|
embed = _embedder()
|
||||||
|
done, total = _index_leafs(conn, leafs, embed)
|
||||||
|
ensure_indexes(conn)
|
||||||
|
s = stats(conn)
|
||||||
|
conn.close()
|
||||||
|
db.close()
|
||||||
|
|
||||||
|
result = {"indexed": done, "corpus_total": total, "mail_leafs": len(mail),
|
||||||
|
**{k: v for k, v in s.items() if k in ("total", "by_root")}}
|
||||||
|
print(json.dumps(result, indent=2) if a.json else
|
||||||
|
f"mail/index_mail: indexed {done}/{total} leafs (mail={len(mail)}); db total {s['total']}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
CORPUS_DEFAULTS = ["README.md", "PLAN.md", "AGENTS.md", "docs", "skills"]
|
||||||
|
|
||||||
|
|
||||||
|
def _load_corpus() -> list[dict]:
|
||||||
|
files: list[Path] = []
|
||||||
|
for entry in CORPUS_DEFAULTS:
|
||||||
|
p = ROOT / entry
|
||||||
|
if p.is_file():
|
||||||
|
files.append(p)
|
||||||
|
elif p.is_dir():
|
||||||
|
files.extend(walk_markdown(p))
|
||||||
|
leafs: list[dict] = []
|
||||||
|
for path in files:
|
||||||
|
try:
|
||||||
|
leafs.extend(to_all(read_markdown(path), path, repo="eSlider/2dph"))
|
||||||
|
except OSError as e:
|
||||||
|
print(f"mail/index_mail: skip {path}: {e}", file=sys.stderr)
|
||||||
|
return leafs
|
||||||
|
|
||||||
|
|
||||||
|
def _index_leafs(conn, leafs: list[dict], embed_fn) -> tuple[int, int]:
|
||||||
|
count = 0
|
||||||
|
for lf in leafs:
|
||||||
|
query = f"{lf['heading']}\n\n{lf['text']}"
|
||||||
|
emb = embed_fn(lf["text"]) if lf["text"] else None
|
||||||
|
upsert_leaf(conn, text=query, root="info", confidence="confirmed",
|
||||||
|
source=lf["source"], source_rev="mail" if lf.get("how") == "mail/import" else "working-tree",
|
||||||
|
how=lf.get("how", "kb/index"), loc=lf["source"], type_=lf.get("type", "reference"),
|
||||||
|
embedding=emb)
|
||||||
|
count += 1
|
||||||
|
return count, len(leafs)
|
||||||
|
|
||||||
|
|
||||||
|
def _embedder():
|
||||||
|
from model2vec import StaticModel
|
||||||
|
model = StaticModel.from_pretrained("minishlab/potion-multilingual-128M")
|
||||||
|
return lambda text: model.encode([text])[0].astype(float).tolist()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
@@ -1,46 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=mail_ocr "$0" "$@"; exit
|
|
||||||
//go:build mail_ocr
|
|
||||||
//
|
|
||||||
// bin/mail/ocr.go - OCR an image or scanned PDF (tesseract eng+deu).
|
|
||||||
//
|
|
||||||
// ./bin/mail/ocr.go scan.png
|
|
||||||
// ./bin/mail/ocr.go scan.pdf
|
|
||||||
// OCR_ENGINE=paddle ./bin/mail/ocr.go scan.png
|
|
||||||
//
|
|
||||||
// PDFs try pdftotext -layout first; empty text layer uses pdftoppm + tesseract.
|
|
||||||
// No gocv. Tesseract CGO bindings are not used (D21 Zig owns Ladybug CGO).
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"fmt"
|
|
||||||
"os"
|
|
||||||
"strings"
|
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
"github.com/eSlider/2dph/internal/ocr"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(run(os.Args[1:]))
|
|
||||||
}
|
|
||||||
|
|
||||||
func run(args []string) int {
|
|
||||||
c, err := ocr.ParseArgs(args)
|
|
||||||
if err != nil {
|
|
||||||
return cliparse.Fail(err)
|
|
||||||
}
|
|
||||||
path := c.Path
|
|
||||||
var text string
|
|
||||||
if strings.HasSuffix(strings.ToLower(path), ".pdf") {
|
|
||||||
text, err = ocr.PDFFile(path)
|
|
||||||
} else {
|
|
||||||
text, err = ocr.ImageFile(path)
|
|
||||||
}
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "mail/ocr: %v\n", err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
fmt.Println(text)
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
+3
-4
@@ -1,13 +1,12 @@
|
|||||||
//usr/bin/env go run "$0" "$@"; exit
|
//usr/bin/env go run "$0" "$@"; exit
|
||||||
// bin/mail/sync.go - async download of OnlyOffice, Gmail and M365 mail to var/mail/.
|
// bin/mail/sync.go - async download of OnlyOffice and Gmail mail to var/mail/.
|
||||||
//
|
//
|
||||||
// ./bin/mail/sync.go --source onlyoffice,gmail,m365 --limit 50 --workers 8
|
// ./bin/mail/sync.go --source onlyoffice,gmail --limit 50 --workers 8
|
||||||
// ./bin/mail/sync.go --source gmail --force
|
// ./bin/mail/sync.go --source gmail --force
|
||||||
// ./bin/mail/sync.go --source m365 --env .secrets/m365.env
|
|
||||||
// ./bin/mail/sync.go --dry-run
|
// ./bin/mail/sync.go --dry-run
|
||||||
//
|
//
|
||||||
// Writes raw message.json + attachments under var/mail/<folder>/<id>/; run
|
// Writes raw message.json + attachments under var/mail/<folder>/<id>/; run
|
||||||
// bin/mail/import.go --from-raw afterwards to convert everything to markdown.
|
// bin/mail/import --from-raw afterwards to convert everything to markdown.
|
||||||
//
|
//
|
||||||
// Shebang trick: first line is a Go `//` comment; the real code lives in the
|
// Shebang trick: first line is a Go `//` comment; the real code lives in the
|
||||||
// importable package (module path, never a relative import).
|
// importable package (module path, never a relative import).
|
||||||
|
|||||||
+42
-89
@@ -5,15 +5,12 @@ package sync
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"errors"
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
"github.com/integrii/flaggy"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// CLIConfig is a superset of SyncConfig plus flag parsing results.
|
// CLIConfig is a superset of SyncConfig plus flag parsing results.
|
||||||
@@ -24,84 +21,58 @@ type CLIConfig struct {
|
|||||||
Help bool
|
Help bool
|
||||||
}
|
}
|
||||||
|
|
||||||
type flagVals struct {
|
// ParseCLI reads os.Args into a CLIConfig. Exit codes: 0 ok, 2 usage.
|
||||||
env, out, srcs, query string
|
|
||||||
workers, limit, offset int
|
|
||||||
force, dryRun bool
|
|
||||||
}
|
|
||||||
|
|
||||||
func Parser() *flaggy.Parser {
|
|
||||||
v := flagVals{workers: 4, query: "in:inbox", srcs: "onlyoffice"}
|
|
||||||
return bind(&v)
|
|
||||||
}
|
|
||||||
|
|
||||||
func bind(v *flagVals) *flaggy.Parser {
|
|
||||||
if v.workers == 0 {
|
|
||||||
v.workers = 4
|
|
||||||
}
|
|
||||||
if v.query == "" {
|
|
||||||
v.query = "in:inbox"
|
|
||||||
}
|
|
||||||
if v.srcs == "" {
|
|
||||||
v.srcs = "onlyoffice"
|
|
||||||
}
|
|
||||||
p := cliparse.New("mail-sync")
|
|
||||||
p.Description = "download mail to var/mail"
|
|
||||||
p.String(&v.env, "", "env", ".env file")
|
|
||||||
p.String(&v.out, "", "out", "var/mail root")
|
|
||||||
p.Int(&v.workers, "", "workers", "concurrent downloads")
|
|
||||||
p.Int(&v.limit, "", "limit", "max messages per source (0 = all)")
|
|
||||||
p.Int(&v.offset, "", "offset", "skip first N messages per source")
|
|
||||||
p.Bool(&v.force, "", "force", "overwrite existing message.json")
|
|
||||||
p.Bool(&v.dryRun, "", "dry-run", "list counts without writing")
|
|
||||||
p.String(&v.query, "", "query", "Gmail search query")
|
|
||||||
p.String(&v.srcs, "", "source", "comma list: onlyoffice,gmail,m365")
|
|
||||||
return p
|
|
||||||
}
|
|
||||||
|
|
||||||
// ParseCLI reads args into a CLIConfig. Exit codes: 0 ok, 2 usage.
|
|
||||||
func ParseCLI(args []string) (CLIConfig, int, error) {
|
func ParseCLI(args []string) (CLIConfig, int, error) {
|
||||||
v := flagVals{workers: 4, query: "in:inbox", srcs: "onlyoffice"}
|
fs := flag.NewFlagSet("mail/sync", flag.ContinueOnError)
|
||||||
p := bind(&v)
|
var (
|
||||||
if err := cliparse.Parse(p, args); err != nil {
|
env = fs.String("env", "", ".env file (default: <cwd>/.env)")
|
||||||
if errors.Is(err, cliparse.ErrHelp) {
|
out = fs.String("out", "", "var/mail root (default: <cwd>/var/mail)")
|
||||||
return CLIConfig{Help: true}, 0, nil
|
workers = fs.Int("workers", 4, "concurrent downloads")
|
||||||
}
|
limit = fs.Int("limit", 0, "max messages per source (0 = all)")
|
||||||
|
offset = fs.Int("offset", 0, "skip first N messages per source")
|
||||||
|
force = fs.Bool("force", false, "overwrite existing message.json + attachments")
|
||||||
|
dryRun = fs.Bool("dry-run", false, "list message counts without writing")
|
||||||
|
query = fs.String("query", "in:inbox", "Gmail search query (gmail source only)")
|
||||||
|
srcs = fs.String("source", "onlyoffice", "comma list: onlyoffice,gmail (default onlyoffice)")
|
||||||
|
help = fs.Bool("help", false, "usage")
|
||||||
|
)
|
||||||
|
fs.SetOutput(os.Stderr)
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
return CLIConfig{}, 2, err
|
return CLIConfig{}, 2, err
|
||||||
}
|
}
|
||||||
if len(p.TrailingArguments) > 0 {
|
if *help || fs.NArg() > 0 {
|
||||||
return CLIConfig{Help: true}, 0, nil
|
return CLIConfig{Help: true}, 0, nil
|
||||||
}
|
}
|
||||||
wd, err := os.Getwd()
|
wd, err := os.Getwd()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return CLIConfig{}, 2, err
|
return CLIConfig{}, 2, err
|
||||||
}
|
}
|
||||||
if v.env == "" {
|
if *env == "" {
|
||||||
v.env = filepath.Join(wd, ".env")
|
*env = filepath.Join(wd, ".env")
|
||||||
}
|
}
|
||||||
if v.out == "" {
|
if *out == "" {
|
||||||
v.out = filepath.Join(wd, "var", "mail")
|
*out = filepath.Join(wd, "var", "mail")
|
||||||
}
|
}
|
||||||
envVars := readEnv(v.env)
|
envVars := readEnv(*env)
|
||||||
cfg := SyncConfig{
|
cfg := SyncConfig{
|
||||||
Out: v.out,
|
Out: *out,
|
||||||
Workers: v.workers,
|
Workers: *workers,
|
||||||
Limit: v.limit,
|
Limit: *limit,
|
||||||
Offset: v.offset,
|
Offset: *offset,
|
||||||
Force: v.force,
|
Force: *force,
|
||||||
DryRun: v.dryRun,
|
DryRun: *dryRun,
|
||||||
Query: v.query,
|
Query: *query,
|
||||||
Policy: RetryPolicy{},
|
Policy: RetryPolicy{},
|
||||||
}
|
}
|
||||||
out := CLIConfig{Sync: cfg, Env: v.env, Sources: v.srcs}
|
cli := CLIConfig{Sync: cfg, Env: *env, Sources: *srcs}
|
||||||
for _, s := range strings.Split(v.srcs, ",") {
|
for _, s := range strings.Split(*srcs, ",") {
|
||||||
switch strings.TrimSpace(s) {
|
switch strings.TrimSpace(s) {
|
||||||
case "onlyoffice":
|
case "onlyoffice":
|
||||||
u := pick(envVars["ONLYOFFICE_URL"], envVars["OO_URL"])
|
u := pick(envVars["ONLYOFFICE_URL"], envVars["OO_URL"])
|
||||||
user := pick(envVars["ONLYOFFICE_USER"], envVars["OO_USER"])
|
user := pick(envVars["ONLYOFFICE_USER"], envVars["OO_USER"])
|
||||||
pass := pick(envVars["ONLYOFFICE_PASS"], envVars["OO_PASSWORD"])
|
pass := pick(envVars["ONLYOFFICE_PASS"], envVars["OO_PASSWORD"])
|
||||||
if u == "" || user == "" || pass == "" {
|
if u == "" || user == "" || pass == "" {
|
||||||
return CLIConfig{}, 2, fmt.Errorf("onlyoffice source needs ONLYOFFICE_URL/USER/PASS in %s", v.env)
|
return CLIConfig{}, 2, fmt.Errorf("onlyoffice source needs ONLYOFFICE_URL/USER/PASS in %s", *env)
|
||||||
}
|
}
|
||||||
cfg.OO = &OOConfig{URL: u, User: user, Password: pass}
|
cfg.OO = &OOConfig{URL: u, User: user, Password: pass}
|
||||||
case "gmail":
|
case "gmail":
|
||||||
@@ -110,52 +81,34 @@ func ParseCLI(args []string) (CLIConfig, int, error) {
|
|||||||
CredentialsPath: filepath.Join(home, ".gmail-mcp", "credentials.json"),
|
CredentialsPath: filepath.Join(home, ".gmail-mcp", "credentials.json"),
|
||||||
KeysPath: filepath.Join(home, ".gmail-mcp", "gcp-oauth.keys.json"),
|
KeysPath: filepath.Join(home, ".gmail-mcp", "gcp-oauth.keys.json"),
|
||||||
}
|
}
|
||||||
case "m365":
|
|
||||||
tenant := pick(envVars["M365_TENANT"], envVars["MS_TENANT"])
|
|
||||||
cid := pick(envVars["M365_CLIENT_ID"], envVars["MS_CLIENT_ID"])
|
|
||||||
sec := pick(envVars["M365_CLIENT_SECRET"], envVars["MS_CLIENT_SECRET"])
|
|
||||||
users := pick(envVars["M365_USERS"], envVars["MS_USERS"])
|
|
||||||
if tenant == "" || cid == "" || sec == "" || users == "" {
|
|
||||||
return CLIConfig{}, 2, fmt.Errorf("m365 source needs M365_TENANT/CLIENT_ID/CLIENT_SECRET/USERS in %s", v.env)
|
|
||||||
}
|
|
||||||
var userList []string
|
|
||||||
for _, u := range strings.Split(users, ",") {
|
|
||||||
if u = strings.TrimSpace(u); u != "" {
|
|
||||||
userList = append(userList, u)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if len(userList) == 0 {
|
|
||||||
return CLIConfig{}, 2, fmt.Errorf("m365 source: M365_USERS empty")
|
|
||||||
}
|
|
||||||
cfg.M365 = &M365Credentials{Tenant: tenant, ClientID: cid, ClientSecret: sec, Users: userList}
|
|
||||||
default:
|
default:
|
||||||
return CLIConfig{}, 2, fmt.Errorf("unknown source %q", s)
|
return CLIConfig{}, 2, fmt.Errorf("unknown source %q", s)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
out.Sync = cfg
|
cli.Sync = cfg
|
||||||
return out, 0, nil
|
return cli, 0, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// Main is the CLI entry: returns process exit code.
|
// Main is the CLI entry: returns process exit code.
|
||||||
func Main(args []string) int {
|
func Main(args []string) int {
|
||||||
cfg, code, err := ParseCLI(args)
|
cli, code, err := ParseCLI(args)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintln(os.Stderr, "mail/sync:", err)
|
fmt.Fprintln(os.Stderr, "mail/sync:", err)
|
||||||
return code
|
return code
|
||||||
}
|
}
|
||||||
if cfg.Help {
|
if cli.Help {
|
||||||
fmt.Fprintln(os.Stderr, "usage: bin/mail/sync.go [--source onlyoffice,gmail,m365] [--query GMAIL_Q] [--limit N] [--offset N] [--workers N] [--force] [--dry-run]")
|
fmt.Fprintln(os.Stderr, "usage: bin/mail/sync.go [--source onlyoffice,gmail] [--query GMAIL_Q] [--limit N] [--offset N] [--workers N] [--force] [--dry-run]")
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
ctx, cancel := context.WithTimeout(context.Background(), 6*time.Hour)
|
ctx, cancel := context.WithTimeout(context.Background(), 6*time.Hour)
|
||||||
defer cancel()
|
defer cancel()
|
||||||
start := time.Now()
|
start := time.Now()
|
||||||
stats, err := Run(ctx, cfg.Sync)
|
stats, err := Run(ctx, cli.Sync)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintln(os.Stderr, "mail/sync:", err)
|
fmt.Fprintln(os.Stderr, "mail/sync:", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
if cfg.Sync.DryRun {
|
if cli.Sync.DryRun {
|
||||||
fmt.Printf("mail/sync: dry-run checked=%d (no writes)\n", stats.Checked)
|
fmt.Printf("mail/sync: dry-run checked=%d (no writes)\n", stats.Checked)
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
@@ -182,13 +135,13 @@ func readEnv(path string) map[string]string {
|
|||||||
k, v, _ := strings.Cut(line, "=")
|
k, v, _ := strings.Cut(line, "=")
|
||||||
out[strings.TrimSpace(k)] = strings.Trim(strings.TrimSpace(v), "\"'")
|
out[strings.TrimSpace(k)] = strings.Trim(strings.TrimSpace(v), "\"'")
|
||||||
}
|
}
|
||||||
|
// env overrides file
|
||||||
for _, kv := range os.Environ() {
|
for _, kv := range os.Environ() {
|
||||||
k, v, ok := strings.Cut(kv, "=")
|
k, v, ok := strings.Cut(kv, "=")
|
||||||
if !ok {
|
if !ok {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if strings.HasPrefix(k, "ONLYOFFICE_") || strings.HasPrefix(k, "OO_") ||
|
if strings.HasPrefix(k, "ONLYOFFICE_") || strings.HasPrefix(k, "OO_") {
|
||||||
strings.HasPrefix(k, "M365_") || strings.HasPrefix(k, "MS_") {
|
|
||||||
out[k] = v
|
out[k] = v
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,382 +0,0 @@
|
|||||||
package sync
|
|
||||||
|
|
||||||
import (
|
|
||||||
"context"
|
|
||||||
"encoding/json"
|
|
||||||
"errors"
|
|
||||||
"fmt"
|
|
||||||
"io"
|
|
||||||
"net/http"
|
|
||||||
"net/url"
|
|
||||||
"os"
|
|
||||||
"path/filepath"
|
|
||||||
"strings"
|
|
||||||
"time"
|
|
||||||
)
|
|
||||||
|
|
||||||
// M365Credentials holds a Microsoft Graph app registration with the Mail.Read
|
|
||||||
// application permission. Client credentials are read from env/.env, never
|
|
||||||
// committed.
|
|
||||||
type M365Credentials struct {
|
|
||||||
Tenant string
|
|
||||||
ClientID string
|
|
||||||
ClientSecret string
|
|
||||||
Users []string // mailbox addresses to sync, e.g. info@example.com
|
|
||||||
}
|
|
||||||
|
|
||||||
// m365Token is the cached access token with expiry.
|
|
||||||
type m365Token struct {
|
|
||||||
AccessToken string
|
|
||||||
Expiry time.Time
|
|
||||||
}
|
|
||||||
|
|
||||||
// M365Client talks to the Microsoft Graph API using the client-credentials
|
|
||||||
// flow (app registration with Mail.Read application permission). GET-only:
|
|
||||||
// messages are never marked as read or deleted.
|
|
||||||
type M365Client struct {
|
|
||||||
creds M365Credentials
|
|
||||||
base string // graph base URL; default https://graph.microsoft.com
|
|
||||||
tokenEndpoint string // login endpoint; default https://login.microsoftonline.com
|
|
||||||
client *http.Client
|
|
||||||
mu chan struct{}
|
|
||||||
token *m365Token
|
|
||||||
}
|
|
||||||
|
|
||||||
func NewM365Client(creds M365Credentials) (*M365Client, error) {
|
|
||||||
if creds.Tenant == "" || creds.ClientID == "" || creds.ClientSecret == "" {
|
|
||||||
return nil, errors.New("m365 needs tenant, client id and client secret")
|
|
||||||
}
|
|
||||||
c := &M365Client{
|
|
||||||
creds: creds,
|
|
||||||
base: "https://graph.microsoft.com",
|
|
||||||
client: &http.Client{Timeout: 90 * time.Second},
|
|
||||||
mu: make(chan struct{}, 1),
|
|
||||||
}
|
|
||||||
c.mu <- struct{}{}
|
|
||||||
return c, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// accessToken returns a fresh bearer token, refreshing via the Azure AD token
|
|
||||||
// endpoint when the cached one is missing or about to expire (within 2 min).
|
|
||||||
func (c *M365Client) accessToken(ctx context.Context) (string, error) {
|
|
||||||
select {
|
|
||||||
case <-c.mu:
|
|
||||||
case <-ctx.Done():
|
|
||||||
return "", ctx.Err()
|
|
||||||
}
|
|
||||||
defer func() { c.mu <- struct{}{} }()
|
|
||||||
if c.token != nil && c.token.AccessToken != "" && time.Now().Before(c.token.Expiry.Add(-2*time.Minute)) {
|
|
||||||
return c.token.AccessToken, nil
|
|
||||||
}
|
|
||||||
return c.refreshLocked(ctx)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (c *M365Client) refreshLocked(ctx context.Context) (string, error) {
|
|
||||||
form := url.Values{}
|
|
||||||
form.Set("grant_type", "client_credentials")
|
|
||||||
form.Set("client_id", c.creds.ClientID)
|
|
||||||
form.Set("client_secret", c.creds.ClientSecret)
|
|
||||||
form.Set("scope", "https://graph.microsoft.com/.default")
|
|
||||||
|
|
||||||
endpoint := c.tokenEndpoint
|
|
||||||
if endpoint == "" {
|
|
||||||
endpoint = fmt.Sprintf("https://login.microsoftonline.com/%s/oauth2/v2.0/token", c.creds.Tenant)
|
|
||||||
}
|
|
||||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, endpoint, strings.NewReader(form.Encode()))
|
|
||||||
if err != nil {
|
|
||||||
return "", err
|
|
||||||
}
|
|
||||||
req.Header.Set("Content-Type", "application/x-www-form-urlencoded")
|
|
||||||
resp, err := c.client.Do(req)
|
|
||||||
if err != nil {
|
|
||||||
return "", fmt.Errorf("m365 token: %w", err)
|
|
||||||
}
|
|
||||||
defer resp.Body.Close()
|
|
||||||
body, _ := io.ReadAll(io.LimitReader(resp.Body, 1<<20))
|
|
||||||
if resp.StatusCode != http.StatusOK {
|
|
||||||
var e struct {
|
|
||||||
Error string `json:"error"`
|
|
||||||
Desc string `json:"error_description"`
|
|
||||||
}
|
|
||||||
_ = json.Unmarshal(body, &e)
|
|
||||||
return "", fmt.Errorf("m365 token status %d: %s (%s)", resp.StatusCode, e.Error, truncate(e.Desc, 200))
|
|
||||||
}
|
|
||||||
var out struct {
|
|
||||||
AccessToken string `json:"access_token"`
|
|
||||||
ExpiresIn int64 `json:"expires_in"`
|
|
||||||
}
|
|
||||||
if err := json.Unmarshal(body, &out); err != nil {
|
|
||||||
return "", fmt.Errorf("m365 token parse: %w", err)
|
|
||||||
}
|
|
||||||
c.token = &m365Token{
|
|
||||||
AccessToken: out.AccessToken,
|
|
||||||
Expiry: time.Now().Add(time.Duration(out.ExpiresIn) * time.Second),
|
|
||||||
}
|
|
||||||
return out.AccessToken, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// deltaPage is one response page of the Graph delta query.
|
|
||||||
type deltaPage struct {
|
|
||||||
Value []struct {
|
|
||||||
ID string `json:"id"`
|
|
||||||
RemovedReason string `json:"@odata.removedReason"`
|
|
||||||
} `json:"value"`
|
|
||||||
NextLink string `json:"@odata.nextLink"`
|
|
||||||
DeltaLink string `json:"@odata.deltaLink"`
|
|
||||||
}
|
|
||||||
|
|
||||||
// ListDeltaIDs walks the inbox delta query and returns live message ids since
|
|
||||||
// the previous deltaLink (or the full inbox when deltaLink is empty). Returns
|
|
||||||
// the new deltaLink for the next run. GET-only; nothing is mutated server-side.
|
|
||||||
func (c *M365Client) ListDeltaIDs(ctx context.Context, mailbox, deltaLink string, limit int) ([]string, string, error) {
|
|
||||||
var (
|
|
||||||
ids []string
|
|
||||||
url string
|
|
||||||
link = deltaLink
|
|
||||||
)
|
|
||||||
if link == "" {
|
|
||||||
url = fmt.Sprintf("/v1.0/users/%s/mailFolders/inbox/messages/delta", pathEscape(mailbox))
|
|
||||||
} else {
|
|
||||||
url = link
|
|
||||||
}
|
|
||||||
for url != "" {
|
|
||||||
var page deltaPage
|
|
||||||
if err := c.getJSON(ctx, url, &page); err != nil {
|
|
||||||
return ids, link, err
|
|
||||||
}
|
|
||||||
for _, m := range page.Value {
|
|
||||||
if m.RemovedReason != "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
if m.ID == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
ids = append(ids, m.ID)
|
|
||||||
if limit > 0 && len(ids) >= limit {
|
|
||||||
if page.DeltaLink != "" {
|
|
||||||
link = page.DeltaLink
|
|
||||||
}
|
|
||||||
return ids, link, nil
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if page.DeltaLink != "" {
|
|
||||||
link = page.DeltaLink
|
|
||||||
url = ""
|
|
||||||
break
|
|
||||||
}
|
|
||||||
url = page.NextLink
|
|
||||||
}
|
|
||||||
return ids, link, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// GetMessage fetches a single message by id and normalizes it to the Message
|
|
||||||
// contract. GET-only.
|
|
||||||
func (c *M365Client) GetMessage(ctx context.Context, mailbox, id string) (*Message, error) {
|
|
||||||
path := fmt.Sprintf("/v1.0/users/%s/messages/%s?$expand=attachments($select=id,name,contentType,size,isInline)",
|
|
||||||
pathEscape(mailbox), pathEscape(id))
|
|
||||||
var raw struct {
|
|
||||||
ID string `json:"id"`
|
|
||||||
Subject string `json:"subject"`
|
|
||||||
From m365Recipient `json:"from"`
|
|
||||||
ToRecipients []m365Recipient `json:"toRecipients"`
|
|
||||||
CCRecipients []m365Recipient `json:"ccRecipients"`
|
|
||||||
BCCRecipients []m365Recipient `json:"bccRecipients"`
|
|
||||||
ReceivedDateTime string `json:"receivedDateTime"`
|
|
||||||
Body m365Body `json:"body"`
|
|
||||||
BodyPreview string `json:"bodyPreview"`
|
|
||||||
InternetMessageID string `json:"internetMessageId"`
|
|
||||||
Attachments []m365Attachment `json:"attachments"`
|
|
||||||
}
|
|
||||||
if err := c.getJSON(ctx, path, &raw); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
m := &Message{
|
|
||||||
Source: "m365",
|
|
||||||
ID: raw.ID,
|
|
||||||
Folder: "m365",
|
|
||||||
Subject: raw.Subject,
|
|
||||||
From: formatRecipient(raw.From),
|
|
||||||
To: formatRecipients(raw.ToRecipients),
|
|
||||||
CC: formatRecipients(raw.CCRecipients),
|
|
||||||
BCC: formatRecipients(raw.BCCRecipients),
|
|
||||||
MimeMessageID: raw.InternetMessageID,
|
|
||||||
}
|
|
||||||
if t, err := time.Parse(time.RFC3339, raw.ReceivedDateTime); err == nil {
|
|
||||||
m.ReceivedAt = t
|
|
||||||
}
|
|
||||||
switch strings.ToLower(raw.Body.ContentType) {
|
|
||||||
case "html":
|
|
||||||
m.HTMLBody = raw.Body.Content
|
|
||||||
if raw.BodyPreview != "" {
|
|
||||||
m.TextBody = raw.BodyPreview
|
|
||||||
}
|
|
||||||
default:
|
|
||||||
m.TextBody = raw.Body.Content
|
|
||||||
if raw.BodyPreview != "" {
|
|
||||||
m.HTMLBody = raw.BodyPreview
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for _, a := range raw.Attachments {
|
|
||||||
if a.IsInline || a.ID == "" || a.Name == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
m.Attachments = append(m.Attachments, Attachment{
|
|
||||||
FileID: a.ID,
|
|
||||||
FileName: a.Name,
|
|
||||||
StoredName: a.Name,
|
|
||||||
Size: a.Size,
|
|
||||||
ContentType: a.ContentType,
|
|
||||||
})
|
|
||||||
}
|
|
||||||
m.HasAttachments = len(m.Attachments) > 0
|
|
||||||
return m, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
type m365Recipient struct {
|
|
||||||
EmailAddress struct {
|
|
||||||
Name string `json:"name"`
|
|
||||||
Address string `json:"address"`
|
|
||||||
} `json:"emailAddress"`
|
|
||||||
}
|
|
||||||
|
|
||||||
type m365Body struct {
|
|
||||||
ContentType string `json:"contentType"`
|
|
||||||
Content string `json:"content"`
|
|
||||||
}
|
|
||||||
|
|
||||||
type m365Attachment struct {
|
|
||||||
ID string `json:"id"`
|
|
||||||
Name string `json:"name"`
|
|
||||||
ContentType string `json:"contentType"`
|
|
||||||
Size int64 `json:"size"`
|
|
||||||
IsInline bool `json:"isInline"`
|
|
||||||
ContentID string `json:"contentId"`
|
|
||||||
}
|
|
||||||
|
|
||||||
func formatRecipients(rs []m365Recipient) string {
|
|
||||||
var parts []string
|
|
||||||
for _, r := range rs {
|
|
||||||
if s := formatRecipient(r); s != "" {
|
|
||||||
parts = append(parts, s)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return strings.Join(parts, ", ")
|
|
||||||
}
|
|
||||||
|
|
||||||
func formatRecipient(r m365Recipient) string { a := r.EmailAddress.Address
|
|
||||||
n := r.EmailAddress.Name
|
|
||||||
switch {
|
|
||||||
case n == "" || n == a:
|
|
||||||
return a
|
|
||||||
case a == "":
|
|
||||||
return n
|
|
||||||
default:
|
|
||||||
return fmt.Sprintf("%s <%s>", n, a)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// DownloadAttachment fetches an attachment's raw bytes via the /$value stream.
|
|
||||||
func (c *M365Client) DownloadAttachment(ctx context.Context, mailbox, msgID, attID string) ([]byte, error) {
|
|
||||||
path := fmt.Sprintf("/v1.0/users/%s/messages/%s/attachments/%s/$value",
|
|
||||||
pathEscape(mailbox), pathEscape(msgID), pathEscape(attID))
|
|
||||||
tok, err := c.accessToken(ctx)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, c.base+path, nil)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
req.Header.Set("Authorization", "Bearer "+tok)
|
|
||||||
resp, err := c.client.Do(req)
|
|
||||||
if err != nil {
|
|
||||||
return nil, fmt.Errorf("m365 attachment: %w", err)
|
|
||||||
}
|
|
||||||
defer resp.Body.Close()
|
|
||||||
body, _ := io.ReadAll(io.LimitReader(resp.Body, 256<<20))
|
|
||||||
if resp.StatusCode != http.StatusOK {
|
|
||||||
return nil, fmt.Errorf("m365 attachment %s: status %d: %s", attID, resp.StatusCode, truncate(string(body), 300))
|
|
||||||
}
|
|
||||||
return body, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (c *M365Client) getJSON(ctx context.Context, path string, out any) error {
|
|
||||||
tok, err := c.accessToken(ctx)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
u := path
|
|
||||||
if !strings.HasPrefix(u, "http") {
|
|
||||||
u = c.base + u
|
|
||||||
}
|
|
||||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, u, nil)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
req.Header.Set("Authorization", "Bearer "+tok)
|
|
||||||
resp, err := c.client.Do(req)
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("m365 %s: %w", path, err)
|
|
||||||
}
|
|
||||||
defer resp.Body.Close()
|
|
||||||
body, _ := io.ReadAll(io.LimitReader(resp.Body, 16<<20))
|
|
||||||
if resp.StatusCode != http.StatusOK {
|
|
||||||
return fmt.Errorf("m365 %s: status %d: %s", path, resp.StatusCode, truncate(string(body), 300))
|
|
||||||
}
|
|
||||||
if out != nil {
|
|
||||||
return json.Unmarshal(body, out)
|
|
||||||
}
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// m365Source adapts a mailbox to the Source worker-pool contract. Each mailbox
|
|
||||||
// gets its own folder under var/mail/m365/<localpart>/ and a delta state file.
|
|
||||||
type m365Source struct {
|
|
||||||
c *M365Client
|
|
||||||
mailbox string
|
|
||||||
localpart string
|
|
||||||
stateDir string
|
|
||||||
pending string // delta link to persist on Commit()
|
|
||||||
hasPending bool
|
|
||||||
}
|
|
||||||
|
|
||||||
func (s *m365Source) Folder() string { return filepath.Join("m365", s.localpart) }
|
|
||||||
|
|
||||||
func (s *m365Source) ListIDs(ctx context.Context, limit int, cursor string) ([]string, string, error) {
|
|
||||||
link, _ := os.ReadFile(filepath.Join(s.stateDir, s.localpart+".deltalink"))
|
|
||||||
ids, newLink, err := s.c.ListDeltaIDs(ctx, s.mailbox, strings.TrimSpace(string(link)), limit)
|
|
||||||
if err != nil {
|
|
||||||
return nil, "", err
|
|
||||||
}
|
|
||||||
// Buffer the new delta link; persist it only in Commit() after the full
|
|
||||||
// batch downloaded, so a failed run stays retryable without gaps.
|
|
||||||
if newLink != "" {
|
|
||||||
s.pending = newLink
|
|
||||||
s.hasPending = true
|
|
||||||
}
|
|
||||||
return ids, "", nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// Commit persists the buffered delta link. Called by the sync runner only when
|
|
||||||
// every listed message downloaded successfully.
|
|
||||||
func (s *m365Source) Commit() error {
|
|
||||||
if !s.hasPending || s.pending == "" {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
if err := os.MkdirAll(s.stateDir, 0o755); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
return os.WriteFile(filepath.Join(s.stateDir, s.localpart+".deltalink"), []byte(s.pending), 0o644)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (s *m365Source) Get(ctx context.Context, id string) (*Message, error) {
|
|
||||||
return s.c.GetMessage(ctx, s.mailbox, id)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (s *m365Source) DownloadAttachment(ctx context.Context, msg *Message, att Attachment) ([]byte, error) {
|
|
||||||
return s.c.DownloadAttachment(ctx, s.mailbox, msg.ID, att.FileID)
|
|
||||||
}
|
|
||||||
|
|
||||||
func pathEscape(s string) string {
|
|
||||||
return url.PathEscape(s)
|
|
||||||
}
|
|
||||||
@@ -1,188 +0,0 @@
|
|||||||
package sync
|
|
||||||
|
|
||||||
import (
|
|
||||||
"context"
|
|
||||||
"net/http"
|
|
||||||
"net/http/httptest"
|
|
||||||
"os"
|
|
||||||
"path/filepath"
|
|
||||||
"testing"
|
|
||||||
)
|
|
||||||
|
|
||||||
// newM365TestClient serves the Graph delta + message endpoints against a fake
|
|
||||||
// token endpoint, so unit tests never touch the network.
|
|
||||||
func newM365TestClient(t *testing.T, graph http.Handler) *M365Client {
|
|
||||||
t.Helper()
|
|
||||||
tok := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
||||||
w.Header().Set("Content-Type", "application/json")
|
|
||||||
w.Write([]byte(`{"access_token":"test-token","expires_in":3600}`))
|
|
||||||
}))
|
|
||||||
gr := httptest.NewServer(graph)
|
|
||||||
t.Cleanup(func() {
|
|
||||||
tok.Close()
|
|
||||||
gr.Close()
|
|
||||||
})
|
|
||||||
client, err := NewM365Client(M365Credentials{Tenant: "t.onmicrosoft.com", ClientID: "c", ClientSecret: "s"})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
client.base = gr.URL
|
|
||||||
client.tokenEndpoint = tok.URL
|
|
||||||
return client
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestM365AccessToken(t *testing.T) {
|
|
||||||
c := newM365TestClient(t, http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {}))
|
|
||||||
got, err := c.accessToken(context.Background())
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("accessToken: %v", err)
|
|
||||||
}
|
|
||||||
if got != "test-token" {
|
|
||||||
t.Errorf("token = %q", got)
|
|
||||||
}
|
|
||||||
// Second call must reuse the cached token (no token request).
|
|
||||||
again, err := c.accessToken(context.Background())
|
|
||||||
if err != nil || again != "test-token" {
|
|
||||||
t.Fatalf("cached token: %q, %v", again, err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestM365DeltaSkipsTombstones(t *testing.T) {
|
|
||||||
first := true
|
|
||||||
graph := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
||||||
w.Header().Set("Content-Type", "application/json")
|
|
||||||
if first {
|
|
||||||
first = false
|
|
||||||
w.Write([]byte(`{"value":[
|
|
||||||
{"id":"m1"},
|
|
||||||
{"id":"m2","@odata.removedReason":"deleted"},
|
|
||||||
{"id":"m3"}
|
|
||||||
],"@odata.deltaLink":"` + deltaNext + `"}`))
|
|
||||||
return
|
|
||||||
}
|
|
||||||
// Second call must use the stored deltaLink (points at this server).
|
|
||||||
if r.URL.Path != "/v1.0/delta-next" {
|
|
||||||
w.WriteHeader(500)
|
|
||||||
w.Write([]byte(`{"error":{"message":"unexpected path"}}`))
|
|
||||||
return
|
|
||||||
}
|
|
||||||
w.Write([]byte(`{"value":[{"id":"m4"}],"@odata.deltaLink":"` + deltaFinal + `"}`))
|
|
||||||
})
|
|
||||||
c := newM365TestClient(t, graph)
|
|
||||||
deltaNext = c.base + "/v1.0/delta-next"
|
|
||||||
deltaFinal = c.base + "/v1.0/delta-final"
|
|
||||||
|
|
||||||
ids, link, err := c.ListDeltaIDs(context.Background(), "a@x.de", "", 0)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("delta: %v", err)
|
|
||||||
}
|
|
||||||
if len(ids) != 2 || ids[0] != "m1" || ids[1] != "m3" {
|
|
||||||
t.Errorf("ids = %v", ids)
|
|
||||||
}
|
|
||||||
if link == "" {
|
|
||||||
t.Error("expected new deltaLink")
|
|
||||||
}
|
|
||||||
// Incremental: pass the deltaLink, get only the new id.
|
|
||||||
ids2, link2, err := c.ListDeltaIDs(context.Background(), "a@x.de", link, 0)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("delta incremental: %v", err)
|
|
||||||
}
|
|
||||||
if len(ids2) != 1 || ids2[0] != "m4" {
|
|
||||||
t.Errorf("ids2 = %v", ids2)
|
|
||||||
}
|
|
||||||
if link2 == "" {
|
|
||||||
t.Error("expected updated deltaLink")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestM365GetMessageNormalizes(t *testing.T) {
|
|
||||||
graph := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
||||||
w.Header().Set("Content-Type", "application/json")
|
|
||||||
if pathLast(r.URL.Path) == "messages" {
|
|
||||||
w.Write([]byte(`{"value":[{"id":"m1"}]}`))
|
|
||||||
return
|
|
||||||
}
|
|
||||||
w.Write([]byte(`{
|
|
||||||
"id":"m1",
|
|
||||||
"subject":"Hallo",
|
|
||||||
"from":{"emailAddress":{"name":"Max","address":"max@x.de"}},
|
|
||||||
"toRecipients":[{"emailAddress":{"address":"a@x.de"}}],
|
|
||||||
"receivedDateTime":"2026-08-14T08:15:00Z",
|
|
||||||
"body":{"contentType":"html","content":"<p>body</p>"},
|
|
||||||
"bodyPreview":"body",
|
|
||||||
"internetMessageId":"<mid@x.de>",
|
|
||||||
"attachments":[
|
|
||||||
{"id":"att1","name":"doc.pdf","contentType":"application/pdf","size":10},
|
|
||||||
{"id":"img1","name":"logo.png","contentType":"image/png","isInline":true}
|
|
||||||
]
|
|
||||||
}`))
|
|
||||||
})
|
|
||||||
c := newM365TestClient(t, graph)
|
|
||||||
m, err := c.GetMessage(context.Background(), "a@x.de", "m1")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("GetMessage: %v", err)
|
|
||||||
}
|
|
||||||
if m.ID != "m1" || m.Subject != "Hallo" || m.From != "Max <max@x.de>" || m.To != "a@x.de" {
|
|
||||||
t.Errorf("headers mismatch: %+v", m)
|
|
||||||
}
|
|
||||||
if m.HTMLBody != "<p>body</p>" {
|
|
||||||
t.Errorf("html = %q", m.HTMLBody)
|
|
||||||
}
|
|
||||||
if m.ReceivedAt.IsZero() {
|
|
||||||
t.Error("receivedAt zero")
|
|
||||||
}
|
|
||||||
if m.MimeMessageID != "<mid@x.de>" {
|
|
||||||
t.Errorf("mime id = %q", m.MimeMessageID)
|
|
||||||
}
|
|
||||||
if len(m.Attachments) != 1 || m.Attachments[0].FileName != "doc.pdf" || m.Attachments[0].FileID != "att1" {
|
|
||||||
t.Errorf("atts = %+v", m.Attachments)
|
|
||||||
}
|
|
||||||
if !m.HasAttachments {
|
|
||||||
t.Error("expected hasAttachments")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestM365SourceDeltaState(t *testing.T) {
|
|
||||||
graph := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
||||||
w.Header().Set("Content-Type", "application/json")
|
|
||||||
w.Write([]byte(`{"value":[{"id":"m1"}],"@odata.deltaLink":"` + deltaNext + `"}`))
|
|
||||||
})
|
|
||||||
c := newM365TestClient(t, graph)
|
|
||||||
deltaNext = c.base + "/v1.0/delta-next"
|
|
||||||
stateDir := filepath.Join(t.TempDir(), ".m365")
|
|
||||||
s := &m365Source{c: c, mailbox: "info@x.de", localpart: "info", stateDir: stateDir}
|
|
||||||
|
|
||||||
if s.Folder() != "m365/info" {
|
|
||||||
t.Errorf("folder = %q", s.Folder())
|
|
||||||
}
|
|
||||||
ids, _, err := s.ListIDs(context.Background(), 0, "")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("ListIDs: %v", err)
|
|
||||||
}
|
|
||||||
if len(ids) != 1 || ids[0] != "m1" {
|
|
||||||
t.Errorf("ids = %v", ids)
|
|
||||||
}
|
|
||||||
if err := s.Commit(); err != nil {
|
|
||||||
t.Fatalf("Commit: %v", err)
|
|
||||||
}
|
|
||||||
data, err := os.ReadFile(filepath.Join(stateDir, "info.deltalink"))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("read delta state: %v", err)
|
|
||||||
}
|
|
||||||
if string(data) != deltaNext {
|
|
||||||
t.Errorf("delta state = %q, want %q", string(data), deltaNext)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func pathLast(p string) string {
|
|
||||||
for i := len(p) - 1; i >= 0; i-- {
|
|
||||||
if p[i] == '/' {
|
|
||||||
return p[i+1:]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return p
|
|
||||||
}
|
|
||||||
|
|
||||||
// deltaNext/deltaFinal are set per-test from the fake graph server URL so
|
|
||||||
// deltaLink values always point back at the fake (never the real Graph).
|
|
||||||
var deltaNext, deltaFinal string
|
|
||||||
+1
-43
@@ -106,7 +106,6 @@ func Retry(ctx context.Context, policy RetryPolicy, fn func() error) error {
|
|||||||
type SyncConfig struct {
|
type SyncConfig struct {
|
||||||
OO *OOConfig // OnlyOffice source (optional)
|
OO *OOConfig // OnlyOffice source (optional)
|
||||||
Gmail *GmailCredentials // Gmail source (optional)
|
Gmail *GmailCredentials // Gmail source (optional)
|
||||||
M365 *M365Credentials // Microsoft 365 Graph source (optional)
|
|
||||||
Out string // var/mail root; default <repo>/var/mail
|
Out string // var/mail root; default <repo>/var/mail
|
||||||
Workers int // concurrency; default 4
|
Workers int // concurrency; default 4
|
||||||
Limit int // max messages per source (0 = all)
|
Limit int // max messages per source (0 = all)
|
||||||
@@ -134,14 +133,6 @@ type Source interface {
|
|||||||
Folder() string
|
Folder() string
|
||||||
}
|
}
|
||||||
|
|
||||||
// Committer is an optional Source capability: Commit is called after all listed
|
|
||||||
// ids have been downloaded successfully. Sources that only advance durable state
|
|
||||||
// on success (e.g. a Graph delta link) implement this so a killed or failed run
|
|
||||||
// stays retryable without gaps.
|
|
||||||
type Committer interface {
|
|
||||||
Commit() error
|
|
||||||
}
|
|
||||||
|
|
||||||
type ooSource struct {
|
type ooSource struct {
|
||||||
c *OOClient
|
c *OOClient
|
||||||
page int
|
page int
|
||||||
@@ -231,22 +222,8 @@ func Run(ctx context.Context, cfg SyncConfig) (*SyncStats, error) {
|
|||||||
}
|
}
|
||||||
sources = append(sources, &gmailSource{c: gm, query: cfg.Query})
|
sources = append(sources, &gmailSource{c: gm, query: cfg.Query})
|
||||||
}
|
}
|
||||||
if cfg.M365 != nil {
|
|
||||||
stateDir := filepath.Join(cfg.Out, ".m365")
|
|
||||||
for _, mb := range cfg.M365.Users {
|
|
||||||
if !strings.Contains(mb, "@") {
|
|
||||||
return nil, fmt.Errorf("m365 user %q is not an email address", mb)
|
|
||||||
}
|
|
||||||
c, err := NewM365Client(*cfg.M365)
|
|
||||||
if err != nil {
|
|
||||||
return nil, fmt.Errorf("m365 init for %s: %w", mb, err)
|
|
||||||
}
|
|
||||||
local := strings.SplitN(mb, "@", 2)[0]
|
|
||||||
sources = append(sources, &m365Source{c: c, mailbox: mb, localpart: strings.ToLower(local), stateDir: stateDir})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if len(sources) == 0 {
|
if len(sources) == 0 {
|
||||||
return nil, errors.New("sync: no source configured (need OO, Gmail, M365, or a combination)")
|
return nil, errors.New("sync: no source configured (need OO, Gmail, or both)")
|
||||||
}
|
}
|
||||||
|
|
||||||
stats := &SyncStats{}
|
stats := &SyncStats{}
|
||||||
@@ -319,25 +296,6 @@ func Run(ctx context.Context, cfg SyncConfig) (*SyncStats, error) {
|
|||||||
close(jobsCh)
|
close(jobsCh)
|
||||||
wg.Wait()
|
wg.Wait()
|
||||||
|
|
||||||
// Only advance durable source state (e.g. delta links) when everything
|
|
||||||
// downloaded. A killed or failed run must be retryable without gaps.
|
|
||||||
if len(failures) == 0 {
|
|
||||||
seen := map[Source]bool{}
|
|
||||||
for _, j := range jobs {
|
|
||||||
if seen[j.src] {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
seen[j.src] = true
|
|
||||||
if c, ok := j.src.(Committer); ok {
|
|
||||||
if err := c.Commit(); err != nil {
|
|
||||||
mu.Lock()
|
|
||||||
failures = append(failures, j.src.Folder()+"/commit: "+err.Error())
|
|
||||||
mu.Unlock()
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(failures) > 0 {
|
if len(failures) > 0 {
|
||||||
fmt.Fprintf(os.Stderr, "sync: %d failures:\n %s\n", len(failures), strings.Join(failures, "\n "))
|
fmt.Fprintf(os.Stderr, "sync: %d failures:\n %s\n", len(failures), strings.Join(failures, "\n "))
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,3 +0,0 @@
|
|||||||
// Commands in this directory are shebang mains (import.go), tagged so
|
|
||||||
// `go build ./bin/markdown` does not see two mains.
|
|
||||||
package main
|
|
||||||
@@ -1,85 +0,0 @@
|
|||||||
//usr/bin/env go run "$0" "$@"; exit
|
|
||||||
//
|
|
||||||
// bin/markdown/import.go - split markdown into leafs (H2 boundaries).
|
|
||||||
//
|
|
||||||
// ./bin/markdown/import.go [dir]
|
|
||||||
// ./bin/markdown/import.go --files a.md,b.md --json
|
|
||||||
//
|
|
||||||
// Conversion only. Brain write is bin/brain/index.go.
|
|
||||||
// Python bin/md/import remains as a fallback.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"fmt"
|
|
||||||
"os"
|
|
||||||
"strings"
|
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
"github.com/eSlider/2dph/internal/mdleaves"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(run(os.Args[1:]))
|
|
||||||
}
|
|
||||||
|
|
||||||
func run(args []string) int {
|
|
||||||
c, err := mdleaves.ParseArgs(args)
|
|
||||||
if err != nil {
|
|
||||||
return cliparse.Fail(err)
|
|
||||||
}
|
|
||||||
jsonOut := c.JSONOut
|
|
||||||
files := c.Files
|
|
||||||
root := c.Root
|
|
||||||
|
|
||||||
var paths []string
|
|
||||||
if files != "" {
|
|
||||||
for _, f := range strings.Split(files, ",") {
|
|
||||||
f = strings.TrimSpace(f)
|
|
||||||
if f != "" {
|
|
||||||
paths = append(paths, f)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
st, err := os.Stat(root)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "md/import: no such path %s\n", root)
|
|
||||||
return 2
|
|
||||||
}
|
|
||||||
if !st.IsDir() {
|
|
||||||
paths = []string{root}
|
|
||||||
} else {
|
|
||||||
var err error
|
|
||||||
paths, err = mdleaves.WalkMarkdown(root)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "md/import: %v\n", err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if len(paths) == 0 {
|
|
||||||
fmt.Fprintln(os.Stderr, "md/import: no markdown files")
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
|
|
||||||
var all []mdleaves.Leaf
|
|
||||||
for _, p := range paths {
|
|
||||||
raw, err := os.ReadFile(p)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "md/import: %s: %v\n", p, err)
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
all = append(all, mdleaves.ToAll(string(raw), p, "")...)
|
|
||||||
}
|
|
||||||
if jsonOut {
|
|
||||||
s, err := mdleaves.EncodeJSON(all)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintln(os.Stderr, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
fmt.Print(s)
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
fmt.Print(mdleaves.EncodeYAML(all))
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
@@ -1,2 +0,0 @@
|
|||||||
// Commands in this directory are shebang mains (query.go).
|
|
||||||
package main
|
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=postgres_query "$0" "$@"; exit
|
|
||||||
//go:build postgres_query
|
|
||||||
//
|
|
||||||
// bin/postgres/query.go - read-only Postgres as YAML.
|
|
||||||
//
|
|
||||||
// ./bin/postgres/query.go --profile onlyoffice -c 'SELECT 1'
|
|
||||||
//
|
|
||||||
// Profiles: $HOME/.config/brain/db-profiles.yml (credentials stay out of git).
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/cmdbin"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(cmdbin.ExecFile("bin/db/psql-yq", os.Args[1:]))
|
|
||||||
}
|
|
||||||
@@ -1,61 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=qa_stats "$0" "$@"; exit
|
|
||||||
//go:build qa_stats
|
|
||||||
//
|
|
||||||
// bin/qa/stats.go - DuckDB quantiles over a JSON number array or JSONL count.
|
|
||||||
//
|
|
||||||
// ./bin/qa/stats.go <<< '[1,2,3,4,5]'
|
|
||||||
// ./bin/qa/stats.go --jsonl rows.jsonl
|
|
||||||
//
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
// DuckDB CGO needs gcc/g++ (not Zig). After eval "$(bin/cgo/zig env)":
|
|
||||||
// CC=gcc CXX=g++ CGO_CFLAGS= CGO_LDFLAGS= ./bin/qa/stats.go
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"encoding/json"
|
|
||||||
"fmt"
|
|
||||||
"io"
|
|
||||||
"os"
|
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
"github.com/eSlider/2dph/internal/duckstats"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(run(os.Args[1:]))
|
|
||||||
}
|
|
||||||
|
|
||||||
func run(args []string) int {
|
|
||||||
c, err := cliparse.ParseQAStats(args)
|
|
||||||
if err != nil {
|
|
||||||
return cliparse.Fail(err)
|
|
||||||
}
|
|
||||||
jsonl := c.JSONL
|
|
||||||
if jsonl != "" {
|
|
||||||
n, err := duckstats.CountJSONL(jsonl)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintln(os.Stderr, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
fmt.Printf("n: %d\n", n)
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
raw, err := io.ReadAll(os.Stdin)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintln(os.Stderr, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
var samples []float64
|
|
||||||
if err := json.Unmarshal(raw, &samples); err != nil {
|
|
||||||
fmt.Fprintln(os.Stderr, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
s, err := duckstats.Quantiles(samples)
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintln(os.Stderr, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
fmt.Printf("n: %d\nmin: %g\np50: %g\np95: %g\nmax: %g\navg: %g\n",
|
|
||||||
s.N, s.Min, s.P50, s.P95, s.Max, s.Avg)
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
@@ -1,73 +0,0 @@
|
|||||||
//usr/bin/env go run -tags=reasoner_bakeoff "$0" "$@"; exit
|
|
||||||
//go:build reasoner_bakeoff
|
|
||||||
//
|
|
||||||
// bin/reasoner/bakeoff.go - CPU tool-call bake-off against an OpenAI-compatible URL (D18).
|
|
||||||
//
|
|
||||||
// REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b ./bin/reasoner/bakeoff.go
|
|
||||||
// ./bin/reasoner/bakeoff.go --model MichelRosselli/bonsai-27b:Q1_0 --json
|
|
||||||
//
|
|
||||||
// Measures OpenAI tool_calls (search/get/audit) and RSS from Ollama /api/ps, not VRAM.
|
|
||||||
// PicoClaw is compose profile picoclaw; tool names match internal/httpapi MCP ops.
|
|
||||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
|
||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"encoding/json"
|
|
||||||
"fmt"
|
|
||||||
"os"
|
|
||||||
|
|
||||||
cliparse "github.com/eSlider/2dph/internal/cli"
|
|
||||||
"github.com/eSlider/2dph/internal/duckstats"
|
|
||||||
"github.com/eSlider/2dph/internal/reasoner"
|
|
||||||
)
|
|
||||||
|
|
||||||
func main() {
|
|
||||||
os.Exit(run(os.Args[1:]))
|
|
||||||
}
|
|
||||||
|
|
||||||
func run(args []string) int {
|
|
||||||
c, err := reasoner.ParseArgs(args)
|
|
||||||
if err != nil {
|
|
||||||
return cliparse.Fail(err)
|
|
||||||
}
|
|
||||||
base, model, jsonOut, device := c.Base, c.Model, c.JSONOut, c.Device
|
|
||||||
client := reasoner.Client{BaseURL: base, Model: model, Device: device}
|
|
||||||
rep := reasoner.Run(client)
|
|
||||||
lat := make([]float64, 0, len(rep.Prompts))
|
|
||||||
for _, p := range rep.Prompts {
|
|
||||||
lat = append(lat, float64(p.LatencyMS))
|
|
||||||
}
|
|
||||||
if st, err := duckstats.Quantiles(lat); err == nil {
|
|
||||||
rep.LatencyP50MS = st.P50
|
|
||||||
rep.LatencyP95MS = st.P95
|
|
||||||
}
|
|
||||||
raw, err := json.MarshalIndent(rep, "", " ")
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintln(os.Stderr, err)
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
if jsonOut {
|
|
||||||
fmt.Println(string(raw))
|
|
||||||
} else {
|
|
||||||
fmt.Printf("model: %s\n", rep.Model)
|
|
||||||
fmt.Printf("hf_id: %s\n", rep.HF)
|
|
||||||
fmt.Printf("device: %s\n", rep.Device)
|
|
||||||
fmt.Printf("tool_call: %d/%d\n", rep.ToolCallOK, rep.ToolCallN)
|
|
||||||
fmt.Printf("xml_leak: %d\n", rep.XMLLeak)
|
|
||||||
fmt.Printf("rss_mb: %d\n", rep.RSSMB)
|
|
||||||
fmt.Printf("vram_mb: %d\n", rep.VRAMMB)
|
|
||||||
fmt.Printf("latency_p50_ms: %g\n", rep.LatencyP50MS)
|
|
||||||
fmt.Printf("latency_p95_ms: %g\n", rep.LatencyP95MS)
|
|
||||||
for _, p := range rep.Prompts {
|
|
||||||
status := "fail"
|
|
||||||
if p.OK {
|
|
||||||
status = "ok"
|
|
||||||
}
|
|
||||||
fmt.Printf(" %s: %s wanted=%s got=%s xml=%v %dms %s\n", p.WantedTool, status, p.WantedTool, p.ToolName, p.XMLLeak, p.LatencyMS, p.Err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if rep.ToolCallN == 0 {
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
@@ -1,2 +0,0 @@
|
|||||||
// Commands in this directory are shebang mains (bakeoff.go).
|
|
||||||
package main
|
|
||||||
+13
-8
@@ -1,22 +1,27 @@
|
|||||||
//usr/bin/env go run -tags=brain_serve "$0" "$@"; exit
|
//usr/bin/env go run "$0" "$@"; exit
|
||||||
//go:build brain_serve
|
// bin/serve.go - async Go HTTP server for the 2dph brain (see bin/server).
|
||||||
//
|
//
|
||||||
// bin/serve.go — deprecated; use bin/brain/serve.go.
|
// KB_ROOT=/path/to/2dph ./bin/serve.go # serve the brain
|
||||||
|
// KB_SEARCH_CMD=... KB_WORKERS=4 KB_PORT=8630 ./bin/serve.go
|
||||||
|
//
|
||||||
|
// Shebang trick: the first line is a Go `//` comment; when executed, env runs
|
||||||
|
// `go run "$0"` so this file doubles as an executable script. The real code
|
||||||
|
// lives in the importable package (module path, never a relative import).
|
||||||
|
// NOTE: never run `gofmt -w` on this file - it rewrites `//usr/bin/env` to
|
||||||
|
// `// usr/...` and breaks the shebang.
|
||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
|
||||||
"os"
|
"os"
|
||||||
|
|
||||||
"github.com/eSlider/2dph/internal/httpapi"
|
"github.com/eSlider/2dph/bin/server"
|
||||||
)
|
)
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
fmt.Fprintln(os.Stderr, "bin/serve.go is deprecated; use bin/brain/serve.go")
|
if env := os.Getenv("KB_ROOT"); env == "" {
|
||||||
if os.Getenv("KB_ROOT") == "" {
|
|
||||||
if wd, err := os.Getwd(); err == nil {
|
if wd, err := os.Getwd(); err == nil {
|
||||||
os.Setenv("KB_ROOT", wd)
|
os.Setenv("KB_ROOT", wd)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
httpapi.Run(nil)
|
server.Run()
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,154 @@
|
|||||||
|
// Package server serves the 2dph brain over HTTP.
|
||||||
|
//
|
||||||
|
// Async by design: every request runs on its own goroutine, and CPU-heavy
|
||||||
|
// searches are serialized through a bounded worker pool (a counting
|
||||||
|
// semaphore) so N requests can't spawn N Python interpreters at once.
|
||||||
|
//
|
||||||
|
// Used by bin/serve.go which is a self-executing shebang script:
|
||||||
|
//
|
||||||
|
// ///usr/bin/env go run "$0" "$@"; exit
|
||||||
|
// package main
|
||||||
|
// import "github.com/eSlider/2dph/bin/server"
|
||||||
|
// func main() { server.Run() }
|
||||||
|
package server
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"log"
|
||||||
|
"net/http"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
type Searcher interface {
|
||||||
|
Search(ctx context.Context, query string, limit int) ([]byte, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
type Server struct {
|
||||||
|
searcher Searcher
|
||||||
|
semaphore chan struct{}
|
||||||
|
}
|
||||||
|
|
||||||
|
const defaultPort = 8630
|
||||||
|
|
||||||
|
func NewServer(searcher Searcher, workers int) http.Handler {
|
||||||
|
return &Server{
|
||||||
|
searcher: searcher,
|
||||||
|
semaphore: make(chan struct{}, workers),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *Server) ServeHTTP(w http.ResponseWriter, r *http.Request) {
|
||||||
|
switch {
|
||||||
|
case r.URL.Path == "/health":
|
||||||
|
writeJSON(w, http.StatusOK, map[string]any{"status": "ok"})
|
||||||
|
case r.URL.Path == "/search":
|
||||||
|
s.handleSearch(w, r)
|
||||||
|
default:
|
||||||
|
writeJSON(w, http.StatusNotFound, map[string]any{"error": "not found"})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *Server) handleSearch(w http.ResponseWriter, r *http.Request) {
|
||||||
|
q := strings.TrimSpace(r.URL.Query().Get("q"))
|
||||||
|
if q == "" {
|
||||||
|
writeJSON(w, http.StatusBadRequest, map[string]any{"error": "q required"})
|
||||||
|
return
|
||||||
|
}
|
||||||
|
limit := 10
|
||||||
|
if raw := r.URL.Query().Get("n"); raw != "" {
|
||||||
|
n, err := strconv.Atoi(raw)
|
||||||
|
if err != nil || n < 1 || n > 100 {
|
||||||
|
writeJSON(w, http.StatusBadRequest, map[string]any{"error": "n must be int 1..100"})
|
||||||
|
return
|
||||||
|
}
|
||||||
|
limit = n
|
||||||
|
}
|
||||||
|
|
||||||
|
// Worker pool: block until a slot frees, so burst concurrency still
|
||||||
|
// bounds memory (no unbounded python processes).
|
||||||
|
select {
|
||||||
|
case s.semaphore <- struct{}{}:
|
||||||
|
defer func() { <-s.semaphore }()
|
||||||
|
case <-r.Context().Done():
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
body, err := s.searcher.Search(r.Context(), q, limit)
|
||||||
|
if err != nil {
|
||||||
|
writeJSON(w, http.StatusGatewayTimeout, map[string]any{"error": err.Error()})
|
||||||
|
return
|
||||||
|
}
|
||||||
|
writeRaw(w, http.StatusOK, body)
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeJSON(w http.ResponseWriter, code int, obj any) {
|
||||||
|
body, _ := json.Marshal(obj)
|
||||||
|
writeRaw(w, code, body)
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeRaw(w http.ResponseWriter, code int, body []byte) {
|
||||||
|
w.Header().Set("Content-Type", "application/json")
|
||||||
|
w.Header().Set("Content-Length", strconv.Itoa(len(body)))
|
||||||
|
w.WriteHeader(code)
|
||||||
|
w.Write(body)
|
||||||
|
}
|
||||||
|
|
||||||
|
// brainSearcher shells out to bin/kb/search --json. A single python search
|
||||||
|
// is bounded and short-lived; the worker pool keeps at most N live.
|
||||||
|
type brainSearcher struct {
|
||||||
|
cmdPath string
|
||||||
|
timeout time.Duration
|
||||||
|
}
|
||||||
|
|
||||||
|
func (b *brainSearcher) Search(ctx context.Context, query string, limit int) ([]byte, error) {
|
||||||
|
ctx, cancel := context.WithTimeout(ctx, b.timeout)
|
||||||
|
defer cancel()
|
||||||
|
cmd := exec.CommandContext(ctx, b.cmdPath, "--json", "-n", strconv.Itoa(limit), query)
|
||||||
|
out, err := cmd.Output()
|
||||||
|
if err != nil {
|
||||||
|
var exitErr *exec.ExitError
|
||||||
|
if errors.As(err, &exitErr) {
|
||||||
|
return nil, errors.New("search backend failed: " + strings.TrimSpace(string(exitErr.Stderr)))
|
||||||
|
}
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Run starts the HTTP server. Reads env: KB_SEARCH_CMD (default bin/kb/search,
|
||||||
|
// relative to the repo root given by KB_ROOT), KB_WORKERS (default 4), KB_PORT
|
||||||
|
// (default 8630).
|
||||||
|
func Run() {
|
||||||
|
root := os.Getenv("KB_ROOT")
|
||||||
|
searchPath := os.Getenv("KB_SEARCH_CMD")
|
||||||
|
if searchPath == "" {
|
||||||
|
searchPath = filepath.Join(root, "bin", "kb", "search")
|
||||||
|
}
|
||||||
|
workers := 4
|
||||||
|
if raw := os.Getenv("KB_WORKERS"); raw != "" {
|
||||||
|
if n, err := strconv.Atoi(raw); err == nil && n > 0 {
|
||||||
|
workers = n
|
||||||
|
}
|
||||||
|
}
|
||||||
|
port := defaultPort
|
||||||
|
if raw := os.Getenv("KB_PORT"); raw != "" {
|
||||||
|
if n, err := strconv.Atoi(raw); err == nil && n > 0 {
|
||||||
|
port = n
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
searcher := &brainSearcher{cmdPath: searchPath, timeout: 60 * time.Second}
|
||||||
|
handler := NewServer(searcher, workers)
|
||||||
|
addr := "127.0.0.1:" + strconv.Itoa(port)
|
||||||
|
log.Printf("serve: %s (workers=%d)", addr, workers)
|
||||||
|
if err := http.ListenAndServe(addr, handler); err != nil {
|
||||||
|
log.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,13 +1,10 @@
|
|||||||
package httpapi
|
package server
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
|
||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"net/http"
|
"net/http"
|
||||||
"net/http/httptest"
|
"net/http/httptest"
|
||||||
"os"
|
|
||||||
"strings"
|
|
||||||
"sync"
|
"sync"
|
||||||
"sync/atomic"
|
"sync/atomic"
|
||||||
"testing"
|
"testing"
|
||||||
@@ -21,10 +18,10 @@ type fakeSearcher struct {
|
|||||||
calls int
|
calls int
|
||||||
active atomic.Int32
|
active atomic.Int32
|
||||||
maxSeen atomic.Int32
|
maxSeen atomic.Int32
|
||||||
callback func(q string, limit int, asOf string) ([]byte, error)
|
callback func(q string, limit int) ([]byte, error)
|
||||||
}
|
}
|
||||||
|
|
||||||
func (f *fakeSearcher) Search(ctx context.Context, query string, limit int, asOf string) ([]byte, error) {
|
func (f *fakeSearcher) Search(ctx context.Context, query string, limit int) ([]byte, error) {
|
||||||
f.mu.Lock()
|
f.mu.Lock()
|
||||||
f.calls++
|
f.calls++
|
||||||
f.mu.Unlock()
|
f.mu.Unlock()
|
||||||
@@ -44,34 +41,11 @@ func (f *fakeSearcher) Search(ctx context.Context, query string, limit int, asOf
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
if f.callback != nil {
|
if f.callback != nil {
|
||||||
return f.callback(query, limit, asOf)
|
return f.callback(query, limit)
|
||||||
}
|
}
|
||||||
return []byte(`{"query":"` + query + `","count":0,"results":[]}`), nil
|
return []byte(`{"query":"` + query + `","count":0,"results":[]}`), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (f *fakeSearcher) Get(_ context.Context, id string, body bool) ([]byte, error) {
|
|
||||||
out := map[string]any{"id": id, "root": "info"}
|
|
||||||
if body {
|
|
||||||
out["text"] = "fake body"
|
|
||||||
}
|
|
||||||
return json.Marshal(out)
|
|
||||||
}
|
|
||||||
|
|
||||||
func (f *fakeSearcher) Stats(context.Context) ([]byte, error) {
|
|
||||||
return []byte(`{"total":0,"by_root":{}}`), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (f *fakeSearcher) Audit(context.Context) ([]byte, error) {
|
|
||||||
return []byte(`{"status":"ok"}`), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (f *fakeSearcher) Ingest(_ context.Context, body []byte) ([]byte, error) {
|
|
||||||
if len(bytes.TrimSpace(body)) == 0 {
|
|
||||||
return []byte(`{"mode":"add","command":"bin/brain/add.go"}`), nil
|
|
||||||
}
|
|
||||||
return []byte(`{"mode":"add","ids":["fake-leaf"]}`), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (f *fakeSearcher) count() int {
|
func (f *fakeSearcher) count() int {
|
||||||
f.mu.Lock()
|
f.mu.Lock()
|
||||||
defer f.mu.Unlock()
|
defer f.mu.Unlock()
|
||||||
@@ -109,7 +83,7 @@ func TestSearchMissingQuery(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestSearchReturnsSearcherResult(t *testing.T) {
|
func TestSearchReturnsSearcherResult(t *testing.T) {
|
||||||
fs := &fakeSearcher{callback: func(q string, limit int, asOf string) ([]byte, error) {
|
fs := &fakeSearcher{callback: func(q string, limit int) ([]byte, error) {
|
||||||
return []byte(`{"query":"` + q + `","count":1,"results":[{"id":"x"}]}`), nil
|
return []byte(`{"query":"` + q + `","count":1,"results":[{"id":"x"}]}`), nil
|
||||||
}}
|
}}
|
||||||
h := NewServer(fs, 1)
|
h := NewServer(fs, 1)
|
||||||
@@ -167,79 +141,6 @@ func TestSearchRejectsBadLimit(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestGetLeaf(t *testing.T) {
|
|
||||||
fs := &fakeSearcher{callback: func(q string, limit int, asOf string) ([]byte, error) {
|
|
||||||
return []byte(`{}`), nil
|
|
||||||
}}
|
|
||||||
h := NewServer(fs, 1)
|
|
||||||
if code, _ := get(t, h, "/get"); code != http.StatusBadRequest {
|
|
||||||
t.Fatalf("missing id code = %d, want 400", code)
|
|
||||||
}
|
|
||||||
code, body := get(t, h, "/get?id=leaf-1&body=1")
|
|
||||||
if code != http.StatusOK {
|
|
||||||
t.Fatalf("get code = %d, want 200 body=%s", code, body)
|
|
||||||
}
|
|
||||||
if !strings.Contains(string(body), "leaf-1") {
|
|
||||||
t.Fatalf("get body %s missing id", body)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestStatsAuditIngest(t *testing.T) {
|
|
||||||
h := NewServer(&fakeSearcher{}, 1)
|
|
||||||
for _, path := range []string{"/stats", "/audit", "/ingest"} {
|
|
||||||
code, body := get(t, h, path)
|
|
||||||
if code != http.StatusOK {
|
|
||||||
t.Fatalf("%s code = %d, want 200 (%s)", path, code, body)
|
|
||||||
}
|
|
||||||
if !json.Valid(body) {
|
|
||||||
t.Fatalf("%s body not json: %s", path, body)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestIngestIsAddNotRebuildHint(t *testing.T) {
|
|
||||||
h := NewServer(&fakeSearcher{}, 1)
|
|
||||||
code, body := get(t, h, "/ingest")
|
|
||||||
if code != http.StatusOK {
|
|
||||||
t.Fatalf("GET /ingest code = %d body=%s", code, body)
|
|
||||||
}
|
|
||||||
if strings.Contains(string(body), `"add":"v2"`) || strings.Contains(string(body), "write is v2") {
|
|
||||||
t.Fatalf("GET /ingest still a v2 hint: %s", body)
|
|
||||||
}
|
|
||||||
if !strings.Contains(string(body), "bin/brain/add.go") {
|
|
||||||
t.Fatalf("GET /ingest should name add.go: %s", body)
|
|
||||||
}
|
|
||||||
code, body = postJSON(t, h, "/ingest", `{"text":"hello","root":"info","source":"t"}`)
|
|
||||||
if code != http.StatusOK {
|
|
||||||
t.Fatalf("POST /ingest code = %d body=%s", code, body)
|
|
||||||
}
|
|
||||||
if !strings.Contains(string(body), "fake-leaf") {
|
|
||||||
t.Fatalf("POST /ingest should add: %s", body)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestHTTPPackageDoesNotExecPython(t *testing.T) {
|
|
||||||
raw, err := os.ReadFile("server.go")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
lower := strings.ToLower(string(raw))
|
|
||||||
if strings.Contains(lower, "python3") || strings.Contains(lower, "bin/kb/search") {
|
|
||||||
t.Fatal("httpapi must not exec Python or bin/kb/search")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestDefaultSearchCmdIsBrainNotPython(t *testing.T) {
|
|
||||||
t.Setenv("KB_SEARCH_CMD", "")
|
|
||||||
cmd := defaultSearchCmd("/repo")
|
|
||||||
if strings.Contains(strings.ToLower(cmd), "python") {
|
|
||||||
t.Fatalf("search path still python: %s", cmd)
|
|
||||||
}
|
|
||||||
if !strings.Contains(cmd, "brain") {
|
|
||||||
t.Fatalf("search path must be the Go brain binary, got %s", cmd)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSearchTimeout(t *testing.T) {
|
func TestSearchTimeout(t *testing.T) {
|
||||||
fs := &fakeSearcher{delay: time.Second}
|
fs := &fakeSearcher{delay: time.Second}
|
||||||
h := NewServer(fs, 1)
|
h := NewServer(fs, 1)
|
||||||
@@ -1,248 +0,0 @@
|
|||||||
# bin/stack/lib.sh — compose helpers for start / start-assistant / stop / status.
|
|
||||||
# Sourced, not executed. No secrets. No host-absolute paths.
|
|
||||||
|
|
||||||
BRAIN_URL="${BRAIN_URL:-http://127.0.0.1:8630}"
|
|
||||||
REASONER_URL="${REASONER_URL:-http://127.0.0.1:11435}"
|
|
||||||
PICOCLAW_URL="${PICOCLAW_URL:-http://127.0.0.1:18790}"
|
|
||||||
REASONER_MODEL="${REASONER_MODEL:-qwen3.5:9b}"
|
|
||||||
STACK_WAIT_SECS="${STACK_WAIT_SECS:-90}"
|
|
||||||
STACK_WAIT_INTERVAL="${STACK_WAIT_INTERVAL:-2}"
|
|
||||||
STACK_PULL_SECS="${STACK_PULL_SECS:-600}"
|
|
||||||
|
|
||||||
if [[ -z "${ROOT:-}" ]]; then
|
|
||||||
STACK_DIR="$(CDPATH= cd -- "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
||||||
ROOT="$(CDPATH= cd -- "$STACK_DIR/../.." && pwd)"
|
|
||||||
fi
|
|
||||||
|
|
||||||
stack_usage() {
|
|
||||||
awk 'NR == 1 { next } /^#/ { sub(/^# ?/, ""); print; next } { exit }' "$1"
|
|
||||||
}
|
|
||||||
|
|
||||||
stack_die() {
|
|
||||||
echo "bin/stack: $*" >&2
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
|
|
||||||
compose() {
|
|
||||||
docker compose -f "$ROOT/compose.yaml" --project-directory "$ROOT" "$@"
|
|
||||||
}
|
|
||||||
|
|
||||||
http_get() {
|
|
||||||
local url=$1
|
|
||||||
local timeout=${2:-5}
|
|
||||||
curl -sS --max-time "$timeout" "$url" 2>/dev/null || return 1
|
|
||||||
}
|
|
||||||
|
|
||||||
health_ok() {
|
|
||||||
local url=$1
|
|
||||||
local timeout=${2:-5}
|
|
||||||
local body
|
|
||||||
body=$(http_get "$url" "$timeout") || return 1
|
|
||||||
printf '%s' "$body" | grep -q '"status":"ok"'
|
|
||||||
}
|
|
||||||
|
|
||||||
wait_health() {
|
|
||||||
local url=$1
|
|
||||||
local n=0
|
|
||||||
while ((n <= STACK_WAIT_SECS)); do
|
|
||||||
if health_ok "$url"; then
|
|
||||||
return 0
|
|
||||||
fi
|
|
||||||
n=$((n + 1))
|
|
||||||
if ((n <= STACK_WAIT_SECS)); then
|
|
||||||
sleep "$STACK_WAIT_INTERVAL"
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
|
|
||||||
wait_http() {
|
|
||||||
local url=$1
|
|
||||||
local n=0
|
|
||||||
while ((n <= STACK_WAIT_SECS)); do
|
|
||||||
if http_get "$url" 5 >/dev/null; then
|
|
||||||
return 0
|
|
||||||
fi
|
|
||||||
n=$((n + 1))
|
|
||||||
if ((n <= STACK_WAIT_SECS)); then
|
|
||||||
sleep "$STACK_WAIT_INTERVAL"
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
|
|
||||||
mcp_body() {
|
|
||||||
curl -sS --max-time 10 \
|
|
||||||
-H 'Content-Type: application/json' \
|
|
||||||
-d '{"jsonrpc":"2.0","id":1,"method":"tools/list"}' \
|
|
||||||
"$BRAIN_URL/mcp" 2>/dev/null || return 1
|
|
||||||
}
|
|
||||||
|
|
||||||
mcp_ok() {
|
|
||||||
local body
|
|
||||||
body=$(mcp_body) || return 1
|
|
||||||
printf '%s' "$body" | grep -Eq '"name": ?"search"' || return 1
|
|
||||||
printf '%s' "$body" | grep -Eq '"name": ?"get"' || return 1
|
|
||||||
printf '%s' "$body" | grep -Eq '"name": ?"audit"' || return 1
|
|
||||||
return 0
|
|
||||||
}
|
|
||||||
|
|
||||||
reasoner_tags() {
|
|
||||||
http_get "$REASONER_URL/api/tags" 5
|
|
||||||
}
|
|
||||||
|
|
||||||
reasoner_has_model() {
|
|
||||||
local body
|
|
||||||
body=$(reasoner_tags) || return 1
|
|
||||||
printf '%s' "$body" | grep -Fq "$REASONER_MODEL"
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_mcp() {
|
|
||||||
mcp_ok || stack_die "MCP tools/list missing search/get/audit at $BRAIN_URL/mcp"
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_brain() {
|
|
||||||
if health_ok "$BRAIN_URL/health"; then
|
|
||||||
echo "brain: reuse $BRAIN_URL" >&2
|
|
||||||
else
|
|
||||||
echo "brain: compose up" >&2
|
|
||||||
compose up -d brain
|
|
||||||
wait_health "$BRAIN_URL/health" || stack_die "brain health failed at $BRAIN_URL/health"
|
|
||||||
fi
|
|
||||||
ensure_mcp
|
|
||||||
}
|
|
||||||
|
|
||||||
pull_reasoner_model() {
|
|
||||||
echo "reasoner: pulling $REASONER_MODEL (CPU, may take minutes)" >&2
|
|
||||||
curl -sS --max-time "$STACK_PULL_SECS" \
|
|
||||||
-H 'Content-Type: application/json' \
|
|
||||||
-d "{\"name\":\"$REASONER_MODEL\"}" \
|
|
||||||
"$REASONER_URL/api/pull" >/dev/null
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_reasoner() {
|
|
||||||
if reasoner_has_model; then
|
|
||||||
echo "reasoner: reuse $REASONER_URL model $REASONER_MODEL" >&2
|
|
||||||
return 0
|
|
||||||
fi
|
|
||||||
if ! reasoner_tags >/dev/null; then
|
|
||||||
echo "reasoner: compose up" >&2
|
|
||||||
compose --profile reasoner up -d reasoner
|
|
||||||
wait_http "$REASONER_URL/api/tags" || stack_die "reasoner not listening at $REASONER_URL"
|
|
||||||
fi
|
|
||||||
if reasoner_has_model; then
|
|
||||||
return 0
|
|
||||||
fi
|
|
||||||
pull_reasoner_model
|
|
||||||
reasoner_has_model || stack_die "reasoner missing model $REASONER_MODEL"
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_picoclaw() {
|
|
||||||
echo "picoclaw: compose up --no-deps (reuse healthy :8630/:11435)" >&2
|
|
||||||
compose --profile picoclaw up -d --no-deps picoclaw
|
|
||||||
wait_health "$PICOCLAW_URL/health" || stack_die "picoclaw health failed at $PICOCLAW_URL/health"
|
|
||||||
}
|
|
||||||
|
|
||||||
mail_sync_running() {
|
|
||||||
compose ps --status running --services 2>/dev/null | grep -qx mail-sync
|
|
||||||
}
|
|
||||||
|
|
||||||
stack_status() {
|
|
||||||
local bh=down mcp=down ph=down present=false ms=down
|
|
||||||
health_ok "$BRAIN_URL/health" && bh=ok
|
|
||||||
mcp_ok && mcp=ok
|
|
||||||
reasoner_has_model && present=true
|
|
||||||
health_ok "$PICOCLAW_URL/health" && ph=ok
|
|
||||||
mail_sync_running && ms=ok
|
|
||||||
cat <<EOF
|
|
||||||
brain:
|
|
||||||
url: $BRAIN_URL
|
|
||||||
health: $bh
|
|
||||||
mcp: $mcp
|
|
||||||
reasoner:
|
|
||||||
url: $REASONER_URL
|
|
||||||
model: $REASONER_MODEL
|
|
||||||
present: $present
|
|
||||||
picoclaw:
|
|
||||||
url: $PICOCLAW_URL
|
|
||||||
health: $ph
|
|
||||||
mail_sync:
|
|
||||||
service: mail-sync
|
|
||||||
running: $ms
|
|
||||||
EOF
|
|
||||||
}
|
|
||||||
|
|
||||||
stack_start() {
|
|
||||||
ensure_brain
|
|
||||||
}
|
|
||||||
|
|
||||||
stack_start_mail_sync() {
|
|
||||||
echo "mail-sync: compose up (ETL sync→import; index only if MAIL_SYNC_INDEX=1)" >&2
|
|
||||||
compose up -d mail-sync
|
|
||||||
}
|
|
||||||
|
|
||||||
stack_attach_agent() {
|
|
||||||
local opts=()
|
|
||||||
if [[ -t 0 && -t 1 ]]; then
|
|
||||||
opts+=(-it)
|
|
||||||
else
|
|
||||||
opts+=(-T)
|
|
||||||
fi
|
|
||||||
if [[ (! -t 0 || ! -t 1) && $# -eq 0 ]]; then
|
|
||||||
echo "picoclaw: no TTY. Attach with:" >&2
|
|
||||||
echo " $ROOT/bin/stack/start-assistant" >&2
|
|
||||||
echo " docker compose --profile picoclaw exec -it picoclaw picoclaw agent" >&2
|
|
||||||
return 0
|
|
||||||
fi
|
|
||||||
echo "picoclaw: agent (search → get → audit before a factual reply)" >&2
|
|
||||||
exec docker compose -f "$ROOT/compose.yaml" --project-directory "$ROOT" \
|
|
||||||
--profile picoclaw exec "${opts[@]}" picoclaw picoclaw agent "$@"
|
|
||||||
}
|
|
||||||
|
|
||||||
stack_start_assistant() {
|
|
||||||
local attach=1
|
|
||||||
local agent_args=()
|
|
||||||
while (($#)); do
|
|
||||||
case "$1" in
|
|
||||||
-h | --help)
|
|
||||||
stack_usage "$ROOT/bin/stack/start-assistant"
|
|
||||||
return 0
|
|
||||||
;;
|
|
||||||
--no-attach)
|
|
||||||
attach=0
|
|
||||||
shift
|
|
||||||
;;
|
|
||||||
--)
|
|
||||||
shift
|
|
||||||
agent_args+=("$@")
|
|
||||||
break
|
|
||||||
;;
|
|
||||||
*)
|
|
||||||
agent_args+=("$1")
|
|
||||||
shift
|
|
||||||
;;
|
|
||||||
esac
|
|
||||||
done
|
|
||||||
stack_start
|
|
||||||
ensure_reasoner
|
|
||||||
ensure_picoclaw
|
|
||||||
stack_status
|
|
||||||
if ((attach == 0)); then
|
|
||||||
echo "picoclaw: gateway $PICOCLAW_URL (agent not attached)" >&2
|
|
||||||
echo "ask the brain: $ROOT/bin/stack/start-assistant" >&2
|
|
||||||
echo "one-shot: $ROOT/bin/stack/start-assistant -- -m \"search the 2dph brain for LadybugDB\"" >&2
|
|
||||||
return 0
|
|
||||||
fi
|
|
||||||
stack_attach_agent "${agent_args[@]}"
|
|
||||||
}
|
|
||||||
|
|
||||||
stack_stop() {
|
|
||||||
case "${1:-}" in
|
|
||||||
-h | --help)
|
|
||||||
stack_usage "$ROOT/bin/stack/stop"
|
|
||||||
return 0
|
|
||||||
;;
|
|
||||||
esac
|
|
||||||
echo "stack: stop brain brain-mcp reasoner picoclaw mail-sync (volumes kept)" >&2
|
|
||||||
compose --profile picoclaw --profile reasoner stop picoclaw brain-mcp reasoner brain mail-sync
|
|
||||||
}
|
|
||||||
@@ -1,21 +0,0 @@
|
|||||||
#!/usr/bin/env bash
|
|
||||||
# bin/stack/start - bring up brain HTTP/MCP and wait until search/get/audit respond.
|
|
||||||
#
|
|
||||||
# bin/stack/start
|
|
||||||
# bin/stack/status
|
|
||||||
#
|
|
||||||
# Reuses a healthy process on :8630 (host serve or compose). Does not start
|
|
||||||
# PicoClaw. Does not rebuild Ladybug.
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
STACK_DIR="$(CDPATH= cd -- "$(dirname "$0")" && pwd)"
|
|
||||||
# shellcheck source=lib.sh
|
|
||||||
source "$STACK_DIR/lib.sh"
|
|
||||||
case "${1:-}" in
|
|
||||||
-h | --help)
|
|
||||||
stack_usage "$0"
|
|
||||||
exit 0
|
|
||||||
;;
|
|
||||||
esac
|
|
||||||
stack_start "$@"
|
|
||||||
stack_status
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
#!/usr/bin/env bash
|
|
||||||
# bin/stack/start-assistant - start + CPU reasoner + PicoClaw, then attach agent.
|
|
||||||
#
|
|
||||||
# bin/stack/start-assistant
|
|
||||||
# bin/stack/start-assistant --no-attach
|
|
||||||
# bin/stack/start-assistant -- -m "search the 2dph brain for LadybugDB"
|
|
||||||
#
|
|
||||||
# Pulls qwen3.5:9b if missing. Gateway :18790. Agent uses MCP search → get → audit.
|
|
||||||
# --no-attach leaves the gateway up without exec.
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
STACK_DIR="$(CDPATH= cd -- "$(dirname "$0")" && pwd)"
|
|
||||||
# shellcheck source=lib.sh
|
|
||||||
source "$STACK_DIR/lib.sh"
|
|
||||||
stack_start_assistant "$@"
|
|
||||||
@@ -1,21 +0,0 @@
|
|||||||
#!/usr/bin/env bash
|
|
||||||
# bin/stack/start-mail-sync - compose up mail-sync ETL (sync → import; optional index).
|
|
||||||
#
|
|
||||||
# bin/stack/start-mail-sync
|
|
||||||
#
|
|
||||||
# Default: onlyoffice,gmail every 300s into kb-var. Full --rebuild only if
|
|
||||||
# MAIL_SYNC_INDEX=1 in compose/env. Secrets: ~/.config/brain/mail.env +
|
|
||||||
# ~/.gmail-mcp (mounted). Does not start brain/picoclaw.
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
STACK_DIR="$(CDPATH= cd -- "$(dirname "$0")" && pwd)"
|
|
||||||
# shellcheck source=lib.sh
|
|
||||||
source "$STACK_DIR/lib.sh"
|
|
||||||
case "${1:-}" in
|
|
||||||
-h | --help)
|
|
||||||
stack_usage "$0"
|
|
||||||
exit 0
|
|
||||||
;;
|
|
||||||
esac
|
|
||||||
stack_start_mail_sync "$@"
|
|
||||||
stack_status
|
|
||||||
@@ -1,17 +0,0 @@
|
|||||||
#!/usr/bin/env bash
|
|
||||||
# bin/stack/status - YAML health for brain MCP, reasoner model, PicoClaw gateway.
|
|
||||||
#
|
|
||||||
# bin/stack/status
|
|
||||||
# bin/stack/status | yq '.picoclaw'
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
STACK_DIR="$(CDPATH= cd -- "$(dirname "$0")" && pwd)"
|
|
||||||
# shellcheck source=lib.sh
|
|
||||||
source "$STACK_DIR/lib.sh"
|
|
||||||
case "${1:-}" in
|
|
||||||
-h | --help)
|
|
||||||
stack_usage "$0"
|
|
||||||
exit 0
|
|
||||||
;;
|
|
||||||
esac
|
|
||||||
stack_status
|
|
||||||
@@ -1,13 +0,0 @@
|
|||||||
#!/usr/bin/env bash
|
|
||||||
# bin/stack/stop - stop compose brain / brain-mcp / reasoner / picoclaw / mail-sync.
|
|
||||||
#
|
|
||||||
# bin/stack/stop
|
|
||||||
#
|
|
||||||
# Volumes kept (kb, reasoner weights, picoclaw-home). Does not kill a host
|
|
||||||
# bin/brain/serve.go that is not a compose service.
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
STACK_DIR="$(CDPATH= cd -- "$(dirname "$0")" && pwd)"
|
|
||||||
# shellcheck source=lib.sh
|
|
||||||
source "$STACK_DIR/lib.sh"
|
|
||||||
stack_stop "$@"
|
|
||||||
@@ -1,103 +0,0 @@
|
|||||||
"""D16 contradiction adjudication (same rules as internal/facts)."""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from typing import Any
|
|
||||||
|
|
||||||
CONF_CONFIRMED = "confirmed"
|
|
||||||
CONF_HYPOTHESIS = "hypothesis"
|
|
||||||
|
|
||||||
RULE_UNRESOLVED = "unresolved"
|
|
||||||
RULE_TEMPORAL = "temporal_freshness"
|
|
||||||
RULE_AUTHORITY = "authority_pairing"
|
|
||||||
RULE_TWO_SOURCE = "two_source"
|
|
||||||
RULE_SINGLE = "single_source"
|
|
||||||
|
|
||||||
KIND_RUNTIME = "runtime"
|
|
||||||
KIND_CONFIG = "config"
|
|
||||||
KIND_NARRATIVE = "narrative"
|
|
||||||
|
|
||||||
|
|
||||||
def _independent(sources: list[dict]) -> int:
|
|
||||||
seen: set[str] = set()
|
|
||||||
for i, s in enumerate(sources):
|
|
||||||
sid = str(s.get("id") or "") or f"{s.get('kind', '')}#{i}"
|
|
||||||
seen.add(sid)
|
|
||||||
return len(seen)
|
|
||||||
|
|
||||||
|
|
||||||
def _fresh_n(sources: list[dict]) -> int:
|
|
||||||
return sum(1 for s in sources if not s.get("stale"))
|
|
||||||
|
|
||||||
|
|
||||||
def _strong_n(sources: list[dict]) -> int:
|
|
||||||
return sum(1 for s in sources if s.get("kind") in (KIND_RUNTIME, KIND_CONFIG))
|
|
||||||
|
|
||||||
|
|
||||||
def adjudicate(claim: dict[str, Any]) -> dict[str, Any]:
|
|
||||||
yes = list(claim.get("yes") or [])
|
|
||||||
no = list(claim.get("no") or [])
|
|
||||||
yes_n, no_n = _independent(yes), _independent(no)
|
|
||||||
text = str(claim.get("text") or "")
|
|
||||||
|
|
||||||
def out(conf: str, rule: str, winner: str = "") -> dict[str, Any]:
|
|
||||||
return {
|
|
||||||
"text": text,
|
|
||||||
"confidence": conf,
|
|
||||||
"confirmed": conf == CONF_CONFIRMED,
|
|
||||||
"rule": rule,
|
|
||||||
"winner": winner,
|
|
||||||
"yes": yes_n,
|
|
||||||
"no": no_n,
|
|
||||||
}
|
|
||||||
|
|
||||||
if yes_n < 2 or no_n < 2:
|
|
||||||
if yes_n >= 2:
|
|
||||||
return out(CONF_CONFIRMED, RULE_TWO_SOURCE, "yes")
|
|
||||||
if no_n >= 2:
|
|
||||||
return out(CONF_CONFIRMED, RULE_TWO_SOURCE, "no")
|
|
||||||
return out(CONF_HYPOTHESIS, RULE_SINGLE)
|
|
||||||
yf, nf = _fresh_n(yes), _fresh_n(no)
|
|
||||||
if yf >= 2 and nf < 2:
|
|
||||||
return out(CONF_CONFIRMED, RULE_TEMPORAL, "yes")
|
|
||||||
if nf >= 2 and yf < 2:
|
|
||||||
return out(CONF_CONFIRMED, RULE_TEMPORAL, "no")
|
|
||||||
ys, ns = _strong_n(yes), _strong_n(no)
|
|
||||||
if ys >= 2 and ns < 2:
|
|
||||||
return out(CONF_CONFIRMED, RULE_AUTHORITY, "yes")
|
|
||||||
if ns >= 2 and ys < 2:
|
|
||||||
return out(CONF_CONFIRMED, RULE_AUTHORITY, "no")
|
|
||||||
return out(CONF_HYPOTHESIS, RULE_UNRESOLVED)
|
|
||||||
|
|
||||||
|
|
||||||
def parse_source_field(source: str) -> tuple[str, str]:
|
|
||||||
"""Split `a x b vs c x d` into (yes, no). Empty no if no ` vs `."""
|
|
||||||
if " vs " not in source:
|
|
||||||
return source, ""
|
|
||||||
yes, _, no = source.partition(" vs ")
|
|
||||||
return yes.strip(), no.strip()
|
|
||||||
|
|
||||||
|
|
||||||
def check_fact_row(lid: str, source: str, loc: str, how: str, conf: str) -> list[str]:
|
|
||||||
"""Lexicon checks for one facts leaf (no Ladybug)."""
|
|
||||||
problems: list[str] = []
|
|
||||||
src = source or ""
|
|
||||||
if conf == CONF_CONFIRMED:
|
|
||||||
if " vs " in src:
|
|
||||||
problems.append(f"{lid}: confirmed fact cannot keep a vs-contradiction")
|
|
||||||
if " x " not in src:
|
|
||||||
problems.append(f"{lid}: needs 2-source evidence in source, got '{source}'")
|
|
||||||
elif conf == CONF_HYPOTHESIS:
|
|
||||||
yes, no = parse_source_field(src)
|
|
||||||
if not no or " x " not in yes or " x " not in no:
|
|
||||||
problems.append(
|
|
||||||
f"{lid}: hypothesis contradiction needs 'a x b vs c x d', got '{source}'"
|
|
||||||
)
|
|
||||||
elif conf == "partial":
|
|
||||||
pass
|
|
||||||
else:
|
|
||||||
problems.append(f"{lid}: unknown confidence '{conf}'")
|
|
||||||
if not loc:
|
|
||||||
problems.append(f"{lid}: missing loc (evidence pointer)")
|
|
||||||
if not how:
|
|
||||||
problems.append(f"{lid}: missing how")
|
|
||||||
return problems
|
|
||||||
+60
-3
@@ -1,12 +1,21 @@
|
|||||||
"""gitimport - Ladybug graph writes for Commit/File/Person (no git binary).
|
"""gitimport - parse `git log` output and turn commits into brain leafs.
|
||||||
|
|
||||||
Commit records come from bin/git/import.go (go-git). This module only MERGEs
|
Pure, testable functions. Field grammar (see bin/git/import):
|
||||||
the version graph File-[:HAS_VERSION]->Commit-[:AUTHORED]->Person.
|
|
||||||
|
git log --no-merges --name-only \
|
||||||
|
--format='%x1e%H%x1f%an%x1f%ae%x1f%aI%x1f%s'
|
||||||
|
|
||||||
|
0x1e = record separator, 0x1f = field separator.
|
||||||
|
Files: newline-separated lines following each record's subject.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
|
|
||||||
|
REC_SEP = "\x1e"
|
||||||
|
FIELD_SEP = "\x1f"
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class Commit:
|
class Commit:
|
||||||
@@ -17,6 +26,54 @@ class Commit:
|
|||||||
subject: str
|
subject: str
|
||||||
files: list[str] = field(default_factory=list)
|
files: list[str] = field(default_factory=list)
|
||||||
|
|
||||||
|
def leaf_text(self, repo: str) -> str:
|
||||||
|
head = f"commit {self.sha[:12]} in {repo} — {self.subject}"
|
||||||
|
body = [head, f"Author: {self.author} <{self.email}>", f"Date: {self.date}"]
|
||||||
|
if self.files:
|
||||||
|
body.append("Changing: " + ", ".join(self.files))
|
||||||
|
return "\n".join(body)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_log(text: str) -> list[Commit]:
|
||||||
|
"""Parse `git log` output into Commit records.
|
||||||
|
|
||||||
|
Records are separated by 0x1e. A record is fields joined by 0x1f,
|
||||||
|
followed by optional newline-separated file paths inside the next
|
||||||
|
segment (git emits blank line + files after each record).
|
||||||
|
"""
|
||||||
|
commits: list[Commit] = []
|
||||||
|
# field records and file lists alternate; simpler: split on REC_SEP,
|
||||||
|
# each chunk = header line, possibly followed by newline + files.
|
||||||
|
for chunk in text.split(REC_SEP):
|
||||||
|
chunk = chunk.strip("\n")
|
||||||
|
if not chunk:
|
||||||
|
continue
|
||||||
|
lines = chunk.split("\n", 1)
|
||||||
|
header = lines[0].split(FIELD_SEP)
|
||||||
|
if len(header) < 5:
|
||||||
|
continue
|
||||||
|
sha, author, email, date, subject = header[:5]
|
||||||
|
files = [ln.strip() for ln in lines[1].splitlines() if ln.strip()] if len(lines) > 1 else []
|
||||||
|
commits.append(Commit(sha=sha, author=author, email=email,
|
||||||
|
date=date, subject=subject, files=files))
|
||||||
|
return commits
|
||||||
|
|
||||||
|
|
||||||
|
def commits_to_leafs(commits: list[Commit], repo: str) -> list[dict]:
|
||||||
|
"""Map commits to the leaf shape bin/kb/index expects (source/repo/...)."""
|
||||||
|
out: list[dict] = []
|
||||||
|
for c in commits:
|
||||||
|
out.append({
|
||||||
|
"source": f"{repo}@{c.sha}",
|
||||||
|
"repo": repo,
|
||||||
|
"heading": f"commit {c.sha[:12]} — {c.subject}",
|
||||||
|
"text": c.leaf_text(repo),
|
||||||
|
"type": "commit",
|
||||||
|
"status": "current",
|
||||||
|
"related": ",".join(c.files),
|
||||||
|
})
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
GIT_SCHEMA = (
|
GIT_SCHEMA = (
|
||||||
"CREATE NODE TABLE IF NOT EXISTS Commit (id STRING, repo STRING, subject STRING, "
|
"CREATE NODE TABLE IF NOT EXISTS Commit (id STRING, repo STRING, subject STRING, "
|
||||||
|
|||||||
+16
-169
@@ -4,7 +4,7 @@ Single embedded graph `var/kb.lbug`. Two roots: facts (assertions backed by
|
|||||||
>=2 independent sources) and info (narrative leafs). Hybrid retrieval: BM25
|
>=2 independent sources) and info (narrative leafs). Hybrid retrieval: BM25
|
||||||
(FTS extension) + HNSW cosine (VECTOR extension) + Cypher graph hops.
|
(FTS extension) + HNSW cosine (VECTOR extension) + Cypher graph hops.
|
||||||
|
|
||||||
All access is read-only unless `--rebuild` (kb/index) or `kb/add`.
|
All access is read-only unless `--rebuild` is passed to kb/index.
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -62,9 +62,7 @@ def init_schema(conn: ladybug.Connection) -> None:
|
|||||||
"CREATE NODE TABLE IF NOT EXISTS Leaf ("
|
"CREATE NODE TABLE IF NOT EXISTS Leaf ("
|
||||||
" id STRING, text STRING, root STRING, confidence STRING, "
|
" id STRING, text STRING, root STRING, confidence STRING, "
|
||||||
" sha256 STRING, source STRING, source_rev STRING, observed_at STRING, "
|
" sha256 STRING, source STRING, source_rev STRING, observed_at STRING, "
|
||||||
" how STRING, loc STRING, type STRING, "
|
" how STRING, loc STRING, type STRING, embedding FLOAT[256], "
|
||||||
" valid_from STRING, valid_to STRING, "
|
|
||||||
" embedding FLOAT[256], "
|
|
||||||
" PRIMARY KEY(id))"
|
" PRIMARY KEY(id))"
|
||||||
)
|
)
|
||||||
conn.execute(
|
conn.execute(
|
||||||
@@ -93,46 +91,6 @@ def init_schema(conn: ladybug.Connection) -> None:
|
|||||||
conn.execute(
|
conn.execute(
|
||||||
"CREATE REL TABLE IF NOT EXISTS AUTHORED (FROM Commit TO Person)"
|
"CREATE REL TABLE IF NOT EXISTS AUTHORED (FROM Commit TO Person)"
|
||||||
)
|
)
|
||||||
ensure_interval_columns(conn)
|
|
||||||
|
|
||||||
|
|
||||||
def ensure_interval_columns(conn: ladybug.Connection) -> None:
|
|
||||||
"""D24: add valid_from/valid_to on older Leaf tables (idempotent ALTER)."""
|
|
||||||
for col in ("valid_from", "valid_to"):
|
|
||||||
try:
|
|
||||||
conn.execute(f"ALTER TABLE Leaf ADD {col} STRING")
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
|
|
||||||
|
|
||||||
def normalize_day(s: str) -> str:
|
|
||||||
s = (s or "").strip()
|
|
||||||
if len(s) >= 10 and s[4] == "-" and s[7] == "-":
|
|
||||||
return s[:10]
|
|
||||||
return s
|
|
||||||
|
|
||||||
|
|
||||||
def active_at(valid_from: str, valid_to: str, as_of: str) -> bool:
|
|
||||||
"""D24: fact interval of truth. Empty ends = always; empty as_of = no filter."""
|
|
||||||
as_of = normalize_day(as_of)
|
|
||||||
if not as_of:
|
|
||||||
return True
|
|
||||||
fro = normalize_day(valid_from)
|
|
||||||
to = normalize_day(valid_to)
|
|
||||||
if fro and as_of < fro:
|
|
||||||
return False
|
|
||||||
if to and as_of > to:
|
|
||||||
return False
|
|
||||||
return True
|
|
||||||
|
|
||||||
|
|
||||||
def filter_as_of(hits: list[dict], as_of: str) -> list[dict]:
|
|
||||||
if not as_of:
|
|
||||||
return hits
|
|
||||||
return [
|
|
||||||
h for h in hits
|
|
||||||
if active_at(str(h.get("valid_from") or ""), str(h.get("valid_to") or ""), as_of)
|
|
||||||
]
|
|
||||||
|
|
||||||
|
|
||||||
def leaf_id(text: str, source: str) -> str:
|
def leaf_id(text: str, source: str) -> str:
|
||||||
@@ -141,124 +99,25 @@ def leaf_id(text: str, source: str) -> str:
|
|||||||
|
|
||||||
def upsert_leaf(conn: ladybug.Connection, *, text: str, root: str, confidence: str,
|
def upsert_leaf(conn: ladybug.Connection, *, text: str, root: str, confidence: str,
|
||||||
source: str, source_rev: str, how: str, loc: str, type_: str,
|
source: str, source_rev: str, how: str, loc: str, type_: str,
|
||||||
embedding: list[float] | None,
|
embedding: list[float] | None) -> str:
|
||||||
valid_from: str = "", valid_to: str = "") -> str:
|
|
||||||
lid = leaf_id(text, source)
|
lid = leaf_id(text, source)
|
||||||
obs = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
obs = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
||||||
vf = normalize_day(valid_from)
|
|
||||||
vt = normalize_day(valid_to)
|
|
||||||
conn.execute(
|
conn.execute(
|
||||||
"MERGE (l:Leaf {id:$id}) "
|
"MERGE (l:Leaf {id:$id}) "
|
||||||
"SET l.text=$text, l.root=$root, l.confidence=$confidence, "
|
"SET l.text=$text, l.root=$root, l.confidence=$confidence, "
|
||||||
" l.sha256=$sha, l.source=$source, l.source_rev=$rev, l.observed_at=$obs, "
|
" l.sha256=$sha, l.source=$source, l.source_rev=$rev, l.observed_at=$obs, "
|
||||||
" l.how=$how, l.loc=$location, l.type=$type, "
|
" l.how=$how, l.loc=$location, l.type=$type"
|
||||||
" l.valid_from=$vf, l.valid_to=$vt"
|
|
||||||
+ (", l.embedding=$emb" if embedding else ""),
|
+ (", l.embedding=$emb" if embedding else ""),
|
||||||
parameters={
|
parameters={
|
||||||
"id": lid, "text": text, "root": root, "confidence": confidence,
|
"id": lid, "text": text, "root": root, "confidence": confidence,
|
||||||
"sha": sha256_b64(text), "source": source, "rev": source_rev,
|
"sha": sha256_b64(text), "source": source, "rev": source_rev,
|
||||||
"obs": obs, "how": how, "location": loc, "type": type_,
|
"obs": obs, "how": how, "location": loc, "type": type_,
|
||||||
"vf": vf, "vt": vt,
|
|
||||||
"emb": (embedding if embedding else None),
|
"emb": (embedding if embedding else None),
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
return lid
|
return lid
|
||||||
|
|
||||||
|
|
||||||
def add_leafs(conn: ladybug.Connection, leafs: list[dict]) -> list[str]:
|
|
||||||
"""Write facts+info leafs in one transaction. Safe while FTS/HNSW exist.
|
|
||||||
|
|
||||||
Each leaf dict: text, source, optional root/confidence/source_rev/how/loc/type/
|
|
||||||
embedding/valid_from/valid_to. Does not delete the database file. Measured on
|
|
||||||
Ladybug 0.19: MERGE of new ids (and updates) stays FTS+HNSW queryable; DROP
|
|
||||||
INDEX is the fatal path.
|
|
||||||
"""
|
|
||||||
if not leafs:
|
|
||||||
return []
|
|
||||||
started = False
|
|
||||||
try:
|
|
||||||
conn.execute("BEGIN TRANSACTION")
|
|
||||||
started = True
|
|
||||||
except Exception:
|
|
||||||
started = False
|
|
||||||
ids: list[str] = []
|
|
||||||
try:
|
|
||||||
for lf in leafs:
|
|
||||||
ids.append(
|
|
||||||
upsert_leaf(
|
|
||||||
conn,
|
|
||||||
text=str(lf["text"]),
|
|
||||||
root=str(lf.get("root") or ROOT_INFO),
|
|
||||||
confidence=str(lf.get("confidence") or CONF_CONFIRMED),
|
|
||||||
source=str(lf["source"]),
|
|
||||||
source_rev=str(lf.get("source_rev") or "working-tree"),
|
|
||||||
how=str(lf.get("how") or "brain/add"),
|
|
||||||
loc=str(lf.get("loc") or lf.get("source") or ""),
|
|
||||||
type_=str(lf.get("type") or lf.get("type_") or "reference"),
|
|
||||||
embedding=lf.get("embedding"),
|
|
||||||
valid_from=str(lf.get("valid_from") or ""),
|
|
||||||
valid_to=str(lf.get("valid_to") or ""),
|
|
||||||
)
|
|
||||||
)
|
|
||||||
if started:
|
|
||||||
conn.execute("COMMIT")
|
|
||||||
except Exception:
|
|
||||||
if started:
|
|
||||||
try:
|
|
||||||
conn.execute("ROLLBACK")
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
raise
|
|
||||||
return ids
|
|
||||||
|
|
||||||
|
|
||||||
def file_id(repo: str, path: str) -> str:
|
|
||||||
"""Stable File.id matching gitimport (`repo:path`)."""
|
|
||||||
return f"{repo}:{path}" if repo else path
|
|
||||||
|
|
||||||
|
|
||||||
def link_from_file(conn: ladybug.Connection, leaf_id: str, path: str,
|
|
||||||
repo: str = "", mtime: str = "") -> str:
|
|
||||||
"""MERGE File and Leaf-[:FROM_FILE]->File so --hop 1 can walk."""
|
|
||||||
fid = file_id(repo, path)
|
|
||||||
conn.execute(
|
|
||||||
"MERGE (f:File {id:$id}) SET f.path=$path, f.repo=$repo, f.mtime=$mtime",
|
|
||||||
parameters={"id": fid, "path": path, "repo": repo, "mtime": mtime},
|
|
||||||
)
|
|
||||||
conn.execute(
|
|
||||||
"MATCH (l:Leaf {id:$lid}), (f:File {id:$fid}) "
|
|
||||||
"MERGE (l)-[:FROM_FILE]->(f)",
|
|
||||||
parameters={"lid": leaf_id, "fid": fid},
|
|
||||||
)
|
|
||||||
return fid
|
|
||||||
|
|
||||||
|
|
||||||
HOP_STMTS = {
|
|
||||||
1: "MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File) RETURN f.id, f.path, 1",
|
|
||||||
2: ("MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File)-[:HAS_VERSION]->(c:Commit) "
|
|
||||||
"RETURN c.id, c.subject, 2"),
|
|
||||||
3: ("MATCH (l:Leaf {id:$id})-[:FROM_FILE]->(f:File)-[:HAS_VERSION]->(c:Commit)"
|
|
||||||
"-[:AUTHORED]->(p:Person) RETURN p.id, p.name, 3"),
|
|
||||||
}
|
|
||||||
HOP_LABELS = {1: "File", 2: "Commit", 3: "Person"}
|
|
||||||
|
|
||||||
|
|
||||||
def hop_walk(conn: ladybug.Connection, leaf_id: str, n: int) -> list[dict]:
|
|
||||||
"""Walk Leaf → File → Commit → Person up to n hops (max 3)."""
|
|
||||||
depth = min(max(int(n), 0), 3)
|
|
||||||
out: list[dict] = []
|
|
||||||
for d in range(1, depth + 1):
|
|
||||||
rows = conn.execute(HOP_STMTS[d], parameters={"id": leaf_id}).get_all()
|
|
||||||
for row in rows:
|
|
||||||
out.append({
|
|
||||||
"id": row[0],
|
|
||||||
"label": HOP_LABELS[d],
|
|
||||||
"name": row[1],
|
|
||||||
"depth": int(row[2]),
|
|
||||||
})
|
|
||||||
return out
|
|
||||||
|
|
||||||
|
|
||||||
def leaf_index_names(conn: ladybug.Connection) -> set[str]:
|
def leaf_index_names(conn: ladybug.Connection) -> set[str]:
|
||||||
"""Return index names on the Leaf table (e.g. {'id', 'Leaf_vec', '_PK'})."""
|
"""Return index names on the Leaf table (e.g. {'id', 'Leaf_vec', '_PK'})."""
|
||||||
rows = conn.execute("CALL SHOW_INDEXES() RETURN *").get_all()
|
rows = conn.execute("CALL SHOW_INDEXES() RETURN *").get_all()
|
||||||
@@ -276,7 +135,7 @@ def create_fts_and_vector(conn: ladybug.Connection, force: bool = False) -> None
|
|||||||
|
|
||||||
`force=True` is accepted for API compatibility but does **not** drop.
|
`force=True` is accepted for API compatibility but does **not** drop.
|
||||||
Fresh indexes require deleting `var/kb.lbug` and rebuilding
|
Fresh indexes require deleting `var/kb.lbug` and rebuilding
|
||||||
(`bin/brain/index.go --rebuild`).
|
(`bin/kb/index --rebuild`).
|
||||||
"""
|
"""
|
||||||
del force # API compat; DROP is unsafe — see docstring
|
del force # API compat; DROP is unsafe — see docstring
|
||||||
names = leaf_index_names(conn)
|
names = leaf_index_names(conn)
|
||||||
@@ -286,7 +145,7 @@ def create_fts_and_vector(conn: ladybug.Connection, force: bool = False) -> None
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
raise RuntimeError(
|
raise RuntimeError(
|
||||||
"CREATE_FTS_INDEX failed (often ghost catalog after DROP INDEX). "
|
"CREATE_FTS_INDEX failed (often ghost catalog after DROP INDEX). "
|
||||||
"Delete var/kb.lbug and run bin/brain/index.go --rebuild. "
|
"Delete var/kb.lbug and run bin/kb/index --rebuild. "
|
||||||
f"Cause: {e}"
|
f"Cause: {e}"
|
||||||
) from e
|
) from e
|
||||||
if "Leaf_vec" not in names:
|
if "Leaf_vec" not in names:
|
||||||
@@ -299,7 +158,7 @@ def create_fts_and_vector(conn: ladybug.Connection, force: bool = False) -> None
|
|||||||
raise RuntimeError(
|
raise RuntimeError(
|
||||||
"CREATE_VECTOR_INDEX failed (often ghost catalog after DROP INDEX "
|
"CREATE_VECTOR_INDEX failed (often ghost catalog after DROP INDEX "
|
||||||
"Leaf.Leaf_vec → `_0_Leaf_vec_UPPER already exists in catalog`). "
|
"Leaf.Leaf_vec → `_0_Leaf_vec_UPPER already exists in catalog`). "
|
||||||
"Delete var/kb.lbug and run bin/brain/index.go --rebuild. "
|
"Delete var/kb.lbug and run bin/kb/index --rebuild. "
|
||||||
f"Cause: {e}"
|
f"Cause: {e}"
|
||||||
) from e
|
) from e
|
||||||
names = leaf_index_names(conn)
|
names = leaf_index_names(conn)
|
||||||
@@ -335,39 +194,29 @@ def drop_indexes(conn: ladybug.Connection) -> None:
|
|||||||
def query_fts(conn: ladybug.Connection, text: str, limit: int = 10) -> list[dict]:
|
def query_fts(conn: ladybug.Connection, text: str, limit: int = 10) -> list[dict]:
|
||||||
r = conn.execute(
|
r = conn.execute(
|
||||||
"CALL QUERY_FTS_INDEX('Leaf', 'id', $q) "
|
"CALL QUERY_FTS_INDEX('Leaf', 'id', $q) "
|
||||||
"RETURN node.id, node.text, node.root, score, node.valid_from, node.valid_to "
|
"RETURN node.id, node.text, node.root, score ORDER BY score DESC LIMIT $n",
|
||||||
"ORDER BY score DESC LIMIT $n",
|
|
||||||
parameters={"q": text, "n": limit},
|
parameters={"q": text, "n": limit},
|
||||||
)
|
)
|
||||||
return [
|
return [{"id": row[0], "text": row[1], "root": row[2], "score": row[3]} for row in r.get_all()]
|
||||||
{
|
|
||||||
"id": row[0], "text": row[1], "root": row[2], "score": row[3],
|
|
||||||
"valid_from": row[4] or "", "valid_to": row[5] or "",
|
|
||||||
}
|
|
||||||
for row in r.get_all()
|
|
||||||
]
|
|
||||||
|
|
||||||
|
|
||||||
def query_vector(conn: ladybug.Connection, embedding: list[float], limit: int = 10) -> list[dict]:
|
def query_vector(conn: ladybug.Connection, embedding: list[float], limit: int = 10) -> list[dict]:
|
||||||
r = conn.execute(
|
r = conn.execute(
|
||||||
"CALL QUERY_VECTOR_INDEX('Leaf', 'Leaf_vec', $q, $n) "
|
"CALL QUERY_VECTOR_INDEX('Leaf', 'Leaf_vec', $q, $n) "
|
||||||
"RETURN node.id, node.text, node.root, distance, node.valid_from, node.valid_to "
|
"RETURN node.id, node.text, node.root, distance ORDER BY distance LIMIT $n",
|
||||||
"ORDER BY distance LIMIT $n",
|
|
||||||
parameters={"q": embedding, "n": limit},
|
parameters={"q": embedding, "n": limit},
|
||||||
)
|
)
|
||||||
out = []
|
out = []
|
||||||
for row in r.get_all():
|
for row in r.get_all():
|
||||||
|
# distance -> similarity reasonable for cosine
|
||||||
score = 1.0 - row[3] if row[3] is not None else 0.0
|
score = 1.0 - row[3] if row[3] is not None else 0.0
|
||||||
out.append({
|
out.append({"id": row[0], "text": row[1], "root": row[2], "score": score})
|
||||||
"id": row[0], "text": row[1], "root": row[2], "score": score,
|
|
||||||
"valid_from": row[4] or "", "valid_to": row[5] or "",
|
|
||||||
})
|
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
def hybrid_search(conn: ladybug.Connection, embedding: list[float], fts_hits: list[dict],
|
def hybrid_search(conn: ladybug.Connection, embedding: list[float], fts_hits: list[dict],
|
||||||
limit: int = 10, as_of: str = "") -> list[dict]:
|
limit: int = 10) -> list[dict]:
|
||||||
"""Merge FTS + vector by reciprocal rank fusion; optional D24 as-of filter."""
|
"""Merge FTS + vector by reciprocal rank fusion."""
|
||||||
fused: dict[str, dict] = {}
|
fused: dict[str, dict] = {}
|
||||||
for rank, hit in enumerate(fts_hits):
|
for rank, hit in enumerate(fts_hits):
|
||||||
fused.setdefault(hit["id"], {**hit, "rrf": 0.0})["rrf"] = 1.0 / (60 + rank + 1)
|
fused.setdefault(hit["id"], {**hit, "rrf": 0.0})["rrf"] = 1.0 / (60 + rank + 1)
|
||||||
@@ -375,10 +224,8 @@ def hybrid_search(conn: ladybug.Connection, embedding: list[float], fts_hits: li
|
|||||||
entry = fused.setdefault(hit["id"], {**hit, "rrf": 0.0})
|
entry = fused.setdefault(hit["id"], {**hit, "rrf": 0.0})
|
||||||
entry["rrf"] += 1.0 / (60 + rank + 1)
|
entry["rrf"] += 1.0 / (60 + rank + 1)
|
||||||
entry.setdefault("score", hit.get("score", 0.0))
|
entry.setdefault("score", hit.get("score", 0.0))
|
||||||
entry.setdefault("valid_from", hit.get("valid_from") or "")
|
|
||||||
entry.setdefault("valid_to", hit.get("valid_to") or "")
|
|
||||||
ranked = sorted(fused.values(), key=lambda h: h.get("rrf", 0.0), reverse=True)
|
ranked = sorted(fused.values(), key=lambda h: h.get("rrf", 0.0), reverse=True)
|
||||||
return filter_as_of(ranked, as_of)[:limit]
|
return ranked[:limit]
|
||||||
|
|
||||||
|
|
||||||
def stats(conn: ladybug.Connection) -> dict:
|
def stats(conn: ladybug.Connection) -> dict:
|
||||||
@@ -390,6 +237,6 @@ def stats(conn: ladybug.Connection) -> dict:
|
|||||||
|
|
||||||
def open_readonly() -> tuple[ladybug.Database, ladybug.Connection]:
|
def open_readonly() -> tuple[ladybug.Database, ladybug.Connection]:
|
||||||
if not DB_PATH.exists():
|
if not DB_PATH.exists():
|
||||||
raise FileNotFoundError(f"{DB_PATH} missing - run bin/brain/index.go --rebuild first")
|
raise FileNotFoundError(f"{DB_PATH} missing - run bin/kb/index first")
|
||||||
db, conn = connect(read_only=True)
|
db, conn = connect(read_only=True)
|
||||||
return db, conn
|
return db, conn
|
||||||
+1
-84
@@ -7,10 +7,7 @@ offline against fixtures.
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import html
|
import html
|
||||||
import os
|
|
||||||
import re
|
import re
|
||||||
import subprocess
|
|
||||||
import tempfile
|
|
||||||
import zipfile
|
import zipfile
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -21,10 +18,9 @@ OFFICE_SUFFIXES = {".docx", ".pptx", ".xlsx", ".html", ".htm", ".epub", ".eml",
|
|||||||
PDF_SUFFIXES = {".pdf"}
|
PDF_SUFFIXES = {".pdf"}
|
||||||
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".tiff", ".tif", ".webp"}
|
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".tiff", ".tif", ".webp"}
|
||||||
ARCHIVE_SUFFIXES = {".zip"}
|
ARCHIVE_SUFFIXES = {".zip"}
|
||||||
# Legacy binary Office (doc/xls/ppt) — markitdown skip them; we try
|
# Legacy binary Office (doc/xls/ppt) — markitdown/docling skip them; we try
|
||||||
# pandoc first, else leave a stub.
|
# pandoc first, else leave a stub.
|
||||||
LEGACY_OFFICE_SUFFIXES = {".doc", ".xls", ".ppt"}
|
LEGACY_OFFICE_SUFFIXES = {".doc", ".xls", ".ppt"}
|
||||||
TESS_LANG = "eng+deu"
|
|
||||||
|
|
||||||
CONVERTIBLE_SUFFIXES = (
|
CONVERTIBLE_SUFFIXES = (
|
||||||
TEXT_SUFFIXES | OFFICE_SUFFIXES | PDF_SUFFIXES | IMAGE_SUFFIXES | ARCHIVE_SUFFIXES | LEGACY_OFFICE_SUFFIXES
|
TEXT_SUFFIXES | OFFICE_SUFFIXES | PDF_SUFFIXES | IMAGE_SUFFIXES | ARCHIVE_SUFFIXES | LEGACY_OFFICE_SUFFIXES
|
||||||
@@ -150,82 +146,3 @@ def zip_extract_safe(zip_path: Path, dest: Path) -> list[Path]:
|
|||||||
|
|
||||||
def is_convertible(suffix: str) -> bool:
|
def is_convertible(suffix: str) -> bool:
|
||||||
return suffix.lower() in CONVERTIBLE_SUFFIXES
|
return suffix.lower() in CONVERTIBLE_SUFFIXES
|
||||||
|
|
||||||
|
|
||||||
def convert_pdf(path: Path, ocr: bool = False) -> str:
|
|
||||||
"""pdftotext -layout first; empty text layer → pdftoppm + tesseract.
|
|
||||||
|
|
||||||
`ocr` is unused for born-digital PDFs (text layer wins). Scans OCR
|
|
||||||
automatically. This path never execs an ONNX document converter.
|
|
||||||
"""
|
|
||||||
del ocr # scans OCR when the text layer is empty; flag is for images
|
|
||||||
text = pdf_fast_text(path)
|
|
||||||
if text and text.strip():
|
|
||||||
return normalize_markdown(text)
|
|
||||||
scanned = ocr_pdf(path)
|
|
||||||
if scanned and scanned.strip():
|
|
||||||
return normalize_markdown(scanned)
|
|
||||||
if text:
|
|
||||||
return normalize_markdown(text)
|
|
||||||
return "\n<!-- pdf has no text layer (ocr unavailable) -->\n"
|
|
||||||
|
|
||||||
|
|
||||||
def pdf_fast_text(path: Path) -> str | None:
|
|
||||||
"""pdftotext -layout; None when poppler is missing or the command fails."""
|
|
||||||
try:
|
|
||||||
proc = subprocess.run(
|
|
||||||
["pdftotext", "-layout", str(path), "-"],
|
|
||||||
capture_output=True, timeout=60)
|
|
||||||
except (OSError, subprocess.TimeoutExpired):
|
|
||||||
return None
|
|
||||||
if proc.returncode != 0:
|
|
||||||
return None
|
|
||||||
return proc.stdout.decode("utf-8", errors="replace")
|
|
||||||
|
|
||||||
|
|
||||||
def ocr_pdf(path: Path) -> str:
|
|
||||||
"""Rasterize with pdftoppm and OCR each page (tesseract or paddle)."""
|
|
||||||
try:
|
|
||||||
with tempfile.TemporaryDirectory(prefix="2dph-ocr-") as tmp:
|
|
||||||
prefix = str(Path(tmp) / "page")
|
|
||||||
proc = subprocess.run(
|
|
||||||
["pdftoppm", "-png", "-r", "200", str(path), prefix],
|
|
||||||
capture_output=True, timeout=120)
|
|
||||||
if proc.returncode != 0:
|
|
||||||
return ""
|
|
||||||
pages = sorted(Path(tmp).glob("page*.png"))
|
|
||||||
parts = [ocr_image(p) for p in pages]
|
|
||||||
return "\n\n".join(p for p in parts if p and p.strip())
|
|
||||||
except (OSError, subprocess.TimeoutExpired):
|
|
||||||
return ""
|
|
||||||
|
|
||||||
|
|
||||||
def ocr_image(path: Path) -> str:
|
|
||||||
engine = os.environ.get("OCR_ENGINE", "tesseract")
|
|
||||||
if engine == "paddle":
|
|
||||||
return _ocr_paddle(path)
|
|
||||||
return _ocr_tesseract(path)
|
|
||||||
|
|
||||||
|
|
||||||
def _ocr_tesseract(path: Path) -> str:
|
|
||||||
try:
|
|
||||||
proc = subprocess.run(
|
|
||||||
["tesseract", str(path), "stdout", "-l", TESS_LANG, "--psm", "6"],
|
|
||||||
capture_output=True, timeout=120)
|
|
||||||
except (OSError, subprocess.TimeoutExpired):
|
|
||||||
return ""
|
|
||||||
if proc.returncode != 0:
|
|
||||||
return ""
|
|
||||||
return proc.stdout.decode("utf-8", errors="replace").strip()
|
|
||||||
|
|
||||||
|
|
||||||
def _ocr_paddle(path: Path) -> str:
|
|
||||||
try:
|
|
||||||
proc = subprocess.run(
|
|
||||||
["paddleocr", "ocr", "-i", str(path)],
|
|
||||||
capture_output=True, timeout=180)
|
|
||||||
except (OSError, subprocess.TimeoutExpired):
|
|
||||||
return ""
|
|
||||||
if proc.returncode != 0:
|
|
||||||
return ""
|
|
||||||
return proc.stdout.decode("utf-8", errors="replace").strip()
|
|
||||||
|
|||||||
@@ -1,37 +0,0 @@
|
|||||||
"""Mail markdown under var/mail → info leafs. Conversion stays off the brain DB."""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from mdleaves import read_markdown, to_all
|
|
||||||
|
|
||||||
|
|
||||||
def msg_date(md: Path) -> str:
|
|
||||||
j = md.parent / "message.json"
|
|
||||||
try:
|
|
||||||
d = json.loads(j.read_text(encoding="utf-8"))
|
|
||||||
return (d.get("receivedDate") or d.get("receivedAt") or "")[:10]
|
|
||||||
except (OSError, json.JSONDecodeError, TypeError):
|
|
||||||
return ""
|
|
||||||
|
|
||||||
|
|
||||||
def from_mail_root(root: Path, limit: int = 0, since: str = "", repo: str = "ooMail") -> list[dict]:
|
|
||||||
if not root.is_dir():
|
|
||||||
return []
|
|
||||||
mds = sorted(root.rglob("message.md"))
|
|
||||||
if since:
|
|
||||||
mds = [m for m in mds if msg_date(m) >= since]
|
|
||||||
if limit:
|
|
||||||
mds = mds[:limit]
|
|
||||||
leafs: list[dict] = []
|
|
||||||
for md in mds:
|
|
||||||
files = [md] + sorted((md.parent / "attachments").glob("*.md"))
|
|
||||||
for f in files:
|
|
||||||
if not f.exists():
|
|
||||||
continue
|
|
||||||
for lf in to_all(read_markdown(f), f, repo=repo):
|
|
||||||
lf["source"] = f"ooMail:{md.parent.name}:{f.name}"
|
|
||||||
lf["how"] = "mail/import"
|
|
||||||
leafs.append(lf)
|
|
||||||
return leafs
|
|
||||||
@@ -1,316 +0,0 @@
|
|||||||
"""D14 layout: bin/{subject}/{method}.go, libs in internal/, one go.mod."""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import os
|
|
||||||
import unittest
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[2]
|
|
||||||
|
|
||||||
|
|
||||||
class BinLayoutTest(unittest.TestCase):
|
|
||||||
def test_brain_search_shebang_exists(self) -> None:
|
|
||||||
p = ROOT / "bin" / "brain" / "search.go"
|
|
||||||
self.assertTrue(p.is_file(), "missing bin/brain/search.go")
|
|
||||||
first = p.read_text().splitlines()[0]
|
|
||||||
self.assertTrue(
|
|
||||||
first.startswith("//usr/bin/env go run"),
|
|
||||||
f"shebang first line, got {first!r}",
|
|
||||||
)
|
|
||||||
|
|
||||||
def test_no_nested_go_mod_under_bin(self) -> None:
|
|
||||||
nested = list((ROOT / "bin").rglob("go.mod"))
|
|
||||||
self.assertEqual(nested, [], f"nested go.mod files: {nested}")
|
|
||||||
|
|
||||||
def test_rank_lives_in_internal_brain(self) -> None:
|
|
||||||
self.assertTrue(
|
|
||||||
(ROOT / "internal" / "brain" / "rank" / "rank.go").is_file(),
|
|
||||||
"ranking must live in internal/brain/rank (cgo-free)",
|
|
||||||
)
|
|
||||||
self.assertFalse(
|
|
||||||
(ROOT / "bin" / "kbsearch").exists(),
|
|
||||||
"bin/kbsearch nested module must be gone",
|
|
||||||
)
|
|
||||||
|
|
||||||
def test_no_main_go_under_bin_brain(self) -> None:
|
|
||||||
main = ROOT / "bin" / "brain" / "main.go"
|
|
||||||
self.assertFalse(main.exists(), "bin/brain/main.go is not a method")
|
|
||||||
|
|
||||||
def test_chats_methods_are_shebangs_not_main(self) -> None:
|
|
||||||
chats = ROOT / "bin" / "chats"
|
|
||||||
self.assertFalse(
|
|
||||||
(chats / "main.go").exists(),
|
|
||||||
"bin/chats/main.go is a dispatcher, not a method",
|
|
||||||
)
|
|
||||||
self.assertFalse(
|
|
||||||
(chats / "index_cmd.go").exists(),
|
|
||||||
"chats index is a brain write hiding under the wrong subject",
|
|
||||||
)
|
|
||||||
for method in ("sync.go", "import.go", "facts.go", "apply.go"):
|
|
||||||
p = chats / method
|
|
||||||
self.assertTrue(p.is_file(), f"missing bin/chats/{method}")
|
|
||||||
first = p.read_text().splitlines()[0]
|
|
||||||
self.assertTrue(
|
|
||||||
first.startswith("//usr/bin/env go run"),
|
|
||||||
f"{method} shebang, got {first!r}",
|
|
||||||
)
|
|
||||||
|
|
||||||
def test_chats_lib_lives_in_internal(self) -> None:
|
|
||||||
self.assertTrue(
|
|
||||||
(ROOT / "internal" / "chats" / "linkedin.go").is_file(),
|
|
||||||
"LinkedIn parser must live in internal/chats",
|
|
||||||
)
|
|
||||||
self.assertFalse(
|
|
||||||
(ROOT / "bin" / "chats" / "linkedin.go").exists(),
|
|
||||||
"parser must not stay under bin/chats as a second main",
|
|
||||||
)
|
|
||||||
|
|
||||||
def _assert_shebang(self, rel: str) -> None:
|
|
||||||
p = ROOT / rel
|
|
||||||
self.assertTrue(p.is_file(), f"missing {rel}")
|
|
||||||
first = p.read_text().splitlines()[0]
|
|
||||||
self.assertTrue(
|
|
||||||
first.startswith("//usr/bin/env go run"),
|
|
||||||
f"{rel} shebang, got {first!r}",
|
|
||||||
)
|
|
||||||
|
|
||||||
def test_brain_methods_are_shebangs(self) -> None:
|
|
||||||
for method in ("index.go", "add.go", "get.go", "stats.go", "eval.go", "watch.go"):
|
|
||||||
self._assert_shebang(f"bin/brain/{method}")
|
|
||||||
|
|
||||||
def test_brain_add_is_python_write_not_rebuild(self) -> None:
|
|
||||||
self._assert_shebang("bin/brain/add.go")
|
|
||||||
text = (ROOT / "bin" / "brain" / "add.go").read_text()
|
|
||||||
self.assertIn("cmdbin.ExecFile", text)
|
|
||||||
self.assertIn("bin/kb/add", text)
|
|
||||||
self.assertNotIn("--rebuild", text)
|
|
||||||
py = (ROOT / "bin" / "kb" / "add").read_text()
|
|
||||||
self.assertIn("add_leafs", py)
|
|
||||||
self.assertIn("--json", py)
|
|
||||||
self.assertNotIn("unlink", py.lower())
|
|
||||||
|
|
||||||
def test_brain_get_stats_eval_are_not_python_exec(self) -> None:
|
|
||||||
for method in ("get.go", "stats.go", "eval.go"):
|
|
||||||
text = (ROOT / "bin" / "brain" / method).read_text()
|
|
||||||
self.assertNotIn(
|
|
||||||
"ExecFile",
|
|
||||||
text,
|
|
||||||
f"bin/brain/{method} must call internal/brain, not ExecFile Python",
|
|
||||||
)
|
|
||||||
self.assertNotIn(
|
|
||||||
"cmdbin",
|
|
||||||
text,
|
|
||||||
f"bin/brain/{method} must not import internal/cmdbin",
|
|
||||||
)
|
|
||||||
self.assertIn(
|
|
||||||
"system_ladybug",
|
|
||||||
text.splitlines()[0],
|
|
||||||
f"bin/brain/{method} shebang must pass -tags=system_ladybug",
|
|
||||||
)
|
|
||||||
self.assertIn(
|
|
||||||
"github.com/eSlider/2dph/internal/brain",
|
|
||||||
text,
|
|
||||||
)
|
|
||||||
|
|
||||||
def test_eval_control_questions_live_in_rank(self) -> None:
|
|
||||||
rank = (ROOT / "internal" / "brain" / "rank" / "evalq.go").read_text()
|
|
||||||
py = (ROOT / "bin" / "kb" / "eval").read_text()
|
|
||||||
for frag in ("BM25", "DevOps", "LadybugDB"):
|
|
||||||
self.assertIn(frag, rank)
|
|
||||||
self.assertIn(frag, py)
|
|
||||||
self.assertIn("0.95", rank)
|
|
||||||
|
|
||||||
def test_facts_methods_are_shebangs(self) -> None:
|
|
||||||
for method in ("audit.go", "extract.go", "crm.go"):
|
|
||||||
self._assert_shebang(f"bin/facts/{method}")
|
|
||||||
text = (ROOT / "bin" / "facts" / method).read_text()
|
|
||||||
self.assertIn("cmdbin.ExecFile", text)
|
|
||||||
self.assertIn(f"bin/facts/{method.removesuffix('.go')}", text)
|
|
||||||
|
|
||||||
def test_d16_adjudication_is_cgo_free(self) -> None:
|
|
||||||
self.assertTrue((ROOT / "internal" / "facts" / "contradict.go").is_file())
|
|
||||||
go = (ROOT / "internal" / "facts" / "contradict.go").read_text()
|
|
||||||
py = (ROOT / "bin" / "tools" / "contradict.py").read_text()
|
|
||||||
audit = (ROOT / "bin" / "facts" / "audit").read_text()
|
|
||||||
for token in ("temporal_freshness", "authority_pairing", "unresolved"):
|
|
||||||
self.assertIn(token, go)
|
|
||||||
self.assertIn(token, py)
|
|
||||||
self.assertIn("contradict", audit)
|
|
||||||
self.assertIn(" vs ", py)
|
|
||||||
plan = (ROOT / "PLAN.md").read_text()
|
|
||||||
self.assertIn("temporal_freshness", plan)
|
|
||||||
self.assertIn("authority_pairing", plan)
|
|
||||||
shebang = (ROOT / "bin" / "facts" / "audit.go").read_text()
|
|
||||||
self.assertIn("contradict", shebang)
|
|
||||||
|
|
||||||
def test_d23_flaggy_cli(self) -> None:
|
|
||||||
self.assertTrue((ROOT / "internal" / "cli" / "cli.go").is_file())
|
|
||||||
self.assertIn("github.com/integrii/flaggy", (ROOT / "go.mod").read_text())
|
|
||||||
plan = (ROOT / "PLAN.md").read_text()
|
|
||||||
self.assertIn("D23", plan)
|
|
||||||
self.assertIn("flaggy", plan)
|
|
||||||
complete = (ROOT / "bin" / "cli" / "complete.go").read_text()
|
|
||||||
first = complete.splitlines()[0]
|
|
||||||
self.assertTrue(first.startswith("//usr/bin/env go run"), first)
|
|
||||||
self.assertIn("complete.go bash", complete)
|
|
||||||
self.assertIn("brain-search", complete)
|
|
||||||
chats_import = (ROOT / "internal" / "chats" / "import.go").read_text()
|
|
||||||
self.assertNotIn("flag.NewFlagSet", chats_import)
|
|
||||||
args = (ROOT / "internal" / "brain" / "rank" / "args.go").read_text()
|
|
||||||
self.assertIn("internal/cli", args)
|
|
||||||
|
|
||||||
def test_mail_import_is_shebang_not_brain_write(self) -> None:
|
|
||||||
self._assert_shebang("bin/mail/import.go")
|
|
||||||
index_mail = (ROOT / "bin" / "mail" / "index_mail").read_text()
|
|
||||||
self.assertIn(
|
|
||||||
"bin/brain/index.go",
|
|
||||||
index_mail,
|
|
||||||
"index_mail must point at bin/brain/index.go",
|
|
||||||
)
|
|
||||||
|
|
||||||
def test_mail_ocr_is_tesseract_not_docling(self) -> None:
|
|
||||||
self._assert_shebang("bin/mail/ocr.go")
|
|
||||||
ocr = (ROOT / "bin" / "mail" / "ocr.go").read_text()
|
|
||||||
self.assertIn("internal/ocr", ocr)
|
|
||||||
self.assertIn("mail_ocr", ocr)
|
|
||||||
self.assertNotIn("github.com/otiai10/gosseract", ocr)
|
|
||||||
py = (ROOT / "bin" / "mail" / "import").read_text()
|
|
||||||
self.assertNotIn("from docling", py)
|
|
||||||
self.assertNotIn("import docling", py)
|
|
||||||
self.assertIn("convert_pdf", py)
|
|
||||||
conv = (ROOT / "bin" / "tools" / "mailconv.py").read_text()
|
|
||||||
self.assertIn("pdftotext", conv)
|
|
||||||
self.assertIn("pdftoppm", conv)
|
|
||||||
self.assertIn("tesseract", conv)
|
|
||||||
self.assertIn("eng+deu", conv)
|
|
||||||
self.assertNotIn("from docling", conv)
|
|
||||||
self.assertNotIn("import docling", conv)
|
|
||||||
self.assertNotIn("gocv", conv.lower())
|
|
||||||
proj = (ROOT / "pyproject.toml").read_text()
|
|
||||||
self.assertNotIn("docling", proj)
|
|
||||||
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
|
|
||||||
self.assertIn("tesseract-ocr", ci)
|
|
||||||
self.assertIn("./internal/ocr", ci)
|
|
||||||
compose = (ROOT / "compose.yaml").read_text()
|
|
||||||
self.assertIn("ocr-paddle", compose)
|
|
||||||
self.assertIn("OCR_ENGINE", compose)
|
|
||||||
|
|
||||||
def test_markdown_import_is_go_not_python_exec(self) -> None:
|
|
||||||
self._assert_shebang("bin/markdown/import.go")
|
|
||||||
text = (ROOT / "bin" / "markdown" / "import.go").read_text()
|
|
||||||
self.assertNotIn("ExecFile", text)
|
|
||||||
self.assertNotIn("cmdbin", text)
|
|
||||||
self.assertIn("internal/mdleaves", text)
|
|
||||||
self.assertNotIn("kb.lbug", text)
|
|
||||||
|
|
||||||
def test_import_adapters_do_not_write_ladybug(self) -> None:
|
|
||||||
for rel in (
|
|
||||||
"bin/mail/import.go",
|
|
||||||
"bin/mail/import",
|
|
||||||
"bin/markdown/import.go",
|
|
||||||
"bin/chats/import.go",
|
|
||||||
"bin/git/import.go",
|
|
||||||
):
|
|
||||||
text = (ROOT / rel).read_text()
|
|
||||||
self.assertNotIn("upsert_leaf", text, rel)
|
|
||||||
self.assertNotIn("kb.lbug", text, rel)
|
|
||||||
self.assertNotIn("var/brain.lbug", text, rel)
|
|
||||||
index = (ROOT / "bin" / "brain" / "index.go").read_text()
|
|
||||||
self.assertIn("bin/kb/index", index)
|
|
||||||
|
|
||||||
def test_postgres_query_is_shebang(self) -> None:
|
|
||||||
self._assert_shebang("bin/postgres/query.go")
|
|
||||||
|
|
||||||
def test_git_import_is_gogit_shebang(self) -> None:
|
|
||||||
self._assert_shebang("bin/git/import.go")
|
|
||||||
py = (ROOT / "bin" / "git" / "import").read_text()
|
|
||||||
self.assertNotIn(
|
|
||||||
'["git"',
|
|
||||||
py,
|
|
||||||
"Python git/import must not subprocess the git binary",
|
|
||||||
)
|
|
||||||
self.assertIn("bin/git/import.go", py)
|
|
||||||
|
|
||||||
def test_web_search_is_shebang(self) -> None:
|
|
||||||
self._assert_shebang("bin/web/search.go")
|
|
||||||
py = (ROOT / "bin" / "web" / "search").read_text()
|
|
||||||
self.assertIn("bin/web/search.go", py)
|
|
||||||
|
|
||||||
def test_gitimport_py_has_no_git_binary(self) -> None:
|
|
||||||
py = (ROOT / "bin" / "tools" / "gitimport.py").read_text()
|
|
||||||
self.assertNotIn("subprocess", py)
|
|
||||||
self.assertNotIn("git log", py)
|
|
||||||
|
|
||||||
def test_gogit_is_direct_go_mod_require(self) -> None:
|
|
||||||
text = (ROOT / "go.mod").read_text()
|
|
||||||
first = text.split("require (")[1].split(")")[0]
|
|
||||||
self.assertRegex(first, r"github.com/go-git/go-git/v5\s+v")
|
|
||||||
for line in first.splitlines():
|
|
||||||
if "go-git/go-git" in line:
|
|
||||||
self.assertNotIn("indirect", line)
|
|
||||||
|
|
||||||
def test_duckdb_go_is_direct_require(self) -> None:
|
|
||||||
text = (ROOT / "go.mod").read_text()
|
|
||||||
first = text.split("require (")[1].split(")")[0]
|
|
||||||
self.assertRegex(first, r"github.com/duckdb/duckdb-go/v2\s+v")
|
|
||||||
for line in first.splitlines():
|
|
||||||
if "duckdb/duckdb-go" in line:
|
|
||||||
self.assertNotIn("indirect", line)
|
|
||||||
skill = (ROOT / "skills" / "duckdb" / "SKILL.md").read_text()
|
|
||||||
self.assertIn("github.com/duckdb/duckdb-go", skill)
|
|
||||||
self.assertIn("Ladybug", skill)
|
|
||||||
self.assertIn("sqlite", skill.lower())
|
|
||||||
self.assertIn("gcc", skill.lower())
|
|
||||||
self.assertIn("Zig", skill)
|
|
||||||
plan = (ROOT / "PLAN.md").read_text()
|
|
||||||
self.assertIn("D22", plan)
|
|
||||||
self.assertIn("duckdb-go", plan)
|
|
||||||
self._assert_shebang("bin/qa/stats.go")
|
|
||||||
reasoner = (ROOT / "internal" / "reasoner" / "client.go").read_text()
|
|
||||||
self.assertNotIn("duckdb", reasoner)
|
|
||||||
self.assertNotIn("duckstats", reasoner)
|
|
||||||
bakeoff = (ROOT / "bin" / "reasoner" / "bakeoff.go").read_text()
|
|
||||||
self.assertIn("internal/duckstats", bakeoff)
|
|
||||||
webcache = (ROOT / "internal" / "websearch" / "cache.go").read_text()
|
|
||||||
self.assertNotIn("duckdb", webcache)
|
|
||||||
self.assertIn("modernc.org/sqlite", webcache)
|
|
||||||
|
|
||||||
def test_cgo_uses_zig_not_gcc(self) -> None:
|
|
||||||
for rel in ("bin/cgo/zig", "bin/cgo/zcc", "bin/cgo/zc++"):
|
|
||||||
p = ROOT / rel
|
|
||||||
self.assertTrue(p.is_file(), f"missing {rel}")
|
|
||||||
self.assertTrue(
|
|
||||||
os.access(p, os.X_OK),
|
|
||||||
f"{rel} must be executable",
|
|
||||||
)
|
|
||||||
zig = (ROOT / "bin" / "cgo" / "zig").read_text()
|
|
||||||
self.assertIn("zig cc", zig)
|
|
||||||
self.assertIn("0.14.1", zig)
|
|
||||||
zcc = (ROOT / "bin" / "cgo" / "zcc").read_text()
|
|
||||||
self.assertIn('exec "$ZIG" cc', zcc)
|
|
||||||
self.assertNotIn("command -v gcc", zcc)
|
|
||||||
search = (ROOT / "bin" / "kb" / "search").read_text()
|
|
||||||
self.assertIn("bin/cgo/zig", search)
|
|
||||||
self.assertNotIn("command -v gcc", search)
|
|
||||||
|
|
||||||
def test_ci_recall_sot_is_zig_brain_eval(self) -> None:
|
|
||||||
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
|
|
||||||
self.assertIn("bin/brain/eval.go", ci)
|
|
||||||
self.assertIn("system_ladybug,brain_eval", ci)
|
|
||||||
self.assertIn("/tmp/brain-eval", ci)
|
|
||||||
self.assertIn("KB_ROOT", ci)
|
|
||||||
self.assertNotIn("bin/kb/eval", ci)
|
|
||||||
self.assertNotIn("gate skipped", ci)
|
|
||||||
self.assertIn("./bin/facts/audit self", ci)
|
|
||||||
|
|
||||||
def test_eval_fragments_live_in_default_corpus(self) -> None:
|
|
||||||
"""CI --rebuild indexes README/PLAN/docs/skills; fragments must be there."""
|
|
||||||
corpus = []
|
|
||||||
for rel in ("README.md", "PLAN.md", "AGENTS.md"):
|
|
||||||
corpus.append((ROOT / rel).read_text())
|
|
||||||
for d in ("docs", "skills"):
|
|
||||||
for p in (ROOT / d).rglob("*.md"):
|
|
||||||
corpus.append(p.read_text())
|
|
||||||
blob = "\n".join(corpus)
|
|
||||||
for frag in ("BM25", "DevOps", "LadybugDB"):
|
|
||||||
self.assertIn(frag, blob, f"{frag} must appear in default index corpus")
|
|
||||||
@@ -1,104 +0,0 @@
|
|||||||
import os
|
|
||||||
import sys
|
|
||||||
import unittest
|
|
||||||
|
|
||||||
sys.path.insert(0, os.path.dirname(__file__))
|
|
||||||
|
|
||||||
from contradict import ( # noqa: E402
|
|
||||||
RULE_AUTHORITY,
|
|
||||||
RULE_SINGLE,
|
|
||||||
RULE_TEMPORAL,
|
|
||||||
RULE_TWO_SOURCE,
|
|
||||||
RULE_UNRESOLVED,
|
|
||||||
adjudicate,
|
|
||||||
check_fact_row,
|
|
||||||
parse_source_field,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def src(i, kind, stale=False):
|
|
||||||
return {"id": i, "kind": kind, "stale": stale}
|
|
||||||
|
|
||||||
|
|
||||||
class TestContradict(unittest.TestCase):
|
|
||||||
def test_two_vs_two_stays_hypothesis(self):
|
|
||||||
r = adjudicate({
|
|
||||||
"text": "svc listens on 443",
|
|
||||||
"yes": [src("docker-ps", "runtime"), src("compose", "config")],
|
|
||||||
"no": [src("docker-old", "runtime"), src("compose-old", "config")],
|
|
||||||
})
|
|
||||||
self.assertFalse(r["confirmed"])
|
|
||||||
self.assertEqual(r["rule"], RULE_UNRESOLVED)
|
|
||||||
self.assertEqual(r["winner"], "")
|
|
||||||
|
|
||||||
def test_temporal_freshness(self):
|
|
||||||
r = adjudicate({
|
|
||||||
"text": "svc listens on 443",
|
|
||||||
"yes": [src("docker-ps", "runtime"), src("compose", "config")],
|
|
||||||
"no": [src("old-readme", "narrative", True), src("old-wiki", "narrative", True)],
|
|
||||||
})
|
|
||||||
self.assertTrue(r["confirmed"])
|
|
||||||
self.assertEqual(r["rule"], RULE_TEMPORAL)
|
|
||||||
self.assertEqual(r["winner"], "yes")
|
|
||||||
|
|
||||||
def test_authority_pairing(self):
|
|
||||||
r = adjudicate({
|
|
||||||
"text": "svc listens on 443",
|
|
||||||
"yes": [src("docker-ps", "runtime"), src("compose", "config")],
|
|
||||||
"no": [src("readme", "narrative"), src("wiki", "narrative")],
|
|
||||||
})
|
|
||||||
self.assertTrue(r["confirmed"])
|
|
||||||
self.assertEqual(r["rule"], RULE_AUTHORITY)
|
|
||||||
self.assertEqual(r["winner"], "yes")
|
|
||||||
|
|
||||||
def test_two_source_and_single(self):
|
|
||||||
two = adjudicate({
|
|
||||||
"text": "arc-1 runs Matrix",
|
|
||||||
"yes": [src("compose", "config"), src("docker-ps", "runtime")],
|
|
||||||
})
|
|
||||||
self.assertTrue(two["confirmed"])
|
|
||||||
self.assertEqual(two["rule"], RULE_TWO_SOURCE)
|
|
||||||
one = adjudicate({"text": "maybe", "yes": [src("readme", "narrative")]})
|
|
||||||
self.assertFalse(one["confirmed"])
|
|
||||||
self.assertEqual(one["rule"], RULE_SINGLE)
|
|
||||||
|
|
||||||
def test_parse_source_field(self):
|
|
||||||
yes, no = parse_source_field("docker ps x compose.yml vs old.md x wiki.md")
|
|
||||||
self.assertIn(" x ", yes)
|
|
||||||
self.assertIn(" x ", no)
|
|
||||||
|
|
||||||
def test_check_fact_row_allows_hypothesis_vs(self):
|
|
||||||
p = check_fact_row(
|
|
||||||
"L1", "a.md x b.md vs c.md x d.md", "var/", "audit", "hypothesis",
|
|
||||||
)
|
|
||||||
self.assertEqual(p, [])
|
|
||||||
p = check_fact_row("L2", "a.md x b.md", "var/", "audit", "confirmed")
|
|
||||||
self.assertEqual(p, [])
|
|
||||||
p = check_fact_row("L3", "a.md x b.md vs c.md x d.md", "var/", "audit", "confirmed")
|
|
||||||
self.assertTrue(any("vs-contradiction" in x for x in p))
|
|
||||||
p = check_fact_row("L4", "only-one.md", "var/", "audit", "hypothesis")
|
|
||||||
self.assertTrue(any("a x b vs" in x for x in p))
|
|
||||||
|
|
||||||
def test_audit_contradict_cli_unresolved(self):
|
|
||||||
import json
|
|
||||||
import subprocess
|
|
||||||
from pathlib import Path
|
|
||||||
root = Path(__file__).resolve().parents[2]
|
|
||||||
payload = json.dumps({
|
|
||||||
"text": "svc 443",
|
|
||||||
"yes": [src("a", "runtime"), src("b", "config")],
|
|
||||||
"no": [src("c", "runtime"), src("d", "config")],
|
|
||||||
})
|
|
||||||
proc = subprocess.run(
|
|
||||||
[sys.executable, str(root / "bin" / "facts" / "audit"), "contradict", "--json"],
|
|
||||||
input=payload, capture_output=True, text=True, check=False,
|
|
||||||
)
|
|
||||||
self.assertEqual(proc.returncode, 0, proc.stderr)
|
|
||||||
out = json.loads(proc.stdout)
|
|
||||||
self.assertTrue(out["ok"])
|
|
||||||
self.assertEqual(out["contradictions"][0]["rule"], RULE_UNRESOLVED)
|
|
||||||
self.assertFalse(out["contradictions"][0]["confirmed"])
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
unittest.main()
|
|
||||||
+10
-13
@@ -9,6 +9,12 @@ sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|||||||
import kblib # noqa: E402
|
import kblib # noqa: E402
|
||||||
import gitimport # noqa: E402
|
import gitimport # noqa: E402
|
||||||
|
|
||||||
|
SAMPLE = (
|
||||||
|
"\x1e" + "a1b2c3d" + "\x1f" + "Ada Lovelace" + "\x1f" + "ada@example.com"
|
||||||
|
+ "\x1f" + "2026-08-10T12:00:00+01:00" + "\x1f" + "feat: first commit"
|
||||||
|
+ "\n\nREADME.md\nsrc/main.c\n"
|
||||||
|
)
|
||||||
|
|
||||||
COMMIT_PERSON_SCHEMA = (
|
COMMIT_PERSON_SCHEMA = (
|
||||||
"CREATE NODE TABLE IF NOT EXISTS Commit (id STRING, repo STRING, subject STRING, "
|
"CREATE NODE TABLE IF NOT EXISTS Commit (id STRING, repo STRING, subject STRING, "
|
||||||
"author STRING, email STRING, date STRING, PRIMARY KEY(id))"
|
"author STRING, email STRING, date STRING, PRIMARY KEY(id))"
|
||||||
@@ -20,17 +26,6 @@ HAS_VERSION_SCHEMA = "CREATE REL TABLE IF NOT EXISTS HAS_VERSION (FROM File TO C
|
|||||||
AUTHORED_SCHEMA = "CREATE REL TABLE IF NOT EXISTS AUTHORED (FROM Commit TO Person)"
|
AUTHORED_SCHEMA = "CREATE REL TABLE IF NOT EXISTS AUTHORED (FROM Commit TO Person)"
|
||||||
|
|
||||||
|
|
||||||
def sample_commit() -> gitimport.Commit:
|
|
||||||
return gitimport.Commit(
|
|
||||||
sha="a1b2c3d",
|
|
||||||
author="Ada Lovelace",
|
|
||||||
email="ada@example.com",
|
|
||||||
date="2026-08-10T12:00:00+01:00",
|
|
||||||
subject="feat: first commit",
|
|
||||||
files=["README.md", "src/main.c"],
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
class GitGraphTest(unittest.TestCase):
|
class GitGraphTest(unittest.TestCase):
|
||||||
def setUp(self):
|
def setUp(self):
|
||||||
self.dir = tempfile.mkdtemp()
|
self.dir = tempfile.mkdtemp()
|
||||||
@@ -47,12 +42,14 @@ class GitGraphTest(unittest.TestCase):
|
|||||||
self.db.close()
|
self.db.close()
|
||||||
|
|
||||||
def test_index_commits_creates_nodes_and_edges(self):
|
def test_index_commits_creates_nodes_and_edges(self):
|
||||||
gitimport.index_commits(self.conn, [sample_commit()], "sample-repo")
|
cs = gitimport.parse_log(SAMPLE)
|
||||||
|
gitimport.index_commits(self.conn, cs, "sample-repo")
|
||||||
rp = self.conn.execute("MATCH (p:Person) RETURN p.name, p.email").get_all()
|
rp = self.conn.execute("MATCH (p:Person) RETURN p.name, p.email").get_all()
|
||||||
self.assertEqual([tuple(r) for r in rp], [("Ada Lovelace", "ada@example.com")])
|
self.assertEqual([tuple(r) for r in rp], [("Ada Lovelace", "ada@example.com")])
|
||||||
rc = self.conn.execute("MATCH (c:Commit) RETURN c.id, c.repo").get_all()
|
rc = self.conn.execute("MATCH (c:Commit) RETURN c.id, c.repo").get_all()
|
||||||
self.assertEqual(len(rc), 1)
|
self.assertEqual(len(rc), 1)
|
||||||
self.assertEqual(rc[0][1], "sample-repo")
|
self.assertEqual(rc[0][1], "sample-repo")
|
||||||
|
# File -[:HAS_VERSION]-> Commit -[:AUTHORED]-> Person
|
||||||
rf = self.conn.execute(
|
rf = self.conn.execute(
|
||||||
"MATCH (f:File)-[:HAS_VERSION]->(c:Commit)-[:AUTHORED]->(p:Person) "
|
"MATCH (f:File)-[:HAS_VERSION]->(c:Commit)-[:AUTHORED]->(p:Person) "
|
||||||
"RETURN f.path, c.id, p.email").get_all()
|
"RETURN f.path, c.id, p.email").get_all()
|
||||||
@@ -61,7 +58,7 @@ class GitGraphTest(unittest.TestCase):
|
|||||||
self.assertTrue(all(r[2] == "ada@example.com" for r in rf))
|
self.assertTrue(all(r[2] == "ada@example.com" for r in rf))
|
||||||
|
|
||||||
def test_index_commits_idempotent(self):
|
def test_index_commits_idempotent(self):
|
||||||
cs = [sample_commit()]
|
cs = gitimport.parse_log(SAMPLE)
|
||||||
gitimport.index_commits(self.conn, cs, "sample-repo")
|
gitimport.index_commits(self.conn, cs, "sample-repo")
|
||||||
gitimport.index_commits(self.conn, cs, "sample-repo")
|
gitimport.index_commits(self.conn, cs, "sample-repo")
|
||||||
n = self.conn.execute("MATCH (c:Commit) RETURN count(*)").get_all()[0][0]
|
n = self.conn.execute("MATCH (c:Commit) RETURN count(*)").get_all()[0][0]
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user