feat: CPU reasoner bake-off for Qwen3.5-9B vs Bonsai (#25)
Measure OpenAI tool_calls (search/get/audit) on a CPU Ollama sidecar instead of gating D18 on GPU or PicoClaw. Weights stay out of the image.
This commit is contained in:
@@ -11,3 +11,4 @@ __pycache__/
|
||||
.secrets/
|
||||
lib-ladybug/
|
||||
go.work.local
|
||||
models/
|
||||
|
||||
@@ -46,7 +46,8 @@ bin/markdown/ import.go (mistune leafs)
|
||||
bin/postgres/ query.go (read-only YAML)
|
||||
bin/git/ import.go (go-git history; Python shim execs it)
|
||||
bin/web/ search.go (SearXNG; Python shim execs it)
|
||||
internal/ shared Go (brain/rank is cgo-free; chats parsers; gitlog; websearch)
|
||||
bin/reasoner/ bakeoff.go (D18 CPU OpenAI tool-call bake-off)
|
||||
internal/ shared Go (brain/rank is cgo-free; chats parsers; gitlog; websearch; reasoner)
|
||||
bin/watch/ corpus watcher (used by bin/brain/watch.go)
|
||||
bin/tools/ vendored python libs behind bin/* (kblib, yamlout, websearch)
|
||||
bin/cgo/ zig zcc zc++ (CGO via zig cc, not gcc)
|
||||
@@ -94,6 +95,7 @@ bin/brain/serve.go # HTTP :8630; GET /openapi.json
|
||||
bin/markdown/import.go [dir] # mistune leaves → YAML
|
||||
bin/git/import.go [REPO] [--json] [--limit N] # go-git history → commit leafs
|
||||
bin/web/search.go "query" [--json] # SearXNG; throttled ≠ absence
|
||||
bin/reasoner/bakeoff.go [--model ID] [--json] # D18 CPU tool-call bake-off
|
||||
bin/postgres/query.go --profile onlyoffice -c 'SELECT 1'
|
||||
bin/md/tables # what the graph holds → YAML
|
||||
bin/brain/deduce "question" # thinking wrapper
|
||||
|
||||
@@ -41,7 +41,7 @@ detective method: **a fact needs ≥2 independent sources or it is
|
||||
| D15 | repo | Gitea [`eSlider/2dph`](https://git.produktor.io/eSlider/2dph) is origin + [issues](https://git.produktor.io/eSlider/2dph/issues). GitHub `eSlider/2dph` is the public clone (PRs + Actions CI). No direct `main` pushes. TDD → PR → CI green → merge. |
|
||||
| D16 | contradictions | ≥2 yes vs ≥2 no → unrelated sources conflict → hypothesis → `(not confirmed)`. Resolution (authority, staleness adjudication) = **v2**, tracked as open question. |
|
||||
| D17 | assertion gate | Fact-check every *claim* (facts → info → live → web), not every edit. `bin/brain/search.go` adds a `web` block when there is no facts hit (`throttled`/`skipped`/`refused` ≠ absence). `--root` and `--no-web` stay local. Missing graph ≠ “does not exist”. |
|
||||
| D18 | reasoner | Pluggable OpenAI-compatible URL. RAM: Qwen3.5-9B. Quality: Bonsai-27B or Qwen3.6-27B. No official Qwen3.6-9B. |
|
||||
| D18 | reasoner | Pluggable OpenAI-compatible URL (`REASONER_BASE_URL`). RAM: `Qwen/Qwen3.5-9B`. Quality: `prism-ml/Bonsai-27B-gguf` or `Qwen/Qwen3.6-27B`. No official Qwen3.6-9B. CPU bake-off: `bin/reasoner/bakeoff.go` + compose profile `reasoner` (`OLLAMA_NUM_GPU=0`, `:11435`). PicoClaw is not shipped; tools are `search`/`get`/`audit`. Weights are not copied into the 2dph image. |
|
||||
| D19 | git history | [go-git](https://github.com/go-git/go-git) via `bin/git/import.go`. No subprocess of the git binary. Conversion prints commit leafs; brain write is `bin/brain/index.go`. |
|
||||
| D20 | agent API | OpenAPI + MCP are generated from the same `internal/httpapi.Ops` table as `bin/brain/serve.go` handlers. `GET /openapi.json`, `POST /mcp` (JSON-RPC tools/list + tools/call). Tool names match OpenAPI paths (`search`/`get`/`stats`/`audit`). |
|
||||
| D21 | CGO | Ladybug/tokenizers CGO is compiled with **Zig** (`bin/cgo/zcc` → `zig cc -target …-linux-gnu`), not gcc. `bin/cgo/zig` pins Zig 0.14.1 + liblbug 0.19.1 + libtokenizers 1.27.0. Compose `target: api` has no CPython; write/rebuild is profile `index`. |
|
||||
@@ -67,6 +67,7 @@ detective method: **a fact needs ≥2 independent sources or it is
|
||||
postgres/query.go read-only YAML (wraps bin/db/psql-yq)
|
||||
git/import.go go-git history (no git binary; conversion only)
|
||||
web/search.go SearXNG client (throttled ≠ absence)
|
||||
reasoner/bakeoff.go CPU tool-call bake-off (D18; OpenAI tools)
|
||||
chats/sync.go import.go facts.go apply.go
|
||||
(libs in internal/chats; no chats index)
|
||||
md/import (deprecated; bin/markdown/import.go)
|
||||
|
||||
@@ -158,6 +158,7 @@ Docker (optional, cached model + var volumes):
|
||||
docker compose up -d brain # API (Zig CGO serve :8630)
|
||||
docker compose --profile index run --rm index # Python Ladybug rebuild
|
||||
docker compose --profile picoclaw up brain-mcp # MCP on 127.0.0.1:8630
|
||||
docker compose --profile reasoner up -d reasoner # CPU Ollama 127.0.0.1:11435
|
||||
docker compose up brain-watch # auto re-index on change
|
||||
```
|
||||
|
||||
|
||||
Executable
+91
@@ -0,0 +1,91 @@
|
||||
//usr/bin/env go run -tags=reasoner_bakeoff "$0" "$@"; exit
|
||||
//go:build reasoner_bakeoff
|
||||
//
|
||||
// bin/reasoner/bakeoff.go - CPU tool-call bake-off against an OpenAI-compatible URL (D18).
|
||||
//
|
||||
// REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b ./bin/reasoner/bakeoff.go
|
||||
// ./bin/reasoner/bakeoff.go --model MichelRosselli/bonsai-27b:Q1_0 --json
|
||||
//
|
||||
// Measures OpenAI tool_calls (search/get/audit) and RSS from Ollama /api/ps, not VRAM.
|
||||
// PicoClaw is not in this repo; the tool names match internal/httpapi MCP ops.
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"github.com/eSlider/2dph/internal/reasoner"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(run(os.Args[1:]))
|
||||
}
|
||||
|
||||
func run(args []string) int {
|
||||
base := os.Getenv("REASONER_BASE_URL")
|
||||
if base == "" {
|
||||
base = "http://127.0.0.1:11435/v1"
|
||||
}
|
||||
model := os.Getenv("REASONER_MODEL")
|
||||
if model == "" {
|
||||
model = reasoner.OllamaRAM
|
||||
}
|
||||
jsonOut := false
|
||||
device := "cpu"
|
||||
for i := 0; i < len(args); i++ {
|
||||
a := args[i]
|
||||
switch {
|
||||
case a == "--json":
|
||||
jsonOut = true
|
||||
case a == "--model" && i+1 < len(args):
|
||||
i++
|
||||
model = args[i]
|
||||
case strings.HasPrefix(a, "--model="):
|
||||
model = strings.TrimPrefix(a, "--model=")
|
||||
case a == "--base-url" && i+1 < len(args):
|
||||
i++
|
||||
base = args[i]
|
||||
case a == "--device" && i+1 < len(args):
|
||||
i++
|
||||
device = args[i]
|
||||
case a == "-h" || a == "--help":
|
||||
fmt.Fprintln(os.Stderr, "bin/reasoner/bakeoff.go [--model ID] [--base-url URL] [--device cpu] [--json]")
|
||||
return 0
|
||||
default:
|
||||
fmt.Fprintln(os.Stderr, "unknown arg:", a)
|
||||
return 2
|
||||
}
|
||||
}
|
||||
c := reasoner.Client{BaseURL: base, Model: model, Device: device}
|
||||
rep := reasoner.Run(c)
|
||||
raw, err := json.MarshalIndent(rep, "", " ")
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
return 1
|
||||
}
|
||||
if jsonOut {
|
||||
fmt.Println(string(raw))
|
||||
} else {
|
||||
fmt.Printf("model: %s\n", rep.Model)
|
||||
fmt.Printf("hf_id: %s\n", rep.HF)
|
||||
fmt.Printf("device: %s\n", rep.Device)
|
||||
fmt.Printf("tool_call: %d/%d\n", rep.ToolCallOK, rep.ToolCallN)
|
||||
fmt.Printf("xml_leak: %d\n", rep.XMLLeak)
|
||||
fmt.Printf("rss_mb: %d\n", rep.RSSMB)
|
||||
fmt.Printf("vram_mb: %d\n", rep.VRAMMB)
|
||||
for _, p := range rep.Prompts {
|
||||
status := "fail"
|
||||
if p.OK {
|
||||
status = "ok"
|
||||
}
|
||||
fmt.Printf(" %s: %s wanted=%s got=%s xml=%v %dms %s\n", p.WantedTool, status, p.WantedTool, p.ToolName, p.XMLLeak, p.LatencyMS, p.Err)
|
||||
}
|
||||
}
|
||||
if rep.ToolCallN == 0 {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
// Commands in this directory are shebang mains (bakeoff.go).
|
||||
package main
|
||||
@@ -103,6 +103,35 @@ class PublishedDocsTest(unittest.TestCase):
|
||||
self.assertIn('profiles: ["index"]', compose)
|
||||
self.assertIn("target: api", compose)
|
||||
|
||||
def test_reasoner_docs_name_real_hf_ids_cpu_sidecar(self) -> None:
|
||||
docs = (ROOT / "docs" / "reasoner.md").read_text()
|
||||
for hf in (
|
||||
"Qwen/Qwen3.5-9B",
|
||||
"Qwen/Qwen3.6-27B",
|
||||
"prism-ml/Bonsai-27B-gguf",
|
||||
):
|
||||
self.assertIn(hf, docs)
|
||||
self.assertIn("no official qwen3.6-9b", docs.lower())
|
||||
self.assertIn("OLLAMA_NUM_GPU", docs)
|
||||
self.assertIn("rss_mb", docs)
|
||||
self.assertIn("vram_mb", docs)
|
||||
self.assertIn("3/3", docs)
|
||||
self.assertIn("Do not claim 9B is better at tools", docs)
|
||||
self.assertNotIn("Qwen/Qwen3.6-9B", docs)
|
||||
plan = (ROOT / "PLAN.md").read_text()
|
||||
self.assertIn("D18", plan)
|
||||
self.assertIn("Qwen/Qwen3.5-9B", plan)
|
||||
compose = (ROOT / "compose.yaml").read_text()
|
||||
self.assertIn('profiles: ["reasoner"]', compose)
|
||||
self.assertIn("OLLAMA_NUM_GPU", compose)
|
||||
self.assertIn("127.0.0.1:11435", compose)
|
||||
dockerfile = (ROOT / "Dockerfile").read_text()
|
||||
self.assertNotIn(".gguf", dockerfile.lower())
|
||||
self.assertNotIn(".safetensors", dockerfile.lower())
|
||||
api = dockerfile[dockerfile.index("FROM debian:bookworm-slim AS api") :]
|
||||
self.assertNotIn("COPY models", api)
|
||||
self.assertNotIn("qwen", api.lower())
|
||||
|
||||
def test_readme_search_escalates_web(self) -> None:
|
||||
text = (ROOT / "README.md").read_text()
|
||||
self.assertIn("--no-web", text)
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
# docker compose up -d brain # API (Zig CGO serve)
|
||||
# docker compose --profile index run --rm index # Python rebuild
|
||||
# docker compose --profile picoclaw up brain-mcp
|
||||
# docker compose --profile reasoner up -d reasoner # CPU Ollama :11435
|
||||
# docker compose --profile searxng up -d
|
||||
#
|
||||
# Secrets never baked in: search.env + db-profiles.yml from ~/.config/brain.
|
||||
@@ -114,6 +115,23 @@ services:
|
||||
- /tmp
|
||||
restart: unless-stopped
|
||||
|
||||
# CPU OpenAI-compatible sidecar (D18). Weights are pulled at runtime, not
|
||||
# baked into the 2dph image. Does not touch host Ollama on :11434.
|
||||
# docker compose --profile reasoner up -d reasoner
|
||||
# docker compose --profile reasoner exec reasoner ollama pull qwen3.5:9b
|
||||
reasoner:
|
||||
profiles: ["reasoner"]
|
||||
image: docker.io/ollama/ollama:latest
|
||||
environment:
|
||||
OLLAMA_NUM_GPU: "0"
|
||||
OLLAMA_HOST: "0.0.0.0:11434"
|
||||
ports:
|
||||
- "127.0.0.1:11435:11434"
|
||||
volumes:
|
||||
- reasoner-ollama:/root/.ollama
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
kb-model:
|
||||
kb-var:
|
||||
reasoner-ollama:
|
||||
|
||||
@@ -6,6 +6,7 @@ Brain/ops/eSlider stack. Facts need proof or they are
|
||||
|
||||
- [PLAN.md](../PLAN.md) — decisions, execution order, open questions (v2)
|
||||
- [design](design.md) — schema, deduction model, sources
|
||||
- [reasoner](reasoner.md) — D18 CPU bake-off (Qwen3.5-9B vs Bonsai / Qwen3.6-27B)
|
||||
- [Gitea issues](https://git.produktor.io/eSlider/2dph/issues) — work board (origin)
|
||||
|
||||
Search: `bin/brain/search.go "query"` (HTTP: `bin/brain/serve.go` —
|
||||
|
||||
+9
-1
@@ -78,4 +78,12 @@ fetches Zig + libs (`bin/cgo/zig`). Index/write is still `bin/kb/index`
|
||||
`bin/brain/serve.go` exposes the same `internal/httpapi.Ops` table as OpenAPI
|
||||
(`GET /openapi.json`) and MCP (`POST /mcp` JSON-RPC `tools/list` +
|
||||
`tools/call`). Tool names match paths: `search`, `get`, `stats`, `audit`.
|
||||
Agents should use these endpoints instead of shebang CLIs.
|
||||
Agents should use these endpoints instead of shebang CLIs.
|
||||
|
||||
## Reasoner (D18)
|
||||
|
||||
Pluggable OpenAI-compatible URL. RAM: `Qwen/Qwen3.5-9B`. Quality:
|
||||
`prism-ml/Bonsai-27B-gguf` or `Qwen/Qwen3.6-27B`. No official Qwen3.6-9B.
|
||||
CPU sidecar: compose profile `reasoner` (`OLLAMA_NUM_GPU=0`,
|
||||
`127.0.0.1:11435`). Bake-off: `bin/reasoner/bakeoff.go`. Weights stay out
|
||||
of the 2dph image. See [docs/reasoner.md](reasoner.md).
|
||||
@@ -0,0 +1,69 @@
|
||||
# Reasoner bake-off (D18)
|
||||
|
||||
Pluggable OpenAI-compatible URL. 2dph does not ship weights. PicoClaw is
|
||||
not in this repo; the bake-off hits the same tool names PicoClaw would
|
||||
(`search` → `get` → `audit` from `internal/httpapi.Ops`).
|
||||
|
||||
```bash
|
||||
docker compose --profile reasoner up -d reasoner
|
||||
docker compose --profile reasoner exec reasoner ollama pull qwen3.5:9b
|
||||
REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b \
|
||||
./bin/reasoner/bakeoff.go --json
|
||||
```
|
||||
|
||||
Host Ollama on `:11434` is left alone. This sidecar binds `127.0.0.1:11435`
|
||||
with `OLLAMA_NUM_GPU=0` (CPU). Measure RSS (`/api/ps` `size`), not VRAM.
|
||||
|
||||
If Compose cannot allocate a project network (Docker IPAM pool exhausted),
|
||||
the same sidecar is:
|
||||
|
||||
```bash
|
||||
docker run -d --name 2dph-reasoner \
|
||||
-e OLLAMA_NUM_GPU=0 \
|
||||
-p 127.0.0.1:11435:11434 \
|
||||
-v 2dph-reasoner-ollama:/root/.ollama \
|
||||
ollama/ollama:latest
|
||||
```
|
||||
|
||||
## Real Hugging Face ids
|
||||
|
||||
| Role | HF id | Ollama tag (this bake-off) |
|
||||
|------|-------|----------------------------|
|
||||
| RAM / 9B | `Qwen/Qwen3.5-9B` | `qwen3.5:9b` |
|
||||
| Quality 27B (CPU) | `prism-ml/Bonsai-27B-gguf` (derived from Qwen3.6-27B) | `MichelRosselli/bonsai-27b:Q1_0` |
|
||||
| Quality 27B (full) | `Qwen/Qwen3.6-27B` | not pulled on this CPU box |
|
||||
|
||||
There is **no official Qwen3.6-9B**. Do not invent that id.
|
||||
|
||||
Qwen3.5-9B has documented upstream tool-call XML bugs. A 9B win on tools
|
||||
is only claimed if this bake-off records OpenAI `tool_calls` (not
|
||||
`<tool_call>` XML in `content`).
|
||||
|
||||
The 2dph API image does not `COPY` GGUF/safetensors. Pull at runtime into
|
||||
the `reasoner-ollama` volume.
|
||||
|
||||
## Live CPU run
|
||||
|
||||
Host sidecar: Ollama **0.32.9**, `OLLAMA_NUM_GPU=0`, `127.0.0.1:11435`,
|
||||
`device: cpu`, `vram_mb: 0`. Date: 2026-08-13. Same three prompts
|
||||
(`search` / `get` / `audit`). PicoClaw binary was not used; the OpenAI
|
||||
tools payload is the surface it would send.
|
||||
|
||||
| Model | HF id | tool_call | xml_leak | rss_mb | latency_ms (search/get/audit) |
|
||||
|-------|-------|-----------|----------|--------|-------------------------------|
|
||||
| `qwen3.5:9b` | `Qwen/Qwen3.5-9B` | 3/3 | 0 | 5790 | 50059 / 51678 / 31980 |
|
||||
| `MichelRosselli/bonsai-27b:Q1_0` | `prism-ml/Bonsai-27B-gguf` | 3/3 | 0 | 21951 | 327702 / 166830 / 119808 |
|
||||
| `Qwen/Qwen3.6-27B` | `Qwen/Qwen3.6-27B` | not loaded | — | — | too heavy for this CPU box |
|
||||
|
||||
Both loaded models emitted OpenAI `tool_calls` (not `<tool_call>` XML) on
|
||||
this runtime. **Do not claim 9B is better at tools** — the score is tied
|
||||
at 3/3. 9B is smaller and faster. Bonsai RSS includes weights + KV
|
||||
(`size` from `/api/ps`); first Bonsai prompt includes cold load.
|
||||
|
||||
Re-run:
|
||||
|
||||
```bash
|
||||
REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b \
|
||||
./bin/reasoner/bakeoff.go --json
|
||||
REASONER_MODEL=MichelRosselli/bonsai-27b:Q1_0 ./bin/reasoner/bakeoff.go --json
|
||||
```
|
||||
@@ -0,0 +1,325 @@
|
||||
package reasoner
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// HF IDs named in docs. No Qwen3.6-9B exists.
|
||||
const (
|
||||
HFQwen35_9B = "Qwen/Qwen3.5-9B"
|
||||
HFQwen36_27B = "Qwen/Qwen3.6-27B"
|
||||
HFBonsai27B = "prism-ml/Bonsai-27B-gguf"
|
||||
OllamaRAM = "qwen3.5:9b"
|
||||
OllamaQuality = "MichelRosselli/bonsai-27b:Q1_0"
|
||||
)
|
||||
|
||||
type ToolCall struct {
|
||||
Name string
|
||||
Arguments string
|
||||
}
|
||||
|
||||
type Result struct {
|
||||
Model string `json:"model"`
|
||||
OK bool `json:"ok"`
|
||||
ToolName string `json:"tool_name,omitempty"`
|
||||
XMLLeak bool `json:"xml_leak"`
|
||||
Err string `json:"error,omitempty"`
|
||||
LatencyMS int64 `json:"latency_ms"`
|
||||
RSSMB int `json:"rss_mb,omitempty"`
|
||||
Device string `json:"device"`
|
||||
WantedTool string `json:"wanted_tool"`
|
||||
}
|
||||
|
||||
type Prompt struct {
|
||||
Name string
|
||||
Want string
|
||||
User string
|
||||
}
|
||||
|
||||
var BakePrompts = []Prompt{
|
||||
{
|
||||
Name: "search-before-claim",
|
||||
Want: "search",
|
||||
User: "Use tools. Search the 2dph brain for LadybugDB before you answer. Call search.",
|
||||
},
|
||||
{
|
||||
Name: "get-leaf",
|
||||
Want: "get",
|
||||
User: "Use tools. Fetch leaf id leaf-demo with get. Do not invent the body.",
|
||||
},
|
||||
{
|
||||
Name: "audit-index",
|
||||
Want: "audit",
|
||||
User: "Use tools. Call audit on the brain index health.",
|
||||
},
|
||||
}
|
||||
|
||||
func MCPTools() []map[string]any {
|
||||
return []map[string]any{
|
||||
openaiTool("search", "deduction search (facts → info → web)", map[string]any{
|
||||
"type": "object",
|
||||
"properties": map[string]any{
|
||||
"q": map[string]any{"type": "string", "description": "search query"},
|
||||
"n": map[string]any{"type": "integer"},
|
||||
},
|
||||
"required": []string{"q"},
|
||||
}),
|
||||
openaiTool("get", "read one leaf by id", map[string]any{
|
||||
"type": "object",
|
||||
"properties": map[string]any{
|
||||
"id": map[string]any{"type": "string"},
|
||||
"body": map[string]any{"type": "boolean"},
|
||||
},
|
||||
"required": []string{"id"},
|
||||
}),
|
||||
openaiTool("audit", "facts confidence histogram", map[string]any{
|
||||
"type": "object",
|
||||
"properties": map[string]any{},
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
func openaiTool(name, desc string, schema map[string]any) map[string]any {
|
||||
return map[string]any{
|
||||
"type": "function",
|
||||
"function": map[string]any{
|
||||
"name": name,
|
||||
"description": desc,
|
||||
"parameters": schema,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
type Client struct {
|
||||
BaseURL string
|
||||
Model string
|
||||
HTTP *http.Client
|
||||
Device string
|
||||
ToolChoice string
|
||||
}
|
||||
|
||||
type Report struct {
|
||||
Model string `json:"model"`
|
||||
HF string `json:"hf_id"`
|
||||
Device string `json:"device"`
|
||||
ToolCallOK int `json:"tool_call_ok"`
|
||||
ToolCallN int `json:"tool_call_n"`
|
||||
XMLLeak int `json:"xml_leak"`
|
||||
RSSMB int `json:"rss_mb"`
|
||||
VRAMMB int `json:"vram_mb"`
|
||||
Prompts []Result `json:"prompts"`
|
||||
}
|
||||
|
||||
func HFFor(model string) string {
|
||||
switch model {
|
||||
case OllamaRAM:
|
||||
return HFQwen35_9B
|
||||
case OllamaQuality:
|
||||
return HFBonsai27B
|
||||
default:
|
||||
if strings.Contains(model, "qwen3.6") || strings.Contains(model, "Qwen3.6") {
|
||||
return HFQwen36_27B
|
||||
}
|
||||
return ""
|
||||
}
|
||||
}
|
||||
|
||||
func (c *Client) httpc() *http.Client {
|
||||
if c.HTTP == nil {
|
||||
c.HTTP = &http.Client{Timeout: 10 * time.Minute}
|
||||
}
|
||||
return c.HTTP
|
||||
}
|
||||
|
||||
func Origin(base string) string {
|
||||
s := strings.TrimRight(base, "/")
|
||||
return strings.TrimSuffix(s, "/v1")
|
||||
}
|
||||
|
||||
func (c Client) ChatTools(user string) (ToolCall, string, error) {
|
||||
choice := c.ToolChoice
|
||||
if choice == "" {
|
||||
choice = "required"
|
||||
}
|
||||
body, _ := json.Marshal(map[string]any{
|
||||
"model": c.Model,
|
||||
"messages": []map[string]string{
|
||||
{"role": "system", "content": "You are PicoClaw talking to 2dph MCP. Always call a tool before a factual claim. search then get then audit."},
|
||||
{"role": "user", "content": user},
|
||||
},
|
||||
"tools": MCPTools(),
|
||||
"tool_choice": choice,
|
||||
})
|
||||
base := strings.TrimRight(c.BaseURL, "/")
|
||||
if !strings.HasSuffix(base, "/v1") {
|
||||
base += "/v1"
|
||||
}
|
||||
url := base + "/chat/completions"
|
||||
req, err := http.NewRequest(http.MethodPost, url, bytes.NewReader(body))
|
||||
if err != nil {
|
||||
return ToolCall{}, "", err
|
||||
}
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
res, err := c.httpc().Do(req)
|
||||
if err != nil {
|
||||
return ToolCall{}, "", err
|
||||
}
|
||||
defer res.Body.Close()
|
||||
raw, _ := io.ReadAll(res.Body)
|
||||
if res.StatusCode >= 300 {
|
||||
return ToolCall{}, string(raw), fmt.Errorf("http %d", res.StatusCode)
|
||||
}
|
||||
return ParseToolResponse(raw)
|
||||
}
|
||||
|
||||
func ParseToolResponse(raw []byte) (ToolCall, string, error) {
|
||||
var wrap struct {
|
||||
Choices []struct {
|
||||
Message struct {
|
||||
Content string `json:"content"`
|
||||
ToolCalls []struct {
|
||||
Function struct {
|
||||
Name string `json:"name"`
|
||||
Arguments json.RawMessage `json:"arguments"`
|
||||
} `json:"function"`
|
||||
} `json:"tool_calls"`
|
||||
} `json:"message"`
|
||||
} `json:"choices"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &wrap); err != nil {
|
||||
return ToolCall{}, "", err
|
||||
}
|
||||
content := ""
|
||||
if len(wrap.Choices) > 0 {
|
||||
content = wrap.Choices[0].Message.Content
|
||||
if n := len(wrap.Choices[0].Message.ToolCalls); n > 0 {
|
||||
fn := wrap.Choices[0].Message.ToolCalls[0].Function
|
||||
return ToolCall{Name: fn.Name, Arguments: rawArgs(fn.Arguments)}, content, nil
|
||||
}
|
||||
}
|
||||
return ToolCall{}, content, fmt.Errorf("no tool_calls")
|
||||
}
|
||||
|
||||
func rawArgs(raw json.RawMessage) string {
|
||||
if len(raw) == 0 {
|
||||
return ""
|
||||
}
|
||||
var s string
|
||||
if err := json.Unmarshal(raw, &s); err == nil {
|
||||
return s
|
||||
}
|
||||
return string(raw)
|
||||
}
|
||||
|
||||
func XMLLeak(content string) bool {
|
||||
s := strings.ToLower(content)
|
||||
return strings.Contains(s, "<tool_call>") ||
|
||||
strings.Contains(s, "<function=") ||
|
||||
strings.Contains(s, "<parameter")
|
||||
}
|
||||
|
||||
func RunPrompt(c Client, p Prompt) Result {
|
||||
start := time.Now()
|
||||
tc, content, err := c.ChatTools(p.User)
|
||||
out := Result{
|
||||
Model: c.Model,
|
||||
WantedTool: p.Want,
|
||||
LatencyMS: time.Since(start).Milliseconds(),
|
||||
Device: c.Device,
|
||||
XMLLeak: XMLLeak(content),
|
||||
}
|
||||
if err != nil {
|
||||
out.Err = err.Error()
|
||||
if content != "" && out.XMLLeak {
|
||||
out.Err = "xml tool call instead of openai tool_calls"
|
||||
}
|
||||
return out
|
||||
}
|
||||
out.ToolName = tc.Name
|
||||
out.OK = tc.Name == p.Want
|
||||
if !out.OK {
|
||||
out.Err = "wanted " + p.Want + " got " + tc.Name
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
type ProcMem struct {
|
||||
Name string
|
||||
SizeMB int
|
||||
VRAMMB int
|
||||
}
|
||||
|
||||
func ParsePS(raw []byte) []ProcMem {
|
||||
var wrap struct {
|
||||
Models []struct {
|
||||
Name string `json:"name"`
|
||||
Size int64 `json:"size"`
|
||||
SizeVRAM int64 `json:"size_vram"`
|
||||
} `json:"models"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &wrap); err != nil {
|
||||
return nil
|
||||
}
|
||||
out := make([]ProcMem, 0, len(wrap.Models))
|
||||
for _, m := range wrap.Models {
|
||||
out = append(out, ProcMem{
|
||||
Name: m.Name,
|
||||
SizeMB: int(m.Size / (1024 * 1024)),
|
||||
VRAMMB: int(m.SizeVRAM / (1024 * 1024)),
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func (c Client) FetchPS() []ProcMem {
|
||||
url := Origin(c.BaseURL) + "/api/ps"
|
||||
res, err := c.httpc().Get(url)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
defer res.Body.Close()
|
||||
raw, _ := io.ReadAll(res.Body)
|
||||
if res.StatusCode >= 300 {
|
||||
return nil
|
||||
}
|
||||
return ParsePS(raw)
|
||||
}
|
||||
|
||||
func Run(c Client) Report {
|
||||
if c.Device == "" {
|
||||
c.Device = "cpu"
|
||||
}
|
||||
rep := Report{
|
||||
Model: c.Model,
|
||||
HF: HFFor(c.Model),
|
||||
Device: c.Device,
|
||||
}
|
||||
for _, p := range BakePrompts {
|
||||
r := RunPrompt(c, p)
|
||||
rep.Prompts = append(rep.Prompts, r)
|
||||
rep.ToolCallN++
|
||||
if r.OK {
|
||||
rep.ToolCallOK++
|
||||
}
|
||||
if r.XMLLeak {
|
||||
rep.XMLLeak++
|
||||
}
|
||||
}
|
||||
if mems := c.FetchPS(); len(mems) > 0 {
|
||||
rep.RSSMB = mems[0].SizeMB
|
||||
rep.VRAMMB = mems[0].VRAMMB
|
||||
if c.Device == "cpu" && mems[0].VRAMMB > 0 {
|
||||
rep.Device = "gpu"
|
||||
}
|
||||
for i := range rep.Prompts {
|
||||
rep.Prompts[i].RSSMB = rep.RSSMB
|
||||
}
|
||||
}
|
||||
return rep
|
||||
}
|
||||
@@ -0,0 +1,166 @@
|
||||
package reasoner
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/eSlider/2dph/internal/httpapi"
|
||||
)
|
||||
|
||||
func TestHFIdsAreRealAndNoQwen36Nine(t *testing.T) {
|
||||
if HFQwen35_9B != "Qwen/Qwen3.5-9B" {
|
||||
t.Fatalf("9B id = %s", HFQwen35_9B)
|
||||
}
|
||||
if HFQwen36_27B != "Qwen/Qwen3.6-27B" {
|
||||
t.Fatalf("27B id = %s", HFQwen36_27B)
|
||||
}
|
||||
if HFBonsai27B != "prism-ml/Bonsai-27B-gguf" {
|
||||
t.Fatalf("bonsai id = %s", HFBonsai27B)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseToolResponseOpenAI(t *testing.T) {
|
||||
raw := []byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"search","arguments":"{\"q\":\"LadybugDB\"}"}}]}}]}`)
|
||||
tc, _, err := ParseToolResponse(raw)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if tc.Name != "search" {
|
||||
t.Fatalf("name=%s", tc.Name)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseToolResponseXMLIsFailure(t *testing.T) {
|
||||
raw := []byte(`{"choices":[{"message":{"content":"<tool_call>search</tool_call>"}}]}`)
|
||||
_, content, err := ParseToolResponse(raw)
|
||||
if err == nil {
|
||||
t.Fatal("expected no tool_calls")
|
||||
}
|
||||
if !XMLLeak(content) {
|
||||
t.Fatal("xml leak not detected")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBakePromptsWantMCPTools(t *testing.T) {
|
||||
if len(BakePrompts) != 3 {
|
||||
t.Fatalf("prompts=%d", len(BakePrompts))
|
||||
}
|
||||
names := map[string]bool{}
|
||||
for _, p := range BakePrompts {
|
||||
names[p.Want] = true
|
||||
}
|
||||
for _, n := range []string{"search", "get", "audit"} {
|
||||
if !names[n] {
|
||||
t.Fatalf("missing want %s", n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseToolResponseObjectArgs(t *testing.T) {
|
||||
raw := []byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"get","arguments":{"id":"leaf-demo"}}}]}}]}`)
|
||||
tc, _, err := ParseToolResponse(raw)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if tc.Name != "get" {
|
||||
t.Fatalf("name=%s", tc.Name)
|
||||
}
|
||||
if !strings.Contains(tc.Arguments, "leaf-demo") {
|
||||
t.Fatalf("args=%s", tc.Arguments)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParsePSCPUNotVRAM(t *testing.T) {
|
||||
raw := []byte(`{"models":[{"name":"qwen3.5:9b","size":6900000000,"size_vram":0}]}`)
|
||||
got := ParsePS(raw)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("n=%d", len(got))
|
||||
}
|
||||
if got[0].VRAMMB != 0 {
|
||||
t.Fatalf("vram=%d", got[0].VRAMMB)
|
||||
}
|
||||
if got[0].SizeMB < 6000 {
|
||||
t.Fatalf("rss=%d", got[0].SizeMB)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMCPToolsArePicoClawSubset(t *testing.T) {
|
||||
mcp := map[string]bool{}
|
||||
for _, op := range httpapi.Ops {
|
||||
if op.MCP {
|
||||
mcp[op.ID] = true
|
||||
}
|
||||
}
|
||||
seen := map[string]bool{}
|
||||
for _, tool := range MCPTools() {
|
||||
fn, _ := tool["function"].(map[string]any)
|
||||
name, _ := fn["name"].(string)
|
||||
if !mcp[name] {
|
||||
t.Fatalf("%s is not an MCP op", name)
|
||||
}
|
||||
seen[name] = true
|
||||
}
|
||||
for _, n := range []string{"search", "get", "audit"} {
|
||||
if !seen[n] {
|
||||
t.Fatalf("bake-off missing %s", n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestChatToolsHitsOpenAIPath(t *testing.T) {
|
||||
var gotTools bool
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.URL.Path != "/v1/chat/completions" {
|
||||
t.Errorf("path=%s", r.URL.Path)
|
||||
}
|
||||
body, _ := io.ReadAll(r.Body)
|
||||
var payload map[string]any
|
||||
_ = json.Unmarshal(body, &payload)
|
||||
if _, ok := payload["tools"]; ok {
|
||||
gotTools = true
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"search","arguments":"{\"q\":\"LadybugDB\"}"}}]}}]}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
c := Client{BaseURL: srv.URL + "/v1", Model: "stub", HTTP: srv.Client(), Device: "cpu"}
|
||||
r := RunPrompt(c, BakePrompts[0])
|
||||
if !gotTools {
|
||||
t.Fatal("tools not sent")
|
||||
}
|
||||
if !r.OK {
|
||||
t.Fatalf("result=%+v", r)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunSamplesPSAfterPrompts(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
if r.URL.Path == "/api/ps" {
|
||||
_, _ = w.Write([]byte(`{"models":[{"name":"stub","size":6500000000,"size_vram":0}]}`))
|
||||
return
|
||||
}
|
||||
_, _ = w.Write([]byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"search","arguments":"{}"}}]}}]}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
c := Client{BaseURL: srv.URL + "/v1", Model: "stub", HTTP: srv.Client(), Device: "cpu"}
|
||||
if mems := c.FetchPS(); len(mems) != 1 || mems[0].VRAMMB != 0 {
|
||||
t.Fatalf("%+v", mems)
|
||||
}
|
||||
if mems := c.FetchPS(); mems[0].SizeMB < 6000 {
|
||||
t.Fatalf("rss=%d", mems[0].SizeMB)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHFForKnownOllamaTags(t *testing.T) {
|
||||
if HFFor(OllamaRAM) != HFQwen35_9B {
|
||||
t.Fatal(HFFor(OllamaRAM))
|
||||
}
|
||||
if HFFor(OllamaQuality) != HFBonsai27B {
|
||||
t.Fatal(HFFor(OllamaQuality))
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user