feat: CPU reasoner bake-off for Qwen3.5-9B vs Bonsai (#25)
Tests / Test (push) Failing after 5s
Tests / Release (semver) (push) Skipped

Measure OpenAI tool_calls (search/get/audit) on a CPU Ollama sidecar
instead of gating D18 on GPU or PicoClaw. Weights stay out of the image.
This commit is contained in:
2026-08-13 23:19:51 +01:00
committed by GitHub
co-authored by GitHub
parent e393cc6a99
commit bb272d7416
13 changed files with 717 additions and 3 deletions
+1
View File
@@ -11,3 +11,4 @@ __pycache__/
.secrets/
lib-ladybug/
go.work.local
models/
+3 -1
View File
@@ -46,7 +46,8 @@ bin/markdown/ import.go (mistune leafs)
bin/postgres/ query.go (read-only YAML)
bin/git/ import.go (go-git history; Python shim execs it)
bin/web/ search.go (SearXNG; Python shim execs it)
internal/ shared Go (brain/rank is cgo-free; chats parsers; gitlog; websearch)
bin/reasoner/ bakeoff.go (D18 CPU OpenAI tool-call bake-off)
internal/ shared Go (brain/rank is cgo-free; chats parsers; gitlog; websearch; reasoner)
bin/watch/ corpus watcher (used by bin/brain/watch.go)
bin/tools/ vendored python libs behind bin/* (kblib, yamlout, websearch)
bin/cgo/ zig zcc zc++ (CGO via zig cc, not gcc)
@@ -94,6 +95,7 @@ bin/brain/serve.go # HTTP :8630; GET /openapi.json
bin/markdown/import.go [dir] # mistune leaves → YAML
bin/git/import.go [REPO] [--json] [--limit N] # go-git history → commit leafs
bin/web/search.go "query" [--json] # SearXNG; throttled ≠ absence
bin/reasoner/bakeoff.go [--model ID] [--json] # D18 CPU tool-call bake-off
bin/postgres/query.go --profile onlyoffice -c 'SELECT 1'
bin/md/tables # what the graph holds → YAML
bin/brain/deduce "question" # thinking wrapper
+2 -1
View File
@@ -41,7 +41,7 @@ detective method: **a fact needs ≥2 independent sources or it is
| D15 | repo | Gitea [`eSlider/2dph`](https://git.produktor.io/eSlider/2dph) is origin + [issues](https://git.produktor.io/eSlider/2dph/issues). GitHub `eSlider/2dph` is the public clone (PRs + Actions CI). No direct `main` pushes. TDD → PR → CI green → merge. |
| D16 | contradictions | ≥2 yes vs ≥2 no → unrelated sources conflict → hypothesis → `(not confirmed)`. Resolution (authority, staleness adjudication) = **v2**, tracked as open question. |
| D17 | assertion gate | Fact-check every *claim* (facts → info → live → web), not every edit. `bin/brain/search.go` adds a `web` block when there is no facts hit (`throttled`/`skipped`/`refused` ≠ absence). `--root` and `--no-web` stay local. Missing graph ≠ “does not exist”. |
| D18 | reasoner | Pluggable OpenAI-compatible URL. RAM: Qwen3.5-9B. Quality: Bonsai-27B or Qwen3.6-27B. No official Qwen3.6-9B. |
| D18 | reasoner | Pluggable OpenAI-compatible URL (`REASONER_BASE_URL`). RAM: `Qwen/Qwen3.5-9B`. Quality: `prism-ml/Bonsai-27B-gguf` or `Qwen/Qwen3.6-27B`. No official Qwen3.6-9B. CPU bake-off: `bin/reasoner/bakeoff.go` + compose profile `reasoner` (`OLLAMA_NUM_GPU=0`, `:11435`). PicoClaw is not shipped; tools are `search`/`get`/`audit`. Weights are not copied into the 2dph image. |
| D19 | git history | [go-git](https://github.com/go-git/go-git) via `bin/git/import.go`. No subprocess of the git binary. Conversion prints commit leafs; brain write is `bin/brain/index.go`. |
| D20 | agent API | OpenAPI + MCP are generated from the same `internal/httpapi.Ops` table as `bin/brain/serve.go` handlers. `GET /openapi.json`, `POST /mcp` (JSON-RPC tools/list + tools/call). Tool names match OpenAPI paths (`search`/`get`/`stats`/`audit`). |
| D21 | CGO | Ladybug/tokenizers CGO is compiled with **Zig** (`bin/cgo/zcc``zig cc -target …-linux-gnu`), not gcc. `bin/cgo/zig` pins Zig 0.14.1 + liblbug 0.19.1 + libtokenizers 1.27.0. Compose `target: api` has no CPython; write/rebuild is profile `index`. |
@@ -67,6 +67,7 @@ detective method: **a fact needs ≥2 independent sources or it is
postgres/query.go read-only YAML (wraps bin/db/psql-yq)
git/import.go go-git history (no git binary; conversion only)
web/search.go SearXNG client (throttled ≠ absence)
reasoner/bakeoff.go CPU tool-call bake-off (D18; OpenAI tools)
chats/sync.go import.go facts.go apply.go
(libs in internal/chats; no chats index)
md/import (deprecated; bin/markdown/import.go)
+1
View File
@@ -158,6 +158,7 @@ Docker (optional, cached model + var volumes):
docker compose up -d brain # API (Zig CGO serve :8630)
docker compose --profile index run --rm index # Python Ladybug rebuild
docker compose --profile picoclaw up brain-mcp # MCP on 127.0.0.1:8630
docker compose --profile reasoner up -d reasoner # CPU Ollama 127.0.0.1:11435
docker compose up brain-watch # auto re-index on change
```
+91
View File
@@ -0,0 +1,91 @@
//usr/bin/env go run -tags=reasoner_bakeoff "$0" "$@"; exit
//go:build reasoner_bakeoff
//
// bin/reasoner/bakeoff.go - CPU tool-call bake-off against an OpenAI-compatible URL (D18).
//
// REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b ./bin/reasoner/bakeoff.go
// ./bin/reasoner/bakeoff.go --model MichelRosselli/bonsai-27b:Q1_0 --json
//
// Measures OpenAI tool_calls (search/get/audit) and RSS from Ollama /api/ps, not VRAM.
// PicoClaw is not in this repo; the tool names match internal/httpapi MCP ops.
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
package main
import (
"encoding/json"
"fmt"
"os"
"strings"
"github.com/eSlider/2dph/internal/reasoner"
)
func main() {
os.Exit(run(os.Args[1:]))
}
func run(args []string) int {
base := os.Getenv("REASONER_BASE_URL")
if base == "" {
base = "http://127.0.0.1:11435/v1"
}
model := os.Getenv("REASONER_MODEL")
if model == "" {
model = reasoner.OllamaRAM
}
jsonOut := false
device := "cpu"
for i := 0; i < len(args); i++ {
a := args[i]
switch {
case a == "--json":
jsonOut = true
case a == "--model" && i+1 < len(args):
i++
model = args[i]
case strings.HasPrefix(a, "--model="):
model = strings.TrimPrefix(a, "--model=")
case a == "--base-url" && i+1 < len(args):
i++
base = args[i]
case a == "--device" && i+1 < len(args):
i++
device = args[i]
case a == "-h" || a == "--help":
fmt.Fprintln(os.Stderr, "bin/reasoner/bakeoff.go [--model ID] [--base-url URL] [--device cpu] [--json]")
return 0
default:
fmt.Fprintln(os.Stderr, "unknown arg:", a)
return 2
}
}
c := reasoner.Client{BaseURL: base, Model: model, Device: device}
rep := reasoner.Run(c)
raw, err := json.MarshalIndent(rep, "", " ")
if err != nil {
fmt.Fprintln(os.Stderr, err)
return 1
}
if jsonOut {
fmt.Println(string(raw))
} else {
fmt.Printf("model: %s\n", rep.Model)
fmt.Printf("hf_id: %s\n", rep.HF)
fmt.Printf("device: %s\n", rep.Device)
fmt.Printf("tool_call: %d/%d\n", rep.ToolCallOK, rep.ToolCallN)
fmt.Printf("xml_leak: %d\n", rep.XMLLeak)
fmt.Printf("rss_mb: %d\n", rep.RSSMB)
fmt.Printf("vram_mb: %d\n", rep.VRAMMB)
for _, p := range rep.Prompts {
status := "fail"
if p.OK {
status = "ok"
}
fmt.Printf(" %s: %s wanted=%s got=%s xml=%v %dms %s\n", p.WantedTool, status, p.WantedTool, p.ToolName, p.XMLLeak, p.LatencyMS, p.Err)
}
}
if rep.ToolCallN == 0 {
return 1
}
return 0
}
+2
View File
@@ -0,0 +1,2 @@
// Commands in this directory are shebang mains (bakeoff.go).
package main
+29
View File
@@ -103,6 +103,35 @@ class PublishedDocsTest(unittest.TestCase):
self.assertIn('profiles: ["index"]', compose)
self.assertIn("target: api", compose)
def test_reasoner_docs_name_real_hf_ids_cpu_sidecar(self) -> None:
docs = (ROOT / "docs" / "reasoner.md").read_text()
for hf in (
"Qwen/Qwen3.5-9B",
"Qwen/Qwen3.6-27B",
"prism-ml/Bonsai-27B-gguf",
):
self.assertIn(hf, docs)
self.assertIn("no official qwen3.6-9b", docs.lower())
self.assertIn("OLLAMA_NUM_GPU", docs)
self.assertIn("rss_mb", docs)
self.assertIn("vram_mb", docs)
self.assertIn("3/3", docs)
self.assertIn("Do not claim 9B is better at tools", docs)
self.assertNotIn("Qwen/Qwen3.6-9B", docs)
plan = (ROOT / "PLAN.md").read_text()
self.assertIn("D18", plan)
self.assertIn("Qwen/Qwen3.5-9B", plan)
compose = (ROOT / "compose.yaml").read_text()
self.assertIn('profiles: ["reasoner"]', compose)
self.assertIn("OLLAMA_NUM_GPU", compose)
self.assertIn("127.0.0.1:11435", compose)
dockerfile = (ROOT / "Dockerfile").read_text()
self.assertNotIn(".gguf", dockerfile.lower())
self.assertNotIn(".safetensors", dockerfile.lower())
api = dockerfile[dockerfile.index("FROM debian:bookworm-slim AS api") :]
self.assertNotIn("COPY models", api)
self.assertNotIn("qwen", api.lower())
def test_readme_search_escalates_web(self) -> None:
text = (ROOT / "README.md").read_text()
self.assertIn("--no-web", text)
+18
View File
@@ -3,6 +3,7 @@
# docker compose up -d brain # API (Zig CGO serve)
# docker compose --profile index run --rm index # Python rebuild
# docker compose --profile picoclaw up brain-mcp
# docker compose --profile reasoner up -d reasoner # CPU Ollama :11435
# docker compose --profile searxng up -d
#
# Secrets never baked in: search.env + db-profiles.yml from ~/.config/brain.
@@ -114,6 +115,23 @@ services:
- /tmp
restart: unless-stopped
# CPU OpenAI-compatible sidecar (D18). Weights are pulled at runtime, not
# baked into the 2dph image. Does not touch host Ollama on :11434.
# docker compose --profile reasoner up -d reasoner
# docker compose --profile reasoner exec reasoner ollama pull qwen3.5:9b
reasoner:
profiles: ["reasoner"]
image: docker.io/ollama/ollama:latest
environment:
OLLAMA_NUM_GPU: "0"
OLLAMA_HOST: "0.0.0.0:11434"
ports:
- "127.0.0.1:11435:11434"
volumes:
- reasoner-ollama:/root/.ollama
restart: unless-stopped
volumes:
kb-model:
kb-var:
reasoner-ollama:
+1
View File
@@ -6,6 +6,7 @@ Brain/ops/eSlider stack. Facts need proof or they are
- [PLAN.md](../PLAN.md) — decisions, execution order, open questions (v2)
- [design](design.md) — schema, deduction model, sources
- [reasoner](reasoner.md) — D18 CPU bake-off (Qwen3.5-9B vs Bonsai / Qwen3.6-27B)
- [Gitea issues](https://git.produktor.io/eSlider/2dph/issues) — work board (origin)
Search: `bin/brain/search.go "query"` (HTTP: `bin/brain/serve.go`
+9 -1
View File
@@ -78,4 +78,12 @@ fetches Zig + libs (`bin/cgo/zig`). Index/write is still `bin/kb/index`
`bin/brain/serve.go` exposes the same `internal/httpapi.Ops` table as OpenAPI
(`GET /openapi.json`) and MCP (`POST /mcp` JSON-RPC `tools/list` +
`tools/call`). Tool names match paths: `search`, `get`, `stats`, `audit`.
Agents should use these endpoints instead of shebang CLIs.
Agents should use these endpoints instead of shebang CLIs.
## Reasoner (D18)
Pluggable OpenAI-compatible URL. RAM: `Qwen/Qwen3.5-9B`. Quality:
`prism-ml/Bonsai-27B-gguf` or `Qwen/Qwen3.6-27B`. No official Qwen3.6-9B.
CPU sidecar: compose profile `reasoner` (`OLLAMA_NUM_GPU=0`,
`127.0.0.1:11435`). Bake-off: `bin/reasoner/bakeoff.go`. Weights stay out
of the 2dph image. See [docs/reasoner.md](reasoner.md).
+69
View File
@@ -0,0 +1,69 @@
# Reasoner bake-off (D18)
Pluggable OpenAI-compatible URL. 2dph does not ship weights. PicoClaw is
not in this repo; the bake-off hits the same tool names PicoClaw would
(`search``get``audit` from `internal/httpapi.Ops`).
```bash
docker compose --profile reasoner up -d reasoner
docker compose --profile reasoner exec reasoner ollama pull qwen3.5:9b
REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b \
./bin/reasoner/bakeoff.go --json
```
Host Ollama on `:11434` is left alone. This sidecar binds `127.0.0.1:11435`
with `OLLAMA_NUM_GPU=0` (CPU). Measure RSS (`/api/ps` `size`), not VRAM.
If Compose cannot allocate a project network (Docker IPAM pool exhausted),
the same sidecar is:
```bash
docker run -d --name 2dph-reasoner \
-e OLLAMA_NUM_GPU=0 \
-p 127.0.0.1:11435:11434 \
-v 2dph-reasoner-ollama:/root/.ollama \
ollama/ollama:latest
```
## Real Hugging Face ids
| Role | HF id | Ollama tag (this bake-off) |
|------|-------|----------------------------|
| RAM / 9B | `Qwen/Qwen3.5-9B` | `qwen3.5:9b` |
| Quality 27B (CPU) | `prism-ml/Bonsai-27B-gguf` (derived from Qwen3.6-27B) | `MichelRosselli/bonsai-27b:Q1_0` |
| Quality 27B (full) | `Qwen/Qwen3.6-27B` | not pulled on this CPU box |
There is **no official Qwen3.6-9B**. Do not invent that id.
Qwen3.5-9B has documented upstream tool-call XML bugs. A 9B win on tools
is only claimed if this bake-off records OpenAI `tool_calls` (not
`<tool_call>` XML in `content`).
The 2dph API image does not `COPY` GGUF/safetensors. Pull at runtime into
the `reasoner-ollama` volume.
## Live CPU run
Host sidecar: Ollama **0.32.9**, `OLLAMA_NUM_GPU=0`, `127.0.0.1:11435`,
`device: cpu`, `vram_mb: 0`. Date: 2026-08-13. Same three prompts
(`search` / `get` / `audit`). PicoClaw binary was not used; the OpenAI
tools payload is the surface it would send.
| Model | HF id | tool_call | xml_leak | rss_mb | latency_ms (search/get/audit) |
|-------|-------|-----------|----------|--------|-------------------------------|
| `qwen3.5:9b` | `Qwen/Qwen3.5-9B` | 3/3 | 0 | 5790 | 50059 / 51678 / 31980 |
| `MichelRosselli/bonsai-27b:Q1_0` | `prism-ml/Bonsai-27B-gguf` | 3/3 | 0 | 21951 | 327702 / 166830 / 119808 |
| `Qwen/Qwen3.6-27B` | `Qwen/Qwen3.6-27B` | not loaded | — | — | too heavy for this CPU box |
Both loaded models emitted OpenAI `tool_calls` (not `<tool_call>` XML) on
this runtime. **Do not claim 9B is better at tools** — the score is tied
at 3/3. 9B is smaller and faster. Bonsai RSS includes weights + KV
(`size` from `/api/ps`); first Bonsai prompt includes cold load.
Re-run:
```bash
REASONER_BASE_URL=http://127.0.0.1:11435/v1 REASONER_MODEL=qwen3.5:9b \
./bin/reasoner/bakeoff.go --json
REASONER_MODEL=MichelRosselli/bonsai-27b:Q1_0 ./bin/reasoner/bakeoff.go --json
```
+325
View File
@@ -0,0 +1,325 @@
package reasoner
import (
"bytes"
"encoding/json"
"fmt"
"io"
"net/http"
"strings"
"time"
)
// HF IDs named in docs. No Qwen3.6-9B exists.
const (
HFQwen35_9B = "Qwen/Qwen3.5-9B"
HFQwen36_27B = "Qwen/Qwen3.6-27B"
HFBonsai27B = "prism-ml/Bonsai-27B-gguf"
OllamaRAM = "qwen3.5:9b"
OllamaQuality = "MichelRosselli/bonsai-27b:Q1_0"
)
type ToolCall struct {
Name string
Arguments string
}
type Result struct {
Model string `json:"model"`
OK bool `json:"ok"`
ToolName string `json:"tool_name,omitempty"`
XMLLeak bool `json:"xml_leak"`
Err string `json:"error,omitempty"`
LatencyMS int64 `json:"latency_ms"`
RSSMB int `json:"rss_mb,omitempty"`
Device string `json:"device"`
WantedTool string `json:"wanted_tool"`
}
type Prompt struct {
Name string
Want string
User string
}
var BakePrompts = []Prompt{
{
Name: "search-before-claim",
Want: "search",
User: "Use tools. Search the 2dph brain for LadybugDB before you answer. Call search.",
},
{
Name: "get-leaf",
Want: "get",
User: "Use tools. Fetch leaf id leaf-demo with get. Do not invent the body.",
},
{
Name: "audit-index",
Want: "audit",
User: "Use tools. Call audit on the brain index health.",
},
}
func MCPTools() []map[string]any {
return []map[string]any{
openaiTool("search", "deduction search (facts → info → web)", map[string]any{
"type": "object",
"properties": map[string]any{
"q": map[string]any{"type": "string", "description": "search query"},
"n": map[string]any{"type": "integer"},
},
"required": []string{"q"},
}),
openaiTool("get", "read one leaf by id", map[string]any{
"type": "object",
"properties": map[string]any{
"id": map[string]any{"type": "string"},
"body": map[string]any{"type": "boolean"},
},
"required": []string{"id"},
}),
openaiTool("audit", "facts confidence histogram", map[string]any{
"type": "object",
"properties": map[string]any{},
}),
}
}
func openaiTool(name, desc string, schema map[string]any) map[string]any {
return map[string]any{
"type": "function",
"function": map[string]any{
"name": name,
"description": desc,
"parameters": schema,
},
}
}
type Client struct {
BaseURL string
Model string
HTTP *http.Client
Device string
ToolChoice string
}
type Report struct {
Model string `json:"model"`
HF string `json:"hf_id"`
Device string `json:"device"`
ToolCallOK int `json:"tool_call_ok"`
ToolCallN int `json:"tool_call_n"`
XMLLeak int `json:"xml_leak"`
RSSMB int `json:"rss_mb"`
VRAMMB int `json:"vram_mb"`
Prompts []Result `json:"prompts"`
}
func HFFor(model string) string {
switch model {
case OllamaRAM:
return HFQwen35_9B
case OllamaQuality:
return HFBonsai27B
default:
if strings.Contains(model, "qwen3.6") || strings.Contains(model, "Qwen3.6") {
return HFQwen36_27B
}
return ""
}
}
func (c *Client) httpc() *http.Client {
if c.HTTP == nil {
c.HTTP = &http.Client{Timeout: 10 * time.Minute}
}
return c.HTTP
}
func Origin(base string) string {
s := strings.TrimRight(base, "/")
return strings.TrimSuffix(s, "/v1")
}
func (c Client) ChatTools(user string) (ToolCall, string, error) {
choice := c.ToolChoice
if choice == "" {
choice = "required"
}
body, _ := json.Marshal(map[string]any{
"model": c.Model,
"messages": []map[string]string{
{"role": "system", "content": "You are PicoClaw talking to 2dph MCP. Always call a tool before a factual claim. search then get then audit."},
{"role": "user", "content": user},
},
"tools": MCPTools(),
"tool_choice": choice,
})
base := strings.TrimRight(c.BaseURL, "/")
if !strings.HasSuffix(base, "/v1") {
base += "/v1"
}
url := base + "/chat/completions"
req, err := http.NewRequest(http.MethodPost, url, bytes.NewReader(body))
if err != nil {
return ToolCall{}, "", err
}
req.Header.Set("Content-Type", "application/json")
res, err := c.httpc().Do(req)
if err != nil {
return ToolCall{}, "", err
}
defer res.Body.Close()
raw, _ := io.ReadAll(res.Body)
if res.StatusCode >= 300 {
return ToolCall{}, string(raw), fmt.Errorf("http %d", res.StatusCode)
}
return ParseToolResponse(raw)
}
func ParseToolResponse(raw []byte) (ToolCall, string, error) {
var wrap struct {
Choices []struct {
Message struct {
Content string `json:"content"`
ToolCalls []struct {
Function struct {
Name string `json:"name"`
Arguments json.RawMessage `json:"arguments"`
} `json:"function"`
} `json:"tool_calls"`
} `json:"message"`
} `json:"choices"`
}
if err := json.Unmarshal(raw, &wrap); err != nil {
return ToolCall{}, "", err
}
content := ""
if len(wrap.Choices) > 0 {
content = wrap.Choices[0].Message.Content
if n := len(wrap.Choices[0].Message.ToolCalls); n > 0 {
fn := wrap.Choices[0].Message.ToolCalls[0].Function
return ToolCall{Name: fn.Name, Arguments: rawArgs(fn.Arguments)}, content, nil
}
}
return ToolCall{}, content, fmt.Errorf("no tool_calls")
}
func rawArgs(raw json.RawMessage) string {
if len(raw) == 0 {
return ""
}
var s string
if err := json.Unmarshal(raw, &s); err == nil {
return s
}
return string(raw)
}
func XMLLeak(content string) bool {
s := strings.ToLower(content)
return strings.Contains(s, "<tool_call>") ||
strings.Contains(s, "<function=") ||
strings.Contains(s, "<parameter")
}
func RunPrompt(c Client, p Prompt) Result {
start := time.Now()
tc, content, err := c.ChatTools(p.User)
out := Result{
Model: c.Model,
WantedTool: p.Want,
LatencyMS: time.Since(start).Milliseconds(),
Device: c.Device,
XMLLeak: XMLLeak(content),
}
if err != nil {
out.Err = err.Error()
if content != "" && out.XMLLeak {
out.Err = "xml tool call instead of openai tool_calls"
}
return out
}
out.ToolName = tc.Name
out.OK = tc.Name == p.Want
if !out.OK {
out.Err = "wanted " + p.Want + " got " + tc.Name
}
return out
}
type ProcMem struct {
Name string
SizeMB int
VRAMMB int
}
func ParsePS(raw []byte) []ProcMem {
var wrap struct {
Models []struct {
Name string `json:"name"`
Size int64 `json:"size"`
SizeVRAM int64 `json:"size_vram"`
} `json:"models"`
}
if err := json.Unmarshal(raw, &wrap); err != nil {
return nil
}
out := make([]ProcMem, 0, len(wrap.Models))
for _, m := range wrap.Models {
out = append(out, ProcMem{
Name: m.Name,
SizeMB: int(m.Size / (1024 * 1024)),
VRAMMB: int(m.SizeVRAM / (1024 * 1024)),
})
}
return out
}
func (c Client) FetchPS() []ProcMem {
url := Origin(c.BaseURL) + "/api/ps"
res, err := c.httpc().Get(url)
if err != nil {
return nil
}
defer res.Body.Close()
raw, _ := io.ReadAll(res.Body)
if res.StatusCode >= 300 {
return nil
}
return ParsePS(raw)
}
func Run(c Client) Report {
if c.Device == "" {
c.Device = "cpu"
}
rep := Report{
Model: c.Model,
HF: HFFor(c.Model),
Device: c.Device,
}
for _, p := range BakePrompts {
r := RunPrompt(c, p)
rep.Prompts = append(rep.Prompts, r)
rep.ToolCallN++
if r.OK {
rep.ToolCallOK++
}
if r.XMLLeak {
rep.XMLLeak++
}
}
if mems := c.FetchPS(); len(mems) > 0 {
rep.RSSMB = mems[0].SizeMB
rep.VRAMMB = mems[0].VRAMMB
if c.Device == "cpu" && mems[0].VRAMMB > 0 {
rep.Device = "gpu"
}
for i := range rep.Prompts {
rep.Prompts[i].RSSMB = rep.RSSMB
}
}
return rep
}
+166
View File
@@ -0,0 +1,166 @@
package reasoner
import (
"encoding/json"
"io"
"net/http"
"net/http/httptest"
"strings"
"testing"
"github.com/eSlider/2dph/internal/httpapi"
)
func TestHFIdsAreRealAndNoQwen36Nine(t *testing.T) {
if HFQwen35_9B != "Qwen/Qwen3.5-9B" {
t.Fatalf("9B id = %s", HFQwen35_9B)
}
if HFQwen36_27B != "Qwen/Qwen3.6-27B" {
t.Fatalf("27B id = %s", HFQwen36_27B)
}
if HFBonsai27B != "prism-ml/Bonsai-27B-gguf" {
t.Fatalf("bonsai id = %s", HFBonsai27B)
}
}
func TestParseToolResponseOpenAI(t *testing.T) {
raw := []byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"search","arguments":"{\"q\":\"LadybugDB\"}"}}]}}]}`)
tc, _, err := ParseToolResponse(raw)
if err != nil {
t.Fatal(err)
}
if tc.Name != "search" {
t.Fatalf("name=%s", tc.Name)
}
}
func TestParseToolResponseXMLIsFailure(t *testing.T) {
raw := []byte(`{"choices":[{"message":{"content":"<tool_call>search</tool_call>"}}]}`)
_, content, err := ParseToolResponse(raw)
if err == nil {
t.Fatal("expected no tool_calls")
}
if !XMLLeak(content) {
t.Fatal("xml leak not detected")
}
}
func TestBakePromptsWantMCPTools(t *testing.T) {
if len(BakePrompts) != 3 {
t.Fatalf("prompts=%d", len(BakePrompts))
}
names := map[string]bool{}
for _, p := range BakePrompts {
names[p.Want] = true
}
for _, n := range []string{"search", "get", "audit"} {
if !names[n] {
t.Fatalf("missing want %s", n)
}
}
}
func TestParseToolResponseObjectArgs(t *testing.T) {
raw := []byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"get","arguments":{"id":"leaf-demo"}}}]}}]}`)
tc, _, err := ParseToolResponse(raw)
if err != nil {
t.Fatal(err)
}
if tc.Name != "get" {
t.Fatalf("name=%s", tc.Name)
}
if !strings.Contains(tc.Arguments, "leaf-demo") {
t.Fatalf("args=%s", tc.Arguments)
}
}
func TestParsePSCPUNotVRAM(t *testing.T) {
raw := []byte(`{"models":[{"name":"qwen3.5:9b","size":6900000000,"size_vram":0}]}`)
got := ParsePS(raw)
if len(got) != 1 {
t.Fatalf("n=%d", len(got))
}
if got[0].VRAMMB != 0 {
t.Fatalf("vram=%d", got[0].VRAMMB)
}
if got[0].SizeMB < 6000 {
t.Fatalf("rss=%d", got[0].SizeMB)
}
}
func TestMCPToolsArePicoClawSubset(t *testing.T) {
mcp := map[string]bool{}
for _, op := range httpapi.Ops {
if op.MCP {
mcp[op.ID] = true
}
}
seen := map[string]bool{}
for _, tool := range MCPTools() {
fn, _ := tool["function"].(map[string]any)
name, _ := fn["name"].(string)
if !mcp[name] {
t.Fatalf("%s is not an MCP op", name)
}
seen[name] = true
}
for _, n := range []string{"search", "get", "audit"} {
if !seen[n] {
t.Fatalf("bake-off missing %s", n)
}
}
}
func TestChatToolsHitsOpenAIPath(t *testing.T) {
var gotTools bool
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if r.URL.Path != "/v1/chat/completions" {
t.Errorf("path=%s", r.URL.Path)
}
body, _ := io.ReadAll(r.Body)
var payload map[string]any
_ = json.Unmarshal(body, &payload)
if _, ok := payload["tools"]; ok {
gotTools = true
}
w.Header().Set("Content-Type", "application/json")
_, _ = w.Write([]byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"search","arguments":"{\"q\":\"LadybugDB\"}"}}]}}]}`))
}))
defer srv.Close()
c := Client{BaseURL: srv.URL + "/v1", Model: "stub", HTTP: srv.Client(), Device: "cpu"}
r := RunPrompt(c, BakePrompts[0])
if !gotTools {
t.Fatal("tools not sent")
}
if !r.OK {
t.Fatalf("result=%+v", r)
}
}
func TestRunSamplesPSAfterPrompts(t *testing.T) {
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "application/json")
if r.URL.Path == "/api/ps" {
_, _ = w.Write([]byte(`{"models":[{"name":"stub","size":6500000000,"size_vram":0}]}`))
return
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"tool_calls":[{"function":{"name":"search","arguments":"{}"}}]}}]}`))
}))
defer srv.Close()
c := Client{BaseURL: srv.URL + "/v1", Model: "stub", HTTP: srv.Client(), Device: "cpu"}
if mems := c.FetchPS(); len(mems) != 1 || mems[0].VRAMMB != 0 {
t.Fatalf("%+v", mems)
}
if mems := c.FetchPS(); mems[0].SizeMB < 6000 {
t.Fatalf("rss=%d", mems[0].SizeMB)
}
}
func TestHFForKnownOllamaTags(t *testing.T) {
if HFFor(OllamaRAM) != HFQwen35_9B {
t.Fatal(HFFor(OllamaRAM))
}
if HFFor(OllamaQuality) != HFBonsai27B {
t.Fatal(HFFor(OllamaQuality))
}
}