feat: D16 adjudication — 2v2 stays hypothesis until a rule fires. (#35)
Tests / Test (push) Skipped
Tests / OCR (tesseract fixture) (push) Skipped
Tests / Release (semver) (push) Skipped
Tests / Test (push) Skipped
Tests / OCR (tesseract fixture) (push) Skipped
Tests / Release (semver) (push) Skipped
temporal_freshness then authority_pairing; unresolved keeps (not confirmed). bin/facts/audit contradict. Gitea #29.
This commit is contained in:
+46
-19
@@ -1,14 +1,15 @@
|
||||
#!/usr/bin/env python3
|
||||
"""facts/audit - evidence & lexicon checks for the 2dph brain.
|
||||
|
||||
bin/facts/audit self # lexicon: every fact in db has >=2 sources
|
||||
bin/facts/audit db # evidence gate: run against var/kb.lbug
|
||||
bin/facts/audit self # lexicon: docs + two-source rule
|
||||
bin/facts/audit db # evidence gate against var/kb.lbug
|
||||
bin/facts/audit contradict # D16 adjudication (JSON claim(s) on stdin)
|
||||
|
||||
`self` mode checks the repo itself (no network, no runtime deps). It greps
|
||||
for known-good two-source pairings and confirms the docs are consistent.
|
||||
`db` mode loads every Leaf with root=facts and asserts each has source_rev
|
||||
and a non-empty `loc` (the "where did you see it" evidence pointer) and that
|
||||
'confirmed' facts carry a two-source `source` field.
|
||||
`self` mode checks the repo itself (no network, no runtime deps).
|
||||
`db` mode loads every Leaf with root=facts. Confirmed facts need ` x `;
|
||||
hypothesis contradictions need `a x b vs c x d` (both sides ≥2).
|
||||
`contradict` applies temporal_freshness then authority_pairing; ≥2 vs ≥2
|
||||
with no rule stays hypothesis / `(not confirmed)`.
|
||||
|
||||
Exit 0 = all checks pass, 1 = audit failures, 2 = could not evaluate.
|
||||
"""
|
||||
@@ -22,6 +23,8 @@ from pathlib import Path
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
||||
|
||||
from contradict import adjudicate, check_fact_row # noqa: E402
|
||||
|
||||
|
||||
def audit_db() -> list[str]:
|
||||
from kblib import connect
|
||||
@@ -33,14 +36,8 @@ def audit_db() -> list[str]:
|
||||
r = conn.execute("MATCH (l:Leaf {root:'facts'}) RETURN l.id, l.source, l.loc, l.how, l.confidence")
|
||||
problems: list[str] = []
|
||||
for lid, source, loc, how, conf in r.get_all():
|
||||
if conf != "confirmed":
|
||||
problems.append(f"{lid}: facts require confidence='confirmed', got '{conf}'")
|
||||
if not source or " x " not in source:
|
||||
problems.append(f"{lid}: needs 2-source evidence in source, got '{source}'")
|
||||
if not loc:
|
||||
problems.append(f"{lid}: missing loc (evidence pointer)")
|
||||
if not how:
|
||||
problems.append(f"{lid}: missing how")
|
||||
problems.extend(check_fact_row(str(lid), str(source or ""), str(loc or ""),
|
||||
str(how or ""), str(conf or "")))
|
||||
conn.close()
|
||||
db.close()
|
||||
return problems
|
||||
@@ -54,20 +51,50 @@ def audit_self() -> list[str]:
|
||||
problems.append("PLAN.md missing recall@5 gate")
|
||||
if re.search(r"(?i)facts must have.*2 sources|2.source", plan) is None:
|
||||
problems.append("PLAN.md missing the two-source evidence rule for facts")
|
||||
if "temporal_freshness" not in plan or "authority_pairing" not in plan:
|
||||
problems.append("PLAN.md missing D16 adjudication rules")
|
||||
if re.search(r"(?i)HNSW|BM25|deduction", (ROOT / "README.md").read_text()) is None:
|
||||
problems.append("README.md missing search/retrieval description")
|
||||
return problems
|
||||
|
||||
|
||||
def audit_contradict(raw: str) -> tuple[list[str], list[dict]]:
|
||||
raw = raw.strip()
|
||||
if not raw:
|
||||
return ["contradict: empty stdin (JSON claim or {claims:[...]})"], []
|
||||
try:
|
||||
payload = json.loads(raw)
|
||||
except json.JSONDecodeError as e:
|
||||
return [f"contradict: invalid JSON: {e}"], []
|
||||
if isinstance(payload, dict) and "claims" in payload:
|
||||
claims = list(payload.get("claims") or [])
|
||||
elif isinstance(payload, dict):
|
||||
claims = [payload]
|
||||
elif isinstance(payload, list):
|
||||
claims = payload
|
||||
else:
|
||||
return ["contradict: expected object or list"], []
|
||||
details = [adjudicate(c) for c in claims]
|
||||
return [], details
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
import argparse
|
||||
p = argparse.ArgumentParser(description="evidence & lexicon audit")
|
||||
p.add_argument("mode", choices=("self", "db"))
|
||||
p.add_argument("mode", choices=("self", "db", "contradict"))
|
||||
p.add_argument("--json", action="store_true")
|
||||
a = p.parse_args(argv)
|
||||
|
||||
problems = audit_self() if a.mode == "self" else audit_db()
|
||||
out = {"mode": a.mode, "ok": not problems, "problems": problems}
|
||||
details: list[dict] = []
|
||||
if a.mode == "self":
|
||||
problems = audit_self()
|
||||
elif a.mode == "db":
|
||||
problems = audit_db()
|
||||
else:
|
||||
problems, details = audit_contradict(sys.stdin.read())
|
||||
out: dict = {"mode": a.mode, "ok": not problems, "problems": problems}
|
||||
if details:
|
||||
out["contradictions"] = details
|
||||
if a.json:
|
||||
print(json.dumps(out, indent=2))
|
||||
else:
|
||||
@@ -77,4 +104,4 @@ def main(argv: list[str]) -> int:
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv[1:]))
|
||||
sys.exit(main(sys.argv[1:]))
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
//
|
||||
// ./bin/facts/audit.go self
|
||||
// ./bin/facts/audit.go db
|
||||
// ./bin/facts/audit.go contradict --json < claim.json
|
||||
//
|
||||
// Python bin/facts/audit is the implementation (CI runs it directly).
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
|
||||
@@ -0,0 +1,103 @@
|
||||
"""D16 contradiction adjudication (same rules as internal/facts)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
CONF_CONFIRMED = "confirmed"
|
||||
CONF_HYPOTHESIS = "hypothesis"
|
||||
|
||||
RULE_UNRESOLVED = "unresolved"
|
||||
RULE_TEMPORAL = "temporal_freshness"
|
||||
RULE_AUTHORITY = "authority_pairing"
|
||||
RULE_TWO_SOURCE = "two_source"
|
||||
RULE_SINGLE = "single_source"
|
||||
|
||||
KIND_RUNTIME = "runtime"
|
||||
KIND_CONFIG = "config"
|
||||
KIND_NARRATIVE = "narrative"
|
||||
|
||||
|
||||
def _independent(sources: list[dict]) -> int:
|
||||
seen: set[str] = set()
|
||||
for i, s in enumerate(sources):
|
||||
sid = str(s.get("id") or "") or f"{s.get('kind', '')}#{i}"
|
||||
seen.add(sid)
|
||||
return len(seen)
|
||||
|
||||
|
||||
def _fresh_n(sources: list[dict]) -> int:
|
||||
return sum(1 for s in sources if not s.get("stale"))
|
||||
|
||||
|
||||
def _strong_n(sources: list[dict]) -> int:
|
||||
return sum(1 for s in sources if s.get("kind") in (KIND_RUNTIME, KIND_CONFIG))
|
||||
|
||||
|
||||
def adjudicate(claim: dict[str, Any]) -> dict[str, Any]:
|
||||
yes = list(claim.get("yes") or [])
|
||||
no = list(claim.get("no") or [])
|
||||
yes_n, no_n = _independent(yes), _independent(no)
|
||||
text = str(claim.get("text") or "")
|
||||
|
||||
def out(conf: str, rule: str, winner: str = "") -> dict[str, Any]:
|
||||
return {
|
||||
"text": text,
|
||||
"confidence": conf,
|
||||
"confirmed": conf == CONF_CONFIRMED,
|
||||
"rule": rule,
|
||||
"winner": winner,
|
||||
"yes": yes_n,
|
||||
"no": no_n,
|
||||
}
|
||||
|
||||
if yes_n < 2 or no_n < 2:
|
||||
if yes_n >= 2:
|
||||
return out(CONF_CONFIRMED, RULE_TWO_SOURCE, "yes")
|
||||
if no_n >= 2:
|
||||
return out(CONF_CONFIRMED, RULE_TWO_SOURCE, "no")
|
||||
return out(CONF_HYPOTHESIS, RULE_SINGLE)
|
||||
yf, nf = _fresh_n(yes), _fresh_n(no)
|
||||
if yf >= 2 and nf < 2:
|
||||
return out(CONF_CONFIRMED, RULE_TEMPORAL, "yes")
|
||||
if nf >= 2 and yf < 2:
|
||||
return out(CONF_CONFIRMED, RULE_TEMPORAL, "no")
|
||||
ys, ns = _strong_n(yes), _strong_n(no)
|
||||
if ys >= 2 and ns < 2:
|
||||
return out(CONF_CONFIRMED, RULE_AUTHORITY, "yes")
|
||||
if ns >= 2 and ys < 2:
|
||||
return out(CONF_CONFIRMED, RULE_AUTHORITY, "no")
|
||||
return out(CONF_HYPOTHESIS, RULE_UNRESOLVED)
|
||||
|
||||
|
||||
def parse_source_field(source: str) -> tuple[str, str]:
|
||||
"""Split `a x b vs c x d` into (yes, no). Empty no if no ` vs `."""
|
||||
if " vs " not in source:
|
||||
return source, ""
|
||||
yes, _, no = source.partition(" vs ")
|
||||
return yes.strip(), no.strip()
|
||||
|
||||
|
||||
def check_fact_row(lid: str, source: str, loc: str, how: str, conf: str) -> list[str]:
|
||||
"""Lexicon checks for one facts leaf (no Ladybug)."""
|
||||
problems: list[str] = []
|
||||
src = source or ""
|
||||
if conf == CONF_CONFIRMED:
|
||||
if " vs " in src:
|
||||
problems.append(f"{lid}: confirmed fact cannot keep a vs-contradiction")
|
||||
if " x " not in src:
|
||||
problems.append(f"{lid}: needs 2-source evidence in source, got '{source}'")
|
||||
elif conf == CONF_HYPOTHESIS:
|
||||
yes, no = parse_source_field(src)
|
||||
if not no or " x " not in yes or " x " not in no:
|
||||
problems.append(
|
||||
f"{lid}: hypothesis contradiction needs 'a x b vs c x d', got '{source}'"
|
||||
)
|
||||
elif conf == "partial":
|
||||
pass
|
||||
else:
|
||||
problems.append(f"{lid}: unknown confidence '{conf}'")
|
||||
if not loc:
|
||||
problems.append(f"{lid}: missing loc (evidence pointer)")
|
||||
if not how:
|
||||
problems.append(f"{lid}: missing how")
|
||||
return problems
|
||||
@@ -127,6 +127,22 @@ class BinLayoutTest(unittest.TestCase):
|
||||
self.assertIn("cmdbin.ExecFile", text)
|
||||
self.assertIn(f"bin/facts/{method.removesuffix('.go')}", text)
|
||||
|
||||
def test_d16_adjudication_is_cgo_free(self) -> None:
|
||||
self.assertTrue((ROOT / "internal" / "facts" / "contradict.go").is_file())
|
||||
go = (ROOT / "internal" / "facts" / "contradict.go").read_text()
|
||||
py = (ROOT / "bin" / "tools" / "contradict.py").read_text()
|
||||
audit = (ROOT / "bin" / "facts" / "audit").read_text()
|
||||
for token in ("temporal_freshness", "authority_pairing", "unresolved"):
|
||||
self.assertIn(token, go)
|
||||
self.assertIn(token, py)
|
||||
self.assertIn("contradict", audit)
|
||||
self.assertIn(" vs ", py)
|
||||
plan = (ROOT / "PLAN.md").read_text()
|
||||
self.assertIn("temporal_freshness", plan)
|
||||
self.assertIn("authority_pairing", plan)
|
||||
shebang = (ROOT / "bin" / "facts" / "audit.go").read_text()
|
||||
self.assertIn("contradict", shebang)
|
||||
|
||||
def test_mail_import_is_shebang_not_brain_write(self) -> None:
|
||||
self._assert_shebang("bin/mail/import.go")
|
||||
index_mail = (ROOT / "bin" / "mail" / "index_mail").read_text()
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
|
||||
from contradict import ( # noqa: E402
|
||||
RULE_AUTHORITY,
|
||||
RULE_SINGLE,
|
||||
RULE_TEMPORAL,
|
||||
RULE_TWO_SOURCE,
|
||||
RULE_UNRESOLVED,
|
||||
adjudicate,
|
||||
check_fact_row,
|
||||
parse_source_field,
|
||||
)
|
||||
|
||||
|
||||
def src(i, kind, stale=False):
|
||||
return {"id": i, "kind": kind, "stale": stale}
|
||||
|
||||
|
||||
class TestContradict(unittest.TestCase):
|
||||
def test_two_vs_two_stays_hypothesis(self):
|
||||
r = adjudicate({
|
||||
"text": "svc listens on 443",
|
||||
"yes": [src("docker-ps", "runtime"), src("compose", "config")],
|
||||
"no": [src("docker-old", "runtime"), src("compose-old", "config")],
|
||||
})
|
||||
self.assertFalse(r["confirmed"])
|
||||
self.assertEqual(r["rule"], RULE_UNRESOLVED)
|
||||
self.assertEqual(r["winner"], "")
|
||||
|
||||
def test_temporal_freshness(self):
|
||||
r = adjudicate({
|
||||
"text": "svc listens on 443",
|
||||
"yes": [src("docker-ps", "runtime"), src("compose", "config")],
|
||||
"no": [src("old-readme", "narrative", True), src("old-wiki", "narrative", True)],
|
||||
})
|
||||
self.assertTrue(r["confirmed"])
|
||||
self.assertEqual(r["rule"], RULE_TEMPORAL)
|
||||
self.assertEqual(r["winner"], "yes")
|
||||
|
||||
def test_authority_pairing(self):
|
||||
r = adjudicate({
|
||||
"text": "svc listens on 443",
|
||||
"yes": [src("docker-ps", "runtime"), src("compose", "config")],
|
||||
"no": [src("readme", "narrative"), src("wiki", "narrative")],
|
||||
})
|
||||
self.assertTrue(r["confirmed"])
|
||||
self.assertEqual(r["rule"], RULE_AUTHORITY)
|
||||
self.assertEqual(r["winner"], "yes")
|
||||
|
||||
def test_two_source_and_single(self):
|
||||
two = adjudicate({
|
||||
"text": "arc-1 runs Matrix",
|
||||
"yes": [src("compose", "config"), src("docker-ps", "runtime")],
|
||||
})
|
||||
self.assertTrue(two["confirmed"])
|
||||
self.assertEqual(two["rule"], RULE_TWO_SOURCE)
|
||||
one = adjudicate({"text": "maybe", "yes": [src("readme", "narrative")]})
|
||||
self.assertFalse(one["confirmed"])
|
||||
self.assertEqual(one["rule"], RULE_SINGLE)
|
||||
|
||||
def test_parse_source_field(self):
|
||||
yes, no = parse_source_field("docker ps x compose.yml vs old.md x wiki.md")
|
||||
self.assertIn(" x ", yes)
|
||||
self.assertIn(" x ", no)
|
||||
|
||||
def test_check_fact_row_allows_hypothesis_vs(self):
|
||||
p = check_fact_row(
|
||||
"L1", "a.md x b.md vs c.md x d.md", "var/", "audit", "hypothesis",
|
||||
)
|
||||
self.assertEqual(p, [])
|
||||
p = check_fact_row("L2", "a.md x b.md", "var/", "audit", "confirmed")
|
||||
self.assertEqual(p, [])
|
||||
p = check_fact_row("L3", "a.md x b.md vs c.md x d.md", "var/", "audit", "confirmed")
|
||||
self.assertTrue(any("vs-contradiction" in x for x in p))
|
||||
p = check_fact_row("L4", "only-one.md", "var/", "audit", "hypothesis")
|
||||
self.assertTrue(any("a x b vs" in x for x in p))
|
||||
|
||||
def test_audit_contradict_cli_unresolved(self):
|
||||
import json
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
root = Path(__file__).resolve().parents[2]
|
||||
payload = json.dumps({
|
||||
"text": "svc 443",
|
||||
"yes": [src("a", "runtime"), src("b", "config")],
|
||||
"no": [src("c", "runtime"), src("d", "config")],
|
||||
})
|
||||
proc = subprocess.run(
|
||||
[sys.executable, str(root / "bin" / "facts" / "audit"), "contradict", "--json"],
|
||||
input=payload, capture_output=True, text=True, check=False,
|
||||
)
|
||||
self.assertEqual(proc.returncode, 0, proc.stderr)
|
||||
out = json.loads(proc.stdout)
|
||||
self.assertTrue(out["ok"])
|
||||
self.assertEqual(out["contradictions"][0]["rule"], RULE_UNRESOLVED)
|
||||
self.assertFalse(out["contradictions"][0]["confirmed"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user