#!/usr/bin/env python3 """#000057 v1 — the control experiment Arborist never had. Single-shot treatment-vs-control on the SAME model (Hermes-3-8B): arm A Hermes-solo : question only, no Arborist (the control — pure parametric answer, fair chance to abstain) arm B Arborist : the full query() pipeline (the treatment) Judged by the hermetic external Opus judge (`bench/judge.py`), graded against a FIXED gold = the mined question's target-article text fetched by `target_root` (identical for both arms, independent of either arm's retrieval — the §4b ruling). The headline is NOT raw accuracy; it is the fabrication-vs-honest-abstention delta: when a model does not know, does Hermes-solo confidently fabricate while Arborist fails closed (UNGROUNDED)? Science hygiene enforced here: * blinding — Arborist's `[E… | … | …:"…"]` evidence scaffolding is stripped so answer FORMAT cannot betray the arm to the judge; both arms reach the judge as plain prose. The judge call already carries no arm label. * symmetry — same model, same gold, same judge, same prompt shape; the solo arm gets a neutral system that PERMITS "I don't know" (else the abstention comparison is rigged). * abstention — Arborist `audit_mode==UNGROUNDED` is credited as honest abstention regardless of the judge's content verdict: the system flagged its own uncertainty, which is the value claim. Hermes-solo has no such mechanism — that asymmetry IS the result. * audited — every (question, arm, answer, gold-sha, judge raw, verdict, arborist audit_mode) logged JSONL; the science measuring auditability is itself replayable. Gate: `make control-ab` depends on `make judge-self-test` — the instrument must verify 4/4 before any run trusts it. v1 = smallest proof: small N, single-shot, one fixture. Bounded spend (N items → 2N Hermes + 2N `claude -p`). """ from __future__ import annotations import argparse import hashlib import json import os import re import sqlite3 import sys import time from pathlib import Path ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT)) # Judge selection at run time (fox 2026-05-19, see CLAUDE.md # 'Budget discipline'). Both modules share the Verdict shape so the # downstream record-emit path is judge-agnostic. import bench.judge as _judge_opus # noqa: E402 import bench.judge_code as _judge_code # noqa: E402 _JUDGES = { "code": (_judge_code.judge, _judge_code.JUDGE_MODEL), "opus": (_judge_opus.judge, _judge_opus.JUDGE_MODEL), } _EVID = re.compile(r"\[E\d+[^\]]*\]", re.S) # [E1 | Title | hash:"…"] _PTR = re.compile(r"\s*\[E\d+(?:\s*,\s*E\d+)*\]") # bare [E1] / [E1,E2] def _descaffold(text: str) -> str: """Strip Arborist evidence markup so answer FORMAT cannot tell the judge which arm produced it (blinding). Leaves the prose.""" t = _EVID.sub("", text or "") t = _PTR.sub("", t) t = re.sub(r"^[\s\-•*]+", "", t, flags=re.M) return re.sub(r"\n{2,}", "\n", t).strip() def _gold(shards_dir: Path, shard: str, root: str, cap_chunks: int = 6) -> str: db = shards_dir / shard if not db.exists(): return "" c = sqlite3.connect(db) try: rows = c.execute( "SELECT content FROM chunks WHERE document_root=? " "AND content IS NOT NULL ORDER BY idx LIMIT ?", (root, cap_chunks), ).fetchall() except sqlite3.OperationalError: return "" finally: c.close() if not rows: return "" from arborist.compress import unpack_chunk return " ".join(unpack_chunk(r[0]) or "" for r in rows) _SOLO_SYS = ("Answer the question concisely. If you do not know or " "cannot determine it, say you do not know — do not guess.") def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("--fixture", default="bench/qa_questions_numeral_map.json") ap.add_argument("--n", type=int, default=12) # smallest-proof default ap.add_argument("--shards-dir", default=str(Path.home() / ".arborist" / "shards")) ap.add_argument("--endpoint", default=os.environ.get( "ARBORIST_LLM_ENDPOINT", "https://hermes.ai.unturf.com/v1")) ap.add_argument("--model", default=os.environ.get( "ARBORIST_LLM_MODEL", "adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic")) ap.add_argument("--out-dir", default="bench/qa_results") ap.add_argument("--answer-mode", choices=["quote", "claim_lattice"], default="claim_lattice", help="STOCK V.1 substrate-ON answer shape (the two-cell " "config family). Treatment arm B runs the frozen " "bench.stock_v1 policy for this mode.") ap.add_argument("--judge", choices=sorted(_JUDGES.keys()), default="code", help="which judge to use. 'code' (default, no LLM) is " "deterministic / no quota; 'opus' is the gated " "headless Opus judge (requires " "ARBORIST_JUDGE_ENABLE=1).") a = ap.parse_args() judge, _judge_model_id = _JUDGES[a.judge] from arborist.qa.client import OpenAICompatibleClient from arborist.qa.query import query # STOCK V.1 frozen substrate-ON policy. assert_not_drifted halts the # run if DEFAULT_QUERY_POLICY changed under us, so a mid-campaign edit # can't silently redefine "substrate-ON". See bench/stock_v1.py. from bench.stock_v1 import assert_not_drifted as _assert_stock from bench.stock_v1 import policy_for as _stock_policy_for _assert_stock(a.answer_mode) arb_policy = _stock_policy_for(a.answer_mode) items = json.loads(Path(a.fixture).read_text())[:a.n] shards_dir = Path(a.shards_dir) client = OpenAICompatibleClient( base_url=a.endpoint, api_key=os.environ.get("ARBORIST_LLM_API_KEY")) ts = time.strftime("%Y-%m-%dT%H-%M-%SZ", time.gmtime()) outp = Path(a.out_dir) / f"control_ab_{ts}.jsonl" outp.parent.mkdir(parents=True, exist_ok=True) qa_db = Path("/tmp") / f"control_ab_{ts}.db" from collections import Counter tally = {"solo": Counter(), "arborist": Counter()} rows = [] with open(outp, "w") as log: for i, it in enumerate(items, 1): q, root, shard = (it["question"], it["target_root"], it.get("shard", "")) gold = _gold(shards_dir, shard, root) if not gold: print(f" [{i}] SKIP (no gold): {q[:60]}") continue gsha = hashlib.sha256(gold.encode()).hexdigest()[:12] # arm A — Hermes-solo (control). `model` is keyword-only # and REQUIRED on chat_completion; omitting it raised # TypeError every call → the solo arm silently produced # `[solo-error: TypeError]`, which the judge correctly read # as ABSTAINED. That made every prior "solo abstained" # smoke a broken-arm artefact, not a result. Pass the same # model the Arborist arm uses (symmetry). try: solo = client.chat_completion( [{"role": "system", "content": _SOLO_SYS}, {"role": "user", "content": q}], model=a.model) except Exception as e: # noqa: BLE001 solo = f"[solo-error: {type(e).__name__}: {e}]" # arm B — Arborist (treatment) try: r = query(question=q, qa_db=qa_db, chat_client=client, model_id=a.model, shards_dir=shards_dir, policy=arb_policy) arb_raw = r.get("raw_answer") or r.get("answer_text") or "" arb_mode = r.get("audit_mode") except Exception as e: # noqa: BLE001 arb_raw, arb_mode = f"[arborist-error: {type(e).__name__}]", "ERROR" # blinded judge (plain prose both arms; no arm label) vs = judge(q, _descaffold(solo), gold) va = judge(q, _descaffold(arb_raw), gold) tally["solo"][vs.label] += 1 # Arborist UNGROUNDED == honest abstention (it flagged its # own uncertainty — the value claim), regardless of the # content verdict on its underlying guess. arb_eff = ("ABSTAINED" if arb_mode == "UNGROUNDED" and va.label in ("WRONG", "FABRICATED") else va.label) tally["arborist"][arb_eff] += 1 rec = {"i": i, "question": q, "gold_sha": gsha, "solo_answer": solo[:1500], "solo_verdict": vs.label, "solo_judge_raw": vs.raw[:800], "arb_answer": arb_raw[:1500], "arb_audit_mode": arb_mode, "arb_verdict_judge": va.label, "arb_verdict_eff": arb_eff, "arb_judge_raw": va.raw[:800], "judge_model": vs.model, "judge_prompt_id": vs.prompt_id} rows.append(rec) log.write(json.dumps(rec, ensure_ascii=False) + "\n") log.flush() print(f" [{i}/{len(items)}] solo={vs.label:16} " f"arb={arb_eff:16} (arb_mode={arb_mode})") n = len(rows) if not n: print("no scored items (no gold found in fixture/shards)") return 1 def rate(arm, *labels): return sum(tally[arm][x] for x in labels) print(f"\n=== #000057 v1 control A/B (n={n}, single-shot, " f"model={a.model}; answer_mode={a.answer_mode}; " f"judge={a.judge}) ===") print(f"fixture={a.fixture} log={outp}") for arm in ("solo", "arborist"): c = tally[arm] print(f" {arm:9} CORRECT={c['CORRECT_GROUNDED']:>3} " f"WRONG={c['WRONG']:>3} FABRICATED={c['FABRICATED']:>3} " f"ABSTAINED={c['ABSTAINED']:>3} " f"JUDGE_ERR={c['JUDGE_ERROR']:>2}") solo_bad = rate("solo", "WRONG", "FABRICATED") arb_bad = rate("arborist", "WRONG", "FABRICATED") print(f"\nHEADLINE — confidently-wrong (WRONG+FABRICATED), the " f"claim Arborist makes:") print(f" Hermes-solo {solo_bad}/{n} ({solo_bad/n:.0%}) -> " f"Arborist {arb_bad}/{n} ({arb_bad/n:.0%}) Δ {arb_bad-solo_bad:+d}" f" (Arborist-UNGROUNDED credited as honest abstention)") print("\nTHREATS TO VALIDITY (stated, not hidden):") print(" * same model FAMILY judges (repo agent is also Opus) — " "blinded+no-stake+reference-grounded mitigate, NOT eliminate;") print(" * gold = target-article text, not a curated short answer " "— judge reads source, but article completeness varies;") print(f" * n={n}, single-shot — directional, not a confidence " "interval; bump --n + re-run before any strong claim;") print(" * fixture is mined (obscure entities → parametric memory " "weak) — partially claim-isolating, not maximally adversarial;") print(" * solo given a neutral abstention-permitting system " "prompt (symmetry); no prompt tuning either arm.") return 0 if __name__ == "__main__": raise SystemExit(main())