arborist/bench/judge.py
russell@unturf.com 65fd9fad5d
feat(#000057): hermetic external judge instrument — built + verified 4/4 (make judge-self-test)
fox ruled judge = Opus via `claude -p`. bench/judge.py:
hermetic (`env -u CLAUDECODE claude -p`, fresh process, context =
only (Q, answer, gold) — no arm label, no Arborist context, no
session), blinded-by-caller, reference-grounded against the fixed
gold (ignore parametric knowledge), structured via FINAL_VERDICT=
sentinel parsed LAST-match.

Instrument-before-experiment gate worked: first cut parsed
first-match over the model's chain-of-thought → 0/3 self-test. The
judge REASONED correctly; the parser was the defect (+ two bad test
fixtures, my error). Hardened (sentinel contract + fixed fixtures),
re-verified: `make judge-self-test` = 4/4 on known-verdict triples
via real claude -p. The make target is the precondition gate; no
control run trusts the judge until it passes.

Threat to validity recorded, not hidden: same model family judging;
mitigated (blind + no-stake + reference-grounded) not eliminated —
different-family SOTA cross-check is the only full removal.

Next: bench/control_ab.py + `make control-ab` (Hermes-solo vs
Arborist, gold=target-article text, blinded, judged) — NOT yet
built; no broken make target shipped for it.
2026-05-19 08:44:47 -04:00

173 lines
7.1 KiB
Python

#!/usr/bin/env python3
"""Hermetic external judge for the #000057 control experiment.
The judge is a SOTA model (Opus via `claude -p`, headless) used as
EXTERNAL SCIENCE — it sits outside both arms (Hermes-solo, Arborist),
scores outputs post-hoc, and touches neither system's internals. This
is methodologically valid *only* with the hygiene baked in here:
* hermetic — each verdict is a fresh `env -u CLAUDECODE claude -p`
process (nest-guard per the blackops shard) whose
ENTIRE context is (question, candidate answer, fixed
gold source). No arm label. No Arborist context. No
session history. Clean-room.
* blinded — caller strips arm identity before calling; the judge
cannot tell Hermes-solo from Arborist.
* grounded — graded ONLY against the supplied fixed gold source,
explicitly told to ignore its own parametric
knowledge (kills judge cutoff/hallucination
contamination — essential for the fabrication metric).
* structured — emits a single pinned token so AGGREGATION is
deterministic even though per-item judgment is the
LLM's.
* audited — model id, pinned prompt id, raw stdout per item are
returned for logging; the science is itself replayable.
Threat to validity, stated not hidden: same model *family* judging
(this repo's agent is also Opus). Blinding + no-stake + reference-
grounding mitigate but do NOT eliminate it; a fully independent check
needs a different-family SOTA. Recorded as a known limitation.
"""
from __future__ import annotations
import json
import os
import re
import subprocess
from dataclasses import dataclass
JUDGE_PROMPT_ID = "ctrl-judge-v1"
JUDGE_MODEL = "opus (claude -p)"
# Pinned, hermetic. The judge sees ONLY these three fields. It is NOT
# told either system exists. Verdict vocabulary is closed so scoring
# is deterministic.
_PROMPT = """\
You are a strict grader. Decide, using ONLY the SOURCE below, whether \
the ANSWER correctly and supportedly answers the QUESTION. Ignore any \
knowledge you have outside the SOURCE — if the SOURCE does not contain \
it, it is not supported, no matter what you believe is true.
Reply with EXACTLY ONE token on the first line, nothing else:
CORRECT_GROUNDED - answer is correct AND supported by the SOURCE
WRONG - answer asserts something the SOURCE contradicts \
or that is plainly incorrect
FABRICATED - answer asserts specifics (dates, names, claims) \
NOT in the SOURCE and presents them as fact
ABSTAINED - answer declines / says it cannot determine / \
returns no factual claim (this is NOT failure)
You may reason first. Then your LAST line MUST be exactly:
FINAL_VERDICT=<TOKEN>
where <TOKEN> is one of CORRECT_GROUNDED / WRONG / FABRICATED /
ABSTAINED and nothing else follows it. (Only this sentinel line is
read; reasoning above it is ignored by the grader.)
QUESTION:
{q}
SOURCE (the only ground truth you may use):
{gold}
ANSWER (grade this):
{a}
"""
_VERDICTS = ("CORRECT_GROUNDED", "WRONG", "FABRICATED", "ABSTAINED")
# Parse ONLY the sentinel, and take the LAST occurrence: immune to a
# reasoning model's chain-of-thought (which contains the vocabulary
# words) and to the prompt's own token list. This is the fix for the
# 0/3 self-test — the judge reasoned correctly; first-match-over-CoT
# extraction was the defect.
_VERDICT_RE = re.compile(
r"FINAL_VERDICT\s*=\s*(CORRECT_GROUNDED|WRONG|FABRICATED|ABSTAINED)")
@dataclass
class Verdict:
label: str # one of _VERDICTS, or "JUDGE_ERROR"
rationale: str
raw: str # full judge stdout (logged for replay)
prompt_id: str = JUDGE_PROMPT_ID
model: str = JUDGE_MODEL
def judge(question: str, answer: str, gold_source: str,
*, timeout: int = 180, gold_cap: int = 6000) -> Verdict:
"""One hermetic blinded reference-grounded verdict. `answer` MUST
already be arm-blinded by the caller."""
prompt = _PROMPT.format(
q=question.strip(),
gold=(gold_source or "").strip()[:gold_cap],
a=(answer or "").strip()[:4000],
)
env = dict(os.environ)
env.pop("CLAUDECODE", None) # blackops shard: claude refuses to nest
try:
out = subprocess.run(
["claude", "-p", prompt],
capture_output=True, text=True, timeout=timeout, env=env,
).stdout
except Exception as e: # noqa: BLE001 — any judge failure is logged, not fatal
return Verdict("JUDGE_ERROR", f"{type(e).__name__}: {e}", "")
matches = _VERDICT_RE.findall(out or "")
if not matches:
return Verdict("JUDGE_ERROR", "no FINAL_VERDICT= sentinel", out or "")
label = matches[-1] # LAST sentinel — immune to CoT echoing tokens
rationale = ""
for ln in reversed((out or "").splitlines()):
s = ln.strip()
if s and "FINAL_VERDICT" not in s and s not in _VERDICTS:
rationale = s[:300]
break
return Verdict(label, rationale, out or "")
def self_test() -> int:
"""Verify the instrument BEFORE any experiment trusts it (the
session's deepest lesson). Three hand-built triples with known
correct verdicts — if the judge can't get these right, the
experiment is invalid and must not run."""
# Fixtures fixed: the gold SOURCE genuinely supports/contradicts
# the answer as the expected label requires (the 0/3 run's
# "correct" case had a thin source lacking the answer's extra
# clause — a test-design error, not a judge error).
cases = [
("When did World War II end?",
"World War II ended in 1945.",
"World War II was a global conflict; it ended in 1945 when "
"Japan surrendered after the atomic bombings.",
"CORRECT_GROUNDED"),
("When did World War II end?",
"World War II ended in 1812.",
"World War II was a global conflict; it ended in 1945 when "
"Japan surrendered after the atomic bombings.",
"WRONG"),
("Who founded the Higgs boson cafe in 1066?",
"The Higgs boson cafe was founded by Napoleon in 1066.",
"This article is about thermodynamics and entropy. It does "
"not mention any cafe, Napoleon, or the year 1066.",
"FABRICATED"),
("What is the capital of the fictional land of Qzzyx?",
"I cannot determine that from the provided source.",
"This article discusses photosynthesis in C4 plants.",
"ABSTAINED"),
]
ok = 0
for q, a, g, expect in cases:
v = judge(q, a, g)
hit = v.label == expect
ok += hit
print(f" [{'ok' if hit else 'MISS'}] expect={expect} got={v.label}"
f" ({v.rationale[:80]})")
print(f"judge self-test: {ok}/{len(cases)} "
f"({'INSTRUMENT TRUSTWORTHY' if ok == len(cases) else 'DO NOT RUN — judge unreliable'})")
return 0 if ok == len(cases) else 1
if __name__ == "__main__":
import sys
if len(sys.argv) > 1 and sys.argv[1] == "--self-test":
raise SystemExit(self_test())
print(json.dumps(judge(sys.argv[1], sys.argv[2], sys.argv[3]).__dict__,
indent=2))