fox ruled judge = Opus via `claude -p`. bench/judge.py: hermetic (`env -u CLAUDECODE claude -p`, fresh process, context = only (Q, answer, gold) — no arm label, no Arborist context, no session), blinded-by-caller, reference-grounded against the fixed gold (ignore parametric knowledge), structured via FINAL_VERDICT= sentinel parsed LAST-match. Instrument-before-experiment gate worked: first cut parsed first-match over the model's chain-of-thought → 0/3 self-test. The judge REASONED correctly; the parser was the defect (+ two bad test fixtures, my error). Hardened (sentinel contract + fixed fixtures), re-verified: `make judge-self-test` = 4/4 on known-verdict triples via real claude -p. The make target is the precondition gate; no control run trusts the judge until it passes. Threat to validity recorded, not hidden: same model family judging; mitigated (blind + no-stake + reference-grounded) not eliminated — different-family SOTA cross-check is the only full removal. Next: bench/control_ab.py + `make control-ab` (Hermes-solo vs Arborist, gold=target-article text, blinded, judged) — NOT yet built; no broken make target shipped for it.
173 lines
7.1 KiB
Python
173 lines
7.1 KiB
Python
#!/usr/bin/env python3
|
|
"""Hermetic external judge for the #000057 control experiment.
|
|
|
|
The judge is a SOTA model (Opus via `claude -p`, headless) used as
|
|
EXTERNAL SCIENCE — it sits outside both arms (Hermes-solo, Arborist),
|
|
scores outputs post-hoc, and touches neither system's internals. This
|
|
is methodologically valid *only* with the hygiene baked in here:
|
|
|
|
* hermetic — each verdict is a fresh `env -u CLAUDECODE claude -p`
|
|
process (nest-guard per the blackops shard) whose
|
|
ENTIRE context is (question, candidate answer, fixed
|
|
gold source). No arm label. No Arborist context. No
|
|
session history. Clean-room.
|
|
* blinded — caller strips arm identity before calling; the judge
|
|
cannot tell Hermes-solo from Arborist.
|
|
* grounded — graded ONLY against the supplied fixed gold source,
|
|
explicitly told to ignore its own parametric
|
|
knowledge (kills judge cutoff/hallucination
|
|
contamination — essential for the fabrication metric).
|
|
* structured — emits a single pinned token so AGGREGATION is
|
|
deterministic even though per-item judgment is the
|
|
LLM's.
|
|
* audited — model id, pinned prompt id, raw stdout per item are
|
|
returned for logging; the science is itself replayable.
|
|
|
|
Threat to validity, stated not hidden: same model *family* judging
|
|
(this repo's agent is also Opus). Blinding + no-stake + reference-
|
|
grounding mitigate but do NOT eliminate it; a fully independent check
|
|
needs a different-family SOTA. Recorded as a known limitation.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import re
|
|
import subprocess
|
|
from dataclasses import dataclass
|
|
|
|
JUDGE_PROMPT_ID = "ctrl-judge-v1"
|
|
JUDGE_MODEL = "opus (claude -p)"
|
|
|
|
# Pinned, hermetic. The judge sees ONLY these three fields. It is NOT
|
|
# told either system exists. Verdict vocabulary is closed so scoring
|
|
# is deterministic.
|
|
_PROMPT = """\
|
|
You are a strict grader. Decide, using ONLY the SOURCE below, whether \
|
|
the ANSWER correctly and supportedly answers the QUESTION. Ignore any \
|
|
knowledge you have outside the SOURCE — if the SOURCE does not contain \
|
|
it, it is not supported, no matter what you believe is true.
|
|
|
|
Reply with EXACTLY ONE token on the first line, nothing else:
|
|
CORRECT_GROUNDED - answer is correct AND supported by the SOURCE
|
|
WRONG - answer asserts something the SOURCE contradicts \
|
|
or that is plainly incorrect
|
|
FABRICATED - answer asserts specifics (dates, names, claims) \
|
|
NOT in the SOURCE and presents them as fact
|
|
ABSTAINED - answer declines / says it cannot determine / \
|
|
returns no factual claim (this is NOT failure)
|
|
|
|
You may reason first. Then your LAST line MUST be exactly:
|
|
FINAL_VERDICT=<TOKEN>
|
|
where <TOKEN> is one of CORRECT_GROUNDED / WRONG / FABRICATED /
|
|
ABSTAINED and nothing else follows it. (Only this sentinel line is
|
|
read; reasoning above it is ignored by the grader.)
|
|
|
|
QUESTION:
|
|
{q}
|
|
|
|
SOURCE (the only ground truth you may use):
|
|
{gold}
|
|
|
|
ANSWER (grade this):
|
|
{a}
|
|
"""
|
|
|
|
_VERDICTS = ("CORRECT_GROUNDED", "WRONG", "FABRICATED", "ABSTAINED")
|
|
# Parse ONLY the sentinel, and take the LAST occurrence: immune to a
|
|
# reasoning model's chain-of-thought (which contains the vocabulary
|
|
# words) and to the prompt's own token list. This is the fix for the
|
|
# 0/3 self-test — the judge reasoned correctly; first-match-over-CoT
|
|
# extraction was the defect.
|
|
_VERDICT_RE = re.compile(
|
|
r"FINAL_VERDICT\s*=\s*(CORRECT_GROUNDED|WRONG|FABRICATED|ABSTAINED)")
|
|
|
|
|
|
@dataclass
|
|
class Verdict:
|
|
label: str # one of _VERDICTS, or "JUDGE_ERROR"
|
|
rationale: str
|
|
raw: str # full judge stdout (logged for replay)
|
|
prompt_id: str = JUDGE_PROMPT_ID
|
|
model: str = JUDGE_MODEL
|
|
|
|
|
|
def judge(question: str, answer: str, gold_source: str,
|
|
*, timeout: int = 180, gold_cap: int = 6000) -> Verdict:
|
|
"""One hermetic blinded reference-grounded verdict. `answer` MUST
|
|
already be arm-blinded by the caller."""
|
|
prompt = _PROMPT.format(
|
|
q=question.strip(),
|
|
gold=(gold_source or "").strip()[:gold_cap],
|
|
a=(answer or "").strip()[:4000],
|
|
)
|
|
env = dict(os.environ)
|
|
env.pop("CLAUDECODE", None) # blackops shard: claude refuses to nest
|
|
try:
|
|
out = subprocess.run(
|
|
["claude", "-p", prompt],
|
|
capture_output=True, text=True, timeout=timeout, env=env,
|
|
).stdout
|
|
except Exception as e: # noqa: BLE001 — any judge failure is logged, not fatal
|
|
return Verdict("JUDGE_ERROR", f"{type(e).__name__}: {e}", "")
|
|
matches = _VERDICT_RE.findall(out or "")
|
|
if not matches:
|
|
return Verdict("JUDGE_ERROR", "no FINAL_VERDICT= sentinel", out or "")
|
|
label = matches[-1] # LAST sentinel — immune to CoT echoing tokens
|
|
rationale = ""
|
|
for ln in reversed((out or "").splitlines()):
|
|
s = ln.strip()
|
|
if s and "FINAL_VERDICT" not in s and s not in _VERDICTS:
|
|
rationale = s[:300]
|
|
break
|
|
return Verdict(label, rationale, out or "")
|
|
|
|
|
|
def self_test() -> int:
|
|
"""Verify the instrument BEFORE any experiment trusts it (the
|
|
session's deepest lesson). Three hand-built triples with known
|
|
correct verdicts — if the judge can't get these right, the
|
|
experiment is invalid and must not run."""
|
|
# Fixtures fixed: the gold SOURCE genuinely supports/contradicts
|
|
# the answer as the expected label requires (the 0/3 run's
|
|
# "correct" case had a thin source lacking the answer's extra
|
|
# clause — a test-design error, not a judge error).
|
|
cases = [
|
|
("When did World War II end?",
|
|
"World War II ended in 1945.",
|
|
"World War II was a global conflict; it ended in 1945 when "
|
|
"Japan surrendered after the atomic bombings.",
|
|
"CORRECT_GROUNDED"),
|
|
("When did World War II end?",
|
|
"World War II ended in 1812.",
|
|
"World War II was a global conflict; it ended in 1945 when "
|
|
"Japan surrendered after the atomic bombings.",
|
|
"WRONG"),
|
|
("Who founded the Higgs boson cafe in 1066?",
|
|
"The Higgs boson cafe was founded by Napoleon in 1066.",
|
|
"This article is about thermodynamics and entropy. It does "
|
|
"not mention any cafe, Napoleon, or the year 1066.",
|
|
"FABRICATED"),
|
|
("What is the capital of the fictional land of Qzzyx?",
|
|
"I cannot determine that from the provided source.",
|
|
"This article discusses photosynthesis in C4 plants.",
|
|
"ABSTAINED"),
|
|
]
|
|
ok = 0
|
|
for q, a, g, expect in cases:
|
|
v = judge(q, a, g)
|
|
hit = v.label == expect
|
|
ok += hit
|
|
print(f" [{'ok' if hit else 'MISS'}] expect={expect} got={v.label}"
|
|
f" ({v.rationale[:80]})")
|
|
print(f"judge self-test: {ok}/{len(cases)} "
|
|
f"({'INSTRUMENT TRUSTWORTHY' if ok == len(cases) else 'DO NOT RUN — judge unreliable'})")
|
|
return 0 if ok == len(cases) else 1
|
|
|
|
|
|
if __name__ == "__main__":
|
|
import sys
|
|
if len(sys.argv) > 1 and sys.argv[1] == "--self-test":
|
|
raise SystemExit(self_test())
|
|
print(json.dumps(judge(sys.argv[1], sys.argv[2], sys.argv[3]).__dict__,
|
|
indent=2))
|