#000052 §3.2 step 1: candidate-bench fixtures (13 POS + 13 NEG) + relevance_shadow_grid.py — sweep manifest models on aboutness, rank by separation margin (not raw score, per §7 #18)
This commit is contained in:
parent
9532c47f0e
commit
ba1b6b904b
2 changed files with 204 additions and 0 deletions
27
bench/fixtures/5f/relevance-aboutness-v1.jsonl
Normal file
27
bench/fixtures/5f/relevance-aboutness-v1.jsonl
Normal file
|
|
@ -0,0 +1,27 @@
|
|||
{"_meta": {"battery": "5f", "sub_battery": "relevance", "version": "relevance-aboutness-v1", "task_count": 26, "notes": "ticket #000052 §3.2 candidate-bench: pairs for evaluating a cross-encoder aboutness reranker. POS = on-topic (should score HIGH / would_demote=False). NEG = off-topic / topic-collision mis-cite / Q-A deflection (should score LOW / would_demote=True). The pair_kind is 'question_answer' OR 'claim_source'. The Zionist-entity case (neg-001) is the motivating field case from §1. Hand-built; small (n=26) but covers the failure modes — the bigger numbers come from the real-traffic shadow sweep + recall-side realism check (§3.2.2 protocol steps 2 + 3)."}}
|
||||
{"id": "pos-qa-001", "kind": "POS", "pair_kind": "question_answer", "query": "who painted the mona lisa?", "document": "Leonardo da Vinci painted the Mona Lisa in the early 16th century."}
|
||||
{"id": "pos-qa-002", "kind": "POS", "pair_kind": "question_answer", "query": "what is the capital of france?", "document": "Paris is the capital and most populous city of France."}
|
||||
{"id": "pos-qa-003", "kind": "POS", "pair_kind": "question_answer", "query": "when did world war 2 end?", "document": "World War 2 ended in 1945 with the surrender of Germany and Japan."}
|
||||
{"id": "pos-qa-004", "kind": "POS", "pair_kind": "question_answer", "query": "where is mount everest?", "document": "Mount Everest lies in the Mahalangur range of the Himalayas on the Nepal-China border."}
|
||||
{"id": "pos-qa-005", "kind": "POS", "pair_kind": "question_answer", "query": "who founded microsoft?", "document": "Bill Gates and Paul Allen founded Microsoft in 1975."}
|
||||
{"id": "pos-qa-006", "kind": "POS", "pair_kind": "question_answer", "query": "what is photosynthesis?", "document": "Photosynthesis is the biological process by which plants convert sunlight, water and carbon dioxide into chemical energy stored as glucose."}
|
||||
{"id": "pos-qa-007", "kind": "POS", "pair_kind": "question_answer", "query": "who wrote hamlet?", "document": "William Shakespeare wrote the tragedy Hamlet around the year 1600."}
|
||||
{"id": "pos-cs-001", "kind": "POS", "pair_kind": "claim_source", "query": "Albert Einstein developed the theory of relativity.", "document": "Albert Einstein was a German-born theoretical physicist whose theory of relativity revolutionized 20th-century physics."}
|
||||
{"id": "pos-cs-002", "kind": "POS", "pair_kind": "claim_source", "query": "Pluto is no longer classified as a planet.", "document": "In 2006 the International Astronomical Union reclassified Pluto as a dwarf planet, removing it from the list of full planets of the Solar System."}
|
||||
{"id": "pos-cs-003", "kind": "POS", "pair_kind": "claim_source", "query": "DNA has a double-helical structure.", "document": "DNA, deoxyribonucleic acid, has a double-helical structure first described by Watson and Crick in 1953."}
|
||||
{"id": "pos-cs-004", "kind": "POS", "pair_kind": "claim_source", "query": "Mount Kilimanjaro is located in Tanzania.", "document": "Mount Kilimanjaro is a dormant volcano in north-eastern Tanzania and the highest mountain in Africa."}
|
||||
{"id": "pos-cs-005", "kind": "POS", "pair_kind": "claim_source", "query": "Paris is the capital of France.", "document": "Paris is the capital and most populous city of France, situated on the Seine River in northern France."}
|
||||
{"id": "pos-cs-006", "kind": "POS", "pair_kind": "claim_source", "query": "The phrase 'Zionist entity' is sometimes used as a pejorative for the State of Israel.", "document": "Zionist entity () is a phrase sometimes used by Arabs and Muslims as a pejorative for the State of Israel."}
|
||||
{"id": "pos-cs-007", "kind": "POS", "pair_kind": "claim_source", "query": "Beethoven composed nine symphonies.", "document": "Ludwig van Beethoven composed nine symphonies, the ninth incorporating Schiller's Ode to Joy in a choral finale."}
|
||||
{"id": "neg-qa-001", "kind": "NEG", "pair_kind": "question_answer", "query": "who painted the mona lisa?", "document": "The Mona Lisa is housed in the Louvre Museum and is one of the most visited paintings in the world, drawing millions of tourists every year."}
|
||||
{"id": "neg-qa-002", "kind": "NEG", "pair_kind": "question_answer", "query": "what is the capital of france?", "document": "France is a country in Western Europe with a population of about 67 million people and a rich culinary tradition."}
|
||||
{"id": "neg-qa-003", "kind": "NEG", "pair_kind": "question_answer", "query": "when did world war 2 end?", "document": "World War 2 was a global conflict that involved most of the world's major nations and reshaped the political map of the 20th century."}
|
||||
{"id": "neg-qa-004", "kind": "NEG", "pair_kind": "question_answer", "query": "where is mount everest?", "document": "Mount Kilimanjaro is the highest mountain in Africa, located in north-eastern Tanzania."}
|
||||
{"id": "neg-qa-005", "kind": "NEG", "pair_kind": "question_answer", "query": "who founded microsoft?", "document": "Microsoft Office is a suite of productivity software including Word, Excel, and PowerPoint, used widely in business and education."}
|
||||
{"id": "neg-qa-006", "kind": "NEG", "pair_kind": "question_answer", "query": "what is photosynthesis?", "document": "Plants are living organisms that grow in soil; they produce oxygen and are a major source of food for many animals."}
|
||||
{"id": "neg-cs-001", "kind": "NEG", "pair_kind": "claim_source", "query": "the phrase 'Zionist entity' is sometimes used as the entity, referring to the State of Israel.", "document": "In grammar, a phrase is a group of words functioning as a single unit in the syntax of a sentence. Phrases can act as nouns, adjectives, or other parts of speech.", "note": "the motivating field case — the garbled claim against the Phrase grammar article"}
|
||||
{"id": "neg-cs-002", "kind": "NEG", "pair_kind": "claim_source", "query": "Mercury is the smallest planet in the Solar System.", "document": "Mercury is the Roman god of commerce, messengers and travelers, the equivalent of the Greek Hermes."}
|
||||
{"id": "neg-cs-003", "kind": "NEG", "pair_kind": "claim_source", "query": "Java is a programming language developed by Sun Microsystems.", "document": "Java is an island in Indonesia with a population of over 140 million, the most populous island in the world."}
|
||||
{"id": "neg-cs-004", "kind": "NEG", "pair_kind": "claim_source", "query": "Apple Inc. was founded by Steve Jobs.", "document": "Apples are a popular fruit grown in many temperate parts of the world; common cultivars include Gala, Fuji and Granny Smith."}
|
||||
{"id": "neg-cs-005", "kind": "NEG", "pair_kind": "claim_source", "query": "The Boltzmann constant relates particle energy to absolute temperature.", "document": "Ludwig Boltzmann was born in Vienna, Austria in 1844; he studied at the University of Vienna and held professorships in several European cities."}
|
||||
{"id": "neg-cs-006", "kind": "NEG", "pair_kind": "claim_source", "query": "Einstein developed general relativity.", "document": "Charles Darwin developed the theory of evolution by natural selection, set out in his 1859 work On the Origin of Species."}
|
||||
177
bench/scripts/relevance_shadow_grid.py
Normal file
177
bench/scripts/relevance_shadow_grid.py
Normal file
|
|
@ -0,0 +1,177 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Relevance / aboutness candidate-bench — #000052 §3.2 step 1.
|
||||
|
||||
Score every (query, document) pair in the candidate-bench fixture set
|
||||
with every model in the relevance manifest's primary + alternates,
|
||||
then report per-model:
|
||||
|
||||
- score distribution by kind (POS / NEG)
|
||||
- separation margin: min(POS_score) − max(NEG_score) (>0 = clean separable)
|
||||
- best θ for fp=0 (NO pos scores below θ) with max recall on NEG (catch = NEG_score < θ)
|
||||
- Pareto frontier (recall on NEG vs FP on POS as θ varies)
|
||||
|
||||
The winner is picked by *separation margin*, not raw score — per the
|
||||
#000049 §7 #18 lesson ('specific checkpoint + score-shape matter, not
|
||||
parameter count'). This is candidate-bench only — it does NOT set the
|
||||
manifest's `demote_below_score`; that requires the real-traffic shadow
|
||||
sweep (#000052 §3.2.2 step 2) over pooled bench-qa STRICT.
|
||||
|
||||
Usage:
|
||||
python3 bench/scripts/relevance_shadow_grid.py
|
||||
python3 bench/scripts/relevance_shadow_grid.py --fixtures path.jsonl --out results.json
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
REPO = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
def _records(path: Path):
|
||||
for ln in path.read_text().splitlines():
|
||||
ln = ln.strip()
|
||||
if not ln:
|
||||
continue
|
||||
o = json.loads(ln)
|
||||
if "_meta" in o:
|
||||
continue
|
||||
yield o
|
||||
|
||||
|
||||
def _models_from_manifest():
|
||||
sys.path.insert(0, str(REPO))
|
||||
from arborist.qa.relevance.shadow import load_manifest
|
||||
m = load_manifest()
|
||||
out = [{"relevance_model_version": m["relevance_model_version"],
|
||||
"hf_repo": m["hf_repo"], "pinned_revision": m.get("pinned_revision")}]
|
||||
for a in m.get("alternates", []):
|
||||
out.append({"relevance_model_version": a["relevance_model_version"],
|
||||
"hf_repo": a["hf_repo"], "pinned_revision": a.get("pinned_revision")})
|
||||
return out
|
||||
|
||||
|
||||
def _frontier(pos_scores: list[float], neg_scores: list[float]) -> dict:
|
||||
"""At each candidate θ (sweep all observed scores), compute
|
||||
catch (frac of NEG < θ) and fp (frac of POS < θ); return:
|
||||
- max-catch-at-fp-zero point
|
||||
- separation margin (min POS − max NEG; >0 = clean linear-separable)
|
||||
- per-θ pareto (sorted)"""
|
||||
if not pos_scores or not neg_scores:
|
||||
return {"separation_margin": None, "fp0_best": None, "pareto": []}
|
||||
sep = min(pos_scores) - max(neg_scores)
|
||||
# sweep θ over the union of observed scores; for "catch NEG below θ":
|
||||
# θ above max(NEG) catches all NEG (fp=full); θ above min(POS) fp's; want fp=0.
|
||||
ts = sorted(set(pos_scores + neg_scores))
|
||||
pts = []
|
||||
for t in ts:
|
||||
catch = sum(1 for s in neg_scores if s < t) / len(neg_scores)
|
||||
fp = sum(1 for s in pos_scores if s < t) / len(pos_scores)
|
||||
pts.append({"theta": round(t, 4), "catch_neg": round(catch, 4),
|
||||
"fp_on_pos": round(fp, 4)})
|
||||
fp0 = [p for p in pts if p["fp_on_pos"] == 0.0]
|
||||
fp0_best = max(fp0, key=lambda p: p["catch_neg"]) if fp0 else None
|
||||
# Pareto (max catch at each fp level)
|
||||
pareto = []
|
||||
best = -1.0
|
||||
for p in sorted(pts, key=lambda p: (p["fp_on_pos"], -p["catch_neg"])):
|
||||
if p["catch_neg"] > best:
|
||||
pareto.append(p); best = p["catch_neg"]
|
||||
return {"separation_margin": round(sep, 4), "fp0_best": fp0_best, "pareto": pareto}
|
||||
|
||||
|
||||
def main(argv=None) -> int:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--fixtures", type=Path,
|
||||
default=REPO / "bench" / "fixtures" / "5f" / "relevance-aboutness-v1.jsonl")
|
||||
ap.add_argument("--out", type=Path,
|
||||
default=REPO / "bench" / "results" / "relevance-aboutness-grid.json")
|
||||
args = ap.parse_args(argv)
|
||||
|
||||
try:
|
||||
import torch # noqa
|
||||
sys.path.insert(0, str(REPO))
|
||||
from arborist.qa.relevance.shadow import ShadowRelevance
|
||||
except ImportError as e:
|
||||
print(f"[relevance-grid] missing dependency: {e} (pip install 'arborist[nli]')",
|
||||
file=sys.stderr)
|
||||
return 2
|
||||
|
||||
recs = list(_records(args.fixtures))
|
||||
pos = [r for r in recs if r["kind"] == "POS"]
|
||||
neg = [r for r in recs if r["kind"] == "NEG"]
|
||||
print(f"[relevance-grid] fixtures: {len(pos)} POS (on-topic) + {len(neg)} NEG (off-topic) "
|
||||
f"= {len(recs)} total", flush=True)
|
||||
if not pos or not neg:
|
||||
print("[relevance-grid] need both POS and NEG", file=sys.stderr); return 2
|
||||
|
||||
models = _models_from_manifest()
|
||||
print(f"[relevance-grid] models ({len(models)}): "
|
||||
f"{[m['relevance_model_version'] for m in models]}", flush=True)
|
||||
|
||||
report = {"generated_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
||||
"fixtures_file": str(args.fixtures),
|
||||
"n_pos": len(pos), "n_neg": len(neg), "models": {}}
|
||||
|
||||
for me in models:
|
||||
mv = me["relevance_model_version"]
|
||||
print(f"[relevance-grid] === {mv} ({me['hf_repo']}) ===", flush=True)
|
||||
manifest_override = {
|
||||
"relevance_model_version": mv, "hf_repo": me["hf_repo"],
|
||||
"pinned_revision": me.get("pinned_revision"),
|
||||
"max_length": 256, "demote_below_score": None,
|
||||
}
|
||||
rel = ShadowRelevance(manifest=manifest_override)
|
||||
rel._ensure_loaded()
|
||||
if not rel.available:
|
||||
print(f"[relevance-grid] skip — {rel._reason}", file=sys.stderr); continue
|
||||
t0 = time.time()
|
||||
pos_scores = [rel._score_batch([(r["query"], r["document"])])[0] for r in pos]
|
||||
neg_scores = [rel._score_batch([(r["query"], r["document"])])[0] for r in neg]
|
||||
infer_s = time.time() - t0
|
||||
fr = _frontier(pos_scores, neg_scores)
|
||||
print(f"[relevance-grid] backend={rel.backend} device={rel.device} · "
|
||||
f"{len(pos)+len(neg)} pairs in {infer_s:.2f}s", flush=True)
|
||||
print(f"[relevance-grid] min(POS)={min(pos_scores):+.3f} max(NEG)={max(neg_scores):+.3f} "
|
||||
f"separation margin = {fr['separation_margin']:+.3f}"
|
||||
+ (" ← CLEAN-SEPARABLE" if fr['separation_margin'] > 0 else " ← NOT clean-separable"), flush=True)
|
||||
if fr["fp0_best"]:
|
||||
print(f"[relevance-grid] best fp=0: θ={fr['fp0_best']['theta']} → "
|
||||
f"catch={fr['fp0_best']['catch_neg']:.3f} on {len(neg)} NEG (off-topic)", flush=True)
|
||||
report["models"][mv] = {
|
||||
"hf_repo": me["hf_repo"], "backend": rel.backend, "device": rel.device,
|
||||
"n_pairs": len(pos) + len(neg), "infer_seconds": round(infer_s, 2),
|
||||
"pos_scores": {r["id"]: round(s, 4) for r, s in zip(pos, pos_scores)},
|
||||
"neg_scores": {r["id"]: round(s, 4) for r, s in zip(neg, neg_scores)},
|
||||
"frontier": fr,
|
||||
}
|
||||
|
||||
args.out.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.out.write_text(json.dumps(report, indent=2))
|
||||
print(f"[relevance-grid] wrote {args.out}")
|
||||
|
||||
print("\n" + "=" * 100)
|
||||
print("RELEVANCE CANDIDATE-BENCH SUMMARY (rank by separation margin — clean-separable wins)")
|
||||
print("=" * 100)
|
||||
print(f"{'model':<48} {'sep margin':>11} {'fp=0 θ':>9} {'fp=0 catch (NEG)':>18}")
|
||||
print("-" * 100)
|
||||
ranked = sorted(report["models"].items(),
|
||||
key=lambda kv: -((kv[1]["frontier"]["separation_margin"] or -99)))
|
||||
for mv, md in ranked:
|
||||
fr = md["frontier"]; sep = fr["separation_margin"]; b = fr["fp0_best"]
|
||||
sep_s = f"{sep:+.3f}" if sep is not None else "—"
|
||||
bθ = f"{b['theta']}" if b else "—"
|
||||
bc = f"{b['catch_neg']:.3f}" if b else "0.000 (no fp=0 θ)"
|
||||
print(f"{mv:<48} {sep_s:>11} {bθ:>9} {bc:>18}")
|
||||
print("-" * 100)
|
||||
print("separation margin = min(POS_score) − max(NEG_score). >0 means a single θ separates all POS from all NEG;")
|
||||
print("the larger the margin, the more robust the model is to threshold drift on bigger samples (§7 #18 lesson).")
|
||||
print("This is candidate-bench ONLY — see #000052 §3.2.2 for the real-traffic shadow + recall-side realism steps.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Loading…
Add table
Add a link
Reference in a new issue