arborist/scripts/bench_emergent.py
russell@unturf.com 8d6961fcc1
aborist/arborist
modified:   .gitlab-ci.yml
	modified:   bench/qa_questions.txt
	modified:   bench/qa_sweep.py
	modified:   bench/run.sh
	modified:   docs/TICKETS.md
	modified:   docs/_source/README.md
	modified:   docs/_source/_ext/makefile_targets.py
	modified:   docs/_source/api/cli.rst
	modified:   docs/_source/api/distill.rst
	modified:   docs/_source/api/mesh.rst
	modified:   docs/_source/api/qa.rst
	modified:   docs/_source/api/retrieval.rst
	modified:   docs/_source/api/storage.rst
	modified:   docs/_source/api/substrate.rst
	modified:   docs/_source/concepts.rst
	modified:   docs/_source/conf.py
	modified:   docs/_source/cookbook.rst
	modified:   docs/_source/index.rst
	modified:   docs/_source/license.rst
	modified:   docs/_source/quickstart.rst
	modified:   docs/bench-maxing.md
	modified:   docs/benchmarks.md
	modified:   docs/cti-architecture.md
	modified:   docs/diagrams/aborist-modules.dot
	modified:   docs/diagrams/aborist-modules.svg
	modified:   docs/diagrams/mesh-data-flow.dot
	modified:   docs/diagrams/mesh-epoch-lifecycle.dot
	modified:   docs/diagrams/mesh-epoch-lifecycle.svg
	modified:   docs/diagrams/mesh-group-decisions.dot
	modified:   docs/diagrams/mesh-group-decisions.svg
	modified:   docs/diagrams/mesh-identity-stack.dot
	modified:   docs/diagrams/mesh-secret-envelope.dot
	modified:   docs/mesh.md
	modified:   docs/qa-modes-bench.md
	modified:   docs/seven-point-program.md
	modified:   docs/tickets/ticket-000001-retrieval-keywords-audit-gap.md
	modified:   docs/tickets/ticket-000002-reference-frame-polarity-contract.md
	modified:   docs/tickets/ticket-000003-anchor-class-warrant.md
	modified:   docs/tickets/ticket-000005-label-ladder-migration.md
	modified:   docs/tickets/ticket-000006-bench-emergent-findings.md
	modified:   docs/tickets/ticket-000007-query-layer-hyphen-fold.md
	modified:   docs/tickets/ticket-000008-broad-quantifier-preflight-guard.md
	modified:   docs/tickets/ticket-000009-quantifier-preflight-dag-binding.md
	modified:   docs/tickets/ticket-000010-metacognition-preflight-guard.md
	modified:   docs/tickets/ticket-000011-soft-preflight-hint-sidecar.md
	modified:   scripts/backfill_concepts.py
	modified:   scripts/bench_emergent.py
	modified:   tests/crawler/test_async_web_fetcher.py
	modified:   tests/crawler/test_bridge.py
	modified:   tests/crawler/test_web_fetch.py
	modified:   tests/test_bench_qa_sweep.py
	modified:   tests/test_burn.py
	modified:   tests/test_burn_doc.py
	modified:   tests/test_claim_lattice.py
	modified:   tests/test_cli_render.py
	modified:   tests/test_compress.py
	modified:   tests/test_concepts.py
	modified:   tests/test_dag.py
	modified:   tests/test_directives.py
	modified:   tests/test_distill.py
	modified:   tests/test_distill_recursive.py
	modified:   tests/test_evict.py
	modified:   tests/test_frame.py
	modified:   tests/test_grok_source.py
	modified:   tests/test_html_source.py
	modified:   tests/test_ingest.py
	modified:   tests/test_inspect.py
	modified:   tests/test_journal.py
	modified:   tests/test_keys.py
	modified:   tests/test_llm_context_base.py
	modified:   tests/test_merkle.py
	modified:   tests/test_mesh.py
	modified:   tests/test_mesh_aead.py
	modified:   tests/test_mesh_chain.py
	modified:   tests/test_mesh_cli.py
	modified:   tests/test_mesh_cli_pull.py
	modified:   tests/test_mesh_wire.py
	modified:   tests/test_mesh_wire_e2e.py
	modified:   tests/test_metacognition.py
	modified:   tests/test_migration_audit_mode.py
	modified:   tests/test_providence_source.py
	modified:   tests/test_qa.py
	modified:   tests/test_qa_quality_live.py
	modified:   tests/test_quantifier_caps.py
	modified:   tests/test_quantifier_classifier.py
	modified:   tests/test_quantifier_phase4.py
	modified:   tests/test_quantifier_reminder.py
	modified:   tests/test_query.py
	modified:   tests/test_reclassify.py
	modified:   tests/test_repair.py
	modified:   tests/test_resume.py
	modified:   tests/test_snapshot.py
	modified:   tests/test_soft_preflight.py
	modified:   tests/test_tfidf.py
	modified:   tests/test_vcs_source.py
	modified:   tests/test_verify.py
	modified:   tests/test_verify_json.py
	modified:   tests/test_versioned_ingest.py
	modified:   tests/test_warrant.py
	modified:   tests/test_wikipedia_old.py
	modified:   tests/test_wikipedia_xml.py
	modified:   tests/test_wikitext.py
2026-05-07 09:31:49 -04:00

375 lines
13 KiB
Python

"""Emergent stress test — random word triangulation against the substrate.
Pick three random words from /usr/share/dict/words, ask Hermes at
temp 0.8 to weave them into a creative question paragraph, send that
paragraph to arborist, log the whole journey (words → question →
answer → audit_mode → sources → timings) as a single JSONL line.
Designed for **blue moon cadence**, NOT every-commit benching. The
combinatoric word space surfaces failure modes the curated bench
(`bench/qa_questions.txt`) doesn't reach: question shapes the
authors didn't anticipate, vocabulary the corpus barely covers,
adversarial premises that emerge by accident.
The teacher review step is intentionally **not** automated here.
Fox brings interesting log entries to a teacher model (Claude
Opus 4.7 in the current setup) and asks for guidance:
- did the answer match the question? (mismatch → bench-max signal)
- novelty class: known_truth_grounding / emergent_synthesis /
novel_claim / no_signal (accounting for the 2010-11 Wikipedia
corpus legitimately lacking post-2010 advances in science /
math / engineering)
- which hyperparam to tune for the next iteration
Future: `--generator-endpoint` & `--student-endpoint` to swap
upstreams (different model per role), and a separate review-side
script that prompts a teacher model via API.
Usage:
make bench-emergent # 10 cycles, default
make bench-emergent N=50 SEED=42 # bigger sample, reproducible
python scripts/bench_emergent.py --n 10 --seed 42
python scripts/bench_emergent.py --print-pending # show un-reviewed entries
"""
from __future__ import annotations
import argparse
import json
import os
import random
import re
import sys
import time
from pathlib import Path
# All paths default to the repo's bench/ tree so re-running across
# branches accumulates a continuous log of every emergent cycle.
DEFAULT_WORDS_PATH = Path("/usr/share/dict/words")
DEFAULT_LOG_PATH = Path("bench/emergent_log.jsonl")
# Word filter: skip too-short / proper-noun / abbreviation / weird
# punctuation. The goal is "common English content tokens" that
# Hermes can actually weave into a coherent paragraph.
_WORD_RE = re.compile(r"^[a-z]{5,12}$")
def pick_words(
n: int = 3,
*,
words_path: Path = DEFAULT_WORDS_PATH,
rng: random.Random | None = None,
) -> list[str]:
"""Sample N random words satisfying the content-token filter."""
rng = rng or random.Random()
pool = [
w for w in words_path.read_text().splitlines()
if _WORD_RE.fullmatch(w.strip().lower())
]
if len(pool) < n:
raise RuntimeError(
f"word pool too small ({len(pool)} usable words at "
f"{words_path}); expected ≥{n}"
)
return rng.sample(pool, n)
GENERATOR_PROMPT = """\
Write a single short paragraph (2-3 sentences) that uses ALL THREE \
of these words:
- {w1}
- {w2}
- {w3}
Then phrase the paragraph as a question that explores the connections \
between them. The question should be the kind a curious reader would \
genuinely ask about how these three concepts relate.
Output ONLY the question paragraph. No prefix, no commentary, no \
quotes around it, no numbering. Just the question.\
"""
def generate_question(
client,
model_id: str,
words: list[str],
*,
temperature: float = 0.8,
max_tokens: int = 256,
) -> str:
"""Hermes at creative temperature weaves the 3 words into a question."""
prompt = GENERATOR_PROMPT.format(w1=words[0], w2=words[1], w3=words[2])
response = client.chat_completion(
messages=[{"role": "user", "content": prompt}],
model=model_id,
temperature=temperature,
max_tokens=max_tokens,
)
text = response.strip() if isinstance(response, str) else str(response).strip()
# Strip wrapping quotes / leading "Question:" labels in case the
# model ignored the no-prefix instruction.
text = text.removeprefix("Question:").strip()
if text.startswith('"') and text.endswith('"'):
text = text[1:-1]
return text
def run_one_cycle(
*,
words: list[str],
client,
model_id: str,
qa_db: Path,
shards_dir: Path,
answer_mode: str,
top_k: int,
) -> dict:
"""One emergent cycle: words → question → answer → log entry dict."""
from arborist.qa.query import DEFAULT_QUERY_POLICY, query
t_pick = time.time()
question = generate_question(client, model_id, words)
t_question = time.time()
policy = dict(DEFAULT_QUERY_POLICY)
policy["answer_mode"] = answer_mode
try:
result = query(
question=question,
qa_db=qa_db,
chat_client=client,
model_id=model_id,
shards_dir=shards_dir,
top_k=top_k,
policy=policy,
burn_existing=True, # always fresh — this is an emergent test
)
student_error = None
except Exception as e:
result = {}
student_error = f"{type(e).__name__}: {e}"
t_answer = time.time()
# Sidecar smell signal: metaphorical question framing the answer
# ignored. Surfaced 2026-05-02 by the swallowtail/upbraided/rockiest
# emergent log entry. Off-the-binary-chain — gives the teacher
# reviewer a flag for poetic-question-vs-literal-answer mismatches.
from arborist.qa.inspect import diagnose_metaphor_deflection
metaphor = diagnose_metaphor_deflection(
question or "", result.get("answer_text") or ""
)
return {
"ts": int(t_pick),
"iso_ts": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(t_pick)),
"words": words,
"question": question,
"question_gen_seconds": round(t_question - t_pick, 2),
"answer": result.get("answer_text"),
"audit_mode": result.get("audit_mode"),
"verifier_method": result.get("verifier_method"),
"n_quotes": result.get("n_quotes"),
"n_verified": result.get("n_verified"),
"violation_kinds": sorted(
{v.get("kind") for v in (result.get("violations") or []) if v.get("kind")}
),
"metaphor_deflection_kind": metaphor["kind"],
"metaphor_cue_count": metaphor["cue_count"],
"metaphor_overlap_count": metaphor["answer_overlap_count"],
"sources": [
{"title": s.get("title"), "uri": s.get("document_uri"), "used": s.get("used")}
for s in (result.get("sources") or [])
],
"cache_key": result.get("cache_key"),
"answer_mode": answer_mode,
"answer_seconds": round(t_answer - t_question, 2),
"total_seconds": round(t_answer - t_pick, 2),
"student_error": student_error,
# Teacher review fields — left null on generation, populated
# by a separate review pass (fox brings entries to Opus, gets
# back an entry to append to the same log).
"teacher": None,
}
def append_log(entry: dict, log_path: Path) -> None:
log_path.parent.mkdir(parents=True, exist_ok=True)
with log_path.open("a") as f:
f.write(json.dumps(entry, ensure_ascii=False) + "\n")
def emit_unfirehose_session(entry: dict, model_id: str) -> None:
"""Write one unfirehose/1.0 session for this bench cycle.
Same auto-discovery path as `arborist.cli._emit_query_journal`:
~/.arborist/unfirehose/{slug}/{session_uuid}.jsonl. Each cycle gets
its own session file (one prompt → one answer = one session).
Failures must NEVER break the bench loop — wrapped at the call
site.
"""
from arborist.journal import SessionWriter
arborist_meta = {
"audit_mode": entry.get("audit_mode"),
"verifier_method": entry.get("verifier_method"),
"n_quotes": entry.get("n_quotes"),
"n_verified": entry.get("n_verified"),
"cache_key": entry.get("cache_key"),
"violation_kinds": entry.get("violation_kinds"),
"metaphor_deflection_kind": entry.get("metaphor_deflection_kind"),
"answer_mode": entry.get("answer_mode"),
"sources": entry.get("sources"),
"timings": {
"question_gen_seconds": entry.get("question_gen_seconds"),
"answer_seconds": entry.get("answer_seconds"),
"total_seconds": entry.get("total_seconds"),
},
"bench": "emergent",
"words": entry.get("words"),
"student_error": entry.get("student_error"),
}
with SessionWriter(first_prompt=entry.get("question") or "") as s:
s.system_message(
"bench-emergent cycle: 3-word triangulation",
subtype="init",
arborist_meta={"words": entry.get("words"), "harness_role": "generator"},
)
s.user_message(entry.get("question") or "")
s.assistant_message(
entry.get("answer") or "",
model=model_id,
provider="hermes",
duration_ms=int((entry.get("answer_seconds") or 0) * 1000) or None,
arborist_meta=arborist_meta,
)
def print_pending(log_path: Path) -> int:
"""Print every log entry with `teacher: None` — what fox should
bring to a teacher model for review."""
if not log_path.exists():
print(f"no log at {log_path}")
return 0
pending = 0
for line in log_path.read_text().splitlines():
if not line.strip():
continue
try:
entry = json.loads(line)
except json.JSONDecodeError:
continue
if entry.get("teacher") is None:
pending += 1
print(json.dumps(entry, indent=2))
print() # blank line between entries
print(f"\n>> {pending} entry/entries awaiting teacher review", file=sys.stderr)
return pending
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__.strip().splitlines()[0])
parser.add_argument("--n", type=int, default=10, help="cycles to run")
parser.add_argument("--seed", type=int, default=None, help="RNG seed for reproducibility")
parser.add_argument(
"--words-path",
type=Path,
default=DEFAULT_WORDS_PATH,
help="path to a unix-style words file",
)
parser.add_argument("--log-path", type=Path, default=DEFAULT_LOG_PATH)
parser.add_argument(
"--shards-dir",
type=Path,
default=Path.home() / ".arborist" / "shards",
)
parser.add_argument(
"--qa-db",
type=Path,
default=Path.home() / ".arborist" / "shards" / "qa.db",
)
parser.add_argument("--answer-mode", default="claim_lattice")
parser.add_argument("--top-k", type=int, default=8)
parser.add_argument(
"--endpoint",
default=os.environ.get("ARBORIST_LLM_ENDPOINT", "https://hermes.ai.unturf.com/v1"),
)
parser.add_argument(
"--model",
default=os.environ.get(
"ARBORIST_LLM_MODEL",
"adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic",
),
)
parser.add_argument(
"--print-pending",
action="store_true",
help="print log entries with teacher==null & exit",
)
ns = parser.parse_args(argv)
if ns.print_pending:
return 0 if print_pending(ns.log_path) >= 0 else 1
if not ns.words_path.exists():
parser.error(f"words file not found: {ns.words_path}")
if not ns.shards_dir.exists():
parser.error(f"shards dir not found: {ns.shards_dir}")
rng = random.Random(ns.seed)
from arborist.qa.client import OpenAICompatibleClient
api_key = os.environ.get("ARBORIST_LLM_API_KEY")
client = OpenAICompatibleClient(base_url=ns.endpoint, api_key=api_key)
print(f">> emergent bench: {ns.n} cycles · seed={ns.seed} · log={ns.log_path}", flush=True)
for i in range(ns.n):
words = pick_words(3, words_path=ns.words_path, rng=rng)
print(f"[{i+1}/{ns.n}] words={words}", flush=True)
entry = run_one_cycle(
words=words,
client=client,
model_id=ns.model,
qa_db=ns.qa_db,
shards_dir=ns.shards_dir,
answer_mode=ns.answer_mode,
top_k=ns.top_k,
)
append_log(entry, ns.log_path)
# Mirror to unfirehose-compatible journal so the bench cycles
# show up in the unfirehose dashboard alongside Claude Code /
# Fetch sessions. Best-effort — never break the bench on a
# journal write failure.
try:
emit_unfirehose_session(entry, ns.model)
except Exception: # pragma: no cover
pass
# Compact stdout summary so a long sweep is observable.
audit = entry.get("audit_mode") or "?"
ratio = (
f"{entry['n_verified']}/{entry['n_quotes']}"
if entry.get("n_quotes") is not None else "?/?"
)
err = entry.get("student_error")
if err:
print(f" → ERROR: {err}", flush=True)
else:
print(
f"{audit} {ratio} {entry['total_seconds']}s "
f"q={entry['question'][:60]!r}",
flush=True,
)
print(f"\n>> done. log: {ns.log_path}")
print(
f">> review: bring un-reviewed entries to Opus via\n"
f" python scripts/bench_emergent.py --print-pending"
)
return 0
if __name__ == "__main__":
sys.exit(main())