arborist/aborist/qa/runner.py
russell@unturf.com 8fb1fe51d7
qa(query): #000007 land — query-layer hyphen folding
Closes the FTS5 hyphen-tokenization asymmetry: `bi-polar is rare?`
retrieved only the Bi-Polar album/disambiguation cluster while the
medical-condition cluster (Bipolar disorder, Bipolar I/II disorder,
etc.) sat in the same shards untouched. `unicode61` splits hyphens
at index AND query time; `Bi-Polar Blues` indexes as [bi, polar, ...]
while `Bipolar disorder` indexes as [bipolar] — non-overlapping
token sets that never met.

Fix is query-layer only — no canonicalization_version bump, no
re-index, existing cache_keys stay valid:

- _hyphen_fold_variants(s): emit joined-no-hyphen variants for
  every hyphenated run.
- _title_query_tokens(s): additively merges variants symmetrically
  (queries AND titles when called on either).
- _filter_by_title_relevance: accept-path 5 — title stem-overlap
  with hyphen-fold anchors passes the breadth gate. Rescues
  `Bipolar disorder` (1-of-N qtoken match) without disrupting
  non-hyphen queries (anchors empty → zero side effect).
- DEFAULT_QUERY_POLICY / DEFAULT_POLICY: hyphen_fold_v1: True
  marker folds into governance_policy_hash; new records
  cache-split cleanly from pre-fold records.

Live verification on /home/fox/.aborist/shards: same query now
retrieves `Bipolar disorder` (#5) and `Bipolar` disambiguation
(#7); model cites both, answer reads "Bi-polar disorder is not
rare; it affects approximately 2.8% of the U.S. population".
EVIDENCE-WARRANTED 2/2, properly grounded.

Tests: 4 new (3 unit, 1 integration with regression-pinned
Bipolar-disorder retrieval). Full suite 760 passed, 34 skipped.

Also: CLAUDE.md gains a close-when-complete hint for tickets — an
open ticket whose code already shipped is a stale map.
2026-05-02 14:12:46 -04:00

812 lines
34 KiB
Python

"""Q&A runner: cache-first lookup -> inference fallback -> provable record.
Implements the v9.8 admissibility invariant:
No record reused unless all 8 cache_key dimensions match AND state
is 'live' (not failed/stale/quarantined).
Cache hit -> persisted audit_mode (STRICT/HYBRID/UNGROUNDED).
Cache miss -> call ChatClient, run faithfulness check, classify, store
record, audit event.
"""
from __future__ import annotations
import json
import sqlite3
import time
from aborist import (
CANONICALIZATION_VERSION,
SCHEMA_VERSION,
)
from aborist.compress import unpack_chunk
from aborist.merkle import MerkleTree, proof_to_dict
from aborist.qa.client import ChatClient
from aborist.qa.prompts import (
CLAIM_LATTICE_GROUNDING_REMINDER,
CLAIM_LATTICE_JSON_GROUNDING_REMINDER,
CLAIM_LATTICE_JSON_SYSTEM_PROMPT,
CLAIM_LATTICE_SYSTEM_PROMPT,
)
from aborist.qa.keys import (
DEFAULT_FIDELITY,
DEFAULT_QUESTION_DEDUP,
FIDELITY_MODES,
QUESTION_DEDUP_MODES,
cache_key,
canonical_question,
conversation_hash,
governance_policy_hash,
model_profile_hash,
question_hash,
verifier_policy_hash,
)
from aborist.qa.dag import build_run_dag
from aborist.qa.evidence import (
build_evidence_map,
evidence_map_root,
render_evidence_map,
render_evidence_map_for_json,
)
from aborist.qa.repair import mechanical_repair, reprompt_repair
from aborist.qa.verify import (
ANSWER_MODES,
CLAIM_LATTICE_JSON_SCHEMA,
DEFAULT_ANSWER_MODE,
verify_claim_lattice,
verify_claim_lattice_json,
verify_quotes,
)
from aborist.store import append_audit, transaction
try:
from aborist.wikitext import BASE_VERSION as _WIKITEXT_BASE_VERSION
from aborist.wikitext import to_base as _wikitext_to_base
except ImportError: # pragma: no cover
_WIKITEXT_BASE_VERSION = None
_wikitext_to_base = None
DEFAULT_POLICY = {
# Ticket #000007 — query-layer hyphen-fold marker. See
# aborist/qa/query.py:DEFAULT_QUERY_POLICY for rationale.
"hyphen_fold_v1": True,
"system_prompt": (
"Answer the user's question based ONLY on the document below. "
"For EVERY factual claim, include a verbatim quote from the "
"document enclosed in double quotes (\"...\"). The quoted span "
"must appear word-for-word. Make a claim only when a verbatim "
"quote in the document directly supports it. "
"If the answer is in the document, write it. "
"If the answer is absent from the document, say 'I don't know "
"based on the provided document.' and stop there. "
"Stay inside the document at all times."
),
# Restated rule fired as a user message right before the document +
# question arrive. See aborist/qa/query.py for the rationale (recent
# user-turn instructions outweigh decayed system-turn rules in 8B
# instruction-tuned models).
"grounding_reminder": (
"REMINDER: wrap every factual claim in double quotes (\"...\") "
"and the quoted span must appear word-for-word in the "
"document. Each claim earns a verbatim quote. "
"Now answer the question on the next message."
),
"temperature": 0.1,
"top_p": 1.0,
"max_tokens": 512,
"entity_policy": "proximity",
"entity_proximity_n": 3,
"entity_proximity_window": 300,
# Mechanical answer repair after first verify. Off by default; see
# aborist/qa/query.py for semantics.
"repair_enabled": False,
"repair_max_reprompts": 0,
# Strip wikitext markup before the LLM ever sees the context. Lets
# Hermes quote prose verbatim and shrinks token bills (~43% on
# Wikipedia chunks). Bumps governance_policy_hash so prior cached
# answers under raw-wikitext policy stay distinct on lookup. Set
# via the wikitext extras; no-op if mwparserfromhell isn't installed.
"base_version": _WIKITEXT_BASE_VERSION,
# G0 / CTI — claim-lattice-pointer answer mode. "quote" (default):
# existing behavior, model writes prose with verbatim quotes inline.
# "claim_lattice_pointer": runtime builds an evidence map and shows
# the model short pointer ids (E1, E2, …); model writes natural
# prose with bracket pointer tags ("Claim. [E12]") instead of
# quoting source text. Renderer interpolates literal spans at
# display time. Synthetic-elision-by-construction-impossible: the
# model never types the quote string. Two-layer id discipline keeps
# the cache & run-DAG keyed on content-addressed evidence_ids.
# Folds into governance_policy_hash so two modes write under
# different cache_keys and never alias. No iterative repair in
# pointer mode (one-shot benchmark discipline).
"answer_mode": DEFAULT_ANSWER_MODE,
"claim_lattice_system_prompt": CLAIM_LATTICE_SYSTEM_PROMPT,
"claim_lattice_grounding_reminder": CLAIM_LATTICE_GROUNDING_REMINDER,
# Allowed source roles for claim-lattice verification. Roles outside
# this set get classified SOURCE_ROLE_BLOCKED and downgrade the
# verdict. Mirrors aborist.qa.verify.DEFAULT_ALLOWED_SOURCE_ROLES;
# noisy_background_source / sequel_background_source are excluded by
# default. Folds into governance_policy_hash on change.
"claim_lattice_allowed_source_roles": [
"primary_answer_source",
"secondary_context_source",
"background_source",
"unclassified",
# Self-promoted providence records (`aborist://providence/`
# URI scheme). Trusted-as-fact substrate per the
# self-reference design — STRICT live records past the
# kindergarten window. See
# docs/self-reference-design.md for the
# falsification trust model.
"self_reference_source",
],
# Hard cap on pointer ids per claim line — mirrors prompt Rule 9.
# Lines exceeding this cap classify as SCHEMA_INVALID and the
# verdict can no longer reach STRICT. Folds into
# governance_policy_hash so changing the cap invalidates prior
# cached records.
"claim_lattice_max_pointers_per_claim": 2,
# Minimum claim-token coverage required for the citation-overlap
# check (Rule 6) to pass. Pre-2026-04-30 the threshold was implicit
# at "≥1 shared token", which let through lazy-anchored claims
# whose only overlap was a single topical word (e.g. "Yale
# University... [E9]" cited to a highway-data span containing only
# "Connecticut"). 0.30 means a 10-token claim needs ≥3 of its
# content tokens to appear in the cited span. Short claims (≤3
# content tokens) keep the old ≥1-token floor so narrow factoids
# like "Steve Jobs co-founded Apple" still pass. Folds into
# governance_policy_hash on change.
"claim_lattice_min_citation_coverage": 0.30,
# Bare-name claim guard. A claim with fewer than this many content
# tokens (>=4 chars, post-spotlight-stopword) is rejected as
# SCHEMA_INVALID. Catches the JP-dinosaurs lazy-anchor where
# "Triceratops. [E16]" passes lexical overlap on a single token
# even when E16 is a video-game tie-in chunk rather than the film
# article. Default 2: bare-entity-name claims (one content token
# after stopword strip) fail; sentence-shape claims pass. Note
# ``_content_tokens`` already filters "appears", "shown", etc. so
# "Trex appears" → 1 content token (filtered), "Trex appears in
# the film" → 2 content tokens (passes). Folds into
# governance_policy_hash on change.
"claim_lattice_min_claim_content_tokens": 2,
# Lazy-anchor smell auto-demote. When >= threshold of verified
# pointer-pairs cite a single pointer AND there are >= min_pairs
# total, cap audit_mode at HYBRID. The smell sidecar was advisory
# only pre-2026-04-30; now it's load-bearing. STRICT requires
# diverse anchoring across pointers.
"claim_lattice_lazy_anchor_demote_threshold": 0.5,
"claim_lattice_lazy_anchor_demote_min_pairs": 3,
# Warrant-lite — relation-question hard check (Ticket H from
# feedback-3, 2026-05-01). Detects relation-shape questions
# ("who is X's boss?", "who founded Y?") and requires the cited
# span to contain at least one named answer entity (proper-noun
# phrase) from the claim. Catches the Homer-Simpson lazy-anchor
# case where claim asserts "Mr. Burns" but cited span is the
# voice-actor bio. WARRANT_MISSING violations cap audit_mode
# at HYBRID. See aborist/qa/warrant.py.
"claim_lattice_warrant_check_enabled": True,
"claim_lattice_deflection_check_enabled": True,
# Claim-count ceiling. Bench finding (2026-04-30 york-england):
# "tell me all there is to know about X" prompted Hermes to spam
# 26-59 encyclopedic claims sourced from training, only 2-4 of
# which grounded in retrieval. Atomic-claim prompt rule (b5925c8)
# cut this to ~10, but a hard structural cap is defense in depth.
# Cap of 12 admits typical entity-list questions (5-7 dinosaurs,
# Simpsons + pets) while flagging the runaway shape. Folds into
# governance_policy_hash on change.
"claim_lattice_max_claims_per_answer": 12,
# JSON variant — `answer_mode="claim_lattice"`. Mirrors the
# multi-source query path. Pairs with grammar-constrained inference
# (vLLM guided_json, Claude/GPT-4 native JSON, Qwen 3.6 reasoner).
# Lenient pre-parser in verify_claim_lattice_json keeps the path
# survivable on inference paths without grammar guidance.
"claim_lattice_json_system_prompt": CLAIM_LATTICE_JSON_SYSTEM_PROMPT,
"claim_lattice_json_grounding_reminder": CLAIM_LATTICE_JSON_GROUNDING_REMINDER,
"claim_lattice_use_guided_json": True,
# JSON-mode stop sequences. Hermes-3-8B sometimes spams whitespace
# / newlines after the closing brace on broad-descriptive shapes
# ("plot of X", "tell me about Y") — the response runs out the
# max_tokens budget and the lenient parser sees truncated JSON.
# Stopping on a blank line cuts the runaway. JSON-mode output
# never legitimately contains a blank line (single object, single
# line) so this is a safe filter. Folds into
# governance_policy_hash on change.
"claim_lattice_json_stop_sequences": ["\n\n"],
}
def _ms_since(t: float) -> float:
return round((time.monotonic() - t) * 1000, 1)
def ask(
conn: sqlite3.Connection,
*,
document_root: str,
question: str,
client: ChatClient,
model_id: str,
revision: str = "",
quantization: str = "",
policy: dict | None = None,
chain: str = "private",
fidelity: str | None = None,
) -> dict:
"""Look up cached answer or run inference. Returns a result dict.
See ``aborist.qa.query.query`` for `fidelity` semantics — it
controls lookup tolerance: ``"strict"`` only checks the cache_key
matching the call's ``policy["question_dedup"]``; the default
``"equivalence_class"`` falls back to the alternate dedup mode's
cache_key on miss so a fast-cache agent can reuse records written
under either mode. Result includes ``lookup_path``.
"""
policy = policy or DEFAULT_POLICY
if fidelity is None:
fidelity = policy.get("fidelity", DEFAULT_FIDELITY)
if fidelity not in FIDELITY_MODES:
raise ValueError(
f"fidelity must be one of {FIDELITY_MODES}, got {fidelity!r}"
)
t_start = time.monotonic()
doc = conn.execute(
"SELECT document_uri, chunking_version FROM documents "
"WHERE document_root = ?",
(document_root,),
).fetchone()
if doc is None:
return {"status": "unknown_document"}
chunk_rows = conn.execute(
"SELECT idx, leaf_hash, content FROM chunks "
"WHERE document_root = ? ORDER BY idx ASC",
(document_root,),
).fetchall()
if not chunk_rows:
return {"status": "unknown_document"}
if any(r["content"] is None for r in chunk_rows):
return {"status": "source_cold", "msg": "rehydrate before asking"}
answer_mode = policy.get("answer_mode", DEFAULT_ANSWER_MODE)
if answer_mode not in ANSWER_MODES:
raise ValueError(
f"policy['answer_mode'] must be one of {ANSWER_MODES}, got {answer_mode!r}"
)
chunk_texts = [unpack_chunk(r["content"]) for r in chunk_rows]
document_text = "\n\n".join(chunk_texts)
# Wikitext → prose before the LLM sees it. The model can then quote
# verbatim against the prose form; the verifier compares like-against-
# like. Idempotent if context is already plain prose. Gated on
# policy["base_version"] so this is part of governance_policy_hash.
if policy.get("base_version") and _wikitext_to_base is not None:
document_text = _wikitext_to_base(document_text)
chunk_texts = [_wikitext_to_base(t) for t in chunk_texts]
evidence_map = []
if answer_mode == "claim_lattice_pointer":
# Quote-by-pointer: one evidence object per chunk. The model sees
# the literal spans labeled with content-addressed IDs and is
# instructed to reference IDs, not type quote text. Synthetic
# elision is impossible by construction — the model never produces
# the quote string.
chunks_for_map = [
{
"source_root": document_root,
"document_uri": doc["document_uri"],
"title": None,
"chunk_idx": r["idx"],
"chunk_root": r["leaf_hash"],
"span": chunk_texts[i],
"source_role": "primary_answer_source",
}
for i, r in enumerate(chunk_rows)
]
evidence_map = build_evidence_map(chunks_for_map)
sys_prompt = policy["claim_lattice_system_prompt"]
grounding_reminder = policy.get("claim_lattice_grounding_reminder")
rendered_evidence = render_evidence_map(evidence_map)
def _user_payload(q: str) -> str:
return f"EVIDENCE:\n\n{rendered_evidence}\n\n---\n\nQUESTION: {q}"
elif answer_mode == "claim_lattice":
# JSON variant — same per-chunk evidence map as pointer mode,
# blocks labeled with content-addressed evidence_id (long hex)
# since the model emits IDs in JSON. Pairs with grammar-
# constrained inference; lenient pre-parser handles drift.
chunks_for_map = [
{
"source_root": document_root,
"document_uri": doc["document_uri"],
"title": None,
"chunk_idx": r["idx"],
"chunk_root": r["leaf_hash"],
"span": chunk_texts[i],
"source_role": "primary_answer_source",
}
for i, r in enumerate(chunk_rows)
]
evidence_map = build_evidence_map(chunks_for_map)
sys_prompt = policy.get(
"claim_lattice_json_system_prompt",
policy["claim_lattice_system_prompt"],
)
grounding_reminder = policy.get(
"claim_lattice_json_grounding_reminder",
policy.get("claim_lattice_grounding_reminder"),
)
rendered_evidence = render_evidence_map_for_json(evidence_map)
def _user_payload(q: str) -> str:
return f"EVIDENCE:\n\n{rendered_evidence}\n\n---\n\nQUESTION: {q}"
else:
sys_prompt = policy["system_prompt"]
grounding_reminder = policy.get("grounding_reminder")
def _user_payload(q: str) -> str:
return f"Document:\n\n{document_text}\n\n---\n\nQuestion: {q}"
# System sets the policy; a user-turn reminder restates the rule one
# message before the payload arrives. Payload (document or evidence
# map + question) lands last as the most-recent tokens before
# generation.
messages = [{"role": "system", "content": sys_prompt}]
if grounding_reminder:
messages.append({"role": "user", "content": grounding_reminder})
messages.append({"role": "user", "content": _user_payload(question)})
mhash = model_profile_hash(model_id, revision, quantization)
# Dedup-mode-aware cache_key. See aborist/qa/query.py for rationale —
# policy_variant matches the alternate mode so governance_policy_hash
# agrees with what an agent under that mode would have written,
# enabling cross-silo fallback.
def _ckey_for_mode(mode: str) -> str:
canon_q = canonical_question(question, mode=mode)
canon_msgs = list(messages[:-1]) + [
{"role": "user", "content": _user_payload(canon_q)},
]
policy_variant = dict(policy, question_dedup=mode)
return cache_key(
document_root,
question_hash(question, mode=mode),
mhash,
conversation_hash(canon_msgs),
governance_policy_hash(policy_variant),
SCHEMA_VERSION,
CANONICALIZATION_VERSION,
doc["chunking_version"],
verifier_policy_hash(policy_variant),
)
ghash = governance_policy_hash(policy) # for the legacy INSERT below
primary_dedup = policy.get("question_dedup", DEFAULT_QUESTION_DEDUP)
if primary_dedup not in QUESTION_DEDUP_MODES:
raise ValueError(
f"policy['question_dedup'] must be one of {QUESTION_DEDUP_MODES}, "
f"got {primary_dedup!r}"
)
# Re-derive the per-mode hashes for use in the INSERT below. _ckey_for_mode
# already builds them, but the legacy INSERT references qhash/chash by name.
qhash = question_hash(question, mode=primary_dedup)
canonical_q_primary = canonical_question(question, mode=primary_dedup)
canonical_messages_primary = list(messages[:-1]) + [
{"role": "user", "content": _user_payload(canonical_q_primary)},
]
chash = conversation_hash(canonical_messages_primary)
primary_ckey = _ckey_for_mode(primary_dedup)
ckey = primary_ckey # legacy name for the rest of the function
t_lookup = time.monotonic()
cached = conn.execute(
"SELECT * FROM providence_cache "
"WHERE cache_key = ? AND falsification_state = 'live'",
(primary_ckey,),
).fetchone()
hit_ckey = primary_ckey
lookup_path = primary_dedup if cached is not None else None
if cached is None and fidelity == "equivalence_class":
other_mode = (
"equivalence_class" if primary_dedup == "strict" else "strict"
)
other_ckey = _ckey_for_mode(other_mode)
if other_ckey != primary_ckey:
cached = conn.execute(
"SELECT * FROM providence_cache "
"WHERE cache_key = ? AND falsification_state = 'live'",
(other_ckey,),
).fetchone()
if cached is not None:
hit_ckey = other_ckey
lookup_path = f"{other_mode}_fallback"
cache_lookup_ms = _ms_since(t_lookup)
if cached is not None:
with transaction(conn):
now = int(time.time())
conn.execute(
"UPDATE providence_cache "
"SET hit_count = hit_count + 1, last_hit_at = ? "
"WHERE cache_key = ?",
(now, hit_ckey),
)
return {
"status": "cache_hit",
"audit_mode": cached["audit_mode"],
"cache_key": hit_ckey,
"lookup_path": lookup_path,
"source_root": document_root,
"answer_text": cached["answer_text"],
"merkle_proof": json.loads(cached["merkle_proof"]),
"n_quotes": cached["n_quotes"],
"n_verified": cached["n_verified"],
"verifier_method": cached["verifier_method"],
"unverified_quotes": (
json.loads(cached["unverified_quotes"])
if cached["unverified_quotes"]
else []
),
"partially_verified_quotes": [],
"timings": {
"cache_lookup_ms": cache_lookup_ms,
"llm_ms": None,
"total_ms": _ms_since(t_start),
},
}
t_llm = time.monotonic()
# JSON mode: pass guided_json schema so vLLM constrains output at
# sampling time. Endpoints without guided-decoding silently drop the
# field; the lenient pre-parser handles whatever drift remains.
extra_body: dict | None = None
stop_seqs: list[str] | None = None
if answer_mode == "claim_lattice" and policy.get(
"claim_lattice_use_guided_json", True
):
extra_body = {"guided_json": CLAIM_LATTICE_JSON_SCHEMA}
if answer_mode == "claim_lattice":
# JSON-mode token-runaway guard. On broad-descriptive /
# comparison questions Hermes-3-8B sometimes spams whitespace
# / newlines after the closing brace until max_tokens
# exhausts; the resulting truncated payload won't parse and
# the run lands UNGROUNDED 0/0 at 12-15s instead of 2-4s.
# Stopping on a blank line (\n\n) cuts the runaway —
# well-formed JSON-mode output never contains a blank line
# since the model emits a single object on one line (or
# with simple internal newlines).
stop_seqs = list(policy.get(
"claim_lattice_json_stop_sequences", ["\n\n"]
))
raw_answer = client.chat_completion(
messages,
model=model_id,
temperature=policy["temperature"],
max_tokens=policy["max_tokens"],
top_p=policy.get("top_p", 1.0),
extra_body=extra_body,
stop=stop_seqs,
)
llm_ms = _ms_since(t_llm)
repair_changes: list[dict] = []
pre_repair_verdict: dict | None = None
if answer_mode == "claim_lattice_pointer":
verdict = verify_claim_lattice(
raw_answer,
evidence_map,
allowed_source_roles=tuple(
policy.get(
"claim_lattice_allowed_source_roles",
[
"primary_answer_source",
"secondary_context_source",
"background_source",
"unclassified",
],
)
),
max_pointers_per_claim=int(policy.get(
"claim_lattice_max_pointers_per_claim", 2
)),
min_citation_coverage=float(policy.get(
"claim_lattice_min_citation_coverage", 0.30
)),
min_claim_content_tokens=int(policy.get(
"claim_lattice_min_claim_content_tokens", 3
)),
lazy_anchor_demote_threshold=float(policy.get(
"claim_lattice_lazy_anchor_demote_threshold", 0.5
)),
lazy_anchor_demote_min_pairs=int(policy.get(
"claim_lattice_lazy_anchor_demote_min_pairs", 3
)),
max_claims_per_answer=int(policy.get(
"claim_lattice_max_claims_per_answer", 12
)),
question=question,
warrant_check_enabled=bool(policy.get(
"claim_lattice_warrant_check_enabled", True
)),
deflection_check_enabled=bool(policy.get(
"claim_lattice_deflection_check_enabled", True
)),
)
# Rendered prose (literal spans interpolated) is the user-facing
# answer text — never the model's raw pointer-line output. If
# rendering produced nothing (no valid claims), persist the raw
# output so an operator can see what the model actually said.
rendered = verdict["rendered_text"]
answer_text = rendered if rendered else raw_answer
elif answer_mode == "claim_lattice":
verdict = verify_claim_lattice_json(
raw_answer,
evidence_map,
allowed_source_roles=tuple(
policy.get(
"claim_lattice_allowed_source_roles",
[
"primary_answer_source",
"secondary_context_source",
"background_source",
"unclassified",
],
)
),
max_evidence_per_claim=int(policy.get(
"claim_lattice_max_pointers_per_claim", 2
)),
min_citation_coverage=float(policy.get(
"claim_lattice_min_citation_coverage", 0.30
)),
max_claims_per_answer=int(policy.get(
"claim_lattice_max_claims_per_answer", 12
)),
question=question,
warrant_check_enabled=bool(policy.get(
"claim_lattice_warrant_check_enabled", True
)),
deflection_check_enabled=bool(policy.get(
"claim_lattice_deflection_check_enabled", True
)),
)
rendered = verdict["rendered_text"]
answer_text = rendered if rendered else raw_answer
else:
answer_text = raw_answer
verdict = verify_quotes(
answer_text,
document_text,
entity_policy=policy.get("entity_policy", "hybrid"),
proximity_n=policy.get("entity_proximity_n", 3),
proximity_window=policy.get("entity_proximity_window", 300),
)
def _verify(text: str) -> dict:
return verify_quotes(
text,
document_text,
entity_policy=policy.get("entity_policy", "hybrid"),
proximity_n=policy.get("entity_proximity_n", 3),
proximity_window=policy.get("entity_proximity_window", 300),
)
if (
policy.get("repair_enabled")
and verdict["audit_mode"] != "STRICT"
and verdict.get("unverified_quotes")
):
repair_result = mechanical_repair(
answer_text, verdict["unverified_quotes"], document_text
)
if repair_result["changes"]:
new_verdict = _verify(repair_result["repaired_text"])
if new_verdict["n_verified"] >= verdict["n_verified"]:
pre_repair_verdict = verdict
answer_text = repair_result["repaired_text"]
verdict = new_verdict
repair_changes = list(repair_result["changes"])
max_reprompts = int(policy.get("repair_max_reprompts", 0))
for _ in range(max_reprompts):
if (
verdict["audit_mode"] == "STRICT"
or not verdict.get("unverified_quotes")
):
break
new_text = reprompt_repair(
chat_client=client,
model_id=model_id,
original_messages=messages,
original_answer=answer_text,
failed_quotes=verdict["unverified_quotes"],
policy=policy,
)
if not new_text:
break
new_verdict = _verify(new_text)
if new_verdict["n_verified"] > verdict["n_verified"]:
if pre_repair_verdict is None:
pre_repair_verdict = verdict
answer_text = new_text
verdict = new_verdict
repair_changes.append({
"action": "reprompt_rewrite",
"diagnosis": "model_feedback_loop",
})
else:
break
unverified_blob = (
json.dumps(verdict["unverified_quotes"], separators=(",", ":"))
if verdict["unverified_quotes"]
else None
)
leaves = [bytes.fromhex(r["leaf_hash"]) for r in chunk_rows]
tree = MerkleTree.build(leaves)
proof_obj = {
"document_root": document_root,
"chunk_0_proof": proof_to_dict(tree.proof(0)),
}
proof_blob = json.dumps(proof_obj, separators=(",", ":"))
# Per-run Merkle-DAG (see aborist/qa/dag.py). Single-doc shape: the
# only "source" is document_root. Pointer mode swaps the 7-stage
# quote shape for the 9-stage CTI shape — context drops out and
# answer splits into raw_answer / parsed_claim_lattice / render.
ev_root = evidence_map_root(evidence_map) if evidence_map else None
parsed_lattice = None
is_lattice_mode = answer_mode in ("claim_lattice_pointer", "claim_lattice")
if is_lattice_mode:
# Per-claim list of {claim_text, content-addressed evidence_ids}
# for the parsed_claim_lattice node hash. Pointer ids are
# run-dependent; evidence_ids are content-addressed → the run-
# DAG hashes the run-stable form. Same shape for JSON and
# pointer; verifier already returns evidence_id_pairs.
evidence_id_pairs = verdict.get("evidence_id_pairs") or []
parsed_lattice = [
{
"claim_text": cs.get("text", ""),
"evidence_ids": evidence_id_pairs[i] if i < len(evidence_id_pairs) else [],
}
for i, cs in enumerate(verdict.get("claim_statuses") or [])
]
run_dag = build_run_dag(
question_hash=qhash,
sources=[{
"document_root": document_root,
"source_role": "primary_answer_source",
"score": None,
"chunk_idx": None,
}],
context_root=document_root,
conversation_hash=chash,
answer_text=answer_text,
audit_mode=verdict["audit_mode"],
verifier_method=verdict["verifier_method"],
n_quotes=verdict["n_quotes"],
n_verified=verdict["n_verified"],
claim_statuses=verdict.get("claim_statuses", []),
lookup_path="miss",
evidence_map_root=ev_root,
answer_mode=answer_mode if answer_mode != "quote" else None,
violations=verdict.get("violations"),
raw_answer_text=raw_answer if is_lattice_mode else None,
parsed_lattice=parsed_lattice,
rendered_text=answer_text if is_lattice_mode else None,
)
run_dag_blob = json.dumps(run_dag, separators=(",", ":"))
now = int(time.time())
with transaction(conn):
if repair_changes and pre_repair_verdict is not None:
append_audit(
conn,
event_type="providence_repair",
subject_root=ckey,
body={
"kind": "mechanical",
"n_changes": len(repair_changes),
"changes": repair_changes,
"pre_audit_mode": pre_repair_verdict["audit_mode"],
"post_audit_mode": verdict["audit_mode"],
"pre_n_verified": pre_repair_verdict["n_verified"],
"post_n_verified": verdict["n_verified"],
},
ts=now,
)
event_hash = append_audit(
conn,
event_type="providence_write",
subject_root=ckey,
body={
"source_root": document_root,
"model_id": model_id,
"revision": revision,
"quantization": quantization,
"chunks_in_context": len(chunk_rows),
"answer_chars": len(answer_text),
"audit_mode": verdict["audit_mode"],
"n_quotes": verdict["n_quotes"],
"n_verified": verdict["n_verified"],
"verifier_method": verdict["verifier_method"],
},
ts=now,
)
conn.execute(
"INSERT INTO providence_cache "
"(cache_key, source_root, document_uri, question_hash, question_text, "
" answer_text, merkle_proof, model_profile_hash, conversation_hash, "
" governance_policy_hash, schema_version, canonicalization_version, "
" chunking_version, falsification_state, chain, audit_event_hash, "
" created_at, hit_count, audit_mode, n_quotes, n_verified, "
" unverified_quotes, verifier_method, run_dag_root, run_dag_blob) "
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 'live', ?, ?, ?, 0, "
" ?, ?, ?, ?, ?, ?, ?)",
(
ckey,
document_root,
doc["document_uri"],
qhash,
question,
answer_text,
proof_blob,
mhash,
chash,
ghash,
SCHEMA_VERSION,
CANONICALIZATION_VERSION,
doc["chunking_version"],
chain,
event_hash,
now,
verdict["audit_mode"],
verdict["n_quotes"],
verdict["n_verified"],
unverified_blob,
verdict["verifier_method"],
run_dag["root"],
run_dag_blob,
),
)
from aborist.qa.dag import localize_failure as _localize
failure_stage = _localize(
audit_mode=verdict["audit_mode"],
n_sources=1, # ask() runs against one document
n_quotes=verdict["n_quotes"],
n_verified=verdict["n_verified"],
)
return {
"status": "cache_miss_then_written",
"audit_mode": verdict["audit_mode"],
"cache_key": ckey,
"run_dag_root": run_dag["root"],
"lookup_path": "miss",
"failure_stage": failure_stage,
"repair_changes": repair_changes,
"pre_repair_audit_mode": (
pre_repair_verdict["audit_mode"] if pre_repair_verdict else None
),
"source_root": document_root,
"answer_text": answer_text,
"merkle_proof": proof_obj,
"n_quotes": verdict["n_quotes"],
"n_verified": verdict["n_verified"],
"verifier_method": verdict["verifier_method"],
"unverified_quotes": verdict["unverified_quotes"],
"partially_verified_quotes": verdict.get("partially_verified_quotes") or [],
# Sidecar smell signals (claim_lattice mode only) — render-
# layer; never persisted, never in run_dag_root.
"pointer_id_distribution": verdict.get("pointer_id_distribution"),
"lazy_anchor_ratio": verdict.get("lazy_anchor_ratio"),
"timings": {
"cache_lookup_ms": cache_lookup_ms,
"llm_ms": llm_ms,
"total_ms": _ms_since(t_start),
},
}