"""JSON-mode claim-lattice verifier (`answer_mode="claim_lattice"`). Pairs with grammar-constrained inference (vLLM guided_json, Claude/GPT-4 native JSON, Qwen 3.6 reasoner). The lenient pre-parser keeps the path survivable on inference paths where the model emits non-strict JSON (markdown fences, prose preamble, curly quotes, trailing commas). """ from __future__ import annotations import json import pytest from arborist.qa.evidence import EvidenceObject from arborist.qa.verify import _lenient_json_parse, verify_claim_lattice_json # ---------------------------------------------------------------- lenient parser def test_lenient_strict_passes_through(): obj, fixups = _lenient_json_parse('{"a": 1}') assert obj == {"a": 1} assert fixups == [] def test_lenient_strips_markdown_fence(): raw = '```json\n{"claims": []}\n```' obj, fixups = _lenient_json_parse(raw) assert obj == {"claims": []} assert "fence" in fixups def test_lenient_strips_unlabeled_fence(): raw = '```\n{"claims": []}\n```' obj, fixups = _lenient_json_parse(raw) assert obj == {"claims": []} assert "fence" in fixups def test_lenient_trims_preamble_and_suffix(): raw = 'Here is the JSON:\n{"claims": []}\nLet me know if you need more.' obj, fixups = _lenient_json_parse(raw) assert obj == {"claims": []} assert "prose_trim" in fixups def test_lenient_normalizes_curly_quotes(): raw = '{“claims”: []}' obj, fixups = _lenient_json_parse(raw) assert obj == {"claims": []} assert "curly_quotes" in fixups def test_lenient_fixes_trailing_comma(): raw = '{"claims": [],}' obj, fixups = _lenient_json_parse(raw) assert obj == {"claims": []} assert "trailing_comma" in fixups def test_lenient_combines_multiple_fixups(): raw = '```json\nHere:\n{“claims”: [],}\n```' obj, fixups = _lenient_json_parse(raw) assert obj == {"claims": []} assert "fence" in fixups # At least one of the inner fixups also fired. assert any(f in fixups for f in ("curly_quotes", "trailing_comma", "prose_trim")) def test_lenient_raises_on_truly_broken(): with pytest.raises(json.JSONDecodeError): _lenient_json_parse("not json at all { ] [") # ---------------------------------------------------------------- JSON verifier def _ev( eid: str, span: str, role: str = "primary_answer_source", pointer_id: str | None = None, title: str = "Test Doc", ) -> EvidenceObject: """Stub evidence object with deterministic eid for the test. JSON-mode prompt-facing surface uses ``pointer_id`` (E1, E2, …) since 2026-04-30; if the test wants the verifier to resolve a citation, it must set ``pointer_id`` explicitly. ``evidence_id`` stays content-addressed for the cache/run-DAG handle. ``title`` defaults to "Test Doc" but tests using Rule 8 (title- relevance, post-2026-05-02) should pass a title whose content tokens overlap the claim text. Otherwise the verifier flags TITLE_MISMATCH and demotes STRICT → HYBRID. """ return EvidenceObject( evidence_id=eid, source_root="00" * 32, document_uri="test://doc", title=title, chunk_idx=0, chunk_root="11" * 32, offset_start=0, offset_end=len(span), source_role=role, text_hash="22" * 32, span=span, pointer_id=pointer_id, ) def test_verify_json_strict_when_all_claims_resolve(): evidence = [ _ev("E1f8e4c2a", "Brachiosaurus appears in the Jurassic Park film as a herbivore.", pointer_id="E1", title="Jurassic Park (film)"), _ev("E2c9d7b3f", "Velociraptor is featured prominently throughout Jurassic Park.", pointer_id="E2", title="Jurassic Park (film)"), ] answer = json.dumps({ "claims": [ {"text": "Brachiosaurus appears in the film", "evidence_ids": ["E1"]}, {"text": "Velociraptor is featured in the film", "evidence_ids": ["E2"]}, ] }) v = verify_claim_lattice_json(answer, evidence) assert v["audit_mode"] == "STRICT" # Same verifier_method as the pointer variant — disambiguated downstream # via `answer_mode` on the run-DAG and `json_fixups` on the verdict. assert v["verifier_method"] == "claim_lattice" assert v["n_verified"] == 2 assert v["violations"] == [] def test_verify_json_hybrid_when_some_unknown_evidence_id(): evidence = [ _ev("E1f8e4c2a", "Brachiosaurus appears in the Jurassic Park film as a herbivore.", pointer_id="E1", title="Jurassic Park (film)"), ] answer = json.dumps({ "claims": [ {"text": "Brachiosaurus appears in the film", "evidence_ids": ["E1"]}, {"text": "Made-up claim", "evidence_ids": ["E99"]}, ] }) v = verify_claim_lattice_json(answer, evidence) assert v["audit_mode"] == "HYBRID" assert any(vio["kind"] == "UNKNOWN_EVIDENCE_ID" for vio in v["violations"]) def test_verify_json_ungrounded_on_schema_invalid(): """Lenient parser fails too → SCHEMA_INVALID → UNGROUNDED.""" v = verify_claim_lattice_json("not json {[", []) assert v["audit_mode"] == "UNGROUNDED" assert any(vio["kind"] == "SCHEMA_INVALID" for vio in v["violations"]) def test_verify_json_recovers_from_markdown_fence(): """JSON-fenced output still parses & verifies; fence fixup logged.""" evidence = [ _ev("E1f8e4c2a", "Brachiosaurus appears in the Jurassic Park film as a herbivore.", pointer_id="E1", title="Jurassic Park (film)"), ] answer = ( '```json\n' '{"claims": [{"text": "Brachiosaurus appears in the film", ' '"evidence_ids": ["E1"]}]}\n' '```' ) v = verify_claim_lattice_json(answer, evidence) assert v["audit_mode"] == "STRICT" assert "fence" in v["json_fixups"] def test_verify_json_manual_quote_violation(): """Strict no-double-quote rule — even valid JSON with double quotes inside a claim's text field fails MANUAL_QUOTE_VIOLATION.""" evidence = [ _ev("E1f8e4c2a", "Brachiosaurus appears in the Jurassic Park film as a herbivore.", pointer_id="E1"), ] answer = json.dumps({ "claims": [ { "text": 'Brachiosaurus is "a herbivore" appears in the film', "evidence_ids": ["E1"], } ] }) v = verify_claim_lattice_json(answer, evidence) assert any(vio["kind"] == "MANUAL_QUOTE_VIOLATION" for vio in v["violations"]) assert v["audit_mode"] == "UNGROUNDED" def test_verify_json_blocks_disallowed_source_role(): """Evidence resolved but source_role outside the allowlist fails SOURCE_ROLE_BLOCKED.""" evidence = [ _ev("E1f8e4c2a", "Brachiosaurus content here.", role="noisy_background_source", pointer_id="E1"), ] answer = json.dumps({ "claims": [ {"text": "Brachiosaurus appears", "evidence_ids": ["E1"]}, ] }) v = verify_claim_lattice_json(answer, evidence) assert any(vio["kind"] == "SOURCE_ROLE_BLOCKED" for vio in v["violations"]) assert v["audit_mode"] == "UNGROUNDED" def test_claim_title_overlap_passes_when_title_shares_token(): """Rule 8 helper: title shares ≥1 stemmed content token with claim.""" from arborist.qa.verify import _claim_title_overlap assert _claim_title_overlap( "Homer Simpson's boss is Mr. Burns.", "Homer Simpson", ) assert _claim_title_overlap( "Brachiosaurus appears in the Jurassic Park film.", "Jurassic Park (film)", ) # Stem-aware: 'simpsons' (plural) collapses to 'simpson'. assert _claim_title_overlap( "The Simpsons family includes Homer.", "Simpson family", ) def test_claim_title_overlap_fails_on_qcd_for_spin_glass(): """Rule 8 helper: spin-glass case from 2026-05-02. Claim about spin glass cited to Quantum chromodynamics → no token overlap.""" from arborist.qa.verify import _claim_title_overlap assert not _claim_title_overlap( "Spin glass modeling involves the use of mathematical tensors.", "Quantum chromodynamics", ) def test_claim_title_overlap_vacuous_pass_on_empty(): """Defensive: empty title or empty claim → vacuous pass.""" from arborist.qa.verify import _claim_title_overlap assert _claim_title_overlap("Some claim here.", None) assert _claim_title_overlap("Some claim here.", "") assert _claim_title_overlap("", "Some Title") def test_verify_json_title_mismatch_demotes_to_ungrounded(): """End-to-end Rule 8: claim cited to a source with no title-token overlap → TITLE_MISMATCH violation. When EVERY resolving claim is title-mismatched (here only 1 claim, 1/1 mismatched), audit_mode demotes to UNGROUNDED — the substrate has zero structural grounding for the user's question. Earlier behavior was HYBRID; tightened 2026-05-02 after the cashback emergent case ('widescreens offer cashback' cited to a generic Coupon article) showed n_verified=1 overclaimed when the citation was meaningless. Span is intentionally rich with claim tokens (covers Rule 5 citation-coverage threshold) so the failure path is Rule 8 cleanly, not earlier rule rejection.""" evidence = [ # Title is QCD; claim is about spin glass modeling. Span # contains enough claim tokens to clear Rule 5 (>=30%). _ev( "Eed1b6e39", ( "Spin glasses are disordered magnetic systems. " "The spin glass modeling literature uses tensors to " "represent interactions between magnetic moments. " "Tensor methods illuminate the mathematical structure." ), pointer_id="E1", title="Quantum chromodynamics", ), ] answer = json.dumps({ "claims": [ { "text": ( "Spin glass modeling involves the use of " "mathematical tensors to represent interactions." ), "evidence_ids": ["E1"], } ] }) v = verify_claim_lattice_json(answer, evidence) assert v["audit_mode"] == "UNGROUNDED", ( f"expected UNGROUNDED via all-claims TITLE_MISMATCH demote; " f"got {v['audit_mode']} with violations " f"{[v['kind'] for v in v['violations']]}" ) assert any(vio["kind"] == "TITLE_MISMATCH" for vio in v["violations"]) def test_verify_json_too_many_claims_demotes_to_hybrid(): """York-england shape: 13 claims (cap default = 12) trips TOO_MANY_CLAIMS even when each individual claim verifies. Demotes STRICT to HYBRID so the runaway is operator-visible.""" # Build 13 evidence objects, each with a pointer id E1..E13 and a # span that contains the claim's full text so per-claim # verification passes. evidence = [] claims = [] for i in range(1, 14): pid = f"E{i}" eid = f"E{i:08x}c2a" text = f"York fact number {i} is described in this span." # Title overlaps "york" so Rule 8 passes; this test exercises # the TOO_MANY_CLAIMS cap, not title-relevance. evidence.append(_ev(eid, text, pointer_id=pid, title="York facts")) claims.append({"text": text, "evidence_ids": [pid]}) answer = json.dumps({"claims": claims}) v = verify_claim_lattice_json(answer, evidence) assert v["audit_mode"] == "HYBRID", \ f"expected HYBRID (TOO_MANY_CLAIMS demote), got {v['audit_mode']}" assert any(vio["kind"] == "TOO_MANY_CLAIMS" for vio in v["violations"]) # The cap doesn't truncate — every claim still verifies. Operator # sees the full evidence of the runaway. assert v["n_verified"] == 13, \ f"all 13 claims should still verify; got {v['n_verified']}" def test_verify_json_at_cap_can_still_strict(): """Boundary: exactly max_claims_per_answer (12) claims is acceptable — no TOO_MANY_CLAIMS violation, and STRICT remains reachable.""" evidence = [] claims = [] for i in range(1, 13): # 12 claims, exactly at cap pid = f"E{i}" eid = f"E{i:08x}c2a" text = f"York fact number {i} appears in span." # Title shares "york" with each claim → Rule 8 passes. evidence.append(_ev(eid, text, pointer_id=pid, title="York facts")) claims.append({"text": text, "evidence_ids": [pid]}) answer = json.dumps({"claims": claims}) v = verify_claim_lattice_json(answer, evidence) assert v["audit_mode"] == "STRICT", \ f"12 claims should not trip the cap; got {v['audit_mode']}" assert not any(vio["kind"] == "TOO_MANY_CLAIMS" for vio in v["violations"]) # ---------------------------------------------------------------- runner integration def test_runner_ask_json_mode_passes_guided_json_extra_body(tmp_path): """`ask()` in JSON mode forwards `extra_body={"guided_json": SCHEMA}` through the chat client. Stub captures the kwargs so we can assert.""" from typing import Iterator from arborist.document import Document from arborist.ingest import ingest_source from arborist.qa import ask from arborist.qa.client import StubClient from arborist.qa.runner import DEFAULT_POLICY from arborist.source import Source from arborist.store import connect class _FakeSource(Source): source_type = "test" def __init__(self, docs): self.docs = docs def iter_documents(self) -> Iterator[Document]: yield from self.docs db = tmp_path / "qa.db" conn = connect(db) try: ingest_source(conn, _FakeSource([ Document( uri="test://doc", content="Cloud Strife is the protagonist. " * 30, source_type="test", title="Cloud", ) ])) root = conn.execute( "SELECT document_root FROM documents WHERE document_uri='test://doc'" ).fetchone()["document_root"] client = StubClient(answer='{"claims":[]}') policy = dict(DEFAULT_POLICY) policy["answer_mode"] = "claim_lattice" ask( conn, document_root=root, question="Who is Cloud?", client=client, model_id="m", policy=policy, ) assert len(client.calls) == 1 kwargs = client.calls[0]["kwargs"] assert "extra_body" in kwargs assert kwargs["extra_body"] is not None assert "guided_json" in kwargs["extra_body"] finally: conn.close() # ---------------------------------------------------------------- Rule 9 # Subject-tokens-absent / premise-parroting check (Ticket #000006 amend # 2026-05-02b). Surfaced by the 200-cycle bench-emergent finding on # `steer/reply/correcter`: claim affirmed three question-distinctive # tokens that appeared zero times in the cited 33.5K-char glossary, # while generic linguistic vocabulary carried Rule 5's coverage check. def test_verify_json_subject_tokens_absent_demotes_strict_to_hybrid(): """Reproduces the steer/reply/correcter false-positive shape: the claim parrots the question's distinctive subject tokens (correcter, steer, reply) but the cited evidence contains only generic linguistic vocabulary (language, communication, terms). Rule 5 passes on the generic overlap; Rule 9 catches that the question-distinctive tokens are absent and demotes.""" cited_span = ( "Sociolinguistics is the study of language in society and how " "social factors influence communication. The exchange of meaning " "between speakers depends on shared terms and discourse " "relationships. Different aspects of language interact with " "communication norms in any given community." ) evidence = [ _ev( "Eparrot01", cited_span, pointer_id="E1", title="Glossary of language teaching terms and ideas", ), ] answer = json.dumps({ "claims": [ { # Three question-distinctive tokens (correcter, steer, # reply) parroted from question into claim — but ZERO # of these tokens appear in cited_span. "text": ( "A correcter can be used to steer a reply by " "identifying errors in language and communication " "between speakers, which involves the exchange of " "meaning across discourse relationships." ), "evidence_ids": ["E1"], } ] }) question = ( "How might a correcter be used to steer a reply in a " "conversation, and what aspects of language or communication " "do these terms encompass?" ) v = verify_claim_lattice_json(answer, evidence, question=question) assert v["audit_mode"] == "HYBRID", ( f"expected HYBRID via SUBJECT_TOKENS_ABSENT demote; got " f"{v['audit_mode']} with violations " f"{[vio['kind'] for vio in v['violations']]}" ) assert any(vio["kind"] == "SUBJECT_TOKENS_ABSENT" for vio in v["violations"]) # The parroted-but-absent tokens should be reported. sta = next(vio for vio in v["violations"] if vio["kind"] == "SUBJECT_TOKENS_ABSENT") absent = set(sta["absent_tokens"]) assert {"correcter", "steer", "reply"}.issubset(absent), ( f"expected correcter/steer/reply in absent_tokens; got {absent}" ) def test_verify_json_subject_tokens_absent_no_question_skips_check(): """No question text → Rule 9 is a no-op. STRICT stays STRICT when every other check passes. Pins that the check requires question text to operate.""" evidence = [ _ev( "Enoq00001", "Brachiosaurus appears in the Jurassic Park film as a herbivore.", pointer_id="E1", title="Jurassic Park (film)", ), ] answer = json.dumps({ "claims": [ {"text": "Brachiosaurus appears in the film", "evidence_ids": ["E1"]}, ] }) v = verify_claim_lattice_json(answer, evidence, question=None) assert v["audit_mode"] == "STRICT" assert not any(vio["kind"] == "SUBJECT_TOKENS_ABSENT" for vio in v["violations"]) def test_verify_json_subject_tokens_absent_below_threshold_passes(): """One absent parroted token → below default threshold of 3 → no demote. Pins the threshold semantics: single-token absence is acceptable noise, three+ is the parrot fingerprint.""" cited_span = ( "Brachiosaurus appears in the Jurassic Park film as a herbivore. " "The dinosaurs in the film were rendered with practical effects " "and CGI by Industrial Light and Magic." ) evidence = [ _ev( "Eonebelow", cited_span, pointer_id="E1", title="Jurassic Park (film)", ), ] # Question token "extinction" doesn't appear in cited; "brachiosaurus" # and "film" do. Only 1 parroted-token absent → below threshold 3. answer = json.dumps({ "claims": [ { "text": ( "Brachiosaurus appears in the film alongside other " "dinosaurs after a long extinction" ), "evidence_ids": ["E1"], } ] }) question = ( "How does Brachiosaurus appear in the Jurassic Park film " "after extinction?" ) v = verify_claim_lattice_json(answer, evidence, question=question) assert v["audit_mode"] == "STRICT", ( f"expected STRICT (below threshold); got {v['audit_mode']} " f"with violations {[vio['kind'] for vio in v['violations']]}" ) # --- multi-engine structured-output extras (claim_lattice JSON mode) ------- def test_claim_lattice_structured_output_extras_default_shape(): """The helper returns extras under all three engine conventions so the same call site enforces the JSON schema across vLLM, llama.cpp, and OpenAI-spec endpoints. Engines silently drop unknown keys, so sending all three is safe.""" from arborist.qa.verify import ( CLAIM_LATTICE_JSON_SCHEMA, claim_lattice_structured_output_extras, ) e = claim_lattice_structured_output_extras() assert set(e.keys()) == {"guided_json", "json_schema", "response_format"} # vLLM convention — the schema directly under `guided_json`. assert e["guided_json"] is CLAIM_LATTICE_JSON_SCHEMA # llama.cpp shorthand — the schema directly under `json_schema`. assert e["json_schema"] is CLAIM_LATTICE_JSON_SCHEMA # OpenAI-spec — wrapped under response_format with the type tag. rf = e["response_format"] assert rf["type"] == "json_schema" inner = rf["json_schema"] assert inner["name"] == "claim_lattice" assert inner["schema"] is CLAIM_LATTICE_JSON_SCHEMA assert inner["strict"] is True def test_claim_lattice_structured_output_extras_accepts_alternate_schema(): """The helper isn't claim-lattice-specific in mechanism — pass a different schema and an alternate name to reuse the multi-engine bridge for any structured-output feature.""" from arborist.qa.verify import claim_lattice_structured_output_extras other = {"type": "object", "properties": {"x": {"type": "string"}}, "required": ["x"], "additionalProperties": False} e = claim_lattice_structured_output_extras(other, name="my_thing") assert e["guided_json"] is other assert e["json_schema"] is other assert e["response_format"]["json_schema"]["name"] == "my_thing" assert e["response_format"]["json_schema"]["schema"] is other def test_query_accepts_extra_body_parameter(): """query() grew an extra_body keyword so per-model knobs (Qwen's enable_thinking, etc.) can flow through to the synthesis call. Schema-enforcement extras are added inside query() and merge with the user-supplied dict (user keys override defaults on collision — common case is disjoint namespaces).""" import inspect from arborist.qa.query import query sig = inspect.signature(query) assert "extra_body" in sig.parameters p = sig.parameters["extra_body"] # keyword-only with default None — accepts any caller, optional. assert p.default is None