From 5a60e8595fb97648923b3fd4dcbb1ed1846daca7 Mon Sep 17 00:00:00 2001 From: "russell@unturf.com" Date: Sun, 3 May 2026 07:41:13 -0400 Subject: [PATCH] =?UTF-8?q?qa(#000008):=20Phase=204=20=E2=80=94=20CLI=20fl?= =?UTF-8?q?ags=20+=20violation=20tails=20+=20reject-broad?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CLI flags on `aborist query`: --no-quantifier-guard Level 2 disable: kills the guard for one call. Telemetry → None. --allow-broad Emergent-search: classifier on, caps off. For exploratory enumeration. --reject-broad Strict reject: ALL/COMPREHENSIVE/ OPEN_REQUEST + scope_bound_hint== "unbounded" returns UNGROUNDED before the LLM call (saves ~10-15s). Bounded universals (Beatles, year-anchored) are NOT rejected per §10.1. --apply-quantifier-caps Flip Phase 2 dry-run gate per-call. Path from dry-run to live cap. Three new soft-demote violation kinds (§10.3) — no new audit_mode token; tails on the existing audit-line: BROAD_QUANTIFIER_RUNAWAY "broad runaway" BROAD_QUANTIFIER_CAP_APPLIED "broad cap N" (cap value rendered) BROAD_QUANTIFIER_SCOPE_UNBOUND "broad unbounded" All three cap the ladder at ANCHOR-WARRANTED. Plus one HARD demote (early-return UNGROUNDED): BROAD_QUANTIFIER_REJECTED "broad rejected" (preflight rejection) The reject-broad path early-returns from query() before the LLM call when policy enables quantifier_reject_broad AND the question is broad-unbounded. Result schema mirrors a normal UNGROUNDED row (answer_text carries the rejection rationale + actionable narrowing hints). _render_query_human gets a dedicated branch for the new status so operators see the rejection without --json. Live verification (post-commit): $ aborist query --reject-broad "Winners of all major sports?" UNGROUNDED · via BROAD_QUANTIFIER_REJECTED · ALL ("all") · cap was 8 0/0 0.0s (preflight) BROAD-QUANTIFIER PREFLIGHT REJECTED · scope unbounded Question matched ALL intensity ("all") with an under-specified universe. Narrow ... or run with --allow-broad for exploratory enumeration. $ aborist query --reject-broad "name all members of the Beatles" UNGROUNDED · via claim_lattice · title mismatch 4/4 20.9s [Beatles enumerated, scope_bound_hint=bounded → not rejected] `quantifier_reject_broad` folded into _VERIFIER_POLICY_FIELDS so flipping reject default invalidates prior cache records. 16 new tests cover: soft-demote registration, hard-demote NOT in soft-demote set, ladder rung mapping for each kind, tail rendering (including cap value interpolation), tail combination with existing kinds, end-to-end render through _render_query_human, governance-hash binding. Two skipped placeholders mark the integration paths exercised by live bench. --- aborist/cli.py | 112 ++++++++++++++++++ aborist/qa/keys.py | 1 + aborist/qa/query.py | 90 +++++++++++++++ aborist/qa/runner.py | 9 ++ tests/test_quantifier_phase4.py | 197 ++++++++++++++++++++++++++++++++ 5 files changed, 409 insertions(+) create mode 100644 tests/test_quantifier_phase4.py diff --git a/aborist/cli.py b/aborist/cli.py index 5772516..8b15799 100644 --- a/aborist/cli.py +++ b/aborist/cli.py @@ -396,6 +396,27 @@ def _cmd_query(args: argparse.Namespace) -> int: call_policy["repair_max_reprompts"] = max( 0, int(getattr(args, "repair_reprompts", 0)) ) + # Ticket #000008 Phase 4 — quantifier-guard CLI overrides. + # Six-level disable hierarchy at Levels 2 (per-call CLI flag) + # via these flags; Level 3 policy fields are reachable via the + # underlying policy dict. + if getattr(args, "no_quantifier_guard", False): + call_policy["quantifier_guard_enabled"] = False + if getattr(args, "allow_broad", False): + # Keeps the classifier on (telemetry stays useful) but + # zeroes out the apply_caps gate so broad shapes don't + # get clipped during emergent search. + call_policy["quantifier_guard_apply_caps"] = False + if getattr(args, "reject_broad", False): + # Phase 4 reject-broad: the actual rejection happens inside + # query() via the policy field; this CLI flag just sets the + # field. See aborist/qa/query.py for the early-return path. + call_policy["quantifier_reject_broad"] = True + if getattr(args, "apply_quantifier_caps", False): + # Operator opts in to flipping the dry-run gate per-call. + # Bench-first per §10.11.3 — this flag is the path from + # dry-run to live-cap. + call_policy["quantifier_guard_apply_caps"] = True result = query( question=args.question, @@ -487,6 +508,17 @@ _SOFT_DEMOTE_VIOLATION_KINDS = frozenset({ # parser found nothing groundable, but at least surfaces the # collapse cause to the operator at audit-line glance. "FORMAT_COLLAPSED", + # Ticket #000008 Phase 4 — broad-quantifier soft-demotes (§10.3). + # Per §10.3 these stay as soft demotes (cap at ANCHOR-WARRANTED) + # rather than minting a new audit_mode token. The audit-line tail + # (rendered by _render_warrant_tail) names which one fired so an + # operator can tell at a glance. + "BROAD_QUANTIFIER_RUNAWAY", # raw_line_count >> pointer_count + "BROAD_QUANTIFIER_CAP_APPLIED", # preflight cap fired below default + "BROAD_QUANTIFIER_SCOPE_UNBOUND", # unbounded universal reached the LLM + # BROAD_QUANTIFIER_REJECTED is a HARD demote (UNGROUNDED via + # early-return) — listed here for completeness but doesn't + # belong in the soft-demote set. }) @@ -618,6 +650,19 @@ def _render_warrant_tail(result: dict) -> str: parts.append("title mismatch") if "FORMAT_COLLAPSED" in kinds: parts.append("format collapsed") + # Ticket #000008 Phase 4 — broad-quantifier tails (§10.3 / §10.7). + # Each names what the preflight detected so operators don't have to + # parse violation lists by hand. Cap value comes from + # `claim_cap_applied` on the result when present. + if "BROAD_QUANTIFIER_REJECTED" in kinds: + parts.append("broad rejected") + elif "BROAD_QUANTIFIER_CAP_APPLIED" in kinds: + cap = result.get("claim_cap_applied") + parts.append(f"broad cap {cap}" if cap else "broad cap") + elif "BROAD_QUANTIFIER_SCOPE_UNBOUND" in kinds: + parts.append("broad unbounded") + elif "BROAD_QUANTIFIER_RUNAWAY" in kinds: + parts.append("broad runaway") if not parts: return "" return " · " + " · ".join(parts) @@ -692,6 +737,27 @@ def _render_query_human(result: dict, question: str) -> str: Errors / no-source paths fall back to a short status line. """ status = result.get("status") + if status == "broad_quantifier_rejected": + # Phase 4 reject-broad early-return path. The result carries + # an answer_text with the rejection rationale + a violations + # list; render both so the operator sees WHY without --json. + answer_text = result.get("answer_text") or "" + violations = result.get("violations") or [] + kind = next( + (v.get("kind") for v in violations + if v.get("kind") == "BROAD_QUANTIFIER_REJECTED"), + "BROAD_QUANTIFIER_REJECTED", + ) + intensity = result.get("quantifier_intensity") or "?" + token = result.get("quantifier_matched_token") or "?" + cap = result.get("claim_cap_applied") + cap_str = f" · cap was {cap}" if cap else "" + return ( + f"{question}\n" + f" UNGROUNDED · via {kind} · {intensity} (\"{token}\")" + f"{cap_str} 0/0 0.0s (preflight)\n\n" + f"{answer_text}" + ) if status not in ("cache_hit", "cache_miss_then_written"): msg = result.get("msg") or status or "unknown error" return f" {status or 'error'}: {msg}" @@ -3394,6 +3460,52 @@ def build_parser() -> argparse.ArgumentParser: "cache-hits ignore them)." ), ) + # Ticket #000008 Phase 4 — quantifier-guard CLI flags. Each + # corresponds to a level of the §10.11.2 disable hierarchy. + query_cmd.add_argument( + "--no-quantifier-guard", + dest="no_quantifier_guard", action="store_true", + help=( + "Disable the broad-quantifier preflight guard for this " + "call. Overrides quantifier_guard_enabled in policy. " + "Bench-side telemetry (quantifier_intensity, etc.) goes " + "to None for the row. Use when the guard misclassifies." + ), + ) + query_cmd.add_argument( + "--allow-broad", + dest="allow_broad", action="store_true", + help=( + "Emergent-search mode: keep the classifier on (telemetry " + "stays useful) but don't apply caps. For broad questions " + "where the operator wants exploratory enumeration, not " + "grounded completeness." + ), + ) + query_cmd.add_argument( + "--reject-broad", + dest="reject_broad", action="store_true", + help=( + "Strict mode: when intensity is ALL/COMPREHENSIVE/" + "OPEN_REQUEST AND scope_bound_hint is unbounded, return " + "UNGROUNDED before the LLM call with a " + "BROAD_QUANTIFIER_REJECTED violation. Saves ~10-15s on " + "rejected runs. Bounded universals (e.g. all members of " + "the Beatles) are NOT rejected." + ), + ) + query_cmd.add_argument( + "--apply-quantifier-caps", + dest="apply_quantifier_caps", action="store_true", + help=( + "Flip the dry-run gate per-call. By default Phase 2 " + "lands with quantifier_guard_apply_caps=False so the " + "cap is reported on the result but not applied to the " + "verifier. This flag enables actual cap enforcement " + "for one call. Use after dry-run bench review confirms " + "the classifier output across the question set." + ), + ) query_cmd.set_defaults(func=_cmd_query) inspect_cmd = sub.add_parser( diff --git a/aborist/qa/keys.py b/aborist/qa/keys.py index 87c321b..f3d4b55 100644 --- a/aborist/qa/keys.py +++ b/aborist/qa/keys.py @@ -228,6 +228,7 @@ _VERIFIER_POLICY_FIELDS = frozenset({ "quantifier_caps_by_intensity", "quantifier_guard_modes", "quantifier_reminder_enabled", + "quantifier_reject_broad", # Quote-mode entity policy "entity_policy", "entity_proximity_n", diff --git a/aborist/qa/query.py b/aborist/qa/query.py index f418d88..b0950b5 100644 --- a/aborist/qa/query.py +++ b/aborist/qa/query.py @@ -466,6 +466,9 @@ DEFAULT_QUERY_POLICY = { "quantifier_guard_modes": ["claim_lattice_pointer", "claim_lattice"], # Phase 3 — see runner.DEFAULT_POLICY for rationale. Default OFF. "quantifier_reminder_enabled": False, + # Phase 4 — strict reject for broad-unbounded. See runner.py for + # rationale. Default OFF. + "quantifier_reject_broad": False, # Claim-count ceiling — see runner.DEFAULT_POLICY for rationale. # Bench finding (york-england "tell me all there is to know") # caught the runaway shape; cap of 12 admits entity-list @@ -1667,6 +1670,93 @@ def query( effective_max_claims = int(claim_cap_lookup) else: effective_max_claims = _policy_max_claims + # Ticket #000008 Phase 4 — strict reject for broad-unbounded. + # When opt-in via policy / --reject-broad CLI flag, return + # UNGROUNDED before the LLM call for ALL/COMPREHENSIVE/ + # OPEN_REQUEST + scope_bound_hint==unbounded shapes. Bounded + # universals (scope_bound_hint==bounded) are NOT rejected per + # §10.1 — those are answerable. Saves the ~10-15s LLM call on + # rejected runs. + quantifier_reject_broad = bool(policy.get("quantifier_reject_broad", False)) + quantifier_should_reject = ( + quantifier_guard_on + and quantifier_mode_gated + and quantifier_reject_broad + and quantifier.get("is_broad") + and quantifier.get("scope_bound_hint") == "unbounded" + ) + if quantifier_should_reject: + # Early-return without an LLM call. Skips retrieval cost too — + # we already know the answer set is undefined. Result schema + # mirrors a normal UNGROUNDED row so bench/CLI rendering + # stays consistent. + return { + "status": "broad_quantifier_rejected", + "audit_mode": "UNGROUNDED", + "cache_key": None, + "lookup_path": "preflight", + "answer_text": ( + "BROAD-QUANTIFIER PREFLIGHT REJECTED · scope unbounded\n\n" + f"Question matched {quantifier['intensity']} intensity " + f"(\"{quantifier['matched_token']}\") with an under-" + "specified universe. Narrow the question (e.g. add a " + "year, league, country, or category) or run with " + "--allow-broad for exploratory enumeration." + ), + "sources": [], + "n_quotes": 0, + "n_verified": 0, + "verifier_method": "claim_lattice_pointer" + if answer_mode == "claim_lattice_pointer" + else "claim_lattice" + if answer_mode == "claim_lattice" + else "quote", + "unverified_quotes": [], + "partially_verified_quotes": [], + "violations": [{ + "kind": "BROAD_QUANTIFIER_REJECTED", + "intensity": quantifier["intensity"], + "matched_token": quantifier["matched_token"], + "scope_bound_hint": quantifier["scope_bound_hint"], + "reason": ( + "preflight rejection — broad-quantifier query with " + "unbounded scope. Operator opted in via " + "quantifier_reject_broad policy." + ), + }], + "format_collapsed": None, + "raw_answer": None, + "quantifier_intensity": quantifier["intensity"], + "quantifier_matched_token": quantifier["matched_token"], + "scope_bound_hint": quantifier["scope_bound_hint"], + "quantifier_explicit_count": quantifier["explicit_count"], + "claim_cap_applied": claim_cap_lookup, + "pointer_id_distribution": None, + "lazy_anchor_ratio": None, + "retrieval_purity": None, + "prompt_chars": { + "system_prompt": 0, + "grounding_reminder": 0, + "user_question": len(question), + "evidence_or_context": 0, + "messages_total": 0, + }, + "answer_chars": 0, + "frame_detection": None, + "burned_existing": 0, + "context_root": None, + "timings": { + "search_ms": 0.0, + "context_ms": 0.0, + "cache_lookup_ms": 0.0, + "llm_ms": None, + "persist_ms": None, + # Preflight rejection runs in <1ms — record 0.0 + # rather than re-fetching wall-time. The point of + # the path is to NOT spend wall time. + "total_ms": 0.0, + }, + } t_start = time.monotonic() # 1. Search. diff --git a/aborist/qa/runner.py b/aborist/qa/runner.py index 0e3c1a1..61c9695 100644 --- a/aborist/qa/runner.py +++ b/aborist/qa/runner.py @@ -231,6 +231,15 @@ DEFAULT_POLICY = { # Operator opts in per-call after Phase 2 dry-run telemetry # confirms which broad-shape rows actually need the reminder. "quantifier_reminder_enabled": False, + # Phase 4 — strict reject for broad-unbounded queries. When True + # AND intensity ∈ {ALL, COMPREHENSIVE, OPEN_REQUEST} AND + # scope_bound_hint == "unbounded", query()/ask() return UNGROUNDED + # before the LLM call with a BROAD_QUANTIFIER_REJECTED violation. + # Saves the ~10-15s LLM call on rejected runs. Default OFF — opt-in + # via --reject-broad CLI flag or per-call policy override. + # Bounded universals (e.g. all members of the Beatles, year-anchored + # questions) are NOT rejected per §10.1. + "quantifier_reject_broad": False, # Claim-count ceiling. Bench finding (2026-04-30 york-england): # "tell me all there is to know about X" prompted Hermes to spam # 26-59 encyclopedic claims sourced from training, only 2-4 of diff --git a/tests/test_quantifier_phase4.py b/tests/test_quantifier_phase4.py new file mode 100644 index 0000000..0c8991f --- /dev/null +++ b/tests/test_quantifier_phase4.py @@ -0,0 +1,197 @@ +"""Phase 4 — broad-quantifier soft-demote tails + reject-broad path. + +Three new violation kinds (§10.3) added to the soft-demote ladder: + - BROAD_QUANTIFIER_RUNAWAY raw_line_count >> pointer_count + - BROAD_QUANTIFIER_CAP_APPLIED preflight cap fired below default + - BROAD_QUANTIFIER_SCOPE_UNBOUND unbounded universal reached LLM + +Plus one HARD demote that returns UNGROUNDED via early-return: + - BROAD_QUANTIFIER_REJECTED preflight rejected before LLM call + +Tests pin: tail rendering, ladder behavior, governance-hash binding +on quantifier_reject_broad, and the early-return shape for the +reject path. +""" + +from __future__ import annotations + +import pytest + +from aborist.cli import ( + _SOFT_DEMOTE_VIOLATION_KINDS, + _ladder_rung_for_lattice, + _render_query_human, + _render_warrant_tail, +) +from aborist.qa.keys import _VERIFIER_POLICY_FIELDS, verifier_policy_hash + + +# ---------------------------------------------------------------- soft-demote registration + +@pytest.mark.parametrize("kind", [ + "BROAD_QUANTIFIER_RUNAWAY", + "BROAD_QUANTIFIER_CAP_APPLIED", + "BROAD_QUANTIFIER_SCOPE_UNBOUND", +]) +def test_broad_quantifier_kinds_in_soft_demote_set(kind): + """All three soft-demote kinds must be registered so they cap + the ladder at ANCHOR-WARRANTED rather than reaching EVIDENCE- + WARRANTED.""" + assert kind in _SOFT_DEMOTE_VIOLATION_KINDS + + +def test_rejected_kind_not_in_soft_demote_set(): + """REJECTED is a HARD demote (UNGROUNDED via early-return). + Listing it as a soft demote would let UNGROUNDED rejected runs + bubble up to ANCHOR-WARRANTED on the ladder.""" + assert "BROAD_QUANTIFIER_REJECTED" not in _SOFT_DEMOTE_VIOLATION_KINDS + + +# ---------------------------------------------------------------- ladder rung + +def test_cap_applied_demotes_to_anchor_warranted(): + """CAP_APPLIED is a soft demote — claim verified, but the cap + bound the answer below default. ANCHOR-WARRANTED, not + EVIDENCE-WARRANTED.""" + rung = _ladder_rung_for_lattice( + "STRICT", + violations=[{"kind": "BROAD_QUANTIFIER_CAP_APPLIED"}], + ) + assert rung == "ANCHOR-WARRANTED" + + +def test_scope_unbound_demotes_to_anchor_warranted(): + rung = _ladder_rung_for_lattice( + "STRICT", + violations=[{"kind": "BROAD_QUANTIFIER_SCOPE_UNBOUND"}], + ) + assert rung == "ANCHOR-WARRANTED" + + +def test_runaway_demotes_to_anchor_warranted(): + rung = _ladder_rung_for_lattice( + "STRICT", + violations=[{"kind": "BROAD_QUANTIFIER_RUNAWAY"}], + ) + assert rung == "ANCHOR-WARRANTED" + + +# ---------------------------------------------------------------- tail rendering + +def test_tail_renders_broad_cap_with_count(): + """Cap value must appear in the tail — operators shouldn't have + to dig into violations to see what cap fired.""" + tail = _render_warrant_tail({ + "violations": [{"kind": "BROAD_QUANTIFIER_CAP_APPLIED"}], + "claim_cap_applied": 8, + }) + assert "broad cap 8" in tail + + +def test_tail_renders_broad_unbounded(): + tail = _render_warrant_tail({ + "violations": [{"kind": "BROAD_QUANTIFIER_SCOPE_UNBOUND"}], + }) + assert "broad unbounded" in tail + + +def test_tail_renders_broad_runaway(): + tail = _render_warrant_tail({ + "violations": [{"kind": "BROAD_QUANTIFIER_RUNAWAY"}], + }) + assert "broad runaway" in tail + + +def test_tail_renders_broad_rejected(): + tail = _render_warrant_tail({ + "violations": [{"kind": "BROAD_QUANTIFIER_REJECTED"}], + }) + assert "broad rejected" in tail + + +def test_tail_combines_with_existing_violations(): + """A run can have both a broad-quantifier soft demote AND a + title mismatch — both should surface on the tail.""" + tail = _render_warrant_tail({ + "violations": [ + {"kind": "TITLE_MISMATCH"}, + {"kind": "BROAD_QUANTIFIER_CAP_APPLIED"}, + ], + "claim_cap_applied": 8, + }) + assert "title mismatch" in tail + assert "broad cap 8" in tail + + +def test_tail_omits_when_no_violations(): + tail = _render_warrant_tail({"violations": []}) + assert tail == "" + + +# ---------------------------------------------------------------- audit-line render + +def test_render_label_with_broad_cap_tail(): + """End-to-end through _render_query_human: HYBRID + broad cap + surfaces both the ladder rung (ANCHOR-WARRANTED-PARTIAL) and + the broad-cap tail.""" + result = { + "status": "cache_miss_then_written", + "audit_mode": "HYBRID", + "cache_key": "abc" * 21, + "context_root": "ab" * 32, + "answer_text": "Some bounded broad-quantifier answer.", + "sources": [], + "n_quotes": 8, + "n_verified": 6, + "verifier_method": "claim_lattice_pointer", + "unverified_quotes": [], + "violations": [{"kind": "BROAD_QUANTIFIER_CAP_APPLIED"}], + "claim_cap_applied": 8, + "timings": {"total_ms": 9000.0}, + } + out = _render_query_human(result, "winners of all major sports?") + assert "ANCHOR-WARRANTED-PARTIAL" in out + assert "broad cap 8" in out + + +# ---------------------------------------------------------------- governance hash + +def test_quantifier_reject_broad_in_verifier_policy_fields(): + """Flipping reject-broad must invalidate prior cache records — + same question with reject_broad=False vs True should produce + different cache_keys.""" + assert "quantifier_reject_broad" in _VERIFIER_POLICY_FIELDS + + +def test_governance_hash_changes_when_reject_broad_flips(): + base_policy = dict.fromkeys(_VERIFIER_POLICY_FIELDS, "default") + base_policy["quantifier_reject_broad"] = False + h_off = verifier_policy_hash(base_policy) + base_policy["quantifier_reject_broad"] = True + h_on = verifier_policy_hash(base_policy) + assert h_off != h_on + + +# ---------------------------------------------------------------- reject-broad path + +def test_reject_broad_returns_early_for_unbounded_all(): + """Integration-shape check: when policy enables reject_broad and + the question is broad-unbounded, query() returns UNGROUNDED with + a BROAD_QUANTIFIER_REJECTED violation BEFORE the LLM call. + + Skipped here because exercising the full query() path requires + a populated shards-dir + qa.db which isn't set up in this unit- + test scope. The behavior is exercised by the live bench cycle + when --reject-broad is passed.""" + pytest.skip( + "integration: requires populated shards; covered by live bench" + ) + + +def test_reject_broad_does_not_fire_for_bounded_universal(): + """`all members of the Beatles` is a bounded universal — should + NOT reject even with quantifier_reject_broad=True. Same skip + rationale as above.""" + pytest.skip( + "integration: requires populated shards; covered by live bench" + )