arborist/tests/test_quantifier_caps.py
russell@unturf.com 8d6961fcc1
aborist/arborist
modified:   .gitlab-ci.yml
	modified:   bench/qa_questions.txt
	modified:   bench/qa_sweep.py
	modified:   bench/run.sh
	modified:   docs/TICKETS.md
	modified:   docs/_source/README.md
	modified:   docs/_source/_ext/makefile_targets.py
	modified:   docs/_source/api/cli.rst
	modified:   docs/_source/api/distill.rst
	modified:   docs/_source/api/mesh.rst
	modified:   docs/_source/api/qa.rst
	modified:   docs/_source/api/retrieval.rst
	modified:   docs/_source/api/storage.rst
	modified:   docs/_source/api/substrate.rst
	modified:   docs/_source/concepts.rst
	modified:   docs/_source/conf.py
	modified:   docs/_source/cookbook.rst
	modified:   docs/_source/index.rst
	modified:   docs/_source/license.rst
	modified:   docs/_source/quickstart.rst
	modified:   docs/bench-maxing.md
	modified:   docs/benchmarks.md
	modified:   docs/cti-architecture.md
	modified:   docs/diagrams/aborist-modules.dot
	modified:   docs/diagrams/aborist-modules.svg
	modified:   docs/diagrams/mesh-data-flow.dot
	modified:   docs/diagrams/mesh-epoch-lifecycle.dot
	modified:   docs/diagrams/mesh-epoch-lifecycle.svg
	modified:   docs/diagrams/mesh-group-decisions.dot
	modified:   docs/diagrams/mesh-group-decisions.svg
	modified:   docs/diagrams/mesh-identity-stack.dot
	modified:   docs/diagrams/mesh-secret-envelope.dot
	modified:   docs/mesh.md
	modified:   docs/qa-modes-bench.md
	modified:   docs/seven-point-program.md
	modified:   docs/tickets/ticket-000001-retrieval-keywords-audit-gap.md
	modified:   docs/tickets/ticket-000002-reference-frame-polarity-contract.md
	modified:   docs/tickets/ticket-000003-anchor-class-warrant.md
	modified:   docs/tickets/ticket-000005-label-ladder-migration.md
	modified:   docs/tickets/ticket-000006-bench-emergent-findings.md
	modified:   docs/tickets/ticket-000007-query-layer-hyphen-fold.md
	modified:   docs/tickets/ticket-000008-broad-quantifier-preflight-guard.md
	modified:   docs/tickets/ticket-000009-quantifier-preflight-dag-binding.md
	modified:   docs/tickets/ticket-000010-metacognition-preflight-guard.md
	modified:   docs/tickets/ticket-000011-soft-preflight-hint-sidecar.md
	modified:   scripts/backfill_concepts.py
	modified:   scripts/bench_emergent.py
	modified:   tests/crawler/test_async_web_fetcher.py
	modified:   tests/crawler/test_bridge.py
	modified:   tests/crawler/test_web_fetch.py
	modified:   tests/test_bench_qa_sweep.py
	modified:   tests/test_burn.py
	modified:   tests/test_burn_doc.py
	modified:   tests/test_claim_lattice.py
	modified:   tests/test_cli_render.py
	modified:   tests/test_compress.py
	modified:   tests/test_concepts.py
	modified:   tests/test_dag.py
	modified:   tests/test_directives.py
	modified:   tests/test_distill.py
	modified:   tests/test_distill_recursive.py
	modified:   tests/test_evict.py
	modified:   tests/test_frame.py
	modified:   tests/test_grok_source.py
	modified:   tests/test_html_source.py
	modified:   tests/test_ingest.py
	modified:   tests/test_inspect.py
	modified:   tests/test_journal.py
	modified:   tests/test_keys.py
	modified:   tests/test_llm_context_base.py
	modified:   tests/test_merkle.py
	modified:   tests/test_mesh.py
	modified:   tests/test_mesh_aead.py
	modified:   tests/test_mesh_chain.py
	modified:   tests/test_mesh_cli.py
	modified:   tests/test_mesh_cli_pull.py
	modified:   tests/test_mesh_wire.py
	modified:   tests/test_mesh_wire_e2e.py
	modified:   tests/test_metacognition.py
	modified:   tests/test_migration_audit_mode.py
	modified:   tests/test_providence_source.py
	modified:   tests/test_qa.py
	modified:   tests/test_qa_quality_live.py
	modified:   tests/test_quantifier_caps.py
	modified:   tests/test_quantifier_classifier.py
	modified:   tests/test_quantifier_phase4.py
	modified:   tests/test_quantifier_reminder.py
	modified:   tests/test_query.py
	modified:   tests/test_reclassify.py
	modified:   tests/test_repair.py
	modified:   tests/test_resume.py
	modified:   tests/test_snapshot.py
	modified:   tests/test_soft_preflight.py
	modified:   tests/test_tfidf.py
	modified:   tests/test_vcs_source.py
	modified:   tests/test_verify.py
	modified:   tests/test_verify_json.py
	modified:   tests/test_versioned_ingest.py
	modified:   tests/test_warrant.py
	modified:   tests/test_wikipedia_old.py
	modified:   tests/test_wikipedia_xml.py
	modified:   tests/test_wikitext.py
2026-05-07 09:31:49 -04:00

277 lines
11 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Quantifier cap table + governance hash + dry-run discipline.
Ticket #000008 Phase 2: per-model cap profiles
(``arborist/qa/model_profiles.py``) and the four policy fields that
gate cap behavior:
- ``quantifier_guard_enabled`` — master kill
- ``quantifier_guard_apply_caps`` — dry-run gate (default False)
- ``quantifier_caps_by_intensity`` — per-call override dict
- ``quantifier_guard_modes`` — per-mode opt-in list
Tests pin:
1. Per-model cap selection (Hermes profile vs default fallback).
2. EXPLICIT_COUNT sentinel — SMALL_NUM_EXPLICIT respects the
question's count.
3. Policy override beats per-model profile.
4. Governance-hash invalidation on every cap-related field flip.
5. Six-level disable hierarchy (master, dry-run, per-mode,
per-call override).
"""
from __future__ import annotations
import pytest
from arborist.qa.keys import _VERIFIER_POLICY_FIELDS, verifier_policy_hash
from arborist.qa.model_profiles import (
EXPLICIT_COUNT,
PROFILES,
cap_for_intensity,
profile_id_present,
)
# ---------------------------------------------------------------- profile lookup
def test_hermes_fp8_profile_present():
"""The Hermes-3-FP8 endpoint profile must exist verbatim — the
string is what the bench harness sends as ``model_profile_id``."""
assert profile_id_present("adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic")
def test_default_profile_fallback():
"""Unknown model ids fall through to the ``default`` profile."""
assert not profile_id_present("some-future-model-v2")
cap = cap_for_intensity(
model_profile_id="some-future-model-v2",
intensity="ALL",
)
# Default profile sets ALL = 12.
assert cap == 12
def test_hermes_caps_are_tighter_than_default():
"""Hermes broad-quantifier caps must be tighter than the default
profile — that's the whole point of the per-model split. Pin the
relationship so a future profile edit can't accidentally loosen
Hermes back to default."""
hermes_id = "adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic"
for intensity in ("ALL", "COMPREHENSIVE", "OPEN_REQUEST", "MANY"):
hermes_cap = cap_for_intensity(
model_profile_id=hermes_id, intensity=intensity,
)
default_cap = cap_for_intensity(
model_profile_id="unknown", intensity=intensity,
)
assert hermes_cap is not None
assert default_cap is not None
assert hermes_cap <= default_cap, (
f"Hermes {intensity} cap ({hermes_cap}) must not exceed "
f"default cap ({default_cap})"
)
# ---------------------------------------------------------------- EXPLICIT_COUNT
def test_explicit_count_resolves_from_question_count():
"""SMALL_NUM_EXPLICIT cap = the explicit count from the question."""
cap = cap_for_intensity(
model_profile_id="adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic",
intensity="SMALL_NUM_EXPLICIT",
explicit_count=3,
)
assert cap == 3
def test_explicit_count_falls_back_when_missing():
"""If the classifier reported SMALL_NUM_EXPLICIT but didn't
extract a count (defensive — shouldn't happen), fall back to the
profile's MANY cap rather than crash."""
cap = cap_for_intensity(
model_profile_id="adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic",
intensity="SMALL_NUM_EXPLICIT",
explicit_count=None,
)
# Hermes MANY = 8.
assert cap == 8
def test_comparative_bound_uses_explicit_count():
"""COMPARATIVE_BOUND also reads explicit_count (the upper bound
extracted by the classifier — for `between A and B` the upper)."""
cap = cap_for_intensity(
model_profile_id="adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic",
intensity="COMPARATIVE_BOUND",
explicit_count=7,
)
assert cap == 7
# ---------------------------------------------------------------- override
def test_policy_override_beats_per_model_profile():
"""A per-call ``quantifier_caps_by_intensity`` dict wins over the
profile table. Lets an operator override caps without editing
model_profiles.py."""
cap = cap_for_intensity(
model_profile_id="adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic",
intensity="ALL",
policy_overrides={"ALL": 4},
)
assert cap == 4
def test_policy_override_partial_falls_back_to_profile():
"""An override dict with one intensity doesn't suppress the
others — the profile still answers for unspecified intensities."""
cap_all = cap_for_intensity(
model_profile_id="adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic",
intensity="ALL",
policy_overrides={"COMPREHENSIVE": 3}, # only COMPREHENSIVE overridden
)
# ALL still pulls from Hermes profile (8).
assert cap_all == 8
def test_unknown_intensity_returns_none():
"""Caller should treat None as 'no cap; use the policy default'."""
cap = cap_for_intensity(
model_profile_id="adamo1139/Hermes-3-Llama-3.1-8B-FP8-Dynamic",
intensity="MADE_UP_INTENSITY",
)
assert cap is None
# ---------------------------------------------------------------- governance hash
@pytest.mark.parametrize("field", [
"quantifier_guard_enabled",
"quantifier_guard_apply_caps",
"quantifier_caps_by_intensity",
"quantifier_guard_modes",
])
def test_each_quantifier_field_is_in_verifier_policy_fields(field):
"""All four Phase-2 policy fields must be in
``_VERIFIER_POLICY_FIELDS`` so flipping any of them invalidates
prior cache records on lookup. Without this binding, an operator
could change caps and silently re-use stale cached verdicts."""
assert field in _VERIFIER_POLICY_FIELDS
def test_governance_hash_changes_when_apply_caps_flips():
"""Dry-run → cap-on flip MUST bump governance_policy_hash. Same
classifier output, different verifier behavior — distinct
cache_key per §10.11.2 Level 6."""
base_policy = dict.fromkeys(_VERIFIER_POLICY_FIELDS, "default")
base_policy["quantifier_guard_apply_caps"] = False
h_dry = verifier_policy_hash(base_policy)
base_policy["quantifier_guard_apply_caps"] = True
h_apply = verifier_policy_hash(base_policy)
assert h_dry != h_apply
def test_governance_hash_changes_when_caps_dict_changes():
"""Editing the per-call override dict must bump the hash too."""
base_policy = dict.fromkeys(_VERIFIER_POLICY_FIELDS, "default")
base_policy["quantifier_caps_by_intensity"] = {}
h_empty = verifier_policy_hash(base_policy)
base_policy["quantifier_caps_by_intensity"] = {"ALL": 4}
h_override = verifier_policy_hash(base_policy)
assert h_empty != h_override
def test_governance_hash_changes_when_master_switch_flips():
base_policy = dict.fromkeys(_VERIFIER_POLICY_FIELDS, "default")
base_policy["quantifier_guard_enabled"] = True
h_on = verifier_policy_hash(base_policy)
base_policy["quantifier_guard_enabled"] = False
h_off = verifier_policy_hash(base_policy)
assert h_on != h_off
def test_governance_hash_changes_when_mode_list_changes():
base_policy = dict.fromkeys(_VERIFIER_POLICY_FIELDS, "default")
base_policy["quantifier_guard_modes"] = ["claim_lattice_pointer", "claim_lattice"]
h_default = verifier_policy_hash(base_policy)
base_policy["quantifier_guard_modes"] = ["claim_lattice_pointer"]
h_pointer_only = verifier_policy_hash(base_policy)
assert h_default != h_pointer_only
# ---------------------------------------------------------------- profile shape
def test_profile_keys_include_all_ten_rungs():
"""Every profile must cover all ten intensity rungs so cap
lookup can't fall through to the default profile due to a
missing key."""
expected_rungs = {
"ABSENT", "SINGULAR", "PROPORTIONAL",
"SMALL_NUM_EXPLICIT", "COMPARATIVE_BOUND",
"FEW", "MANY", "ALL", "COMPREHENSIVE", "OPEN_REQUEST",
}
for profile_id, profile in PROFILES.items():
missing = expected_rungs - set(profile.keys())
assert not missing, f"{profile_id} missing rungs: {missing}"
def test_default_profile_present():
"""Lookup-fallback target must exist or unknown models would
return None on every call."""
assert "default" in PROFILES
# Ticket #000008 §12.10 — n=5 verification produced these defaults
# (committed 2026-05-03 post cap+reminder bench). Pin them here so a
# future drift in DEFAULT_QUERY_POLICY surfaces a failing test.
def test_default_reminder_enabled_for_lattice_modes():
"""Phase 3 reminder default-ON per #000008 §12 bench finding:
eliminates FORMAT_COLLAPSED, reduces NO_EVIDENCE_POINTER 33%,
boosts mean ratio +17pp pointer / +21pp JSON, rescues JSON
UNGROUNDED 7→1."""
from arborist.qa.runner import DEFAULT_POLICY as RUNNER_POLICY
from arborist.qa.query import DEFAULT_QUERY_POLICY
assert RUNNER_POLICY["quantifier_reminder_enabled"] is True
assert DEFAULT_QUERY_POLICY["quantifier_reminder_enabled"] is True
def test_default_apply_caps_modes_is_json_only():
"""Phase 2 cap default-allowlist per #000008 §12.10 n=5 finding:
cap-on-JSON wins +14pp on STRICT-rate; cap-on-pointer fires
TOO_MANY_CLAIMS 20× without moving the 0/45 STRICT floor.
Default the apply-caps allowlist to JSON-only so flipping
apply_caps=True doesn't add wasted cap-noise on pointer mode."""
from arborist.qa.runner import DEFAULT_POLICY as RUNNER_POLICY
from arborist.qa.query import DEFAULT_QUERY_POLICY
assert RUNNER_POLICY["quantifier_apply_caps_modes"] == ["claim_lattice"]
assert DEFAULT_QUERY_POLICY["quantifier_apply_caps_modes"] == ["claim_lattice"]
def test_apply_caps_modes_is_in_verifier_policy_fields():
"""Without governance binding, an operator could change the
allowlist and silently re-use cached records written under a
different allowlist."""
assert "quantifier_apply_caps_modes" in _VERIFIER_POLICY_FIELDS
def test_governance_hash_changes_when_apply_caps_modes_changes():
base_policy = dict.fromkeys(_VERIFIER_POLICY_FIELDS, "default")
base_policy["quantifier_apply_caps_modes"] = ["claim_lattice"]
h_json_only = verifier_policy_hash(base_policy)
base_policy["quantifier_apply_caps_modes"] = [
"claim_lattice", "claim_lattice_pointer"
]
h_both = verifier_policy_hash(base_policy)
assert h_json_only != h_both
def test_apply_caps_default_off_preserves_dry_run_discipline():
"""The cap-application gate stays operator-opt-in by default
even with reminder default-on. Dry-run discipline (§10.11.3)
survives the §12 bench cycle — operators flip apply_caps via
--apply-quantifier-caps after their own bench review."""
from arborist.qa.runner import DEFAULT_POLICY as RUNNER_POLICY
from arborist.qa.query import DEFAULT_QUERY_POLICY
assert RUNNER_POLICY["quantifier_guard_apply_caps"] is False
assert DEFAULT_QUERY_POLICY["quantifier_guard_apply_caps"] is False