"""Soft preflight sidecar (#000011) tests. The sidecar produces ONLY soft hints labeled SOFT_*; it cannot create PREFLIGHT_OK or PREFLIGHT_BLOCKED. These tests pin: - default-OFF (sidecar disabled by default) - parse-failure modes (all return stub, never raise) - label normalization (SOFT_ prefix enforced) - rationale extraction - the dataclass schema Uses a fake ChatClient (mock chat_completion) — no LLM round-trip. """ from __future__ import annotations import pytest from arborist.qa.soft_preflight import ( SOFT_PREFLIGHT_VERSION, SoftPreflightHint, _normalize_label, _parse_soft_hint_response, soft_preflight_question, ) class _FakeChatClient: """Mock ChatClient that returns a configured response.""" def __init__(self, response: str): self.response = response self.call_count = 0 def chat_completion(self, *args, **kwargs) -> str: self.call_count += 1 return self.response class _RaisingChatClient: """Mock that raises on call — tests fail-closed behavior.""" def chat_completion(self, *args, **kwargs) -> str: raise RuntimeError("simulated network failure") # ----------------------------------------------------------- defaults def test_sidecar_default_off_returns_stub_hint(): """Without soft_preflight_enabled=True in policy, sidecar returns SOFT_DISABLED stub. No LLM call happens.""" client = _FakeChatClient("LABEL: SOFT_WELL_FORMED\nRATIONALE: ok") hint = soft_preflight_question( "what is the capital of france?", chat_client=client, model_id="test-model", policy={}, # no soft_preflight_enabled ) assert hint.classifier_label == "SOFT_DISABLED" assert hint.confidence == 0.0 assert client.call_count == 0 # no LLM call def test_sidecar_explicit_off_returns_stub(): client = _FakeChatClient("LABEL: SOFT_WELL_FORMED\nRATIONALE: ok") hint = soft_preflight_question( "anything", chat_client=client, model_id="test", policy={"soft_preflight_enabled": False}, ) assert hint.classifier_label == "SOFT_DISABLED" assert client.call_count == 0 def test_empty_question_returns_stub_even_when_enabled(): client = _FakeChatClient("LABEL: SOFT_WELL_FORMED\nRATIONALE: ok") hint = soft_preflight_question( "", chat_client=client, model_id="test", policy={"soft_preflight_enabled": True}, ) assert hint.classifier_label == "SOFT_DISABLED" assert client.call_count == 0 # ----------------------------------------------------------- happy path def test_sidecar_returns_soft_hint_when_enabled(): client = _FakeChatClient( "LABEL: SOFT_FALSE_PREMISE_SUSPECTED\n" "RATIONALE: question presupposes Mr. Burns is Homer's father" ) hint = soft_preflight_question( "When did Mr. Burns become Homer's biological father?", chat_client=client, model_id="hermes-test", policy={"soft_preflight_enabled": True}, ) assert hint.classifier_label == "SOFT_FALSE_PREMISE_SUSPECTED" assert hint.confidence == 0.5 assert "Mr. Burns" in hint.rationale assert hint.model_profile_id == "hermes-test" assert hint.sidecar_version == SOFT_PREFLIGHT_VERSION assert client.call_count == 1 @pytest.mark.parametrize("label", [ "SOFT_WELL_FORMED", "SOFT_CONTRADICTION_SUSPECTED", "SOFT_TIME_SENSITIVE", "SOFT_OUT_OF_CORPUS_LIKELY", "SOFT_BROAD_QUANTIFIER", "SOFT_MULTI_HOP_REASONING", "SOFT_SUBJECTIVE", "SOFT_SCOPE_AMBIGUOUS", ]) def test_each_canonical_label_passes_through(label): client = _FakeChatClient(f"LABEL: {label}\nRATIONALE: test") hint = soft_preflight_question( "test", chat_client=client, model_id="m", policy={"soft_preflight_enabled": True}, ) assert hint.classifier_label == label # ----------------------------------------------------------- failure modes def test_chat_client_exception_returns_parse_fail(): """Sidecar fails-closed on chat_completion exceptions.""" client = _RaisingChatClient() hint = soft_preflight_question( "test", chat_client=client, model_id="m", policy={"soft_preflight_enabled": True}, ) assert hint.classifier_label == "SOFT_PARSE_FAIL" assert hint.confidence == 0.0 assert "RuntimeError" in hint.rationale def test_unparseable_response_returns_parse_fail(): client = _FakeChatClient("garbage output no label here") hint = soft_preflight_question( "test", chat_client=client, model_id="m", policy={"soft_preflight_enabled": True}, ) assert hint.classifier_label == "SOFT_PARSE_FAIL" assert hint.confidence == 0.0 def test_model_drift_label_outside_enum_returns_parse_fail(): """Model returns a label not in the enum → fail-closed.""" client = _FakeChatClient( "LABEL: SOFT_INVENTED_NEW_LABEL\nRATIONALE: model drift" ) hint = soft_preflight_question( "test", chat_client=client, model_id="m", policy={"soft_preflight_enabled": True}, ) assert hint.classifier_label == "SOFT_PARSE_FAIL" def test_label_without_soft_prefix_gets_normalized(): """Model drops the SOFT_ prefix → normalizer adds it back.""" client = _FakeChatClient("LABEL: WELL_FORMED\nRATIONALE: ok") hint = soft_preflight_question( "test", chat_client=client, model_id="m", policy={"soft_preflight_enabled": True}, ) assert hint.classifier_label == "SOFT_WELL_FORMED" def test_label_with_trailing_punctuation_normalized(): client = _FakeChatClient("LABEL: SOFT_WELL_FORMED.\nRATIONALE: ok") hint = soft_preflight_question( "test", chat_client=client, model_id="m", policy={"soft_preflight_enabled": True}, ) assert hint.classifier_label == "SOFT_WELL_FORMED" # ----------------------------------------------------------- pure helpers def test_normalize_label_handles_whitespace(): assert _normalize_label(" SOFT_WELL_FORMED ") == "SOFT_WELL_FORMED" def test_normalize_label_uppercases(): assert _normalize_label("soft_well_formed") == "SOFT_WELL_FORMED" def test_normalize_label_returns_parse_fail_for_garbage(): assert _normalize_label("just garbage") == "SOFT_PARSE_FAIL" def test_parse_soft_hint_handles_mixed_case_keys(): label, rationale = _parse_soft_hint_response( "label: SOFT_WELL_FORMED\nrationale: ok" ) assert label == "SOFT_WELL_FORMED" assert rationale == "ok" def test_parse_soft_hint_handles_missing_rationale_line(): """When the model only writes LABEL: but no RATIONALE: line.""" label, rationale = _parse_soft_hint_response("LABEL: SOFT_WELL_FORMED") assert label == "SOFT_WELL_FORMED" assert rationale # something, even if "(no rationale)" def test_parse_soft_hint_caps_rationale_length(): """Long rationales get capped to keep payload bounded.""" long_rationale = "x" * 500 label, rationale = _parse_soft_hint_response( f"LABEL: SOFT_WELL_FORMED\nRATIONALE: {long_rationale}" ) assert len(rationale) <= 200 # ----------------------------------------------------------- dataclass schema def test_hint_to_dict_is_json_serializable(): """Bench rows / run-DAG persist soft hints as JSON; the dataclass round-trips cleanly.""" import json client = _FakeChatClient("LABEL: SOFT_WELL_FORMED\nRATIONALE: ok") hint = soft_preflight_question( "test", chat_client=client, model_id="m", policy={"soft_preflight_enabled": True}, ) d = hint.to_dict() json.dumps(d, ensure_ascii=False) # raises if non-serializable def test_version_pinned(): assert SOFT_PREFLIGHT_VERSION == "soft-preflight-v0.1"