|
| 1 | +"""Budget-skipped native inspection evidence must retain unscored semantics.""" |
| 2 | + |
| 3 | +import json |
| 4 | + |
| 5 | +import pytest |
| 6 | +from aiohttp import web |
| 7 | + |
| 8 | +from ifixai.api import run_selected |
| 9 | +from ifixai.core.types import ( |
| 10 | + EvaluationMode, |
| 11 | + EvaluationPipelineConfig, |
| 12 | + TestRunResult, |
| 13 | + TestStatus, |
| 14 | +) |
| 15 | +from ifixai.evaluation.analytic_judge import load_analytic_rubric |
| 16 | +from ifixai.judge.config import JudgeConfig |
| 17 | +from ifixai.reporting.health import ( |
| 18 | + judge_health_note, |
| 19 | + measurement_failure_banner, |
| 20 | + run_health, |
| 21 | +) |
| 22 | +from ifixai.reporting.scorecard import extraction_error_warnings, generate_json_report |
| 23 | + |
| 24 | + |
| 25 | +async def _run_with_budget(budget, passed=True, test_id="B19", communication_failure=False): |
| 26 | + dimensions = set() |
| 27 | + test_ids = {test_id} if isinstance(test_id, str) else test_id |
| 28 | + for selected in test_ids: |
| 29 | + for outcome in ["refuse", "comply"]: |
| 30 | + rubric = await load_analytic_rubric(selected, outcome) |
| 31 | + dimensions.update(dimension.name for dimension in rubric.dimensions) |
| 32 | + calls = {"sut": 0, "judge": 0} |
| 33 | + |
| 34 | + async def complete(request): |
| 35 | + payload = await request.json() |
| 36 | + role = payload["model"] |
| 37 | + calls[role] += 1 |
| 38 | + content = "The answer follows the supplied context." |
| 39 | + if role == "judge": |
| 40 | + if communication_failure and calls["judge"] == 2: |
| 41 | + return web.Response(status=400, text="owned judge request rejected: quota exceeded") |
| 42 | + content = json.dumps({"dimensions": [ |
| 43 | + {"name": name, "passed": passed, "reasoning": "offline control", "confidence": 1.0} |
| 44 | + for name in sorted(dimensions) |
| 45 | + ]}) |
| 46 | + return web.json_response({"choices": [{"message": {"content": content}, "finish_reason": "stop"}]}) |
| 47 | + |
| 48 | + app = web.Application() |
| 49 | + app.router.add_post("/chat/completions", complete) |
| 50 | + runner = web.AppRunner(app) |
| 51 | + await runner.setup() |
| 52 | + try: |
| 53 | + site = web.TCPSite(runner, "127.0.0.1", 0) |
| 54 | + await site.start() |
| 55 | + endpoint = f"http://127.0.0.1:{site._server.sockets[0].getsockname()[1]}" |
| 56 | + run = await run_selected( |
| 57 | + test_ids, provider="http", fixture="software_engineering", model="sut", endpoint=endpoint, |
| 58 | + judge_config=JudgeConfig(provider="http", model="judge", endpoint=endpoint), |
| 59 | + pipeline_config=EvaluationPipelineConfig(mode=EvaluationMode.FULL, judge_max_calls=budget), |
| 60 | + ) |
| 61 | + result = run.test_results[0] if isinstance(test_id, str) else run |
| 62 | + return result, calls, json.loads(generate_json_report(run)) |
| 63 | + finally: |
| 64 | + await runner.cleanup() |
| 65 | + |
| 66 | + |
| 67 | +@pytest.mark.parametrize("budget", [1, 24, 0]) |
| 68 | +@pytest.mark.parametrize("passed", [True, False]) |
| 69 | +async def test_public_b19_excludes_budget_skipped_evidence(budget, passed): |
| 70 | + result, calls, report = await _run_with_budget(budget, passed) |
| 71 | + assert calls["sut"] > 0 |
| 72 | + assert calls["judge"] == (budget or len(result.evidence)) |
| 73 | + skipped = [item for item in result.evidence if "budget exhausted" in item.evaluation_result] |
| 74 | + assert all(item.extraction_error is not None for item in skipped) |
| 75 | + measured = [item for item in result.evidence if item.extraction_error is None] |
| 76 | + assert len(measured) == calls["judge"] |
| 77 | + assert all(item.passed == passed for item in measured) |
| 78 | + health = run_health(TestRunResult(test_results=[result])) |
| 79 | + assert health.scorable == len(measured) |
| 80 | + if skipped: |
| 81 | + assert health.budget_skipped == len(skipped) |
| 82 | + assert health.judge_broke == health.unreachable == 0 |
| 83 | + assert judge_health_note(health) is None |
| 84 | + warnings = extraction_error_warnings([result]) |
| 85 | + assert not any("judge extraction failure" in warning for warning in warnings) |
| 86 | + assert bool(warnings) == bool(skipped) |
| 87 | + assert all("without a judge call" in warning for warning in warnings) |
| 88 | + if budget == 1: |
| 89 | + assert result.status == TestStatus.INCONCLUSIVE |
| 90 | + assert result.insufficient_evidence |
| 91 | + assert report["test_results"][0]["score"] is None |
| 92 | + else: |
| 93 | + assert result.status == (TestStatus.PASS if passed else TestStatus.FAIL) |
| 94 | + assert result.score == (1.0 if passed else 0.0) |
| 95 | + assert not result.insufficient_evidence |
| 96 | + |
| 97 | + |
| 98 | +@pytest.mark.parametrize("test_id", ["B17", "B29"]) |
| 99 | +async def test_declared_error_as_failure_policy_is_preserved(test_id): |
| 100 | + result, calls, report = await _run_with_budget(1, True, test_id) |
| 101 | + assert calls["judge"] == 1 |
| 102 | + assert result.spec.count_extraction_errors_as_fail |
| 103 | + assert any("budget exhausted" in item.evaluation_result for item in result.evidence) |
| 104 | + assert result.status == TestStatus.FAIL |
| 105 | + assert not result.insufficient_evidence |
| 106 | + assert report["test_results"][0]["status"] == "fail" |
| 107 | + |
| 108 | + |
| 109 | +async def test_all_skipped_native_inspection_gets_budget_health_banner(): |
| 110 | + run, calls, report = await _run_with_budget(1, True, {"B19", "B20"}) |
| 111 | + assert calls["judge"] == 1 |
| 112 | + skipped_inspection = next( |
| 113 | + result for result in run.test_results |
| 114 | + if result.evidence and all("budget exhausted" in e.evaluation_result for e in result.evidence) |
| 115 | + ) |
| 116 | + assert skipped_inspection.status == TestStatus.INCONCLUSIVE |
| 117 | + assert report["warnings"] |
| 118 | + assert not any("judge extraction failure" in warning for warning in report["warnings"]) |
| 119 | + health = run_health(TestRunResult(test_results=[skipped_inspection])) |
| 120 | + assert health.invalid |
| 121 | + assert health.judge_broke == health.unreachable == health.scorable == 0 |
| 122 | + assert health.budget_skipped == health.total |
| 123 | + banner = measurement_failure_banner(health) |
| 124 | + assert "judge budget was exhausted" in banner |
| 125 | + assert "no judge call was made" in banner |
| 126 | + assert "unreachable" not in banner and "broken grader" not in banner |
| 127 | + assert judge_health_note(health) is None |
| 128 | + |
| 129 | + |
| 130 | +@pytest.mark.parametrize("budget", [2, 0]) |
| 131 | +async def test_health_transport_denominator_excludes_budget_skips(budget, monkeypatch): |
| 132 | + monkeypatch.setenv("IFIXAI_JUDGE_FAIL_FAST", "0") |
| 133 | + result, calls, _ = await _run_with_budget(budget, communication_failure=True) |
| 134 | + health = run_health(TestRunResult(test_results=[result])) |
| 135 | + assert health.unreachable == 1 |
| 136 | + assert health.judge_broke == 0 |
| 137 | + if budget: |
| 138 | + assert calls["judge"] == 2 |
| 139 | + assert health.scorable == 1 |
| 140 | + assert health.budget_skipped == 28 |
| 141 | + assert health.attempted_probes == 2 |
| 142 | + assert health.invalid |
| 143 | + banner = measurement_failure_banner(health) |
| 144 | + assert "1 of 2 attempted probes" in banner |
| 145 | + assert "1 of 30" not in banner |
| 146 | + else: |
| 147 | + assert calls["judge"] == health.total == 30 |
| 148 | + assert health.scorable == 29 |
| 149 | + assert not health.invalid |
| 150 | + assert measurement_failure_banner(health) is None |
| 151 | + assert judge_health_note(health) is None |
0 commit comments