Skip to content

Commit 33369e1

Browse files
fix(evaluation): keep budget-skipped probes unscored (#169)
Signed-off-by: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> Co-authored-by: n-papaioannou <n.papaioannou@ime.life>
1 parent d6e2e80 commit 33369e1

6 files changed

Lines changed: 195 additions & 13 deletions

File tree

‎ifixai/cli/run.py‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1840,7 +1840,8 @@ def run(
18401840
result.validation_warnings.append(
18411841
"run_invalid: measurement failure — "
18421842
f"errored={health.errored}/{health.n_inspections} inspections, "
1843-
f"unreachable={health.unreachable}/{health.total} model calls, "
1843+
f"unreachable={health.unreachable}/{health.attempted_probes} attempted probes, "
1844+
f"budget_skipped={health.budget_skipped}, "
18441845
f"judge_broke={health.judge_broke}, scorable={health.scorable}. "
18451846
"Ignore the grade."
18461847
)

‎ifixai/core/types.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -201,6 +201,7 @@ class JudgeErrorKind(str, Enum):
201201
COMMUNICATION = "communication"
202202
EXTRACTION = "extraction"
203203
CONTRACT = "contract"
204+
BUDGET = "budget"
204205

205206

206207
class EvaluationMode(str, Enum):

‎ifixai/evaluation/pipeline.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -218,6 +218,7 @@ async def evaluate(
218218
passed=False,
219219
evaluation_result="inconclusive: judge budget exhausted",
220220
evaluation_method=EvaluationMethod.JUDGE,
221+
extraction_error=JudgeErrorKind.BUDGET,
221222
)
222223

223224

‎ifixai/reporting/health.py‎

Lines changed: 19 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -21,6 +21,11 @@ class _RunHealth(NamedTuple):
2121
scorable: int # produced a graded result (no extraction_error)
2222
unreachable: int # a model call failed to communicate (agent OR judge)
2323
judge_broke: int # the judge replied but the verdict was unusable
24+
budget_skipped: int = 0 # no judge call was made after the run budget ran out
25+
26+
@property
27+
def attempted_probes(self) -> int:
28+
return self.total - self.budget_skipped
2429

2530
@property
2631
def invalid(self) -> bool:
@@ -36,7 +41,7 @@ def invalid(self) -> bool:
3641
return self.n_inspections > 0
3742
if self.scorable == 0:
3843
return True # every probe was unscorable
39-
return self.unreachable / self.total >= 0.5
44+
return self.attempted_probes > 0 and self.unreachable / self.attempted_probes >= 0.5
4045

4146
@property
4247
def low_confidence(self) -> bool:
@@ -61,8 +66,9 @@ def run_health(result: TestRunResult) -> _RunHealth:
6166
6267
COMMUNICATION is a failed model call on either seam (agent under test or
6368
judge); CONTRACT/EXTRACTION mean the judge answered but the verdict was
64-
unusable — unambiguously a grader-health problem."""
65-
errored = total = scorable = unreachable = judge_broke = 0
69+
unusable — unambiguously a grader-health problem. BUDGET means no judge
70+
call was made, so it is neither a transport nor a verdict failure."""
71+
errored = total = scorable = unreachable = judge_broke = budget_skipped = 0
6672
for br in result.test_results:
6773
if br.status == TestStatus.ERROR:
6874
errored += 1
@@ -77,9 +83,11 @@ def run_health(result: TestRunResult) -> _RunHealth:
7783
unreachable += 1
7884
elif kind in (JudgeErrorKind.CONTRACT, JudgeErrorKind.EXTRACTION):
7985
judge_broke += 1
86+
elif kind == JudgeErrorKind.BUDGET:
87+
budget_skipped += 1
8088
else:
8189
unreachable += 1
82-
return _RunHealth(len(result.test_results), errored, total, scorable, unreachable, judge_broke)
90+
return _RunHealth(len(result.test_results), errored, total, scorable, unreachable, judge_broke, budget_skipped)
8391

8492

8593
def measurement_failure_banner(health: _RunHealth) -> str | None:
@@ -94,6 +102,12 @@ def measurement_failure_banner(health: _RunHealth) -> str | None:
94102
f"all {health.n_inspections} inspection(s) ran but produced no gradeable "
95103
"evidence. Check the model id / key / endpoint and re-run."
96104
)
105+
elif health.budget_skipped == health.total and not health.errored:
106+
cause = (
107+
f"all {health.budget_skipped} probes were skipped because the judge "
108+
"budget was exhausted; no judge call was made for those probes. "
109+
"Increase the judge budget or select fewer inspections and re-run."
110+
)
97111
elif health.scorable == 0 and health.judge_broke and health.unreachable == 0 and not health.errored:
98112
# Every probe reached a model but the judge produced no usable verdict —
99113
# a broken grader, not a model problem. Don't send the user to the model.
@@ -110,7 +124,7 @@ def measurement_failure_banner(health: _RunHealth) -> str | None:
110124
)
111125
else:
112126
cause = (
113-
f"{health.unreachable} of {health.total} model calls failed to complete "
127+
f"{health.unreachable} of {health.attempted_probes} attempted probes had model calls fail to complete "
114128
"(the agent under test or a judge was unreachable). Check the model id / "
115129
"key / endpoint (the preflight catches a bad model id before billing) "
116130
"and re-run."

‎ifixai/reporting/scorecard.py‎

Lines changed: 21 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -2,7 +2,13 @@
22
from datetime import datetime, timezone
33
from typing import Final
44

5-
from ifixai.core.types import RegulatoryFramework, TestResult, TestRunResult, TestStatus
5+
from ifixai.core.types import (
6+
JudgeErrorKind,
7+
RegulatoryFramework,
8+
TestResult,
9+
TestRunResult,
10+
TestStatus,
11+
)
612
from ifixai.judge.config import JudgeConfig
713
from ifixai.mappings.loader import load_all_mappings
814
from ifixai.reporting.regulatory import (
@@ -181,13 +187,21 @@ def extraction_error_warnings(
181187
) -> list[str]:
182188
messages: list[str] = []
183189
for br in test_results:
184-
affected = sum(1 for ev in br.evidence if ev.extraction_error is not None)
185-
if affected == 0:
186-
continue
187-
messages.append(
188-
EXTRACTION_ERROR_PREFIX
189-
+ f"{br.test_id} ({affected} evidence items affected)"
190+
affected = sum(
191+
1 for ev in br.evidence
192+
if ev.extraction_error is not None and ev.extraction_error != JudgeErrorKind.BUDGET
190193
)
194+
if affected:
195+
messages.append(
196+
EXTRACTION_ERROR_PREFIX
197+
+ f"{br.test_id} ({affected} evidence items affected)"
198+
)
199+
skipped = sum(ev.extraction_error == JudgeErrorKind.BUDGET for ev in br.evidence)
200+
if skipped:
201+
messages.append(
202+
f"judge budget exhausted: {br.test_id} "
203+
f"({skipped} evidence items skipped without a judge call)"
204+
)
191205
return messages
192206

193207

Lines changed: 151 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,151 @@
1+
"""Budget-skipped native inspection evidence must retain unscored semantics."""
2+
3+
import json
4+
5+
import pytest
6+
from aiohttp import web
7+
8+
from ifixai.api import run_selected
9+
from ifixai.core.types import (
10+
EvaluationMode,
11+
EvaluationPipelineConfig,
12+
TestRunResult,
13+
TestStatus,
14+
)
15+
from ifixai.evaluation.analytic_judge import load_analytic_rubric
16+
from ifixai.judge.config import JudgeConfig
17+
from ifixai.reporting.health import (
18+
judge_health_note,
19+
measurement_failure_banner,
20+
run_health,
21+
)
22+
from ifixai.reporting.scorecard import extraction_error_warnings, generate_json_report
23+
24+
25+
async def _run_with_budget(budget, passed=True, test_id="B19", communication_failure=False):
26+
dimensions = set()
27+
test_ids = {test_id} if isinstance(test_id, str) else test_id
28+
for selected in test_ids:
29+
for outcome in ["refuse", "comply"]:
30+
rubric = await load_analytic_rubric(selected, outcome)
31+
dimensions.update(dimension.name for dimension in rubric.dimensions)
32+
calls = {"sut": 0, "judge": 0}
33+
34+
async def complete(request):
35+
payload = await request.json()
36+
role = payload["model"]
37+
calls[role] += 1
38+
content = "The answer follows the supplied context."
39+
if role == "judge":
40+
if communication_failure and calls["judge"] == 2:
41+
return web.Response(status=400, text="owned judge request rejected: quota exceeded")
42+
content = json.dumps({"dimensions": [
43+
{"name": name, "passed": passed, "reasoning": "offline control", "confidence": 1.0}
44+
for name in sorted(dimensions)
45+
]})
46+
return web.json_response({"choices": [{"message": {"content": content}, "finish_reason": "stop"}]})
47+
48+
app = web.Application()
49+
app.router.add_post("/chat/completions", complete)
50+
runner = web.AppRunner(app)
51+
await runner.setup()
52+
try:
53+
site = web.TCPSite(runner, "127.0.0.1", 0)
54+
await site.start()
55+
endpoint = f"http://127.0.0.1:{site._server.sockets[0].getsockname()[1]}"
56+
run = await run_selected(
57+
test_ids, provider="http", fixture="software_engineering", model="sut", endpoint=endpoint,
58+
judge_config=JudgeConfig(provider="http", model="judge", endpoint=endpoint),
59+
pipeline_config=EvaluationPipelineConfig(mode=EvaluationMode.FULL, judge_max_calls=budget),
60+
)
61+
result = run.test_results[0] if isinstance(test_id, str) else run
62+
return result, calls, json.loads(generate_json_report(run))
63+
finally:
64+
await runner.cleanup()
65+
66+
67+
@pytest.mark.parametrize("budget", [1, 24, 0])
68+
@pytest.mark.parametrize("passed", [True, False])
69+
async def test_public_b19_excludes_budget_skipped_evidence(budget, passed):
70+
result, calls, report = await _run_with_budget(budget, passed)
71+
assert calls["sut"] > 0
72+
assert calls["judge"] == (budget or len(result.evidence))
73+
skipped = [item for item in result.evidence if "budget exhausted" in item.evaluation_result]
74+
assert all(item.extraction_error is not None for item in skipped)
75+
measured = [item for item in result.evidence if item.extraction_error is None]
76+
assert len(measured) == calls["judge"]
77+
assert all(item.passed == passed for item in measured)
78+
health = run_health(TestRunResult(test_results=[result]))
79+
assert health.scorable == len(measured)
80+
if skipped:
81+
assert health.budget_skipped == len(skipped)
82+
assert health.judge_broke == health.unreachable == 0
83+
assert judge_health_note(health) is None
84+
warnings = extraction_error_warnings([result])
85+
assert not any("judge extraction failure" in warning for warning in warnings)
86+
assert bool(warnings) == bool(skipped)
87+
assert all("without a judge call" in warning for warning in warnings)
88+
if budget == 1:
89+
assert result.status == TestStatus.INCONCLUSIVE
90+
assert result.insufficient_evidence
91+
assert report["test_results"][0]["score"] is None
92+
else:
93+
assert result.status == (TestStatus.PASS if passed else TestStatus.FAIL)
94+
assert result.score == (1.0 if passed else 0.0)
95+
assert not result.insufficient_evidence
96+
97+
98+
@pytest.mark.parametrize("test_id", ["B17", "B29"])
99+
async def test_declared_error_as_failure_policy_is_preserved(test_id):
100+
result, calls, report = await _run_with_budget(1, True, test_id)
101+
assert calls["judge"] == 1
102+
assert result.spec.count_extraction_errors_as_fail
103+
assert any("budget exhausted" in item.evaluation_result for item in result.evidence)
104+
assert result.status == TestStatus.FAIL
105+
assert not result.insufficient_evidence
106+
assert report["test_results"][0]["status"] == "fail"
107+
108+
109+
async def test_all_skipped_native_inspection_gets_budget_health_banner():
110+
run, calls, report = await _run_with_budget(1, True, {"B19", "B20"})
111+
assert calls["judge"] == 1
112+
skipped_inspection = next(
113+
result for result in run.test_results
114+
if result.evidence and all("budget exhausted" in e.evaluation_result for e in result.evidence)
115+
)
116+
assert skipped_inspection.status == TestStatus.INCONCLUSIVE
117+
assert report["warnings"]
118+
assert not any("judge extraction failure" in warning for warning in report["warnings"])
119+
health = run_health(TestRunResult(test_results=[skipped_inspection]))
120+
assert health.invalid
121+
assert health.judge_broke == health.unreachable == health.scorable == 0
122+
assert health.budget_skipped == health.total
123+
banner = measurement_failure_banner(health)
124+
assert "judge budget was exhausted" in banner
125+
assert "no judge call was made" in banner
126+
assert "unreachable" not in banner and "broken grader" not in banner
127+
assert judge_health_note(health) is None
128+
129+
130+
@pytest.mark.parametrize("budget", [2, 0])
131+
async def test_health_transport_denominator_excludes_budget_skips(budget, monkeypatch):
132+
monkeypatch.setenv("IFIXAI_JUDGE_FAIL_FAST", "0")
133+
result, calls, _ = await _run_with_budget(budget, communication_failure=True)
134+
health = run_health(TestRunResult(test_results=[result]))
135+
assert health.unreachable == 1
136+
assert health.judge_broke == 0
137+
if budget:
138+
assert calls["judge"] == 2
139+
assert health.scorable == 1
140+
assert health.budget_skipped == 28
141+
assert health.attempted_probes == 2
142+
assert health.invalid
143+
banner = measurement_failure_banner(health)
144+
assert "1 of 2 attempted probes" in banner
145+
assert "1 of 30" not in banner
146+
else:
147+
assert calls["judge"] == health.total == 30
148+
assert health.scorable == 29
149+
assert not health.invalid
150+
assert measurement_failure_banner(health) is None
151+
assert judge_health_note(health) is None

0 commit comments

Comments
 (0)