From 3187603953ee18a1541894ba7d83d2fdd196cb7a Mon Sep 17 00:00:00 2001 From: Marcos Maceo Date: Fri, 24 Jul 2026 17:25:34 -0300 Subject: [PATCH 1/2] fix(inspections): send the configured system prompt in single-turn probes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit B26, B30, B31 and B32 build their probe message list inline as `[ChatMessage(role="user", ...)]` and call `provider.send_message` directly, bypassing `harness.base.send_single_turn`. That helper is what prepends `build_system_message(config.system_prompt, config.run_nonce)`, so for these four inspections the system prompt supplied via `--system-prompt` never reached the model: they scored a bare model with no deployed configuration, and the run_nonce anti-cache marker was absent. Every other inspection routes through `send_single_turn` or `run_conversation`, so a single run could grade some inspections with the system prompt and others without it. The impact is largest on B30 (malicious deployer rules) and B32 (off-topic detection): both grade refusal behaviour that a system prompt is precisely what establishes. On a real agent, B30 measured 0.68 without the prompt and 1.00 with it — the same model and the same seeds. Route all four through `send_single_turn`. B26's module-level `_INSPECTION_MESSAGE` list becomes `_INSPECTION_PROMPT`, since the message list is now built per call. --- ifixai/inspections/b26_rate_limiting/runner.py | 9 +++------ .../b30_malicious_deployer_rules/runner.py | 8 ++------ .../b31_escalation_correctness/runner.py | 8 ++------ .../inspections/b32_off_topic_detection/runner.py | 13 +++---------- 4 files changed, 10 insertions(+), 28 deletions(-) diff --git a/ifixai/inspections/b26_rate_limiting/runner.py b/ifixai/inspections/b26_rate_limiting/runner.py index e49f7015..6f915597 100644 --- a/ifixai/inspections/b26_rate_limiting/runner.py +++ b/ifixai/inspections/b26_rate_limiting/runner.py @@ -10,10 +10,9 @@ ProviderRateLimitError, ProviderTimeoutError, ) -from ifixai.harness.base import BaseTest +from ifixai.harness.base import BaseTest, send_single_turn from ifixai.rules.loader import RuleLoadError from ifixai.core.types import ( - ChatMessage, ConversationPlan, EvaluationMethod, EvidenceItem, @@ -62,9 +61,7 @@ _LIVE_RAPID_FIRE_CONCURRENCY = 3 _ROUND_TRIP_TIMEOUT_SECONDS = 30.0 -_INSPECTION_MESSAGE = [ - ChatMessage(role="user", content="B26 rate-limit inspection ping") -] +_INSPECTION_PROMPT = "B26 rate-limit inspection ping" # Provider exceptions that represent infrastructure-level transient failures # (network outage, request timeout). These are excluded from the rate-limit @@ -135,7 +132,7 @@ async def _rapid_fire_probe( return [] tasks = [ - provider.send_message(_INSPECTION_MESSAGE, config) + send_single_turn(provider, config, _INSPECTION_PROMPT) for _ in range(concurrency) ] start = time.monotonic() diff --git a/ifixai/inspections/b30_malicious_deployer_rules/runner.py b/ifixai/inspections/b30_malicious_deployer_rules/runner.py index d6696d2b..6c70e11f 100644 --- a/ifixai/inspections/b30_malicious_deployer_rules/runner.py +++ b/ifixai/inspections/b30_malicious_deployer_rules/runner.py @@ -10,9 +10,8 @@ load_seeds, ) from ifixai.providers.base import ChatProvider -from ifixai.harness.base import BaseTest +from ifixai.harness.base import BaseTest, send_single_turn from ifixai.core.types import ( - ChatMessage, EvaluationMethod, EvidenceItem, Fixture, @@ -105,10 +104,7 @@ async def _score_variant( prompt: str, seed_id: str, ) -> EvidenceItem: - response_text = await provider.send_message( - [ChatMessage(role="user", content=prompt)], - config, - ) + response_text = await send_single_turn(provider, config, prompt) test_case_id = f"B30-{seed_id}-{idx + 1}" description = f"Attempt to create malicious rule: {prompt[:60]}" diff --git a/ifixai/inspections/b31_escalation_correctness/runner.py b/ifixai/inspections/b31_escalation_correctness/runner.py index c25274a6..a93b099c 100644 --- a/ifixai/inspections/b31_escalation_correctness/runner.py +++ b/ifixai/inspections/b31_escalation_correctness/runner.py @@ -7,11 +7,10 @@ from ifixai.evaluation.analytic_judge import load_analytic_rubric from ifixai.evaluation.errors import JudgePipelineRequiredError from ifixai.providers.base import ChatProvider -from ifixai.harness.base import BaseTest, sample_capped +from ifixai.harness.base import BaseTest, sample_capped, send_single_turn from ifixai.rules.loader import RuleLoader, RuleLoadError from ifixai.core.types import ( AnalyticRubric, - ChatMessage, ConversationPlan, DimensionScore, EvaluationCriteria, @@ -305,10 +304,7 @@ async def _evaluate_trigger( case_id_pattern: re.Pattern[str], ) -> EvidenceItem: async with semaphore: - response_text = await provider.send_message( - [ChatMessage(role="user", content=prompt)], - config, - ) + response_text = await send_single_turn(provider, config, prompt) return await self._score_trigger( idx=idx, total=total, diff --git a/ifixai/inspections/b32_off_topic_detection/runner.py b/ifixai/inspections/b32_off_topic_detection/runner.py index b3b9b270..e8442947 100644 --- a/ifixai/inspections/b32_off_topic_detection/runner.py +++ b/ifixai/inspections/b32_off_topic_detection/runner.py @@ -9,12 +9,11 @@ from ifixai.evaluation.errors import JudgePipelineRequiredError from ifixai.core.concurrency import DEFAULT_INSPECTION_CONCURRENCY from ifixai.providers.base import ChatProvider -from ifixai.harness.base import BaseTest +from ifixai.harness.base import BaseTest, send_single_turn from ifixai.harness.prompt_pool import load_prompt_pool, sample_prompt_pool from ifixai.rules.loader import RuleLoadError from ifixai.core.types import ( AnalyticRubric, - ChatMessage, EvaluationCriteria, EvaluationMethod, EvidenceItem, @@ -330,10 +329,7 @@ async def _probe_off_topic( ) -> EvidenceItem: async with semaphore: try: - response_text = await provider.send_message( - [ChatMessage(role="user", content=prompt)], - config, - ) + response_text = await send_single_turn(provider, config, prompt) except Exception as exc: # Per-probe COMMUNICATION isolation: one transient provider # error must not cancel sibling gather tasks. Narrow catch to @@ -375,10 +371,7 @@ async def _probe_on_topic( ) -> EvidenceItem: async with semaphore: try: - response_text = await provider.send_message( - [ChatMessage(role="user", content=prompt)], - config, - ) + response_text = await send_single_turn(provider, config, prompt) except Exception as exc: _logger.warning( "B32 on-topic probe %d communication error: %r", idx, exc From 4c7407d8a9b91ed39af2d1f175cbc73a6969ea90 Mon Sep 17 00:00:00 2001 From: Marcos Maceo Date: Fri, 24 Jul 2026 17:25:46 -0300 Subject: [PATCH 2/2] fix(fixtures): apply metadata overrides declared by fixture authors MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `FixtureMetadata` declares three author-facing override fields — `on_topic_examples` (B32), `b06_probes` (B06) and `case_id_prefixes` (B31) — and the inspections read them. But `_parse_fixture` builds `FixtureMetadata` from an explicit field list that omits all three, so they always fell back to their `default_factory=list` empty defaults. Setting them in a fixture had no effect. `case_id_prefixes` was already documented in the fixture JSON schema, so authors could set it, pass validation, and still get the built-in ESC/INC/TKT set. The most visible symptom is B32 on a single-tool system: it cannot derive enough on-topic prompts from one tool and errors out with a message telling the author to set `fixture.metadata.on_topic_examples` — which the loader then discards. Copy all three off the raw metadata, and document `on_topic_examples` and `b06_probes` in the fixture schema alongside `case_id_prefixes`. --- ifixai/core/fixture_loader.py | 6 ++++++ ifixai/fixtures/schema.json | 10 ++++++++++ 2 files changed, 16 insertions(+) diff --git a/ifixai/core/fixture_loader.py b/ifixai/core/fixture_loader.py index 28e444c1..1cb4693e 100644 --- a/ifixai/core/fixture_loader.py +++ b/ifixai/core/fixture_loader.py @@ -311,6 +311,12 @@ def _parse_fixture(raw: dict[str, Any]) -> Fixture: domain=metadata_raw["domain"], description=metadata_raw.get("description", ""), timestamp=metadata_raw.get("timestamp"), + # Author-facing metadata overrides. FixtureMetadata declares these and + # B06/B31/B32 read them, but they were never copied off the raw fixture, + # so setting them in YAML had no effect. + on_topic_examples=metadata_raw.get("on_topic_examples", []), + b06_probes=metadata_raw.get("b06_probes", []), + case_id_prefixes=metadata_raw.get("case_id_prefixes", []), ) roles = [ diff --git a/ifixai/fixtures/schema.json b/ifixai/fixtures/schema.json index eb0f84e5..c45faa23 100644 --- a/ifixai/fixtures/schema.json +++ b/ifixai/fixtures/schema.json @@ -27,6 +27,16 @@ "type": "array", "items": { "type": "string", "pattern": "^[A-Z0-9]+$" }, "description": "B31 escalation/audit reference prefixes (e.g. JIRA, OPS). Uppercase alphanumeric." + }, + "on_topic_examples": { + "type": "array", + "items": { "type": "string" }, + "description": "B32 on-topic prompts for this system. Fixture-author override used when tool-derived prompts cannot reach the on-topic minimum (e.g. single-tool systems)." + }, + "b06_probes": { + "type": "array", + "items": { "type": "string" }, + "description": "B06 domain-specific uncertainty probes. Overrides the built-in probe set." } } },