Skip to content

Commit 0f63005

Browse files
committed
sync: RAG integrity, concurrency hardening, CLI polish from dev
Single squashed sync from ifixai-ai/diagnostic-dev to keep the public release tree current with internal development. Highlights - RAG context integrity: B28 inspection rewritten to test prompt-injection resistance through retrieved context, with structural typed cases. - Judge prompt isolation: SUT response moved out of the system prompt into a delimited user message to mitigate self-judging the response. - Rubric cache: lazy-init the asyncio lock so multi-loop test runs don't collide. - Concurrency governor: ramp waiters back up gradually after a 429 instead of releasing all parked coroutines in a thundering herd. - Category summary: stop rendering "✓ all passed" when zero tests were scored; show "— no scored tests" instead. - CLI polish: per-test folder layout, benchmark progress display, run summary terminology, PowerShell rendering fix. - Restore iMe Core branding modules (_branding.py, _imecore_prompt.py) and rewire run.py to use print_startup_banner and print_imecore_conclusion, plus the --quiet flag. Public-side intent from PR #2 preserved. - Docs: README repo-prep, methodology trim, drop internal spec IDs from public surface. Sync window: public/main (deb9ecb) → diagnostic-dev/main (293a62d), 81 non-merge commits.
1 parent deb9ecb commit 0f63005

34 files changed

Lines changed: 951 additions & 327 deletions

File tree

‎docs/methodology.md‎

Lines changed: 3 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -1,8 +1,8 @@
1-
# ifixai Methodology
1+
# iFixAi Methodology
22

3-
This page states, in one read, *how* ifixai scores an AI agent and *why* each choice is defensible. It exists so a reviewer does not have to reconstruct the rules from the code.
3+
This page states, in one read, *how* iFixAi scores an AI Agent or Deployment and *why* each choice is defensible. It exists so a reviewer does not have to reconstruct the rules from the code.
44

5-
ifixai is a diagnostic, not a certification. It runs 32 inspections against any agent and reports where the deployment's response behaviour differs from common governance expectations. It is useful for CI regression tracking, vendor comparisons under a controlled fixture, and pre-audit spot checks. It is not a substitute for domain-specific threat modelling or a formal safety argument.
5+
iFixAi is a diagnostic, not a certification. It runs 32 inspections against any agent and reports where the deployment's response behaviour differs from common governance expectations. It is useful for CI regression tracking, vendor comparisons under a controlled fixture, and pre-audit spot checks. It is not a substitute for domain-specific threat modelling or a formal safety argument.
66

77
## Evaluation paths
88

‎ifixai/cli/main.py‎

Lines changed: 19 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,5 @@
1+
import sys
2+
13
import click
24

35
from ifixai._version import VERSION
@@ -21,7 +23,24 @@ def ifixai_cli() -> None:
2123
ifixai_cli.add_command(compare)
2224

2325

26+
def _ensure_utf8_stdout() -> None:
27+
if sys.platform != "win32":
28+
return
29+
import io
30+
def _fix(stream): # noqa: E306
31+
enc = getattr(stream, "encoding", "") or ""
32+
if enc.lower().replace("-", "") == "utf8":
33+
return stream
34+
buf = getattr(stream, "buffer", None)
35+
if buf is None:
36+
return stream
37+
return io.TextIOWrapper(buf, encoding="utf-8", errors="replace")
38+
sys.stdout = _fix(sys.stdout)
39+
sys.stderr = _fix(sys.stderr)
40+
41+
2442
def main() -> None:
43+
_ensure_utf8_stdout()
2544
ifixai_cli()
2645

2746

‎ifixai/cli/orchestrator.py‎

Lines changed: 161 additions & 15 deletions
Original file line numberDiff line numberDiff line change
@@ -1,23 +1,27 @@
11
import os
22
import sys
3+
import threading
34
from collections import Counter
45

56
import click
67

78
from ifixai.api import run_inspections, run_single, run_strategic
89
from ifixai.core.concurrency import ConcurrencyGovernor
910
from ifixai.core.fixture_loader import load_fixture
11+
from ifixai.harness.registry import ALL_SPECS, SPEC_BY_ID
1012
from ifixai.judge.config import JudgeConfig, JudgeProviderSpec
1113
from ifixai.providers.resolver import (
1214
_PROVIDER_CREDENTIAL_ENV_VARS,
1315
detect_available_credentials,
1416
select_cross_provider_judge,
1517
)
18+
from ifixai.scoring.category_weights import STRATEGIC_TEST_IDS
1619
from ifixai.core.types import (
1720
TestResult,
1821
TestRunResult,
1922
EvaluationMethod,
2023
EvaluationPipelineConfig,
24+
InspectionCategory,
2125
)
2226

2327

@@ -185,25 +189,138 @@ def _print_insufficient_evidence_summary(result: TestRunResult) -> None:
185189
insufficient = [br for br in result.test_results if br.insufficient_evidence]
186190
total = len(result.test_results)
187191
if not insufficient:
188-
click.echo(click.style(f"0 out of {total} tests have failed.", fg="green"))
192+
click.echo(
193+
click.style(
194+
f"All {total} tests produced sufficient evidence to be scored.",
195+
fg="green",
196+
)
197+
)
189198
return
190199
inspection_ids = ", ".join(sorted(br.test_id for br in insufficient))
191200
click.echo(
192201
click.style(
193-
f"{len(insufficient)} out of {total} tests have failed "
194-
f"({inspection_ids}). Wrap your provider in a governance layer "
195-
f"or run with ≥2 provider credentials. See docs/methodology.md.",
202+
f"{len(insufficient)} out of {total} tests had insufficient evidence "
203+
f"to be scored ({inspection_ids}). The remaining tests were scored but "
204+
f"may still be below threshold -- see the per-category bars above. "
205+
f"Wrap your provider in a governance layer or run with ≥2 provider "
206+
f"credentials. See docs/methodology.md.",
196207
fg="yellow",
197208
)
198209
)
199210

200211

201-
def _progress_callback(
212+
_CATEGORY_BAR_COLOR: dict[str, str] = {
213+
InspectionCategory.FABRICATION.value: "\033[91m", # bright red
214+
InspectionCategory.MANIPULATION.value: "\033[91m", # bright red
215+
InspectionCategory.DECEPTION.value: "\033[95m", # bright magenta
216+
InspectionCategory.UNPREDICTABILITY.value: "\033[93m", # bright yellow
217+
InspectionCategory.OPACITY.value: "\033[94m", # bright blue
218+
}
219+
_RESET = "\033[0m"
220+
_RED = "\033[91m"
221+
_GREEN = "\033[92m"
222+
_YELLOW = "\033[93m"
223+
_DIM = "\033[2m"
224+
_BOLD = "\033[1m"
225+
_SPINNER_FRAMES = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏"]
226+
227+
228+
class BenchmarkProgressDisplay:
229+
"""Live animated display: pre-prints all benchmarks then updates in-place."""
230+
231+
def __init__(self, tests: list[tuple[str, str]]) -> None:
232+
self._tests = tests # [(test_id, name), ...]
233+
self._results: dict[str, TestResult] = {}
234+
self._frame_idx = 0
235+
self._lock = threading.Lock()
236+
self._done = threading.Event()
237+
self._thread: threading.Thread | None = None
238+
239+
def start(self) -> None:
240+
for test_id, name in self._tests:
241+
sys.stdout.write(f" {_YELLOW}⠋{_RESET} {_DIM}{test_id}{_RESET} {name}\n")
242+
sys.stdout.flush()
243+
self._thread = threading.Thread(target=self._animate, daemon=True)
244+
self._thread.start()
245+
246+
def update(self, test_id: str, index: int, total: int, result: TestResult) -> None:
247+
with self._lock:
248+
self._results[test_id] = result
249+
250+
def stop(self) -> None:
251+
self._done.set()
252+
if self._thread:
253+
self._thread.join(timeout=2.0)
254+
self._redraw(final=True)
255+
256+
def _animate(self) -> None:
257+
while not self._done.wait(timeout=0.1):
258+
self._frame_idx = (self._frame_idx + 1) % len(_SPINNER_FRAMES)
259+
self._redraw()
260+
261+
def _build_lines(self, final: bool = False) -> list[str]:
262+
lines: list[str] = []
263+
with self._lock:
264+
frame = _SPINNER_FRAMES[self._frame_idx]
265+
for test_id, name in self._tests:
266+
if test_id in self._results:
267+
result = self._results[test_id]
268+
if result.passing:
269+
icon = f"{_GREEN}✓{_RESET}"
270+
status = f"{_GREEN}PASS{_RESET}"
271+
else:
272+
icon = f"{_RED}✗{_RESET}"
273+
status = f"{_RED}FAIL{_RESET}"
274+
lines.append(
275+
f" {icon} {_BOLD}{test_id}{_RESET} {name} "
276+
f"... {status} ({result.score:.0%})"
277+
)
278+
else:
279+
spinner = "·" if final else frame
280+
lines.append(
281+
f" {_YELLOW}{spinner}{_RESET} {_DIM}{test_id}{_RESET} {name}"
282+
)
283+
return lines
284+
285+
def _redraw(self, final: bool = False) -> None:
286+
n = len(self._tests)
287+
if n == 0:
288+
return
289+
lines = self._build_lines(final=final)
290+
sys.stdout.write(f"\033[{n}A")
291+
for line in lines:
292+
sys.stdout.write(f"\r\033[K{line}\n")
293+
sys.stdout.flush()
294+
295+
296+
def _print_category_summary(result: TestRunResult) -> None:
297+
if not result.category_scores:
298+
return
299+
bar_width = 22
300+
click.echo()
301+
for cs in result.category_scores:
302+
color = _CATEGORY_BAR_COLOR.get(cs.category.value, "\033[96m")
303+
bar = color + ("█" * bar_width) + _RESET
304+
failed = cs.test_count - cs.tests_passed
305+
count_str = f"{cs.test_count}/{cs.test_count}"
306+
if cs.test_count == 0:
307+
fail_str = f"{_DIM}— no scored tests{_RESET}"
308+
elif failed > 0:
309+
fail_str = f"{_RED}× {failed} failed{_RESET}"
310+
else:
311+
fail_str = f"{_GREEN}✓ all passed{_RESET}"
312+
name = cs.category.value.ljust(16)
313+
click.echo(f" {name} [{bar}] {count_str:>5} {fail_str}")
314+
click.echo()
315+
316+
317+
def _progress_callback_plain(
202318
bid: str,
203319
index: int,
204320
total: int,
205321
bench_result: TestResult,
206322
) -> None:
323+
"""Fallback used when stdout is not a TTY (e.g. piped/redirected)."""
207324
status_label = (
208325
click.style("PASS", fg="green")
209326
if bench_result.passing
@@ -215,6 +332,20 @@ def _progress_callback(
215332
)
216333

217334

335+
def _build_display_tests(
336+
strategic: bool,
337+
test_id: str | None,
338+
) -> list[tuple[str, str]]:
339+
if test_id:
340+
uid = test_id.upper()
341+
spec = SPEC_BY_ID.get(uid)
342+
return [(uid, spec.name if spec else uid)] # type: ignore[union-attr]
343+
if strategic:
344+
strategic_set = set(STRATEGIC_TEST_IDS)
345+
return [(s.test_id, s.name) for s in ALL_SPECS if s.test_id in strategic_set]
346+
return [(s.test_id, s.name) for s in ALL_SPECS]
347+
348+
218349
async def execute_tests(
219350
provider: str,
220351
api_key: str,
@@ -242,7 +373,15 @@ async def execute_tests(
242373
click.echo(click.style(f"Fixture error: {exc}", fg="red"))
243374
return None
244375

245-
effective_callback = progress_callback or _progress_callback
376+
use_display = progress_callback is None and sys.stdout.isatty()
377+
display: BenchmarkProgressDisplay | None = None
378+
379+
if use_display:
380+
display = BenchmarkProgressDisplay(_build_display_tests(strategic, test_id))
381+
display.start()
382+
effective_callback = display.update
383+
else:
384+
effective_callback = progress_callback or _progress_callback_plain
246385

247386
try:
248387
if test_id:
@@ -260,15 +399,18 @@ async def execute_tests(
260399
sut_temperature=sut_temperature,
261400
sut_seed=sut_seed,
262401
)
263-
status_label = (
264-
click.style("PASS", fg="green")
265-
if single_result.passing
266-
else click.style("FAIL", fg="red")
267-
)
268-
click.echo(
269-
f" [1/1] {test_id} {single_result.name} ... "
270-
f"{status_label} ({single_result.score:.0%})"
271-
)
402+
if display:
403+
display.update(test_id, 1, 1, single_result)
404+
else:
405+
status_label = (
406+
click.style("PASS", fg="green")
407+
if single_result.passing
408+
else click.style("FAIL", fg="red")
409+
)
410+
click.echo(
411+
f" [1/1] {test_id} {single_result.name} ... "
412+
f"{status_label} ({single_result.score:.0%})"
413+
)
272414
return TestRunResult(
273415
system_name=system_name,
274416
system_version=system_version,
@@ -324,3 +466,7 @@ async def execute_tests(
324466
except Exception as exc:
325467
click.echo(click.style(f"Test execution failed: {exc}", fg="red"))
326468
return None
469+
470+
finally:
471+
if display:
472+
display.stop()

‎ifixai/cli/reports.py‎

Lines changed: 4 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -28,12 +28,14 @@ def save_reports(
2828
fixture_slug = _slugify(result.fixture_name)
2929
base_name = f"ifixai-{system_slug}-{fixture_slug}"
3030

31+
click.echo(click.style("Access your Full Report here:", bold=True))
32+
3133
if report_format in ("json", "both"):
3234
json_path = out_path / f"{base_name}.json"
3335
json_path.write_text(generate_json_report(result), encoding="utf-8")
34-
click.echo(f" JSON report: {json_path}")
36+
click.echo(f" JSON report: {json_path}")
3537

3638
if report_format in ("markdown", "both"):
3739
md_path = out_path / f"{base_name}.md"
3840
md_path.write_text(generate_markdown_report(result), encoding="utf-8")
39-
click.echo(f" Markdown report: {md_path}")
41+
click.echo(f" Markdown report: {md_path}")

0 commit comments

Comments
 (0)