Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,14 @@
<a href="CONTRIBUTING.md">Contributing</a>
</p>

<p align="center">
<a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue.svg" alt="license: Apache 2.0" /></a>
<a href="pyproject.toml"><img src="https://img.shields.io/badge/python-3.10%2B-blue.svg" alt="python 3.10+" /></a>
<a href="https://github.com/ifixai-ai/diagnostic/actions/workflows/ci.yml"><img src="https://github.com/ifixai-ai/diagnostic/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
<img src="https://img.shields.io/badge/inspections-32-orange.svg" alt="32 inspections" />
<a href="https://github.com/ifixai-ai/diagnostic/issues?q=is%3Aopen+label%3A%22good+first+issue%22"><img src="https://img.shields.io/github/issues/ifixai-ai/diagnostic/good%20first%20issue?label=good%20first%20issues&color=7057ff" alt="good first issues" /></a>
</p>

---

iFixAi runs up to 32 inspections against any AI agent and reports where its
Expand Down
6 changes: 3 additions & 3 deletions docs/methodology.md
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
# ifixai Methodology
# iFixAi Methodology

This page states, in one read, *how* ifixai scores an AI agent and *why* each choice is defensible. It exists so a reviewer does not have to reconstruct the rules from the code.
This page states, in one read, *how* iFixAi scores an AI Agent or Deployment and *why* each choice is defensible. It exists so a reviewer does not have to reconstruct the rules from the code.

ifixai is a diagnostic, not a certification. It runs 32 inspections against any agent and reports where the deployment's response behaviour differs from common governance expectations. It is useful for CI regression tracking, vendor comparisons under a controlled fixture, and pre-audit spot checks. It is not a substitute for domain-specific threat modelling or a formal safety argument.
iFixAi is a diagnostic, not a certification. It runs 32 inspections against any agent and reports where the deployment's response behaviour differs from common governance expectations. It is useful for CI regression tracking, vendor comparisons under a controlled fixture, and pre-audit spot checks. It is not a substitute for domain-specific threat modelling or a formal safety argument.

## Evaluation paths

Expand Down
19 changes: 19 additions & 0 deletions ifixai/cli/main.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,5 @@
import sys

import click

from ifixai._version import VERSION
Expand All @@ -21,7 +23,24 @@ def ifixai_cli() -> None:
ifixai_cli.add_command(compare)


def _ensure_utf8_stdout() -> None:
if sys.platform != "win32":
return
import io
def _fix(stream): # noqa: E306
enc = getattr(stream, "encoding", "") or ""
if enc.lower().replace("-", "") == "utf8":
return stream
buf = getattr(stream, "buffer", None)
if buf is None:
return stream
return io.TextIOWrapper(buf, encoding="utf-8", errors="replace")
sys.stdout = _fix(sys.stdout)
sys.stderr = _fix(sys.stderr)


def main() -> None:
_ensure_utf8_stdout()
ifixai_cli()


Expand Down
238 changes: 201 additions & 37 deletions ifixai/cli/orchestrator.py
Original file line number Diff line number Diff line change
@@ -1,23 +1,27 @@
import os
import sys
import threading
from collections import Counter

import click

from ifixai.api import run_inspections, run_single, run_strategic
from ifixai.core.concurrency import ConcurrencyGovernor
from ifixai.core.fixture_loader import load_fixture
from ifixai.harness.registry import ALL_SPECS, SPEC_BY_ID
from ifixai.judge.config import JudgeConfig, JudgeProviderSpec
from ifixai.providers.resolver import (
_PROVIDER_CREDENTIAL_ENV_VARS,
detect_available_credentials,
select_cross_provider_judge,
)
from ifixai.scoring.category_weights import STRATEGIC_TEST_IDS
from ifixai.core.types import (
TestResult,
TestRunResult,
EvaluationMethod,
EvaluationPipelineConfig,
InspectionCategory,
TestStatus,
)


Expand Down Expand Up @@ -157,53 +161,184 @@ def _build_judge_config(


def _print_inconclusive_summary(result: TestRunResult) -> None:
inconclusive_count = 0
inconclusive_tests: set[str] = set()
by_category: Counter[str] = Counter()

for br in result.test_results:
for evidence in br.evidence:
if evidence.evaluation_method == EvaluationMethod.JUDGE:
inconclusive_count += 1
inconclusive_tests.add(br.test_id)
by_category[br.category.value] += 1

if inconclusive_count == 0:
click.echo(click.style("Inconclusive: 0 evidence items.", fg="green"))
else:
breakdown = ", ".join(f"{cat}={n}" for cat, n in by_category.most_common())
click.echo(
click.style(
f"Inconclusive: {inconclusive_count} evidence items across "
f"{len(inconclusive_tests)} tests ({breakdown})",
fg="yellow",
)
inconclusive = [
br for br in result.test_results
if br.status == TestStatus.INCONCLUSIVE
]

if not inconclusive:
click.echo(click.style("Inconclusive: 0 tests.", fg="green"))
return

by_category: Counter[str] = Counter(br.category.value for br in inconclusive)
breakdown = ", ".join(f"{cat}={n}" for cat, n in by_category.most_common())
click.echo(
click.style(
f"Inconclusive: {len(inconclusive)} tests ({breakdown})",
fg="yellow",
)
)


def _print_insufficient_evidence_summary(result: TestRunResult) -> None:
insufficient = [br for br in result.test_results if br.insufficient_evidence]
total = len(result.test_results)
if not insufficient:
click.echo(click.style(f"0 out of {total} tests have failed.", fg="green"))
click.echo(
click.style(
f"All {total} tests produced sufficient evidence to be scored.",
fg="green",
)
)
return
inspection_ids = ", ".join(sorted(br.test_id for br in insufficient))
click.echo(
click.style(
f"{len(insufficient)} out of {total} tests have failed "
f"({inspection_ids}). Wrap your provider in a governance layer "
f"or run with ≥2 provider credentials. See docs/methodology.md.",
f"{len(insufficient)} out of {total} tests had insufficient evidence "
f"to be scored ({inspection_ids}). The remaining tests were scored but "
f"may still be below threshold -- see the per-category bars above. "
f"Wrap your provider in a governance layer or run with ≥2 provider "
f"credentials. See docs/methodology.md.",
fg="yellow",
)
)


def _progress_callback(
_CATEGORY_BAR_COLOR: dict[str, str] = {
InspectionCategory.FABRICATION.value: "\033[38;5;208m", # orange
InspectionCategory.MANIPULATION.value: "\033[93m", # yellow
InspectionCategory.DECEPTION.value: "\033[92m", # green
InspectionCategory.UNPREDICTABILITY.value: "\033[94m", # blue
InspectionCategory.OPACITY.value: "\033[38;5;213m", # pink
}
_RESET = "\033[0m"
_RED = "\033[91m"
_GREEN = "\033[92m"
_YELLOW = "\033[93m"
_DIM = "\033[2m"
_BOLD = "\033[1m"
_SPINNER_FRAMES = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏"]


class BenchmarkProgressDisplay:
"""Live animated display: pre-prints all benchmarks then updates in-place."""

def __init__(self, tests: list[tuple[str, str]]) -> None:
self._tests = tests # [(test_id, name), ...]
self._results: dict[str, TestResult] = {}
self._frame_idx = 0
self._lock = threading.Lock()
self._done = threading.Event()
self._thread: threading.Thread | None = None

def start(self) -> None:
for test_id, name in self._tests:
sys.stdout.write(f" {_YELLOW}⠋{_RESET} {_DIM}{test_id}{_RESET} {name}\n")
sys.stdout.flush()
self._thread = threading.Thread(target=self._animate, daemon=True)
self._thread.start()

def update(self, test_id: str, index: int, total: int, result: TestResult) -> None:
with self._lock:
self._results[test_id] = result

def stop(self) -> None:
self._done.set()
if self._thread:
self._thread.join(timeout=2.0)
self._redraw(final=True)

def _animate(self) -> None:
while not self._done.wait(timeout=0.1):
self._frame_idx = (self._frame_idx + 1) % len(_SPINNER_FRAMES)
self._redraw()

def _build_lines(self, final: bool = False) -> list[str]:
lines: list[str] = []
with self._lock:
frame = _SPINNER_FRAMES[self._frame_idx]
for test_id, name in self._tests:
if test_id in self._results:
result = self._results[test_id]
if result.insufficient_evidence:
icon = f"{_YELLOW}⊘{_RESET}"
status = f"{_YELLOW}INCONCLUSIVE{_RESET}"
lines.append(
f" {icon} {_BOLD}{test_id}{_RESET} {name} "
f"... {status} (insufficient evidence)"
)
continue
if result.passing:
icon = f"{_GREEN}✓{_RESET}"
status = f"{_GREEN}PASS{_RESET}"
else:
icon = f"{_RED}✗{_RESET}"
status = f"{_RED}FAIL{_RESET}"
lines.append(
f" {icon} {_BOLD}{test_id}{_RESET} {name} "
f"... {status} ({result.score:.0%})"
)
else:
spinner = "·" if final else frame
lines.append(
f" {_YELLOW}{spinner}{_RESET} {_DIM}{test_id}{_RESET} {name}"
)
return lines

def _redraw(self, final: bool = False) -> None:
n = len(self._tests)
if n == 0:
return
lines = self._build_lines(final=final)
sys.stdout.write(f"\033[{n}A")
for line in lines:
sys.stdout.write(f"\r\033[K{line}\n")
sys.stdout.flush()


def _print_category_summary(result: TestRunResult) -> None:
if not result.category_scores:
return
bar_width = 22
click.echo()
for cs in result.category_scores:
color = _CATEGORY_BAR_COLOR.get(cs.category.value, "\033[96m")
bar = color + ("█" * bar_width) + _RESET
total_in_suite = len(cs.test_ids)
failed = cs.test_count - cs.tests_passed
inconclusive = total_in_suite - cs.test_count
count_str = f"{cs.test_count}/{total_in_suite}"
if total_in_suite == 0:
count_str = "—"
fail_str = f"{_DIM}not in this suite{_RESET}"
elif cs.test_count == 0:
fail_str = f"{_YELLOW}⊘ {inconclusive} inconclusive{_RESET}"
elif failed > 0 and inconclusive > 0:
fail_str = f"{_RED}× {failed} failed{_RESET}, {_YELLOW}⊘ {inconclusive} inconclusive{_RESET}"
elif failed > 0:
fail_str = f"{_RED}× {failed} failed{_RESET}"
elif inconclusive > 0:
fail_str = f"{_GREEN}✓ {cs.tests_passed} passed{_RESET}, {_YELLOW}⊘ {inconclusive} inconclusive{_RESET}"
else:
fail_str = f"{_GREEN}✓ all passed{_RESET}"
name = cs.category.value.ljust(16)
click.echo(f" {name} [{bar}] {count_str:>5} {fail_str}")
click.echo()


def _progress_callback_plain(
bid: str,
index: int,
total: int,
bench_result: TestResult,
) -> None:
"""Fallback used when stdout is not a TTY (e.g. piped/redirected)."""
if bench_result.insufficient_evidence:
click.echo(
f" [{index}/{total}] {bid} {bench_result.name} ... "
f"{click.style('INCONCLUSIVE', fg='yellow')} (insufficient evidence)"
)
return
status_label = (
click.style("PASS", fg="green")
if bench_result.passing
Expand All @@ -215,6 +350,20 @@ def _progress_callback(
)


def _build_display_tests(
strategic: bool,
test_id: str | None,
) -> list[tuple[str, str]]:
if test_id:
uid = test_id.upper()
spec = SPEC_BY_ID.get(uid)
return [(uid, spec.name if spec else uid)] # type: ignore[union-attr]
if strategic:
strategic_set = set(STRATEGIC_TEST_IDS)
return [(s.test_id, s.name) for s in ALL_SPECS if s.test_id in strategic_set]
return [(s.test_id, s.name) for s in ALL_SPECS]


async def execute_tests(
provider: str,
api_key: str,
Expand Down Expand Up @@ -242,7 +391,15 @@ async def execute_tests(
click.echo(click.style(f"Fixture error: {exc}", fg="red"))
return None

effective_callback = progress_callback or _progress_callback
use_display = progress_callback is None and sys.stdout.isatty()
display: BenchmarkProgressDisplay | None = None

if use_display:
display = BenchmarkProgressDisplay(_build_display_tests(strategic, test_id))
display.start()
effective_callback = display.update
else:
effective_callback = progress_callback or _progress_callback_plain

try:
if test_id:
Expand All @@ -260,15 +417,18 @@ async def execute_tests(
sut_temperature=sut_temperature,
sut_seed=sut_seed,
)
status_label = (
click.style("PASS", fg="green")
if single_result.passing
else click.style("FAIL", fg="red")
)
click.echo(
f" [1/1] {test_id} {single_result.name} ... "
f"{status_label} ({single_result.score:.0%})"
)
if display:
display.update(test_id, 1, 1, single_result)
else:
status_label = (
click.style("PASS", fg="green")
if single_result.passing
else click.style("FAIL", fg="red")
)
click.echo(
f" [1/1] {test_id} {single_result.name} ... "
f"{status_label} ({single_result.score:.0%})"
)
return TestRunResult(
system_name=system_name,
system_version=system_version,
Expand Down Expand Up @@ -324,3 +484,7 @@ async def execute_tests(
except Exception as exc:
click.echo(click.style(f"Test execution failed: {exc}", fg="red"))
return None

finally:
if display:
display.stop()
6 changes: 4 additions & 2 deletions ifixai/cli/reports.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,12 +28,14 @@ def save_reports(
fixture_slug = _slugify(result.fixture_name)
base_name = f"ifixai-{system_slug}-{fixture_slug}"

click.echo(click.style("Access your Full Report here:", bold=True))

if report_format in ("json", "both"):
json_path = out_path / f"{base_name}.json"
json_path.write_text(generate_json_report(result), encoding="utf-8")
click.echo(f" JSON report: {json_path}")
click.echo(f" JSON report: {json_path}")

if report_format in ("markdown", "both"):
md_path = out_path / f"{base_name}.md"
md_path.write_text(generate_markdown_report(result), encoding="utf-8")
click.echo(f" Markdown report: {md_path}")
click.echo(f" Markdown report: {md_path}")
Loading
Loading