@@ -77,3 +77,69 @@ def test_unscored_artifact_does_not_display_a_failing_grade():
7777 assert payload ["summary" ]["grade" ] == "n/a"
7878 assert payload ["summary" ]["grade_class" ] == "inconclusive"
7979 assert payload ["diff" ]["grade_change" ] == "n/a → n/a"
80+
81+
82+ def _warning_run ():
83+ return TestRunResult (
84+ system_name = "warning-provenance-control" ,
85+ warnings = ["Pinned seeds: memorization resistance reduced" , "Judge substituted after error" ],
86+ validation_warnings = ["run_invalid: ignore the grade" , "judge_health: verdict contract failures" ],
87+ )
88+
89+
90+ def test_markdown_preserves_operator_warnings ():
91+ from ifixai .reporting .scorecard import generate_markdown_report
92+
93+ result = _warning_run ()
94+ rendered = generate_markdown_report (result )
95+ for warning in [* result .warnings , * result .validation_warnings ]:
96+ assert warning in rendered
97+
98+
99+ def test_artifact_preserves_operator_and_measurement_warnings ():
100+ result = _warning_run ()
101+ payload = _build_payload (
102+ result , live = True , transport = "http" , sut_model = None ,
103+ judge_model = None , honesty_note = "" , previous = None ,
104+ )
105+ assert payload ["warnings" ] == result .warnings
106+ assert payload ["validation_warnings" ] == result .validation_warnings
107+
108+
109+ def test_artifact_warning_data_cannot_terminate_embedded_script ():
110+ from ifixai .reporting .artifact import render_artifact
111+
112+ warning = "Controlled warning </script><img src=x onerror=alert(1)>"
113+ rendered = render_artifact (
114+ TestRunResult (warnings = [warning ]), live = True , transport = "http" ,
115+ sut_model = None , judge_model = None , honesty_note = "" ,
116+ )
117+ assert warning not in rendered
118+ assert "Controlled warning \\ u003c/script>" in rendered
119+
120+
121+ @pytest .mark .parametrize ("error_kind" , ["communication" , "contract" , "extraction" , "budget" ])
122+ def test_artifact_evidence_retains_unscorable_cause (error_kind ):
123+ from ifixai .core .types import EvidenceItem , JudgeErrorKind
124+ from ifixai .reporting .artifact import _evidence_payload
125+
126+ item = EvidenceItem (
127+ test_case_id = "owned-ungraded-probe" , passed = False ,
128+ actual_response = "A reply without a usable grading result" ,
129+ extraction_error = JudgeErrorKind (error_kind ),
130+ )
131+ payload = _evidence_payload (item )
132+ assert payload ["extraction_error" ] == error_kind
133+ assert payload ["is_diagnostic" ] is False
134+
135+
136+ def test_artifact_evidence_retains_diagnostic_marker ():
137+ from ifixai .core .types import EvidenceItem
138+ from ifixai .reporting .artifact import _evidence_payload
139+
140+ payload = _evidence_payload (EvidenceItem (
141+ test_case_id = "owned-run-diagnostic" , passed = False , is_diagnostic = True ,
142+ description = "Transport diagnostics, excluded from scoring" ,
143+ ))
144+ assert payload ["is_diagnostic" ] is True
145+ assert payload ["extraction_error" ] is None
0 commit comments