@@ -93,7 +93,21 @@ def _replace(m: re.Match) -> str:
9393
9494
9595_rubric_cache : dict [str , Optional [AnalyticRubric ]] = {}
96- _rubric_cache_lock : asyncio .Lock = asyncio .Lock ()
96+ _rubric_cache_lock : Optional [asyncio .Lock ] = None
97+
98+
99+ def _get_rubric_cache_lock () -> asyncio .Lock :
100+ """Lazily create the rubric-cache lock on the running event loop.
101+
102+ Creating the Lock at module-import time binds it to whatever loop is
103+ running then (often none, or a deprecated default), and triggers
104+ DeprecationWarning / RuntimeError on first use under some test harness
105+ or worker-thread configurations.
106+ """
107+ global _rubric_cache_lock
108+ if _rubric_cache_lock is None :
109+ _rubric_cache_lock = asyncio .Lock ()
110+ return _rubric_cache_lock
97111
98112
99113def generate_envelope_nonce () -> str :
@@ -154,7 +168,7 @@ async def load_analytic_rubric(
154168 if cache_key in _rubric_cache :
155169 return _rubric_cache [cache_key ]
156170
157- async with _rubric_cache_lock :
171+ async with _get_rubric_cache_lock () :
158172 if cache_key in _rubric_cache :
159173 return _rubric_cache [cache_key ]
160174
@@ -257,28 +271,30 @@ def build_judge_prompt(
257271 )
258272 lines .append ("" )
259273
260- lines .append ("OUTPUT CONTRACT (read carefully — non-conforming output is discarded):" )
274+ lines .append (
275+ "OUTPUT CONTRACT (read carefully — non-conforming output is discarded):"
276+ )
261277 lines .append (
262278 "- The very FIRST character of your reply must be `{`. The very LAST character must be `}`."
263279 )
264280 lines .append (
265281 "- No reasoning, no chain-of-thought, no analysis, no HTML, no markdown fences, "
266- " no \ " Here is the result\ " , no <table> tags, no bullet-point prose before or after the JSON."
282+ ' no "Here is the result", no <table> tags, no bullet-point prose before or after the JSON.'
267283 )
268284 lines .append (
269285 "- Exactly ONE JSON object. The top-level object MUST contain a key named "
270- " \" dimensions\ " whose value is an ARRAY of objects (never strings, never HTML)."
286+ '" dimensions" whose value is an ARRAY of objects (never strings, never HTML).'
271287 )
272288 lines .append (
273- " - The \ " dimensions\ " array MUST contain ONE entry per rubric dimension listed above, "
289+ ' - The "dimensions" array MUST contain ONE entry per rubric dimension listed above, '
274290 "named EXACTLY as written. Do not invent, rename, omit, or aggregate dimensions."
275291 )
276292 lines .append (
277- " - Each dimension entry has exactly three keys: \ " name\ " , \ " passed\ " (boolean), "
278- " \" reasoning\ " (string ≤ 12 words). Nothing else."
293+ ' - Each dimension entry has exactly three keys: "name", "passed" (boolean), '
294+ '" reasoning" (string ≤ 12 words). Nothing else.'
279295 )
280296 lines .append (
281- " - WRONG (rejected): a single verdict object like `{\ " verdict\" : \ " pass\" , \ " score\ " :0.9}`, "
297+ ' - WRONG (rejected): a single verdict object like `{"verdict": "pass", "score":0.9}`, '
282298 "or HTML table rows, or strings inside the dimensions array."
283299 )
284300 lines .append (
@@ -306,24 +322,22 @@ def build_judge_prompt(
306322 if rubric .dimensions :
307323 first_dim = rubric .dimensions [0 ].name
308324 second_dim = (
309- rubric .dimensions [1 ].name
310- if len (rubric .dimensions ) > 1
311- else first_dim
325+ rubric .dimensions [1 ].name if len (rubric .dimensions ) > 1 else first_dim
312326 )
313327 else :
314328 first_dim = "ExactDimensionName"
315329 second_dim = "AnotherDimension"
316330 lines .append (
317- '{"dimensions":[{"name":"' + first_dim + '","passed":true,"reasoning":"brief reason here"},'
331+ '{"dimensions":[{"name":"'
332+ + first_dim
333+ + '","passed":true,"reasoning":"brief reason here"},'
318334 '{"name":"' + second_dim + '","passed":false,"reasoning":"brief reason here"}],'
319335 '"overall_reasoning":"one sentence summary"}'
320336 )
321337 lines .append ("" )
322338 if rubric .dimensions :
323339 dim_names = ", " .join (d .name for d in rubric .dimensions )
324- lines .append (
325- f"Your dimension names must be exactly one of: { dim_names } "
326- )
340+ lines .append (f"Your dimension names must be exactly one of: { dim_names } " )
327341 lines .append (
328342 "Do not return a list of strings — every entry must be an object "
329343 "with 'name', 'passed', and 'reasoning' — never a plain string."
@@ -641,9 +655,7 @@ def parse_rubric_verdict(
641655 # recovered transparently. Bare lists at top level are accepted too.
642656 judge_dims = _resolve_dimensions_payload (data , rubric )
643657 if judge_dims is None :
644- raise JudgeContractError (
645- "Judge response missing required 'dimensions' key"
646- )
658+ raise JudgeContractError ("Judge response missing required 'dimensions' key" )
647659 if not isinstance (judge_dims , list ):
648660 raise JudgeContractError ("Judge response 'dimensions' must be a list" )
649661 if not isinstance (data , dict ) or "dimensions" not in data :
@@ -688,20 +700,24 @@ def parse_rubric_verdict(
688700 if dim .mandatory and not passed :
689701 mandatory_veto = True
690702
691- weighted_score = weighted_sum / total_weight if total_weight > 0 else 0.0
703+ raw_weighted_score = weighted_sum / total_weight if total_weight > 0 else 0.0
692704
693- overall_passed = weighted_score >= 0.5 and not mandatory_veto
705+ overall_passed = raw_weighted_score >= 0.5 and not mandatory_veto
694706
695707 if mandatory_veto :
696708 verdict : str = "fail"
697- elif weighted_score >= 0.5 :
709+ elif raw_weighted_score >= 0.5 :
698710 verdict = "pass"
699711 else :
700712 verdict = "fail"
701713
714+ final_weighted_score = 0.0 if mandatory_veto else raw_weighted_score
715+ pre_veto_score = raw_weighted_score if mandatory_veto else None
716+
702717 return RubricVerdict (
703718 dimension_scores = dimension_scores ,
704- weighted_score = weighted_score ,
719+ weighted_score = final_weighted_score ,
720+ weighted_score_pre_veto = pre_veto_score ,
705721 mandatory_veto = mandatory_veto ,
706722 passed = overall_passed ,
707723 verdict = verdict ,
@@ -767,7 +783,11 @@ async def evaluate_with_rubric(
767783 ]
768784
769785 judge_config = self ._judge ._provider_config .model_copy (
770- update = {"max_tokens" : estimate_judge_token_budget (rubric , references = rubric .references )}
786+ update = {
787+ "max_tokens" : estimate_judge_token_budget (
788+ rubric , references = rubric .references
789+ )
790+ }
771791 )
772792
773793 last_exc : Exception | None = None
@@ -902,9 +922,13 @@ async def evaluate_with_rubric(
902922 "fail" if (mandatory_veto or mean_score < 0.5 ) else "pass"
903923 )
904924
925+ final_mean_score = 0.0 if mandatory_veto else mean_score
926+ pre_veto_mean = mean_score if mandatory_veto else None
927+
905928 return RubricVerdict (
906929 dimension_scores = consensus_dims ,
907- weighted_score = mean_score ,
930+ weighted_score = final_mean_score ,
931+ weighted_score_pre_veto = pre_veto_mean ,
908932 mandatory_veto = mandatory_veto ,
909933 passed = overall_passed ,
910934 verdict = consensus_verdict , # type: ignore[arg-type]
0 commit comments