@@ -501,6 +501,22 @@ def _normalize_dim_name(name: str) -> str:
501501 return re .sub (r"[^a-z0-9]" , "" , cleaned .lower ())
502502
503503
504+ def _parse_dimension_passed (value : object ) -> bool :
505+ if isinstance (value , bool ):
506+ return value
507+ if isinstance (value , int ) and value in (0 , 1 ):
508+ return bool (value )
509+ if isinstance (value , str ):
510+ token = value .strip ().lower ()
511+ if token in _VERDICT_TRUE_TOKENS :
512+ return True
513+ if token in _VERDICT_FALSE_TOKENS :
514+ return False
515+ raise JudgeContractError (
516+ f"Judge dimension 'passed' must be a boolean, integer 0/1, or a recognized verdict token; got { value !r} "
517+ )
518+
519+
504520def _fuzzy_match_dim (name : str , rubric : AnalyticRubric ) -> Optional [str ]:
505521 """Return the canonical rubric dim name for a (possibly typo'd) judge name.
506522
@@ -524,7 +540,7 @@ def _recover_dimensions_from_top_level(
524540
525541 Llama sometimes returns `{"dim_name_a": {...}, "dim_name_b": {...},
526542 "overall_reasoning": "..."}` instead of the contract envelope. If any
527- top-level key fuzzy-matches a rubric dim name and carries a bool / dict
543+ top-level key fuzzy-matches a rubric dim name and carries a bool / integer 0/1 / dict
528544 verdict, recover it.
529545 """
530546 reserved = {"overall_reasoning" , "dimensions" }
@@ -541,18 +557,12 @@ def _recover_dimensions_from_top_level(
541557 recovered .append (
542558 {
543559 "name" : canonical ,
544- "passed" : bool (value .get ("passed" , False )),
560+ "passed" : _parse_dimension_passed (value .get ("passed" , False )),
545561 "reasoning" : str (value .get ("reasoning" , "" )),
546562 }
547563 )
548- elif isinstance (value , str ):
549- token = value .strip ().lower ()
550- if token in _VERDICT_TRUE_TOKENS :
551- recovered .append ({"name" : canonical , "passed" : True , "reasoning" : "" })
552- elif token in _VERDICT_FALSE_TOKENS :
553- recovered .append ({"name" : canonical , "passed" : False , "reasoning" : "" })
554- # Unknown tokens fall through; caller raises JudgeContractError so
555- # the existing retry loop still triggers.
564+ elif isinstance (value , (str , int )):
565+ recovered .append ({"name" : canonical , "passed" : _parse_dimension_passed (value ), "reasoning" : "" })
556566 return recovered or None
557567
558568
@@ -661,6 +671,8 @@ def build_judge_dim_map(
661671 entry = rescued
662672 if "name" not in entry :
663673 continue
674+ if not isinstance (entry ["name" ], str ):
675+ raise JudgeContractError ("Judge dimension 'name' must be a string" )
664676 canonical = _fuzzy_match_dim (entry ["name" ], rubric )
665677 if canonical is None :
666678 logger .warning (
@@ -670,10 +682,13 @@ def build_judge_dim_map(
670682 key = canonical .lower ()
671683 if key in conflicts :
672684 continue
673- normalized_entry = {** entry , "name" : canonical }
685+ normalized_entry = {
686+ ** entry , "name" : canonical ,
687+ "passed" : _parse_dimension_passed (entry .get ("passed" , False )),
688+ }
674689 if key in first_occurrence :
675- existing = bool ( first_occurrence [key ]. get ( "passed" , False ))
676- incoming = bool ( entry . get ( "passed" , False ))
690+ existing = first_occurrence [key ][ "passed" ]
691+ incoming = normalized_entry [ "passed" ]
677692 if existing != incoming :
678693 conflicts .add (key )
679694 del first_occurrence [key ]
@@ -796,7 +811,7 @@ def _dimension_score_from_judge(
796811 if match is not None :
797812 return DimensionScore (
798813 dimension_name = dim .name ,
799- passed = bool (match .get ("passed" , False )),
814+ passed = _parse_dimension_passed (match .get ("passed" , False )),
800815 reasoning = str (match .get ("reasoning" , "" )),
801816 confidence = 0.9 ,
802817 is_mandatory = dim .mandatory ,
@@ -937,6 +952,13 @@ async def evaluate_with_rubric(
937952 references = rubric .references ,
938953 )
939954 safe_response = sanitize_response_payload (response )
955+ if len (safe_response ) > JUDGE_PAYLOAD_MAX_CHARS :
956+ # The sanitizer appends a marker when it cuts the input. A verdict
957+ # over that prefix cannot establish what the complete answer did.
958+ raise JudgeExtractionError (
959+ f"response exceeds the { JUDGE_PAYLOAD_MAX_CHARS } -character judge payload limit; "
960+ "the complete response cannot be graded"
961+ )
940962
941963 messages = [
942964 ChatMessage (role = "system" , content = prompt ),
0 commit comments