run_eval: construct the judge only when the recipe uses llm_judge
Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
parent
dc8d4605ac
commit
fe902db330
@ -821,7 +821,8 @@ async def run_eval(
|
|||||||
status_callback('Scores cached in checkpoint -- replaying '
|
status_callback('Scores cached in checkpoint -- replaying '
|
||||||
'(no scorers run; --rescore re-evaluates)')
|
'(no scorers run; --rescore re-evaluates)')
|
||||||
|
|
||||||
if judge is None and judge_spec and _records is None:
|
if judge is None and judge_spec and _records is None \
|
||||||
|
and _recipe_needs_judge(recipe):
|
||||||
if status_callback:
|
if status_callback:
|
||||||
status_callback('loading judge model')
|
status_callback('loading judge model')
|
||||||
judge_adapter = _make_adapter(judge_spec, api_key=judge_api_key or api_key)
|
judge_adapter = _make_adapter(judge_spec, api_key=judge_api_key or api_key)
|
||||||
@ -893,6 +894,20 @@ async def run_eval(
|
|||||||
return report
|
return report
|
||||||
|
|
||||||
|
|
||||||
|
def _recipe_needs_judge(recipe) -> bool:
|
||||||
|
"""Only construct the judge when a scorer actually consumes it --
|
||||||
|
rule-based benches (longbench_v2 etc.) never touch a judge, and
|
||||||
|
building one there crashed on malformed specs for no benefit."""
|
||||||
|
try:
|
||||||
|
for spec in (recipe.scorers or {}).values():
|
||||||
|
p = spec if isinstance(spec, dict) else {}
|
||||||
|
if p.get('name') in ('llm_judge', 'judge'):
|
||||||
|
return True
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
def _env_user_adapter(spec: str):
|
def _env_user_adapter(spec: str):
|
||||||
"""Build (once per spec) the separate USER-simulator adapter for env
|
"""Build (once per spec) the separate USER-simulator adapter for env
|
||||||
benches (tau2 strong-user parity mode)."""
|
benches (tau2 strong-user parity mode)."""
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user