sora 370953729b Fix perf stats (wrong import path), per-repeat checkpoints, README
- perf_stats aggregator lives in eval/, not model/: the import failed
  silently and EVERY perf column was empty (not just ttft). Now warns
  on stderr instead of swallowing.
- repeats > 1 get their own checkpoint key (:rep2, :rep3, ...): repeat 2
  previously restored repeat 1's predictions and finished instantly with
  identical scores. rep1 keeps the legacy key (existing checkpoints still
  resume).
- repeats summary: report the MEAN score and aggregate time/tokens over
  ALL runs (was: last run only).
- README: six-benchmark command as the primary example.

Co-Authored-By: Claude <noreply@anthropic.com>
2026-09-11 13:38:04 +00:00

65 lines
1.8 KiB
Python

"""EvalHarness: a plugin-based LLM/agent evaluation harness (data layer first)."""
from .data import (
ChatMessage,
Dataset,
DatasetSpec,
FieldSpec,
Sample,
SandboxSpec,
ToolInfo,
get_dataset,
list_datasets,
register_dataset,
)
__version__ = '0.1.0'
async def arun(bench, model, **kwargs):
"""ASYNC entry: for callers already inside an event loop.
import evalharness
rep = await evalharness.arun('gsm8k', 'openai/http://...?m', limit=200)
"""
return await _run_dispatch(bench, model, **kwargs)
def run(bench, model, **kwargs):
"""SYNC entry (primary API, mirroring inspect_ai.run / lm_eval).
import evalharness
rep = evalharness.run('gsm8k', 'openai/http://...?m', limit=200)
Creates its own event loop; safe to call from scripts/notebooks.
"""
import asyncio
try: # already inside a loop (notebook)? run in a thread
asyncio.get_running_loop()
import concurrent.futures
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
return pool.submit(asyncio.run, _run_dispatch(bench, model, **kwargs)).result()
except RuntimeError:
return asyncio.run(_run_dispatch(bench, model, **kwargs))
async def _run_dispatch(bench, model, **kwargs):
from .data import Dataset as _Dataset, get_dataset as _gd
from .model import run_eval as _run_eval
if isinstance(bench, str):
ds = _gd(bench, subset=kwargs.pop('subset', None))
elif isinstance(bench, (_Dataset, list)):
ds = bench
else:
raise TypeError(f'bench must be str/Dataset/list, got {type(bench)}')
return await _run_eval(ds, model, **kwargs)
__all__ = [
'Dataset', 'DatasetSpec', 'FieldSpec', 'Sample', 'ChatMessage', 'SandboxSpec', 'ToolInfo',
'get_dataset', 'list_datasets', 'register_dataset', 'run', 'arun',
]