- perf_stats aggregator lives in eval/, not model/: the import failed silently and EVERY perf column was empty (not just ttft). Now warns on stderr instead of swallowing. - repeats > 1 get their own checkpoint key (:rep2, :rep3, ...): repeat 2 previously restored repeat 1's predictions and finished instantly with identical scores. rep1 keeps the legacy key (existing checkpoints still resume). - repeats summary: report the MEAN score and aggregate time/tokens over ALL runs (was: last run only). - README: six-benchmark command as the primary example. Co-Authored-By: Claude <noreply@anthropic.com>
65 lines
1.8 KiB
Python
65 lines
1.8 KiB
Python
"""EvalHarness: a plugin-based LLM/agent evaluation harness (data layer first)."""
|
|
|
|
from .data import (
|
|
ChatMessage,
|
|
Dataset,
|
|
DatasetSpec,
|
|
FieldSpec,
|
|
Sample,
|
|
SandboxSpec,
|
|
ToolInfo,
|
|
get_dataset,
|
|
list_datasets,
|
|
register_dataset,
|
|
)
|
|
|
|
__version__ = '0.1.0'
|
|
|
|
|
|
async def arun(bench, model, **kwargs):
|
|
"""ASYNC entry: for callers already inside an event loop.
|
|
|
|
import evalharness
|
|
rep = await evalharness.arun('gsm8k', 'openai/http://...?m', limit=200)
|
|
"""
|
|
return await _run_dispatch(bench, model, **kwargs)
|
|
|
|
|
|
def run(bench, model, **kwargs):
|
|
"""SYNC entry (primary API, mirroring inspect_ai.run / lm_eval).
|
|
|
|
import evalharness
|
|
rep = evalharness.run('gsm8k', 'openai/http://...?m', limit=200)
|
|
|
|
Creates its own event loop; safe to call from scripts/notebooks.
|
|
"""
|
|
import asyncio
|
|
|
|
try: # already inside a loop (notebook)? run in a thread
|
|
asyncio.get_running_loop()
|
|
import concurrent.futures
|
|
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
|
|
return pool.submit(asyncio.run, _run_dispatch(bench, model, **kwargs)).result()
|
|
except RuntimeError:
|
|
return asyncio.run(_run_dispatch(bench, model, **kwargs))
|
|
|
|
|
|
async def _run_dispatch(bench, model, **kwargs):
|
|
from .data import Dataset as _Dataset, get_dataset as _gd
|
|
from .model import run_eval as _run_eval
|
|
|
|
if isinstance(bench, str):
|
|
ds = _gd(bench, subset=kwargs.pop('subset', None))
|
|
elif isinstance(bench, (_Dataset, list)):
|
|
ds = bench
|
|
else:
|
|
raise TypeError(f'bench must be str/Dataset/list, got {type(bench)}')
|
|
return await _run_eval(ds, model, **kwargs)
|
|
|
|
|
|
__all__ = [
|
|
'Dataset', 'DatasetSpec', 'FieldSpec', 'Sample', 'ChatMessage', 'SandboxSpec', 'ToolInfo',
|
|
'get_dataset', 'list_datasets', 'register_dataset', 'run', 'arun',
|
|
]
|