Benchmark categories in results (English)
Five categories (Code & Engineering / Reasoning & Math / Knowledge & Language / Long Context / Agents & Tools) now annotate the console summary table (new column), summary.csv (new field), and each report's run_info.category. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
parent
1fd0dcbdbb
commit
1e72a553ec
@ -163,6 +163,30 @@ def _rich_console():
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
BENCH_CATEGORIES = {
|
||||||
|
'bigcodebench': 'Code & Engineering', 'humaneval': 'Code & Engineering',
|
||||||
|
'live_code_bench': 'Code & Engineering',
|
||||||
|
'swe_bench_verified': 'Code & Engineering',
|
||||||
|
'aime24': 'Reasoning & Math', 'aime25': 'Reasoning & Math',
|
||||||
|
'aime26': 'Reasoning & Math', 'hmmt26': 'Reasoning & Math',
|
||||||
|
'imo_answerbench': 'Reasoning & Math', 'hle': 'Reasoning & Math',
|
||||||
|
'gsm8k': 'Reasoning & Math', 'competition_math': 'Reasoning & Math',
|
||||||
|
'bbh': 'Reasoning & Math', 'drop': 'Reasoning & Math',
|
||||||
|
'gpqa_diamond': 'Knowledge & Language', 'mmlu_pro': 'Knowledge & Language',
|
||||||
|
'simple_qa': 'Knowledge & Language', 'mmlu': 'Knowledge & Language',
|
||||||
|
'cmmlu': 'Knowledge & Language', 'arc': 'Knowledge & Language',
|
||||||
|
'hellaswag': 'Knowledge & Language', 'trivia_qa': 'Knowledge & Language',
|
||||||
|
'winogrande': 'Knowledge & Language',
|
||||||
|
'longbench_v2': 'Long Context', 'openai_mrcr': 'Long Context',
|
||||||
|
'tau2_bench': 'Agents & Tools', 'general_fc': 'Agents & Tools',
|
||||||
|
'bfcl_v3': 'Agents & Tools',
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def bench_category(name: str) -> str:
|
||||||
|
return BENCH_CATEGORIES.get(name, '')
|
||||||
|
|
||||||
|
|
||||||
def _load_bench_cfg(args, name: str) -> dict:
|
def _load_bench_cfg(args, name: str) -> dict:
|
||||||
"""Merged YAML config for one bench: {default 段, bench 段}.
|
"""Merged YAML config for one bench: {default 段, bench 段}.
|
||||||
|
|
||||||
@ -664,7 +688,7 @@ def _cmd_eval_run(args) -> int:
|
|||||||
|
|
||||||
with open(f'{out_dir}/summary.csv', 'w', newline='', encoding='utf-8') as f:
|
with open(f'{out_dir}/summary.csv', 'w', newline='', encoding='utf-8') as f:
|
||||||
w = _csv.writer(f)
|
w = _csv.writer(f)
|
||||||
w.writerow(['benchmark', 'score', 'metric', 'num_samples',
|
w.writerow(['benchmark', 'category', 'score', 'metric', 'num_samples',
|
||||||
'time_h', 'time_s', 'extract_fail',
|
'time_h', 'time_s', 'extract_fail',
|
||||||
'success_rate', 'latency_mean_s', 'output_tps', 'request_qps',
|
'success_rate', 'latency_mean_s', 'output_tps', 'request_qps',
|
||||||
'input_tokens_mean', 'output_tokens_mean', 'total_tokens',
|
'input_tokens_mean', 'output_tokens_mean', 'total_tokens',
|
||||||
@ -678,7 +702,8 @@ def _cmd_eval_run(args) -> int:
|
|||||||
if gname != 'perf'
|
if gname != 'perf'
|
||||||
for g, v in (gv or {}).items()
|
for g, v in (gv or {}).items()
|
||||||
if isinstance(v, (int, float)))[:2000]
|
if isinstance(v, (int, float)))[:2000]
|
||||||
w.writerow([r['name'], _f3(r.get('value')), r['metric'], r.get('n', ''),
|
w.writerow([r['name'], r.get('category', ''), _f3(r.get('value')),
|
||||||
|
r['metric'], r.get('n', ''),
|
||||||
r.get('hours', ''), r.get('secs', ''),
|
r.get('hours', ''), r.get('secs', ''),
|
||||||
r.get('extract_fail', 0)] +
|
r.get('extract_fail', 0)] +
|
||||||
[perf.get(k, '') for k in (
|
[perf.get(k, '') for k in (
|
||||||
@ -943,7 +968,9 @@ def _cmd_eval_run(args) -> int:
|
|||||||
# (days old, slower setup) -- that once reported 15.9h for a
|
# (days old, slower setup) -- that once reported 15.9h for a
|
||||||
# one-hour run
|
# one-hour run
|
||||||
_wall = round(_time.time() - t0, 1)
|
_wall = round(_time.time() - t0, 1)
|
||||||
rows.append({'name': name, 'metric': primary,
|
_cat = bench_category(name)
|
||||||
|
report.metric_groups.setdefault('run_info', {})['category'] = _cat
|
||||||
|
rows.append({'name': name, 'metric': primary, 'category': _cat,
|
||||||
'cached': (report.metric_groups.get('run_info', {})
|
'cached': (report.metric_groups.get('run_info', {})
|
||||||
.get('gen_fresh') == 0) if _repeats <= 1 else False,
|
.get('gen_fresh') == 0) if _repeats <= 1 else False,
|
||||||
'value': report.metrics.get(primary),
|
'value': report.metrics.get(primary),
|
||||||
@ -971,6 +998,7 @@ def _cmd_eval_run(args) -> int:
|
|||||||
'done', _time.time() - t0)
|
'done', _time.time() - t0)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
rows.append({'name': name, 'metric': '-', 'value': None,
|
rows.append({'name': name, 'metric': '-', 'value': None,
|
||||||
|
'category': bench_category(name),
|
||||||
'secs': round(_time.time() - t0, 1), 'ok': False,
|
'secs': round(_time.time() - t0, 1), 'ok': False,
|
||||||
'err': f'{type(e).__name__}: {str(e)[:100]}'})
|
'err': f'{type(e).__name__}: {str(e)[:100]}'})
|
||||||
print(f'{name}: FAILED {type(e).__name__}: {str(e)[:160]}', file=sys.stderr)
|
print(f'{name}: FAILED {type(e).__name__}: {str(e)[:160]}', file=sys.stderr)
|
||||||
@ -1040,7 +1068,8 @@ def _print_summary_table(console, rows):
|
|||||||
|
|
||||||
t = Table(title='Run Summary', header_style='bold cyan',
|
t = Table(title='Run Summary', header_style='bold cyan',
|
||||||
title_style='bold', expand=False)
|
title_style='bold', expand=False)
|
||||||
for col, just in (('benchmark', 'left'), ('metric', 'left'),
|
for col, just in (('benchmark', 'left'), ('category', 'left'),
|
||||||
|
('metric', 'left'),
|
||||||
('score', 'right'), ('n', 'right'), ('time', 'right'),
|
('score', 'right'), ('n', 'right'), ('time', 'right'),
|
||||||
('tok in', 'right'), ('tok out', 'right'),
|
('tok in', 'right'), ('tok out', 'right'),
|
||||||
('in/s', 'right'), ('out/s', 'right')):
|
('in/s', 'right'), ('out/s', 'right')):
|
||||||
@ -1055,7 +1084,8 @@ def _print_summary_table(console, rows):
|
|||||||
tout = f'{to:,}' if to else '—'
|
tout = f'{to:,}' if to else '—'
|
||||||
tis = f'{ti / wall:.0f}' if (wall > 1 and ti) else '—'
|
tis = f'{ti / wall:.0f}' if (wall > 1 and ti) else '—'
|
||||||
tos = f'{to / wall:.0f}' if (wall > 1 and to) else '—'
|
tos = f'{to / wall:.0f}' if (wall > 1 and to) else '—'
|
||||||
t.add_row(r['name'], r['metric'], v, str(r.get('n', '')), tm,
|
t.add_row(r['name'], r.get('category', ''), r['metric'], v,
|
||||||
|
str(r.get('n', '')), tm,
|
||||||
tin, tout, tis, tos,
|
tin, tout, tis, tos,
|
||||||
style='green' if r['ok'] else 'red')
|
style='green' if r['ok'] else 'red')
|
||||||
wall_all = sum(r.get('wall') or r.get('secs') or 0 for r in rows)
|
wall_all = sum(r.get('wall') or r.get('secs') or 0 for r in rows)
|
||||||
@ -1066,7 +1096,7 @@ def _print_summary_table(console, rows):
|
|||||||
t.add_section()
|
t.add_section()
|
||||||
tis_all = f'{ti_all / wall_all:.0f}' if wall_all > 1 else ''
|
tis_all = f'{ti_all / wall_all:.0f}' if wall_all > 1 else ''
|
||||||
tos_all = f'{to_all / wall_all:.0f}' if wall_all > 1 else ''
|
tos_all = f'{to_all / wall_all:.0f}' if wall_all > 1 else ''
|
||||||
t.add_row(f'[bold]{len(rows)} benchmarks[/bold]', '',
|
t.add_row(f'[bold]{len(rows)} benchmarks[/bold]', '', '',
|
||||||
f'{sum(1 for r in rows if r["ok"])}/{len(rows)} ok',
|
f'{sum(1 for r in rows if r["ok"])}/{len(rows)} ok',
|
||||||
str(n_all), tm_all, f'{ti_all:,}', f'{to_all:,}',
|
str(n_all), tm_all, f'{ti_all:,}', f'{to_all:,}',
|
||||||
tis_all, tos_all)
|
tis_all, tos_all)
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user