This commit is contained in:
sora 2026-07-22 03:26:58 +00:00
parent cfa58f869f
commit 03b49a39d0
15 changed files with 448 additions and 1177 deletions

View File

@ -0,0 +1,29 @@
分类,Benchmark,得分,实测时间(h),总样本数,延迟_mean(s),输出TPS,请求QPS,输入tokens_mean,输出tokens_mean,累计总tokens,TTFT_mean(s),TTFT P90,TTFT P99,TPOT_mean(s),TPOT P90,TPOT P99,,,,,,,,,,,,,,,
代码与工程,bigcodebench,0.9956,0.9915,1140,10.15508,32.99,0.0985,151.77,335.03,554946,0.26441,0.27379,0.42952,0.02957,0.03059,0.03143,,,,,,,,,,,,,,,
,humaneval,0.9044,0.1183,164,2.776502,42.58667,0.3602,163.0793,118.2398,45994,0.25521,0.304186,0.451449,0.021783,0.024498,0.027453,,,,,,,,,,,,,,,
,live_code_bench,0.6756,7.9486,100,10.41723,43.578,0.09644,518.1118,453.5784,1040894,0.253597,0.272382,0.464771,0.02093,0.023769,0.026398,,,,,,,,,,,,,,,
推理与数学,aime24,0.6167,1.2217,30,42.44225,44.85,0.024108,119.3333,1901.842,55169,0.258426,0.274508,0.444511,0.021698,0.025218,hle,,,,,,,,,,,,,,,
,aime25,0.4639,2.0414,30,59.98314,46.03583,0.017092,188.4333,2761.283,80482,0.257489,0.273448,0.4417,0.021655,0.024778,0.026626,,,,,,,,,,,,,,,
,aime26,0.5611,1.4847,30,51.64098,44.66417,0.019925,139.7667,2305.069,60102,0.25465,0.268527,0.435013,0.021667,0.024596,0.026089,,,,,,,,,,,,,,,
,hmmt26,0.4015,1.5481,30,50.8562,44.73,0.020008,108.4242,2274.237,63775,0.255456,0.266018,0.431919,0.022514,0.025679,0.027263,,,,,,,,,,,,,,,
,imo_answerbench,0.36,1.2508,400,39.6005,44.94,0.0253,122.1,1779.8,753159,0.2466,0.2641,0.3167,0.0226,0.0257,0.0285,,,,,,,,,,,,,,,
,hle,0.0508,14.6863,3000,61.787217,34.42,0.0162,292.3252,2126.7916,6047792,0.275126,0.282138,0.474372,0.028997,0.029831,0.031026,,,,,,,,,,,,,,,
,gsm8k,0.9704,0.3178,1319,3.031098,43.8,0.3299,582.9507,132.7726,944039,0.253185,0.267307,0.45469,0.021127,0.02357,0.025568,,,,,,,,,,,,,,,
,competition_math,0.941,3.1817,500,8.681756,51,0.1152,296.5698,442.732,3696509,0.245456,0.263328,0.435642,0.019541,0.021677,0.024484,,,,,,,,,,,,,,,
,bbh,0.9032,1.8036,6513,3.561695,42.95,0.2808,916.98,152.9719,6966457,0.269139,0.276913,0.467974,0.023935,0.030921,0.065977,,,,,,,,,,,,,,,
,drop,0.8183,1.5336,9536,1.824962,36.31,0.548,1273.607,66.25849,12776955,0.297489,0.4593,0.505599,0.023644,0.028072,0.034489,,,,,,,,,,,,,,,
知识与语言理解,gpqa_diamond,0.6944,0.3164,198,10.54948,41.335,0.0948,252.4343,436.0202,135770,0.245972,0.26274,0.456931,0.023819,0.027653,0.030237,,,,,,,,,,,,,,,
,mmlu_pro,0.8285,5.6167,12032,6.264368,31.05,0.1596,1312.528,194.5092,18132672,0.293428,0.437219,0.477191,0.031274,0.034267,0.036824,,,,,,,,,,,,,,,
,simple_qa,0.3814,1.9014,4326,1.88825,30.69,0.529590891,30.351826,57.955155,382016,0.26334,0.269454,0.452375,0.027793,0.031603,0.035322,,,,,,,,,,,,,,,
,mmlu,0.9045,3.0297,14042,2.6957,35.98,0.371,727.6,97,11532337,0.2643,0.2751,0.4663,0.026,0.0305,0.0348,,,,,,,,,,,,,,,
,cmmlu,0.9001,3.4935,11515,3.9225,36.65,0.2549,115.3,143.8,2983005,0.2553,0.2956,0.443,0.0266,0.0307,0.0353,,,,,,,,,,,,,,,
,arc,0.9476,0.2103,1172,0.662452,13.65,1.5095,102.0333,9.042841,394098,0.321567,0.437672,0.449777,0.066642,0.129754,0.132577,,,,,,,,,,,,,,,
,hellaswag,0.8666,0.4453,10042,0.597052,7.32,1.6749,219.8686,4.370444,2251808,0.341421,0.448449,0.461157,0.081246,0.133338,0.136106,,,,,,,,,,,,,,,
,trivia_qa,0.7768,5.3011,11313,9.163784,3.78,0.1091,11964.972,34.6152,95912700,4.722344,8.951851,16.512039,0.131478,0.292489,0.622806,,,,,,,,,,,,,,,
,winogrande,0.7845,0.0775,1267,0.695127,13.88,1.4386,76.11997,9.646409,108666,0.318495,0.433513,0.438349,0.069826,0.129328,0.130936,,,,,,,,,,,,,,,
长上下文,longbench_v2,0.5686,3.0638,503,84.120374,2.75,0.0119,85638.8191,231.6441,43192843,25.148194,53.202063,64.267424,0.248375,0.534405,0.946107,,,,,,,,,,,,,,,
,openai_mrcr,0.7768,5.2828,2399,29.1532,12.95,0.0343,24774.8,377.6,60340773,6.6088,21.0867,37.983,0.0608,0.1125,0.2004,,,,,,,,,,,,,,,
智能体与工具,tau2_bench,0.7684,1.7625,,,,,,,,,,,,,,,,,,,,,,,,,,,,
,general_fc,0.6988,1.3803,2000,9.4082,39.21,0.1063,1619.2,368.9,3976219,0.4226,0.7249,1.2941,0.026,0.0307,0.0398,,,,,,,,,,,,,,,
,bfcl_v3,0.6643,5.23,4441,4.296605,25.32,0.2327,7268.4069,108.7864,97061732,0.418217,0.521432,1.106471,0.035785,0.039796,0.053177,,,,,,,,,,,,,,,
,总计,0.711992593,75.2394,,,,,,,,,,,,,,,,,,,,,,,,,,,,
1 分类 Benchmark 得分 实测时间(h) 总样本数 延迟_mean(s) 输出TPS 请求QPS 输入tokens_mean 输出tokens_mean 累计总tokens TTFT_mean(s) TTFT P90 TTFT P99 TPOT_mean(s) TPOT P90 TPOT P99
2 代码与工程 bigcodebench 0.9956 0.9915 1140 10.15508 32.99 0.0985 151.77 335.03 554946 0.26441 0.27379 0.42952 0.02957 0.03059 0.03143
3 humaneval 0.9044 0.1183 164 2.776502 42.58667 0.3602 163.0793 118.2398 45994 0.25521 0.304186 0.451449 0.021783 0.024498 0.027453
4 live_code_bench 0.6756 7.9486 100 10.41723 43.578 0.09644 518.1118 453.5784 1040894 0.253597 0.272382 0.464771 0.02093 0.023769 0.026398
5 推理与数学 aime24 0.6167 1.2217 30 42.44225 44.85 0.024108 119.3333 1901.842 55169 0.258426 0.274508 0.444511 0.021698 0.025218 hle
6 aime25 0.4639 2.0414 30 59.98314 46.03583 0.017092 188.4333 2761.283 80482 0.257489 0.273448 0.4417 0.021655 0.024778 0.026626
7 aime26 0.5611 1.4847 30 51.64098 44.66417 0.019925 139.7667 2305.069 60102 0.25465 0.268527 0.435013 0.021667 0.024596 0.026089
8 hmmt26 0.4015 1.5481 30 50.8562 44.73 0.020008 108.4242 2274.237 63775 0.255456 0.266018 0.431919 0.022514 0.025679 0.027263
9 imo_answerbench 0.36 1.2508 400 39.6005 44.94 0.0253 122.1 1779.8 753159 0.2466 0.2641 0.3167 0.0226 0.0257 0.0285
10 hle 0.0508 14.6863 3000 61.787217 34.42 0.0162 292.3252 2126.7916 6047792 0.275126 0.282138 0.474372 0.028997 0.029831 0.031026
11 gsm8k 0.9704 0.3178 1319 3.031098 43.8 0.3299 582.9507 132.7726 944039 0.253185 0.267307 0.45469 0.021127 0.02357 0.025568
12 competition_math 0.941 3.1817 500 8.681756 51 0.1152 296.5698 442.732 3696509 0.245456 0.263328 0.435642 0.019541 0.021677 0.024484
13 bbh 0.9032 1.8036 6513 3.561695 42.95 0.2808 916.98 152.9719 6966457 0.269139 0.276913 0.467974 0.023935 0.030921 0.065977
14 drop 0.8183 1.5336 9536 1.824962 36.31 0.548 1273.607 66.25849 12776955 0.297489 0.4593 0.505599 0.023644 0.028072 0.034489
15 知识与语言理解 gpqa_diamond 0.6944 0.3164 198 10.54948 41.335 0.0948 252.4343 436.0202 135770 0.245972 0.26274 0.456931 0.023819 0.027653 0.030237
16 mmlu_pro 0.8285 5.6167 12032 6.264368 31.05 0.1596 1312.528 194.5092 18132672 0.293428 0.437219 0.477191 0.031274 0.034267 0.036824
17 simple_qa 0.3814 1.9014 4326 1.88825 30.69 0.529590891 30.351826 57.955155 382016 0.26334 0.269454 0.452375 0.027793 0.031603 0.035322
18 mmlu 0.9045 3.0297 14042 2.6957 35.98 0.371 727.6 97 11532337 0.2643 0.2751 0.4663 0.026 0.0305 0.0348
19 cmmlu 0.9001 3.4935 11515 3.9225 36.65 0.2549 115.3 143.8 2983005 0.2553 0.2956 0.443 0.0266 0.0307 0.0353
20 arc 0.9476 0.2103 1172 0.662452 13.65 1.5095 102.0333 9.042841 394098 0.321567 0.437672 0.449777 0.066642 0.129754 0.132577
21 hellaswag 0.8666 0.4453 10042 0.597052 7.32 1.6749 219.8686 4.370444 2251808 0.341421 0.448449 0.461157 0.081246 0.133338 0.136106
22 trivia_qa 0.7768 5.3011 11313 9.163784 3.78 0.1091 11964.972 34.6152 95912700 4.722344 8.951851 16.512039 0.131478 0.292489 0.622806
23 winogrande 0.7845 0.0775 1267 0.695127 13.88 1.4386 76.11997 9.646409 108666 0.318495 0.433513 0.438349 0.069826 0.129328 0.130936
24 长上下文 longbench_v2 0.5686 3.0638 503 84.120374 2.75 0.0119 85638.8191 231.6441 43192843 25.148194 53.202063 64.267424 0.248375 0.534405 0.946107
25 openai_mrcr 0.7768 5.2828 2399 29.1532 12.95 0.0343 24774.8 377.6 60340773 6.6088 21.0867 37.983 0.0608 0.1125 0.2004
26 智能体与工具 tau2_bench 0.7684 1.7625
27 general_fc 0.6988 1.3803 2000 9.4082 39.21 0.1063 1619.2 368.9 3976219 0.4226 0.7249 1.2941 0.026 0.0307 0.0398
28 bfcl_v3 0.6643 5.23 4441 4.296605 25.32 0.2327 7268.4069 108.7864 97061732 0.418217 0.521432 1.106471 0.035785 0.039796 0.053177
29 总计 0.711992593 75.2394

View File

@ -1,94 +0,0 @@
#!/usr/bin/env python3
"""Fill simple_qa (fresh re-run) + bfcl_v3 (official OVERALL) into the Excel.
simple_qa: old cells had D=0.0005 (cache-resume artifact) and empty perf.
Overwrite C/D/F-Q from the NEW report (real perf + ~1.9h duration).
bfcl_v3 : top-level score 0.6643 is evalscope's macro-average (non-standard).
Official BFCL score = OVERALL subset = 0.569. Update C only.
"""
import json
import glob
import openpyxl
XLSX = '/data1/sora/P800模型能力评测结果_统一格式_filled.xlsx'
SHEETS = ['2.0-FULL', '2.0-Lite']
# --- read reports ---
sq = json.load(open('/data1/sora/evalscope/output/simple_qa/seed_42/reports/DeepSeek-V4-Flash-Int8/simple_qa.json'))
bc = json.load(open(glob.glob('/data1/sora/evalscope/output/bfcl_v3/seed_42.bak/reports/DeepSeek-V4-Flash-Int8/bfcl_v3.json')[0]))
def perf(d):
pm = d.get('perf_metrics') or {}
s = pm.get('summary', {}) or {}
u = s.get('usage', {}) or {}
tf = s.get('ttft') or {}
tp = s.get('tpot') or {}
lat = (s.get('latency') or {}).get('mean')
return {
'score': d.get('score'),
'duration_h': (d.get('duration_sec') or 0) / 3600,
'latency': lat,
'tps': (s.get('throughput') or {}).get('avg_output_tps'),
'qps': (1.0 / lat) if lat else None,
'in_tok': (u.get('input_tokens') or {}).get('mean'),
'out_tok': (u.get('output_tokens') or {}).get('mean'),
'ttc': u.get('total_tokens_count'),
'ttft_m': tf.get('mean'),
'ttft90': tf.get('90%') or tf.get('p90'),
'ttft99': tf.get('99%') or tf.get('p99'),
'tpot_m': tp.get('mean'),
'tpot90': tp.get('90%') or tp.get('p90'),
'tpot99': tp.get('99%') or tp.get('p99'),
}
def bfcl_overall(d):
for metric in d.get('metrics', []):
for cat in metric.get('categories', []):
for sub in cat.get('subsets', []):
nm = sub.get('name')
if nm == 'OVERALL' or (isinstance(nm, list) and 'OVERALL' in nm):
return sub.get('score')
return None
P = perf(sq)
BC_OVERALL = bfcl_overall(bc)
print('simple_qa:', {k: (round(v, 4) if isinstance(v, float) else v) for k, v in P.items()})
print('bfcl_v3 OVERALL:', BC_OVERALL)
# col -> simple_qa perf key
SQ = {'C': P['score'], 'D': round(P['duration_h'], 4), 'F': P['latency'], 'G': P['tps'],
'H': P['qps'], 'I': P['in_tok'], 'J': P['out_tok'], 'K': P['ttc'],
'L': P['ttft_m'], 'M': P['ttft90'], 'N': P['ttft99'],
'O': P['tpot_m'], 'P': P['tpot90'], 'Q': P['tpot99']}
wb = openpyxl.load_workbook(XLSX)
for sn in SHEETS:
ws = wb[sn]
for r in range(2, 30):
b = ws.cell(r, 2).value
if b == 'simple_qa':
for col, v in SQ.items():
if v is not None:
ws[f'{col}{r}'] = int(round(v)) if col == 'K' else round(v, 4) if col in ('C', 'D') else v
print(f'{sn} simple_qa row{r}: C={ws[f"C{r}"].value} D={ws[f"D{r}"].value} F={ws[f"F{r}"].value} K={ws[f"K{r}"].value}')
elif b == 'bfcl_v3' and BC_OVERALL is not None:
ws.cell(r, 3).value = round(BC_OVERALL, 4)
print(f'{sn} bfcl_v3 row{r}: C={ws.cell(r,3).value}')
# recompute total D
tr = next((rr for rr in range(2, ws.max_row + 1) if ws.cell(rr, 2).value == '总计'), None)
if tr:
tot = 0.0
for rr in range(2, tr):
v = ws.cell(rr, 4).value
try:
tot += float(v)
except (TypeError, ValueError):
pass
ws.cell(tr, 4).value = round(tot, 4)
print(f'{sn} 总计 D = {round(tot, 4)}h')
wb.save(XLSX)
print('saved ->', XLSX)

View File

@ -1,38 +0,0 @@
"""Re-run only the review stage for bigcodebench using existing predictions.
The upstream `bigcodebench/bigcodebench-evaluate` image has an ENTRYPOINT that
runs the evaluator and exits, which kills the ms_enclave sandbox containers.
We use the locally built `bigcodebench-sandbox:latest` image (same libraries,
entrypoint dropped, runs `tail -f /dev/null`) and re-score the cached
predictions without regenerating them.
"""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent))
import run as run_mod
from evalscope import run_task
DATASET_NAME = 'bigcodebench'
BATCH_SIZE = 4
if DATASET_NAME not in run_mod.DATASET_CONFIGS:
raise SystemExit(f'{DATASET_NAME} not found in config')
ds_cfg = run_mod.DATASET_CONFIGS[DATASET_NAME]
task_cfg = run_mod.build_task_config(
dataset_name=DATASET_NAME,
ds_cfg=ds_cfg,
batch_size=BATCH_SIZE,
enable_thinking=run_mod.ENABLE_THINKING,
seed=run_mod.SEED,
run_idx=0,
)
task_cfg.rerun_review = True
print(f'Re-running review for {DATASET_NAME}')
print(f'Cache/work dir: {task_cfg.work_dir}')
print(f'Sandbox config: {task_cfg.sandbox.default_config}')
run_task(task_cfg)
print('Done.')

View File

@ -1,8 +1,40 @@
import yaml #!/usr/bin/env python3
"""
Unified benchmark runner for EvalScope.
A single entry point for lite / mid / full / group1 / group2 / group3 evaluations.
All tunable parameters can be controlled via command-line arguments.
Examples:
# Full evaluation (all benchmarks, multi-run for stability)
python bash/run.py \
--model DeepSeek-V4-Flash-Int8 \
--api-url http://localhost:30000/v1 \
--dataset-dir /data1/sora/evalscope \
--output-dir /data1/sora/evalscope/output \
--suite full \
--limit none
# Lite smoke test (~5h with full samples)
python bash/run.py --suite lite --limit none
# Run only selected benchmarks
python bash/run.py --datasets aime24,gsm8k,arc --limit 20
# Custom judge model
python bash/run.py \
--judge-model deepseek-v4-pro \
--judge-api-url https://api.deepseek.com/v1 \
--judge-api-key sk-xxx
"""
import argparse
import json
import sys
from copy import deepcopy from copy import deepcopy
from pathlib import Path from pathlib import Path
import sys
import os import yaml
from evalscope import run_task, TaskConfig from evalscope import run_task, TaskConfig
from evalscope.api.agent import NativeAgentConfig from evalscope.api.agent import NativeAgentConfig
@ -12,116 +44,112 @@ SCRIPT_DIR = Path(__file__).parent.resolve()
PROJECT_ROOT = SCRIPT_DIR.parent PROJECT_ROOT = SCRIPT_DIR.parent
# ============================================================ # ============================================================
# ★★★ 必改参数 (USER CONFIG) ★★★ # Default configuration (override via CLI)
# 每次评测新模型前,只需要检查/修改以下参数。
# 也可以通过命令行覆盖:--model / --api-url / --dataset-dir /
# --output-dir / --limit / --config
# ============================================================ # ============================================================
# 模型名served model name DEFAULT_MODEL = 'DeepSeek-V4-Flash-Int8'
MODEL = 'DeepSeek-V4-Flash-Int8' DEFAULT_API_URL = 'http://localhost:30000/v1'
DEFAULT_DATASET_DIR = str(PROJECT_ROOT)
DEFAULT_OUTPUT_DIR = str(PROJECT_ROOT / 'output')
DEFAULT_CONFIG = str(PROJECT_ROOT / 'config' / 'dpv4-int8_nothinking.yaml')
DEFAULT_TOKENIZER_PATH = '/data1/models/DeepSeek-V4-Flash-INT8'
DEFAULT_LIMIT = None
DEFAULT_SEED = 42
DEFAULT_BATCH_SIZE = 4
DEFAULT_ENABLE_THINKING = False
# 模型服务地址OpenAI 兼容 API DEFAULT_JUDGE_MODEL = 'DeepSeek/DeepSeek-V4-Pro'
API_URL = 'http://localhost:30000/v1' DEFAULT_JUDGE_API_URL = 'https://api.vectron.meta-stone.com/v1'
DEFAULT_JUDGE_API_KEY = 'sk-dbd8a665f7634081b87ec409c7636500'
DEFAULT_JUDGE_MAX_TOKENS = 10240
# 本地数据集根目录(提前下载好的 datasets 目录) # 长文本 middle-truncation 上限token 数)。当前默认 128k。
DATASET_DIR = str(PROJECT_ROOT / 'datasets') DEFAULT_TRUNCATION_TOKENS = 32768 * 4
# 评测输出目录(每个 benchmark 单独子目录) # ============================================================
OUTPUT_DIR = str(PROJECT_ROOT / 'output') # Benchmark suites
# ============================================================
# 采样数量上限None 表示跑全部样本 # 多次采样配置:总样本数控制在 ~400-500
LIMIT = None MULTI_RUN_CONFIG = {
'aime24': 12,
'aime25': 12,
'aime26': 12,
'hmmt26': 12,
'live_code_bench': 5,
'imo_answerbench': 4,
'humaneval': 3,
'gpqa_diamond': 2,
}
# 每个 benchmark 的生成参数配置文件max_tokens / temperature 等) # 能力域完整列表
CONFIG = None ALL_MULTI_RUN = [
'humaneval', 'live_code_bench',
'aime24', 'aime25', 'aime26', 'hmmt26',
'imo_answerbench', 'gpqa_diamond',
]
ALL_SINGLE_RUN = [
'bigcodebench', 'bfcl_v3', 'competition_math', 'gsm8k', 'hle', 'super_gpqa',
'arc', 'bbh', 'cmmlu', 'drop', 'hellaswag', 'mmlu', 'mmlu_pro',
'simple_qa', 'trivia_qa', 'winogrande',
'openai_mrcr', 'longbench_v2',
]
ALL_AGENT = ['tau2_bench', 'general_fc']
# 是否开启 thinking 模式sglang chat_template_kwargs.thinking # 分组基于 CSV 单次时间 + multi-run 后的 wall time 平衡:
ENABLE_THINKING = False # Group1: ~61h | Group2: ~62h | Group3: ~55h
SUITES = {
# 测试哪些 benchmark见下方 DATASETS = _multi_run_order + _single_run_order + _agent_order 'full': {
# 想只跑部分 benchmark 时,注释掉对应行即可。 'multi': ALL_MULTI_RUN,
MODEL_PATH = '/data1/models/DeepSeek-V4-Flash-INT8' 'single': ALL_SINGLE_RUN,
'agent': ALL_AGENT,
# Judge model for benchmarks that require LLM-as-judge (simple_qa, tau2_bench, etc.) },
# Vectron endpoint with DeepSeek-V4-Pro 'lite': {
JUDGE_MODEL_ARGS = { 'multi': ['aime24', 'humaneval'],
'model_id': 'DeepSeek/DeepSeek-V4-Pro', 'single': ['gsm8k', 'arc', 'longbench_v2'],
'api_url': 'https://api.vectron.meta-stone.com/v1', 'agent': ['general_fc'],
'api_key': 'sk-dbd8a665f7634081b87ec409c7636500', },
'eval_type': 'openai_api', 'mid': {
'generation_config': { 'multi': ['aime24', 'humaneval'],
'temperature': 0.0, 'single': [
'max_tokens': 10240, 'live_code_bench', 'bigcodebench', 'competition_math', 'gsm8k',
'gpqa_diamond', 'mmlu_pro', 'simple_qa', 'longbench_v2', 'openai_mrcr',
],
'agent': ['general_fc', 'tau2_bench'],
},
# 多机组分组,基于 CSV 实测完整时间(已含 multi-run平衡
# Group1: ~22.7h | Group2: ~25.6h | Group3: ~27.0h | 合计 ~75.2h
'group1': {
'multi': ['live_code_bench', 'aime24', 'aime25', 'aime26', 'hmmt26', 'imo_answerbench', 'humaneval'],
'single': ['bigcodebench', 'competition_math', 'gsm8k', 'drop', 'arc', 'hellaswag', 'winogrande'],
'agent': [],
},
'group2': {
'multi': [],
'single': ['hle', 'mmlu_pro', 'trivia_qa'],
'agent': [],
},
'group3': {
'multi': ['gpqa_diamond'],
'single': ['openai_mrcr', 'longbench_v2', 'bfcl_v3', 'mmlu', 'cmmlu', 'bbh', 'simple_qa'],
'agent': ['tau2_bench', 'general_fc'],
}, },
} }
TRUNCATION_CONFIG = { # ============================================================
'longbench_v2': 32768*4, # Fixed configuration
'openai_mrcr': 32768*4, # ============================================================
MATH_DATASETS = {
'aime24', 'aime25', 'aime26', 'hmmt26',
'gsm8k', 'competition_math', 'imo_answerbench',
} }
# Combine all datasets in order (shortest estimated time first)
# 1. Multi-run small benchmarks (temperature=1.0 for sampling diversity)
_multi_run_order = [
# 'humaneval', # ~164 samples x 3 runs = 492
# 'live_code_bench', # ~100 samples x 5 runs = 500
# 'aime26', # ~30 samples x 12 runs = 360
# 'aime24', # ~30 samples x 12 runs = 360
# 'aime25', # ~30 samples x 12 runs = 360
# 'gpqa_diamond', # ~198 samples x 2 runs = 396
# 'imo_answerbench', # ~120 samples x 4 runs = 480
# 'hmmt26', # ~30 samples x 12 runs = 360
]
# 2. Single-run benchmarks (temperature=0.0 greedy) MATH_PROMPT_TEMPLATE = (
_single_run_order = [ "{question}\n"
# 'arc', "Please reason step by step, and put your final answer within \\boxed{{}}."
# 'bfcl_v3', # function calling: greedy decoding )
# 'winogrande',
# 'competition_math',
# 'gsm8k',
# 'hellaswag',
# 'bigcodebench',
# 'drop',
# 'bbh',
# 'openai_mrcr',
# 'longbench_v2',
# 'mmlu',
# 'cmmlu',
# 'simple_qa',
# 'mmlu_pro',
'hle',
# 'trivia_qa',
]
# 3. Agent / tool benchmarks (last)
_agent_order = [
# 'tau2_bench',
# 'general_fc',
]
# ============================================================
# Command line argument overrides (不需要修改)
# ============================================================
for i, arg in enumerate(sys.argv):
if arg == '--limit' and i + 1 < len(sys.argv):
limit_val = sys.argv[i + 1]
if limit_val.lower() == 'none' or limit_val.lower() == 'all':
LIMIT = None
else:
LIMIT = int(limit_val)
elif arg == '--model' and i + 1 < len(sys.argv):
MODEL = sys.argv[i + 1]
elif arg == '--api-url' and i + 1 < len(sys.argv):
API_URL = sys.argv[i + 1]
elif arg == '--dataset-dir' and i + 1 < len(sys.argv):
DATASET_DIR = sys.argv[i + 1]
elif arg == '--output-dir' and i + 1 < len(sys.argv):
OUTPUT_DIR = sys.argv[i + 1]
elif arg == '--config' and i + 1 < len(sys.argv):
CONFIG = sys.argv[i + 1]
# Benchmarks that require sandboxed code execution
SANDBOX_DATASETS = {'humaneval', 'bigcodebench'} SANDBOX_DATASETS = {'humaneval', 'bigcodebench'}
SANDBOX_CONFIGS = { SANDBOX_CONFIGS = {
'bigcodebench': { 'bigcodebench': {
@ -141,91 +169,69 @@ SANDBOX_CONFIGS = {
}, },
} }
# ============================================================
# User-tunable parameters
# ============================================================
# Fixed seed for reproducibility. With temperature > 0 the model still samples
# randomly, so running N times with the same seed still yields variance.
SEED = 42
# Benchmarks to run multiple times with temperature=1.0.
# Format: {benchmark_name: num_runs}
# Number of runs chosen so total samples ≈ 400-500 per benchmark.
MULTI_RUN_CONFIG = {
# ~30 samples each -> 12 runs = ~360 samples
'aime24': 12,
'aime25': 12,
'aime26': 12,
'hmmt26': 12,
# ~100 samples -> 5 runs = ~500 samples
'live_code_bench': 5,
# ~120 samples -> 4 runs = ~480 samples
'imo_answerbench': 4,
# ~164 samples -> 3 runs = ~492 samples
'humaneval': 3,
# ~198 samples -> 2 runs = ~396 samples
'gpqa_diamond': 2,
}
# Single-run benchmarks (temperature=0.0 greedy)
SINGLE_RUN_DATASETS = [
'bigcodebench',
'bfcl_v3', # function calling: greedy decoding
'competition_math',
'gsm8k',
'hle',
'super_gpqa',
'arc',
'bbh',
'cmmlu',
'drop',
'hellaswag',
'mmlu',
'mmlu_pro',
'simple_qa',
'trivia_qa',
'winogrande',
'openai_mrcr', # long-context, run once
'longbench_v2', # long-context, run once
]
# Agent / tool benchmarks (temperature=0.0 greedy, run once)
AGENT_DATASETS = [
'tau2_bench',
'general_fc',
]
DATASETS = _multi_run_order + _single_run_order + _agent_order
# ENABLE_THINKING 已移至文件头部「必改参数」区
BATCH_SIZE_LIST = [4]
# LIMIT is parsed from command line: --limit N
SHUFFLE = True
# ============================================================ # ============================================================
# Fixed configuration # CLI parser
# ============================================================ # ============================================================
CONFIG_PATH = str(Path(CONFIG) if CONFIG else SCRIPT_DIR / 'config' / 'dpv4-int8_nothinking.yaml') def build_parser():
if not Path(CONFIG_PATH).exists(): parser = argparse.ArgumentParser(
raise FileNotFoundError(f'Config file not found: {CONFIG_PATH}') description='Unified EvalScope benchmark runner',
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog='Suites: full, lite, mid, group1, group2, group3',
)
# Model / API
parser.add_argument('--model', default=DEFAULT_MODEL,
help='Served model name (default: %(default)s)')
parser.add_argument('--api-url', default=DEFAULT_API_URL,
help='OpenAI-compatible API URL (default: %(default)s)')
MATH_DATASETS = { # Paths
'aime24', 'aime25', 'aime26', 'hmmt26', parser.add_argument('--dataset-dir', default=DEFAULT_DATASET_DIR,
'gsm8k', 'competition_math', 'imo_answerbench', help='Parent directory containing datasets/ subdir (default: %(default)s)')
} parser.add_argument('--output-dir', default=DEFAULT_OUTPUT_DIR,
help='Output root directory (default: %(default)s)')
parser.add_argument('--config', default=DEFAULT_CONFIG,
help='YAML config path (default: %(default)s)')
parser.add_argument('--tokenizer-path', default=DEFAULT_TOKENIZER_PATH,
help='Local tokenizer path for middle-truncation (default: %(default)s)')
MATH_PROMPT_TEMPLATE = ( # Run control
"{question}\n" parser.add_argument('--suite', default='full', choices=list(SUITES.keys()),
"Please reason step by step, and put your final answer within \\boxed{{}}." help='Benchmark suite to run (default: %(default)s)')
) parser.add_argument('--datasets', '--benchmarks', dest='datasets', default=None,
help='Override suite with comma-separated benchmark names, e.g. aime24,gsm8k')
parser.add_argument('--exclude', default=None,
help='Comma-separated benchmarks to exclude from the chosen suite')
parser.add_argument('--limit', default=None,
help='Max samples per benchmark; "none"/"all" for no limit (default: none)')
parser.add_argument('--seed', type=int, default=DEFAULT_SEED,
help='Random seed (default: %(default)s)')
parser.add_argument('--batch-size', type=int, default=DEFAULT_BATCH_SIZE,
help='Evaluation batch size (default: %(default)s)')
with open(CONFIG_PATH, 'r', encoding='utf-8') as f: # Decoding / thinking
DATASET_CONFIGS = yaml.safe_load(f) parser.add_argument('--thinking', action='store_true', default=None,
help='Enable thinking mode (sglang chat_template_kwargs.thinking=True)')
parser.add_argument('--no-thinking', dest='thinking', action='store_false',
help='Disable thinking mode (default)')
# Judge model
parser.add_argument('--judge-model', default=DEFAULT_JUDGE_MODEL,
help='Judge model name (default: %(default)s)')
parser.add_argument('--judge-api-url', default=DEFAULT_JUDGE_API_URL,
help='Judge model API URL (default: %(default)s)')
parser.add_argument('--judge-api-key', default=DEFAULT_JUDGE_API_KEY,
help='Judge model API key')
parser.add_argument('--judge-max-tokens', type=int, default=DEFAULT_JUDGE_MAX_TOKENS,
help='Judge model max_tokens (default: %(default)s)')
# Truncation
parser.add_argument('--truncation-tokens', type=int, default=DEFAULT_TRUNCATION_TOKENS,
help='Middle-truncation token budget for long-context benchmarks (default: %(default)s)')
return parser
# ============================================================ # ============================================================
@ -235,23 +241,21 @@ with open(CONFIG_PATH, 'r', encoding='utf-8') as f:
_TOKENIZER = None _TOKENIZER = None
def get_tokenizer(): def get_tokenizer(tokenizer_path: str):
global _TOKENIZER global _TOKENIZER
if _TOKENIZER is None: if _TOKENIZER is None:
from transformers import AutoTokenizer from transformers import AutoTokenizer
try: try:
# Try local path first _TOKENIZER = AutoTokenizer.from_pretrained(tokenizer_path, trust_remote_code=True)
_TOKENIZER = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True)
except Exception: except Exception:
# Fallback to HuggingFace model ID
_TOKENIZER = AutoTokenizer.from_pretrained('deepseek-ai/DeepSeek-V4-Flash', trust_remote_code=True) _TOKENIZER = AutoTokenizer.from_pretrained('deepseek-ai/DeepSeek-V4-Flash', trust_remote_code=True)
return _TOKENIZER return _TOKENIZER
def truncate_middle(text: str, max_tokens: int) -> str: def truncate_middle(text: str, max_tokens: int, tokenizer_path: str) -> str:
if max_tokens <= 0: if max_tokens <= 0:
return text return text
tokenizer = get_tokenizer() tokenizer = get_tokenizer(tokenizer_path)
token_ids = tokenizer.encode(text, add_special_tokens=False) token_ids = tokenizer.encode(text, add_special_tokens=False)
if len(token_ids) <= max_tokens: if len(token_ids) <= max_tokens:
return text return text
@ -261,41 +265,35 @@ def truncate_middle(text: str, max_tokens: int) -> str:
return tokenizer.decode(truncated_ids, skip_special_tokens=True) return tokenizer.decode(truncated_ids, skip_special_tokens=True)
def _patch_adapters_for_truncation(): def _patch_adapters_for_truncation(tokenizer_path: str, truncation_tokens: int):
from evalscope.benchmarks.longbench_v2.longbench_v2_adapter import LongBenchV2Adapter from evalscope.benchmarks.longbench_v2.longbench_v2_adapter import LongBenchV2Adapter
from evalscope.benchmarks.openai_mrcr.openai_mrcr_adapter import OpenAIMRCRAdapter from evalscope.benchmarks.openai_mrcr.openai_mrcr_adapter import OpenAIMRCRAdapter
# Patch LongBenchV2Adapter.format_prompt_template
_orig_longbench_format = LongBenchV2Adapter.format_prompt_template _orig_longbench_format = LongBenchV2Adapter.format_prompt_template
def _patched_longbench_format(self, sample): def _patched_longbench_format(self, sample):
max_tok = TRUNCATION_CONFIG.get('longbench_v2') if sample.metadata and 'context' in sample.metadata:
if max_tok and sample.metadata and 'context' in sample.metadata: sample.metadata['context'] = truncate_middle(sample.metadata['context'], truncation_tokens, tokenizer_path)
sample.metadata['context'] = truncate_middle(sample.metadata['context'], max_tok)
return _orig_longbench_format(self, sample) return _orig_longbench_format(self, sample)
LongBenchV2Adapter.format_prompt_template = _patched_longbench_format LongBenchV2Adapter.format_prompt_template = _patched_longbench_format
# Patch OpenAIMRCRAdapter.record_to_sample with needle-aware truncation.
# MRCR is a long chat history with needles hidden at desired_msg_index.
# We keep the head, tail, and a window around the needle, and truncate
# each kept message if it is still too long. This preserves the retrieval
# task while fitting GPU memory.
_orig_mrcr_record = OpenAIMRCRAdapter.record_to_sample _orig_mrcr_record = OpenAIMRCRAdapter.record_to_sample
def _patched_mrcr_record(self, record): def _patched_mrcr_record(self, record):
import json
max_total_tok = TRUNCATION_CONFIG.get('openai_mrcr')
per_msg_max_tok = 8192 per_msg_max_tok = 8192
if max_total_tok and 'prompt' in record: if 'prompt' in record:
try: try:
prompt_data = json.loads(record['prompt']) prompt_data = json.loads(record['prompt'])
if not isinstance(prompt_data, list) or len(prompt_data) == 0: if not isinstance(prompt_data, list) or len(prompt_data) == 0:
return _orig_mrcr_record(self, record) return _orig_mrcr_record(self, record)
tokenizer = get_tokenizer() tokenizer = get_tokenizer(tokenizer_path)
total_tok = sum( total_tok = sum(
len(tokenizer.encode(msg.get('content', '') if isinstance(msg, dict) else '', add_special_tokens=False)) len(tokenizer.encode(msg.get('content', '') if isinstance(msg, dict) else '', add_special_tokens=False))
for msg in prompt_data for msg in prompt_data
) )
if total_tok <= max_total_tok: if total_tok <= truncation_tokens:
return _orig_mrcr_record(self, record) return _orig_mrcr_record(self, record)
desired_idx = record.get('desired_msg_index', 0) desired_idx = record.get('desired_msg_index', 0)
@ -304,10 +302,8 @@ def _patch_adapters_for_truncation():
n = len(prompt_data) n = len(prompt_data)
keep = set() keep = set()
# Head and tail context
keep.update(range(min(2, n))) keep.update(range(min(2, n)))
keep.update(range(max(0, n - 2), n)) keep.update(range(max(0, n - 2), n))
# Window around the needle
window = 2 window = 2
keep.update(range(max(0, desired_idx - window), min(n, desired_idx + window + 1))) keep.update(range(max(0, desired_idx - window), min(n, desired_idx + window + 1)))
keep = sorted(keep) keep = sorted(keep)
@ -319,7 +315,7 @@ def _patch_adapters_for_truncation():
msg = dict(msg) msg = dict(msg)
content = msg.get('content', '') content = msg.get('content', '')
if len(tokenizer.encode(content, add_special_tokens=False)) > per_msg_max_tok: if len(tokenizer.encode(content, add_special_tokens=False)) > per_msg_max_tok:
msg['content'] = truncate_middle(content, per_msg_max_tok) msg['content'] = truncate_middle(content, per_msg_max_tok, tokenizer_path)
new_prompt.append(msg) new_prompt.append(msg)
record = dict(record) record = dict(record)
@ -327,16 +323,21 @@ def _patch_adapters_for_truncation():
except (json.JSONDecodeError, TypeError): except (json.JSONDecodeError, TypeError):
pass pass
return _orig_mrcr_record(self, record) return _orig_mrcr_record(self, record)
OpenAIMRCRAdapter.record_to_sample = _patched_mrcr_record OpenAIMRCRAdapter.record_to_sample = _patched_mrcr_record
_patch_adapters_for_truncation()
# ============================================================ # ============================================================
# Helpers # Helpers
# ============================================================ # ============================================================
def load_dataset_configs(config_path: str):
if not Path(config_path).exists():
raise FileNotFoundError(f'Config file not found: {config_path}')
with open(config_path, 'r', encoding='utf-8') as f:
return yaml.safe_load(f)
def configure_thinking(generation_config: dict, enable: bool) -> dict: def configure_thinking(generation_config: dict, enable: bool) -> dict:
extra_body = generation_config.get('extra_body', {}) extra_body = generation_config.get('extra_body', {})
chat_template_kwargs = extra_body.get('chat_template_kwargs', {}) chat_template_kwargs = extra_body.get('chat_template_kwargs', {})
@ -363,14 +364,24 @@ def build_agent_config(agent_cfg: dict) -> NativeAgentConfig:
return NativeAgentConfig(**agent_cfg) return NativeAgentConfig(**agent_cfg)
def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_thinking: bool, seed: int, run_idx: int = 0) -> TaskConfig: def build_task_config(
# Each run gets its own work_dir so use_cache does not reuse predictions dataset_name: str,
# across repeated samples. This is required for temperature=1.0 multi-run ds_cfg: dict,
# benchmarks to actually measure variance. batch_size: int,
enable_thinking: bool,
seed: int,
limit,
output_dir: str,
model: str,
api_url: str,
dataset_dir: str,
judge_model_args: dict,
run_idx: int = 0,
) -> TaskConfig:
if run_idx > 0: if run_idx > 0:
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}_run_{run_idx}' work_dir = Path(output_dir) / dataset_name / f'seed_{seed}_run_{run_idx}'
else: else:
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}' work_dir = Path(output_dir) / dataset_name / f'seed_{seed}'
work_dir.mkdir(parents=True, exist_ok=True) work_dir.mkdir(parents=True, exist_ok=True)
work_dir = str(work_dir) work_dir = str(work_dir)
@ -388,13 +399,13 @@ def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_t
agent_config = build_agent_config(ds_cfg['agent_config']) agent_config = build_agent_config(ds_cfg['agent_config'])
return TaskConfig( return TaskConfig(
model=MODEL, model=model,
api_url=API_URL, api_url=api_url,
eval_type='openai_api', eval_type='openai_api',
dataset_dir=DATASET_DIR, dataset_dir=dataset_dir,
judge_model_args=JUDGE_MODEL_ARGS, judge_model_args=judge_model_args,
seed=seed, seed=seed,
limit=LIMIT, limit=limit,
collect_perf=True, collect_perf=True,
no_timestamp=True, no_timestamp=True,
work_dir=work_dir, work_dir=work_dir,
@ -419,71 +430,135 @@ def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_t
# ============================================================ # ============================================================
# Main loop # Main
# ============================================================ # ============================================================
def main(): def main():
print(f"Config: {CONFIG_PATH}") parser = build_parser()
print(f"Model: {MODEL}") args = parser.parse_args()
print(f"API URL: {API_URL}")
print(f"Dataset Dir: {DATASET_DIR}")
print(f"Output Dir: {OUTPUT_DIR}")
print(f"Limit: {LIMIT if LIMIT is not None else 'ALL'}")
print(f"Total benchmarks: {len(DATASETS)}")
print("="*60)
for batch_size in BATCH_SIZE_LIST: # Resolve limit
# 1. Run multi-run benchmarks (temperature=1.0, same seed, variance from sampling) limit = args.limit
for dataset_name in _multi_run_order: if limit is not None:
if dataset_name not in DATASET_CONFIGS: if str(limit).lower() in ('none', 'all'):
print(f"WARNING: {dataset_name} not in YAML config, skipping") limit = None
continue else:
ds_cfg = DATASET_CONFIGS[dataset_name] limit = int(limit)
num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1)
for run_idx in range(num_runs):
print(f"\n{'='*60}")
print(f"Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={SEED})")
print(f"{'='*60}")
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED, run_idx=run_idx)
try:
run_task(task_cfg)
except Exception as e:
print(f"ERROR in {dataset_name} (run {run_idx + 1}): {e}")
continue
# 2. Run single-run benchmarks (temperature=0.0 greedy) enable_thinking = DEFAULT_ENABLE_THINKING if args.thinking is None else args.thinking
for dataset_name in _single_run_order:
if dataset_name not in DATASET_CONFIGS: # Resolve suite or custom datasets
print(f"WARNING: {dataset_name} not in YAML config, skipping") if args.datasets:
continue custom = [d.strip() for d in args.datasets.split(',') if d.strip()]
ds_cfg = DATASET_CONFIGS[dataset_name] multi_run = [d for d in custom if d in MULTI_RUN_CONFIG]
single_run = [d for d in custom if d not in MULTI_RUN_CONFIG]
agent = [d for d in custom if d in ALL_AGENT]
single_run = [d for d in single_run if d not in ALL_AGENT]
else:
suite = SUITES[args.suite]
multi_run = list(suite['multi'])
single_run = list(suite['single'])
agent = list(suite['agent'])
# Apply --exclude
if args.exclude:
exclude = {d.strip() for d in args.exclude.split(',') if d.strip()}
multi_run = [d for d in multi_run if d not in exclude]
single_run = [d for d in single_run if d not in exclude]
agent = [d for d in agent if d not in exclude]
judge_model_args = {
'model_id': args.judge_model,
'api_url': args.judge_api_url,
'api_key': args.judge_api_key,
'eval_type': 'openai_api',
'generation_config': {
'temperature': 0.0,
'max_tokens': args.judge_max_tokens,
},
}
truncation_tokens = args.truncation_tokens
_patch_adapters_for_truncation(args.tokenizer_path, truncation_tokens)
dataset_configs = load_dataset_configs(args.config)
print('=' * 60)
print(f'Config: {args.config}')
print(f'Model: {args.model}')
print(f'API URL: {args.api_url}')
print(f'Dataset Dir: {args.dataset_dir}')
print(f'Output Dir: {args.output_dir}')
print(f'Suite: {args.suite}')
print(f'Limit: {limit if limit is not None else "ALL"}')
print(f'Thinking: {enable_thinking}')
print(f'Seed: {args.seed}')
print(f'Batch Size: {args.batch_size}')
print(f'Tokenizer Path: {args.tokenizer_path}')
print(f'Truncation Tokens: {truncation_tokens}')
print(f'Multi-run datasets: {multi_run}')
print(f'Single-run datasets: {single_run}')
print(f'Agent datasets: {agent}')
print('=' * 60)
for dataset_name in multi_run:
if dataset_name not in dataset_configs:
print(f'WARNING: {dataset_name} not in YAML config, skipping')
continue
ds_cfg = dataset_configs[dataset_name]
num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1)
for run_idx in range(num_runs):
print(f"\n{'='*60}") print(f"\n{'='*60}")
print(f"Running: {dataset_name} (seed={SEED})") print(f'Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={args.seed})')
print(f"{'='*60}") print(f"{'='*60}")
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED) task_cfg = build_task_config(
dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit,
args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args,
run_idx=run_idx,
)
try: try:
run_task(task_cfg) run_task(task_cfg)
except Exception as e: except Exception as e:
print(f"ERROR in {dataset_name}: {e}") print(f'ERROR in {dataset_name} (run {run_idx + 1}): {e}')
continue continue
# 3. Run agent benchmarks for dataset_name in single_run:
for dataset_name in _agent_order: if dataset_name not in dataset_configs:
if dataset_name not in DATASET_CONFIGS: print(f'WARNING: {dataset_name} not in YAML config, skipping')
print(f"WARNING: {dataset_name} not in YAML config, skipping") continue
continue ds_cfg = dataset_configs[dataset_name]
ds_cfg = DATASET_CONFIGS[dataset_name] print(f"\n{'='*60}")
print(f"\n{'='*60}") print(f'Running: {dataset_name} (seed={args.seed})')
print(f"Running: {dataset_name} (seed={SEED})") print(f"{'='*60}")
print(f"{'='*60}") task_cfg = build_task_config(
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED) dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit,
try: args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args,
run_task(task_cfg) )
except Exception as e: try:
print(f"ERROR in {dataset_name}: {e}") run_task(task_cfg)
continue except Exception as e:
print(f'ERROR in {dataset_name}: {e}')
continue
print("\nAll benchmarks done!") for dataset_name in agent:
if dataset_name not in dataset_configs:
print(f'WARNING: {dataset_name} not in YAML config, skipping')
continue
ds_cfg = dataset_configs[dataset_name]
print(f"\n{'='*60}")
print(f'Running: {dataset_name} (seed={args.seed})')
print(f"{'='*60}")
task_cfg = build_task_config(
dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit,
args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args,
)
try:
run_task(task_cfg)
except Exception as e:
print(f'ERROR in {dataset_name}: {e}')
continue
print('\nAll benchmarks done!')
if __name__ == '__main__': if __name__ == '__main__':

View File

@ -1,74 +0,0 @@
#!/usr/bin/env python3
"""
Full evaluation suite: ~3-5 days
Runs all benchmarks with full datasets and multiple runs for stability.
Use --limit none (default) for the complete evaluation.
"""
from pathlib import Path
import sys
sys.path.insert(0, str(Path(__file__).parent))
import run as run_module
# Use the default full configuration from run.py
# (already ordered by estimated runtime)
run_module._multi_run_order = [
'humaneval',
'live_code_bench',
'aime26',
'aime24',
'aime25',
'gpqa_diamond',
'imo_answerbench',
'hmmt26',
]
run_module._single_run_order = [
'arc',
'bfcl_v3',
'winogrande',
'competition_math',
'gsm8k',
'hellaswag',
'bigcodebench',
'drop',
'bbh',
'openai_mrcr',
'longbench_v2',
'mmlu',
'cmmlu',
'super_gpqa',
'simple_qa',
'mmlu_pro',
'hle',
'trivia_qa',
]
run_module._agent_order = [
'tau2_bench',
'general_fc',
]
run_module.DATASETS = (
run_module._multi_run_order
+ run_module._single_run_order
+ run_module._agent_order
)
# Full: many runs for statistical stability
run_module.MULTI_RUN_CONFIG = {
'aime24': 12,
'aime25': 12,
'aime26': 12,
'hmmt26': 12,
'live_code_bench': 5,
'imo_answerbench': 4,
'humaneval': 3,
'gpqa_diamond': 2,
}
if __name__ == '__main__':
print('Full benchmarks:', run_module.DATASETS)
print('Estimated time: ~3-5 days with --limit none')
run_module.main()

View File

@ -1,48 +0,0 @@
"""
Group 1: Code + Reasoning + short Knowledge + Agent (est. ~18-22h)
Run on machine 1.
"""
from pathlib import Path
import sys
# Ensure we can import run.py from the same directory inside or outside Docker
sys.path.insert(0, str(Path(__file__).parent))
import run as run_module
# Override dataset lists for group 1
run_module._multi_run_order = [
'humaneval',
'live_code_bench',
'aime26',
'aime24',
'aime25',
'gpqa_diamond',
'imo_answerbench',
'hmmt26',
]
run_module._single_run_order = [
'arc',
'winogrande',
'competition_math',
'gsm8k',
'hellaswag',
'bigcodebench',
]
run_module._agent_order = [
'bfcl_v3',
'tau2_bench',
]
run_module.DATASETS = (
run_module._multi_run_order
+ run_module._single_run_order
+ run_module._agent_order
)
if __name__ == '__main__':
print('Group 1 benchmarks:', run_module.DATASETS)
print('Estimated time: ~18-22h')
run_module.main()

View File

@ -1,40 +0,0 @@
"""
Group 2: Knowledge + Long-context + Agent (est. ~20-24h)
Run on machine 2.
"""
from pathlib import Path
import sys
# Ensure we can import run.py from the same directory inside or outside Docker
sys.path.insert(0, str(Path(__file__).parent))
import run as run_module
# Override dataset lists for group 2
run_module._multi_run_order = []
run_module._single_run_order = [
'drop',
'bbh',
'openai_mrcr',
'longbench_v2',
'mmlu',
'cmmlu',
'super_gpqa',
'simple_qa',
]
run_module._agent_order = [
'general_fc',
]
run_module.DATASETS = (
run_module._multi_run_order
+ run_module._single_run_order
+ run_module._agent_order
)
if __name__ == '__main__':
print('Group 2 benchmarks:', run_module.DATASETS)
print('Estimated time: ~20-24h')
run_module.main()

View File

@ -1,35 +0,0 @@
"""
Group 3: Long-running Knowledge benchmarks (est. ~35-40h)
Run on machine 3.
Note: trivia_qa and hle are inherently slow; consider using --limit to control time.
"""
from pathlib import Path
import sys
# Ensure we can import run.py from the same directory inside or outside Docker
sys.path.insert(0, str(Path(__file__).parent))
import run as run_module
# Override dataset lists for group 3
run_module._multi_run_order = []
run_module._single_run_order = [
'mmlu_pro',
'hle',
'trivia_qa',
]
run_module._agent_order = []
run_module.DATASETS = (
run_module._multi_run_order
+ run_module._single_run_order
+ run_module._agent_order
)
if __name__ == '__main__':
print('Group 3 benchmarks:', run_module.DATASETS)
print('Estimated time: ~35-40h (trivia_qa and hle are slow)')
print('Tip: use --limit 500 to cap runtime if needed')
run_module.main()

View File

@ -1,40 +0,0 @@
#!/usr/bin/env python3
"""
Lite evaluation suite: ~2-4h
Covers all 5 capability domains with 1-2 benchmarks each.
Use --limit 20 (default) for quick smoke testing.
"""
from pathlib import Path
import sys
sys.path.insert(0, str(Path(__file__).parent))
import run as run_module
run_module._multi_run_order = [
'aime24', # reasoning
'humaneval', # code
]
run_module._single_run_order = [
'gsm8k', # reasoning
'mmlu_pro', # knowledge
'simple_qa', # knowledge
'longbench_v2', # long-context
]
run_module._agent_order = [
'bfcl_v3', # tool/function calling
]
run_module.DATASETS = (
run_module._multi_run_order
+ run_module._single_run_order
+ run_module._agent_order
)
if __name__ == '__main__':
print('Lite benchmarks:', run_module.DATASETS)
run_module.main()

View File

@ -1,45 +0,0 @@
#!/usr/bin/env python3
"""Run a single benchmark for quick smoke tests.
Usage:
python bash/run_single.py bigcodebench --limit 1
python bash/run_single.py bfcl_v3 --limit 5 --model MODEL --api-url URL
"""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent))
import run as run_mod
from evalscope import run_task
if len(sys.argv) < 2:
print('Usage: python bash/run_single.py <dataset_name> [--limit N] [--model M] [--api-url U]')
sys.exit(1)
dataset_name = sys.argv[1]
limit = None
for i, arg in enumerate(sys.argv):
if arg == '--limit' and i + 1 < len(sys.argv):
v = sys.argv[i + 1]
limit = None if v.lower() in ('none', 'all') else int(v)
elif arg == '--model' and i + 1 < len(sys.argv):
run_mod.MODEL = sys.argv[i + 1]
elif arg == '--api-url' and i + 1 < len(sys.argv):
run_mod.API_URL = sys.argv[i + 1]
if dataset_name not in run_mod.DATASET_CONFIGS:
raise SystemExit(f'{dataset_name} not in config')
run_mod.LIMIT = limit
ds_cfg = run_mod.DATASET_CONFIGS[dataset_name]
task_cfg = run_mod.build_task_config(
dataset_name=dataset_name,
ds_cfg=ds_cfg,
batch_size=4,
enable_thinking=run_mod.ENABLE_THINKING,
seed=run_mod.SEED,
run_idx=0,
)
print(f'Running {dataset_name} with limit={limit}')
run_task(task_cfg)
print('Done.')

View File

@ -1,497 +0,0 @@
import yaml
from copy import deepcopy
from pathlib import Path
import sys
import os
from evalscope import run_task, TaskConfig
from evalscope.api.agent import NativeAgentConfig
from evalscope.config import SandboxTaskConfig
SCRIPT_DIR = Path(__file__).parent.resolve()
PROJECT_ROOT = SCRIPT_DIR.parent
# ============================================================
# ★★★ 必改参数 (USER CONFIG) ★★★
# 每次评测新模型前,只需要检查/修改以下参数。
# 也可以通过命令行覆盖:--model / --api-url / --dataset-dir /
# --output-dir / --limit / --config
# ============================================================
# 模型名served model name
MODEL = 'DeepSeek-V4-Flash-Int8'
# 模型服务地址OpenAI 兼容 API
API_URL = 'http://localhost:30000/v1'
# 本地数据集根目录(提前下载好的 datasets 目录)
DATASET_DIR = str(PROJECT_ROOT / 'datasets')
# 评测输出目录(每个 benchmark 单独子目录)
OUTPUT_DIR = str(PROJECT_ROOT / 'output')
# 采样数量上限None 表示跑全部样本
LIMIT = None
# 每个 benchmark 的生成参数配置文件max_tokens / temperature 等)
CONFIG = None
# 是否开启 thinking 模式sglang chat_template_kwargs.thinking
ENABLE_THINKING = False
# 测试哪些 benchmark见下方 DATASETS = _multi_run_order + _single_run_order + _agent_order
# 想只跑部分 benchmark 时,注释掉对应行即可。
# ============================================================
# Command line argument overrides (不需要修改)
# ============================================================
for i, arg in enumerate(sys.argv):
if arg == '--limit' and i + 1 < len(sys.argv):
limit_val = sys.argv[i + 1]
if limit_val.lower() == 'none' or limit_val.lower() == 'all':
LIMIT = None
else:
LIMIT = int(limit_val)
elif arg == '--model' and i + 1 < len(sys.argv):
MODEL = sys.argv[i + 1]
elif arg == '--api-url' and i + 1 < len(sys.argv):
API_URL = sys.argv[i + 1]
elif arg == '--dataset-dir' and i + 1 < len(sys.argv):
DATASET_DIR = sys.argv[i + 1]
elif arg == '--output-dir' and i + 1 < len(sys.argv):
OUTPUT_DIR = sys.argv[i + 1]
elif arg == '--config' and i + 1 < len(sys.argv):
CONFIG = sys.argv[i + 1]
# Benchmarks that require sandboxed code execution
SANDBOX_DATASETS = {'humaneval', 'bigcodebench'}
# Per-dataset sandbox configs.
# bigcodebench's upstream image has ENTRYPOINT ["python3", "-m", "bigcodebench.evaluate"],
# which exits immediately and breaks ms_enclave exec. We built a derivative image
# `bigcodebench-sandbox:latest` that keeps the same Python environment but drops the
# entrypoint and runs `tail -f /dev/null` so the container stays alive.
SANDBOX_CONFIGS = {
'bigcodebench': {
'image': 'bigcodebench-sandbox:latest',
'working_dir': '/tmp',
'tools_config': {
'shell_executor': {},
'python_executor': {}
}
},
'humaneval': {
'image': 'python:3.11-slim',
'tools_config': {
'shell_executor': {},
'python_executor': {}
}
},
}
# ============================================================
# User-tunable parameters
# ============================================================
# Fixed seed for reproducibility. With temperature > 0 the model still samples
# randomly, so running N times with the same seed still yields variance.
SEED = 42
# Benchmarks to run multiple times with temperature=1.0.
# Format: {benchmark_name: num_runs}
# Number of runs chosen so total samples ≈ 400-500 per benchmark.
MULTI_RUN_CONFIG = {
# ~30 samples each -> 12 runs = ~360 samples
'aime24': 12,
'aime25': 12,
'aime26': 12,
'hmmt26': 12,
# ~100 samples -> 5 runs = ~500 samples
'live_code_bench': 5,
# ~120 samples -> 4 runs = ~480 samples
'imo_answerbench': 4,
# ~164 samples -> 3 runs = ~492 samples
'humaneval': 3,
# ~198 samples -> 2 runs = ~396 samples
'gpqa_diamond': 2,
}
# Single-run benchmarks (temperature=0.0 greedy)
SINGLE_RUN_DATASETS = [
'bigcodebench',
'bfcl_v3', # function calling: greedy decoding
'competition_math',
'gsm8k',
'hle',
'super_gpqa',
'arc',
'bbh',
'cmmlu',
'drop',
'hellaswag',
'mmlu',
'mmlu_pro',
'simple_qa',
'trivia_qa',
'winogrande',
'openai_mrcr', # long-context, run once
'longbench_v2', # long-context, run once
]
# Agent / tool benchmarks (temperature=0.0 greedy, run once)
AGENT_DATASETS = [
'tau2_bench',
'general_fc',
]
# Combine all datasets in order (shortest estimated time first)
# 1. Multi-run small benchmarks (temperature=1.0 for sampling diversity)
_multi_run_order = [
# 'humaneval', # ~164 samples x 3 runs = 492
# 'live_code_bench', # ~100 samples x 5 runs = 500
# 'aime26', # ~30 samples x 12 runs = 360
# 'aime24', # ~30 samples x 12 runs = 360
# 'aime25', # ~30 samples x 12 runs = 360
# 'gpqa_diamond', # ~198 samples x 2 runs = 396
# 'imo_answerbench', # ~120 samples x 4 runs = 480
# 'hmmt26', # ~30 samples x 12 runs = 360
]
# 2. Single-run benchmarks (temperature=0.0 greedy)
_single_run_order = [
# 'arc',
# 'bfcl_v3', # function calling: greedy decoding
# 'winogrande',
# 'competition_math',
# 'gsm8k',
# 'hellaswag',
'bigcodebench',
# 'drop',
# 'bbh',
# 'openai_mrcr',
# 'longbench_v2',
# 'mmlu',
# 'cmmlu',
# 'super_gpqa',
# 'simple_qa',
# 'mmlu_pro',
'hle',
# 'trivia_qa',
]
# 3. Agent / tool benchmarks (last)
_agent_order = [
# 'tau2_bench',
# 'general_fc',
]
DATASETS = _multi_run_order + _single_run_order + _agent_order
# ENABLE_THINKING 已移至文件头部「必改参数」区
BATCH_SIZE_LIST = [4]
# LIMIT is parsed from command line: --limit N
SHUFFLE = True
# ============================================================
# Fixed configuration
# ============================================================
CONFIG_PATH = str(Path(CONFIG) if CONFIG else SCRIPT_DIR / 'config' / 'dpv4-int8_nothinking.yaml')
if not Path(CONFIG_PATH).exists():
raise FileNotFoundError(f'Config file not found: {CONFIG_PATH}')
MODEL_PATH = '/data1/models/DeepSeek-V4-Flash-INT8'
# Judge model for benchmarks that require LLM-as-judge (simple_qa, tau2_bench, etc.)
# Vectron endpoint with DeepSeek-V4-Pro
JUDGE_MODEL_ARGS = {
'model_id': 'DeepSeek/DeepSeek-V4-Pro',
'api_url': 'https://api.vectron.meta-stone.com/v1',
'api_key': 'sk-dbd8a665f7634081b87ec409c7636500',
'eval_type': 'openai_api',
'generation_config': {
'temperature': 0.0,
'max_tokens': 10240,
},
}
TRUNCATION_CONFIG = {
'longbench_v2': 32768*4,
'openai_mrcr': 32768*4,
}
MATH_DATASETS = {
'aime24', 'aime25', 'aime26', 'hmmt26',
'gsm8k', 'competition_math', 'imo_answerbench',
}
MATH_PROMPT_TEMPLATE = (
"{question}\n"
"Please reason step by step, and put your final answer within \\boxed{{}}."
)
with open(CONFIG_PATH, 'r', encoding='utf-8') as f:
DATASET_CONFIGS = yaml.safe_load(f)
# ============================================================
# Middle-truncation helpers
# ============================================================
_TOKENIZER = None
def get_tokenizer():
global _TOKENIZER
if _TOKENIZER is None:
from transformers import AutoTokenizer
try:
# Try local path first
_TOKENIZER = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True)
except Exception:
# Fallback to HuggingFace model ID
_TOKENIZER = AutoTokenizer.from_pretrained('deepseek-ai/DeepSeek-V4-Flash', trust_remote_code=True)
return _TOKENIZER
def truncate_middle(text: str, max_tokens: int) -> str:
if max_tokens <= 0:
return text
tokenizer = get_tokenizer()
token_ids = tokenizer.encode(text, add_special_tokens=False)
if len(token_ids) <= max_tokens:
return text
keep_head = max_tokens // 2
keep_tail = max_tokens - keep_head
truncated_ids = token_ids[:keep_head] + token_ids[-keep_tail:]
return tokenizer.decode(truncated_ids, skip_special_tokens=True)
def _patch_adapters_for_truncation():
from evalscope.benchmarks.longbench_v2.longbench_v2_adapter import LongBenchV2Adapter
from evalscope.benchmarks.openai_mrcr.openai_mrcr_adapter import OpenAIMRCRAdapter
# Patch LongBenchV2Adapter.format_prompt_template
_orig_longbench_format = LongBenchV2Adapter.format_prompt_template
def _patched_longbench_format(self, sample):
max_tok = TRUNCATION_CONFIG.get('longbench_v2')
if max_tok and sample.metadata and 'context' in sample.metadata:
sample.metadata['context'] = truncate_middle(sample.metadata['context'], max_tok)
return _orig_longbench_format(self, sample)
LongBenchV2Adapter.format_prompt_template = _patched_longbench_format
# Patch OpenAIMRCRAdapter.record_to_sample with needle-aware truncation.
# MRCR is a long chat history with needles hidden at desired_msg_index.
# We keep the head, tail, and a window around the needle, and truncate
# each kept message if it is still too long. This preserves the retrieval
# task while fitting GPU memory.
_orig_mrcr_record = OpenAIMRCRAdapter.record_to_sample
def _patched_mrcr_record(self, record):
import json
max_total_tok = TRUNCATION_CONFIG.get('openai_mrcr')
per_msg_max_tok = 8192
if max_total_tok and 'prompt' in record:
try:
prompt_data = json.loads(record['prompt'])
if not isinstance(prompt_data, list) or len(prompt_data) == 0:
return _orig_mrcr_record(self, record)
tokenizer = get_tokenizer()
total_tok = sum(
len(tokenizer.encode(msg.get('content', '') if isinstance(msg, dict) else '', add_special_tokens=False))
for msg in prompt_data
)
if total_tok <= max_total_tok:
return _orig_mrcr_record(self, record)
desired_idx = record.get('desired_msg_index', 0)
if not isinstance(desired_idx, int) or desired_idx < 0 or desired_idx >= len(prompt_data):
desired_idx = 0
n = len(prompt_data)
keep = set()
# Head and tail context
keep.update(range(min(2, n)))
keep.update(range(max(0, n - 2), n))
# Window around the needle
window = 2
keep.update(range(max(0, desired_idx - window), min(n, desired_idx + window + 1)))
keep = sorted(keep)
new_prompt = []
for idx in keep:
msg = prompt_data[idx]
if isinstance(msg, dict):
msg = dict(msg)
content = msg.get('content', '')
if len(tokenizer.encode(content, add_special_tokens=False)) > per_msg_max_tok:
msg['content'] = truncate_middle(content, per_msg_max_tok)
new_prompt.append(msg)
record = dict(record)
record['prompt'] = json.dumps(new_prompt)
except (json.JSONDecodeError, TypeError):
pass
return _orig_mrcr_record(self, record)
OpenAIMRCRAdapter.record_to_sample = _patched_mrcr_record
_patch_adapters_for_truncation()
# ============================================================
# Helpers
# ============================================================
def configure_thinking(generation_config: dict, enable: bool) -> dict:
extra_body = generation_config.get('extra_body', {})
chat_template_kwargs = extra_body.get('chat_template_kwargs', {})
if enable:
chat_template_kwargs['thinking'] = True
else:
chat_template_kwargs.pop('thinking', None)
if chat_template_kwargs:
extra_body['chat_template_kwargs'] = chat_template_kwargs
if extra_body:
generation_config['extra_body'] = extra_body
return generation_config
def build_agent_config(agent_cfg: dict) -> NativeAgentConfig:
agent_cfg = deepcopy(agent_cfg or {})
known_fields = {'mode', 'strategy', 'tools', 'max_steps', 'mcp_servers', 'environment', 'environment_extra'}
kwargs = agent_cfg.pop('kwargs', {})
for key in list(agent_cfg.keys()):
if key not in known_fields:
kwargs[key] = agent_cfg.pop(key)
if kwargs:
agent_cfg['kwargs'] = kwargs
return NativeAgentConfig(**agent_cfg)
def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_thinking: bool, seed: int, run_idx: int = 0) -> TaskConfig:
# Each run gets its own work_dir so use_cache does not reuse predictions
# across repeated samples. This is required for temperature=1.0 multi-run
# benchmarks to actually measure variance.
if run_idx > 0:
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}_run_{run_idx}'
else:
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}'
work_dir.mkdir(parents=True, exist_ok=True)
work_dir = str(work_dir)
generation_config = configure_thinking(deepcopy(ds_cfg['generation_config']), enable_thinking)
dataset_args = deepcopy(ds_cfg.get('dataset_args', {}))
dataset_args.setdefault('shuffle', True)
if dataset_name in MATH_DATASETS:
dataset_args['prompt_template'] = MATH_PROMPT_TEMPLATE
dataset_args_dict = {dataset_name: dataset_args}
agent_config = None
if 'agent_config' in ds_cfg:
agent_config = build_agent_config(ds_cfg['agent_config'])
return TaskConfig(
model=MODEL,
api_url=API_URL,
eval_type='openai_api',
dataset_dir=DATASET_DIR,
judge_model_args=JUDGE_MODEL_ARGS,
seed=seed,
limit=LIMIT,
collect_perf=True,
no_timestamp=True,
work_dir=work_dir,
use_cache=work_dir,
datasets=[dataset_name],
generation_config=generation_config,
dataset_args=dataset_args_dict,
agent_config=agent_config,
eval_batch_size=batch_size,
sandbox=SandboxTaskConfig(
enabled=True,
engine='docker',
default_config=SANDBOX_CONFIGS.get(dataset_name, {
'image': 'python:3.11-slim',
'tools_config': {
'shell_executor': {},
'python_executor': {}
}
})
) if dataset_name in SANDBOX_DATASETS else None,
)
# ============================================================
# Main loop
# ============================================================
def main():
print(f"Config: {CONFIG_PATH}")
print(f"Model: {MODEL}")
print(f"API URL: {API_URL}")
print(f"Dataset Dir: {DATASET_DIR}")
print(f"Output Dir: {OUTPUT_DIR}")
print(f"Limit: {LIMIT if LIMIT is not None else 'ALL'}")
print(f"Total benchmarks: {len(DATASETS)}")
print("="*60)
for batch_size in BATCH_SIZE_LIST:
# 1. Run multi-run benchmarks (temperature=1.0, same seed, variance from sampling)
for dataset_name in _multi_run_order:
if dataset_name not in DATASET_CONFIGS:
print(f"WARNING: {dataset_name} not in YAML config, skipping")
continue
ds_cfg = DATASET_CONFIGS[dataset_name]
num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1)
for run_idx in range(num_runs):
print(f"\n{'='*60}")
print(f"Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={SEED})")
print(f"{'='*60}")
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED, run_idx=run_idx)
try:
run_task(task_cfg)
except Exception as e:
print(f"ERROR in {dataset_name} (run {run_idx + 1}): {e}")
continue
# 2. Run single-run benchmarks (temperature=0.0 greedy)
for dataset_name in _single_run_order:
if dataset_name not in DATASET_CONFIGS:
print(f"WARNING: {dataset_name} not in YAML config, skipping")
continue
ds_cfg = DATASET_CONFIGS[dataset_name]
print(f"\n{'='*60}")
print(f"Running: {dataset_name} (seed={SEED})")
print(f"{'='*60}")
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED)
try:
run_task(task_cfg)
except Exception as e:
print(f"ERROR in {dataset_name}: {e}")
continue
# 3. Run agent benchmarks
for dataset_name in _agent_order:
if dataset_name not in DATASET_CONFIGS:
print(f"WARNING: {dataset_name} not in YAML config, skipping")
continue
ds_cfg = DATASET_CONFIGS[dataset_name]
print(f"\n{'='*60}")
print(f"Running: {dataset_name} (seed={SEED})")
print(f"{'='*60}")
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED)
try:
run_task(task_cfg)
except Exception as e:
print(f"ERROR in {dataset_name}: {e}")
continue
print("\nAll benchmarks done!")
if __name__ == '__main__':
main()

26
scripts/run_docker_full.sh Executable file
View File

@ -0,0 +1,26 @@
#!/bin/bash
# Docker 全量评测Full
# 请根据实际机器修改 MODEL / API_URL / HOST_EVALSCOPE 路径
MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}"
API_URL="${API_URL:-http://localhost:30000/v1}"
HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}"
OUTPUT_DIR="${OUTPUT_DIR:-/opt/evalscope/output_full}"
docker run -it --rm \
--network host \
-v "${HOST_EVALSCOPE}:/opt/evalscope" \
-v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \
-v "${HOST_EVALSCOPE}/output_full:/opt/evalscope/output_full" \
-v /var/run/docker.sock:/var/run/docker.sock \
evalscope-complete-py312:latest \
bash -c "
cd /opt/evalscope &&
python bash/run.py \
--model ${MODEL} \
--api-url ${API_URL} \
--dataset-dir /opt/evalscope \
--output-dir ${OUTPUT_DIR} \
--suite full \
--limit none
"

27
scripts/run_docker_group.sh Executable file
View File

@ -0,0 +1,27 @@
#!/bin/bash
# Docker 多机分组评测
# 用法GROUP=1 ./run_docker_group.sh
GROUP="${GROUP:-1}"
MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}"
API_URL="${API_URL:-http://localhost:30000/v1}"
HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}"
OUTPUT_DIR="/opt/evalscope/output_group${GROUP}"
docker run -it --rm \
--network host \
-v "${HOST_EVALSCOPE}:/opt/evalscope" \
-v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \
-v "${HOST_EVALSCOPE}/output_group${GROUP}:/opt/evalscope/output_group${GROUP}" \
-v /var/run/docker.sock:/var/run/docker.sock \
evalscope-complete-py312:latest \
bash -c "
cd /opt/evalscope &&
python bash/run.py \
--model ${MODEL} \
--api-url ${API_URL} \
--dataset-dir /opt/evalscope \
--output-dir ${OUTPUT_DIR} \
--suite group${GROUP} \
--limit none
"

25
scripts/run_docker_lite.sh Executable file
View File

@ -0,0 +1,25 @@
#!/bin/bash
# Docker Lite 快速冒烟
# 方式一:使用 --suite lite
MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}"
API_URL="${API_URL:-http://localhost:30000/v1}"
HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}"
docker run -it --rm \
--network host \
-v "${HOST_EVALSCOPE}:/opt/evalscope" \
-v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \
-v "${HOST_EVALSCOPE}/output_lite:/opt/evalscope/output_lite" \
-v /var/run/docker.sock:/var/run/docker.sock \
evalscope-complete-py312:latest \
bash -c "
cd /opt/evalscope &&
python bash/run.py \
--model ${MODEL} \
--api-url ${API_URL} \
--dataset-dir /opt/evalscope \
--output-dir /opt/evalscope/output_lite \
--suite lite \
--limit none
"