all
This commit is contained in:
parent
cfa58f869f
commit
03b49a39d0
29
P800模型能力评测结果 - DS4-Flash-INT8-NO-Thinking-2.0-FULL.csv
Normal file
29
P800模型能力评测结果 - DS4-Flash-INT8-NO-Thinking-2.0-FULL.csv
Normal file
@ -0,0 +1,29 @@
|
|||||||
|
分类,Benchmark,得分,实测时间(h),总样本数,延迟_mean(s),输出TPS,请求QPS,输入tokens_mean,输出tokens_mean,累计总tokens,TTFT_mean(s),TTFT P90,TTFT P99,TPOT_mean(s),TPOT P90,TPOT P99,,,,,,,,,,,,,,,
|
||||||
|
代码与工程,bigcodebench,0.9956,0.9915,1140,10.15508,32.99,0.0985,151.77,335.03,554946,0.26441,0.27379,0.42952,0.02957,0.03059,0.03143,,,,,,,,,,,,,,,
|
||||||
|
,humaneval,0.9044,0.1183,164,2.776502,42.58667,0.3602,163.0793,118.2398,45994,0.25521,0.304186,0.451449,0.021783,0.024498,0.027453,,,,,,,,,,,,,,,
|
||||||
|
,live_code_bench,0.6756,7.9486,100,10.41723,43.578,0.09644,518.1118,453.5784,1040894,0.253597,0.272382,0.464771,0.02093,0.023769,0.026398,,,,,,,,,,,,,,,
|
||||||
|
推理与数学,aime24,0.6167,1.2217,30,42.44225,44.85,0.024108,119.3333,1901.842,55169,0.258426,0.274508,0.444511,0.021698,0.025218,hle,,,,,,,,,,,,,,,
|
||||||
|
,aime25,0.4639,2.0414,30,59.98314,46.03583,0.017092,188.4333,2761.283,80482,0.257489,0.273448,0.4417,0.021655,0.024778,0.026626,,,,,,,,,,,,,,,
|
||||||
|
,aime26,0.5611,1.4847,30,51.64098,44.66417,0.019925,139.7667,2305.069,60102,0.25465,0.268527,0.435013,0.021667,0.024596,0.026089,,,,,,,,,,,,,,,
|
||||||
|
,hmmt26,0.4015,1.5481,30,50.8562,44.73,0.020008,108.4242,2274.237,63775,0.255456,0.266018,0.431919,0.022514,0.025679,0.027263,,,,,,,,,,,,,,,
|
||||||
|
,imo_answerbench,0.36,1.2508,400,39.6005,44.94,0.0253,122.1,1779.8,753159,0.2466,0.2641,0.3167,0.0226,0.0257,0.0285,,,,,,,,,,,,,,,
|
||||||
|
,hle,0.0508,14.6863,3000,61.787217,34.42,0.0162,292.3252,2126.7916,6047792,0.275126,0.282138,0.474372,0.028997,0.029831,0.031026,,,,,,,,,,,,,,,
|
||||||
|
,gsm8k,0.9704,0.3178,1319,3.031098,43.8,0.3299,582.9507,132.7726,944039,0.253185,0.267307,0.45469,0.021127,0.02357,0.025568,,,,,,,,,,,,,,,
|
||||||
|
,competition_math,0.941,3.1817,500,8.681756,51,0.1152,296.5698,442.732,3696509,0.245456,0.263328,0.435642,0.019541,0.021677,0.024484,,,,,,,,,,,,,,,
|
||||||
|
,bbh,0.9032,1.8036,6513,3.561695,42.95,0.2808,916.98,152.9719,6966457,0.269139,0.276913,0.467974,0.023935,0.030921,0.065977,,,,,,,,,,,,,,,
|
||||||
|
,drop,0.8183,1.5336,9536,1.824962,36.31,0.548,1273.607,66.25849,12776955,0.297489,0.4593,0.505599,0.023644,0.028072,0.034489,,,,,,,,,,,,,,,
|
||||||
|
知识与语言理解,gpqa_diamond,0.6944,0.3164,198,10.54948,41.335,0.0948,252.4343,436.0202,135770,0.245972,0.26274,0.456931,0.023819,0.027653,0.030237,,,,,,,,,,,,,,,
|
||||||
|
,mmlu_pro,0.8285,5.6167,12032,6.264368,31.05,0.1596,1312.528,194.5092,18132672,0.293428,0.437219,0.477191,0.031274,0.034267,0.036824,,,,,,,,,,,,,,,
|
||||||
|
,simple_qa,0.3814,1.9014,4326,1.88825,30.69,0.529590891,30.351826,57.955155,382016,0.26334,0.269454,0.452375,0.027793,0.031603,0.035322,,,,,,,,,,,,,,,
|
||||||
|
,mmlu,0.9045,3.0297,14042,2.6957,35.98,0.371,727.6,97,11532337,0.2643,0.2751,0.4663,0.026,0.0305,0.0348,,,,,,,,,,,,,,,
|
||||||
|
,cmmlu,0.9001,3.4935,11515,3.9225,36.65,0.2549,115.3,143.8,2983005,0.2553,0.2956,0.443,0.0266,0.0307,0.0353,,,,,,,,,,,,,,,
|
||||||
|
,arc,0.9476,0.2103,1172,0.662452,13.65,1.5095,102.0333,9.042841,394098,0.321567,0.437672,0.449777,0.066642,0.129754,0.132577,,,,,,,,,,,,,,,
|
||||||
|
,hellaswag,0.8666,0.4453,10042,0.597052,7.32,1.6749,219.8686,4.370444,2251808,0.341421,0.448449,0.461157,0.081246,0.133338,0.136106,,,,,,,,,,,,,,,
|
||||||
|
,trivia_qa,0.7768,5.3011,11313,9.163784,3.78,0.1091,11964.972,34.6152,95912700,4.722344,8.951851,16.512039,0.131478,0.292489,0.622806,,,,,,,,,,,,,,,
|
||||||
|
,winogrande,0.7845,0.0775,1267,0.695127,13.88,1.4386,76.11997,9.646409,108666,0.318495,0.433513,0.438349,0.069826,0.129328,0.130936,,,,,,,,,,,,,,,
|
||||||
|
长上下文,longbench_v2,0.5686,3.0638,503,84.120374,2.75,0.0119,85638.8191,231.6441,43192843,25.148194,53.202063,64.267424,0.248375,0.534405,0.946107,,,,,,,,,,,,,,,
|
||||||
|
,openai_mrcr,0.7768,5.2828,2399,29.1532,12.95,0.0343,24774.8,377.6,60340773,6.6088,21.0867,37.983,0.0608,0.1125,0.2004,,,,,,,,,,,,,,,
|
||||||
|
智能体与工具,tau2_bench,0.7684,1.7625,,,,,,,,,,,,,,,,,,,,,,,,,,,,
|
||||||
|
,general_fc,0.6988,1.3803,2000,9.4082,39.21,0.1063,1619.2,368.9,3976219,0.4226,0.7249,1.2941,0.026,0.0307,0.0398,,,,,,,,,,,,,,,
|
||||||
|
,bfcl_v3,0.6643,5.23,4441,4.296605,25.32,0.2327,7268.4069,108.7864,97061732,0.418217,0.521432,1.106471,0.035785,0.039796,0.053177,,,,,,,,,,,,,,,
|
||||||
|
,总计,0.711992593,75.2394,,,,,,,,,,,,,,,,,,,,,,,,,,,,
|
||||||
|
@ -1,94 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""Fill simple_qa (fresh re-run) + bfcl_v3 (official OVERALL) into the Excel.
|
|
||||||
|
|
||||||
simple_qa: old cells had D=0.0005 (cache-resume artifact) and empty perf.
|
|
||||||
Overwrite C/D/F-Q from the NEW report (real perf + ~1.9h duration).
|
|
||||||
bfcl_v3 : top-level score 0.6643 is evalscope's macro-average (non-standard).
|
|
||||||
Official BFCL score = OVERALL subset = 0.569. Update C only.
|
|
||||||
"""
|
|
||||||
import json
|
|
||||||
import glob
|
|
||||||
import openpyxl
|
|
||||||
|
|
||||||
XLSX = '/data1/sora/P800模型能力评测结果_统一格式_filled.xlsx'
|
|
||||||
SHEETS = ['2.0-FULL', '2.0-Lite']
|
|
||||||
|
|
||||||
# --- read reports ---
|
|
||||||
sq = json.load(open('/data1/sora/evalscope/output/simple_qa/seed_42/reports/DeepSeek-V4-Flash-Int8/simple_qa.json'))
|
|
||||||
bc = json.load(open(glob.glob('/data1/sora/evalscope/output/bfcl_v3/seed_42.bak/reports/DeepSeek-V4-Flash-Int8/bfcl_v3.json')[0]))
|
|
||||||
|
|
||||||
|
|
||||||
def perf(d):
|
|
||||||
pm = d.get('perf_metrics') or {}
|
|
||||||
s = pm.get('summary', {}) or {}
|
|
||||||
u = s.get('usage', {}) or {}
|
|
||||||
tf = s.get('ttft') or {}
|
|
||||||
tp = s.get('tpot') or {}
|
|
||||||
lat = (s.get('latency') or {}).get('mean')
|
|
||||||
return {
|
|
||||||
'score': d.get('score'),
|
|
||||||
'duration_h': (d.get('duration_sec') or 0) / 3600,
|
|
||||||
'latency': lat,
|
|
||||||
'tps': (s.get('throughput') or {}).get('avg_output_tps'),
|
|
||||||
'qps': (1.0 / lat) if lat else None,
|
|
||||||
'in_tok': (u.get('input_tokens') or {}).get('mean'),
|
|
||||||
'out_tok': (u.get('output_tokens') or {}).get('mean'),
|
|
||||||
'ttc': u.get('total_tokens_count'),
|
|
||||||
'ttft_m': tf.get('mean'),
|
|
||||||
'ttft90': tf.get('90%') or tf.get('p90'),
|
|
||||||
'ttft99': tf.get('99%') or tf.get('p99'),
|
|
||||||
'tpot_m': tp.get('mean'),
|
|
||||||
'tpot90': tp.get('90%') or tp.get('p90'),
|
|
||||||
'tpot99': tp.get('99%') or tp.get('p99'),
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def bfcl_overall(d):
|
|
||||||
for metric in d.get('metrics', []):
|
|
||||||
for cat in metric.get('categories', []):
|
|
||||||
for sub in cat.get('subsets', []):
|
|
||||||
nm = sub.get('name')
|
|
||||||
if nm == 'OVERALL' or (isinstance(nm, list) and 'OVERALL' in nm):
|
|
||||||
return sub.get('score')
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
P = perf(sq)
|
|
||||||
BC_OVERALL = bfcl_overall(bc)
|
|
||||||
print('simple_qa:', {k: (round(v, 4) if isinstance(v, float) else v) for k, v in P.items()})
|
|
||||||
print('bfcl_v3 OVERALL:', BC_OVERALL)
|
|
||||||
|
|
||||||
# col -> simple_qa perf key
|
|
||||||
SQ = {'C': P['score'], 'D': round(P['duration_h'], 4), 'F': P['latency'], 'G': P['tps'],
|
|
||||||
'H': P['qps'], 'I': P['in_tok'], 'J': P['out_tok'], 'K': P['ttc'],
|
|
||||||
'L': P['ttft_m'], 'M': P['ttft90'], 'N': P['ttft99'],
|
|
||||||
'O': P['tpot_m'], 'P': P['tpot90'], 'Q': P['tpot99']}
|
|
||||||
|
|
||||||
wb = openpyxl.load_workbook(XLSX)
|
|
||||||
for sn in SHEETS:
|
|
||||||
ws = wb[sn]
|
|
||||||
for r in range(2, 30):
|
|
||||||
b = ws.cell(r, 2).value
|
|
||||||
if b == 'simple_qa':
|
|
||||||
for col, v in SQ.items():
|
|
||||||
if v is not None:
|
|
||||||
ws[f'{col}{r}'] = int(round(v)) if col == 'K' else round(v, 4) if col in ('C', 'D') else v
|
|
||||||
print(f'{sn} simple_qa row{r}: C={ws[f"C{r}"].value} D={ws[f"D{r}"].value} F={ws[f"F{r}"].value} K={ws[f"K{r}"].value}')
|
|
||||||
elif b == 'bfcl_v3' and BC_OVERALL is not None:
|
|
||||||
ws.cell(r, 3).value = round(BC_OVERALL, 4)
|
|
||||||
print(f'{sn} bfcl_v3 row{r}: C={ws.cell(r,3).value}')
|
|
||||||
# recompute total D
|
|
||||||
tr = next((rr for rr in range(2, ws.max_row + 1) if ws.cell(rr, 2).value == '总计'), None)
|
|
||||||
if tr:
|
|
||||||
tot = 0.0
|
|
||||||
for rr in range(2, tr):
|
|
||||||
v = ws.cell(rr, 4).value
|
|
||||||
try:
|
|
||||||
tot += float(v)
|
|
||||||
except (TypeError, ValueError):
|
|
||||||
pass
|
|
||||||
ws.cell(tr, 4).value = round(tot, 4)
|
|
||||||
print(f'{sn} 总计 D = {round(tot, 4)}h')
|
|
||||||
|
|
||||||
wb.save(XLSX)
|
|
||||||
print('saved ->', XLSX)
|
|
||||||
@ -1,38 +0,0 @@
|
|||||||
"""Re-run only the review stage for bigcodebench using existing predictions.
|
|
||||||
|
|
||||||
The upstream `bigcodebench/bigcodebench-evaluate` image has an ENTRYPOINT that
|
|
||||||
runs the evaluator and exits, which kills the ms_enclave sandbox containers.
|
|
||||||
We use the locally built `bigcodebench-sandbox:latest` image (same libraries,
|
|
||||||
entrypoint dropped, runs `tail -f /dev/null`) and re-score the cached
|
|
||||||
predictions without regenerating them.
|
|
||||||
"""
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
sys.path.insert(0, str(Path(__file__).parent))
|
|
||||||
import run as run_mod
|
|
||||||
from evalscope import run_task
|
|
||||||
|
|
||||||
DATASET_NAME = 'bigcodebench'
|
|
||||||
BATCH_SIZE = 4
|
|
||||||
|
|
||||||
if DATASET_NAME not in run_mod.DATASET_CONFIGS:
|
|
||||||
raise SystemExit(f'{DATASET_NAME} not found in config')
|
|
||||||
|
|
||||||
ds_cfg = run_mod.DATASET_CONFIGS[DATASET_NAME]
|
|
||||||
task_cfg = run_mod.build_task_config(
|
|
||||||
dataset_name=DATASET_NAME,
|
|
||||||
ds_cfg=ds_cfg,
|
|
||||||
batch_size=BATCH_SIZE,
|
|
||||||
enable_thinking=run_mod.ENABLE_THINKING,
|
|
||||||
seed=run_mod.SEED,
|
|
||||||
run_idx=0,
|
|
||||||
)
|
|
||||||
task_cfg.rerun_review = True
|
|
||||||
|
|
||||||
print(f'Re-running review for {DATASET_NAME}')
|
|
||||||
print(f'Cache/work dir: {task_cfg.work_dir}')
|
|
||||||
print(f'Sandbox config: {task_cfg.sandbox.default_config}')
|
|
||||||
|
|
||||||
run_task(task_cfg)
|
|
||||||
print('Done.')
|
|
||||||
607
bash/run.py
607
bash/run.py
@ -1,8 +1,40 @@
|
|||||||
import yaml
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
Unified benchmark runner for EvalScope.
|
||||||
|
|
||||||
|
A single entry point for lite / mid / full / group1 / group2 / group3 evaluations.
|
||||||
|
All tunable parameters can be controlled via command-line arguments.
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
# Full evaluation (all benchmarks, multi-run for stability)
|
||||||
|
python bash/run.py \
|
||||||
|
--model DeepSeek-V4-Flash-Int8 \
|
||||||
|
--api-url http://localhost:30000/v1 \
|
||||||
|
--dataset-dir /data1/sora/evalscope \
|
||||||
|
--output-dir /data1/sora/evalscope/output \
|
||||||
|
--suite full \
|
||||||
|
--limit none
|
||||||
|
|
||||||
|
# Lite smoke test (~5h with full samples)
|
||||||
|
python bash/run.py --suite lite --limit none
|
||||||
|
|
||||||
|
# Run only selected benchmarks
|
||||||
|
python bash/run.py --datasets aime24,gsm8k,arc --limit 20
|
||||||
|
|
||||||
|
# Custom judge model
|
||||||
|
python bash/run.py \
|
||||||
|
--judge-model deepseek-v4-pro \
|
||||||
|
--judge-api-url https://api.deepseek.com/v1 \
|
||||||
|
--judge-api-key sk-xxx
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
import sys
|
|
||||||
import os
|
import yaml
|
||||||
|
|
||||||
from evalscope import run_task, TaskConfig
|
from evalscope import run_task, TaskConfig
|
||||||
from evalscope.api.agent import NativeAgentConfig
|
from evalscope.api.agent import NativeAgentConfig
|
||||||
@ -12,116 +44,112 @@ SCRIPT_DIR = Path(__file__).parent.resolve()
|
|||||||
PROJECT_ROOT = SCRIPT_DIR.parent
|
PROJECT_ROOT = SCRIPT_DIR.parent
|
||||||
|
|
||||||
# ============================================================
|
# ============================================================
|
||||||
# ★★★ 必改参数 (USER CONFIG) ★★★
|
# Default configuration (override via CLI)
|
||||||
# 每次评测新模型前,只需要检查/修改以下参数。
|
|
||||||
# 也可以通过命令行覆盖:--model / --api-url / --dataset-dir /
|
|
||||||
# --output-dir / --limit / --config
|
|
||||||
# ============================================================
|
# ============================================================
|
||||||
|
|
||||||
# 模型名(served model name)
|
DEFAULT_MODEL = 'DeepSeek-V4-Flash-Int8'
|
||||||
MODEL = 'DeepSeek-V4-Flash-Int8'
|
DEFAULT_API_URL = 'http://localhost:30000/v1'
|
||||||
|
DEFAULT_DATASET_DIR = str(PROJECT_ROOT)
|
||||||
|
DEFAULT_OUTPUT_DIR = str(PROJECT_ROOT / 'output')
|
||||||
|
DEFAULT_CONFIG = str(PROJECT_ROOT / 'config' / 'dpv4-int8_nothinking.yaml')
|
||||||
|
DEFAULT_TOKENIZER_PATH = '/data1/models/DeepSeek-V4-Flash-INT8'
|
||||||
|
DEFAULT_LIMIT = None
|
||||||
|
DEFAULT_SEED = 42
|
||||||
|
DEFAULT_BATCH_SIZE = 4
|
||||||
|
DEFAULT_ENABLE_THINKING = False
|
||||||
|
|
||||||
# 模型服务地址(OpenAI 兼容 API)
|
DEFAULT_JUDGE_MODEL = 'DeepSeek/DeepSeek-V4-Pro'
|
||||||
API_URL = 'http://localhost:30000/v1'
|
DEFAULT_JUDGE_API_URL = 'https://api.vectron.meta-stone.com/v1'
|
||||||
|
DEFAULT_JUDGE_API_KEY = 'sk-dbd8a665f7634081b87ec409c7636500'
|
||||||
|
DEFAULT_JUDGE_MAX_TOKENS = 10240
|
||||||
|
|
||||||
# 本地数据集根目录(提前下载好的 datasets 目录)
|
# 长文本 middle-truncation 上限(token 数)。当前默认 128k。
|
||||||
DATASET_DIR = str(PROJECT_ROOT / 'datasets')
|
DEFAULT_TRUNCATION_TOKENS = 32768 * 4
|
||||||
|
|
||||||
# 评测输出目录(每个 benchmark 单独子目录)
|
# ============================================================
|
||||||
OUTPUT_DIR = str(PROJECT_ROOT / 'output')
|
# Benchmark suites
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
# 采样数量上限;None 表示跑全部样本
|
# 多次采样配置:总样本数控制在 ~400-500
|
||||||
LIMIT = None
|
MULTI_RUN_CONFIG = {
|
||||||
|
'aime24': 12,
|
||||||
|
'aime25': 12,
|
||||||
|
'aime26': 12,
|
||||||
|
'hmmt26': 12,
|
||||||
|
'live_code_bench': 5,
|
||||||
|
'imo_answerbench': 4,
|
||||||
|
'humaneval': 3,
|
||||||
|
'gpqa_diamond': 2,
|
||||||
|
}
|
||||||
|
|
||||||
# 每个 benchmark 的生成参数配置文件(max_tokens / temperature 等)
|
# 能力域完整列表
|
||||||
CONFIG = None
|
ALL_MULTI_RUN = [
|
||||||
|
'humaneval', 'live_code_bench',
|
||||||
|
'aime24', 'aime25', 'aime26', 'hmmt26',
|
||||||
|
'imo_answerbench', 'gpqa_diamond',
|
||||||
|
]
|
||||||
|
ALL_SINGLE_RUN = [
|
||||||
|
'bigcodebench', 'bfcl_v3', 'competition_math', 'gsm8k', 'hle', 'super_gpqa',
|
||||||
|
'arc', 'bbh', 'cmmlu', 'drop', 'hellaswag', 'mmlu', 'mmlu_pro',
|
||||||
|
'simple_qa', 'trivia_qa', 'winogrande',
|
||||||
|
'openai_mrcr', 'longbench_v2',
|
||||||
|
]
|
||||||
|
ALL_AGENT = ['tau2_bench', 'general_fc']
|
||||||
|
|
||||||
# 是否开启 thinking 模式(sglang chat_template_kwargs.thinking)
|
# 分组基于 CSV 单次时间 + multi-run 后的 wall time 平衡:
|
||||||
ENABLE_THINKING = False
|
# Group1: ~61h | Group2: ~62h | Group3: ~55h
|
||||||
|
SUITES = {
|
||||||
# 测试哪些 benchmark:见下方 DATASETS = _multi_run_order + _single_run_order + _agent_order
|
'full': {
|
||||||
# 想只跑部分 benchmark 时,注释掉对应行即可。
|
'multi': ALL_MULTI_RUN,
|
||||||
MODEL_PATH = '/data1/models/DeepSeek-V4-Flash-INT8'
|
'single': ALL_SINGLE_RUN,
|
||||||
|
'agent': ALL_AGENT,
|
||||||
# Judge model for benchmarks that require LLM-as-judge (simple_qa, tau2_bench, etc.)
|
},
|
||||||
# Vectron endpoint with DeepSeek-V4-Pro
|
'lite': {
|
||||||
JUDGE_MODEL_ARGS = {
|
'multi': ['aime24', 'humaneval'],
|
||||||
'model_id': 'DeepSeek/DeepSeek-V4-Pro',
|
'single': ['gsm8k', 'arc', 'longbench_v2'],
|
||||||
'api_url': 'https://api.vectron.meta-stone.com/v1',
|
'agent': ['general_fc'],
|
||||||
'api_key': 'sk-dbd8a665f7634081b87ec409c7636500',
|
},
|
||||||
'eval_type': 'openai_api',
|
'mid': {
|
||||||
'generation_config': {
|
'multi': ['aime24', 'humaneval'],
|
||||||
'temperature': 0.0,
|
'single': [
|
||||||
'max_tokens': 10240,
|
'live_code_bench', 'bigcodebench', 'competition_math', 'gsm8k',
|
||||||
|
'gpqa_diamond', 'mmlu_pro', 'simple_qa', 'longbench_v2', 'openai_mrcr',
|
||||||
|
],
|
||||||
|
'agent': ['general_fc', 'tau2_bench'],
|
||||||
|
},
|
||||||
|
# 多机组分组,基于 CSV 实测完整时间(已含 multi-run)平衡:
|
||||||
|
# Group1: ~22.7h | Group2: ~25.6h | Group3: ~27.0h | 合计 ~75.2h
|
||||||
|
'group1': {
|
||||||
|
'multi': ['live_code_bench', 'aime24', 'aime25', 'aime26', 'hmmt26', 'imo_answerbench', 'humaneval'],
|
||||||
|
'single': ['bigcodebench', 'competition_math', 'gsm8k', 'drop', 'arc', 'hellaswag', 'winogrande'],
|
||||||
|
'agent': [],
|
||||||
|
},
|
||||||
|
'group2': {
|
||||||
|
'multi': [],
|
||||||
|
'single': ['hle', 'mmlu_pro', 'trivia_qa'],
|
||||||
|
'agent': [],
|
||||||
|
},
|
||||||
|
'group3': {
|
||||||
|
'multi': ['gpqa_diamond'],
|
||||||
|
'single': ['openai_mrcr', 'longbench_v2', 'bfcl_v3', 'mmlu', 'cmmlu', 'bbh', 'simple_qa'],
|
||||||
|
'agent': ['tau2_bench', 'general_fc'],
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
TRUNCATION_CONFIG = {
|
# ============================================================
|
||||||
'longbench_v2': 32768*4,
|
# Fixed configuration
|
||||||
'openai_mrcr': 32768*4,
|
# ============================================================
|
||||||
|
|
||||||
|
MATH_DATASETS = {
|
||||||
|
'aime24', 'aime25', 'aime26', 'hmmt26',
|
||||||
|
'gsm8k', 'competition_math', 'imo_answerbench',
|
||||||
}
|
}
|
||||||
# Combine all datasets in order (shortest estimated time first)
|
|
||||||
# 1. Multi-run small benchmarks (temperature=1.0 for sampling diversity)
|
|
||||||
_multi_run_order = [
|
|
||||||
# 'humaneval', # ~164 samples x 3 runs = 492
|
|
||||||
# 'live_code_bench', # ~100 samples x 5 runs = 500
|
|
||||||
# 'aime26', # ~30 samples x 12 runs = 360
|
|
||||||
# 'aime24', # ~30 samples x 12 runs = 360
|
|
||||||
# 'aime25', # ~30 samples x 12 runs = 360
|
|
||||||
# 'gpqa_diamond', # ~198 samples x 2 runs = 396
|
|
||||||
# 'imo_answerbench', # ~120 samples x 4 runs = 480
|
|
||||||
# 'hmmt26', # ~30 samples x 12 runs = 360
|
|
||||||
]
|
|
||||||
|
|
||||||
# 2. Single-run benchmarks (temperature=0.0 greedy)
|
MATH_PROMPT_TEMPLATE = (
|
||||||
_single_run_order = [
|
"{question}\n"
|
||||||
# 'arc',
|
"Please reason step by step, and put your final answer within \\boxed{{}}."
|
||||||
# 'bfcl_v3', # function calling: greedy decoding
|
)
|
||||||
# 'winogrande',
|
|
||||||
# 'competition_math',
|
|
||||||
# 'gsm8k',
|
|
||||||
# 'hellaswag',
|
|
||||||
# 'bigcodebench',
|
|
||||||
# 'drop',
|
|
||||||
# 'bbh',
|
|
||||||
# 'openai_mrcr',
|
|
||||||
# 'longbench_v2',
|
|
||||||
# 'mmlu',
|
|
||||||
# 'cmmlu',
|
|
||||||
# 'simple_qa',
|
|
||||||
# 'mmlu_pro',
|
|
||||||
'hle',
|
|
||||||
# 'trivia_qa',
|
|
||||||
]
|
|
||||||
|
|
||||||
# 3. Agent / tool benchmarks (last)
|
|
||||||
_agent_order = [
|
|
||||||
# 'tau2_bench',
|
|
||||||
# 'general_fc',
|
|
||||||
]
|
|
||||||
# ============================================================
|
|
||||||
# Command line argument overrides (不需要修改)
|
|
||||||
# ============================================================
|
|
||||||
|
|
||||||
for i, arg in enumerate(sys.argv):
|
|
||||||
if arg == '--limit' and i + 1 < len(sys.argv):
|
|
||||||
limit_val = sys.argv[i + 1]
|
|
||||||
if limit_val.lower() == 'none' or limit_val.lower() == 'all':
|
|
||||||
LIMIT = None
|
|
||||||
else:
|
|
||||||
LIMIT = int(limit_val)
|
|
||||||
elif arg == '--model' and i + 1 < len(sys.argv):
|
|
||||||
MODEL = sys.argv[i + 1]
|
|
||||||
elif arg == '--api-url' and i + 1 < len(sys.argv):
|
|
||||||
API_URL = sys.argv[i + 1]
|
|
||||||
elif arg == '--dataset-dir' and i + 1 < len(sys.argv):
|
|
||||||
DATASET_DIR = sys.argv[i + 1]
|
|
||||||
elif arg == '--output-dir' and i + 1 < len(sys.argv):
|
|
||||||
OUTPUT_DIR = sys.argv[i + 1]
|
|
||||||
elif arg == '--config' and i + 1 < len(sys.argv):
|
|
||||||
CONFIG = sys.argv[i + 1]
|
|
||||||
|
|
||||||
# Benchmarks that require sandboxed code execution
|
|
||||||
SANDBOX_DATASETS = {'humaneval', 'bigcodebench'}
|
SANDBOX_DATASETS = {'humaneval', 'bigcodebench'}
|
||||||
SANDBOX_CONFIGS = {
|
SANDBOX_CONFIGS = {
|
||||||
'bigcodebench': {
|
'bigcodebench': {
|
||||||
@ -141,91 +169,69 @@ SANDBOX_CONFIGS = {
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
# ============================================================
|
|
||||||
# User-tunable parameters
|
|
||||||
# ============================================================
|
|
||||||
|
|
||||||
# Fixed seed for reproducibility. With temperature > 0 the model still samples
|
|
||||||
# randomly, so running N times with the same seed still yields variance.
|
|
||||||
SEED = 42
|
|
||||||
|
|
||||||
# Benchmarks to run multiple times with temperature=1.0.
|
|
||||||
# Format: {benchmark_name: num_runs}
|
|
||||||
# Number of runs chosen so total samples ≈ 400-500 per benchmark.
|
|
||||||
MULTI_RUN_CONFIG = {
|
|
||||||
# ~30 samples each -> 12 runs = ~360 samples
|
|
||||||
'aime24': 12,
|
|
||||||
'aime25': 12,
|
|
||||||
'aime26': 12,
|
|
||||||
'hmmt26': 12,
|
|
||||||
# ~100 samples -> 5 runs = ~500 samples
|
|
||||||
'live_code_bench': 5,
|
|
||||||
# ~120 samples -> 4 runs = ~480 samples
|
|
||||||
'imo_answerbench': 4,
|
|
||||||
# ~164 samples -> 3 runs = ~492 samples
|
|
||||||
'humaneval': 3,
|
|
||||||
# ~198 samples -> 2 runs = ~396 samples
|
|
||||||
'gpqa_diamond': 2,
|
|
||||||
}
|
|
||||||
|
|
||||||
# Single-run benchmarks (temperature=0.0 greedy)
|
|
||||||
SINGLE_RUN_DATASETS = [
|
|
||||||
'bigcodebench',
|
|
||||||
'bfcl_v3', # function calling: greedy decoding
|
|
||||||
'competition_math',
|
|
||||||
'gsm8k',
|
|
||||||
'hle',
|
|
||||||
'super_gpqa',
|
|
||||||
'arc',
|
|
||||||
'bbh',
|
|
||||||
'cmmlu',
|
|
||||||
'drop',
|
|
||||||
'hellaswag',
|
|
||||||
'mmlu',
|
|
||||||
'mmlu_pro',
|
|
||||||
'simple_qa',
|
|
||||||
'trivia_qa',
|
|
||||||
'winogrande',
|
|
||||||
'openai_mrcr', # long-context, run once
|
|
||||||
'longbench_v2', # long-context, run once
|
|
||||||
]
|
|
||||||
|
|
||||||
# Agent / tool benchmarks (temperature=0.0 greedy, run once)
|
|
||||||
AGENT_DATASETS = [
|
|
||||||
'tau2_bench',
|
|
||||||
'general_fc',
|
|
||||||
]
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
DATASETS = _multi_run_order + _single_run_order + _agent_order
|
|
||||||
|
|
||||||
# ENABLE_THINKING 已移至文件头部「必改参数」区
|
|
||||||
BATCH_SIZE_LIST = [4]
|
|
||||||
# LIMIT is parsed from command line: --limit N
|
|
||||||
SHUFFLE = True
|
|
||||||
|
|
||||||
# ============================================================
|
# ============================================================
|
||||||
# Fixed configuration
|
# CLI parser
|
||||||
# ============================================================
|
# ============================================================
|
||||||
|
|
||||||
CONFIG_PATH = str(Path(CONFIG) if CONFIG else SCRIPT_DIR / 'config' / 'dpv4-int8_nothinking.yaml')
|
def build_parser():
|
||||||
if not Path(CONFIG_PATH).exists():
|
parser = argparse.ArgumentParser(
|
||||||
raise FileNotFoundError(f'Config file not found: {CONFIG_PATH}')
|
description='Unified EvalScope benchmark runner',
|
||||||
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||||
|
epilog='Suites: full, lite, mid, group1, group2, group3',
|
||||||
|
)
|
||||||
|
|
||||||
|
# Model / API
|
||||||
|
parser.add_argument('--model', default=DEFAULT_MODEL,
|
||||||
|
help='Served model name (default: %(default)s)')
|
||||||
|
parser.add_argument('--api-url', default=DEFAULT_API_URL,
|
||||||
|
help='OpenAI-compatible API URL (default: %(default)s)')
|
||||||
|
|
||||||
MATH_DATASETS = {
|
# Paths
|
||||||
'aime24', 'aime25', 'aime26', 'hmmt26',
|
parser.add_argument('--dataset-dir', default=DEFAULT_DATASET_DIR,
|
||||||
'gsm8k', 'competition_math', 'imo_answerbench',
|
help='Parent directory containing datasets/ subdir (default: %(default)s)')
|
||||||
}
|
parser.add_argument('--output-dir', default=DEFAULT_OUTPUT_DIR,
|
||||||
|
help='Output root directory (default: %(default)s)')
|
||||||
|
parser.add_argument('--config', default=DEFAULT_CONFIG,
|
||||||
|
help='YAML config path (default: %(default)s)')
|
||||||
|
parser.add_argument('--tokenizer-path', default=DEFAULT_TOKENIZER_PATH,
|
||||||
|
help='Local tokenizer path for middle-truncation (default: %(default)s)')
|
||||||
|
|
||||||
MATH_PROMPT_TEMPLATE = (
|
# Run control
|
||||||
"{question}\n"
|
parser.add_argument('--suite', default='full', choices=list(SUITES.keys()),
|
||||||
"Please reason step by step, and put your final answer within \\boxed{{}}."
|
help='Benchmark suite to run (default: %(default)s)')
|
||||||
)
|
parser.add_argument('--datasets', '--benchmarks', dest='datasets', default=None,
|
||||||
|
help='Override suite with comma-separated benchmark names, e.g. aime24,gsm8k')
|
||||||
|
parser.add_argument('--exclude', default=None,
|
||||||
|
help='Comma-separated benchmarks to exclude from the chosen suite')
|
||||||
|
parser.add_argument('--limit', default=None,
|
||||||
|
help='Max samples per benchmark; "none"/"all" for no limit (default: none)')
|
||||||
|
parser.add_argument('--seed', type=int, default=DEFAULT_SEED,
|
||||||
|
help='Random seed (default: %(default)s)')
|
||||||
|
parser.add_argument('--batch-size', type=int, default=DEFAULT_BATCH_SIZE,
|
||||||
|
help='Evaluation batch size (default: %(default)s)')
|
||||||
|
|
||||||
with open(CONFIG_PATH, 'r', encoding='utf-8') as f:
|
# Decoding / thinking
|
||||||
DATASET_CONFIGS = yaml.safe_load(f)
|
parser.add_argument('--thinking', action='store_true', default=None,
|
||||||
|
help='Enable thinking mode (sglang chat_template_kwargs.thinking=True)')
|
||||||
|
parser.add_argument('--no-thinking', dest='thinking', action='store_false',
|
||||||
|
help='Disable thinking mode (default)')
|
||||||
|
|
||||||
|
# Judge model
|
||||||
|
parser.add_argument('--judge-model', default=DEFAULT_JUDGE_MODEL,
|
||||||
|
help='Judge model name (default: %(default)s)')
|
||||||
|
parser.add_argument('--judge-api-url', default=DEFAULT_JUDGE_API_URL,
|
||||||
|
help='Judge model API URL (default: %(default)s)')
|
||||||
|
parser.add_argument('--judge-api-key', default=DEFAULT_JUDGE_API_KEY,
|
||||||
|
help='Judge model API key')
|
||||||
|
parser.add_argument('--judge-max-tokens', type=int, default=DEFAULT_JUDGE_MAX_TOKENS,
|
||||||
|
help='Judge model max_tokens (default: %(default)s)')
|
||||||
|
|
||||||
|
# Truncation
|
||||||
|
parser.add_argument('--truncation-tokens', type=int, default=DEFAULT_TRUNCATION_TOKENS,
|
||||||
|
help='Middle-truncation token budget for long-context benchmarks (default: %(default)s)')
|
||||||
|
|
||||||
|
return parser
|
||||||
|
|
||||||
|
|
||||||
# ============================================================
|
# ============================================================
|
||||||
@ -235,23 +241,21 @@ with open(CONFIG_PATH, 'r', encoding='utf-8') as f:
|
|||||||
_TOKENIZER = None
|
_TOKENIZER = None
|
||||||
|
|
||||||
|
|
||||||
def get_tokenizer():
|
def get_tokenizer(tokenizer_path: str):
|
||||||
global _TOKENIZER
|
global _TOKENIZER
|
||||||
if _TOKENIZER is None:
|
if _TOKENIZER is None:
|
||||||
from transformers import AutoTokenizer
|
from transformers import AutoTokenizer
|
||||||
try:
|
try:
|
||||||
# Try local path first
|
_TOKENIZER = AutoTokenizer.from_pretrained(tokenizer_path, trust_remote_code=True)
|
||||||
_TOKENIZER = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True)
|
|
||||||
except Exception:
|
except Exception:
|
||||||
# Fallback to HuggingFace model ID
|
|
||||||
_TOKENIZER = AutoTokenizer.from_pretrained('deepseek-ai/DeepSeek-V4-Flash', trust_remote_code=True)
|
_TOKENIZER = AutoTokenizer.from_pretrained('deepseek-ai/DeepSeek-V4-Flash', trust_remote_code=True)
|
||||||
return _TOKENIZER
|
return _TOKENIZER
|
||||||
|
|
||||||
|
|
||||||
def truncate_middle(text: str, max_tokens: int) -> str:
|
def truncate_middle(text: str, max_tokens: int, tokenizer_path: str) -> str:
|
||||||
if max_tokens <= 0:
|
if max_tokens <= 0:
|
||||||
return text
|
return text
|
||||||
tokenizer = get_tokenizer()
|
tokenizer = get_tokenizer(tokenizer_path)
|
||||||
token_ids = tokenizer.encode(text, add_special_tokens=False)
|
token_ids = tokenizer.encode(text, add_special_tokens=False)
|
||||||
if len(token_ids) <= max_tokens:
|
if len(token_ids) <= max_tokens:
|
||||||
return text
|
return text
|
||||||
@ -261,41 +265,35 @@ def truncate_middle(text: str, max_tokens: int) -> str:
|
|||||||
return tokenizer.decode(truncated_ids, skip_special_tokens=True)
|
return tokenizer.decode(truncated_ids, skip_special_tokens=True)
|
||||||
|
|
||||||
|
|
||||||
def _patch_adapters_for_truncation():
|
def _patch_adapters_for_truncation(tokenizer_path: str, truncation_tokens: int):
|
||||||
from evalscope.benchmarks.longbench_v2.longbench_v2_adapter import LongBenchV2Adapter
|
from evalscope.benchmarks.longbench_v2.longbench_v2_adapter import LongBenchV2Adapter
|
||||||
from evalscope.benchmarks.openai_mrcr.openai_mrcr_adapter import OpenAIMRCRAdapter
|
from evalscope.benchmarks.openai_mrcr.openai_mrcr_adapter import OpenAIMRCRAdapter
|
||||||
|
|
||||||
# Patch LongBenchV2Adapter.format_prompt_template
|
|
||||||
_orig_longbench_format = LongBenchV2Adapter.format_prompt_template
|
_orig_longbench_format = LongBenchV2Adapter.format_prompt_template
|
||||||
|
|
||||||
def _patched_longbench_format(self, sample):
|
def _patched_longbench_format(self, sample):
|
||||||
max_tok = TRUNCATION_CONFIG.get('longbench_v2')
|
if sample.metadata and 'context' in sample.metadata:
|
||||||
if max_tok and sample.metadata and 'context' in sample.metadata:
|
sample.metadata['context'] = truncate_middle(sample.metadata['context'], truncation_tokens, tokenizer_path)
|
||||||
sample.metadata['context'] = truncate_middle(sample.metadata['context'], max_tok)
|
|
||||||
return _orig_longbench_format(self, sample)
|
return _orig_longbench_format(self, sample)
|
||||||
|
|
||||||
LongBenchV2Adapter.format_prompt_template = _patched_longbench_format
|
LongBenchV2Adapter.format_prompt_template = _patched_longbench_format
|
||||||
|
|
||||||
# Patch OpenAIMRCRAdapter.record_to_sample with needle-aware truncation.
|
|
||||||
# MRCR is a long chat history with needles hidden at desired_msg_index.
|
|
||||||
# We keep the head, tail, and a window around the needle, and truncate
|
|
||||||
# each kept message if it is still too long. This preserves the retrieval
|
|
||||||
# task while fitting GPU memory.
|
|
||||||
_orig_mrcr_record = OpenAIMRCRAdapter.record_to_sample
|
_orig_mrcr_record = OpenAIMRCRAdapter.record_to_sample
|
||||||
|
|
||||||
def _patched_mrcr_record(self, record):
|
def _patched_mrcr_record(self, record):
|
||||||
import json
|
|
||||||
max_total_tok = TRUNCATION_CONFIG.get('openai_mrcr')
|
|
||||||
per_msg_max_tok = 8192
|
per_msg_max_tok = 8192
|
||||||
if max_total_tok and 'prompt' in record:
|
if 'prompt' in record:
|
||||||
try:
|
try:
|
||||||
prompt_data = json.loads(record['prompt'])
|
prompt_data = json.loads(record['prompt'])
|
||||||
if not isinstance(prompt_data, list) or len(prompt_data) == 0:
|
if not isinstance(prompt_data, list) or len(prompt_data) == 0:
|
||||||
return _orig_mrcr_record(self, record)
|
return _orig_mrcr_record(self, record)
|
||||||
|
|
||||||
tokenizer = get_tokenizer()
|
tokenizer = get_tokenizer(tokenizer_path)
|
||||||
total_tok = sum(
|
total_tok = sum(
|
||||||
len(tokenizer.encode(msg.get('content', '') if isinstance(msg, dict) else '', add_special_tokens=False))
|
len(tokenizer.encode(msg.get('content', '') if isinstance(msg, dict) else '', add_special_tokens=False))
|
||||||
for msg in prompt_data
|
for msg in prompt_data
|
||||||
)
|
)
|
||||||
if total_tok <= max_total_tok:
|
if total_tok <= truncation_tokens:
|
||||||
return _orig_mrcr_record(self, record)
|
return _orig_mrcr_record(self, record)
|
||||||
|
|
||||||
desired_idx = record.get('desired_msg_index', 0)
|
desired_idx = record.get('desired_msg_index', 0)
|
||||||
@ -304,10 +302,8 @@ def _patch_adapters_for_truncation():
|
|||||||
|
|
||||||
n = len(prompt_data)
|
n = len(prompt_data)
|
||||||
keep = set()
|
keep = set()
|
||||||
# Head and tail context
|
|
||||||
keep.update(range(min(2, n)))
|
keep.update(range(min(2, n)))
|
||||||
keep.update(range(max(0, n - 2), n))
|
keep.update(range(max(0, n - 2), n))
|
||||||
# Window around the needle
|
|
||||||
window = 2
|
window = 2
|
||||||
keep.update(range(max(0, desired_idx - window), min(n, desired_idx + window + 1)))
|
keep.update(range(max(0, desired_idx - window), min(n, desired_idx + window + 1)))
|
||||||
keep = sorted(keep)
|
keep = sorted(keep)
|
||||||
@ -319,7 +315,7 @@ def _patch_adapters_for_truncation():
|
|||||||
msg = dict(msg)
|
msg = dict(msg)
|
||||||
content = msg.get('content', '')
|
content = msg.get('content', '')
|
||||||
if len(tokenizer.encode(content, add_special_tokens=False)) > per_msg_max_tok:
|
if len(tokenizer.encode(content, add_special_tokens=False)) > per_msg_max_tok:
|
||||||
msg['content'] = truncate_middle(content, per_msg_max_tok)
|
msg['content'] = truncate_middle(content, per_msg_max_tok, tokenizer_path)
|
||||||
new_prompt.append(msg)
|
new_prompt.append(msg)
|
||||||
|
|
||||||
record = dict(record)
|
record = dict(record)
|
||||||
@ -327,16 +323,21 @@ def _patch_adapters_for_truncation():
|
|||||||
except (json.JSONDecodeError, TypeError):
|
except (json.JSONDecodeError, TypeError):
|
||||||
pass
|
pass
|
||||||
return _orig_mrcr_record(self, record)
|
return _orig_mrcr_record(self, record)
|
||||||
|
|
||||||
OpenAIMRCRAdapter.record_to_sample = _patched_mrcr_record
|
OpenAIMRCRAdapter.record_to_sample = _patched_mrcr_record
|
||||||
|
|
||||||
|
|
||||||
_patch_adapters_for_truncation()
|
|
||||||
|
|
||||||
|
|
||||||
# ============================================================
|
# ============================================================
|
||||||
# Helpers
|
# Helpers
|
||||||
# ============================================================
|
# ============================================================
|
||||||
|
|
||||||
|
def load_dataset_configs(config_path: str):
|
||||||
|
if not Path(config_path).exists():
|
||||||
|
raise FileNotFoundError(f'Config file not found: {config_path}')
|
||||||
|
with open(config_path, 'r', encoding='utf-8') as f:
|
||||||
|
return yaml.safe_load(f)
|
||||||
|
|
||||||
|
|
||||||
def configure_thinking(generation_config: dict, enable: bool) -> dict:
|
def configure_thinking(generation_config: dict, enable: bool) -> dict:
|
||||||
extra_body = generation_config.get('extra_body', {})
|
extra_body = generation_config.get('extra_body', {})
|
||||||
chat_template_kwargs = extra_body.get('chat_template_kwargs', {})
|
chat_template_kwargs = extra_body.get('chat_template_kwargs', {})
|
||||||
@ -363,14 +364,24 @@ def build_agent_config(agent_cfg: dict) -> NativeAgentConfig:
|
|||||||
return NativeAgentConfig(**agent_cfg)
|
return NativeAgentConfig(**agent_cfg)
|
||||||
|
|
||||||
|
|
||||||
def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_thinking: bool, seed: int, run_idx: int = 0) -> TaskConfig:
|
def build_task_config(
|
||||||
# Each run gets its own work_dir so use_cache does not reuse predictions
|
dataset_name: str,
|
||||||
# across repeated samples. This is required for temperature=1.0 multi-run
|
ds_cfg: dict,
|
||||||
# benchmarks to actually measure variance.
|
batch_size: int,
|
||||||
|
enable_thinking: bool,
|
||||||
|
seed: int,
|
||||||
|
limit,
|
||||||
|
output_dir: str,
|
||||||
|
model: str,
|
||||||
|
api_url: str,
|
||||||
|
dataset_dir: str,
|
||||||
|
judge_model_args: dict,
|
||||||
|
run_idx: int = 0,
|
||||||
|
) -> TaskConfig:
|
||||||
if run_idx > 0:
|
if run_idx > 0:
|
||||||
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}_run_{run_idx}'
|
work_dir = Path(output_dir) / dataset_name / f'seed_{seed}_run_{run_idx}'
|
||||||
else:
|
else:
|
||||||
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}'
|
work_dir = Path(output_dir) / dataset_name / f'seed_{seed}'
|
||||||
work_dir.mkdir(parents=True, exist_ok=True)
|
work_dir.mkdir(parents=True, exist_ok=True)
|
||||||
work_dir = str(work_dir)
|
work_dir = str(work_dir)
|
||||||
|
|
||||||
@ -388,13 +399,13 @@ def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_t
|
|||||||
agent_config = build_agent_config(ds_cfg['agent_config'])
|
agent_config = build_agent_config(ds_cfg['agent_config'])
|
||||||
|
|
||||||
return TaskConfig(
|
return TaskConfig(
|
||||||
model=MODEL,
|
model=model,
|
||||||
api_url=API_URL,
|
api_url=api_url,
|
||||||
eval_type='openai_api',
|
eval_type='openai_api',
|
||||||
dataset_dir=DATASET_DIR,
|
dataset_dir=dataset_dir,
|
||||||
judge_model_args=JUDGE_MODEL_ARGS,
|
judge_model_args=judge_model_args,
|
||||||
seed=seed,
|
seed=seed,
|
||||||
limit=LIMIT,
|
limit=limit,
|
||||||
collect_perf=True,
|
collect_perf=True,
|
||||||
no_timestamp=True,
|
no_timestamp=True,
|
||||||
work_dir=work_dir,
|
work_dir=work_dir,
|
||||||
@ -419,71 +430,135 @@ def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_t
|
|||||||
|
|
||||||
|
|
||||||
# ============================================================
|
# ============================================================
|
||||||
# Main loop
|
# Main
|
||||||
# ============================================================
|
# ============================================================
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
print(f"Config: {CONFIG_PATH}")
|
parser = build_parser()
|
||||||
print(f"Model: {MODEL}")
|
args = parser.parse_args()
|
||||||
print(f"API URL: {API_URL}")
|
|
||||||
print(f"Dataset Dir: {DATASET_DIR}")
|
|
||||||
print(f"Output Dir: {OUTPUT_DIR}")
|
|
||||||
print(f"Limit: {LIMIT if LIMIT is not None else 'ALL'}")
|
|
||||||
print(f"Total benchmarks: {len(DATASETS)}")
|
|
||||||
print("="*60)
|
|
||||||
|
|
||||||
for batch_size in BATCH_SIZE_LIST:
|
# Resolve limit
|
||||||
# 1. Run multi-run benchmarks (temperature=1.0, same seed, variance from sampling)
|
limit = args.limit
|
||||||
for dataset_name in _multi_run_order:
|
if limit is not None:
|
||||||
if dataset_name not in DATASET_CONFIGS:
|
if str(limit).lower() in ('none', 'all'):
|
||||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
limit = None
|
||||||
continue
|
else:
|
||||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
limit = int(limit)
|
||||||
num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1)
|
|
||||||
for run_idx in range(num_runs):
|
|
||||||
print(f"\n{'='*60}")
|
|
||||||
print(f"Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={SEED})")
|
|
||||||
print(f"{'='*60}")
|
|
||||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED, run_idx=run_idx)
|
|
||||||
try:
|
|
||||||
run_task(task_cfg)
|
|
||||||
except Exception as e:
|
|
||||||
print(f"ERROR in {dataset_name} (run {run_idx + 1}): {e}")
|
|
||||||
continue
|
|
||||||
|
|
||||||
# 2. Run single-run benchmarks (temperature=0.0 greedy)
|
enable_thinking = DEFAULT_ENABLE_THINKING if args.thinking is None else args.thinking
|
||||||
for dataset_name in _single_run_order:
|
|
||||||
if dataset_name not in DATASET_CONFIGS:
|
# Resolve suite or custom datasets
|
||||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
if args.datasets:
|
||||||
continue
|
custom = [d.strip() for d in args.datasets.split(',') if d.strip()]
|
||||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
multi_run = [d for d in custom if d in MULTI_RUN_CONFIG]
|
||||||
|
single_run = [d for d in custom if d not in MULTI_RUN_CONFIG]
|
||||||
|
agent = [d for d in custom if d in ALL_AGENT]
|
||||||
|
single_run = [d for d in single_run if d not in ALL_AGENT]
|
||||||
|
else:
|
||||||
|
suite = SUITES[args.suite]
|
||||||
|
multi_run = list(suite['multi'])
|
||||||
|
single_run = list(suite['single'])
|
||||||
|
agent = list(suite['agent'])
|
||||||
|
|
||||||
|
# Apply --exclude
|
||||||
|
if args.exclude:
|
||||||
|
exclude = {d.strip() for d in args.exclude.split(',') if d.strip()}
|
||||||
|
multi_run = [d for d in multi_run if d not in exclude]
|
||||||
|
single_run = [d for d in single_run if d not in exclude]
|
||||||
|
agent = [d for d in agent if d not in exclude]
|
||||||
|
|
||||||
|
judge_model_args = {
|
||||||
|
'model_id': args.judge_model,
|
||||||
|
'api_url': args.judge_api_url,
|
||||||
|
'api_key': args.judge_api_key,
|
||||||
|
'eval_type': 'openai_api',
|
||||||
|
'generation_config': {
|
||||||
|
'temperature': 0.0,
|
||||||
|
'max_tokens': args.judge_max_tokens,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
truncation_tokens = args.truncation_tokens
|
||||||
|
_patch_adapters_for_truncation(args.tokenizer_path, truncation_tokens)
|
||||||
|
|
||||||
|
dataset_configs = load_dataset_configs(args.config)
|
||||||
|
|
||||||
|
print('=' * 60)
|
||||||
|
print(f'Config: {args.config}')
|
||||||
|
print(f'Model: {args.model}')
|
||||||
|
print(f'API URL: {args.api_url}')
|
||||||
|
print(f'Dataset Dir: {args.dataset_dir}')
|
||||||
|
print(f'Output Dir: {args.output_dir}')
|
||||||
|
print(f'Suite: {args.suite}')
|
||||||
|
print(f'Limit: {limit if limit is not None else "ALL"}')
|
||||||
|
print(f'Thinking: {enable_thinking}')
|
||||||
|
print(f'Seed: {args.seed}')
|
||||||
|
print(f'Batch Size: {args.batch_size}')
|
||||||
|
print(f'Tokenizer Path: {args.tokenizer_path}')
|
||||||
|
print(f'Truncation Tokens: {truncation_tokens}')
|
||||||
|
print(f'Multi-run datasets: {multi_run}')
|
||||||
|
print(f'Single-run datasets: {single_run}')
|
||||||
|
print(f'Agent datasets: {agent}')
|
||||||
|
print('=' * 60)
|
||||||
|
|
||||||
|
for dataset_name in multi_run:
|
||||||
|
if dataset_name not in dataset_configs:
|
||||||
|
print(f'WARNING: {dataset_name} not in YAML config, skipping')
|
||||||
|
continue
|
||||||
|
ds_cfg = dataset_configs[dataset_name]
|
||||||
|
num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1)
|
||||||
|
for run_idx in range(num_runs):
|
||||||
print(f"\n{'='*60}")
|
print(f"\n{'='*60}")
|
||||||
print(f"Running: {dataset_name} (seed={SEED})")
|
print(f'Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={args.seed})')
|
||||||
print(f"{'='*60}")
|
print(f"{'='*60}")
|
||||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED)
|
task_cfg = build_task_config(
|
||||||
|
dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit,
|
||||||
|
args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args,
|
||||||
|
run_idx=run_idx,
|
||||||
|
)
|
||||||
try:
|
try:
|
||||||
run_task(task_cfg)
|
run_task(task_cfg)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"ERROR in {dataset_name}: {e}")
|
print(f'ERROR in {dataset_name} (run {run_idx + 1}): {e}')
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# 3. Run agent benchmarks
|
for dataset_name in single_run:
|
||||||
for dataset_name in _agent_order:
|
if dataset_name not in dataset_configs:
|
||||||
if dataset_name not in DATASET_CONFIGS:
|
print(f'WARNING: {dataset_name} not in YAML config, skipping')
|
||||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
continue
|
||||||
continue
|
ds_cfg = dataset_configs[dataset_name]
|
||||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
print(f"\n{'='*60}")
|
||||||
print(f"\n{'='*60}")
|
print(f'Running: {dataset_name} (seed={args.seed})')
|
||||||
print(f"Running: {dataset_name} (seed={SEED})")
|
print(f"{'='*60}")
|
||||||
print(f"{'='*60}")
|
task_cfg = build_task_config(
|
||||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED)
|
dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit,
|
||||||
try:
|
args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args,
|
||||||
run_task(task_cfg)
|
)
|
||||||
except Exception as e:
|
try:
|
||||||
print(f"ERROR in {dataset_name}: {e}")
|
run_task(task_cfg)
|
||||||
continue
|
except Exception as e:
|
||||||
|
print(f'ERROR in {dataset_name}: {e}')
|
||||||
|
continue
|
||||||
|
|
||||||
print("\nAll benchmarks done!")
|
for dataset_name in agent:
|
||||||
|
if dataset_name not in dataset_configs:
|
||||||
|
print(f'WARNING: {dataset_name} not in YAML config, skipping')
|
||||||
|
continue
|
||||||
|
ds_cfg = dataset_configs[dataset_name]
|
||||||
|
print(f"\n{'='*60}")
|
||||||
|
print(f'Running: {dataset_name} (seed={args.seed})')
|
||||||
|
print(f"{'='*60}")
|
||||||
|
task_cfg = build_task_config(
|
||||||
|
dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit,
|
||||||
|
args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
run_task(task_cfg)
|
||||||
|
except Exception as e:
|
||||||
|
print(f'ERROR in {dataset_name}: {e}')
|
||||||
|
continue
|
||||||
|
|
||||||
|
print('\nAll benchmarks done!')
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
|
|||||||
@ -1,74 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
Full evaluation suite: ~3-5 days
|
|
||||||
Runs all benchmarks with full datasets and multiple runs for stability.
|
|
||||||
Use --limit none (default) for the complete evaluation.
|
|
||||||
"""
|
|
||||||
from pathlib import Path
|
|
||||||
import sys
|
|
||||||
|
|
||||||
sys.path.insert(0, str(Path(__file__).parent))
|
|
||||||
|
|
||||||
import run as run_module
|
|
||||||
|
|
||||||
# Use the default full configuration from run.py
|
|
||||||
# (already ordered by estimated runtime)
|
|
||||||
run_module._multi_run_order = [
|
|
||||||
'humaneval',
|
|
||||||
'live_code_bench',
|
|
||||||
'aime26',
|
|
||||||
'aime24',
|
|
||||||
'aime25',
|
|
||||||
'gpqa_diamond',
|
|
||||||
'imo_answerbench',
|
|
||||||
'hmmt26',
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module._single_run_order = [
|
|
||||||
'arc',
|
|
||||||
'bfcl_v3',
|
|
||||||
'winogrande',
|
|
||||||
'competition_math',
|
|
||||||
'gsm8k',
|
|
||||||
'hellaswag',
|
|
||||||
'bigcodebench',
|
|
||||||
'drop',
|
|
||||||
'bbh',
|
|
||||||
'openai_mrcr',
|
|
||||||
'longbench_v2',
|
|
||||||
'mmlu',
|
|
||||||
'cmmlu',
|
|
||||||
'super_gpqa',
|
|
||||||
'simple_qa',
|
|
||||||
'mmlu_pro',
|
|
||||||
'hle',
|
|
||||||
'trivia_qa',
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module._agent_order = [
|
|
||||||
'tau2_bench',
|
|
||||||
'general_fc',
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module.DATASETS = (
|
|
||||||
run_module._multi_run_order
|
|
||||||
+ run_module._single_run_order
|
|
||||||
+ run_module._agent_order
|
|
||||||
)
|
|
||||||
|
|
||||||
# Full: many runs for statistical stability
|
|
||||||
run_module.MULTI_RUN_CONFIG = {
|
|
||||||
'aime24': 12,
|
|
||||||
'aime25': 12,
|
|
||||||
'aime26': 12,
|
|
||||||
'hmmt26': 12,
|
|
||||||
'live_code_bench': 5,
|
|
||||||
'imo_answerbench': 4,
|
|
||||||
'humaneval': 3,
|
|
||||||
'gpqa_diamond': 2,
|
|
||||||
}
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
print('Full benchmarks:', run_module.DATASETS)
|
|
||||||
print('Estimated time: ~3-5 days with --limit none')
|
|
||||||
run_module.main()
|
|
||||||
@ -1,48 +0,0 @@
|
|||||||
"""
|
|
||||||
Group 1: Code + Reasoning + short Knowledge + Agent (est. ~18-22h)
|
|
||||||
Run on machine 1.
|
|
||||||
"""
|
|
||||||
from pathlib import Path
|
|
||||||
import sys
|
|
||||||
|
|
||||||
# Ensure we can import run.py from the same directory inside or outside Docker
|
|
||||||
sys.path.insert(0, str(Path(__file__).parent))
|
|
||||||
|
|
||||||
import run as run_module
|
|
||||||
|
|
||||||
# Override dataset lists for group 1
|
|
||||||
run_module._multi_run_order = [
|
|
||||||
'humaneval',
|
|
||||||
'live_code_bench',
|
|
||||||
'aime26',
|
|
||||||
'aime24',
|
|
||||||
'aime25',
|
|
||||||
'gpqa_diamond',
|
|
||||||
'imo_answerbench',
|
|
||||||
'hmmt26',
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module._single_run_order = [
|
|
||||||
'arc',
|
|
||||||
'winogrande',
|
|
||||||
'competition_math',
|
|
||||||
'gsm8k',
|
|
||||||
'hellaswag',
|
|
||||||
'bigcodebench',
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module._agent_order = [
|
|
||||||
'bfcl_v3',
|
|
||||||
'tau2_bench',
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module.DATASETS = (
|
|
||||||
run_module._multi_run_order
|
|
||||||
+ run_module._single_run_order
|
|
||||||
+ run_module._agent_order
|
|
||||||
)
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
print('Group 1 benchmarks:', run_module.DATASETS)
|
|
||||||
print('Estimated time: ~18-22h')
|
|
||||||
run_module.main()
|
|
||||||
@ -1,40 +0,0 @@
|
|||||||
"""
|
|
||||||
Group 2: Knowledge + Long-context + Agent (est. ~20-24h)
|
|
||||||
Run on machine 2.
|
|
||||||
"""
|
|
||||||
from pathlib import Path
|
|
||||||
import sys
|
|
||||||
|
|
||||||
# Ensure we can import run.py from the same directory inside or outside Docker
|
|
||||||
sys.path.insert(0, str(Path(__file__).parent))
|
|
||||||
|
|
||||||
import run as run_module
|
|
||||||
|
|
||||||
# Override dataset lists for group 2
|
|
||||||
run_module._multi_run_order = []
|
|
||||||
|
|
||||||
run_module._single_run_order = [
|
|
||||||
'drop',
|
|
||||||
'bbh',
|
|
||||||
'openai_mrcr',
|
|
||||||
'longbench_v2',
|
|
||||||
'mmlu',
|
|
||||||
'cmmlu',
|
|
||||||
'super_gpqa',
|
|
||||||
'simple_qa',
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module._agent_order = [
|
|
||||||
'general_fc',
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module.DATASETS = (
|
|
||||||
run_module._multi_run_order
|
|
||||||
+ run_module._single_run_order
|
|
||||||
+ run_module._agent_order
|
|
||||||
)
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
print('Group 2 benchmarks:', run_module.DATASETS)
|
|
||||||
print('Estimated time: ~20-24h')
|
|
||||||
run_module.main()
|
|
||||||
@ -1,35 +0,0 @@
|
|||||||
"""
|
|
||||||
Group 3: Long-running Knowledge benchmarks (est. ~35-40h)
|
|
||||||
Run on machine 3.
|
|
||||||
Note: trivia_qa and hle are inherently slow; consider using --limit to control time.
|
|
||||||
"""
|
|
||||||
from pathlib import Path
|
|
||||||
import sys
|
|
||||||
|
|
||||||
# Ensure we can import run.py from the same directory inside or outside Docker
|
|
||||||
sys.path.insert(0, str(Path(__file__).parent))
|
|
||||||
|
|
||||||
import run as run_module
|
|
||||||
|
|
||||||
# Override dataset lists for group 3
|
|
||||||
run_module._multi_run_order = []
|
|
||||||
|
|
||||||
run_module._single_run_order = [
|
|
||||||
'mmlu_pro',
|
|
||||||
'hle',
|
|
||||||
'trivia_qa',
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module._agent_order = []
|
|
||||||
|
|
||||||
run_module.DATASETS = (
|
|
||||||
run_module._multi_run_order
|
|
||||||
+ run_module._single_run_order
|
|
||||||
+ run_module._agent_order
|
|
||||||
)
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
print('Group 3 benchmarks:', run_module.DATASETS)
|
|
||||||
print('Estimated time: ~35-40h (trivia_qa and hle are slow)')
|
|
||||||
print('Tip: use --limit 500 to cap runtime if needed')
|
|
||||||
run_module.main()
|
|
||||||
@ -1,40 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
Lite evaluation suite: ~2-4h
|
|
||||||
Covers all 5 capability domains with 1-2 benchmarks each.
|
|
||||||
Use --limit 20 (default) for quick smoke testing.
|
|
||||||
"""
|
|
||||||
from pathlib import Path
|
|
||||||
import sys
|
|
||||||
|
|
||||||
sys.path.insert(0, str(Path(__file__).parent))
|
|
||||||
|
|
||||||
import run as run_module
|
|
||||||
|
|
||||||
run_module._multi_run_order = [
|
|
||||||
'aime24', # reasoning
|
|
||||||
'humaneval', # code
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module._single_run_order = [
|
|
||||||
'gsm8k', # reasoning
|
|
||||||
'mmlu_pro', # knowledge
|
|
||||||
'simple_qa', # knowledge
|
|
||||||
'longbench_v2', # long-context
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module._agent_order = [
|
|
||||||
'bfcl_v3', # tool/function calling
|
|
||||||
]
|
|
||||||
|
|
||||||
run_module.DATASETS = (
|
|
||||||
run_module._multi_run_order
|
|
||||||
+ run_module._single_run_order
|
|
||||||
+ run_module._agent_order
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
print('Lite benchmarks:', run_module.DATASETS)
|
|
||||||
run_module.main()
|
|
||||||
|
|
||||||
@ -1,45 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""Run a single benchmark for quick smoke tests.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
python bash/run_single.py bigcodebench --limit 1
|
|
||||||
python bash/run_single.py bfcl_v3 --limit 5 --model MODEL --api-url URL
|
|
||||||
"""
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
sys.path.insert(0, str(Path(__file__).parent))
|
|
||||||
import run as run_mod
|
|
||||||
from evalscope import run_task
|
|
||||||
|
|
||||||
if len(sys.argv) < 2:
|
|
||||||
print('Usage: python bash/run_single.py <dataset_name> [--limit N] [--model M] [--api-url U]')
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
dataset_name = sys.argv[1]
|
|
||||||
limit = None
|
|
||||||
for i, arg in enumerate(sys.argv):
|
|
||||||
if arg == '--limit' and i + 1 < len(sys.argv):
|
|
||||||
v = sys.argv[i + 1]
|
|
||||||
limit = None if v.lower() in ('none', 'all') else int(v)
|
|
||||||
elif arg == '--model' and i + 1 < len(sys.argv):
|
|
||||||
run_mod.MODEL = sys.argv[i + 1]
|
|
||||||
elif arg == '--api-url' and i + 1 < len(sys.argv):
|
|
||||||
run_mod.API_URL = sys.argv[i + 1]
|
|
||||||
|
|
||||||
if dataset_name not in run_mod.DATASET_CONFIGS:
|
|
||||||
raise SystemExit(f'{dataset_name} not in config')
|
|
||||||
|
|
||||||
run_mod.LIMIT = limit
|
|
||||||
ds_cfg = run_mod.DATASET_CONFIGS[dataset_name]
|
|
||||||
task_cfg = run_mod.build_task_config(
|
|
||||||
dataset_name=dataset_name,
|
|
||||||
ds_cfg=ds_cfg,
|
|
||||||
batch_size=4,
|
|
||||||
enable_thinking=run_mod.ENABLE_THINKING,
|
|
||||||
seed=run_mod.SEED,
|
|
||||||
run_idx=0,
|
|
||||||
)
|
|
||||||
print(f'Running {dataset_name} with limit={limit}')
|
|
||||||
run_task(task_cfg)
|
|
||||||
print('Done.')
|
|
||||||
497
bash/test.py
497
bash/test.py
@ -1,497 +0,0 @@
|
|||||||
import yaml
|
|
||||||
from copy import deepcopy
|
|
||||||
from pathlib import Path
|
|
||||||
import sys
|
|
||||||
import os
|
|
||||||
|
|
||||||
from evalscope import run_task, TaskConfig
|
|
||||||
from evalscope.api.agent import NativeAgentConfig
|
|
||||||
from evalscope.config import SandboxTaskConfig
|
|
||||||
|
|
||||||
SCRIPT_DIR = Path(__file__).parent.resolve()
|
|
||||||
PROJECT_ROOT = SCRIPT_DIR.parent
|
|
||||||
|
|
||||||
|
|
||||||
# ============================================================
|
|
||||||
# ★★★ 必改参数 (USER CONFIG) ★★★
|
|
||||||
# 每次评测新模型前,只需要检查/修改以下参数。
|
|
||||||
# 也可以通过命令行覆盖:--model / --api-url / --dataset-dir /
|
|
||||||
# --output-dir / --limit / --config
|
|
||||||
# ============================================================
|
|
||||||
|
|
||||||
# 模型名(served model name)
|
|
||||||
MODEL = 'DeepSeek-V4-Flash-Int8'
|
|
||||||
|
|
||||||
# 模型服务地址(OpenAI 兼容 API)
|
|
||||||
API_URL = 'http://localhost:30000/v1'
|
|
||||||
|
|
||||||
# 本地数据集根目录(提前下载好的 datasets 目录)
|
|
||||||
DATASET_DIR = str(PROJECT_ROOT / 'datasets')
|
|
||||||
|
|
||||||
# 评测输出目录(每个 benchmark 单独子目录)
|
|
||||||
OUTPUT_DIR = str(PROJECT_ROOT / 'output')
|
|
||||||
|
|
||||||
# 采样数量上限;None 表示跑全部样本
|
|
||||||
LIMIT = None
|
|
||||||
|
|
||||||
# 每个 benchmark 的生成参数配置文件(max_tokens / temperature 等)
|
|
||||||
CONFIG = None
|
|
||||||
|
|
||||||
# 是否开启 thinking 模式(sglang chat_template_kwargs.thinking)
|
|
||||||
ENABLE_THINKING = False
|
|
||||||
|
|
||||||
# 测试哪些 benchmark:见下方 DATASETS = _multi_run_order + _single_run_order + _agent_order
|
|
||||||
# 想只跑部分 benchmark 时,注释掉对应行即可。
|
|
||||||
|
|
||||||
# ============================================================
|
|
||||||
# Command line argument overrides (不需要修改)
|
|
||||||
# ============================================================
|
|
||||||
|
|
||||||
for i, arg in enumerate(sys.argv):
|
|
||||||
if arg == '--limit' and i + 1 < len(sys.argv):
|
|
||||||
limit_val = sys.argv[i + 1]
|
|
||||||
if limit_val.lower() == 'none' or limit_val.lower() == 'all':
|
|
||||||
LIMIT = None
|
|
||||||
else:
|
|
||||||
LIMIT = int(limit_val)
|
|
||||||
elif arg == '--model' and i + 1 < len(sys.argv):
|
|
||||||
MODEL = sys.argv[i + 1]
|
|
||||||
elif arg == '--api-url' and i + 1 < len(sys.argv):
|
|
||||||
API_URL = sys.argv[i + 1]
|
|
||||||
elif arg == '--dataset-dir' and i + 1 < len(sys.argv):
|
|
||||||
DATASET_DIR = sys.argv[i + 1]
|
|
||||||
elif arg == '--output-dir' and i + 1 < len(sys.argv):
|
|
||||||
OUTPUT_DIR = sys.argv[i + 1]
|
|
||||||
elif arg == '--config' and i + 1 < len(sys.argv):
|
|
||||||
CONFIG = sys.argv[i + 1]
|
|
||||||
|
|
||||||
# Benchmarks that require sandboxed code execution
|
|
||||||
SANDBOX_DATASETS = {'humaneval', 'bigcodebench'}
|
|
||||||
# Per-dataset sandbox configs.
|
|
||||||
# bigcodebench's upstream image has ENTRYPOINT ["python3", "-m", "bigcodebench.evaluate"],
|
|
||||||
# which exits immediately and breaks ms_enclave exec. We built a derivative image
|
|
||||||
# `bigcodebench-sandbox:latest` that keeps the same Python environment but drops the
|
|
||||||
# entrypoint and runs `tail -f /dev/null` so the container stays alive.
|
|
||||||
SANDBOX_CONFIGS = {
|
|
||||||
'bigcodebench': {
|
|
||||||
'image': 'bigcodebench-sandbox:latest',
|
|
||||||
'working_dir': '/tmp',
|
|
||||||
'tools_config': {
|
|
||||||
'shell_executor': {},
|
|
||||||
'python_executor': {}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
'humaneval': {
|
|
||||||
'image': 'python:3.11-slim',
|
|
||||||
'tools_config': {
|
|
||||||
'shell_executor': {},
|
|
||||||
'python_executor': {}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
# ============================================================
|
|
||||||
# User-tunable parameters
|
|
||||||
# ============================================================
|
|
||||||
|
|
||||||
# Fixed seed for reproducibility. With temperature > 0 the model still samples
|
|
||||||
# randomly, so running N times with the same seed still yields variance.
|
|
||||||
SEED = 42
|
|
||||||
|
|
||||||
# Benchmarks to run multiple times with temperature=1.0.
|
|
||||||
# Format: {benchmark_name: num_runs}
|
|
||||||
# Number of runs chosen so total samples ≈ 400-500 per benchmark.
|
|
||||||
MULTI_RUN_CONFIG = {
|
|
||||||
# ~30 samples each -> 12 runs = ~360 samples
|
|
||||||
'aime24': 12,
|
|
||||||
'aime25': 12,
|
|
||||||
'aime26': 12,
|
|
||||||
'hmmt26': 12,
|
|
||||||
# ~100 samples -> 5 runs = ~500 samples
|
|
||||||
'live_code_bench': 5,
|
|
||||||
# ~120 samples -> 4 runs = ~480 samples
|
|
||||||
'imo_answerbench': 4,
|
|
||||||
# ~164 samples -> 3 runs = ~492 samples
|
|
||||||
'humaneval': 3,
|
|
||||||
# ~198 samples -> 2 runs = ~396 samples
|
|
||||||
'gpqa_diamond': 2,
|
|
||||||
}
|
|
||||||
|
|
||||||
# Single-run benchmarks (temperature=0.0 greedy)
|
|
||||||
SINGLE_RUN_DATASETS = [
|
|
||||||
'bigcodebench',
|
|
||||||
'bfcl_v3', # function calling: greedy decoding
|
|
||||||
'competition_math',
|
|
||||||
'gsm8k',
|
|
||||||
'hle',
|
|
||||||
'super_gpqa',
|
|
||||||
'arc',
|
|
||||||
'bbh',
|
|
||||||
'cmmlu',
|
|
||||||
'drop',
|
|
||||||
'hellaswag',
|
|
||||||
'mmlu',
|
|
||||||
'mmlu_pro',
|
|
||||||
'simple_qa',
|
|
||||||
'trivia_qa',
|
|
||||||
'winogrande',
|
|
||||||
'openai_mrcr', # long-context, run once
|
|
||||||
'longbench_v2', # long-context, run once
|
|
||||||
]
|
|
||||||
|
|
||||||
# Agent / tool benchmarks (temperature=0.0 greedy, run once)
|
|
||||||
AGENT_DATASETS = [
|
|
||||||
'tau2_bench',
|
|
||||||
'general_fc',
|
|
||||||
]
|
|
||||||
|
|
||||||
# Combine all datasets in order (shortest estimated time first)
|
|
||||||
# 1. Multi-run small benchmarks (temperature=1.0 for sampling diversity)
|
|
||||||
_multi_run_order = [
|
|
||||||
# 'humaneval', # ~164 samples x 3 runs = 492
|
|
||||||
# 'live_code_bench', # ~100 samples x 5 runs = 500
|
|
||||||
# 'aime26', # ~30 samples x 12 runs = 360
|
|
||||||
# 'aime24', # ~30 samples x 12 runs = 360
|
|
||||||
# 'aime25', # ~30 samples x 12 runs = 360
|
|
||||||
# 'gpqa_diamond', # ~198 samples x 2 runs = 396
|
|
||||||
# 'imo_answerbench', # ~120 samples x 4 runs = 480
|
|
||||||
# 'hmmt26', # ~30 samples x 12 runs = 360
|
|
||||||
]
|
|
||||||
|
|
||||||
# 2. Single-run benchmarks (temperature=0.0 greedy)
|
|
||||||
_single_run_order = [
|
|
||||||
# 'arc',
|
|
||||||
# 'bfcl_v3', # function calling: greedy decoding
|
|
||||||
# 'winogrande',
|
|
||||||
# 'competition_math',
|
|
||||||
# 'gsm8k',
|
|
||||||
# 'hellaswag',
|
|
||||||
'bigcodebench',
|
|
||||||
# 'drop',
|
|
||||||
# 'bbh',
|
|
||||||
# 'openai_mrcr',
|
|
||||||
# 'longbench_v2',
|
|
||||||
# 'mmlu',
|
|
||||||
# 'cmmlu',
|
|
||||||
# 'super_gpqa',
|
|
||||||
# 'simple_qa',
|
|
||||||
# 'mmlu_pro',
|
|
||||||
'hle',
|
|
||||||
# 'trivia_qa',
|
|
||||||
]
|
|
||||||
|
|
||||||
# 3. Agent / tool benchmarks (last)
|
|
||||||
_agent_order = [
|
|
||||||
# 'tau2_bench',
|
|
||||||
# 'general_fc',
|
|
||||||
]
|
|
||||||
|
|
||||||
DATASETS = _multi_run_order + _single_run_order + _agent_order
|
|
||||||
|
|
||||||
# ENABLE_THINKING 已移至文件头部「必改参数」区
|
|
||||||
BATCH_SIZE_LIST = [4]
|
|
||||||
# LIMIT is parsed from command line: --limit N
|
|
||||||
SHUFFLE = True
|
|
||||||
|
|
||||||
# ============================================================
|
|
||||||
# Fixed configuration
|
|
||||||
# ============================================================
|
|
||||||
|
|
||||||
CONFIG_PATH = str(Path(CONFIG) if CONFIG else SCRIPT_DIR / 'config' / 'dpv4-int8_nothinking.yaml')
|
|
||||||
if not Path(CONFIG_PATH).exists():
|
|
||||||
raise FileNotFoundError(f'Config file not found: {CONFIG_PATH}')
|
|
||||||
|
|
||||||
MODEL_PATH = '/data1/models/DeepSeek-V4-Flash-INT8'
|
|
||||||
|
|
||||||
# Judge model for benchmarks that require LLM-as-judge (simple_qa, tau2_bench, etc.)
|
|
||||||
# Vectron endpoint with DeepSeek-V4-Pro
|
|
||||||
JUDGE_MODEL_ARGS = {
|
|
||||||
'model_id': 'DeepSeek/DeepSeek-V4-Pro',
|
|
||||||
'api_url': 'https://api.vectron.meta-stone.com/v1',
|
|
||||||
'api_key': 'sk-dbd8a665f7634081b87ec409c7636500',
|
|
||||||
'eval_type': 'openai_api',
|
|
||||||
'generation_config': {
|
|
||||||
'temperature': 0.0,
|
|
||||||
'max_tokens': 10240,
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
TRUNCATION_CONFIG = {
|
|
||||||
'longbench_v2': 32768*4,
|
|
||||||
'openai_mrcr': 32768*4,
|
|
||||||
}
|
|
||||||
|
|
||||||
MATH_DATASETS = {
|
|
||||||
'aime24', 'aime25', 'aime26', 'hmmt26',
|
|
||||||
'gsm8k', 'competition_math', 'imo_answerbench',
|
|
||||||
}
|
|
||||||
|
|
||||||
MATH_PROMPT_TEMPLATE = (
|
|
||||||
"{question}\n"
|
|
||||||
"Please reason step by step, and put your final answer within \\boxed{{}}."
|
|
||||||
)
|
|
||||||
|
|
||||||
with open(CONFIG_PATH, 'r', encoding='utf-8') as f:
|
|
||||||
DATASET_CONFIGS = yaml.safe_load(f)
|
|
||||||
|
|
||||||
|
|
||||||
# ============================================================
|
|
||||||
# Middle-truncation helpers
|
|
||||||
# ============================================================
|
|
||||||
|
|
||||||
_TOKENIZER = None
|
|
||||||
|
|
||||||
|
|
||||||
def get_tokenizer():
|
|
||||||
global _TOKENIZER
|
|
||||||
if _TOKENIZER is None:
|
|
||||||
from transformers import AutoTokenizer
|
|
||||||
try:
|
|
||||||
# Try local path first
|
|
||||||
_TOKENIZER = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True)
|
|
||||||
except Exception:
|
|
||||||
# Fallback to HuggingFace model ID
|
|
||||||
_TOKENIZER = AutoTokenizer.from_pretrained('deepseek-ai/DeepSeek-V4-Flash', trust_remote_code=True)
|
|
||||||
return _TOKENIZER
|
|
||||||
|
|
||||||
|
|
||||||
def truncate_middle(text: str, max_tokens: int) -> str:
|
|
||||||
if max_tokens <= 0:
|
|
||||||
return text
|
|
||||||
tokenizer = get_tokenizer()
|
|
||||||
token_ids = tokenizer.encode(text, add_special_tokens=False)
|
|
||||||
if len(token_ids) <= max_tokens:
|
|
||||||
return text
|
|
||||||
keep_head = max_tokens // 2
|
|
||||||
keep_tail = max_tokens - keep_head
|
|
||||||
truncated_ids = token_ids[:keep_head] + token_ids[-keep_tail:]
|
|
||||||
return tokenizer.decode(truncated_ids, skip_special_tokens=True)
|
|
||||||
|
|
||||||
|
|
||||||
def _patch_adapters_for_truncation():
|
|
||||||
from evalscope.benchmarks.longbench_v2.longbench_v2_adapter import LongBenchV2Adapter
|
|
||||||
from evalscope.benchmarks.openai_mrcr.openai_mrcr_adapter import OpenAIMRCRAdapter
|
|
||||||
|
|
||||||
# Patch LongBenchV2Adapter.format_prompt_template
|
|
||||||
_orig_longbench_format = LongBenchV2Adapter.format_prompt_template
|
|
||||||
def _patched_longbench_format(self, sample):
|
|
||||||
max_tok = TRUNCATION_CONFIG.get('longbench_v2')
|
|
||||||
if max_tok and sample.metadata and 'context' in sample.metadata:
|
|
||||||
sample.metadata['context'] = truncate_middle(sample.metadata['context'], max_tok)
|
|
||||||
return _orig_longbench_format(self, sample)
|
|
||||||
LongBenchV2Adapter.format_prompt_template = _patched_longbench_format
|
|
||||||
|
|
||||||
# Patch OpenAIMRCRAdapter.record_to_sample with needle-aware truncation.
|
|
||||||
# MRCR is a long chat history with needles hidden at desired_msg_index.
|
|
||||||
# We keep the head, tail, and a window around the needle, and truncate
|
|
||||||
# each kept message if it is still too long. This preserves the retrieval
|
|
||||||
# task while fitting GPU memory.
|
|
||||||
_orig_mrcr_record = OpenAIMRCRAdapter.record_to_sample
|
|
||||||
def _patched_mrcr_record(self, record):
|
|
||||||
import json
|
|
||||||
max_total_tok = TRUNCATION_CONFIG.get('openai_mrcr')
|
|
||||||
per_msg_max_tok = 8192
|
|
||||||
if max_total_tok and 'prompt' in record:
|
|
||||||
try:
|
|
||||||
prompt_data = json.loads(record['prompt'])
|
|
||||||
if not isinstance(prompt_data, list) or len(prompt_data) == 0:
|
|
||||||
return _orig_mrcr_record(self, record)
|
|
||||||
|
|
||||||
tokenizer = get_tokenizer()
|
|
||||||
total_tok = sum(
|
|
||||||
len(tokenizer.encode(msg.get('content', '') if isinstance(msg, dict) else '', add_special_tokens=False))
|
|
||||||
for msg in prompt_data
|
|
||||||
)
|
|
||||||
if total_tok <= max_total_tok:
|
|
||||||
return _orig_mrcr_record(self, record)
|
|
||||||
|
|
||||||
desired_idx = record.get('desired_msg_index', 0)
|
|
||||||
if not isinstance(desired_idx, int) or desired_idx < 0 or desired_idx >= len(prompt_data):
|
|
||||||
desired_idx = 0
|
|
||||||
|
|
||||||
n = len(prompt_data)
|
|
||||||
keep = set()
|
|
||||||
# Head and tail context
|
|
||||||
keep.update(range(min(2, n)))
|
|
||||||
keep.update(range(max(0, n - 2), n))
|
|
||||||
# Window around the needle
|
|
||||||
window = 2
|
|
||||||
keep.update(range(max(0, desired_idx - window), min(n, desired_idx + window + 1)))
|
|
||||||
keep = sorted(keep)
|
|
||||||
|
|
||||||
new_prompt = []
|
|
||||||
for idx in keep:
|
|
||||||
msg = prompt_data[idx]
|
|
||||||
if isinstance(msg, dict):
|
|
||||||
msg = dict(msg)
|
|
||||||
content = msg.get('content', '')
|
|
||||||
if len(tokenizer.encode(content, add_special_tokens=False)) > per_msg_max_tok:
|
|
||||||
msg['content'] = truncate_middle(content, per_msg_max_tok)
|
|
||||||
new_prompt.append(msg)
|
|
||||||
|
|
||||||
record = dict(record)
|
|
||||||
record['prompt'] = json.dumps(new_prompt)
|
|
||||||
except (json.JSONDecodeError, TypeError):
|
|
||||||
pass
|
|
||||||
return _orig_mrcr_record(self, record)
|
|
||||||
OpenAIMRCRAdapter.record_to_sample = _patched_mrcr_record
|
|
||||||
|
|
||||||
|
|
||||||
_patch_adapters_for_truncation()
|
|
||||||
|
|
||||||
|
|
||||||
# ============================================================
|
|
||||||
# Helpers
|
|
||||||
# ============================================================
|
|
||||||
|
|
||||||
def configure_thinking(generation_config: dict, enable: bool) -> dict:
|
|
||||||
extra_body = generation_config.get('extra_body', {})
|
|
||||||
chat_template_kwargs = extra_body.get('chat_template_kwargs', {})
|
|
||||||
if enable:
|
|
||||||
chat_template_kwargs['thinking'] = True
|
|
||||||
else:
|
|
||||||
chat_template_kwargs.pop('thinking', None)
|
|
||||||
if chat_template_kwargs:
|
|
||||||
extra_body['chat_template_kwargs'] = chat_template_kwargs
|
|
||||||
if extra_body:
|
|
||||||
generation_config['extra_body'] = extra_body
|
|
||||||
return generation_config
|
|
||||||
|
|
||||||
|
|
||||||
def build_agent_config(agent_cfg: dict) -> NativeAgentConfig:
|
|
||||||
agent_cfg = deepcopy(agent_cfg or {})
|
|
||||||
known_fields = {'mode', 'strategy', 'tools', 'max_steps', 'mcp_servers', 'environment', 'environment_extra'}
|
|
||||||
kwargs = agent_cfg.pop('kwargs', {})
|
|
||||||
for key in list(agent_cfg.keys()):
|
|
||||||
if key not in known_fields:
|
|
||||||
kwargs[key] = agent_cfg.pop(key)
|
|
||||||
if kwargs:
|
|
||||||
agent_cfg['kwargs'] = kwargs
|
|
||||||
return NativeAgentConfig(**agent_cfg)
|
|
||||||
|
|
||||||
|
|
||||||
def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_thinking: bool, seed: int, run_idx: int = 0) -> TaskConfig:
|
|
||||||
# Each run gets its own work_dir so use_cache does not reuse predictions
|
|
||||||
# across repeated samples. This is required for temperature=1.0 multi-run
|
|
||||||
# benchmarks to actually measure variance.
|
|
||||||
if run_idx > 0:
|
|
||||||
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}_run_{run_idx}'
|
|
||||||
else:
|
|
||||||
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}'
|
|
||||||
work_dir.mkdir(parents=True, exist_ok=True)
|
|
||||||
work_dir = str(work_dir)
|
|
||||||
|
|
||||||
generation_config = configure_thinking(deepcopy(ds_cfg['generation_config']), enable_thinking)
|
|
||||||
dataset_args = deepcopy(ds_cfg.get('dataset_args', {}))
|
|
||||||
dataset_args.setdefault('shuffle', True)
|
|
||||||
|
|
||||||
if dataset_name in MATH_DATASETS:
|
|
||||||
dataset_args['prompt_template'] = MATH_PROMPT_TEMPLATE
|
|
||||||
|
|
||||||
dataset_args_dict = {dataset_name: dataset_args}
|
|
||||||
|
|
||||||
agent_config = None
|
|
||||||
if 'agent_config' in ds_cfg:
|
|
||||||
agent_config = build_agent_config(ds_cfg['agent_config'])
|
|
||||||
|
|
||||||
return TaskConfig(
|
|
||||||
model=MODEL,
|
|
||||||
api_url=API_URL,
|
|
||||||
eval_type='openai_api',
|
|
||||||
dataset_dir=DATASET_DIR,
|
|
||||||
judge_model_args=JUDGE_MODEL_ARGS,
|
|
||||||
seed=seed,
|
|
||||||
limit=LIMIT,
|
|
||||||
collect_perf=True,
|
|
||||||
no_timestamp=True,
|
|
||||||
work_dir=work_dir,
|
|
||||||
use_cache=work_dir,
|
|
||||||
datasets=[dataset_name],
|
|
||||||
generation_config=generation_config,
|
|
||||||
dataset_args=dataset_args_dict,
|
|
||||||
agent_config=agent_config,
|
|
||||||
eval_batch_size=batch_size,
|
|
||||||
sandbox=SandboxTaskConfig(
|
|
||||||
enabled=True,
|
|
||||||
engine='docker',
|
|
||||||
default_config=SANDBOX_CONFIGS.get(dataset_name, {
|
|
||||||
'image': 'python:3.11-slim',
|
|
||||||
'tools_config': {
|
|
||||||
'shell_executor': {},
|
|
||||||
'python_executor': {}
|
|
||||||
}
|
|
||||||
})
|
|
||||||
) if dataset_name in SANDBOX_DATASETS else None,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
# ============================================================
|
|
||||||
# Main loop
|
|
||||||
# ============================================================
|
|
||||||
|
|
||||||
def main():
|
|
||||||
print(f"Config: {CONFIG_PATH}")
|
|
||||||
print(f"Model: {MODEL}")
|
|
||||||
print(f"API URL: {API_URL}")
|
|
||||||
print(f"Dataset Dir: {DATASET_DIR}")
|
|
||||||
print(f"Output Dir: {OUTPUT_DIR}")
|
|
||||||
print(f"Limit: {LIMIT if LIMIT is not None else 'ALL'}")
|
|
||||||
print(f"Total benchmarks: {len(DATASETS)}")
|
|
||||||
print("="*60)
|
|
||||||
|
|
||||||
for batch_size in BATCH_SIZE_LIST:
|
|
||||||
# 1. Run multi-run benchmarks (temperature=1.0, same seed, variance from sampling)
|
|
||||||
for dataset_name in _multi_run_order:
|
|
||||||
if dataset_name not in DATASET_CONFIGS:
|
|
||||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
|
||||||
continue
|
|
||||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
|
||||||
num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1)
|
|
||||||
for run_idx in range(num_runs):
|
|
||||||
print(f"\n{'='*60}")
|
|
||||||
print(f"Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={SEED})")
|
|
||||||
print(f"{'='*60}")
|
|
||||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED, run_idx=run_idx)
|
|
||||||
try:
|
|
||||||
run_task(task_cfg)
|
|
||||||
except Exception as e:
|
|
||||||
print(f"ERROR in {dataset_name} (run {run_idx + 1}): {e}")
|
|
||||||
continue
|
|
||||||
|
|
||||||
# 2. Run single-run benchmarks (temperature=0.0 greedy)
|
|
||||||
for dataset_name in _single_run_order:
|
|
||||||
if dataset_name not in DATASET_CONFIGS:
|
|
||||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
|
||||||
continue
|
|
||||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
|
||||||
print(f"\n{'='*60}")
|
|
||||||
print(f"Running: {dataset_name} (seed={SEED})")
|
|
||||||
print(f"{'='*60}")
|
|
||||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED)
|
|
||||||
try:
|
|
||||||
run_task(task_cfg)
|
|
||||||
except Exception as e:
|
|
||||||
print(f"ERROR in {dataset_name}: {e}")
|
|
||||||
continue
|
|
||||||
|
|
||||||
# 3. Run agent benchmarks
|
|
||||||
for dataset_name in _agent_order:
|
|
||||||
if dataset_name not in DATASET_CONFIGS:
|
|
||||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
|
||||||
continue
|
|
||||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
|
||||||
print(f"\n{'='*60}")
|
|
||||||
print(f"Running: {dataset_name} (seed={SEED})")
|
|
||||||
print(f"{'='*60}")
|
|
||||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED)
|
|
||||||
try:
|
|
||||||
run_task(task_cfg)
|
|
||||||
except Exception as e:
|
|
||||||
print(f"ERROR in {dataset_name}: {e}")
|
|
||||||
continue
|
|
||||||
|
|
||||||
print("\nAll benchmarks done!")
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
main()
|
|
||||||
26
scripts/run_docker_full.sh
Executable file
26
scripts/run_docker_full.sh
Executable file
@ -0,0 +1,26 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# Docker 全量评测(Full)
|
||||||
|
# 请根据实际机器修改 MODEL / API_URL / HOST_EVALSCOPE 路径
|
||||||
|
|
||||||
|
MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}"
|
||||||
|
API_URL="${API_URL:-http://localhost:30000/v1}"
|
||||||
|
HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}"
|
||||||
|
OUTPUT_DIR="${OUTPUT_DIR:-/opt/evalscope/output_full}"
|
||||||
|
|
||||||
|
docker run -it --rm \
|
||||||
|
--network host \
|
||||||
|
-v "${HOST_EVALSCOPE}:/opt/evalscope" \
|
||||||
|
-v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \
|
||||||
|
-v "${HOST_EVALSCOPE}/output_full:/opt/evalscope/output_full" \
|
||||||
|
-v /var/run/docker.sock:/var/run/docker.sock \
|
||||||
|
evalscope-complete-py312:latest \
|
||||||
|
bash -c "
|
||||||
|
cd /opt/evalscope &&
|
||||||
|
python bash/run.py \
|
||||||
|
--model ${MODEL} \
|
||||||
|
--api-url ${API_URL} \
|
||||||
|
--dataset-dir /opt/evalscope \
|
||||||
|
--output-dir ${OUTPUT_DIR} \
|
||||||
|
--suite full \
|
||||||
|
--limit none
|
||||||
|
"
|
||||||
27
scripts/run_docker_group.sh
Executable file
27
scripts/run_docker_group.sh
Executable file
@ -0,0 +1,27 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# Docker 多机分组评测
|
||||||
|
# 用法:GROUP=1 ./run_docker_group.sh
|
||||||
|
|
||||||
|
GROUP="${GROUP:-1}"
|
||||||
|
MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}"
|
||||||
|
API_URL="${API_URL:-http://localhost:30000/v1}"
|
||||||
|
HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}"
|
||||||
|
OUTPUT_DIR="/opt/evalscope/output_group${GROUP}"
|
||||||
|
|
||||||
|
docker run -it --rm \
|
||||||
|
--network host \
|
||||||
|
-v "${HOST_EVALSCOPE}:/opt/evalscope" \
|
||||||
|
-v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \
|
||||||
|
-v "${HOST_EVALSCOPE}/output_group${GROUP}:/opt/evalscope/output_group${GROUP}" \
|
||||||
|
-v /var/run/docker.sock:/var/run/docker.sock \
|
||||||
|
evalscope-complete-py312:latest \
|
||||||
|
bash -c "
|
||||||
|
cd /opt/evalscope &&
|
||||||
|
python bash/run.py \
|
||||||
|
--model ${MODEL} \
|
||||||
|
--api-url ${API_URL} \
|
||||||
|
--dataset-dir /opt/evalscope \
|
||||||
|
--output-dir ${OUTPUT_DIR} \
|
||||||
|
--suite group${GROUP} \
|
||||||
|
--limit none
|
||||||
|
"
|
||||||
25
scripts/run_docker_lite.sh
Executable file
25
scripts/run_docker_lite.sh
Executable file
@ -0,0 +1,25 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# Docker Lite 快速冒烟
|
||||||
|
# 方式一:使用 --suite lite
|
||||||
|
|
||||||
|
MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}"
|
||||||
|
API_URL="${API_URL:-http://localhost:30000/v1}"
|
||||||
|
HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}"
|
||||||
|
|
||||||
|
docker run -it --rm \
|
||||||
|
--network host \
|
||||||
|
-v "${HOST_EVALSCOPE}:/opt/evalscope" \
|
||||||
|
-v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \
|
||||||
|
-v "${HOST_EVALSCOPE}/output_lite:/opt/evalscope/output_lite" \
|
||||||
|
-v /var/run/docker.sock:/var/run/docker.sock \
|
||||||
|
evalscope-complete-py312:latest \
|
||||||
|
bash -c "
|
||||||
|
cd /opt/evalscope &&
|
||||||
|
python bash/run.py \
|
||||||
|
--model ${MODEL} \
|
||||||
|
--api-url ${API_URL} \
|
||||||
|
--dataset-dir /opt/evalscope \
|
||||||
|
--output-dir /opt/evalscope/output_lite \
|
||||||
|
--suite lite \
|
||||||
|
--limit none
|
||||||
|
"
|
||||||
Loading…
x
Reference in New Issue
Block a user