all
This commit is contained in:
parent
cfa58f869f
commit
03b49a39d0
29
P800模型能力评测结果 - DS4-Flash-INT8-NO-Thinking-2.0-FULL.csv
Normal file
29
P800模型能力评测结果 - DS4-Flash-INT8-NO-Thinking-2.0-FULL.csv
Normal file
@ -0,0 +1,29 @@
|
||||
分类,Benchmark,得分,实测时间(h),总样本数,延迟_mean(s),输出TPS,请求QPS,输入tokens_mean,输出tokens_mean,累计总tokens,TTFT_mean(s),TTFT P90,TTFT P99,TPOT_mean(s),TPOT P90,TPOT P99,,,,,,,,,,,,,,,
|
||||
代码与工程,bigcodebench,0.9956,0.9915,1140,10.15508,32.99,0.0985,151.77,335.03,554946,0.26441,0.27379,0.42952,0.02957,0.03059,0.03143,,,,,,,,,,,,,,,
|
||||
,humaneval,0.9044,0.1183,164,2.776502,42.58667,0.3602,163.0793,118.2398,45994,0.25521,0.304186,0.451449,0.021783,0.024498,0.027453,,,,,,,,,,,,,,,
|
||||
,live_code_bench,0.6756,7.9486,100,10.41723,43.578,0.09644,518.1118,453.5784,1040894,0.253597,0.272382,0.464771,0.02093,0.023769,0.026398,,,,,,,,,,,,,,,
|
||||
推理与数学,aime24,0.6167,1.2217,30,42.44225,44.85,0.024108,119.3333,1901.842,55169,0.258426,0.274508,0.444511,0.021698,0.025218,hle,,,,,,,,,,,,,,,
|
||||
,aime25,0.4639,2.0414,30,59.98314,46.03583,0.017092,188.4333,2761.283,80482,0.257489,0.273448,0.4417,0.021655,0.024778,0.026626,,,,,,,,,,,,,,,
|
||||
,aime26,0.5611,1.4847,30,51.64098,44.66417,0.019925,139.7667,2305.069,60102,0.25465,0.268527,0.435013,0.021667,0.024596,0.026089,,,,,,,,,,,,,,,
|
||||
,hmmt26,0.4015,1.5481,30,50.8562,44.73,0.020008,108.4242,2274.237,63775,0.255456,0.266018,0.431919,0.022514,0.025679,0.027263,,,,,,,,,,,,,,,
|
||||
,imo_answerbench,0.36,1.2508,400,39.6005,44.94,0.0253,122.1,1779.8,753159,0.2466,0.2641,0.3167,0.0226,0.0257,0.0285,,,,,,,,,,,,,,,
|
||||
,hle,0.0508,14.6863,3000,61.787217,34.42,0.0162,292.3252,2126.7916,6047792,0.275126,0.282138,0.474372,0.028997,0.029831,0.031026,,,,,,,,,,,,,,,
|
||||
,gsm8k,0.9704,0.3178,1319,3.031098,43.8,0.3299,582.9507,132.7726,944039,0.253185,0.267307,0.45469,0.021127,0.02357,0.025568,,,,,,,,,,,,,,,
|
||||
,competition_math,0.941,3.1817,500,8.681756,51,0.1152,296.5698,442.732,3696509,0.245456,0.263328,0.435642,0.019541,0.021677,0.024484,,,,,,,,,,,,,,,
|
||||
,bbh,0.9032,1.8036,6513,3.561695,42.95,0.2808,916.98,152.9719,6966457,0.269139,0.276913,0.467974,0.023935,0.030921,0.065977,,,,,,,,,,,,,,,
|
||||
,drop,0.8183,1.5336,9536,1.824962,36.31,0.548,1273.607,66.25849,12776955,0.297489,0.4593,0.505599,0.023644,0.028072,0.034489,,,,,,,,,,,,,,,
|
||||
知识与语言理解,gpqa_diamond,0.6944,0.3164,198,10.54948,41.335,0.0948,252.4343,436.0202,135770,0.245972,0.26274,0.456931,0.023819,0.027653,0.030237,,,,,,,,,,,,,,,
|
||||
,mmlu_pro,0.8285,5.6167,12032,6.264368,31.05,0.1596,1312.528,194.5092,18132672,0.293428,0.437219,0.477191,0.031274,0.034267,0.036824,,,,,,,,,,,,,,,
|
||||
,simple_qa,0.3814,1.9014,4326,1.88825,30.69,0.529590891,30.351826,57.955155,382016,0.26334,0.269454,0.452375,0.027793,0.031603,0.035322,,,,,,,,,,,,,,,
|
||||
,mmlu,0.9045,3.0297,14042,2.6957,35.98,0.371,727.6,97,11532337,0.2643,0.2751,0.4663,0.026,0.0305,0.0348,,,,,,,,,,,,,,,
|
||||
,cmmlu,0.9001,3.4935,11515,3.9225,36.65,0.2549,115.3,143.8,2983005,0.2553,0.2956,0.443,0.0266,0.0307,0.0353,,,,,,,,,,,,,,,
|
||||
,arc,0.9476,0.2103,1172,0.662452,13.65,1.5095,102.0333,9.042841,394098,0.321567,0.437672,0.449777,0.066642,0.129754,0.132577,,,,,,,,,,,,,,,
|
||||
,hellaswag,0.8666,0.4453,10042,0.597052,7.32,1.6749,219.8686,4.370444,2251808,0.341421,0.448449,0.461157,0.081246,0.133338,0.136106,,,,,,,,,,,,,,,
|
||||
,trivia_qa,0.7768,5.3011,11313,9.163784,3.78,0.1091,11964.972,34.6152,95912700,4.722344,8.951851,16.512039,0.131478,0.292489,0.622806,,,,,,,,,,,,,,,
|
||||
,winogrande,0.7845,0.0775,1267,0.695127,13.88,1.4386,76.11997,9.646409,108666,0.318495,0.433513,0.438349,0.069826,0.129328,0.130936,,,,,,,,,,,,,,,
|
||||
长上下文,longbench_v2,0.5686,3.0638,503,84.120374,2.75,0.0119,85638.8191,231.6441,43192843,25.148194,53.202063,64.267424,0.248375,0.534405,0.946107,,,,,,,,,,,,,,,
|
||||
,openai_mrcr,0.7768,5.2828,2399,29.1532,12.95,0.0343,24774.8,377.6,60340773,6.6088,21.0867,37.983,0.0608,0.1125,0.2004,,,,,,,,,,,,,,,
|
||||
智能体与工具,tau2_bench,0.7684,1.7625,,,,,,,,,,,,,,,,,,,,,,,,,,,,
|
||||
,general_fc,0.6988,1.3803,2000,9.4082,39.21,0.1063,1619.2,368.9,3976219,0.4226,0.7249,1.2941,0.026,0.0307,0.0398,,,,,,,,,,,,,,,
|
||||
,bfcl_v3,0.6643,5.23,4441,4.296605,25.32,0.2327,7268.4069,108.7864,97061732,0.418217,0.521432,1.106471,0.035785,0.039796,0.053177,,,,,,,,,,,,,,,
|
||||
,总计,0.711992593,75.2394,,,,,,,,,,,,,,,,,,,,,,,,,,,,
|
||||
|
@ -1,94 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Fill simple_qa (fresh re-run) + bfcl_v3 (official OVERALL) into the Excel.
|
||||
|
||||
simple_qa: old cells had D=0.0005 (cache-resume artifact) and empty perf.
|
||||
Overwrite C/D/F-Q from the NEW report (real perf + ~1.9h duration).
|
||||
bfcl_v3 : top-level score 0.6643 is evalscope's macro-average (non-standard).
|
||||
Official BFCL score = OVERALL subset = 0.569. Update C only.
|
||||
"""
|
||||
import json
|
||||
import glob
|
||||
import openpyxl
|
||||
|
||||
XLSX = '/data1/sora/P800模型能力评测结果_统一格式_filled.xlsx'
|
||||
SHEETS = ['2.0-FULL', '2.0-Lite']
|
||||
|
||||
# --- read reports ---
|
||||
sq = json.load(open('/data1/sora/evalscope/output/simple_qa/seed_42/reports/DeepSeek-V4-Flash-Int8/simple_qa.json'))
|
||||
bc = json.load(open(glob.glob('/data1/sora/evalscope/output/bfcl_v3/seed_42.bak/reports/DeepSeek-V4-Flash-Int8/bfcl_v3.json')[0]))
|
||||
|
||||
|
||||
def perf(d):
|
||||
pm = d.get('perf_metrics') or {}
|
||||
s = pm.get('summary', {}) or {}
|
||||
u = s.get('usage', {}) or {}
|
||||
tf = s.get('ttft') or {}
|
||||
tp = s.get('tpot') or {}
|
||||
lat = (s.get('latency') or {}).get('mean')
|
||||
return {
|
||||
'score': d.get('score'),
|
||||
'duration_h': (d.get('duration_sec') or 0) / 3600,
|
||||
'latency': lat,
|
||||
'tps': (s.get('throughput') or {}).get('avg_output_tps'),
|
||||
'qps': (1.0 / lat) if lat else None,
|
||||
'in_tok': (u.get('input_tokens') or {}).get('mean'),
|
||||
'out_tok': (u.get('output_tokens') or {}).get('mean'),
|
||||
'ttc': u.get('total_tokens_count'),
|
||||
'ttft_m': tf.get('mean'),
|
||||
'ttft90': tf.get('90%') or tf.get('p90'),
|
||||
'ttft99': tf.get('99%') or tf.get('p99'),
|
||||
'tpot_m': tp.get('mean'),
|
||||
'tpot90': tp.get('90%') or tp.get('p90'),
|
||||
'tpot99': tp.get('99%') or tp.get('p99'),
|
||||
}
|
||||
|
||||
|
||||
def bfcl_overall(d):
|
||||
for metric in d.get('metrics', []):
|
||||
for cat in metric.get('categories', []):
|
||||
for sub in cat.get('subsets', []):
|
||||
nm = sub.get('name')
|
||||
if nm == 'OVERALL' or (isinstance(nm, list) and 'OVERALL' in nm):
|
||||
return sub.get('score')
|
||||
return None
|
||||
|
||||
|
||||
P = perf(sq)
|
||||
BC_OVERALL = bfcl_overall(bc)
|
||||
print('simple_qa:', {k: (round(v, 4) if isinstance(v, float) else v) for k, v in P.items()})
|
||||
print('bfcl_v3 OVERALL:', BC_OVERALL)
|
||||
|
||||
# col -> simple_qa perf key
|
||||
SQ = {'C': P['score'], 'D': round(P['duration_h'], 4), 'F': P['latency'], 'G': P['tps'],
|
||||
'H': P['qps'], 'I': P['in_tok'], 'J': P['out_tok'], 'K': P['ttc'],
|
||||
'L': P['ttft_m'], 'M': P['ttft90'], 'N': P['ttft99'],
|
||||
'O': P['tpot_m'], 'P': P['tpot90'], 'Q': P['tpot99']}
|
||||
|
||||
wb = openpyxl.load_workbook(XLSX)
|
||||
for sn in SHEETS:
|
||||
ws = wb[sn]
|
||||
for r in range(2, 30):
|
||||
b = ws.cell(r, 2).value
|
||||
if b == 'simple_qa':
|
||||
for col, v in SQ.items():
|
||||
if v is not None:
|
||||
ws[f'{col}{r}'] = int(round(v)) if col == 'K' else round(v, 4) if col in ('C', 'D') else v
|
||||
print(f'{sn} simple_qa row{r}: C={ws[f"C{r}"].value} D={ws[f"D{r}"].value} F={ws[f"F{r}"].value} K={ws[f"K{r}"].value}')
|
||||
elif b == 'bfcl_v3' and BC_OVERALL is not None:
|
||||
ws.cell(r, 3).value = round(BC_OVERALL, 4)
|
||||
print(f'{sn} bfcl_v3 row{r}: C={ws.cell(r,3).value}')
|
||||
# recompute total D
|
||||
tr = next((rr for rr in range(2, ws.max_row + 1) if ws.cell(rr, 2).value == '总计'), None)
|
||||
if tr:
|
||||
tot = 0.0
|
||||
for rr in range(2, tr):
|
||||
v = ws.cell(rr, 4).value
|
||||
try:
|
||||
tot += float(v)
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
ws.cell(tr, 4).value = round(tot, 4)
|
||||
print(f'{sn} 总计 D = {round(tot, 4)}h')
|
||||
|
||||
wb.save(XLSX)
|
||||
print('saved ->', XLSX)
|
||||
@ -1,38 +0,0 @@
|
||||
"""Re-run only the review stage for bigcodebench using existing predictions.
|
||||
|
||||
The upstream `bigcodebench/bigcodebench-evaluate` image has an ENTRYPOINT that
|
||||
runs the evaluator and exits, which kills the ms_enclave sandbox containers.
|
||||
We use the locally built `bigcodebench-sandbox:latest` image (same libraries,
|
||||
entrypoint dropped, runs `tail -f /dev/null`) and re-score the cached
|
||||
predictions without regenerating them.
|
||||
"""
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
import run as run_mod
|
||||
from evalscope import run_task
|
||||
|
||||
DATASET_NAME = 'bigcodebench'
|
||||
BATCH_SIZE = 4
|
||||
|
||||
if DATASET_NAME not in run_mod.DATASET_CONFIGS:
|
||||
raise SystemExit(f'{DATASET_NAME} not found in config')
|
||||
|
||||
ds_cfg = run_mod.DATASET_CONFIGS[DATASET_NAME]
|
||||
task_cfg = run_mod.build_task_config(
|
||||
dataset_name=DATASET_NAME,
|
||||
ds_cfg=ds_cfg,
|
||||
batch_size=BATCH_SIZE,
|
||||
enable_thinking=run_mod.ENABLE_THINKING,
|
||||
seed=run_mod.SEED,
|
||||
run_idx=0,
|
||||
)
|
||||
task_cfg.rerun_review = True
|
||||
|
||||
print(f'Re-running review for {DATASET_NAME}')
|
||||
print(f'Cache/work dir: {task_cfg.work_dir}')
|
||||
print(f'Sandbox config: {task_cfg.sandbox.default_config}')
|
||||
|
||||
run_task(task_cfg)
|
||||
print('Done.')
|
||||
573
bash/run.py
573
bash/run.py
@ -1,8 +1,40 @@
|
||||
import yaml
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Unified benchmark runner for EvalScope.
|
||||
|
||||
A single entry point for lite / mid / full / group1 / group2 / group3 evaluations.
|
||||
All tunable parameters can be controlled via command-line arguments.
|
||||
|
||||
Examples:
|
||||
# Full evaluation (all benchmarks, multi-run for stability)
|
||||
python bash/run.py \
|
||||
--model DeepSeek-V4-Flash-Int8 \
|
||||
--api-url http://localhost:30000/v1 \
|
||||
--dataset-dir /data1/sora/evalscope \
|
||||
--output-dir /data1/sora/evalscope/output \
|
||||
--suite full \
|
||||
--limit none
|
||||
|
||||
# Lite smoke test (~5h with full samples)
|
||||
python bash/run.py --suite lite --limit none
|
||||
|
||||
# Run only selected benchmarks
|
||||
python bash/run.py --datasets aime24,gsm8k,arc --limit 20
|
||||
|
||||
# Custom judge model
|
||||
python bash/run.py \
|
||||
--judge-model deepseek-v4-pro \
|
||||
--judge-api-url https://api.deepseek.com/v1 \
|
||||
--judge-api-key sk-xxx
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from copy import deepcopy
|
||||
from pathlib import Path
|
||||
import sys
|
||||
import os
|
||||
|
||||
import yaml
|
||||
|
||||
from evalscope import run_task, TaskConfig
|
||||
from evalscope.api.agent import NativeAgentConfig
|
||||
@ -12,116 +44,112 @@ SCRIPT_DIR = Path(__file__).parent.resolve()
|
||||
PROJECT_ROOT = SCRIPT_DIR.parent
|
||||
|
||||
# ============================================================
|
||||
# ★★★ 必改参数 (USER CONFIG) ★★★
|
||||
# 每次评测新模型前,只需要检查/修改以下参数。
|
||||
# 也可以通过命令行覆盖:--model / --api-url / --dataset-dir /
|
||||
# --output-dir / --limit / --config
|
||||
# Default configuration (override via CLI)
|
||||
# ============================================================
|
||||
|
||||
# 模型名(served model name)
|
||||
MODEL = 'DeepSeek-V4-Flash-Int8'
|
||||
DEFAULT_MODEL = 'DeepSeek-V4-Flash-Int8'
|
||||
DEFAULT_API_URL = 'http://localhost:30000/v1'
|
||||
DEFAULT_DATASET_DIR = str(PROJECT_ROOT)
|
||||
DEFAULT_OUTPUT_DIR = str(PROJECT_ROOT / 'output')
|
||||
DEFAULT_CONFIG = str(PROJECT_ROOT / 'config' / 'dpv4-int8_nothinking.yaml')
|
||||
DEFAULT_TOKENIZER_PATH = '/data1/models/DeepSeek-V4-Flash-INT8'
|
||||
DEFAULT_LIMIT = None
|
||||
DEFAULT_SEED = 42
|
||||
DEFAULT_BATCH_SIZE = 4
|
||||
DEFAULT_ENABLE_THINKING = False
|
||||
|
||||
# 模型服务地址(OpenAI 兼容 API)
|
||||
API_URL = 'http://localhost:30000/v1'
|
||||
DEFAULT_JUDGE_MODEL = 'DeepSeek/DeepSeek-V4-Pro'
|
||||
DEFAULT_JUDGE_API_URL = 'https://api.vectron.meta-stone.com/v1'
|
||||
DEFAULT_JUDGE_API_KEY = 'sk-dbd8a665f7634081b87ec409c7636500'
|
||||
DEFAULT_JUDGE_MAX_TOKENS = 10240
|
||||
|
||||
# 本地数据集根目录(提前下载好的 datasets 目录)
|
||||
DATASET_DIR = str(PROJECT_ROOT / 'datasets')
|
||||
# 长文本 middle-truncation 上限(token 数)。当前默认 128k。
|
||||
DEFAULT_TRUNCATION_TOKENS = 32768 * 4
|
||||
|
||||
# 评测输出目录(每个 benchmark 单独子目录)
|
||||
OUTPUT_DIR = str(PROJECT_ROOT / 'output')
|
||||
# ============================================================
|
||||
# Benchmark suites
|
||||
# ============================================================
|
||||
|
||||
# 采样数量上限;None 表示跑全部样本
|
||||
LIMIT = None
|
||||
# 多次采样配置:总样本数控制在 ~400-500
|
||||
MULTI_RUN_CONFIG = {
|
||||
'aime24': 12,
|
||||
'aime25': 12,
|
||||
'aime26': 12,
|
||||
'hmmt26': 12,
|
||||
'live_code_bench': 5,
|
||||
'imo_answerbench': 4,
|
||||
'humaneval': 3,
|
||||
'gpqa_diamond': 2,
|
||||
}
|
||||
|
||||
# 每个 benchmark 的生成参数配置文件(max_tokens / temperature 等)
|
||||
CONFIG = None
|
||||
# 能力域完整列表
|
||||
ALL_MULTI_RUN = [
|
||||
'humaneval', 'live_code_bench',
|
||||
'aime24', 'aime25', 'aime26', 'hmmt26',
|
||||
'imo_answerbench', 'gpqa_diamond',
|
||||
]
|
||||
ALL_SINGLE_RUN = [
|
||||
'bigcodebench', 'bfcl_v3', 'competition_math', 'gsm8k', 'hle', 'super_gpqa',
|
||||
'arc', 'bbh', 'cmmlu', 'drop', 'hellaswag', 'mmlu', 'mmlu_pro',
|
||||
'simple_qa', 'trivia_qa', 'winogrande',
|
||||
'openai_mrcr', 'longbench_v2',
|
||||
]
|
||||
ALL_AGENT = ['tau2_bench', 'general_fc']
|
||||
|
||||
# 是否开启 thinking 模式(sglang chat_template_kwargs.thinking)
|
||||
ENABLE_THINKING = False
|
||||
|
||||
# 测试哪些 benchmark:见下方 DATASETS = _multi_run_order + _single_run_order + _agent_order
|
||||
# 想只跑部分 benchmark 时,注释掉对应行即可。
|
||||
MODEL_PATH = '/data1/models/DeepSeek-V4-Flash-INT8'
|
||||
|
||||
# Judge model for benchmarks that require LLM-as-judge (simple_qa, tau2_bench, etc.)
|
||||
# Vectron endpoint with DeepSeek-V4-Pro
|
||||
JUDGE_MODEL_ARGS = {
|
||||
'model_id': 'DeepSeek/DeepSeek-V4-Pro',
|
||||
'api_url': 'https://api.vectron.meta-stone.com/v1',
|
||||
'api_key': 'sk-dbd8a665f7634081b87ec409c7636500',
|
||||
'eval_type': 'openai_api',
|
||||
'generation_config': {
|
||||
'temperature': 0.0,
|
||||
'max_tokens': 10240,
|
||||
# 分组基于 CSV 单次时间 + multi-run 后的 wall time 平衡:
|
||||
# Group1: ~61h | Group2: ~62h | Group3: ~55h
|
||||
SUITES = {
|
||||
'full': {
|
||||
'multi': ALL_MULTI_RUN,
|
||||
'single': ALL_SINGLE_RUN,
|
||||
'agent': ALL_AGENT,
|
||||
},
|
||||
'lite': {
|
||||
'multi': ['aime24', 'humaneval'],
|
||||
'single': ['gsm8k', 'arc', 'longbench_v2'],
|
||||
'agent': ['general_fc'],
|
||||
},
|
||||
'mid': {
|
||||
'multi': ['aime24', 'humaneval'],
|
||||
'single': [
|
||||
'live_code_bench', 'bigcodebench', 'competition_math', 'gsm8k',
|
||||
'gpqa_diamond', 'mmlu_pro', 'simple_qa', 'longbench_v2', 'openai_mrcr',
|
||||
],
|
||||
'agent': ['general_fc', 'tau2_bench'],
|
||||
},
|
||||
# 多机组分组,基于 CSV 实测完整时间(已含 multi-run)平衡:
|
||||
# Group1: ~22.7h | Group2: ~25.6h | Group3: ~27.0h | 合计 ~75.2h
|
||||
'group1': {
|
||||
'multi': ['live_code_bench', 'aime24', 'aime25', 'aime26', 'hmmt26', 'imo_answerbench', 'humaneval'],
|
||||
'single': ['bigcodebench', 'competition_math', 'gsm8k', 'drop', 'arc', 'hellaswag', 'winogrande'],
|
||||
'agent': [],
|
||||
},
|
||||
'group2': {
|
||||
'multi': [],
|
||||
'single': ['hle', 'mmlu_pro', 'trivia_qa'],
|
||||
'agent': [],
|
||||
},
|
||||
'group3': {
|
||||
'multi': ['gpqa_diamond'],
|
||||
'single': ['openai_mrcr', 'longbench_v2', 'bfcl_v3', 'mmlu', 'cmmlu', 'bbh', 'simple_qa'],
|
||||
'agent': ['tau2_bench', 'general_fc'],
|
||||
},
|
||||
}
|
||||
|
||||
TRUNCATION_CONFIG = {
|
||||
'longbench_v2': 32768*4,
|
||||
'openai_mrcr': 32768*4,
|
||||
# ============================================================
|
||||
# Fixed configuration
|
||||
# ============================================================
|
||||
|
||||
MATH_DATASETS = {
|
||||
'aime24', 'aime25', 'aime26', 'hmmt26',
|
||||
'gsm8k', 'competition_math', 'imo_answerbench',
|
||||
}
|
||||
# Combine all datasets in order (shortest estimated time first)
|
||||
# 1. Multi-run small benchmarks (temperature=1.0 for sampling diversity)
|
||||
_multi_run_order = [
|
||||
# 'humaneval', # ~164 samples x 3 runs = 492
|
||||
# 'live_code_bench', # ~100 samples x 5 runs = 500
|
||||
# 'aime26', # ~30 samples x 12 runs = 360
|
||||
# 'aime24', # ~30 samples x 12 runs = 360
|
||||
# 'aime25', # ~30 samples x 12 runs = 360
|
||||
# 'gpqa_diamond', # ~198 samples x 2 runs = 396
|
||||
# 'imo_answerbench', # ~120 samples x 4 runs = 480
|
||||
# 'hmmt26', # ~30 samples x 12 runs = 360
|
||||
]
|
||||
|
||||
# 2. Single-run benchmarks (temperature=0.0 greedy)
|
||||
_single_run_order = [
|
||||
# 'arc',
|
||||
# 'bfcl_v3', # function calling: greedy decoding
|
||||
# 'winogrande',
|
||||
# 'competition_math',
|
||||
# 'gsm8k',
|
||||
# 'hellaswag',
|
||||
# 'bigcodebench',
|
||||
# 'drop',
|
||||
# 'bbh',
|
||||
# 'openai_mrcr',
|
||||
# 'longbench_v2',
|
||||
# 'mmlu',
|
||||
# 'cmmlu',
|
||||
# 'simple_qa',
|
||||
# 'mmlu_pro',
|
||||
'hle',
|
||||
# 'trivia_qa',
|
||||
]
|
||||
MATH_PROMPT_TEMPLATE = (
|
||||
"{question}\n"
|
||||
"Please reason step by step, and put your final answer within \\boxed{{}}."
|
||||
)
|
||||
|
||||
# 3. Agent / tool benchmarks (last)
|
||||
_agent_order = [
|
||||
# 'tau2_bench',
|
||||
# 'general_fc',
|
||||
]
|
||||
# ============================================================
|
||||
# Command line argument overrides (不需要修改)
|
||||
# ============================================================
|
||||
|
||||
for i, arg in enumerate(sys.argv):
|
||||
if arg == '--limit' and i + 1 < len(sys.argv):
|
||||
limit_val = sys.argv[i + 1]
|
||||
if limit_val.lower() == 'none' or limit_val.lower() == 'all':
|
||||
LIMIT = None
|
||||
else:
|
||||
LIMIT = int(limit_val)
|
||||
elif arg == '--model' and i + 1 < len(sys.argv):
|
||||
MODEL = sys.argv[i + 1]
|
||||
elif arg == '--api-url' and i + 1 < len(sys.argv):
|
||||
API_URL = sys.argv[i + 1]
|
||||
elif arg == '--dataset-dir' and i + 1 < len(sys.argv):
|
||||
DATASET_DIR = sys.argv[i + 1]
|
||||
elif arg == '--output-dir' and i + 1 < len(sys.argv):
|
||||
OUTPUT_DIR = sys.argv[i + 1]
|
||||
elif arg == '--config' and i + 1 < len(sys.argv):
|
||||
CONFIG = sys.argv[i + 1]
|
||||
|
||||
# Benchmarks that require sandboxed code execution
|
||||
SANDBOX_DATASETS = {'humaneval', 'bigcodebench'}
|
||||
SANDBOX_CONFIGS = {
|
||||
'bigcodebench': {
|
||||
@ -141,91 +169,69 @@ SANDBOX_CONFIGS = {
|
||||
},
|
||||
}
|
||||
|
||||
# ============================================================
|
||||
# User-tunable parameters
|
||||
# ============================================================
|
||||
|
||||
# Fixed seed for reproducibility. With temperature > 0 the model still samples
|
||||
# randomly, so running N times with the same seed still yields variance.
|
||||
SEED = 42
|
||||
|
||||
# Benchmarks to run multiple times with temperature=1.0.
|
||||
# Format: {benchmark_name: num_runs}
|
||||
# Number of runs chosen so total samples ≈ 400-500 per benchmark.
|
||||
MULTI_RUN_CONFIG = {
|
||||
# ~30 samples each -> 12 runs = ~360 samples
|
||||
'aime24': 12,
|
||||
'aime25': 12,
|
||||
'aime26': 12,
|
||||
'hmmt26': 12,
|
||||
# ~100 samples -> 5 runs = ~500 samples
|
||||
'live_code_bench': 5,
|
||||
# ~120 samples -> 4 runs = ~480 samples
|
||||
'imo_answerbench': 4,
|
||||
# ~164 samples -> 3 runs = ~492 samples
|
||||
'humaneval': 3,
|
||||
# ~198 samples -> 2 runs = ~396 samples
|
||||
'gpqa_diamond': 2,
|
||||
}
|
||||
|
||||
# Single-run benchmarks (temperature=0.0 greedy)
|
||||
SINGLE_RUN_DATASETS = [
|
||||
'bigcodebench',
|
||||
'bfcl_v3', # function calling: greedy decoding
|
||||
'competition_math',
|
||||
'gsm8k',
|
||||
'hle',
|
||||
'super_gpqa',
|
||||
'arc',
|
||||
'bbh',
|
||||
'cmmlu',
|
||||
'drop',
|
||||
'hellaswag',
|
||||
'mmlu',
|
||||
'mmlu_pro',
|
||||
'simple_qa',
|
||||
'trivia_qa',
|
||||
'winogrande',
|
||||
'openai_mrcr', # long-context, run once
|
||||
'longbench_v2', # long-context, run once
|
||||
]
|
||||
|
||||
# Agent / tool benchmarks (temperature=0.0 greedy, run once)
|
||||
AGENT_DATASETS = [
|
||||
'tau2_bench',
|
||||
'general_fc',
|
||||
]
|
||||
|
||||
|
||||
|
||||
DATASETS = _multi_run_order + _single_run_order + _agent_order
|
||||
|
||||
# ENABLE_THINKING 已移至文件头部「必改参数」区
|
||||
BATCH_SIZE_LIST = [4]
|
||||
# LIMIT is parsed from command line: --limit N
|
||||
SHUFFLE = True
|
||||
|
||||
# ============================================================
|
||||
# Fixed configuration
|
||||
# CLI parser
|
||||
# ============================================================
|
||||
|
||||
CONFIG_PATH = str(Path(CONFIG) if CONFIG else SCRIPT_DIR / 'config' / 'dpv4-int8_nothinking.yaml')
|
||||
if not Path(CONFIG_PATH).exists():
|
||||
raise FileNotFoundError(f'Config file not found: {CONFIG_PATH}')
|
||||
def build_parser():
|
||||
parser = argparse.ArgumentParser(
|
||||
description='Unified EvalScope benchmark runner',
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog='Suites: full, lite, mid, group1, group2, group3',
|
||||
)
|
||||
|
||||
# Model / API
|
||||
parser.add_argument('--model', default=DEFAULT_MODEL,
|
||||
help='Served model name (default: %(default)s)')
|
||||
parser.add_argument('--api-url', default=DEFAULT_API_URL,
|
||||
help='OpenAI-compatible API URL (default: %(default)s)')
|
||||
|
||||
MATH_DATASETS = {
|
||||
'aime24', 'aime25', 'aime26', 'hmmt26',
|
||||
'gsm8k', 'competition_math', 'imo_answerbench',
|
||||
}
|
||||
# Paths
|
||||
parser.add_argument('--dataset-dir', default=DEFAULT_DATASET_DIR,
|
||||
help='Parent directory containing datasets/ subdir (default: %(default)s)')
|
||||
parser.add_argument('--output-dir', default=DEFAULT_OUTPUT_DIR,
|
||||
help='Output root directory (default: %(default)s)')
|
||||
parser.add_argument('--config', default=DEFAULT_CONFIG,
|
||||
help='YAML config path (default: %(default)s)')
|
||||
parser.add_argument('--tokenizer-path', default=DEFAULT_TOKENIZER_PATH,
|
||||
help='Local tokenizer path for middle-truncation (default: %(default)s)')
|
||||
|
||||
MATH_PROMPT_TEMPLATE = (
|
||||
"{question}\n"
|
||||
"Please reason step by step, and put your final answer within \\boxed{{}}."
|
||||
)
|
||||
# Run control
|
||||
parser.add_argument('--suite', default='full', choices=list(SUITES.keys()),
|
||||
help='Benchmark suite to run (default: %(default)s)')
|
||||
parser.add_argument('--datasets', '--benchmarks', dest='datasets', default=None,
|
||||
help='Override suite with comma-separated benchmark names, e.g. aime24,gsm8k')
|
||||
parser.add_argument('--exclude', default=None,
|
||||
help='Comma-separated benchmarks to exclude from the chosen suite')
|
||||
parser.add_argument('--limit', default=None,
|
||||
help='Max samples per benchmark; "none"/"all" for no limit (default: none)')
|
||||
parser.add_argument('--seed', type=int, default=DEFAULT_SEED,
|
||||
help='Random seed (default: %(default)s)')
|
||||
parser.add_argument('--batch-size', type=int, default=DEFAULT_BATCH_SIZE,
|
||||
help='Evaluation batch size (default: %(default)s)')
|
||||
|
||||
with open(CONFIG_PATH, 'r', encoding='utf-8') as f:
|
||||
DATASET_CONFIGS = yaml.safe_load(f)
|
||||
# Decoding / thinking
|
||||
parser.add_argument('--thinking', action='store_true', default=None,
|
||||
help='Enable thinking mode (sglang chat_template_kwargs.thinking=True)')
|
||||
parser.add_argument('--no-thinking', dest='thinking', action='store_false',
|
||||
help='Disable thinking mode (default)')
|
||||
|
||||
# Judge model
|
||||
parser.add_argument('--judge-model', default=DEFAULT_JUDGE_MODEL,
|
||||
help='Judge model name (default: %(default)s)')
|
||||
parser.add_argument('--judge-api-url', default=DEFAULT_JUDGE_API_URL,
|
||||
help='Judge model API URL (default: %(default)s)')
|
||||
parser.add_argument('--judge-api-key', default=DEFAULT_JUDGE_API_KEY,
|
||||
help='Judge model API key')
|
||||
parser.add_argument('--judge-max-tokens', type=int, default=DEFAULT_JUDGE_MAX_TOKENS,
|
||||
help='Judge model max_tokens (default: %(default)s)')
|
||||
|
||||
# Truncation
|
||||
parser.add_argument('--truncation-tokens', type=int, default=DEFAULT_TRUNCATION_TOKENS,
|
||||
help='Middle-truncation token budget for long-context benchmarks (default: %(default)s)')
|
||||
|
||||
return parser
|
||||
|
||||
|
||||
# ============================================================
|
||||
@ -235,23 +241,21 @@ with open(CONFIG_PATH, 'r', encoding='utf-8') as f:
|
||||
_TOKENIZER = None
|
||||
|
||||
|
||||
def get_tokenizer():
|
||||
def get_tokenizer(tokenizer_path: str):
|
||||
global _TOKENIZER
|
||||
if _TOKENIZER is None:
|
||||
from transformers import AutoTokenizer
|
||||
try:
|
||||
# Try local path first
|
||||
_TOKENIZER = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True)
|
||||
_TOKENIZER = AutoTokenizer.from_pretrained(tokenizer_path, trust_remote_code=True)
|
||||
except Exception:
|
||||
# Fallback to HuggingFace model ID
|
||||
_TOKENIZER = AutoTokenizer.from_pretrained('deepseek-ai/DeepSeek-V4-Flash', trust_remote_code=True)
|
||||
return _TOKENIZER
|
||||
|
||||
|
||||
def truncate_middle(text: str, max_tokens: int) -> str:
|
||||
def truncate_middle(text: str, max_tokens: int, tokenizer_path: str) -> str:
|
||||
if max_tokens <= 0:
|
||||
return text
|
||||
tokenizer = get_tokenizer()
|
||||
tokenizer = get_tokenizer(tokenizer_path)
|
||||
token_ids = tokenizer.encode(text, add_special_tokens=False)
|
||||
if len(token_ids) <= max_tokens:
|
||||
return text
|
||||
@ -261,41 +265,35 @@ def truncate_middle(text: str, max_tokens: int) -> str:
|
||||
return tokenizer.decode(truncated_ids, skip_special_tokens=True)
|
||||
|
||||
|
||||
def _patch_adapters_for_truncation():
|
||||
def _patch_adapters_for_truncation(tokenizer_path: str, truncation_tokens: int):
|
||||
from evalscope.benchmarks.longbench_v2.longbench_v2_adapter import LongBenchV2Adapter
|
||||
from evalscope.benchmarks.openai_mrcr.openai_mrcr_adapter import OpenAIMRCRAdapter
|
||||
|
||||
# Patch LongBenchV2Adapter.format_prompt_template
|
||||
_orig_longbench_format = LongBenchV2Adapter.format_prompt_template
|
||||
|
||||
def _patched_longbench_format(self, sample):
|
||||
max_tok = TRUNCATION_CONFIG.get('longbench_v2')
|
||||
if max_tok and sample.metadata and 'context' in sample.metadata:
|
||||
sample.metadata['context'] = truncate_middle(sample.metadata['context'], max_tok)
|
||||
if sample.metadata and 'context' in sample.metadata:
|
||||
sample.metadata['context'] = truncate_middle(sample.metadata['context'], truncation_tokens, tokenizer_path)
|
||||
return _orig_longbench_format(self, sample)
|
||||
|
||||
LongBenchV2Adapter.format_prompt_template = _patched_longbench_format
|
||||
|
||||
# Patch OpenAIMRCRAdapter.record_to_sample with needle-aware truncation.
|
||||
# MRCR is a long chat history with needles hidden at desired_msg_index.
|
||||
# We keep the head, tail, and a window around the needle, and truncate
|
||||
# each kept message if it is still too long. This preserves the retrieval
|
||||
# task while fitting GPU memory.
|
||||
_orig_mrcr_record = OpenAIMRCRAdapter.record_to_sample
|
||||
|
||||
def _patched_mrcr_record(self, record):
|
||||
import json
|
||||
max_total_tok = TRUNCATION_CONFIG.get('openai_mrcr')
|
||||
per_msg_max_tok = 8192
|
||||
if max_total_tok and 'prompt' in record:
|
||||
if 'prompt' in record:
|
||||
try:
|
||||
prompt_data = json.loads(record['prompt'])
|
||||
if not isinstance(prompt_data, list) or len(prompt_data) == 0:
|
||||
return _orig_mrcr_record(self, record)
|
||||
|
||||
tokenizer = get_tokenizer()
|
||||
tokenizer = get_tokenizer(tokenizer_path)
|
||||
total_tok = sum(
|
||||
len(tokenizer.encode(msg.get('content', '') if isinstance(msg, dict) else '', add_special_tokens=False))
|
||||
for msg in prompt_data
|
||||
)
|
||||
if total_tok <= max_total_tok:
|
||||
if total_tok <= truncation_tokens:
|
||||
return _orig_mrcr_record(self, record)
|
||||
|
||||
desired_idx = record.get('desired_msg_index', 0)
|
||||
@ -304,10 +302,8 @@ def _patch_adapters_for_truncation():
|
||||
|
||||
n = len(prompt_data)
|
||||
keep = set()
|
||||
# Head and tail context
|
||||
keep.update(range(min(2, n)))
|
||||
keep.update(range(max(0, n - 2), n))
|
||||
# Window around the needle
|
||||
window = 2
|
||||
keep.update(range(max(0, desired_idx - window), min(n, desired_idx + window + 1)))
|
||||
keep = sorted(keep)
|
||||
@ -319,7 +315,7 @@ def _patch_adapters_for_truncation():
|
||||
msg = dict(msg)
|
||||
content = msg.get('content', '')
|
||||
if len(tokenizer.encode(content, add_special_tokens=False)) > per_msg_max_tok:
|
||||
msg['content'] = truncate_middle(content, per_msg_max_tok)
|
||||
msg['content'] = truncate_middle(content, per_msg_max_tok, tokenizer_path)
|
||||
new_prompt.append(msg)
|
||||
|
||||
record = dict(record)
|
||||
@ -327,16 +323,21 @@ def _patch_adapters_for_truncation():
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
return _orig_mrcr_record(self, record)
|
||||
|
||||
OpenAIMRCRAdapter.record_to_sample = _patched_mrcr_record
|
||||
|
||||
|
||||
_patch_adapters_for_truncation()
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Helpers
|
||||
# ============================================================
|
||||
|
||||
def load_dataset_configs(config_path: str):
|
||||
if not Path(config_path).exists():
|
||||
raise FileNotFoundError(f'Config file not found: {config_path}')
|
||||
with open(config_path, 'r', encoding='utf-8') as f:
|
||||
return yaml.safe_load(f)
|
||||
|
||||
|
||||
def configure_thinking(generation_config: dict, enable: bool) -> dict:
|
||||
extra_body = generation_config.get('extra_body', {})
|
||||
chat_template_kwargs = extra_body.get('chat_template_kwargs', {})
|
||||
@ -363,14 +364,24 @@ def build_agent_config(agent_cfg: dict) -> NativeAgentConfig:
|
||||
return NativeAgentConfig(**agent_cfg)
|
||||
|
||||
|
||||
def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_thinking: bool, seed: int, run_idx: int = 0) -> TaskConfig:
|
||||
# Each run gets its own work_dir so use_cache does not reuse predictions
|
||||
# across repeated samples. This is required for temperature=1.0 multi-run
|
||||
# benchmarks to actually measure variance.
|
||||
def build_task_config(
|
||||
dataset_name: str,
|
||||
ds_cfg: dict,
|
||||
batch_size: int,
|
||||
enable_thinking: bool,
|
||||
seed: int,
|
||||
limit,
|
||||
output_dir: str,
|
||||
model: str,
|
||||
api_url: str,
|
||||
dataset_dir: str,
|
||||
judge_model_args: dict,
|
||||
run_idx: int = 0,
|
||||
) -> TaskConfig:
|
||||
if run_idx > 0:
|
||||
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}_run_{run_idx}'
|
||||
work_dir = Path(output_dir) / dataset_name / f'seed_{seed}_run_{run_idx}'
|
||||
else:
|
||||
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}'
|
||||
work_dir = Path(output_dir) / dataset_name / f'seed_{seed}'
|
||||
work_dir.mkdir(parents=True, exist_ok=True)
|
||||
work_dir = str(work_dir)
|
||||
|
||||
@ -388,13 +399,13 @@ def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_t
|
||||
agent_config = build_agent_config(ds_cfg['agent_config'])
|
||||
|
||||
return TaskConfig(
|
||||
model=MODEL,
|
||||
api_url=API_URL,
|
||||
model=model,
|
||||
api_url=api_url,
|
||||
eval_type='openai_api',
|
||||
dataset_dir=DATASET_DIR,
|
||||
judge_model_args=JUDGE_MODEL_ARGS,
|
||||
dataset_dir=dataset_dir,
|
||||
judge_model_args=judge_model_args,
|
||||
seed=seed,
|
||||
limit=LIMIT,
|
||||
limit=limit,
|
||||
collect_perf=True,
|
||||
no_timestamp=True,
|
||||
work_dir=work_dir,
|
||||
@ -419,71 +430,135 @@ def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_t
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Main loop
|
||||
# Main
|
||||
# ============================================================
|
||||
|
||||
def main():
|
||||
print(f"Config: {CONFIG_PATH}")
|
||||
print(f"Model: {MODEL}")
|
||||
print(f"API URL: {API_URL}")
|
||||
print(f"Dataset Dir: {DATASET_DIR}")
|
||||
print(f"Output Dir: {OUTPUT_DIR}")
|
||||
print(f"Limit: {LIMIT if LIMIT is not None else 'ALL'}")
|
||||
print(f"Total benchmarks: {len(DATASETS)}")
|
||||
print("="*60)
|
||||
parser = build_parser()
|
||||
args = parser.parse_args()
|
||||
|
||||
for batch_size in BATCH_SIZE_LIST:
|
||||
# 1. Run multi-run benchmarks (temperature=1.0, same seed, variance from sampling)
|
||||
for dataset_name in _multi_run_order:
|
||||
if dataset_name not in DATASET_CONFIGS:
|
||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
||||
# Resolve limit
|
||||
limit = args.limit
|
||||
if limit is not None:
|
||||
if str(limit).lower() in ('none', 'all'):
|
||||
limit = None
|
||||
else:
|
||||
limit = int(limit)
|
||||
|
||||
enable_thinking = DEFAULT_ENABLE_THINKING if args.thinking is None else args.thinking
|
||||
|
||||
# Resolve suite or custom datasets
|
||||
if args.datasets:
|
||||
custom = [d.strip() for d in args.datasets.split(',') if d.strip()]
|
||||
multi_run = [d for d in custom if d in MULTI_RUN_CONFIG]
|
||||
single_run = [d for d in custom if d not in MULTI_RUN_CONFIG]
|
||||
agent = [d for d in custom if d in ALL_AGENT]
|
||||
single_run = [d for d in single_run if d not in ALL_AGENT]
|
||||
else:
|
||||
suite = SUITES[args.suite]
|
||||
multi_run = list(suite['multi'])
|
||||
single_run = list(suite['single'])
|
||||
agent = list(suite['agent'])
|
||||
|
||||
# Apply --exclude
|
||||
if args.exclude:
|
||||
exclude = {d.strip() for d in args.exclude.split(',') if d.strip()}
|
||||
multi_run = [d for d in multi_run if d not in exclude]
|
||||
single_run = [d for d in single_run if d not in exclude]
|
||||
agent = [d for d in agent if d not in exclude]
|
||||
|
||||
judge_model_args = {
|
||||
'model_id': args.judge_model,
|
||||
'api_url': args.judge_api_url,
|
||||
'api_key': args.judge_api_key,
|
||||
'eval_type': 'openai_api',
|
||||
'generation_config': {
|
||||
'temperature': 0.0,
|
||||
'max_tokens': args.judge_max_tokens,
|
||||
},
|
||||
}
|
||||
|
||||
truncation_tokens = args.truncation_tokens
|
||||
_patch_adapters_for_truncation(args.tokenizer_path, truncation_tokens)
|
||||
|
||||
dataset_configs = load_dataset_configs(args.config)
|
||||
|
||||
print('=' * 60)
|
||||
print(f'Config: {args.config}')
|
||||
print(f'Model: {args.model}')
|
||||
print(f'API URL: {args.api_url}')
|
||||
print(f'Dataset Dir: {args.dataset_dir}')
|
||||
print(f'Output Dir: {args.output_dir}')
|
||||
print(f'Suite: {args.suite}')
|
||||
print(f'Limit: {limit if limit is not None else "ALL"}')
|
||||
print(f'Thinking: {enable_thinking}')
|
||||
print(f'Seed: {args.seed}')
|
||||
print(f'Batch Size: {args.batch_size}')
|
||||
print(f'Tokenizer Path: {args.tokenizer_path}')
|
||||
print(f'Truncation Tokens: {truncation_tokens}')
|
||||
print(f'Multi-run datasets: {multi_run}')
|
||||
print(f'Single-run datasets: {single_run}')
|
||||
print(f'Agent datasets: {agent}')
|
||||
print('=' * 60)
|
||||
|
||||
for dataset_name in multi_run:
|
||||
if dataset_name not in dataset_configs:
|
||||
print(f'WARNING: {dataset_name} not in YAML config, skipping')
|
||||
continue
|
||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
||||
ds_cfg = dataset_configs[dataset_name]
|
||||
num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1)
|
||||
for run_idx in range(num_runs):
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={SEED})")
|
||||
print(f'Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={args.seed})')
|
||||
print(f"{'='*60}")
|
||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED, run_idx=run_idx)
|
||||
task_cfg = build_task_config(
|
||||
dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit,
|
||||
args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args,
|
||||
run_idx=run_idx,
|
||||
)
|
||||
try:
|
||||
run_task(task_cfg)
|
||||
except Exception as e:
|
||||
print(f"ERROR in {dataset_name} (run {run_idx + 1}): {e}")
|
||||
print(f'ERROR in {dataset_name} (run {run_idx + 1}): {e}')
|
||||
continue
|
||||
|
||||
# 2. Run single-run benchmarks (temperature=0.0 greedy)
|
||||
for dataset_name in _single_run_order:
|
||||
if dataset_name not in DATASET_CONFIGS:
|
||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
||||
for dataset_name in single_run:
|
||||
if dataset_name not in dataset_configs:
|
||||
print(f'WARNING: {dataset_name} not in YAML config, skipping')
|
||||
continue
|
||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
||||
ds_cfg = dataset_configs[dataset_name]
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Running: {dataset_name} (seed={SEED})")
|
||||
print(f'Running: {dataset_name} (seed={args.seed})')
|
||||
print(f"{'='*60}")
|
||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED)
|
||||
task_cfg = build_task_config(
|
||||
dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit,
|
||||
args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args,
|
||||
)
|
||||
try:
|
||||
run_task(task_cfg)
|
||||
except Exception as e:
|
||||
print(f"ERROR in {dataset_name}: {e}")
|
||||
print(f'ERROR in {dataset_name}: {e}')
|
||||
continue
|
||||
|
||||
# 3. Run agent benchmarks
|
||||
for dataset_name in _agent_order:
|
||||
if dataset_name not in DATASET_CONFIGS:
|
||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
||||
for dataset_name in agent:
|
||||
if dataset_name not in dataset_configs:
|
||||
print(f'WARNING: {dataset_name} not in YAML config, skipping')
|
||||
continue
|
||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
||||
ds_cfg = dataset_configs[dataset_name]
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Running: {dataset_name} (seed={SEED})")
|
||||
print(f'Running: {dataset_name} (seed={args.seed})')
|
||||
print(f"{'='*60}")
|
||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED)
|
||||
task_cfg = build_task_config(
|
||||
dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit,
|
||||
args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args,
|
||||
)
|
||||
try:
|
||||
run_task(task_cfg)
|
||||
except Exception as e:
|
||||
print(f"ERROR in {dataset_name}: {e}")
|
||||
print(f'ERROR in {dataset_name}: {e}')
|
||||
continue
|
||||
|
||||
print("\nAll benchmarks done!")
|
||||
print('\nAll benchmarks done!')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
@ -1,74 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Full evaluation suite: ~3-5 days
|
||||
Runs all benchmarks with full datasets and multiple runs for stability.
|
||||
Use --limit none (default) for the complete evaluation.
|
||||
"""
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
import run as run_module
|
||||
|
||||
# Use the default full configuration from run.py
|
||||
# (already ordered by estimated runtime)
|
||||
run_module._multi_run_order = [
|
||||
'humaneval',
|
||||
'live_code_bench',
|
||||
'aime26',
|
||||
'aime24',
|
||||
'aime25',
|
||||
'gpqa_diamond',
|
||||
'imo_answerbench',
|
||||
'hmmt26',
|
||||
]
|
||||
|
||||
run_module._single_run_order = [
|
||||
'arc',
|
||||
'bfcl_v3',
|
||||
'winogrande',
|
||||
'competition_math',
|
||||
'gsm8k',
|
||||
'hellaswag',
|
||||
'bigcodebench',
|
||||
'drop',
|
||||
'bbh',
|
||||
'openai_mrcr',
|
||||
'longbench_v2',
|
||||
'mmlu',
|
||||
'cmmlu',
|
||||
'super_gpqa',
|
||||
'simple_qa',
|
||||
'mmlu_pro',
|
||||
'hle',
|
||||
'trivia_qa',
|
||||
]
|
||||
|
||||
run_module._agent_order = [
|
||||
'tau2_bench',
|
||||
'general_fc',
|
||||
]
|
||||
|
||||
run_module.DATASETS = (
|
||||
run_module._multi_run_order
|
||||
+ run_module._single_run_order
|
||||
+ run_module._agent_order
|
||||
)
|
||||
|
||||
# Full: many runs for statistical stability
|
||||
run_module.MULTI_RUN_CONFIG = {
|
||||
'aime24': 12,
|
||||
'aime25': 12,
|
||||
'aime26': 12,
|
||||
'hmmt26': 12,
|
||||
'live_code_bench': 5,
|
||||
'imo_answerbench': 4,
|
||||
'humaneval': 3,
|
||||
'gpqa_diamond': 2,
|
||||
}
|
||||
|
||||
if __name__ == '__main__':
|
||||
print('Full benchmarks:', run_module.DATASETS)
|
||||
print('Estimated time: ~3-5 days with --limit none')
|
||||
run_module.main()
|
||||
@ -1,48 +0,0 @@
|
||||
"""
|
||||
Group 1: Code + Reasoning + short Knowledge + Agent (est. ~18-22h)
|
||||
Run on machine 1.
|
||||
"""
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
# Ensure we can import run.py from the same directory inside or outside Docker
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
import run as run_module
|
||||
|
||||
# Override dataset lists for group 1
|
||||
run_module._multi_run_order = [
|
||||
'humaneval',
|
||||
'live_code_bench',
|
||||
'aime26',
|
||||
'aime24',
|
||||
'aime25',
|
||||
'gpqa_diamond',
|
||||
'imo_answerbench',
|
||||
'hmmt26',
|
||||
]
|
||||
|
||||
run_module._single_run_order = [
|
||||
'arc',
|
||||
'winogrande',
|
||||
'competition_math',
|
||||
'gsm8k',
|
||||
'hellaswag',
|
||||
'bigcodebench',
|
||||
]
|
||||
|
||||
run_module._agent_order = [
|
||||
'bfcl_v3',
|
||||
'tau2_bench',
|
||||
]
|
||||
|
||||
run_module.DATASETS = (
|
||||
run_module._multi_run_order
|
||||
+ run_module._single_run_order
|
||||
+ run_module._agent_order
|
||||
)
|
||||
|
||||
if __name__ == '__main__':
|
||||
print('Group 1 benchmarks:', run_module.DATASETS)
|
||||
print('Estimated time: ~18-22h')
|
||||
run_module.main()
|
||||
@ -1,40 +0,0 @@
|
||||
"""
|
||||
Group 2: Knowledge + Long-context + Agent (est. ~20-24h)
|
||||
Run on machine 2.
|
||||
"""
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
# Ensure we can import run.py from the same directory inside or outside Docker
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
import run as run_module
|
||||
|
||||
# Override dataset lists for group 2
|
||||
run_module._multi_run_order = []
|
||||
|
||||
run_module._single_run_order = [
|
||||
'drop',
|
||||
'bbh',
|
||||
'openai_mrcr',
|
||||
'longbench_v2',
|
||||
'mmlu',
|
||||
'cmmlu',
|
||||
'super_gpqa',
|
||||
'simple_qa',
|
||||
]
|
||||
|
||||
run_module._agent_order = [
|
||||
'general_fc',
|
||||
]
|
||||
|
||||
run_module.DATASETS = (
|
||||
run_module._multi_run_order
|
||||
+ run_module._single_run_order
|
||||
+ run_module._agent_order
|
||||
)
|
||||
|
||||
if __name__ == '__main__':
|
||||
print('Group 2 benchmarks:', run_module.DATASETS)
|
||||
print('Estimated time: ~20-24h')
|
||||
run_module.main()
|
||||
@ -1,35 +0,0 @@
|
||||
"""
|
||||
Group 3: Long-running Knowledge benchmarks (est. ~35-40h)
|
||||
Run on machine 3.
|
||||
Note: trivia_qa and hle are inherently slow; consider using --limit to control time.
|
||||
"""
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
# Ensure we can import run.py from the same directory inside or outside Docker
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
import run as run_module
|
||||
|
||||
# Override dataset lists for group 3
|
||||
run_module._multi_run_order = []
|
||||
|
||||
run_module._single_run_order = [
|
||||
'mmlu_pro',
|
||||
'hle',
|
||||
'trivia_qa',
|
||||
]
|
||||
|
||||
run_module._agent_order = []
|
||||
|
||||
run_module.DATASETS = (
|
||||
run_module._multi_run_order
|
||||
+ run_module._single_run_order
|
||||
+ run_module._agent_order
|
||||
)
|
||||
|
||||
if __name__ == '__main__':
|
||||
print('Group 3 benchmarks:', run_module.DATASETS)
|
||||
print('Estimated time: ~35-40h (trivia_qa and hle are slow)')
|
||||
print('Tip: use --limit 500 to cap runtime if needed')
|
||||
run_module.main()
|
||||
@ -1,40 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Lite evaluation suite: ~2-4h
|
||||
Covers all 5 capability domains with 1-2 benchmarks each.
|
||||
Use --limit 20 (default) for quick smoke testing.
|
||||
"""
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
import run as run_module
|
||||
|
||||
run_module._multi_run_order = [
|
||||
'aime24', # reasoning
|
||||
'humaneval', # code
|
||||
]
|
||||
|
||||
run_module._single_run_order = [
|
||||
'gsm8k', # reasoning
|
||||
'mmlu_pro', # knowledge
|
||||
'simple_qa', # knowledge
|
||||
'longbench_v2', # long-context
|
||||
]
|
||||
|
||||
run_module._agent_order = [
|
||||
'bfcl_v3', # tool/function calling
|
||||
]
|
||||
|
||||
run_module.DATASETS = (
|
||||
run_module._multi_run_order
|
||||
+ run_module._single_run_order
|
||||
+ run_module._agent_order
|
||||
)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
print('Lite benchmarks:', run_module.DATASETS)
|
||||
run_module.main()
|
||||
|
||||
@ -1,45 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run a single benchmark for quick smoke tests.
|
||||
|
||||
Usage:
|
||||
python bash/run_single.py bigcodebench --limit 1
|
||||
python bash/run_single.py bfcl_v3 --limit 5 --model MODEL --api-url URL
|
||||
"""
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
import run as run_mod
|
||||
from evalscope import run_task
|
||||
|
||||
if len(sys.argv) < 2:
|
||||
print('Usage: python bash/run_single.py <dataset_name> [--limit N] [--model M] [--api-url U]')
|
||||
sys.exit(1)
|
||||
|
||||
dataset_name = sys.argv[1]
|
||||
limit = None
|
||||
for i, arg in enumerate(sys.argv):
|
||||
if arg == '--limit' and i + 1 < len(sys.argv):
|
||||
v = sys.argv[i + 1]
|
||||
limit = None if v.lower() in ('none', 'all') else int(v)
|
||||
elif arg == '--model' and i + 1 < len(sys.argv):
|
||||
run_mod.MODEL = sys.argv[i + 1]
|
||||
elif arg == '--api-url' and i + 1 < len(sys.argv):
|
||||
run_mod.API_URL = sys.argv[i + 1]
|
||||
|
||||
if dataset_name not in run_mod.DATASET_CONFIGS:
|
||||
raise SystemExit(f'{dataset_name} not in config')
|
||||
|
||||
run_mod.LIMIT = limit
|
||||
ds_cfg = run_mod.DATASET_CONFIGS[dataset_name]
|
||||
task_cfg = run_mod.build_task_config(
|
||||
dataset_name=dataset_name,
|
||||
ds_cfg=ds_cfg,
|
||||
batch_size=4,
|
||||
enable_thinking=run_mod.ENABLE_THINKING,
|
||||
seed=run_mod.SEED,
|
||||
run_idx=0,
|
||||
)
|
||||
print(f'Running {dataset_name} with limit={limit}')
|
||||
run_task(task_cfg)
|
||||
print('Done.')
|
||||
497
bash/test.py
497
bash/test.py
@ -1,497 +0,0 @@
|
||||
import yaml
|
||||
from copy import deepcopy
|
||||
from pathlib import Path
|
||||
import sys
|
||||
import os
|
||||
|
||||
from evalscope import run_task, TaskConfig
|
||||
from evalscope.api.agent import NativeAgentConfig
|
||||
from evalscope.config import SandboxTaskConfig
|
||||
|
||||
SCRIPT_DIR = Path(__file__).parent.resolve()
|
||||
PROJECT_ROOT = SCRIPT_DIR.parent
|
||||
|
||||
|
||||
# ============================================================
|
||||
# ★★★ 必改参数 (USER CONFIG) ★★★
|
||||
# 每次评测新模型前,只需要检查/修改以下参数。
|
||||
# 也可以通过命令行覆盖:--model / --api-url / --dataset-dir /
|
||||
# --output-dir / --limit / --config
|
||||
# ============================================================
|
||||
|
||||
# 模型名(served model name)
|
||||
MODEL = 'DeepSeek-V4-Flash-Int8'
|
||||
|
||||
# 模型服务地址(OpenAI 兼容 API)
|
||||
API_URL = 'http://localhost:30000/v1'
|
||||
|
||||
# 本地数据集根目录(提前下载好的 datasets 目录)
|
||||
DATASET_DIR = str(PROJECT_ROOT / 'datasets')
|
||||
|
||||
# 评测输出目录(每个 benchmark 单独子目录)
|
||||
OUTPUT_DIR = str(PROJECT_ROOT / 'output')
|
||||
|
||||
# 采样数量上限;None 表示跑全部样本
|
||||
LIMIT = None
|
||||
|
||||
# 每个 benchmark 的生成参数配置文件(max_tokens / temperature 等)
|
||||
CONFIG = None
|
||||
|
||||
# 是否开启 thinking 模式(sglang chat_template_kwargs.thinking)
|
||||
ENABLE_THINKING = False
|
||||
|
||||
# 测试哪些 benchmark:见下方 DATASETS = _multi_run_order + _single_run_order + _agent_order
|
||||
# 想只跑部分 benchmark 时,注释掉对应行即可。
|
||||
|
||||
# ============================================================
|
||||
# Command line argument overrides (不需要修改)
|
||||
# ============================================================
|
||||
|
||||
for i, arg in enumerate(sys.argv):
|
||||
if arg == '--limit' and i + 1 < len(sys.argv):
|
||||
limit_val = sys.argv[i + 1]
|
||||
if limit_val.lower() == 'none' or limit_val.lower() == 'all':
|
||||
LIMIT = None
|
||||
else:
|
||||
LIMIT = int(limit_val)
|
||||
elif arg == '--model' and i + 1 < len(sys.argv):
|
||||
MODEL = sys.argv[i + 1]
|
||||
elif arg == '--api-url' and i + 1 < len(sys.argv):
|
||||
API_URL = sys.argv[i + 1]
|
||||
elif arg == '--dataset-dir' and i + 1 < len(sys.argv):
|
||||
DATASET_DIR = sys.argv[i + 1]
|
||||
elif arg == '--output-dir' and i + 1 < len(sys.argv):
|
||||
OUTPUT_DIR = sys.argv[i + 1]
|
||||
elif arg == '--config' and i + 1 < len(sys.argv):
|
||||
CONFIG = sys.argv[i + 1]
|
||||
|
||||
# Benchmarks that require sandboxed code execution
|
||||
SANDBOX_DATASETS = {'humaneval', 'bigcodebench'}
|
||||
# Per-dataset sandbox configs.
|
||||
# bigcodebench's upstream image has ENTRYPOINT ["python3", "-m", "bigcodebench.evaluate"],
|
||||
# which exits immediately and breaks ms_enclave exec. We built a derivative image
|
||||
# `bigcodebench-sandbox:latest` that keeps the same Python environment but drops the
|
||||
# entrypoint and runs `tail -f /dev/null` so the container stays alive.
|
||||
SANDBOX_CONFIGS = {
|
||||
'bigcodebench': {
|
||||
'image': 'bigcodebench-sandbox:latest',
|
||||
'working_dir': '/tmp',
|
||||
'tools_config': {
|
||||
'shell_executor': {},
|
||||
'python_executor': {}
|
||||
}
|
||||
},
|
||||
'humaneval': {
|
||||
'image': 'python:3.11-slim',
|
||||
'tools_config': {
|
||||
'shell_executor': {},
|
||||
'python_executor': {}
|
||||
}
|
||||
},
|
||||
}
|
||||
|
||||
# ============================================================
|
||||
# User-tunable parameters
|
||||
# ============================================================
|
||||
|
||||
# Fixed seed for reproducibility. With temperature > 0 the model still samples
|
||||
# randomly, so running N times with the same seed still yields variance.
|
||||
SEED = 42
|
||||
|
||||
# Benchmarks to run multiple times with temperature=1.0.
|
||||
# Format: {benchmark_name: num_runs}
|
||||
# Number of runs chosen so total samples ≈ 400-500 per benchmark.
|
||||
MULTI_RUN_CONFIG = {
|
||||
# ~30 samples each -> 12 runs = ~360 samples
|
||||
'aime24': 12,
|
||||
'aime25': 12,
|
||||
'aime26': 12,
|
||||
'hmmt26': 12,
|
||||
# ~100 samples -> 5 runs = ~500 samples
|
||||
'live_code_bench': 5,
|
||||
# ~120 samples -> 4 runs = ~480 samples
|
||||
'imo_answerbench': 4,
|
||||
# ~164 samples -> 3 runs = ~492 samples
|
||||
'humaneval': 3,
|
||||
# ~198 samples -> 2 runs = ~396 samples
|
||||
'gpqa_diamond': 2,
|
||||
}
|
||||
|
||||
# Single-run benchmarks (temperature=0.0 greedy)
|
||||
SINGLE_RUN_DATASETS = [
|
||||
'bigcodebench',
|
||||
'bfcl_v3', # function calling: greedy decoding
|
||||
'competition_math',
|
||||
'gsm8k',
|
||||
'hle',
|
||||
'super_gpqa',
|
||||
'arc',
|
||||
'bbh',
|
||||
'cmmlu',
|
||||
'drop',
|
||||
'hellaswag',
|
||||
'mmlu',
|
||||
'mmlu_pro',
|
||||
'simple_qa',
|
||||
'trivia_qa',
|
||||
'winogrande',
|
||||
'openai_mrcr', # long-context, run once
|
||||
'longbench_v2', # long-context, run once
|
||||
]
|
||||
|
||||
# Agent / tool benchmarks (temperature=0.0 greedy, run once)
|
||||
AGENT_DATASETS = [
|
||||
'tau2_bench',
|
||||
'general_fc',
|
||||
]
|
||||
|
||||
# Combine all datasets in order (shortest estimated time first)
|
||||
# 1. Multi-run small benchmarks (temperature=1.0 for sampling diversity)
|
||||
_multi_run_order = [
|
||||
# 'humaneval', # ~164 samples x 3 runs = 492
|
||||
# 'live_code_bench', # ~100 samples x 5 runs = 500
|
||||
# 'aime26', # ~30 samples x 12 runs = 360
|
||||
# 'aime24', # ~30 samples x 12 runs = 360
|
||||
# 'aime25', # ~30 samples x 12 runs = 360
|
||||
# 'gpqa_diamond', # ~198 samples x 2 runs = 396
|
||||
# 'imo_answerbench', # ~120 samples x 4 runs = 480
|
||||
# 'hmmt26', # ~30 samples x 12 runs = 360
|
||||
]
|
||||
|
||||
# 2. Single-run benchmarks (temperature=0.0 greedy)
|
||||
_single_run_order = [
|
||||
# 'arc',
|
||||
# 'bfcl_v3', # function calling: greedy decoding
|
||||
# 'winogrande',
|
||||
# 'competition_math',
|
||||
# 'gsm8k',
|
||||
# 'hellaswag',
|
||||
'bigcodebench',
|
||||
# 'drop',
|
||||
# 'bbh',
|
||||
# 'openai_mrcr',
|
||||
# 'longbench_v2',
|
||||
# 'mmlu',
|
||||
# 'cmmlu',
|
||||
# 'super_gpqa',
|
||||
# 'simple_qa',
|
||||
# 'mmlu_pro',
|
||||
'hle',
|
||||
# 'trivia_qa',
|
||||
]
|
||||
|
||||
# 3. Agent / tool benchmarks (last)
|
||||
_agent_order = [
|
||||
# 'tau2_bench',
|
||||
# 'general_fc',
|
||||
]
|
||||
|
||||
DATASETS = _multi_run_order + _single_run_order + _agent_order
|
||||
|
||||
# ENABLE_THINKING 已移至文件头部「必改参数」区
|
||||
BATCH_SIZE_LIST = [4]
|
||||
# LIMIT is parsed from command line: --limit N
|
||||
SHUFFLE = True
|
||||
|
||||
# ============================================================
|
||||
# Fixed configuration
|
||||
# ============================================================
|
||||
|
||||
CONFIG_PATH = str(Path(CONFIG) if CONFIG else SCRIPT_DIR / 'config' / 'dpv4-int8_nothinking.yaml')
|
||||
if not Path(CONFIG_PATH).exists():
|
||||
raise FileNotFoundError(f'Config file not found: {CONFIG_PATH}')
|
||||
|
||||
MODEL_PATH = '/data1/models/DeepSeek-V4-Flash-INT8'
|
||||
|
||||
# Judge model for benchmarks that require LLM-as-judge (simple_qa, tau2_bench, etc.)
|
||||
# Vectron endpoint with DeepSeek-V4-Pro
|
||||
JUDGE_MODEL_ARGS = {
|
||||
'model_id': 'DeepSeek/DeepSeek-V4-Pro',
|
||||
'api_url': 'https://api.vectron.meta-stone.com/v1',
|
||||
'api_key': 'sk-dbd8a665f7634081b87ec409c7636500',
|
||||
'eval_type': 'openai_api',
|
||||
'generation_config': {
|
||||
'temperature': 0.0,
|
||||
'max_tokens': 10240,
|
||||
},
|
||||
}
|
||||
|
||||
TRUNCATION_CONFIG = {
|
||||
'longbench_v2': 32768*4,
|
||||
'openai_mrcr': 32768*4,
|
||||
}
|
||||
|
||||
MATH_DATASETS = {
|
||||
'aime24', 'aime25', 'aime26', 'hmmt26',
|
||||
'gsm8k', 'competition_math', 'imo_answerbench',
|
||||
}
|
||||
|
||||
MATH_PROMPT_TEMPLATE = (
|
||||
"{question}\n"
|
||||
"Please reason step by step, and put your final answer within \\boxed{{}}."
|
||||
)
|
||||
|
||||
with open(CONFIG_PATH, 'r', encoding='utf-8') as f:
|
||||
DATASET_CONFIGS = yaml.safe_load(f)
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Middle-truncation helpers
|
||||
# ============================================================
|
||||
|
||||
_TOKENIZER = None
|
||||
|
||||
|
||||
def get_tokenizer():
|
||||
global _TOKENIZER
|
||||
if _TOKENIZER is None:
|
||||
from transformers import AutoTokenizer
|
||||
try:
|
||||
# Try local path first
|
||||
_TOKENIZER = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True)
|
||||
except Exception:
|
||||
# Fallback to HuggingFace model ID
|
||||
_TOKENIZER = AutoTokenizer.from_pretrained('deepseek-ai/DeepSeek-V4-Flash', trust_remote_code=True)
|
||||
return _TOKENIZER
|
||||
|
||||
|
||||
def truncate_middle(text: str, max_tokens: int) -> str:
|
||||
if max_tokens <= 0:
|
||||
return text
|
||||
tokenizer = get_tokenizer()
|
||||
token_ids = tokenizer.encode(text, add_special_tokens=False)
|
||||
if len(token_ids) <= max_tokens:
|
||||
return text
|
||||
keep_head = max_tokens // 2
|
||||
keep_tail = max_tokens - keep_head
|
||||
truncated_ids = token_ids[:keep_head] + token_ids[-keep_tail:]
|
||||
return tokenizer.decode(truncated_ids, skip_special_tokens=True)
|
||||
|
||||
|
||||
def _patch_adapters_for_truncation():
|
||||
from evalscope.benchmarks.longbench_v2.longbench_v2_adapter import LongBenchV2Adapter
|
||||
from evalscope.benchmarks.openai_mrcr.openai_mrcr_adapter import OpenAIMRCRAdapter
|
||||
|
||||
# Patch LongBenchV2Adapter.format_prompt_template
|
||||
_orig_longbench_format = LongBenchV2Adapter.format_prompt_template
|
||||
def _patched_longbench_format(self, sample):
|
||||
max_tok = TRUNCATION_CONFIG.get('longbench_v2')
|
||||
if max_tok and sample.metadata and 'context' in sample.metadata:
|
||||
sample.metadata['context'] = truncate_middle(sample.metadata['context'], max_tok)
|
||||
return _orig_longbench_format(self, sample)
|
||||
LongBenchV2Adapter.format_prompt_template = _patched_longbench_format
|
||||
|
||||
# Patch OpenAIMRCRAdapter.record_to_sample with needle-aware truncation.
|
||||
# MRCR is a long chat history with needles hidden at desired_msg_index.
|
||||
# We keep the head, tail, and a window around the needle, and truncate
|
||||
# each kept message if it is still too long. This preserves the retrieval
|
||||
# task while fitting GPU memory.
|
||||
_orig_mrcr_record = OpenAIMRCRAdapter.record_to_sample
|
||||
def _patched_mrcr_record(self, record):
|
||||
import json
|
||||
max_total_tok = TRUNCATION_CONFIG.get('openai_mrcr')
|
||||
per_msg_max_tok = 8192
|
||||
if max_total_tok and 'prompt' in record:
|
||||
try:
|
||||
prompt_data = json.loads(record['prompt'])
|
||||
if not isinstance(prompt_data, list) or len(prompt_data) == 0:
|
||||
return _orig_mrcr_record(self, record)
|
||||
|
||||
tokenizer = get_tokenizer()
|
||||
total_tok = sum(
|
||||
len(tokenizer.encode(msg.get('content', '') if isinstance(msg, dict) else '', add_special_tokens=False))
|
||||
for msg in prompt_data
|
||||
)
|
||||
if total_tok <= max_total_tok:
|
||||
return _orig_mrcr_record(self, record)
|
||||
|
||||
desired_idx = record.get('desired_msg_index', 0)
|
||||
if not isinstance(desired_idx, int) or desired_idx < 0 or desired_idx >= len(prompt_data):
|
||||
desired_idx = 0
|
||||
|
||||
n = len(prompt_data)
|
||||
keep = set()
|
||||
# Head and tail context
|
||||
keep.update(range(min(2, n)))
|
||||
keep.update(range(max(0, n - 2), n))
|
||||
# Window around the needle
|
||||
window = 2
|
||||
keep.update(range(max(0, desired_idx - window), min(n, desired_idx + window + 1)))
|
||||
keep = sorted(keep)
|
||||
|
||||
new_prompt = []
|
||||
for idx in keep:
|
||||
msg = prompt_data[idx]
|
||||
if isinstance(msg, dict):
|
||||
msg = dict(msg)
|
||||
content = msg.get('content', '')
|
||||
if len(tokenizer.encode(content, add_special_tokens=False)) > per_msg_max_tok:
|
||||
msg['content'] = truncate_middle(content, per_msg_max_tok)
|
||||
new_prompt.append(msg)
|
||||
|
||||
record = dict(record)
|
||||
record['prompt'] = json.dumps(new_prompt)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
return _orig_mrcr_record(self, record)
|
||||
OpenAIMRCRAdapter.record_to_sample = _patched_mrcr_record
|
||||
|
||||
|
||||
_patch_adapters_for_truncation()
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Helpers
|
||||
# ============================================================
|
||||
|
||||
def configure_thinking(generation_config: dict, enable: bool) -> dict:
|
||||
extra_body = generation_config.get('extra_body', {})
|
||||
chat_template_kwargs = extra_body.get('chat_template_kwargs', {})
|
||||
if enable:
|
||||
chat_template_kwargs['thinking'] = True
|
||||
else:
|
||||
chat_template_kwargs.pop('thinking', None)
|
||||
if chat_template_kwargs:
|
||||
extra_body['chat_template_kwargs'] = chat_template_kwargs
|
||||
if extra_body:
|
||||
generation_config['extra_body'] = extra_body
|
||||
return generation_config
|
||||
|
||||
|
||||
def build_agent_config(agent_cfg: dict) -> NativeAgentConfig:
|
||||
agent_cfg = deepcopy(agent_cfg or {})
|
||||
known_fields = {'mode', 'strategy', 'tools', 'max_steps', 'mcp_servers', 'environment', 'environment_extra'}
|
||||
kwargs = agent_cfg.pop('kwargs', {})
|
||||
for key in list(agent_cfg.keys()):
|
||||
if key not in known_fields:
|
||||
kwargs[key] = agent_cfg.pop(key)
|
||||
if kwargs:
|
||||
agent_cfg['kwargs'] = kwargs
|
||||
return NativeAgentConfig(**agent_cfg)
|
||||
|
||||
|
||||
def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_thinking: bool, seed: int, run_idx: int = 0) -> TaskConfig:
|
||||
# Each run gets its own work_dir so use_cache does not reuse predictions
|
||||
# across repeated samples. This is required for temperature=1.0 multi-run
|
||||
# benchmarks to actually measure variance.
|
||||
if run_idx > 0:
|
||||
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}_run_{run_idx}'
|
||||
else:
|
||||
work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}'
|
||||
work_dir.mkdir(parents=True, exist_ok=True)
|
||||
work_dir = str(work_dir)
|
||||
|
||||
generation_config = configure_thinking(deepcopy(ds_cfg['generation_config']), enable_thinking)
|
||||
dataset_args = deepcopy(ds_cfg.get('dataset_args', {}))
|
||||
dataset_args.setdefault('shuffle', True)
|
||||
|
||||
if dataset_name in MATH_DATASETS:
|
||||
dataset_args['prompt_template'] = MATH_PROMPT_TEMPLATE
|
||||
|
||||
dataset_args_dict = {dataset_name: dataset_args}
|
||||
|
||||
agent_config = None
|
||||
if 'agent_config' in ds_cfg:
|
||||
agent_config = build_agent_config(ds_cfg['agent_config'])
|
||||
|
||||
return TaskConfig(
|
||||
model=MODEL,
|
||||
api_url=API_URL,
|
||||
eval_type='openai_api',
|
||||
dataset_dir=DATASET_DIR,
|
||||
judge_model_args=JUDGE_MODEL_ARGS,
|
||||
seed=seed,
|
||||
limit=LIMIT,
|
||||
collect_perf=True,
|
||||
no_timestamp=True,
|
||||
work_dir=work_dir,
|
||||
use_cache=work_dir,
|
||||
datasets=[dataset_name],
|
||||
generation_config=generation_config,
|
||||
dataset_args=dataset_args_dict,
|
||||
agent_config=agent_config,
|
||||
eval_batch_size=batch_size,
|
||||
sandbox=SandboxTaskConfig(
|
||||
enabled=True,
|
||||
engine='docker',
|
||||
default_config=SANDBOX_CONFIGS.get(dataset_name, {
|
||||
'image': 'python:3.11-slim',
|
||||
'tools_config': {
|
||||
'shell_executor': {},
|
||||
'python_executor': {}
|
||||
}
|
||||
})
|
||||
) if dataset_name in SANDBOX_DATASETS else None,
|
||||
)
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Main loop
|
||||
# ============================================================
|
||||
|
||||
def main():
|
||||
print(f"Config: {CONFIG_PATH}")
|
||||
print(f"Model: {MODEL}")
|
||||
print(f"API URL: {API_URL}")
|
||||
print(f"Dataset Dir: {DATASET_DIR}")
|
||||
print(f"Output Dir: {OUTPUT_DIR}")
|
||||
print(f"Limit: {LIMIT if LIMIT is not None else 'ALL'}")
|
||||
print(f"Total benchmarks: {len(DATASETS)}")
|
||||
print("="*60)
|
||||
|
||||
for batch_size in BATCH_SIZE_LIST:
|
||||
# 1. Run multi-run benchmarks (temperature=1.0, same seed, variance from sampling)
|
||||
for dataset_name in _multi_run_order:
|
||||
if dataset_name not in DATASET_CONFIGS:
|
||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
||||
continue
|
||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
||||
num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1)
|
||||
for run_idx in range(num_runs):
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={SEED})")
|
||||
print(f"{'='*60}")
|
||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED, run_idx=run_idx)
|
||||
try:
|
||||
run_task(task_cfg)
|
||||
except Exception as e:
|
||||
print(f"ERROR in {dataset_name} (run {run_idx + 1}): {e}")
|
||||
continue
|
||||
|
||||
# 2. Run single-run benchmarks (temperature=0.0 greedy)
|
||||
for dataset_name in _single_run_order:
|
||||
if dataset_name not in DATASET_CONFIGS:
|
||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
||||
continue
|
||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Running: {dataset_name} (seed={SEED})")
|
||||
print(f"{'='*60}")
|
||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED)
|
||||
try:
|
||||
run_task(task_cfg)
|
||||
except Exception as e:
|
||||
print(f"ERROR in {dataset_name}: {e}")
|
||||
continue
|
||||
|
||||
# 3. Run agent benchmarks
|
||||
for dataset_name in _agent_order:
|
||||
if dataset_name not in DATASET_CONFIGS:
|
||||
print(f"WARNING: {dataset_name} not in YAML config, skipping")
|
||||
continue
|
||||
ds_cfg = DATASET_CONFIGS[dataset_name]
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Running: {dataset_name} (seed={SEED})")
|
||||
print(f"{'='*60}")
|
||||
task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED)
|
||||
try:
|
||||
run_task(task_cfg)
|
||||
except Exception as e:
|
||||
print(f"ERROR in {dataset_name}: {e}")
|
||||
continue
|
||||
|
||||
print("\nAll benchmarks done!")
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
26
scripts/run_docker_full.sh
Executable file
26
scripts/run_docker_full.sh
Executable file
@ -0,0 +1,26 @@
|
||||
#!/bin/bash
|
||||
# Docker 全量评测(Full)
|
||||
# 请根据实际机器修改 MODEL / API_URL / HOST_EVALSCOPE 路径
|
||||
|
||||
MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}"
|
||||
API_URL="${API_URL:-http://localhost:30000/v1}"
|
||||
HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}"
|
||||
OUTPUT_DIR="${OUTPUT_DIR:-/opt/evalscope/output_full}"
|
||||
|
||||
docker run -it --rm \
|
||||
--network host \
|
||||
-v "${HOST_EVALSCOPE}:/opt/evalscope" \
|
||||
-v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \
|
||||
-v "${HOST_EVALSCOPE}/output_full:/opt/evalscope/output_full" \
|
||||
-v /var/run/docker.sock:/var/run/docker.sock \
|
||||
evalscope-complete-py312:latest \
|
||||
bash -c "
|
||||
cd /opt/evalscope &&
|
||||
python bash/run.py \
|
||||
--model ${MODEL} \
|
||||
--api-url ${API_URL} \
|
||||
--dataset-dir /opt/evalscope \
|
||||
--output-dir ${OUTPUT_DIR} \
|
||||
--suite full \
|
||||
--limit none
|
||||
"
|
||||
27
scripts/run_docker_group.sh
Executable file
27
scripts/run_docker_group.sh
Executable file
@ -0,0 +1,27 @@
|
||||
#!/bin/bash
|
||||
# Docker 多机分组评测
|
||||
# 用法:GROUP=1 ./run_docker_group.sh
|
||||
|
||||
GROUP="${GROUP:-1}"
|
||||
MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}"
|
||||
API_URL="${API_URL:-http://localhost:30000/v1}"
|
||||
HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}"
|
||||
OUTPUT_DIR="/opt/evalscope/output_group${GROUP}"
|
||||
|
||||
docker run -it --rm \
|
||||
--network host \
|
||||
-v "${HOST_EVALSCOPE}:/opt/evalscope" \
|
||||
-v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \
|
||||
-v "${HOST_EVALSCOPE}/output_group${GROUP}:/opt/evalscope/output_group${GROUP}" \
|
||||
-v /var/run/docker.sock:/var/run/docker.sock \
|
||||
evalscope-complete-py312:latest \
|
||||
bash -c "
|
||||
cd /opt/evalscope &&
|
||||
python bash/run.py \
|
||||
--model ${MODEL} \
|
||||
--api-url ${API_URL} \
|
||||
--dataset-dir /opt/evalscope \
|
||||
--output-dir ${OUTPUT_DIR} \
|
||||
--suite group${GROUP} \
|
||||
--limit none
|
||||
"
|
||||
25
scripts/run_docker_lite.sh
Executable file
25
scripts/run_docker_lite.sh
Executable file
@ -0,0 +1,25 @@
|
||||
#!/bin/bash
|
||||
# Docker Lite 快速冒烟
|
||||
# 方式一:使用 --suite lite
|
||||
|
||||
MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}"
|
||||
API_URL="${API_URL:-http://localhost:30000/v1}"
|
||||
HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}"
|
||||
|
||||
docker run -it --rm \
|
||||
--network host \
|
||||
-v "${HOST_EVALSCOPE}:/opt/evalscope" \
|
||||
-v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \
|
||||
-v "${HOST_EVALSCOPE}/output_lite:/opt/evalscope/output_lite" \
|
||||
-v /var/run/docker.sock:/var/run/docker.sock \
|
||||
evalscope-complete-py312:latest \
|
||||
bash -c "
|
||||
cd /opt/evalscope &&
|
||||
python bash/run.py \
|
||||
--model ${MODEL} \
|
||||
--api-url ${API_URL} \
|
||||
--dataset-dir /opt/evalscope \
|
||||
--output-dir /opt/evalscope/output_lite \
|
||||
--suite lite \
|
||||
--limit none
|
||||
"
|
||||
Loading…
x
Reference in New Issue
Block a user