From 03b49a39d07ad4f65bac7a0220937fb162fb8802 Mon Sep 17 00:00:00 2001 From: sora <2075279110@qq.com> Date: Wed, 22 Jul 2026 03:26:58 +0000 Subject: [PATCH] all --- ...结果 - DS4-Flash-INT8-NO-Thinking-2.0-FULL.csv | 29 + bash/fill_simpleqa_bfcl.py | 94 --- bash/rerun_bigcodebench_review.py | 38 -- bash/run.py | 607 ++++++++++-------- bash/run_full.py | 74 --- bash/run_group1.py | 48 -- bash/run_group2.py | 40 -- bash/run_group3.py | 35 - bash/run_lite.py | 40 -- bash/run_single.py | 45 -- bash/test.py | 497 -------------- .../dpv4-int8_nothinking.yaml | 0 scripts/run_docker_full.sh | 26 + scripts/run_docker_group.sh | 27 + scripts/run_docker_lite.sh | 25 + 15 files changed, 448 insertions(+), 1177 deletions(-) create mode 100644 P800模型能力评测结果 - DS4-Flash-INT8-NO-Thinking-2.0-FULL.csv delete mode 100644 bash/fill_simpleqa_bfcl.py delete mode 100644 bash/rerun_bigcodebench_review.py delete mode 100644 bash/run_full.py delete mode 100644 bash/run_group1.py delete mode 100644 bash/run_group2.py delete mode 100644 bash/run_group3.py delete mode 100644 bash/run_lite.py delete mode 100644 bash/run_single.py delete mode 100644 bash/test.py rename {bash/config => config}/dpv4-int8_nothinking.yaml (100%) create mode 100755 scripts/run_docker_full.sh create mode 100755 scripts/run_docker_group.sh create mode 100755 scripts/run_docker_lite.sh diff --git a/P800模型能力评测结果 - DS4-Flash-INT8-NO-Thinking-2.0-FULL.csv b/P800模型能力评测结果 - DS4-Flash-INT8-NO-Thinking-2.0-FULL.csv new file mode 100644 index 0000000..e35e548 --- /dev/null +++ b/P800模型能力评测结果 - DS4-Flash-INT8-NO-Thinking-2.0-FULL.csv @@ -0,0 +1,29 @@ +分类,Benchmark,得分,实测时间(h),总样本数,延迟_mean(s),输出TPS,请求QPS,输入tokens_mean,输出tokens_mean,累计总tokens,TTFT_mean(s),TTFT P90,TTFT P99,TPOT_mean(s),TPOT P90,TPOT P99,,,,,,,,,,,,,,, +代码与工程,bigcodebench,0.9956,0.9915,1140,10.15508,32.99,0.0985,151.77,335.03,554946,0.26441,0.27379,0.42952,0.02957,0.03059,0.03143,,,,,,,,,,,,,,, +,humaneval,0.9044,0.1183,164,2.776502,42.58667,0.3602,163.0793,118.2398,45994,0.25521,0.304186,0.451449,0.021783,0.024498,0.027453,,,,,,,,,,,,,,, +,live_code_bench,0.6756,7.9486,100,10.41723,43.578,0.09644,518.1118,453.5784,1040894,0.253597,0.272382,0.464771,0.02093,0.023769,0.026398,,,,,,,,,,,,,,, +推理与数学,aime24,0.6167,1.2217,30,42.44225,44.85,0.024108,119.3333,1901.842,55169,0.258426,0.274508,0.444511,0.021698,0.025218,hle,,,,,,,,,,,,,,, +,aime25,0.4639,2.0414,30,59.98314,46.03583,0.017092,188.4333,2761.283,80482,0.257489,0.273448,0.4417,0.021655,0.024778,0.026626,,,,,,,,,,,,,,, +,aime26,0.5611,1.4847,30,51.64098,44.66417,0.019925,139.7667,2305.069,60102,0.25465,0.268527,0.435013,0.021667,0.024596,0.026089,,,,,,,,,,,,,,, +,hmmt26,0.4015,1.5481,30,50.8562,44.73,0.020008,108.4242,2274.237,63775,0.255456,0.266018,0.431919,0.022514,0.025679,0.027263,,,,,,,,,,,,,,, +,imo_answerbench,0.36,1.2508,400,39.6005,44.94,0.0253,122.1,1779.8,753159,0.2466,0.2641,0.3167,0.0226,0.0257,0.0285,,,,,,,,,,,,,,, +,hle,0.0508,14.6863,3000,61.787217,34.42,0.0162,292.3252,2126.7916,6047792,0.275126,0.282138,0.474372,0.028997,0.029831,0.031026,,,,,,,,,,,,,,, +,gsm8k,0.9704,0.3178,1319,3.031098,43.8,0.3299,582.9507,132.7726,944039,0.253185,0.267307,0.45469,0.021127,0.02357,0.025568,,,,,,,,,,,,,,, +,competition_math,0.941,3.1817,500,8.681756,51,0.1152,296.5698,442.732,3696509,0.245456,0.263328,0.435642,0.019541,0.021677,0.024484,,,,,,,,,,,,,,, +,bbh,0.9032,1.8036,6513,3.561695,42.95,0.2808,916.98,152.9719,6966457,0.269139,0.276913,0.467974,0.023935,0.030921,0.065977,,,,,,,,,,,,,,, +,drop,0.8183,1.5336,9536,1.824962,36.31,0.548,1273.607,66.25849,12776955,0.297489,0.4593,0.505599,0.023644,0.028072,0.034489,,,,,,,,,,,,,,, +知识与语言理解,gpqa_diamond,0.6944,0.3164,198,10.54948,41.335,0.0948,252.4343,436.0202,135770,0.245972,0.26274,0.456931,0.023819,0.027653,0.030237,,,,,,,,,,,,,,, +,mmlu_pro,0.8285,5.6167,12032,6.264368,31.05,0.1596,1312.528,194.5092,18132672,0.293428,0.437219,0.477191,0.031274,0.034267,0.036824,,,,,,,,,,,,,,, +,simple_qa,0.3814,1.9014,4326,1.88825,30.69,0.529590891,30.351826,57.955155,382016,0.26334,0.269454,0.452375,0.027793,0.031603,0.035322,,,,,,,,,,,,,,, +,mmlu,0.9045,3.0297,14042,2.6957,35.98,0.371,727.6,97,11532337,0.2643,0.2751,0.4663,0.026,0.0305,0.0348,,,,,,,,,,,,,,, +,cmmlu,0.9001,3.4935,11515,3.9225,36.65,0.2549,115.3,143.8,2983005,0.2553,0.2956,0.443,0.0266,0.0307,0.0353,,,,,,,,,,,,,,, +,arc,0.9476,0.2103,1172,0.662452,13.65,1.5095,102.0333,9.042841,394098,0.321567,0.437672,0.449777,0.066642,0.129754,0.132577,,,,,,,,,,,,,,, +,hellaswag,0.8666,0.4453,10042,0.597052,7.32,1.6749,219.8686,4.370444,2251808,0.341421,0.448449,0.461157,0.081246,0.133338,0.136106,,,,,,,,,,,,,,, +,trivia_qa,0.7768,5.3011,11313,9.163784,3.78,0.1091,11964.972,34.6152,95912700,4.722344,8.951851,16.512039,0.131478,0.292489,0.622806,,,,,,,,,,,,,,, +,winogrande,0.7845,0.0775,1267,0.695127,13.88,1.4386,76.11997,9.646409,108666,0.318495,0.433513,0.438349,0.069826,0.129328,0.130936,,,,,,,,,,,,,,, +长上下文,longbench_v2,0.5686,3.0638,503,84.120374,2.75,0.0119,85638.8191,231.6441,43192843,25.148194,53.202063,64.267424,0.248375,0.534405,0.946107,,,,,,,,,,,,,,, +,openai_mrcr,0.7768,5.2828,2399,29.1532,12.95,0.0343,24774.8,377.6,60340773,6.6088,21.0867,37.983,0.0608,0.1125,0.2004,,,,,,,,,,,,,,, +智能体与工具,tau2_bench,0.7684,1.7625,,,,,,,,,,,,,,,,,,,,,,,,,,,, +,general_fc,0.6988,1.3803,2000,9.4082,39.21,0.1063,1619.2,368.9,3976219,0.4226,0.7249,1.2941,0.026,0.0307,0.0398,,,,,,,,,,,,,,, +,bfcl_v3,0.6643,5.23,4441,4.296605,25.32,0.2327,7268.4069,108.7864,97061732,0.418217,0.521432,1.106471,0.035785,0.039796,0.053177,,,,,,,,,,,,,,, +,总计,0.711992593,75.2394,,,,,,,,,,,,,,,,,,,,,,,,,,,, \ No newline at end of file diff --git a/bash/fill_simpleqa_bfcl.py b/bash/fill_simpleqa_bfcl.py deleted file mode 100644 index d9f238c..0000000 --- a/bash/fill_simpleqa_bfcl.py +++ /dev/null @@ -1,94 +0,0 @@ -#!/usr/bin/env python3 -"""Fill simple_qa (fresh re-run) + bfcl_v3 (official OVERALL) into the Excel. - -simple_qa: old cells had D=0.0005 (cache-resume artifact) and empty perf. - Overwrite C/D/F-Q from the NEW report (real perf + ~1.9h duration). -bfcl_v3 : top-level score 0.6643 is evalscope's macro-average (non-standard). - Official BFCL score = OVERALL subset = 0.569. Update C only. -""" -import json -import glob -import openpyxl - -XLSX = '/data1/sora/P800模型能力评测结果_统一格式_filled.xlsx' -SHEETS = ['2.0-FULL', '2.0-Lite'] - -# --- read reports --- -sq = json.load(open('/data1/sora/evalscope/output/simple_qa/seed_42/reports/DeepSeek-V4-Flash-Int8/simple_qa.json')) -bc = json.load(open(glob.glob('/data1/sora/evalscope/output/bfcl_v3/seed_42.bak/reports/DeepSeek-V4-Flash-Int8/bfcl_v3.json')[0])) - - -def perf(d): - pm = d.get('perf_metrics') or {} - s = pm.get('summary', {}) or {} - u = s.get('usage', {}) or {} - tf = s.get('ttft') or {} - tp = s.get('tpot') or {} - lat = (s.get('latency') or {}).get('mean') - return { - 'score': d.get('score'), - 'duration_h': (d.get('duration_sec') or 0) / 3600, - 'latency': lat, - 'tps': (s.get('throughput') or {}).get('avg_output_tps'), - 'qps': (1.0 / lat) if lat else None, - 'in_tok': (u.get('input_tokens') or {}).get('mean'), - 'out_tok': (u.get('output_tokens') or {}).get('mean'), - 'ttc': u.get('total_tokens_count'), - 'ttft_m': tf.get('mean'), - 'ttft90': tf.get('90%') or tf.get('p90'), - 'ttft99': tf.get('99%') or tf.get('p99'), - 'tpot_m': tp.get('mean'), - 'tpot90': tp.get('90%') or tp.get('p90'), - 'tpot99': tp.get('99%') or tp.get('p99'), - } - - -def bfcl_overall(d): - for metric in d.get('metrics', []): - for cat in metric.get('categories', []): - for sub in cat.get('subsets', []): - nm = sub.get('name') - if nm == 'OVERALL' or (isinstance(nm, list) and 'OVERALL' in nm): - return sub.get('score') - return None - - -P = perf(sq) -BC_OVERALL = bfcl_overall(bc) -print('simple_qa:', {k: (round(v, 4) if isinstance(v, float) else v) for k, v in P.items()}) -print('bfcl_v3 OVERALL:', BC_OVERALL) - -# col -> simple_qa perf key -SQ = {'C': P['score'], 'D': round(P['duration_h'], 4), 'F': P['latency'], 'G': P['tps'], - 'H': P['qps'], 'I': P['in_tok'], 'J': P['out_tok'], 'K': P['ttc'], - 'L': P['ttft_m'], 'M': P['ttft90'], 'N': P['ttft99'], - 'O': P['tpot_m'], 'P': P['tpot90'], 'Q': P['tpot99']} - -wb = openpyxl.load_workbook(XLSX) -for sn in SHEETS: - ws = wb[sn] - for r in range(2, 30): - b = ws.cell(r, 2).value - if b == 'simple_qa': - for col, v in SQ.items(): - if v is not None: - ws[f'{col}{r}'] = int(round(v)) if col == 'K' else round(v, 4) if col in ('C', 'D') else v - print(f'{sn} simple_qa row{r}: C={ws[f"C{r}"].value} D={ws[f"D{r}"].value} F={ws[f"F{r}"].value} K={ws[f"K{r}"].value}') - elif b == 'bfcl_v3' and BC_OVERALL is not None: - ws.cell(r, 3).value = round(BC_OVERALL, 4) - print(f'{sn} bfcl_v3 row{r}: C={ws.cell(r,3).value}') - # recompute total D - tr = next((rr for rr in range(2, ws.max_row + 1) if ws.cell(rr, 2).value == '总计'), None) - if tr: - tot = 0.0 - for rr in range(2, tr): - v = ws.cell(rr, 4).value - try: - tot += float(v) - except (TypeError, ValueError): - pass - ws.cell(tr, 4).value = round(tot, 4) - print(f'{sn} 总计 D = {round(tot, 4)}h') - -wb.save(XLSX) -print('saved ->', XLSX) diff --git a/bash/rerun_bigcodebench_review.py b/bash/rerun_bigcodebench_review.py deleted file mode 100644 index c73a66b..0000000 --- a/bash/rerun_bigcodebench_review.py +++ /dev/null @@ -1,38 +0,0 @@ -"""Re-run only the review stage for bigcodebench using existing predictions. - -The upstream `bigcodebench/bigcodebench-evaluate` image has an ENTRYPOINT that -runs the evaluator and exits, which kills the ms_enclave sandbox containers. -We use the locally built `bigcodebench-sandbox:latest` image (same libraries, -entrypoint dropped, runs `tail -f /dev/null`) and re-score the cached -predictions without regenerating them. -""" -import sys -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import run as run_mod -from evalscope import run_task - -DATASET_NAME = 'bigcodebench' -BATCH_SIZE = 4 - -if DATASET_NAME not in run_mod.DATASET_CONFIGS: - raise SystemExit(f'{DATASET_NAME} not found in config') - -ds_cfg = run_mod.DATASET_CONFIGS[DATASET_NAME] -task_cfg = run_mod.build_task_config( - dataset_name=DATASET_NAME, - ds_cfg=ds_cfg, - batch_size=BATCH_SIZE, - enable_thinking=run_mod.ENABLE_THINKING, - seed=run_mod.SEED, - run_idx=0, -) -task_cfg.rerun_review = True - -print(f'Re-running review for {DATASET_NAME}') -print(f'Cache/work dir: {task_cfg.work_dir}') -print(f'Sandbox config: {task_cfg.sandbox.default_config}') - -run_task(task_cfg) -print('Done.') diff --git a/bash/run.py b/bash/run.py index 35a7ce1..850e6a0 100644 --- a/bash/run.py +++ b/bash/run.py @@ -1,8 +1,40 @@ -import yaml +#!/usr/bin/env python3 +""" +Unified benchmark runner for EvalScope. + +A single entry point for lite / mid / full / group1 / group2 / group3 evaluations. +All tunable parameters can be controlled via command-line arguments. + +Examples: + # Full evaluation (all benchmarks, multi-run for stability) + python bash/run.py \ + --model DeepSeek-V4-Flash-Int8 \ + --api-url http://localhost:30000/v1 \ + --dataset-dir /data1/sora/evalscope \ + --output-dir /data1/sora/evalscope/output \ + --suite full \ + --limit none + + # Lite smoke test (~5h with full samples) + python bash/run.py --suite lite --limit none + + # Run only selected benchmarks + python bash/run.py --datasets aime24,gsm8k,arc --limit 20 + + # Custom judge model + python bash/run.py \ + --judge-model deepseek-v4-pro \ + --judge-api-url https://api.deepseek.com/v1 \ + --judge-api-key sk-xxx +""" + +import argparse +import json +import sys from copy import deepcopy from pathlib import Path -import sys -import os + +import yaml from evalscope import run_task, TaskConfig from evalscope.api.agent import NativeAgentConfig @@ -12,116 +44,112 @@ SCRIPT_DIR = Path(__file__).parent.resolve() PROJECT_ROOT = SCRIPT_DIR.parent # ============================================================ -# ★★★ 必改参数 (USER CONFIG) ★★★ -# 每次评测新模型前,只需要检查/修改以下参数。 -# 也可以通过命令行覆盖:--model / --api-url / --dataset-dir / -# --output-dir / --limit / --config +# Default configuration (override via CLI) # ============================================================ -# 模型名(served model name) -MODEL = 'DeepSeek-V4-Flash-Int8' +DEFAULT_MODEL = 'DeepSeek-V4-Flash-Int8' +DEFAULT_API_URL = 'http://localhost:30000/v1' +DEFAULT_DATASET_DIR = str(PROJECT_ROOT) +DEFAULT_OUTPUT_DIR = str(PROJECT_ROOT / 'output') +DEFAULT_CONFIG = str(PROJECT_ROOT / 'config' / 'dpv4-int8_nothinking.yaml') +DEFAULT_TOKENIZER_PATH = '/data1/models/DeepSeek-V4-Flash-INT8' +DEFAULT_LIMIT = None +DEFAULT_SEED = 42 +DEFAULT_BATCH_SIZE = 4 +DEFAULT_ENABLE_THINKING = False -# 模型服务地址(OpenAI 兼容 API) -API_URL = 'http://localhost:30000/v1' +DEFAULT_JUDGE_MODEL = 'DeepSeek/DeepSeek-V4-Pro' +DEFAULT_JUDGE_API_URL = 'https://api.vectron.meta-stone.com/v1' +DEFAULT_JUDGE_API_KEY = 'sk-dbd8a665f7634081b87ec409c7636500' +DEFAULT_JUDGE_MAX_TOKENS = 10240 -# 本地数据集根目录(提前下载好的 datasets 目录) -DATASET_DIR = str(PROJECT_ROOT / 'datasets') +# 长文本 middle-truncation 上限(token 数)。当前默认 128k。 +DEFAULT_TRUNCATION_TOKENS = 32768 * 4 -# 评测输出目录(每个 benchmark 单独子目录) -OUTPUT_DIR = str(PROJECT_ROOT / 'output') +# ============================================================ +# Benchmark suites +# ============================================================ -# 采样数量上限;None 表示跑全部样本 -LIMIT = None +# 多次采样配置:总样本数控制在 ~400-500 +MULTI_RUN_CONFIG = { + 'aime24': 12, + 'aime25': 12, + 'aime26': 12, + 'hmmt26': 12, + 'live_code_bench': 5, + 'imo_answerbench': 4, + 'humaneval': 3, + 'gpqa_diamond': 2, +} -# 每个 benchmark 的生成参数配置文件(max_tokens / temperature 等) -CONFIG = None +# 能力域完整列表 +ALL_MULTI_RUN = [ + 'humaneval', 'live_code_bench', + 'aime24', 'aime25', 'aime26', 'hmmt26', + 'imo_answerbench', 'gpqa_diamond', +] +ALL_SINGLE_RUN = [ + 'bigcodebench', 'bfcl_v3', 'competition_math', 'gsm8k', 'hle', 'super_gpqa', + 'arc', 'bbh', 'cmmlu', 'drop', 'hellaswag', 'mmlu', 'mmlu_pro', + 'simple_qa', 'trivia_qa', 'winogrande', + 'openai_mrcr', 'longbench_v2', +] +ALL_AGENT = ['tau2_bench', 'general_fc'] -# 是否开启 thinking 模式(sglang chat_template_kwargs.thinking) -ENABLE_THINKING = False - -# 测试哪些 benchmark:见下方 DATASETS = _multi_run_order + _single_run_order + _agent_order -# 想只跑部分 benchmark 时,注释掉对应行即可。 -MODEL_PATH = '/data1/models/DeepSeek-V4-Flash-INT8' - -# Judge model for benchmarks that require LLM-as-judge (simple_qa, tau2_bench, etc.) -# Vectron endpoint with DeepSeek-V4-Pro -JUDGE_MODEL_ARGS = { - 'model_id': 'DeepSeek/DeepSeek-V4-Pro', - 'api_url': 'https://api.vectron.meta-stone.com/v1', - 'api_key': 'sk-dbd8a665f7634081b87ec409c7636500', - 'eval_type': 'openai_api', - 'generation_config': { - 'temperature': 0.0, - 'max_tokens': 10240, +# 分组基于 CSV 单次时间 + multi-run 后的 wall time 平衡: +# Group1: ~61h | Group2: ~62h | Group3: ~55h +SUITES = { + 'full': { + 'multi': ALL_MULTI_RUN, + 'single': ALL_SINGLE_RUN, + 'agent': ALL_AGENT, + }, + 'lite': { + 'multi': ['aime24', 'humaneval'], + 'single': ['gsm8k', 'arc', 'longbench_v2'], + 'agent': ['general_fc'], + }, + 'mid': { + 'multi': ['aime24', 'humaneval'], + 'single': [ + 'live_code_bench', 'bigcodebench', 'competition_math', 'gsm8k', + 'gpqa_diamond', 'mmlu_pro', 'simple_qa', 'longbench_v2', 'openai_mrcr', + ], + 'agent': ['general_fc', 'tau2_bench'], + }, + # 多机组分组,基于 CSV 实测完整时间(已含 multi-run)平衡: + # Group1: ~22.7h | Group2: ~25.6h | Group3: ~27.0h | 合计 ~75.2h + 'group1': { + 'multi': ['live_code_bench', 'aime24', 'aime25', 'aime26', 'hmmt26', 'imo_answerbench', 'humaneval'], + 'single': ['bigcodebench', 'competition_math', 'gsm8k', 'drop', 'arc', 'hellaswag', 'winogrande'], + 'agent': [], + }, + 'group2': { + 'multi': [], + 'single': ['hle', 'mmlu_pro', 'trivia_qa'], + 'agent': [], + }, + 'group3': { + 'multi': ['gpqa_diamond'], + 'single': ['openai_mrcr', 'longbench_v2', 'bfcl_v3', 'mmlu', 'cmmlu', 'bbh', 'simple_qa'], + 'agent': ['tau2_bench', 'general_fc'], }, } -TRUNCATION_CONFIG = { - 'longbench_v2': 32768*4, - 'openai_mrcr': 32768*4, +# ============================================================ +# Fixed configuration +# ============================================================ + +MATH_DATASETS = { + 'aime24', 'aime25', 'aime26', 'hmmt26', + 'gsm8k', 'competition_math', 'imo_answerbench', } -# Combine all datasets in order (shortest estimated time first) -# 1. Multi-run small benchmarks (temperature=1.0 for sampling diversity) -_multi_run_order = [ - # 'humaneval', # ~164 samples x 3 runs = 492 - # 'live_code_bench', # ~100 samples x 5 runs = 500 - # 'aime26', # ~30 samples x 12 runs = 360 - # 'aime24', # ~30 samples x 12 runs = 360 - # 'aime25', # ~30 samples x 12 runs = 360 - # 'gpqa_diamond', # ~198 samples x 2 runs = 396 - # 'imo_answerbench', # ~120 samples x 4 runs = 480 - # 'hmmt26', # ~30 samples x 12 runs = 360 -] -# 2. Single-run benchmarks (temperature=0.0 greedy) -_single_run_order = [ - # 'arc', - # 'bfcl_v3', # function calling: greedy decoding - # 'winogrande', - # 'competition_math', - # 'gsm8k', - # 'hellaswag', - # 'bigcodebench', - # 'drop', - # 'bbh', - # 'openai_mrcr', - # 'longbench_v2', - # 'mmlu', - # 'cmmlu', - # 'simple_qa', - # 'mmlu_pro', - 'hle', - # 'trivia_qa', -] +MATH_PROMPT_TEMPLATE = ( + "{question}\n" + "Please reason step by step, and put your final answer within \\boxed{{}}." +) -# 3. Agent / tool benchmarks (last) -_agent_order = [ - # 'tau2_bench', - # 'general_fc', -] -# ============================================================ -# Command line argument overrides (不需要修改) -# ============================================================ - -for i, arg in enumerate(sys.argv): - if arg == '--limit' and i + 1 < len(sys.argv): - limit_val = sys.argv[i + 1] - if limit_val.lower() == 'none' or limit_val.lower() == 'all': - LIMIT = None - else: - LIMIT = int(limit_val) - elif arg == '--model' and i + 1 < len(sys.argv): - MODEL = sys.argv[i + 1] - elif arg == '--api-url' and i + 1 < len(sys.argv): - API_URL = sys.argv[i + 1] - elif arg == '--dataset-dir' and i + 1 < len(sys.argv): - DATASET_DIR = sys.argv[i + 1] - elif arg == '--output-dir' and i + 1 < len(sys.argv): - OUTPUT_DIR = sys.argv[i + 1] - elif arg == '--config' and i + 1 < len(sys.argv): - CONFIG = sys.argv[i + 1] - -# Benchmarks that require sandboxed code execution SANDBOX_DATASETS = {'humaneval', 'bigcodebench'} SANDBOX_CONFIGS = { 'bigcodebench': { @@ -141,91 +169,69 @@ SANDBOX_CONFIGS = { }, } -# ============================================================ -# User-tunable parameters -# ============================================================ - -# Fixed seed for reproducibility. With temperature > 0 the model still samples -# randomly, so running N times with the same seed still yields variance. -SEED = 42 - -# Benchmarks to run multiple times with temperature=1.0. -# Format: {benchmark_name: num_runs} -# Number of runs chosen so total samples ≈ 400-500 per benchmark. -MULTI_RUN_CONFIG = { - # ~30 samples each -> 12 runs = ~360 samples - 'aime24': 12, - 'aime25': 12, - 'aime26': 12, - 'hmmt26': 12, - # ~100 samples -> 5 runs = ~500 samples - 'live_code_bench': 5, - # ~120 samples -> 4 runs = ~480 samples - 'imo_answerbench': 4, - # ~164 samples -> 3 runs = ~492 samples - 'humaneval': 3, - # ~198 samples -> 2 runs = ~396 samples - 'gpqa_diamond': 2, -} - -# Single-run benchmarks (temperature=0.0 greedy) -SINGLE_RUN_DATASETS = [ - 'bigcodebench', - 'bfcl_v3', # function calling: greedy decoding - 'competition_math', - 'gsm8k', - 'hle', - 'super_gpqa', - 'arc', - 'bbh', - 'cmmlu', - 'drop', - 'hellaswag', - 'mmlu', - 'mmlu_pro', - 'simple_qa', - 'trivia_qa', - 'winogrande', - 'openai_mrcr', # long-context, run once - 'longbench_v2', # long-context, run once -] - -# Agent / tool benchmarks (temperature=0.0 greedy, run once) -AGENT_DATASETS = [ - 'tau2_bench', - 'general_fc', -] - - - -DATASETS = _multi_run_order + _single_run_order + _agent_order - -# ENABLE_THINKING 已移至文件头部「必改参数」区 -BATCH_SIZE_LIST = [4] -# LIMIT is parsed from command line: --limit N -SHUFFLE = True # ============================================================ -# Fixed configuration +# CLI parser # ============================================================ -CONFIG_PATH = str(Path(CONFIG) if CONFIG else SCRIPT_DIR / 'config' / 'dpv4-int8_nothinking.yaml') -if not Path(CONFIG_PATH).exists(): - raise FileNotFoundError(f'Config file not found: {CONFIG_PATH}') +def build_parser(): + parser = argparse.ArgumentParser( + description='Unified EvalScope benchmark runner', + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog='Suites: full, lite, mid, group1, group2, group3', + ) + # Model / API + parser.add_argument('--model', default=DEFAULT_MODEL, + help='Served model name (default: %(default)s)') + parser.add_argument('--api-url', default=DEFAULT_API_URL, + help='OpenAI-compatible API URL (default: %(default)s)') -MATH_DATASETS = { - 'aime24', 'aime25', 'aime26', 'hmmt26', - 'gsm8k', 'competition_math', 'imo_answerbench', -} + # Paths + parser.add_argument('--dataset-dir', default=DEFAULT_DATASET_DIR, + help='Parent directory containing datasets/ subdir (default: %(default)s)') + parser.add_argument('--output-dir', default=DEFAULT_OUTPUT_DIR, + help='Output root directory (default: %(default)s)') + parser.add_argument('--config', default=DEFAULT_CONFIG, + help='YAML config path (default: %(default)s)') + parser.add_argument('--tokenizer-path', default=DEFAULT_TOKENIZER_PATH, + help='Local tokenizer path for middle-truncation (default: %(default)s)') -MATH_PROMPT_TEMPLATE = ( - "{question}\n" - "Please reason step by step, and put your final answer within \\boxed{{}}." -) + # Run control + parser.add_argument('--suite', default='full', choices=list(SUITES.keys()), + help='Benchmark suite to run (default: %(default)s)') + parser.add_argument('--datasets', '--benchmarks', dest='datasets', default=None, + help='Override suite with comma-separated benchmark names, e.g. aime24,gsm8k') + parser.add_argument('--exclude', default=None, + help='Comma-separated benchmarks to exclude from the chosen suite') + parser.add_argument('--limit', default=None, + help='Max samples per benchmark; "none"/"all" for no limit (default: none)') + parser.add_argument('--seed', type=int, default=DEFAULT_SEED, + help='Random seed (default: %(default)s)') + parser.add_argument('--batch-size', type=int, default=DEFAULT_BATCH_SIZE, + help='Evaluation batch size (default: %(default)s)') -with open(CONFIG_PATH, 'r', encoding='utf-8') as f: - DATASET_CONFIGS = yaml.safe_load(f) + # Decoding / thinking + parser.add_argument('--thinking', action='store_true', default=None, + help='Enable thinking mode (sglang chat_template_kwargs.thinking=True)') + parser.add_argument('--no-thinking', dest='thinking', action='store_false', + help='Disable thinking mode (default)') + + # Judge model + parser.add_argument('--judge-model', default=DEFAULT_JUDGE_MODEL, + help='Judge model name (default: %(default)s)') + parser.add_argument('--judge-api-url', default=DEFAULT_JUDGE_API_URL, + help='Judge model API URL (default: %(default)s)') + parser.add_argument('--judge-api-key', default=DEFAULT_JUDGE_API_KEY, + help='Judge model API key') + parser.add_argument('--judge-max-tokens', type=int, default=DEFAULT_JUDGE_MAX_TOKENS, + help='Judge model max_tokens (default: %(default)s)') + + # Truncation + parser.add_argument('--truncation-tokens', type=int, default=DEFAULT_TRUNCATION_TOKENS, + help='Middle-truncation token budget for long-context benchmarks (default: %(default)s)') + + return parser # ============================================================ @@ -235,23 +241,21 @@ with open(CONFIG_PATH, 'r', encoding='utf-8') as f: _TOKENIZER = None -def get_tokenizer(): +def get_tokenizer(tokenizer_path: str): global _TOKENIZER if _TOKENIZER is None: from transformers import AutoTokenizer try: - # Try local path first - _TOKENIZER = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True) + _TOKENIZER = AutoTokenizer.from_pretrained(tokenizer_path, trust_remote_code=True) except Exception: - # Fallback to HuggingFace model ID _TOKENIZER = AutoTokenizer.from_pretrained('deepseek-ai/DeepSeek-V4-Flash', trust_remote_code=True) return _TOKENIZER -def truncate_middle(text: str, max_tokens: int) -> str: +def truncate_middle(text: str, max_tokens: int, tokenizer_path: str) -> str: if max_tokens <= 0: return text - tokenizer = get_tokenizer() + tokenizer = get_tokenizer(tokenizer_path) token_ids = tokenizer.encode(text, add_special_tokens=False) if len(token_ids) <= max_tokens: return text @@ -261,41 +265,35 @@ def truncate_middle(text: str, max_tokens: int) -> str: return tokenizer.decode(truncated_ids, skip_special_tokens=True) -def _patch_adapters_for_truncation(): +def _patch_adapters_for_truncation(tokenizer_path: str, truncation_tokens: int): from evalscope.benchmarks.longbench_v2.longbench_v2_adapter import LongBenchV2Adapter from evalscope.benchmarks.openai_mrcr.openai_mrcr_adapter import OpenAIMRCRAdapter - # Patch LongBenchV2Adapter.format_prompt_template _orig_longbench_format = LongBenchV2Adapter.format_prompt_template + def _patched_longbench_format(self, sample): - max_tok = TRUNCATION_CONFIG.get('longbench_v2') - if max_tok and sample.metadata and 'context' in sample.metadata: - sample.metadata['context'] = truncate_middle(sample.metadata['context'], max_tok) + if sample.metadata and 'context' in sample.metadata: + sample.metadata['context'] = truncate_middle(sample.metadata['context'], truncation_tokens, tokenizer_path) return _orig_longbench_format(self, sample) + LongBenchV2Adapter.format_prompt_template = _patched_longbench_format - # Patch OpenAIMRCRAdapter.record_to_sample with needle-aware truncation. - # MRCR is a long chat history with needles hidden at desired_msg_index. - # We keep the head, tail, and a window around the needle, and truncate - # each kept message if it is still too long. This preserves the retrieval - # task while fitting GPU memory. _orig_mrcr_record = OpenAIMRCRAdapter.record_to_sample + def _patched_mrcr_record(self, record): - import json - max_total_tok = TRUNCATION_CONFIG.get('openai_mrcr') per_msg_max_tok = 8192 - if max_total_tok and 'prompt' in record: + if 'prompt' in record: try: prompt_data = json.loads(record['prompt']) if not isinstance(prompt_data, list) or len(prompt_data) == 0: return _orig_mrcr_record(self, record) - tokenizer = get_tokenizer() + tokenizer = get_tokenizer(tokenizer_path) total_tok = sum( len(tokenizer.encode(msg.get('content', '') if isinstance(msg, dict) else '', add_special_tokens=False)) for msg in prompt_data ) - if total_tok <= max_total_tok: + if total_tok <= truncation_tokens: return _orig_mrcr_record(self, record) desired_idx = record.get('desired_msg_index', 0) @@ -304,10 +302,8 @@ def _patch_adapters_for_truncation(): n = len(prompt_data) keep = set() - # Head and tail context keep.update(range(min(2, n))) keep.update(range(max(0, n - 2), n)) - # Window around the needle window = 2 keep.update(range(max(0, desired_idx - window), min(n, desired_idx + window + 1))) keep = sorted(keep) @@ -319,7 +315,7 @@ def _patch_adapters_for_truncation(): msg = dict(msg) content = msg.get('content', '') if len(tokenizer.encode(content, add_special_tokens=False)) > per_msg_max_tok: - msg['content'] = truncate_middle(content, per_msg_max_tok) + msg['content'] = truncate_middle(content, per_msg_max_tok, tokenizer_path) new_prompt.append(msg) record = dict(record) @@ -327,16 +323,21 @@ def _patch_adapters_for_truncation(): except (json.JSONDecodeError, TypeError): pass return _orig_mrcr_record(self, record) + OpenAIMRCRAdapter.record_to_sample = _patched_mrcr_record -_patch_adapters_for_truncation() - - # ============================================================ # Helpers # ============================================================ +def load_dataset_configs(config_path: str): + if not Path(config_path).exists(): + raise FileNotFoundError(f'Config file not found: {config_path}') + with open(config_path, 'r', encoding='utf-8') as f: + return yaml.safe_load(f) + + def configure_thinking(generation_config: dict, enable: bool) -> dict: extra_body = generation_config.get('extra_body', {}) chat_template_kwargs = extra_body.get('chat_template_kwargs', {}) @@ -363,14 +364,24 @@ def build_agent_config(agent_cfg: dict) -> NativeAgentConfig: return NativeAgentConfig(**agent_cfg) -def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_thinking: bool, seed: int, run_idx: int = 0) -> TaskConfig: - # Each run gets its own work_dir so use_cache does not reuse predictions - # across repeated samples. This is required for temperature=1.0 multi-run - # benchmarks to actually measure variance. +def build_task_config( + dataset_name: str, + ds_cfg: dict, + batch_size: int, + enable_thinking: bool, + seed: int, + limit, + output_dir: str, + model: str, + api_url: str, + dataset_dir: str, + judge_model_args: dict, + run_idx: int = 0, +) -> TaskConfig: if run_idx > 0: - work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}_run_{run_idx}' + work_dir = Path(output_dir) / dataset_name / f'seed_{seed}_run_{run_idx}' else: - work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}' + work_dir = Path(output_dir) / dataset_name / f'seed_{seed}' work_dir.mkdir(parents=True, exist_ok=True) work_dir = str(work_dir) @@ -388,13 +399,13 @@ def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_t agent_config = build_agent_config(ds_cfg['agent_config']) return TaskConfig( - model=MODEL, - api_url=API_URL, + model=model, + api_url=api_url, eval_type='openai_api', - dataset_dir=DATASET_DIR, - judge_model_args=JUDGE_MODEL_ARGS, + dataset_dir=dataset_dir, + judge_model_args=judge_model_args, seed=seed, - limit=LIMIT, + limit=limit, collect_perf=True, no_timestamp=True, work_dir=work_dir, @@ -419,71 +430,135 @@ def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_t # ============================================================ -# Main loop +# Main # ============================================================ def main(): - print(f"Config: {CONFIG_PATH}") - print(f"Model: {MODEL}") - print(f"API URL: {API_URL}") - print(f"Dataset Dir: {DATASET_DIR}") - print(f"Output Dir: {OUTPUT_DIR}") - print(f"Limit: {LIMIT if LIMIT is not None else 'ALL'}") - print(f"Total benchmarks: {len(DATASETS)}") - print("="*60) + parser = build_parser() + args = parser.parse_args() - for batch_size in BATCH_SIZE_LIST: - # 1. Run multi-run benchmarks (temperature=1.0, same seed, variance from sampling) - for dataset_name in _multi_run_order: - if dataset_name not in DATASET_CONFIGS: - print(f"WARNING: {dataset_name} not in YAML config, skipping") - continue - ds_cfg = DATASET_CONFIGS[dataset_name] - num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1) - for run_idx in range(num_runs): - print(f"\n{'='*60}") - print(f"Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={SEED})") - print(f"{'='*60}") - task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED, run_idx=run_idx) - try: - run_task(task_cfg) - except Exception as e: - print(f"ERROR in {dataset_name} (run {run_idx + 1}): {e}") - continue + # Resolve limit + limit = args.limit + if limit is not None: + if str(limit).lower() in ('none', 'all'): + limit = None + else: + limit = int(limit) - # 2. Run single-run benchmarks (temperature=0.0 greedy) - for dataset_name in _single_run_order: - if dataset_name not in DATASET_CONFIGS: - print(f"WARNING: {dataset_name} not in YAML config, skipping") - continue - ds_cfg = DATASET_CONFIGS[dataset_name] + enable_thinking = DEFAULT_ENABLE_THINKING if args.thinking is None else args.thinking + + # Resolve suite or custom datasets + if args.datasets: + custom = [d.strip() for d in args.datasets.split(',') if d.strip()] + multi_run = [d for d in custom if d in MULTI_RUN_CONFIG] + single_run = [d for d in custom if d not in MULTI_RUN_CONFIG] + agent = [d for d in custom if d in ALL_AGENT] + single_run = [d for d in single_run if d not in ALL_AGENT] + else: + suite = SUITES[args.suite] + multi_run = list(suite['multi']) + single_run = list(suite['single']) + agent = list(suite['agent']) + + # Apply --exclude + if args.exclude: + exclude = {d.strip() for d in args.exclude.split(',') if d.strip()} + multi_run = [d for d in multi_run if d not in exclude] + single_run = [d for d in single_run if d not in exclude] + agent = [d for d in agent if d not in exclude] + + judge_model_args = { + 'model_id': args.judge_model, + 'api_url': args.judge_api_url, + 'api_key': args.judge_api_key, + 'eval_type': 'openai_api', + 'generation_config': { + 'temperature': 0.0, + 'max_tokens': args.judge_max_tokens, + }, + } + + truncation_tokens = args.truncation_tokens + _patch_adapters_for_truncation(args.tokenizer_path, truncation_tokens) + + dataset_configs = load_dataset_configs(args.config) + + print('=' * 60) + print(f'Config: {args.config}') + print(f'Model: {args.model}') + print(f'API URL: {args.api_url}') + print(f'Dataset Dir: {args.dataset_dir}') + print(f'Output Dir: {args.output_dir}') + print(f'Suite: {args.suite}') + print(f'Limit: {limit if limit is not None else "ALL"}') + print(f'Thinking: {enable_thinking}') + print(f'Seed: {args.seed}') + print(f'Batch Size: {args.batch_size}') + print(f'Tokenizer Path: {args.tokenizer_path}') + print(f'Truncation Tokens: {truncation_tokens}') + print(f'Multi-run datasets: {multi_run}') + print(f'Single-run datasets: {single_run}') + print(f'Agent datasets: {agent}') + print('=' * 60) + + for dataset_name in multi_run: + if dataset_name not in dataset_configs: + print(f'WARNING: {dataset_name} not in YAML config, skipping') + continue + ds_cfg = dataset_configs[dataset_name] + num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1) + for run_idx in range(num_runs): print(f"\n{'='*60}") - print(f"Running: {dataset_name} (seed={SEED})") + print(f'Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={args.seed})') print(f"{'='*60}") - task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED) + task_cfg = build_task_config( + dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit, + args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args, + run_idx=run_idx, + ) try: run_task(task_cfg) except Exception as e: - print(f"ERROR in {dataset_name}: {e}") + print(f'ERROR in {dataset_name} (run {run_idx + 1}): {e}') continue - # 3. Run agent benchmarks - for dataset_name in _agent_order: - if dataset_name not in DATASET_CONFIGS: - print(f"WARNING: {dataset_name} not in YAML config, skipping") - continue - ds_cfg = DATASET_CONFIGS[dataset_name] - print(f"\n{'='*60}") - print(f"Running: {dataset_name} (seed={SEED})") - print(f"{'='*60}") - task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED) - try: - run_task(task_cfg) - except Exception as e: - print(f"ERROR in {dataset_name}: {e}") - continue + for dataset_name in single_run: + if dataset_name not in dataset_configs: + print(f'WARNING: {dataset_name} not in YAML config, skipping') + continue + ds_cfg = dataset_configs[dataset_name] + print(f"\n{'='*60}") + print(f'Running: {dataset_name} (seed={args.seed})') + print(f"{'='*60}") + task_cfg = build_task_config( + dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit, + args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args, + ) + try: + run_task(task_cfg) + except Exception as e: + print(f'ERROR in {dataset_name}: {e}') + continue - print("\nAll benchmarks done!") + for dataset_name in agent: + if dataset_name not in dataset_configs: + print(f'WARNING: {dataset_name} not in YAML config, skipping') + continue + ds_cfg = dataset_configs[dataset_name] + print(f"\n{'='*60}") + print(f'Running: {dataset_name} (seed={args.seed})') + print(f"{'='*60}") + task_cfg = build_task_config( + dataset_name, ds_cfg, args.batch_size, enable_thinking, args.seed, limit, + args.output_dir, args.model, args.api_url, args.dataset_dir, judge_model_args, + ) + try: + run_task(task_cfg) + except Exception as e: + print(f'ERROR in {dataset_name}: {e}') + continue + + print('\nAll benchmarks done!') if __name__ == '__main__': diff --git a/bash/run_full.py b/bash/run_full.py deleted file mode 100644 index 462c952..0000000 --- a/bash/run_full.py +++ /dev/null @@ -1,74 +0,0 @@ -#!/usr/bin/env python3 -""" -Full evaluation suite: ~3-5 days -Runs all benchmarks with full datasets and multiple runs for stability. -Use --limit none (default) for the complete evaluation. -""" -from pathlib import Path -import sys - -sys.path.insert(0, str(Path(__file__).parent)) - -import run as run_module - -# Use the default full configuration from run.py -# (already ordered by estimated runtime) -run_module._multi_run_order = [ - 'humaneval', - 'live_code_bench', - 'aime26', - 'aime24', - 'aime25', - 'gpqa_diamond', - 'imo_answerbench', - 'hmmt26', -] - -run_module._single_run_order = [ - 'arc', - 'bfcl_v3', - 'winogrande', - 'competition_math', - 'gsm8k', - 'hellaswag', - 'bigcodebench', - 'drop', - 'bbh', - 'openai_mrcr', - 'longbench_v2', - 'mmlu', - 'cmmlu', - 'super_gpqa', - 'simple_qa', - 'mmlu_pro', - 'hle', - 'trivia_qa', -] - -run_module._agent_order = [ - 'tau2_bench', - 'general_fc', -] - -run_module.DATASETS = ( - run_module._multi_run_order - + run_module._single_run_order - + run_module._agent_order -) - -# Full: many runs for statistical stability -run_module.MULTI_RUN_CONFIG = { - 'aime24': 12, - 'aime25': 12, - 'aime26': 12, - 'hmmt26': 12, - 'live_code_bench': 5, - 'imo_answerbench': 4, - 'humaneval': 3, - 'gpqa_diamond': 2, -} - -if __name__ == '__main__': - print('Full benchmarks:', run_module.DATASETS) - print('Estimated time: ~3-5 days with --limit none') - run_module.main() diff --git a/bash/run_group1.py b/bash/run_group1.py deleted file mode 100644 index aa0377a..0000000 --- a/bash/run_group1.py +++ /dev/null @@ -1,48 +0,0 @@ -""" -Group 1: Code + Reasoning + short Knowledge + Agent (est. ~18-22h) -Run on machine 1. -""" -from pathlib import Path -import sys - -# Ensure we can import run.py from the same directory inside or outside Docker -sys.path.insert(0, str(Path(__file__).parent)) - -import run as run_module - -# Override dataset lists for group 1 -run_module._multi_run_order = [ - 'humaneval', - 'live_code_bench', - 'aime26', - 'aime24', - 'aime25', - 'gpqa_diamond', - 'imo_answerbench', - 'hmmt26', -] - -run_module._single_run_order = [ - 'arc', - 'winogrande', - 'competition_math', - 'gsm8k', - 'hellaswag', - 'bigcodebench', -] - -run_module._agent_order = [ - 'bfcl_v3', - 'tau2_bench', -] - -run_module.DATASETS = ( - run_module._multi_run_order - + run_module._single_run_order - + run_module._agent_order -) - -if __name__ == '__main__': - print('Group 1 benchmarks:', run_module.DATASETS) - print('Estimated time: ~18-22h') - run_module.main() diff --git a/bash/run_group2.py b/bash/run_group2.py deleted file mode 100644 index fd33035..0000000 --- a/bash/run_group2.py +++ /dev/null @@ -1,40 +0,0 @@ -""" -Group 2: Knowledge + Long-context + Agent (est. ~20-24h) -Run on machine 2. -""" -from pathlib import Path -import sys - -# Ensure we can import run.py from the same directory inside or outside Docker -sys.path.insert(0, str(Path(__file__).parent)) - -import run as run_module - -# Override dataset lists for group 2 -run_module._multi_run_order = [] - -run_module._single_run_order = [ - 'drop', - 'bbh', - 'openai_mrcr', - 'longbench_v2', - 'mmlu', - 'cmmlu', - 'super_gpqa', - 'simple_qa', -] - -run_module._agent_order = [ - 'general_fc', -] - -run_module.DATASETS = ( - run_module._multi_run_order - + run_module._single_run_order - + run_module._agent_order -) - -if __name__ == '__main__': - print('Group 2 benchmarks:', run_module.DATASETS) - print('Estimated time: ~20-24h') - run_module.main() diff --git a/bash/run_group3.py b/bash/run_group3.py deleted file mode 100644 index 8917309..0000000 --- a/bash/run_group3.py +++ /dev/null @@ -1,35 +0,0 @@ -""" -Group 3: Long-running Knowledge benchmarks (est. ~35-40h) -Run on machine 3. -Note: trivia_qa and hle are inherently slow; consider using --limit to control time. -""" -from pathlib import Path -import sys - -# Ensure we can import run.py from the same directory inside or outside Docker -sys.path.insert(0, str(Path(__file__).parent)) - -import run as run_module - -# Override dataset lists for group 3 -run_module._multi_run_order = [] - -run_module._single_run_order = [ - 'mmlu_pro', - 'hle', - 'trivia_qa', -] - -run_module._agent_order = [] - -run_module.DATASETS = ( - run_module._multi_run_order - + run_module._single_run_order - + run_module._agent_order -) - -if __name__ == '__main__': - print('Group 3 benchmarks:', run_module.DATASETS) - print('Estimated time: ~35-40h (trivia_qa and hle are slow)') - print('Tip: use --limit 500 to cap runtime if needed') - run_module.main() diff --git a/bash/run_lite.py b/bash/run_lite.py deleted file mode 100644 index 1b923ea..0000000 --- a/bash/run_lite.py +++ /dev/null @@ -1,40 +0,0 @@ -#!/usr/bin/env python3 -""" -Lite evaluation suite: ~2-4h -Covers all 5 capability domains with 1-2 benchmarks each. -Use --limit 20 (default) for quick smoke testing. -""" -from pathlib import Path -import sys - -sys.path.insert(0, str(Path(__file__).parent)) - -import run as run_module - -run_module._multi_run_order = [ - 'aime24', # reasoning - 'humaneval', # code -] - -run_module._single_run_order = [ - 'gsm8k', # reasoning - 'mmlu_pro', # knowledge - 'simple_qa', # knowledge - 'longbench_v2', # long-context -] - -run_module._agent_order = [ - 'bfcl_v3', # tool/function calling -] - -run_module.DATASETS = ( - run_module._multi_run_order - + run_module._single_run_order - + run_module._agent_order -) - - -if __name__ == '__main__': - print('Lite benchmarks:', run_module.DATASETS) - run_module.main() - diff --git a/bash/run_single.py b/bash/run_single.py deleted file mode 100644 index 13b57f5..0000000 --- a/bash/run_single.py +++ /dev/null @@ -1,45 +0,0 @@ -#!/usr/bin/env python3 -"""Run a single benchmark for quick smoke tests. - -Usage: - python bash/run_single.py bigcodebench --limit 1 - python bash/run_single.py bfcl_v3 --limit 5 --model MODEL --api-url URL -""" -import sys -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import run as run_mod -from evalscope import run_task - -if len(sys.argv) < 2: - print('Usage: python bash/run_single.py [--limit N] [--model M] [--api-url U]') - sys.exit(1) - -dataset_name = sys.argv[1] -limit = None -for i, arg in enumerate(sys.argv): - if arg == '--limit' and i + 1 < len(sys.argv): - v = sys.argv[i + 1] - limit = None if v.lower() in ('none', 'all') else int(v) - elif arg == '--model' and i + 1 < len(sys.argv): - run_mod.MODEL = sys.argv[i + 1] - elif arg == '--api-url' and i + 1 < len(sys.argv): - run_mod.API_URL = sys.argv[i + 1] - -if dataset_name not in run_mod.DATASET_CONFIGS: - raise SystemExit(f'{dataset_name} not in config') - -run_mod.LIMIT = limit -ds_cfg = run_mod.DATASET_CONFIGS[dataset_name] -task_cfg = run_mod.build_task_config( - dataset_name=dataset_name, - ds_cfg=ds_cfg, - batch_size=4, - enable_thinking=run_mod.ENABLE_THINKING, - seed=run_mod.SEED, - run_idx=0, -) -print(f'Running {dataset_name} with limit={limit}') -run_task(task_cfg) -print('Done.') diff --git a/bash/test.py b/bash/test.py deleted file mode 100644 index f955845..0000000 --- a/bash/test.py +++ /dev/null @@ -1,497 +0,0 @@ -import yaml -from copy import deepcopy -from pathlib import Path -import sys -import os - -from evalscope import run_task, TaskConfig -from evalscope.api.agent import NativeAgentConfig -from evalscope.config import SandboxTaskConfig - -SCRIPT_DIR = Path(__file__).parent.resolve() -PROJECT_ROOT = SCRIPT_DIR.parent - - -# ============================================================ -# ★★★ 必改参数 (USER CONFIG) ★★★ -# 每次评测新模型前,只需要检查/修改以下参数。 -# 也可以通过命令行覆盖:--model / --api-url / --dataset-dir / -# --output-dir / --limit / --config -# ============================================================ - -# 模型名(served model name) -MODEL = 'DeepSeek-V4-Flash-Int8' - -# 模型服务地址(OpenAI 兼容 API) -API_URL = 'http://localhost:30000/v1' - -# 本地数据集根目录(提前下载好的 datasets 目录) -DATASET_DIR = str(PROJECT_ROOT / 'datasets') - -# 评测输出目录(每个 benchmark 单独子目录) -OUTPUT_DIR = str(PROJECT_ROOT / 'output') - -# 采样数量上限;None 表示跑全部样本 -LIMIT = None - -# 每个 benchmark 的生成参数配置文件(max_tokens / temperature 等) -CONFIG = None - -# 是否开启 thinking 模式(sglang chat_template_kwargs.thinking) -ENABLE_THINKING = False - -# 测试哪些 benchmark:见下方 DATASETS = _multi_run_order + _single_run_order + _agent_order -# 想只跑部分 benchmark 时,注释掉对应行即可。 - -# ============================================================ -# Command line argument overrides (不需要修改) -# ============================================================ - -for i, arg in enumerate(sys.argv): - if arg == '--limit' and i + 1 < len(sys.argv): - limit_val = sys.argv[i + 1] - if limit_val.lower() == 'none' or limit_val.lower() == 'all': - LIMIT = None - else: - LIMIT = int(limit_val) - elif arg == '--model' and i + 1 < len(sys.argv): - MODEL = sys.argv[i + 1] - elif arg == '--api-url' and i + 1 < len(sys.argv): - API_URL = sys.argv[i + 1] - elif arg == '--dataset-dir' and i + 1 < len(sys.argv): - DATASET_DIR = sys.argv[i + 1] - elif arg == '--output-dir' and i + 1 < len(sys.argv): - OUTPUT_DIR = sys.argv[i + 1] - elif arg == '--config' and i + 1 < len(sys.argv): - CONFIG = sys.argv[i + 1] - -# Benchmarks that require sandboxed code execution -SANDBOX_DATASETS = {'humaneval', 'bigcodebench'} -# Per-dataset sandbox configs. -# bigcodebench's upstream image has ENTRYPOINT ["python3", "-m", "bigcodebench.evaluate"], -# which exits immediately and breaks ms_enclave exec. We built a derivative image -# `bigcodebench-sandbox:latest` that keeps the same Python environment but drops the -# entrypoint and runs `tail -f /dev/null` so the container stays alive. -SANDBOX_CONFIGS = { - 'bigcodebench': { - 'image': 'bigcodebench-sandbox:latest', - 'working_dir': '/tmp', - 'tools_config': { - 'shell_executor': {}, - 'python_executor': {} - } - }, - 'humaneval': { - 'image': 'python:3.11-slim', - 'tools_config': { - 'shell_executor': {}, - 'python_executor': {} - } - }, -} - -# ============================================================ -# User-tunable parameters -# ============================================================ - -# Fixed seed for reproducibility. With temperature > 0 the model still samples -# randomly, so running N times with the same seed still yields variance. -SEED = 42 - -# Benchmarks to run multiple times with temperature=1.0. -# Format: {benchmark_name: num_runs} -# Number of runs chosen so total samples ≈ 400-500 per benchmark. -MULTI_RUN_CONFIG = { - # ~30 samples each -> 12 runs = ~360 samples - 'aime24': 12, - 'aime25': 12, - 'aime26': 12, - 'hmmt26': 12, - # ~100 samples -> 5 runs = ~500 samples - 'live_code_bench': 5, - # ~120 samples -> 4 runs = ~480 samples - 'imo_answerbench': 4, - # ~164 samples -> 3 runs = ~492 samples - 'humaneval': 3, - # ~198 samples -> 2 runs = ~396 samples - 'gpqa_diamond': 2, -} - -# Single-run benchmarks (temperature=0.0 greedy) -SINGLE_RUN_DATASETS = [ - 'bigcodebench', - 'bfcl_v3', # function calling: greedy decoding - 'competition_math', - 'gsm8k', - 'hle', - 'super_gpqa', - 'arc', - 'bbh', - 'cmmlu', - 'drop', - 'hellaswag', - 'mmlu', - 'mmlu_pro', - 'simple_qa', - 'trivia_qa', - 'winogrande', - 'openai_mrcr', # long-context, run once - 'longbench_v2', # long-context, run once -] - -# Agent / tool benchmarks (temperature=0.0 greedy, run once) -AGENT_DATASETS = [ - 'tau2_bench', - 'general_fc', -] - -# Combine all datasets in order (shortest estimated time first) -# 1. Multi-run small benchmarks (temperature=1.0 for sampling diversity) -_multi_run_order = [ - # 'humaneval', # ~164 samples x 3 runs = 492 - # 'live_code_bench', # ~100 samples x 5 runs = 500 - # 'aime26', # ~30 samples x 12 runs = 360 - # 'aime24', # ~30 samples x 12 runs = 360 - # 'aime25', # ~30 samples x 12 runs = 360 - # 'gpqa_diamond', # ~198 samples x 2 runs = 396 - # 'imo_answerbench', # ~120 samples x 4 runs = 480 - # 'hmmt26', # ~30 samples x 12 runs = 360 -] - -# 2. Single-run benchmarks (temperature=0.0 greedy) -_single_run_order = [ - # 'arc', - # 'bfcl_v3', # function calling: greedy decoding - # 'winogrande', - # 'competition_math', - # 'gsm8k', - # 'hellaswag', - 'bigcodebench', - # 'drop', - # 'bbh', - # 'openai_mrcr', - # 'longbench_v2', - # 'mmlu', - # 'cmmlu', - # 'super_gpqa', - # 'simple_qa', - # 'mmlu_pro', - 'hle', - # 'trivia_qa', -] - -# 3. Agent / tool benchmarks (last) -_agent_order = [ - # 'tau2_bench', - # 'general_fc', -] - -DATASETS = _multi_run_order + _single_run_order + _agent_order - -# ENABLE_THINKING 已移至文件头部「必改参数」区 -BATCH_SIZE_LIST = [4] -# LIMIT is parsed from command line: --limit N -SHUFFLE = True - -# ============================================================ -# Fixed configuration -# ============================================================ - -CONFIG_PATH = str(Path(CONFIG) if CONFIG else SCRIPT_DIR / 'config' / 'dpv4-int8_nothinking.yaml') -if not Path(CONFIG_PATH).exists(): - raise FileNotFoundError(f'Config file not found: {CONFIG_PATH}') - -MODEL_PATH = '/data1/models/DeepSeek-V4-Flash-INT8' - -# Judge model for benchmarks that require LLM-as-judge (simple_qa, tau2_bench, etc.) -# Vectron endpoint with DeepSeek-V4-Pro -JUDGE_MODEL_ARGS = { - 'model_id': 'DeepSeek/DeepSeek-V4-Pro', - 'api_url': 'https://api.vectron.meta-stone.com/v1', - 'api_key': 'sk-dbd8a665f7634081b87ec409c7636500', - 'eval_type': 'openai_api', - 'generation_config': { - 'temperature': 0.0, - 'max_tokens': 10240, - }, -} - -TRUNCATION_CONFIG = { - 'longbench_v2': 32768*4, - 'openai_mrcr': 32768*4, -} - -MATH_DATASETS = { - 'aime24', 'aime25', 'aime26', 'hmmt26', - 'gsm8k', 'competition_math', 'imo_answerbench', -} - -MATH_PROMPT_TEMPLATE = ( - "{question}\n" - "Please reason step by step, and put your final answer within \\boxed{{}}." -) - -with open(CONFIG_PATH, 'r', encoding='utf-8') as f: - DATASET_CONFIGS = yaml.safe_load(f) - - -# ============================================================ -# Middle-truncation helpers -# ============================================================ - -_TOKENIZER = None - - -def get_tokenizer(): - global _TOKENIZER - if _TOKENIZER is None: - from transformers import AutoTokenizer - try: - # Try local path first - _TOKENIZER = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True) - except Exception: - # Fallback to HuggingFace model ID - _TOKENIZER = AutoTokenizer.from_pretrained('deepseek-ai/DeepSeek-V4-Flash', trust_remote_code=True) - return _TOKENIZER - - -def truncate_middle(text: str, max_tokens: int) -> str: - if max_tokens <= 0: - return text - tokenizer = get_tokenizer() - token_ids = tokenizer.encode(text, add_special_tokens=False) - if len(token_ids) <= max_tokens: - return text - keep_head = max_tokens // 2 - keep_tail = max_tokens - keep_head - truncated_ids = token_ids[:keep_head] + token_ids[-keep_tail:] - return tokenizer.decode(truncated_ids, skip_special_tokens=True) - - -def _patch_adapters_for_truncation(): - from evalscope.benchmarks.longbench_v2.longbench_v2_adapter import LongBenchV2Adapter - from evalscope.benchmarks.openai_mrcr.openai_mrcr_adapter import OpenAIMRCRAdapter - - # Patch LongBenchV2Adapter.format_prompt_template - _orig_longbench_format = LongBenchV2Adapter.format_prompt_template - def _patched_longbench_format(self, sample): - max_tok = TRUNCATION_CONFIG.get('longbench_v2') - if max_tok and sample.metadata and 'context' in sample.metadata: - sample.metadata['context'] = truncate_middle(sample.metadata['context'], max_tok) - return _orig_longbench_format(self, sample) - LongBenchV2Adapter.format_prompt_template = _patched_longbench_format - - # Patch OpenAIMRCRAdapter.record_to_sample with needle-aware truncation. - # MRCR is a long chat history with needles hidden at desired_msg_index. - # We keep the head, tail, and a window around the needle, and truncate - # each kept message if it is still too long. This preserves the retrieval - # task while fitting GPU memory. - _orig_mrcr_record = OpenAIMRCRAdapter.record_to_sample - def _patched_mrcr_record(self, record): - import json - max_total_tok = TRUNCATION_CONFIG.get('openai_mrcr') - per_msg_max_tok = 8192 - if max_total_tok and 'prompt' in record: - try: - prompt_data = json.loads(record['prompt']) - if not isinstance(prompt_data, list) or len(prompt_data) == 0: - return _orig_mrcr_record(self, record) - - tokenizer = get_tokenizer() - total_tok = sum( - len(tokenizer.encode(msg.get('content', '') if isinstance(msg, dict) else '', add_special_tokens=False)) - for msg in prompt_data - ) - if total_tok <= max_total_tok: - return _orig_mrcr_record(self, record) - - desired_idx = record.get('desired_msg_index', 0) - if not isinstance(desired_idx, int) or desired_idx < 0 or desired_idx >= len(prompt_data): - desired_idx = 0 - - n = len(prompt_data) - keep = set() - # Head and tail context - keep.update(range(min(2, n))) - keep.update(range(max(0, n - 2), n)) - # Window around the needle - window = 2 - keep.update(range(max(0, desired_idx - window), min(n, desired_idx + window + 1))) - keep = sorted(keep) - - new_prompt = [] - for idx in keep: - msg = prompt_data[idx] - if isinstance(msg, dict): - msg = dict(msg) - content = msg.get('content', '') - if len(tokenizer.encode(content, add_special_tokens=False)) > per_msg_max_tok: - msg['content'] = truncate_middle(content, per_msg_max_tok) - new_prompt.append(msg) - - record = dict(record) - record['prompt'] = json.dumps(new_prompt) - except (json.JSONDecodeError, TypeError): - pass - return _orig_mrcr_record(self, record) - OpenAIMRCRAdapter.record_to_sample = _patched_mrcr_record - - -_patch_adapters_for_truncation() - - -# ============================================================ -# Helpers -# ============================================================ - -def configure_thinking(generation_config: dict, enable: bool) -> dict: - extra_body = generation_config.get('extra_body', {}) - chat_template_kwargs = extra_body.get('chat_template_kwargs', {}) - if enable: - chat_template_kwargs['thinking'] = True - else: - chat_template_kwargs.pop('thinking', None) - if chat_template_kwargs: - extra_body['chat_template_kwargs'] = chat_template_kwargs - if extra_body: - generation_config['extra_body'] = extra_body - return generation_config - - -def build_agent_config(agent_cfg: dict) -> NativeAgentConfig: - agent_cfg = deepcopy(agent_cfg or {}) - known_fields = {'mode', 'strategy', 'tools', 'max_steps', 'mcp_servers', 'environment', 'environment_extra'} - kwargs = agent_cfg.pop('kwargs', {}) - for key in list(agent_cfg.keys()): - if key not in known_fields: - kwargs[key] = agent_cfg.pop(key) - if kwargs: - agent_cfg['kwargs'] = kwargs - return NativeAgentConfig(**agent_cfg) - - -def build_task_config(dataset_name: str, ds_cfg: dict, batch_size: int, enable_thinking: bool, seed: int, run_idx: int = 0) -> TaskConfig: - # Each run gets its own work_dir so use_cache does not reuse predictions - # across repeated samples. This is required for temperature=1.0 multi-run - # benchmarks to actually measure variance. - if run_idx > 0: - work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}_run_{run_idx}' - else: - work_dir = Path(OUTPUT_DIR) / dataset_name / f'seed_{seed}' - work_dir.mkdir(parents=True, exist_ok=True) - work_dir = str(work_dir) - - generation_config = configure_thinking(deepcopy(ds_cfg['generation_config']), enable_thinking) - dataset_args = deepcopy(ds_cfg.get('dataset_args', {})) - dataset_args.setdefault('shuffle', True) - - if dataset_name in MATH_DATASETS: - dataset_args['prompt_template'] = MATH_PROMPT_TEMPLATE - - dataset_args_dict = {dataset_name: dataset_args} - - agent_config = None - if 'agent_config' in ds_cfg: - agent_config = build_agent_config(ds_cfg['agent_config']) - - return TaskConfig( - model=MODEL, - api_url=API_URL, - eval_type='openai_api', - dataset_dir=DATASET_DIR, - judge_model_args=JUDGE_MODEL_ARGS, - seed=seed, - limit=LIMIT, - collect_perf=True, - no_timestamp=True, - work_dir=work_dir, - use_cache=work_dir, - datasets=[dataset_name], - generation_config=generation_config, - dataset_args=dataset_args_dict, - agent_config=agent_config, - eval_batch_size=batch_size, - sandbox=SandboxTaskConfig( - enabled=True, - engine='docker', - default_config=SANDBOX_CONFIGS.get(dataset_name, { - 'image': 'python:3.11-slim', - 'tools_config': { - 'shell_executor': {}, - 'python_executor': {} - } - }) - ) if dataset_name in SANDBOX_DATASETS else None, - ) - - -# ============================================================ -# Main loop -# ============================================================ - -def main(): - print(f"Config: {CONFIG_PATH}") - print(f"Model: {MODEL}") - print(f"API URL: {API_URL}") - print(f"Dataset Dir: {DATASET_DIR}") - print(f"Output Dir: {OUTPUT_DIR}") - print(f"Limit: {LIMIT if LIMIT is not None else 'ALL'}") - print(f"Total benchmarks: {len(DATASETS)}") - print("="*60) - - for batch_size in BATCH_SIZE_LIST: - # 1. Run multi-run benchmarks (temperature=1.0, same seed, variance from sampling) - for dataset_name in _multi_run_order: - if dataset_name not in DATASET_CONFIGS: - print(f"WARNING: {dataset_name} not in YAML config, skipping") - continue - ds_cfg = DATASET_CONFIGS[dataset_name] - num_runs = MULTI_RUN_CONFIG.get(dataset_name, 1) - for run_idx in range(num_runs): - print(f"\n{'='*60}") - print(f"Running: {dataset_name} (run {run_idx + 1}/{num_runs}, seed={SEED})") - print(f"{'='*60}") - task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED, run_idx=run_idx) - try: - run_task(task_cfg) - except Exception as e: - print(f"ERROR in {dataset_name} (run {run_idx + 1}): {e}") - continue - - # 2. Run single-run benchmarks (temperature=0.0 greedy) - for dataset_name in _single_run_order: - if dataset_name not in DATASET_CONFIGS: - print(f"WARNING: {dataset_name} not in YAML config, skipping") - continue - ds_cfg = DATASET_CONFIGS[dataset_name] - print(f"\n{'='*60}") - print(f"Running: {dataset_name} (seed={SEED})") - print(f"{'='*60}") - task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED) - try: - run_task(task_cfg) - except Exception as e: - print(f"ERROR in {dataset_name}: {e}") - continue - - # 3. Run agent benchmarks - for dataset_name in _agent_order: - if dataset_name not in DATASET_CONFIGS: - print(f"WARNING: {dataset_name} not in YAML config, skipping") - continue - ds_cfg = DATASET_CONFIGS[dataset_name] - print(f"\n{'='*60}") - print(f"Running: {dataset_name} (seed={SEED})") - print(f"{'='*60}") - task_cfg = build_task_config(dataset_name, ds_cfg, batch_size, ENABLE_THINKING, SEED) - try: - run_task(task_cfg) - except Exception as e: - print(f"ERROR in {dataset_name}: {e}") - continue - - print("\nAll benchmarks done!") - - -if __name__ == '__main__': - main() diff --git a/bash/config/dpv4-int8_nothinking.yaml b/config/dpv4-int8_nothinking.yaml similarity index 100% rename from bash/config/dpv4-int8_nothinking.yaml rename to config/dpv4-int8_nothinking.yaml diff --git a/scripts/run_docker_full.sh b/scripts/run_docker_full.sh new file mode 100755 index 0000000..4a1504f --- /dev/null +++ b/scripts/run_docker_full.sh @@ -0,0 +1,26 @@ +#!/bin/bash +# Docker 全量评测(Full) +# 请根据实际机器修改 MODEL / API_URL / HOST_EVALSCOPE 路径 + +MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}" +API_URL="${API_URL:-http://localhost:30000/v1}" +HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}" +OUTPUT_DIR="${OUTPUT_DIR:-/opt/evalscope/output_full}" + +docker run -it --rm \ + --network host \ + -v "${HOST_EVALSCOPE}:/opt/evalscope" \ + -v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \ + -v "${HOST_EVALSCOPE}/output_full:/opt/evalscope/output_full" \ + -v /var/run/docker.sock:/var/run/docker.sock \ + evalscope-complete-py312:latest \ + bash -c " + cd /opt/evalscope && + python bash/run.py \ + --model ${MODEL} \ + --api-url ${API_URL} \ + --dataset-dir /opt/evalscope \ + --output-dir ${OUTPUT_DIR} \ + --suite full \ + --limit none + " diff --git a/scripts/run_docker_group.sh b/scripts/run_docker_group.sh new file mode 100755 index 0000000..c68ed2c --- /dev/null +++ b/scripts/run_docker_group.sh @@ -0,0 +1,27 @@ +#!/bin/bash +# Docker 多机分组评测 +# 用法:GROUP=1 ./run_docker_group.sh + +GROUP="${GROUP:-1}" +MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}" +API_URL="${API_URL:-http://localhost:30000/v1}" +HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}" +OUTPUT_DIR="/opt/evalscope/output_group${GROUP}" + +docker run -it --rm \ + --network host \ + -v "${HOST_EVALSCOPE}:/opt/evalscope" \ + -v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \ + -v "${HOST_EVALSCOPE}/output_group${GROUP}:/opt/evalscope/output_group${GROUP}" \ + -v /var/run/docker.sock:/var/run/docker.sock \ + evalscope-complete-py312:latest \ + bash -c " + cd /opt/evalscope && + python bash/run.py \ + --model ${MODEL} \ + --api-url ${API_URL} \ + --dataset-dir /opt/evalscope \ + --output-dir ${OUTPUT_DIR} \ + --suite group${GROUP} \ + --limit none + " diff --git a/scripts/run_docker_lite.sh b/scripts/run_docker_lite.sh new file mode 100755 index 0000000..79deb85 --- /dev/null +++ b/scripts/run_docker_lite.sh @@ -0,0 +1,25 @@ +#!/bin/bash +# Docker Lite 快速冒烟 +# 方式一:使用 --suite lite + +MODEL="${MODEL:-DeepSeek-V4-Flash-Int8}" +API_URL="${API_URL:-http://localhost:30000/v1}" +HOST_EVALSCOPE="${HOST_EVALSCOPE:-/data1/sora/evalscope}" + +docker run -it --rm \ + --network host \ + -v "${HOST_EVALSCOPE}:/opt/evalscope" \ + -v "${HOST_EVALSCOPE}/datasets:/opt/evalscope/datasets" \ + -v "${HOST_EVALSCOPE}/output_lite:/opt/evalscope/output_lite" \ + -v /var/run/docker.sock:/var/run/docker.sock \ + evalscope-complete-py312:latest \ + bash -c " + cd /opt/evalscope && + python bash/run.py \ + --model ${MODEL} \ + --api-url ${API_URL} \ + --dataset-dir /opt/evalscope \ + --output-dir /opt/evalscope/output_lite \ + --suite lite \ + --limit none + "