Update thinking config auto-selection, deploy script, README, and SWE-bench utils

This commit is contained in:
sora 2026-07-29 06:37:36 +00:00
parent b307e01cbf
commit b585c1e814
7 changed files with 302 additions and 35 deletions

1
MARKER_TEST Normal file
View File

@ -0,0 +1 @@
MARKER_FROM_HOST

View File

@ -129,18 +129,15 @@ docker load -i /data1/sora/evalscope/docker/evalscope-complete-py312.tar.gz
# 5. 准备代码执行 sandbox 镜像humaneval / bigcodebench 必须) # 5. 准备代码执行 sandbox 镜像humaneval / bigcodebench 必须)
# 从 ModelScope 下载预打包的 sandbox 镜像(避免 DockerHub 限流/超时) # 从 ModelScope 下载预打包的 sandbox 镜像(避免 DockerHub 限流/超时)
python3 -c " python3 -c "
from modelscope_hub import HubApi from modelscope.hub.file_download import model_file_download
api = HubApi() model_file_download(
api.download_file( model_id='SoraAmami/evalscope-sandbox-images',
repo_id='SoraAmami/evalscope-sandbox-images', file_path='bigcodebench-sandbox.tar.gz',
repo_type='dataset',
path='bigcodebench-sandbox.tar.gz',
local_dir='/data1/sora/evalscope/docker/sandbox_images' local_dir='/data1/sora/evalscope/docker/sandbox_images'
) )
api.download_file( model_file_download(
repo_id='SoraAmami/evalscope-sandbox-images', model_id='SoraAmami/evalscope-sandbox-images',
repo_type='dataset', file_path='python-3.11-slim.tar.gz',
path='python-3.11-slim.tar.gz',
local_dir='/data1/sora/evalscope/docker/sandbox_images' local_dir='/data1/sora/evalscope/docker/sandbox_images'
) )
" "
@ -205,18 +202,15 @@ docker load -i /data1/sora/evalscope/docker/evalscope-complete-py312.tar.gz
# 7. 准备代码执行 sandbox 镜像 # 7. 准备代码执行 sandbox 镜像
# 从 ModelScope 下载预打包的 sandbox 镜像(避免 DockerHub 限流/超时) # 从 ModelScope 下载预打包的 sandbox 镜像(避免 DockerHub 限流/超时)
python3 -c " python3 -c "
from modelscope_hub import HubApi from modelscope.hub.file_download import model_file_download
api = HubApi() model_file_download(
api.download_file( model_id='SoraAmami/evalscope-sandbox-images',
repo_id='SoraAmami/evalscope-sandbox-images', file_path='bigcodebench-sandbox.tar.gz',
repo_type='dataset',
path='bigcodebench-sandbox.tar.gz',
local_dir='/data1/sora/evalscope/docker/sandbox_images' local_dir='/data1/sora/evalscope/docker/sandbox_images'
) )
api.download_file( model_file_download(
repo_id='SoraAmami/evalscope-sandbox-images', model_id='SoraAmami/evalscope-sandbox-images',
repo_type='dataset', file_path='python-3.11-slim.tar.gz',
path='python-3.11-slim.tar.gz',
local_dir='/data1/sora/evalscope/docker/sandbox_images' local_dir='/data1/sora/evalscope/docker/sandbox_images'
) )
" "
@ -276,15 +270,15 @@ docker images | grep -E 'evalscope-complete-py312|bigcodebench-sandbox|python:3.
### 4.2 `official` 官方对比套件 ### 4.2 `official` 官方对比套件
对齐腾讯入库标准对比表核心子集 当前默认配置
| 类型 | Benchmark | | 类型 | Benchmark |
|------|-----------| |------|-----------|
| multi-run | `aime25`, `aime26`, `live_code_bench` | | multi-run | `aime25`, `aime26`, `live_code_bench` |
| single-run | `hle`, `mmlu_pro`, `gpqa_diamond`, `longbench_v2` | | single-run | `hle`, `mmlu_pro`, `gpqa_diamond`, `longbench_v2`, `swe_bench_verified` |
| agent | `tau2_bench` | | agent | `tau2_bench` |
> `swe_bench_verified` 默认注释掉,如需启用可修改 `bash/run.py``official` suite > **注意**`hle``tau2_bench` 需要 LLM-as-judge默认 `DeepSeek/DeepSeek-V4-Pro` @ Vectron。如果裁判 API 余额/配额不足,会报 `402`/`429`,需要充值或换用其他 judge 模型
预计时间:视模型速度和是否启用 swe 而定,通常 **数小时到数十小时** 预计时间:视模型速度和是否启用 swe 而定,通常 **数小时到数十小时**
@ -314,6 +308,35 @@ docker images | grep -E 'evalscope-complete-py312|bigcodebench-sandbox|python:3.
| `group2` | - | `hle`, `mmlu_pro`, `trivia_qa` | - | ~26h | | `group2` | - | `hle`, `mmlu_pro`, `trivia_qa` | - | ~26h |
| `group3` | `gpqa_diamond` | `openai_mrcr`, `longbench_v2`, `bfcl_v3`, `mmlu`, `cmmlu`, `bbh`, `simple_qa` | `tau2_bench`, `general_fc` | ~27h | | `group3` | `gpqa_diamond` | `openai_mrcr`, `longbench_v2`, `bfcl_v3`, `mmlu`, `cmmlu`, `bbh`, `simple_qa` | `tau2_bench`, `general_fc` | ~27h |
### 4.6 多机同时跑 `official` 套件示例
`official` 套件可在多台机器上并行执行,每台调用各自的本地模型服务:
```bash
# 01 跑 swe_bench_verified
python bash/run.py \
--api-url http://174.1.51.1:30000/v1 \
--datasets swe_bench_verified \
--folder-name DP4-flash-int8-thinking-add0 \
--thinking --max-tokens-add 0
# 02 跑 mmlu_pro / gpqa_diamond / longbench_v2
python bash/run.py \
--api-url http://174.1.51.2:30000/v1 \
--datasets mmlu_pro,gpqa_diamond,longbench_v2 \
--folder-name DP4-flash-int8-thinking-add0 \
--thinking --max-tokens-add 0
# 04 跑 hle需要可用 judge API否则会失败
python bash/run.py \
--api-url http://174.1.51.4:30000/v1 \
--datasets hle \
--folder-name DP4-flash-int8-thinking-add0 \
--thinking --max-tokens-add 0
```
跑完后所有机器的 `output/DP4-flash-int8-thinking-add0/` 可合并到同一目录,再运行 `bash/collect_results.py` 生成汇总表。
--- ---
## 6. Multi-run 配置 ## 6. Multi-run 配置
@ -395,7 +418,7 @@ for f in sorted(glob.glob('output/{folder_name}/*/seed_*/reports/*/*.json')):
conda create -n evalscope python=3.12 -y conda create -n evalscope python=3.12 -y
conda activate evalscope conda activate evalscope
cd /data1/sora/evalscope cd /data1/sora/evalscope
pip install -e ./evalscope pip install -e .
``` ```
### 7.2 运行单个 benchmark ### 7.2 运行单个 benchmark

View File

@ -652,7 +652,18 @@ def main():
truncation_tokens = args.truncation_tokens truncation_tokens = args.truncation_tokens
_patch_adapters_for_truncation(args.tokenizer_path, truncation_tokens) _patch_adapters_for_truncation(args.tokenizer_path, truncation_tokens)
dataset_configs = load_dataset_configs(args.config) # Auto-select thinking config if --thinking is enabled and user did not
# explicitly provide a different --config.
config_path = args.config
if enable_thinking and args.config == DEFAULT_CONFIG:
thinking_config = str(PROJECT_ROOT / 'config' / 'dpv4-int8_thinking.yaml')
if Path(thinking_config).exists():
config_path = thinking_config
print(f'Auto-selected thinking config: {config_path}')
else:
print(f'WARNING: thinking config not found at {thinking_config}, falling back to {args.config}')
dataset_configs = load_dataset_configs(config_path)
# Resolve top-level output folder name # Resolve top-level output folder name
if args.folder_name: if args.folder_name:

View File

@ -0,0 +1,205 @@
gpqa_diamond:
generation_config:
temperature: 1.0
top_p: 1.0
stream: true
max_tokens: 8192
hle:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 32768
aime24:
generation_config:
temperature: 1.0
top_p: 1.0
stream: true
max_tokens: 32768
aime25:
generation_config:
temperature: 1.0
top_p: 1.0
stream: true
max_tokens: 65536
mmlu_pro:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
simple_qa:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
arc:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
bbh:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 32768
live_code_bench:
generation_config:
temperature: 1.0
top_p: 1.0
stream: true
max_tokens: 65536
dataset_args:
subset_list:
- release_v6
aime26:
generation_config:
temperature: 1.0
top_p: 1.0
stream: true
max_tokens: 65536
hmmt26:
generation_config:
temperature: 1.0
top_p: 1.0
stream: true
max_tokens: 32768
imo_answerbench:
generation_config:
temperature: 1.0
top_p: 1.0
stream: true
max_tokens: 32768
super_gpqa:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
drop:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 32768
hellaswag:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
mmlu:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
openai_mrcr:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
bigcodebench:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 32768
humaneval:
generation_config:
temperature: 1.0
top_p: 1.0
stream: true
max_tokens: 32768
gsm8k:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 32768
competition_math:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 32768
cmmlu:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
trivia_qa:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
winogrande:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
longbench_v2:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 8192
dataset_args:
subset_list:
- short
- medium
- long
tau2_bench:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 16384
dataset_args:
extra_params:
user_model: deepseek-v4-pro
api_key: sk-9ed86ef546ca47e3afa7c3b014dea268
api_base: https://api.deepseek.com/v1
generation_config:
temperature: 0.0
max_tokens: 4096
agent_config:
mode: native
strategy: react
max_steps: 50
general_fc:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 4096
bfcl_v3:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 4096
parallel_tool_calls: true
dataset_args:
extra_params:
is_fc_model: true
underscore_to_dot: true
swe_bench_verified:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 32768
swe_bench_pro:
generation_config:
temperature: 0.0
top_p: 1.0
stream: true
max_tokens: 32768

View File

@ -83,6 +83,8 @@ def eval_instance(
# Build + start instance container (instance image should already be built) # Build + start instance container (instance image should already be built)
container = None container = None
client = None
instance_id = None
eval_completed = False eval_completed = False
report = {} report = {}
try: try:
@ -198,15 +200,20 @@ def eval_instance(
f"Result for {instance_id}: resolved: {report[instance_id]['resolved']}") f"Result for {instance_id}: resolved: {report[instance_id]['resolved']}")
eval_completed = True eval_completed = True
except Exception as e: except Exception as e:
error_msg = (f'Error in evaluating model for {instance_id}: {e}\n' display_id = instance_id or 'unknown'
error_msg = (f'Error in evaluating model for {display_id}: {e}\n'
f'{traceback.format_exc()}') f'{traceback.format_exc()}')
logger.error(error_msg) logger.error(error_msg)
report['error'] = error_msg report['error'] = error_msg
finally: finally:
# Remove instance container # Remove instance container
cleanup_container(client, container, logger) if client is not None:
cleanup_container(client, container, logger)
resolved = False
if instance_id is not None:
resolved = report.get(instance_id, {}).get('resolved', False)
return { return {
'completed': eval_completed, 'completed': eval_completed,
'resolved': report.get(instance_id, {}).get('resolved', False), 'resolved': resolved,
'report': report, 'report': report,
} }

View File

@ -0,0 +1,23 @@
#!/bin/bash
# ============================================================
# 把本地新镜像 scp 到远程机器并加载
# 用法: bash scripts/copy_image_to_machine.sh <IP>
# ============================================================
set -euo pipefail
IP="${1:?请提供远程机器 IP}"
PASS="${SSHPASS:-sskj2025}"
LOCAL_TAR="/data1/sora/evalscope/docker/evalscope-complete-py312.tar.gz"
REMOTE_TAR="/data1/sora/evalscope/docker/evalscope-complete-py312.tar.gz"
echo "==> [$IP] 复制镜像 tar 包..."
sshpass -p "$PASS" scp -o StrictHostKeyChecking=no "$LOCAL_TAR" "root@$IP:$REMOTE_TAR"
echo "==> [$IP] 加载 Docker 镜像..."
sshpass -p "$PASS" ssh -o StrictHostKeyChecking=no "root@$IP" "docker load -i $REMOTE_TAR"
echo "==> [$IP] 验证 torch..."
sshpass -p "$PASS" ssh -o StrictHostKeyChecking=no "root@$IP" "docker run --rm evalscope-complete-py312:latest python -c 'import torch; print(\"torch\", torch.__version__)'"
echo "==> [$IP] 完成"

View File

@ -237,14 +237,11 @@ download_sandbox_image_from_modelscope() {
return 0 return 0
fi fi
echo "==> 从 ModelScope 下载 $file ..." echo "==> 从 ModelScope 下载 $file ..."
# 优先使用新版 modelscope_hub API支持 dataset repo
python3 -c " python3 -c "
from modelscope_hub import HubApi from modelscope.hub.file_download import model_file_download
api = HubApi() model_file_download(
api.download_file( model_id='SoraAmami/evalscope-sandbox-images',
repo_id='SoraAmami/evalscope-sandbox-images', file_path='$file',
repo_type='dataset',
path='$file',
local_dir='$SANDBOX_IMAGES_DIR' local_dir='$SANDBOX_IMAGES_DIR'
) )
" && return 0 " && return 0