Config exactly matches evalscope dpv4-int8_nothinking.yaml (verified all match); concurrency removed (CLI controls it)

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
sora 2026-09-11 08:25:15 +00:00
parent 787a8a28e9
commit b8ed18771a

View File

@ -2,125 +2,87 @@ default:
temperature: 0.0
max_tokens: 32768
top_p: 1.0
concurrency: 8
aime24:
temperature: 1.0
max_tokens: 8192
repeats: 12
concurrency: 4
aime25:
temperature: 1.0
max_tokens: 8192
repeats: 12
concurrency: 4
aime26:
temperature: 1.0
max_tokens: 8192
repeats: 12
concurrency: 4
hmmt26:
temperature: 1.0
max_tokens: 8192
repeats: 12
concurrency: 4
gsm8k:
limit: 200
competition_math:
limit_per_task: 40
mmlu:
limit_per_task: 4
cmmlu:
limit_per_task: 3
mmlu_pro:
limit_per_task: 20
arc:
limit: 200
hellaswag:
limit: 400
winogrande:
limit: 200
bbh:
limit_per_task: 10
gpqa_diamond:
temperature: 1.0
max_tokens: 8192
limit: 200
simple_qa:
max_tokens: 1024
limit: 200
trivia_qa:
limit: 200
drop:
limit: 200
hle:
limit_per_task: 25
judge: dp4-flash
judge_url: http://174.1.51.4:30000/v1
imo_answerbench:
aime24:
temperature: 1.0
repeats: 12
aime25:
temperature: 1.0
repeats: 12
mmlu_pro:
max_tokens: 8192
limit_per_task: 20
simple_qa:
max_tokens: 8192
limit_per_task: 25
judge: dp4-flash
judge_url: http://174.1.51.4:30000/v1
humaneval:
temperature: 1.0
concurrency: 4
bigcodebench:
limit: 200
concurrency: 4
arc:
max_tokens: 8192
limit: 200
bbh:
limit_per_task: 10
live_code_bench:
temperature: 1.0
limit: 200
concurrency: 4
longbench_v2:
aime26:
temperature: 1.0
repeats: 12
hmmt26:
temperature: 1.0
repeats: 12
imo_answerbench:
temperature: 1.0
limit_per_task: 25
judge: dp4-flash
judge_url: http://174.1.51.4:30000/v1
drop:
limit: 200
hellaswag:
max_tokens: 8192
limit_per_task: 66
concurrency: 4
max_input_tokens: 120000
limit: 400
mmlu:
max_tokens: 8192
limit_per_task: 4
openai_mrcr:
max_tokens: 8192
limit: 200
concurrency: 4
max_input_tokens: 120000
bfcl_v3:
max_tokens: 4096
limit_per_task: 20
env: bfcl_mock
general_fc:
max_tokens: 4096
limit: 400
bigcodebench:
limit: 200
humaneval:
temperature: 1.0
gsm8k:
limit: 200
competition_math:
limit_per_task: 40
cmmlu:
max_tokens: 8192
limit_per_task: 3
trivia_qa:
max_tokens: 8192
limit: 200
winogrande:
max_tokens: 8192
limit: 200
longbench_v2:
max_tokens: 8192
limit_per_task: 66
max_input_tokens: 120000
tau2_bench:
max_tokens: 16384
env: tau2_official
max_turns: 50
general_fc:
max_tokens: 4096
limit: 400
bfcl_v3:
max_tokens: 4096
limit_per_task: 20
env: bfcl_mock
swe_bench_verified:
limit: 70
concurrency: 4