EvalHarness/evalharness/config/dp4-nothink.yaml
sora 787a8a28e9 Standard YAML indentation, no inline braces
Co-Authored-By: Claude <noreply@anthropic.com>
2026-09-11 08:09:09 +00:00

127 lines
1.6 KiB
YAML

default:
temperature: 0.0
max_tokens: 32768
top_p: 1.0
concurrency: 8
aime24:
temperature: 1.0
max_tokens: 8192
repeats: 12
concurrency: 4
aime25:
temperature: 1.0
max_tokens: 8192
repeats: 12
concurrency: 4
aime26:
temperature: 1.0
max_tokens: 8192
repeats: 12
concurrency: 4
hmmt26:
temperature: 1.0
max_tokens: 8192
repeats: 12
concurrency: 4
gsm8k:
limit: 200
competition_math:
limit_per_task: 40
mmlu:
limit_per_task: 4
cmmlu:
limit_per_task: 3
mmlu_pro:
limit_per_task: 20
arc:
limit: 200
hellaswag:
limit: 400
winogrande:
limit: 200
bbh:
limit_per_task: 10
gpqa_diamond:
temperature: 1.0
max_tokens: 8192
limit: 200
simple_qa:
max_tokens: 1024
limit: 200
trivia_qa:
limit: 200
drop:
limit: 200
hle:
limit_per_task: 25
judge: dp4-flash
judge_url: http://174.1.51.4:30000/v1
imo_answerbench:
temperature: 1.0
max_tokens: 8192
limit_per_task: 25
judge: dp4-flash
judge_url: http://174.1.51.4:30000/v1
humaneval:
temperature: 1.0
concurrency: 4
bigcodebench:
limit: 200
concurrency: 4
live_code_bench:
temperature: 1.0
limit: 200
concurrency: 4
longbench_v2:
max_tokens: 8192
limit_per_task: 66
concurrency: 4
max_input_tokens: 120000
openai_mrcr:
max_tokens: 8192
limit: 200
concurrency: 4
max_input_tokens: 120000
bfcl_v3:
max_tokens: 4096
limit_per_task: 20
env: bfcl_mock
general_fc:
max_tokens: 4096
limit: 400
tau2_bench:
max_tokens: 16384
env: tau2_official
max_turns: 50
swe_bench_verified:
limit: 70
concurrency: 4