EvalHarness/evalharness/config/dp4-nothink.yaml
sora 4d7567adf8 Flatten config: direct key-value per bench, no nested groups
Co-Authored-By: Claude <noreply@anthropic.com>
2026-09-11 08:05:36 +00:00

52 lines
2.0 KiB
YAML
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# dp4-nothink 配置
# 用法: evalharness eval run <bench> --config dp4-nothink --api-url ... --model ...
# 优先级: DatasetSpec < default < <bench> < 命令行参数
default:
temperature: 0.0
max_tokens: 32768
top_p: 1.0
concurrency: 8
# ── 数学temp=1跑 12 遍取均值)──
aime24: {temperature: 1.0, max_tokens: 8192, repeats: 12, concurrency: 4}
aime25: {temperature: 1.0, max_tokens: 8192, repeats: 12, concurrency: 4}
aime26: {temperature: 1.0, max_tokens: 8192, repeats: 12, concurrency: 4}
hmmt26: {temperature: 1.0, max_tokens: 8192, repeats: 12, concurrency: 4}
gsm8k: {limit: 200}
competition_math: {limit_per_task: 40}
# ── 知识/选择题 ──
mmlu: {limit_per_task: 4}
cmmlu: {limit_per_task: 3}
mmlu_pro: {limit_per_task: 20}
arc: {limit: 200}
hellaswag: {limit: 400}
winogrande: {limit: 200}
bbh: {limit_per_task: 10}
gpqa_diamond: {temperature: 1.0, max_tokens: 8192, limit: 200}
# ── 问答 ──
simple_qa: {max_tokens: 1024, limit: 200}
trivia_qa: {limit: 200}
drop: {limit: 200}
# ── Judge 类 ──
hle: {limit_per_task: 25, judge: dp4-flash, judge_url: 'http://174.1.51.4:30000/v1'}
imo_answerbench: {temperature: 1.0, max_tokens: 8192, limit_per_task: 25, judge: dp4-flash, judge_url: 'http://174.1.51.4:30000/v1'}
# ── 代码docker 沙箱)──
humaneval: {temperature: 1.0, concurrency: 4}
bigcodebench: {limit: 200, concurrency: 4}
live_code_bench: {temperature: 1.0, limit: 200, concurrency: 4}
# ── 长上下文 ──
longbench_v2: {max_tokens: 8192, limit_per_task: 66, concurrency: 4, max_input_tokens: 120000}
openai_mrcr: {max_tokens: 8192, limit: 200, concurrency: 4, max_input_tokens: 120000}
# ── Agent ──
bfcl_v3: {max_tokens: 4096, limit_per_task: 20, env: bfcl_mock}
general_fc: {max_tokens: 4096, limit: 400}
tau2_bench: {max_tokens: 16384, env: tau2_official, max_turns: 50}
swe_bench_verified: {limit: 70, concurrency: 4}