From 4d7567adf8119b01311b96aff8c6b8b1a6b4395c Mon Sep 17 00:00:00 2001 From: sora <2075279110@qq.com> Date: Fri, 11 Sep 2026 08:05:36 +0000 Subject: [PATCH] Flatten config: direct key-value per bench, no nested groups Co-Authored-By: Claude --- evalharness/config/dp4-nothink.yaml | 152 ++++++++-------------------- 1 file changed, 42 insertions(+), 110 deletions(-) diff --git a/evalharness/config/dp4-nothink.yaml b/evalharness/config/dp4-nothink.yaml index 19caaad..a88a79f 100644 --- a/evalharness/config/dp4-nothink.yaml +++ b/evalharness/config/dp4-nothink.yaml @@ -1,119 +1,51 @@ -# dp4-nothink: DeepSeek-V4-Flash nothinking protocol -# Usage: evalharness eval run --config dp4-nothink --api-url ... --model ... -# Priority: DatasetSpec.gen_config < default < < CLI args +# dp4-nothink 配置 +# 用法: evalharness eval run --config dp4-nothink --api-url ... --model ... +# 优先级: DatasetSpec < default < < 命令行参数 default: - generation: {temperature: 0.0, max_tokens: 32768, top_p: 1.0, stream: true} - run: {concurrency: 8, checkpoint: true, resume: true} + temperature: 0.0 + max_tokens: 32768 + top_p: 1.0 + concurrency: 8 -# ── Math (temp=1, 12 runs averaged) ── -aime24: - generation: {temperature: 1.0, max_tokens: 8192, top_p: 1.0} - run: {repeats: 12, concurrency: 4} +# ── 数学(temp=1,跑 12 遍取均值)── +aime24: {temperature: 1.0, max_tokens: 8192, repeats: 12, concurrency: 4} +aime25: {temperature: 1.0, max_tokens: 8192, repeats: 12, concurrency: 4} +aime26: {temperature: 1.0, max_tokens: 8192, repeats: 12, concurrency: 4} +hmmt26: {temperature: 1.0, max_tokens: 8192, repeats: 12, concurrency: 4} +gsm8k: {limit: 200} +competition_math: {limit_per_task: 40} -aime25: - generation: {temperature: 1.0, max_tokens: 8192, top_p: 1.0} - run: {repeats: 12, concurrency: 4} +# ── 知识/选择题 ── +mmlu: {limit_per_task: 4} +cmmlu: {limit_per_task: 3} +mmlu_pro: {limit_per_task: 20} +arc: {limit: 200} +hellaswag: {limit: 400} +winogrande: {limit: 200} +bbh: {limit_per_task: 10} +gpqa_diamond: {temperature: 1.0, max_tokens: 8192, limit: 200} -aime26: - generation: {temperature: 1.0, max_tokens: 8192, top_p: 1.0} - run: {repeats: 12, concurrency: 4} +# ── 问答 ── +simple_qa: {max_tokens: 1024, limit: 200} +trivia_qa: {limit: 200} +drop: {limit: 200} -hmmt26: - generation: {temperature: 1.0, max_tokens: 8192, top_p: 1.0} - run: {repeats: 12, concurrency: 4} +# ── Judge 类 ── +hle: {limit_per_task: 25, judge: dp4-flash, judge_url: 'http://174.1.51.4:30000/v1'} +imo_answerbench: {temperature: 1.0, max_tokens: 8192, limit_per_task: 25, judge: dp4-flash, judge_url: 'http://174.1.51.4:30000/v1'} -competition_math: - generation: {temperature: 0.0, max_tokens: 32768, top_p: 1.0} - run: {limit_per_task: 40, concurrency: 8} +# ── 代码(docker 沙箱)── +humaneval: {temperature: 1.0, concurrency: 4} +bigcodebench: {limit: 200, concurrency: 4} +live_code_bench: {temperature: 1.0, limit: 200, concurrency: 4} -gsm8k: - generation: {temperature: 0.0, max_tokens: 32768, top_p: 1.0} - run: {limit: 200, concurrency: 8} +# ── 长上下文 ── +longbench_v2: {max_tokens: 8192, limit_per_task: 66, concurrency: 4, max_input_tokens: 120000} +openai_mrcr: {max_tokens: 8192, limit: 200, concurrency: 4, max_input_tokens: 120000} -# ── Knowledge / MCQ ── -mmlu: - run: {limit_per_task: 4, concurrency: 8} - -cmmlu: - run: {limit_per_task: 3, concurrency: 8} - -mmlu_pro: - run: {limit_per_task: 20, concurrency: 8} - -arc: - run: {limit: 200, concurrency: 8} - -hellaswag: - run: {limit: 400, concurrency: 8} - -winogrande: - run: {limit: 200, concurrency: 8} - -bbh: - generation: {temperature: 0.0, max_tokens: 32768, top_p: 1.0} - run: {limit_per_task: 10, concurrency: 8} - -gpqa_diamond: - generation: {temperature: 1.0, max_tokens: 8192, top_p: 1.0} - run: {limit: 200, concurrency: 8} - -# ── QA ── -simple_qa: - generation: {temperature: 0.0, max_tokens: 1024, top_p: 1.0} - run: {limit: 200, concurrency: 8} - -trivia_qa: - run: {limit: 200, concurrency: 8} - -drop: - run: {limit: 200, concurrency: 8} - -# ── Judged ── -hle: - generation: {temperature: 0.0, max_tokens: 32768, top_p: 1.0} - run: {limit_per_task: 25, concurrency: 8} - judge: {model: dp4-flash, api_url: 'http://174.1.51.4:30000/v1'} - -imo_answerbench: - generation: {temperature: 1.0, max_tokens: 8192, top_p: 1.0} - run: {limit_per_task: 25, concurrency: 8} - judge: {model: dp4-flash, api_url: 'http://174.1.51.4:30000/v1'} - -# ── Code (docker sandbox) ── -humaneval: - generation: {temperature: 1.0, max_tokens: 32768, top_p: 1.0} - run: {concurrency: 4} - -bigcodebench: - generation: {temperature: 0.0, max_tokens: 32768, top_p: 1.0} - run: {limit: 200, concurrency: 4} - -live_code_bench: - generation: {temperature: 1.0, max_tokens: 32768, top_p: 1.0} - run: {limit: 200, concurrency: 4} - -# ── Long context ── -longbench_v2: - generation: {temperature: 0.0, max_tokens: 8192, top_p: 1.0} - run: {limit_per_task: 66, concurrency: 4, max_input_tokens: 120000} - -openai_mrcr: - generation: {temperature: 0.0, max_tokens: 8192, top_p: 1.0} - run: {limit: 200, concurrency: 4, max_input_tokens: 120000} - -# ── Agent / tools ── -bfcl_v3: - generation: {temperature: 0.0, max_tokens: 4096, top_p: 1.0, parallel_tool_calls: true} - run: {limit_per_task: 20, concurrency: 8, env: bfcl_mock} - -general_fc: - generation: {temperature: 0.0, max_tokens: 4096, top_p: 1.0} - run: {limit: 400, concurrency: 8} - -tau2_bench: - generation: {temperature: 0.0, max_tokens: 16384, top_p: 1.0} - run: {concurrency: 8, env: tau2_official, max_turns: 50} - -swe_bench_verified: - run: {limit: 70, concurrency: 4} +# ── Agent ── +bfcl_v3: {max_tokens: 4096, limit_per_task: 20, env: bfcl_mock} +general_fc: {max_tokens: 4096, limit: 400} +tau2_bench: {max_tokens: 16384, env: tau2_official, max_turns: 50} +swe_bench_verified: {limit: 70, concurrency: 4}