# bbh: knowledge/MCQ benchmark generation: temperature: 0.0 max_tokens: 32768 top_p: 1.0 run: limit_per_task: 10 concurrency: 8