EvalHarness/evalharness/data/datasets/imo_answerbench.py

33 lines
1.1 KiB
Python

"""IMO Answer Bench. Community-curated (evalscope); no official upstream release."""
from ..sample import Sample
from ..registry import register_dataset
from ..spec import DatasetSpec
@register_dataset(
DatasetSpec(
name='imo_answerbench',
source='evalscope/imo-answerbench', # curated; no official upstream release (ModelScope)
split='train', # the mirror ships a single train split
task_type='math',
tags=['math', 'competition', 'imo'],
description='IMO-level answer bench (community-curated, no official upstream).',
params={'hub': 'modelscope'},
)
)
def imo_answerbench():
def to_sample(record: dict) -> Sample:
return Sample(
input=record['Problem'],
target=str(record['Short Answer']).strip(),
metadata={
'id': record.get('Problem ID'),
'category': record.get('Category'),
'subcategory': record.get('Subcategory'),
'source': record.get('Source'),
},
)
return to_sample