sora 13274243a0 Bump vendored EvalScope and add K3-ready DPV4 configs.
Keep K3 suite selection and report-schema scoring in bash, merge K3/vision dataset_args into dpv4 yamls, and pin EvalScope at 735d920ee911 with local patches.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-09-02 07:30:48 +00:00

87 lines
3.8 KiB
Python

"""Unit tests for SLAKE answer parsing and normalization.
Scoring is normalized exact match, so a regression here fails silently: a mis-parsed or
over-normalized answer still looks like a plausible short answer and the run completes with a
wrong score.
"""
from evalscope.benchmarks.slake.slake_adapter import EN_PROMPT_TEMPLATE, ZH_PROMPT_TEMPLATE
from evalscope.benchmarks.slake.utils import normalize_answer, parse_answer
def test_parse_answer_reads_the_marker_line() -> None:
assert parse_answer('The scan is axial.\nANSWER: CT') == 'CT'
assert parse_answer('answer: 胸腔') == '胸腔'
assert parse_answer('ANSWER: "Lung"') == 'Lung'
def test_parse_answer_ignores_a_restated_instruction() -> None:
"""The prompt contains the marker, so an echoed instruction must not shadow the answer."""
prediction = f'{EN_PROMPT_TEMPLATE.format(question="Which organ is shown?")}\nANSWER: Lung'
assert parse_answer(prediction) == 'Lung'
def test_parse_answer_keeps_one_line() -> None:
assert parse_answer('ANSWER: Chest\n(the lower ribs are visible too)') == 'Chest'
def test_parse_answer_falls_back_to_the_whole_reply() -> None:
# Models frequently answer with the bare short answer the prompt asks for.
assert parse_answer('CT') == 'CT'
assert parse_answer('') == ''
assert parse_answer(None) == ''
def test_normalize_answer_ignores_case_and_punctuation() -> None:
assert normalize_answer('Lung.') == 'lung'
assert normalize_answer('CT (Computed Tomography)') == 'ct'
assert normalize_answer('腹部。') == '腹部'
assert normalize_answer('Lung, Spinal Cord') == normalize_answer('lung spinal cord')
def test_normalize_answer_unifies_yes_no_surface_forms() -> None:
# The Chinese references use several polarity spellings for the same closed-ended answer.
for yes in ['Yes', 'yes.', '是的', '', '', '包含', '可以', '存在']:
assert normalize_answer(yes) == 'yes', yes
for no in ['No', '不是', '', '没有', '不包含', '不可以', '不正常']:
assert normalize_answer(no) == 'no', no
def test_normalize_answer_unifies_xray_spellings() -> None:
# Modality references stay in English even for Chinese questions.
for spelling in ['X-Ray', 'x-ray', 'Xray', 'X光', 'X射线']:
assert normalize_answer(spelling) == 'xray', spelling
def test_normalize_answer_maps_word_numbers_to_digits() -> None:
"""Quantity references are Arabic digits, so word-form numbers must resolve to them.
The English mapping mirrors the official ``preprocess_answer``; the Chinese one is needed
because the Chinese prompt asks for a Chinese answer.
"""
assert normalize_answer('Two') == normalize_answer('2')
assert normalize_answer('none') == normalize_answer('0')
assert normalize_answer('两个') == normalize_answer('2')
assert normalize_answer('') == normalize_answer('2')
assert normalize_answer('3个') == normalize_answer('3')
# A counter word is only dropped after a number, so ordinary answers are untouched
assert normalize_answer('半个') == '半个'
assert normalize_answer('两侧') == '两侧'
def test_normalize_answer_drops_articles() -> None:
assert normalize_answer('The lung') == normalize_answer('Lung')
assert normalize_answer('a chest') == normalize_answer('chest')
def test_normalize_answer_keeps_distinct_answers_distinct() -> None:
assert normalize_answer('健康') != normalize_answer('是的')
assert normalize_answer('异常') != normalize_answer('不是')
assert normalize_answer('T2-weighted') != normalize_answer('T2')
def test_prompts_request_the_answer_marker() -> None:
for template in [EN_PROMPT_TEMPLATE, ZH_PROMPT_TEMPLATE]:
prompt = template.format(question='Which organ is shown?')
assert prompt.startswith('Which organ is shown?')
assert 'ANSWER:' in prompt