evalstone/evalscope/tests/report/test_combinator.py
sora 13274243a0 Bump vendored EvalScope and add K3-ready DPV4 configs.
Keep K3 suite selection and report-schema scoring in bash, merge K3/vision dataset_args into dpv4 yamls, and pin EvalScope at 735d920ee911 with local patches.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-09-02 07:30:48 +00:00

200 lines
7.6 KiB
Python

import json
from typing import List, Optional
from evalscope.api.metric import AggScore
from evalscope.api.metric.semantics import MetricIdentity, MetricSelector
from evalscope.metrics.semantics import get_semantics_resolver
from evalscope.metrics.semantics.identity import migrate_legacy_identity
from evalscope.report import gen_table, get_display_data_frame, get_report_list
from evalscope.report.generator import ReportGenerator
from evalscope.report.report import Category, Metric, Report, Subset
class _StubAdapter:
def __init__(self, name: str, primary_metric: Optional[str] = None, aggregation: str = 'mean') -> None:
self.name = name
self.primary_metric = MetricSelector(
name=migrate_legacy_identity(primary_metric, aggregation).name
) if primary_metric else None
self.aggregation = aggregation
self.pretty_name = name
self.description = ''
self.category_map = {}
def _report(benchmark: str, scores: List[AggScore], primary_metric: Optional[str] = None) -> Report:
return ReportGenerator.generate_report(
score_dict={'default': scores},
model_name='test-model',
data_adapter=_StubAdapter(benchmark, primary_metric=primary_metric),
)
def _categorized_report(category_name: tuple[str, ...]) -> Report:
report = _report('gsm8k', [AggScore(score=0.8, metric_name='accuracy', aggregation='mean', num=5)])
report.metrics[0].categories = [Category(name=category_name, subsets=[Subset(name='main', score=0.8, num=5)])]
return report
def test_get_report_list_skips_non_report_json(tmp_path):
reports_dir = tmp_path / 'reports'
report_file = reports_dir / 'qwen-plus' / 'gdpval.json'
submission_info_file = reports_dir / 'qwen-plus' / 'gdpval_submission' / 'submission_info.json'
empty_report_like_file = reports_dir / 'qwen-plus' / 'empty_report_like.json'
report = Report(
name='qwen-plus@gdpval',
dataset_name='gdpval',
model_name='qwen-plus',
metrics=[
Metric(
identity=MetricIdentity(name='submission_ready', aggregation='mean'),
semantics=get_semantics_resolver().resolve(
'gdpval', MetricIdentity(name='submission_ready', aggregation='mean')
).semantics,
categories=[Category(
name=('default', ),
subsets=[Subset(name='default', score=0.8679, num=1)],
)],
)
],
)
report.to_json(str(report_file))
submission_info_file.parent.mkdir(parents=True, exist_ok=True)
with open(submission_info_file, 'w', encoding='utf-8') as f:
json.dump({'benchmark': 'gdpval', 'samples': [{'id': 'case-1'}]}, f)
with open(empty_report_like_file, 'w', encoding='utf-8') as f:
json.dump({'name': 'empty', 'dataset_name': 'empty', 'model_name': 'qwen-plus', 'metrics': []}, f)
reports = get_report_list([str(reports_dir)])
assert len(reports) == 1
assert reports[0].dataset_name == 'gdpval'
assert reports[0].model_name == 'qwen-plus'
table = gen_table(reports_path_list=[str(reports_dir)], add_overall_metric=True)
assert 'gdpval' in table
assert 'qwen-plus' in table
assert 'default_dataset' not in table
def test_gen_table_formats_metric_values_without_changing_dataframe_scores() -> None:
accuracy = _report('gsm8k', [AggScore(score=0.8567, metric_name='accuracy', aggregation='mean', num=100)])
accuracy.dataset_pretty_name = 'GSM8K Pretty'
wer = _report(
'torgo', [AggScore(score=0.0432, metric_name='wer', aggregation='identity', num=50)], primary_metric='wer'
)
cider = _report('cider', [AggScore(score=1.23456, metric_name='cider', aggregation='mean', num=10)])
diagnostic = _report(
'third_party', [AggScore(score=0.87654321, metric_name='mystery', aggregation='identity', num=2)]
)
table = gen_table(report_list=[accuracy, wer, cider, diagnostic])
assert 'Accuracy ↑' in table
assert 'GSM8K Pretty' in table
assert '85.7%' in table
assert 'WER ↓' in table
assert '4.3%' in table
assert 'CIDEr ↑' in table
assert '1.235' in table
assert 'mystery' in table
assert '0.8765' in table
assert '0.8567' not in table
assert accuracy.to_dataframe()['Score'].tolist() == [0.8567]
def test_display_data_frame_uses_the_same_semantics_as_the_cli_table() -> None:
report = _report('gsm8k', [AggScore(score=0.8567, metric_name='accuracy', aggregation='mean', num=100)])
display_table = get_display_data_frame([report])
assert display_table['Metric'].tolist() == ['Accuracy ↑']
assert display_table['Score'].tolist() == ['85.7%']
assert report.to_dataframe()['Score'].tolist() == [0.8567]
def test_gen_table_disambiguates_repeated_metric_display_names() -> None:
report = _report(
'plawbench',
[
AggScore(score=0.72, metric_name='accuracy', aggregation='mean', num=30),
AggScore(score=0.75, metric_name='fact_acc', aggregation='mean', num=30),
],
primary_metric='accuracy',
)
table = gen_table(report_list=[report])
assert 'Accuracy ↑ (accuracy:mean)' in table
assert 'Accuracy ↑ (fact_acc:mean)' in table
def test_gen_table_adds_overall_row_for_each_metric() -> None:
adapter = _StubAdapter('multi_metric', primary_metric='accuracy')
report = ReportGenerator.generate_report(
score_dict={
'first': [
AggScore(score=0.5, metric_name='accuracy', aggregation='mean', num=2),
AggScore(score=0.25, metric_name='f1', aggregation='mean', num=2),
],
'second': [
AggScore(score=1.0, metric_name='accuracy', aggregation='mean', num=1),
AggScore(score=1.0, metric_name='f1', aggregation='mean', num=1),
],
},
model_name='test-model',
data_adapter=adapter,
)
table = report.to_dataframe(add_overall_metric=True)
overall_rows = table[table['Subset'] == 'OVERALL']
assert overall_rows['Metric'].tolist() == ['accuracy:mean', 'f1:mean']
assert overall_rows['Score'].tolist() == [0.6667, 0.5]
def test_gen_table_hides_placeholder_category_columns_without_changing_dataframe() -> None:
report = _categorized_report(('default', ))
raw_dataframe = report.to_dataframe()
table = gen_table(report_list=[report])
assert 'Cat.0' not in table
assert 'Category' not in table
assert 'default' not in table
assert raw_dataframe['Cat.0'].tolist() == ['default']
def test_gen_table_renames_one_informative_category_level() -> None:
table = gen_table(report_list=[_categorized_report(('math', ))])
assert 'Category' in table
assert 'Cat.0' not in table
assert 'math' in table
def test_gen_table_numbers_multiple_informative_category_levels() -> None:
table = gen_table(report_list=[_categorized_report(('knowledge', 'math'))])
assert 'Category 1' in table
assert 'Category 2' in table
assert 'Cat.0' not in table
assert 'Cat.1' not in table
assert 'knowledge' in table
assert 'math' in table
def test_gen_table_replaces_placeholder_categories_when_mixed_with_informative_values() -> None:
default_report = _categorized_report(('default', ))
categorized_report = _categorized_report(('math', ))
categorized_report.dataset_name = 'categorized'
table = gen_table(report_list=[default_report, categorized_report])
assert 'Category' in table
assert 'math' in table
assert 'default' not in table
assert default_report.to_dataframe()['Cat.0'].tolist() == ['default']