#!/usr/bin/env python3 """ 诊断 swe_bench_verified 镜像准备阶段卡在哪里。 用法: python bash/diagnose_swe_images.py 它会模拟 evalscope 内部 build_images() 的前半段: 1. 加载数据集 2. 生成 500 个 instance 的镜像名 3. 列出本地 Docker 镜像 4. 对比哪些已存在、哪些缺失 5. 打印每一步耗时 """ import os import sys import time from pathlib import Path # 让 evalscope 可导入 PROJECT_ROOT = Path(__file__).parent.parent.resolve() sys.path.insert(0, str(PROJECT_ROOT / 'evalscope')) os.environ.setdefault('USE_MODELSCOPE_HUB', '1') os.environ.setdefault('HF_ENDPOINT', 'https://hf-mirror.com') os.environ.setdefault('PYTHONUNBUFFERED', '1') from evalscope.api.dataset import RemoteDataLoader, FieldSpec from evalscope.benchmarks.swe_bench.utils import resolve_swebench_arch from swebench.harness.test_spec.test_spec import make_test_spec def get_local_images(): import subprocess result = subprocess.run( ['docker', 'images', '--format', '{{.Repository}}:{{.Tag}}'], capture_output=True, text=True ) if result.returncode != 0: print(f'ERROR: docker images failed: {result.stderr}') return set() return {line.strip() for line in result.stdout.splitlines() if line.strip()} def main(): t0 = time.time() print('==> 1. 加载 SWE-bench_Verified 数据集', flush=True) loader = RemoteDataLoader( data_id_or_path='princeton-nlp/SWE-bench_Verified', split='test', sample_fields=FieldSpec( input='problem_statement', metadata=[ 'instance_id', 'repo', 'base_commit', 'patch', 'PASS_TO_PASS', 'FAIL_TO_PASS', 'test_patch', 'version', 'environment_setup_commit', 'hints_text', 'created_at', ], ), ) samples = loader.load() print(f' 样本数: {len(samples)}, 耗时: {time.time()-t0:.2f}s', flush=True) t1 = time.time() print('==> 2. 生成 500 个镜像名 (make_test_spec)', flush=True) missing_repo = 0 image_names = [] last_log = time.time() for idx, s in enumerate(samples): metadata = s.metadata if hasattr(s, 'metadata') else s instance_id = metadata.get('instance_id') try: arch = resolve_swebench_arch(instance_id, '') spec = make_test_spec(metadata, namespace='swebench', arch=arch) image_names.append(spec.instance_image_key) except KeyError as e: missing_repo += 1 if missing_repo <= 3: print(f' WARNING: instance {instance_id} 缺少字段: {e}', flush=True) if time.time() - last_log > 10: print(f' 进度: {idx+1}/{len(samples)} ...', flush=True) last_log = time.time() print(f' 生成镜像名: {len(image_names)}, 缺失字段: {missing_repo}, 耗时: {time.time()-t1:.2f}s', flush=True) t2 = time.time() print('==> 3. 扫描本地 Docker 镜像', flush=True) local_images = get_local_images() print(f' 本地镜像总数: {len(local_images)}, 耗时: {time.time()-t2:.2f}s', flush=True) t3 = time.time() print('==> 4. 对比缺失镜像', flush=True) needed = set(image_names) present = needed & local_images missing = needed - local_images print(f' 需要: {len(needed)}, 已存在: {len(present)}, 缺失: {len(missing)}, 耗时: {time.time()-t3:.2f}s', flush=True) if missing: print('\n前 10 个缺失镜像:', flush=True) for name in list(missing)[:10]: print(f' {name}', flush=True) else: print('\n所有需要镜像均已存在,evalscope 不应该再拉取/构建。', flush=True) print(f'\n总耗时: {time.time()-t0:.2f}s', flush=True) if __name__ == '__main__': main()