From ba2b5c177293a28f7cc4a6eeaab201b51d481fd5 Mon Sep 17 00:00:00 2001 From: sora <2075279110@qq.com> Date: Wed, 5 Aug 2026 03:56:44 +0000 Subject: [PATCH] fix: terminal_bench_v2_1 result collection, apt mirror, and restore run.py env var --- bash/collect_results.py | 115 +++++++++++++++++- .../preload_terminal_bench_images.sh | 12 +- .../terminal_bench/terminal_bench_adapter.py | 55 +++++++++ 3 files changed, 174 insertions(+), 8 deletions(-) rename bash/{case => images_load}/preload_terminal_bench_images.sh (89%) diff --git a/bash/collect_results.py b/bash/collect_results.py index 56af21f..db771a5 100644 --- a/bash/collect_results.py +++ b/bash/collect_results.py @@ -53,6 +53,7 @@ BENCHMARK_DOMAIN = { 'tau2_bench': '智能体与工具', 'general_fc': '智能体与工具', 'bfcl_v3': '智能体与工具', + 'terminal_bench_v2_1': '智能体与工具', } # Column order matching the reference CSV @@ -121,6 +122,69 @@ def read_predictions_with_index(pred_file: Path): yield {'index': obj.get('index'), 'perf_metrics': pm} +def read_agent_perf_from_trajectory(pred_file: Path): + """Extract approximate perf metrics from agent trajectory files. + + Agent/sandbox benchmarks (e.g. terminal_bench_v2_1) do not record per-call + latency/TTFT/TPOT through EvalScope's model wrapper. However, the trial + trajectory contains step timestamps and final token counts. This function + yields synthetic ``{'index', 'perf_metrics'}`` records with: + + - ``latency``: wall-clock trial duration in seconds + - ``input_tokens``: total prompt tokens from the agent run + - ``output_tokens``: total completion tokens from the agent run + - ``ttft`` / ``tpot``: not available, left as None + """ + if not pred_file.exists(): + return + from datetime import datetime + with open(pred_file, 'r', encoding='utf-8') as f: + for line in f: + line = line.strip() + if not line: + continue + try: + obj = json.loads(line) + except json.JSONDecodeError: + continue + idx = obj.get('index') + model_output = obj.get('model_output', {}) + content = '' + choices = model_output.get('choices', []) + if choices and 'message' in choices[0]: + content = choices[0]['message'].get('content', '') + if not content or not isinstance(content, str) or not content.startswith('file://'): + continue + traj_path = Path(content[7:]) / 'agent' / 'trajectory.json' + if not traj_path.exists(): + continue + try: + traj = json.loads(traj_path.read_text(encoding='utf-8')) + except Exception: + continue + steps = traj.get('steps', []) + duration = None + if len(steps) >= 2: + try: + first_ts = datetime.fromisoformat(steps[0]['timestamp']) + last_ts = datetime.fromisoformat(steps[-1]['timestamp']) + duration = (last_ts - first_ts).total_seconds() + except Exception: + pass + final_metrics = traj.get('final_metrics', {}) + prompt_tokens = final_metrics.get('total_prompt_tokens') + completion_tokens = final_metrics.get('total_completion_tokens') + pm = {} + if duration is not None: + pm['latency'] = duration + if prompt_tokens is not None: + pm['input_tokens'] = int(prompt_tokens) + if completion_tokens is not None: + pm['output_tokens'] = int(completion_tokens) + if pm: + yield {'index': idx, 'perf_metrics': pm} + + def load_backup_summary(output_dir: Path, benchmark: str, model_name: str): """Load the durable ``perf_stats_backup/__.json``. @@ -280,8 +344,9 @@ def collect_benchmark(output_dir: Path, benchmark: str, model_name: str): data = json.loads(report.read_text(encoding='utf-8')) scores.append(extract_score(data)) if summary0 is None: - summary0 = data.get('perf_metrics', {}).get('summary', {}) - n_samples_unique = summary0.get('n_samples', 0) + perf_metrics = data.get('perf_metrics') or {} + summary0 = perf_metrics.get('summary', {}) + n_samples_unique = summary0.get('n_samples', data.get('num', 0)) except Exception: continue @@ -345,6 +410,27 @@ def collect_benchmark(output_dir: Path, benchmark: str, model_name: str): if otok is not None: output_tokens.append(int(otok)) + # Agent/sandbox benchmarks do not expose per-call perf metrics through + # EvalScope's model wrapper. Fall back to parsing the trial trajectory + # files (timestamps + final token counts) to get approximate latency and + # token usage. + if not latencies and pred_files: + for obj in read_agent_perf_from_trajectory(pred_files[0]): + idx = obj['index'] + key = ('trajectory', idx) + if idx is not None: + if key in seen_keys: + continue + seen_keys.add(key) + sample_indexes.append(idx) + pm = obj['perf_metrics'] + if pm.get('latency') is not None: + latencies.append(float(pm['latency'])) + if pm.get('input_tokens') is not None: + input_tokens.append(int(pm['input_tokens'])) + if pm.get('output_tokens') is not None: + output_tokens.append(int(pm['output_tokens'])) + # If we don't have raw predictions but have a perf_stats backup, that # represents a known-good summary captured right after a clean run — # preferable to summary0 (which may be the just-reset run). @@ -371,7 +457,8 @@ def collect_benchmark(output_dir: Path, benchmark: str, model_name: str): # reliable than the (possibly reset) report summary's n_samples. # For multi-seed / multi-run benchmarks we report the total number of # evaluated predictions (all seeds/runs) rather than unique problem IDs. - n_samples_unique = len(latencies) + if n_samples_unique < len(latencies): + n_samples_unique = len(latencies) elif summary0: # Fallback to report summary if raw predictions are unavailable latency_mean = summary0.get('latency', {}).get('mean', np.nan) @@ -425,7 +512,27 @@ def collect_benchmark(output_dir: Path, benchmark: str, model_name: str): tpot_p90 = backup_summary.get('tpot', {}).get('90%', np.nan) tpot_p99 = backup_summary.get('tpot', {}).get('99%', np.nan) else: - return None + # Agent / sandbox benchmarks (e.g. terminal_bench_v2_1) provide a score + # but do not record per-sample perf metrics. Keep the score and leave + # perf columns empty rather than dropping the benchmark entirely. + latency_mean = np.nan + avg_output_tps = np.nan + avg_req_ps = np.nan + input_tok_mean = np.nan + output_tok_mean = np.nan + total_tokens = np.nan + ttft_mean = np.nan + ttft_p90 = np.nan + ttft_p99 = np.nan + tpot_mean = np.nan + tpot_p90 = np.nan + tpot_p99 = np.nan + if not n_samples_unique and reports: + try: + data = json.loads(reports[0].read_text(encoding='utf-8')) + n_samples_unique = data.get('num', 0) + except Exception: + pass # Duration: prefer the durable active timer maintained by perf_backup.py, # which only counts time when run_task() is actually executing. This avoids diff --git a/bash/case/preload_terminal_bench_images.sh b/bash/images_load/preload_terminal_bench_images.sh similarity index 89% rename from bash/case/preload_terminal_bench_images.sh rename to bash/images_load/preload_terminal_bench_images.sh index c7b7795..24a6f01 100755 --- a/bash/case/preload_terminal_bench_images.sh +++ b/bash/images_load/preload_terminal_bench_images.sh @@ -8,7 +8,7 @@ # 后续评测就只需要跑 agent,不用等镜像下载。 # # 用法: -# bash bash/case/preload_terminal_bench_images.sh +# bash bash/images_load/preload_terminal_bench_images.sh # # 注意: # - 需要 Docker 环境 @@ -47,12 +47,16 @@ echo "" echo "============================================================" echo "3. 并行拉取 Docker 镜像" echo "============================================================" -echo "开始拉取镜像(后台 4 并发),日志保存在 /tmp/pull_terminal_images.log" -mkdir -p logs -LOG_FILE="logs/pull_terminal_images.log" +# 脚本放在 bash/images_load/,项目根目录是上一级再上一级 +PROJECT_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +LOG_DIR="${PROJECT_ROOT}/logs" +LOG_FILE="${LOG_DIR}/pull_terminal_images.log" +mkdir -p "${LOG_DIR}" > "${LOG_FILE}" +echo "开始拉取镜像(后台 4 并发),日志保存在 ${LOG_FILE}" + PULL_FAILED=0 while IFS= read -r image; do [[ -z "$image" ]] && continue diff --git a/evalscope/evalscope/benchmarks/terminal_bench/terminal_bench_adapter.py b/evalscope/evalscope/benchmarks/terminal_bench/terminal_bench_adapter.py index e306ce8..b83a8df 100644 --- a/evalscope/evalscope/benchmarks/terminal_bench/terminal_bench_adapter.py +++ b/evalscope/evalscope/benchmarks/terminal_bench/terminal_bench_adapter.py @@ -22,6 +22,61 @@ from evalscope.utils.logger import get_logger logger = get_logger() +# Harbor 从 Supabase 下载任务包时默认超时只有 120s,网络慢时容易 ReadTimeout。 +# 允许通过环境变量调大,默认 600s。 +try: + import harbor.storage.supabase as _harbor_supabase + + _harbor_supabase.PACKAGE_STORAGE_TIMEOUT_SEC = int( + os.environ.get('HARBOR_PACKAGE_STORAGE_TIMEOUT_SEC', '600') + ) +except Exception: + pass + +# terminus-2 agent 启动时会在容器里 apt-get install tmux/asciinema,默认只等 120s, +# 国内 apt 源慢时很容易超时,导致任务直接失败。允许通过环境变量调大,默认 600s。 +try: + from harbor.agents.terminus_2 import tmux_session as _harbor_tmux_session + + _harbor_tmux_session.TmuxSession._TOOL_INSTALL_TIMEOUT_SEC = int( + os.environ.get('HARBOR_TOOL_INSTALL_TIMEOUT_SEC', '600') + ) +except Exception: + pass + +# 把容器内的 apt 源换成清华镜像,避免 tmux/asciinema 安装时 apt-get update 慢/超时。 +# 可以通过环境变量 HARBOR_APT_MIRROR 切换镜像地址。 +try: + from harbor.agents.terminus_2.tmux_session import TmuxSession as _TmuxSession + + _APT_MIRROR = os.environ.get('HARBOR_APT_MIRROR', 'https://mirrors.tuna.tsinghua.edu.cn') + _ORIG_GET_COMBINED_INSTALL_COMMAND = _TmuxSession._get_combined_install_command + + def _patched_get_combined_install_command(self, system_info, tools): + package_manager = system_info.get('package_manager') if isinstance(system_info, dict) else None + if package_manager == 'apt-get': + packages = ' '.join(tools) + return ( + f"sed -i 's|http://archive.ubuntu.com/ubuntu/|{_APT_MIRROR}/ubuntu/|g; " + f"s|https://archive.ubuntu.com/ubuntu/|{_APT_MIRROR}/ubuntu/|g; " + f"s|http://security.ubuntu.com/ubuntu/|{_APT_MIRROR}/ubuntu/|g; " + f"s|https://security.ubuntu.com/ubuntu/|{_APT_MIRROR}/ubuntu/|g; " + f"s|http://ports.ubuntu.com/ubuntu-ports/|{_APT_MIRROR}/ubuntu-ports/|g; " + f"s|https://ports.ubuntu.com/ubuntu-ports/|{_APT_MIRROR}/ubuntu-ports/|g; " + f"s|http://deb.debian.org/debian|{_APT_MIRROR}/debian|g; " + f"s|https://deb.debian.org/debian|{_APT_MIRROR}/debian|g; " + f"s|http://security.debian.org/debian-security|{_APT_MIRROR}/debian-security|g; " + f"s|https://security.debian.org/debian-security|{_APT_MIRROR}/debian-security|g' " + f"/etc/apt/sources.list /etc/apt/sources.list.d/*.list 2>/dev/null; " + f"DEBIAN_FRONTEND=noninteractive apt-get update && " + f"DEBIAN_FRONTEND=noninteractive apt-get install -y {packages}" + ) + return _ORIG_GET_COMBINED_INSTALL_COMMAND(self, system_info, tools) + + _TmuxSession._get_combined_install_command = _patched_get_combined_install_command +except Exception: + pass + COMMON_EXTRA_PARAMS = { 'environment_type': { 'type': 'str',