From 0f9399baf68caf514cef8b2e62c84397a6aa1434 Mon Sep 17 00:00:00 2001 From: yy-fighting Date: Wed, 8 Jul 2026 09:44:13 +0000 Subject: [PATCH] feat(64k): add SGLang vs vLLM 64k-context comparison and results run 20260708-092556 --- .../dsv4_h200_64k_sglang_vs_vllm/compare.py | 85 ++++++ .../dsv4_h200_64k_sglang_vs_vllm/config.env | 34 +++ .../parse_backend.py | 219 ++++++++++++++ .../results/20260708-092556/comparison.md | 23 ++ .../results/20260708-092556/sglang/report.md | 16 + .../20260708-092556/sglang/results.json | 140 +++++++++ .../results/20260708-092556/vllm/report.md | 16 + .../results/20260708-092556/vllm/results.json | 140 +++++++++ .../dsv4_h200_64k_sglang_vs_vllm/run_bench.sh | 275 ++++++++++++++++++ .../start_sglang.sh | 62 ++++ .../start_vllm.sh | 65 +++++ .../dsv4_h200_64k_sglang_vs_vllm/warmup.py | 76 +++++ 12 files changed, 1151 insertions(+) create mode 100755 experiments/dsv4_h200_64k_sglang_vs_vllm/compare.py create mode 100644 experiments/dsv4_h200_64k_sglang_vs_vllm/config.env create mode 100755 experiments/dsv4_h200_64k_sglang_vs_vllm/parse_backend.py create mode 100644 experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/comparison.md create mode 100644 experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang/report.md create mode 100644 experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang/results.json create mode 100644 experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm/report.md create mode 100644 experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm/results.json create mode 100644 experiments/dsv4_h200_64k_sglang_vs_vllm/run_bench.sh create mode 100755 experiments/dsv4_h200_64k_sglang_vs_vllm/start_sglang.sh create mode 100755 experiments/dsv4_h200_64k_sglang_vs_vllm/start_vllm.sh create mode 100755 experiments/dsv4_h200_64k_sglang_vs_vllm/warmup.py diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/compare.py b/experiments/dsv4_h200_64k_sglang_vs_vllm/compare.py new file mode 100755 index 0000000..0f4eb51 --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/compare.py @@ -0,0 +1,85 @@ +#!/usr/bin/env python3 +"""Generate a side-by-side comparison of SGLang and vLLM results. + +Usage: + python3 compare.py --sglang --vllm \ + [--output comparison.md] +""" +import argparse +import json +from collections import defaultdict +from pathlib import Path + + +def load_result(result_root: Path) -> dict: + path = result_root / "results.json" + with open(path, "r", encoding="utf-8") as f: + return json.load(f) + + +def slo_status(ttft_p95_ms: float, tpot_mean_ms: float) -> str: + # S2 tier targets from scripts/SLO_STANDARDS.md. + ttft_ok = ttft_p95_ms < 3000.0 + tpot_ok = tpot_mean_ms < 50.0 + if ttft_ok and tpot_ok: + return "✅" + if ttft_ok or tpot_ok: + return "⚠️" + return "❌" + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--sglang", type=Path, required=True) + parser.add_argument("--vllm", type=Path, required=True) + parser.add_argument("-o", "--output", type=Path, default=Path("comparison.md")) + args = parser.parse_args() + + sglang_data = load_result(args.sglang) + vllm_data = load_result(args.vllm) + + by_scenario = defaultdict(dict) + for data in (sglang_data, vllm_data): + backend = data["metadata"]["engine"] + for s in data.get("scenarios", []): + key = s["name"] + by_scenario[key][backend] = s + + with open(args.output, "w", encoding="utf-8") as f: + f.write("# SGLang vs vLLM on DeepSeek-V4-Flash (H200, TP=8)\n\n") + f.write("## Summary\n\n") + f.write("- Model: `/data/models/DeepSeek-V4-Flash`\n") + f.write("- Hardware: 8x NVIDIA H200 143GB\n") + f.write("- Tensor Parallelism: 8\n") + f.write("- Benchmark client: `sglang.bench_serving`\n") + f.write("- SLO reference: S2 tier (TTFT P95 < 3s, TPOT < 50ms)\n\n") + + f.write("## Side-by-side results\n\n") + f.write("| Scenario | Backend | Conc | Input | Output | Req/s | OutTok/s | TTFT P95(ms) | TTFT P99(ms) | TPOT Mean(ms) | TPOT P95(ms) | TPOT P99(ms) | E2E P99(ms) | SLO |\n") + f.write("|---|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|\n") + + for scenario_name in sorted(by_scenario.keys()): + for backend in ("sglang", "vllm"): + s = by_scenario[scenario_name].get(backend) + if s is None: + continue + cfg = s["config"] + m = s["metrics"] + status = slo_status(m["ttft_ms"]["p95"], m["tpot_ms"]["mean"]) + f.write( + f"| {scenario_name} | {backend} | {cfg['concurrency']} | {cfg['input_len']} | {cfg['output_len']} | " + f"{m['request_throughput']:.2f} | {m['output_token_throughput']:.2f} | " + f"{m['ttft_ms']['p95']:.2f} | {m['ttft_ms']['p99']:.2f} | " + f"{m['tpot_ms']['mean']:.2f} | {m['tpot_ms']['p95']:.2f} | {m['tpot_ms']['p99']:.2f} | " + f"{m['e2e_ms']['p99']:.2f} | {status} |\n" + ) + + f.write("\n## Notes\n\n") + f.write("- SLO check uses TTFT P95 and TPOT mean (the same criteria as `scripts/SLO_STANDARDS.md`).\n") + f.write("- A ⚠️ indicates one of the two metrics is out of target; ❌ indicates both are out.\n") + + print(f"Wrote comparison to {args.output}") + + +if __name__ == "__main__": + main() diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/config.env b/experiments/dsv4_h200_64k_sglang_vs_vllm/config.env new file mode 100644 index 0000000..c64aff4 --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/config.env @@ -0,0 +1,34 @@ +# 64k-context focused comparison for SGLang vs vLLM on H200. +# Scenarios requested: +# 1) input=65536, output=256, concurrency=25 +# 2) input=65536, output=1024, concurrency=8, num_prompts=100 + +EXPERIMENT="dsv4_h200_64k_sglang_vs_vllm" +MODEL_NAME="DeepSeek-V4-Flash" +MODEL_PATH="/data/models/DeepSeek-V4-Flash" +SERVED_MODEL_NAME="deepseek-v4-flash" + +SGLANG_PORT="${SGLANG_PORT:-30006}" +VLLM_PORT="${VLLM_PORT:-30005}" + +VENV_SGLANG="${VENV_SGLANG:-/data/user1/yy/envs/sglang}" +VENV_VLLM="${VENV_VLLM:-/data/user1/yy/envs/vllm}" + +export CUDA_VISIBLE_DEVICES="0,1,2,3,4,5,6,7" +TP=8 + +# 64k + 1024 output + padding -> use 70000 to keep both backends identical. +MAX_MODEL_LEN=70000 +MAX_NUM_SEQS=64 +MAX_RUNNING=64 + +# "concurrency input_len output_len num_prompts" +declare -a SCENARIOS=( + "25 65536 256 100" + "8 65536 1024 100" +) + +VENV_CLIENT="${VENV_CLIENT:-$VENV_SGLANG}" + +SGLANG_START_SCRIPT="${SCRIPT_DIR:-.}/start_sglang.sh" +VLLM_START_SCRIPT="${SCRIPT_DIR:-.}/start_vllm.sh" diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/parse_backend.py b/experiments/dsv4_h200_64k_sglang_vs_vllm/parse_backend.py new file mode 100755 index 0000000..935d870 --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/parse_backend.py @@ -0,0 +1,219 @@ +#!/usr/bin/env python3 +"""Parse raw sglang.bench_serving JSONL outputs for one backend. + +Reads JSONL files like {sglang|vllm}_phase1_MMDD_concurrency_inputlen_outputlen.jsonl +and generates results.json + report.md in the given result root. + +Usage: + python3 parse_backend.py [--backend sglang|vllm] +""" +import argparse +import json +import re +import sys +from pathlib import Path + + +def parse_jsonl(path: Path) -> dict | None: + with open(path, "r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + return json.loads(line) + except json.JSONDecodeError: + continue + return None + + +def compute_metrics(data: dict) -> dict: + completed = data.get("completed", 0) + total = len(data.get("input_lens", [])) + failed = total - completed if total > 0 else 0 + duration_s = data.get("duration", 0.0) + + return { + "success": completed, + "failed": failed, + "duration_s": duration_s, + "request_throughput": data.get("request_throughput", 0.0), + "input_token_throughput": data.get("input_throughput", 0.0), + "output_token_throughput": data.get("output_throughput", 0.0), + "total_token_throughput": data.get("total_throughput", 0.0), + "total_input_tokens": data.get("total_input_tokens", 0), + "total_output_tokens": data.get("total_output_tokens", 0), + "e2e_ms": { + "mean": data.get("mean_e2e_latency_ms", 0.0), + "p50": data.get("median_e2e_latency_ms", 0.0), + "p90": data.get("p90_e2e_latency_ms", 0.0), + "p95": data.get("p95_e2e_latency_ms", 0.0), + "p99": data.get("p99_e2e_latency_ms", 0.0), + }, + "ttft_ms": { + "mean": data.get("mean_ttft_ms", 0.0), + "p50": data.get("median_ttft_ms", 0.0), + "p90": data.get("p90_ttft_ms", 0.0), + "p95": data.get("p95_ttft_ms", 0.0), + "p99": data.get("p99_ttft_ms", 0.0), + }, + "tpot_ms": { + "mean": data.get("mean_tpot_ms", 0.0), + "p50": data.get("median_tpot_ms", 0.0), + "p90": data.get("p90_tpot_ms", 0.0), + "p95": data.get("p95_tpot_ms", 0.0), + "p99": data.get("p99_tpot_ms", 0.0), + }, + "itl_ms": { + "mean": data.get("mean_itl_ms", 0.0), + "p50": data.get("median_itl_ms", 0.0), + "p90": data.get("p90_itl_ms", 0.0), + "p95": data.get("p95_itl_ms", 0.0), + "p99": data.get("p99_itl_ms", 0.0), + }, + } + + +def scenario_name(concurrency: int, input_len: int, output_len: int) -> str: + return f"c{concurrency}_i{input_len}_o{output_len}" + + +def slo_status(metrics: dict) -> dict: + """Check S2-tier SLO: TTFT P95 < 3s, TPOT mean < 50ms.""" + ttft_ok = metrics["ttft_ms"]["p95"] < 3000.0 + tpot_ok = metrics["tpot_ms"]["mean"] < 50.0 + if ttft_ok and tpot_ok: + mark = "✅" + elif ttft_ok or tpot_ok: + mark = "⚠️" + else: + mark = "❌" + return { + "ttft_p95_ok": ttft_ok, + "tpot_mean_ok": tpot_ok, + "overall": mark, + } + + +def append_scenario(results_json: Path, scenario: dict) -> None: + with open(results_json, "r", encoding="utf-8") as f: + data = json.load(f) + data["scenarios"].append(scenario) + with open(results_json, "w", encoding="utf-8") as f: + json.dump(data, f, indent=2, ensure_ascii=False) + + +def generate_report(result_root: Path, backend: str, scenarios: list[dict]) -> None: + report_path = result_root / "report.md" + with open(report_path, "w", encoding="utf-8") as f: + f.write(f"# H200 {backend.upper()} Comparison Benchmark Report\n\n") + f.write(f"- Result root: `{result_root}`\n") + f.write(f"- Model: `/data/models/DeepSeek-V4-Flash`\n") + f.write(f"- Backend: {backend.upper()} (TP=8)\n") + f.write(f"- Benchmark client: `sglang.bench_serving --backend {backend}`\n\n") + + f.write("## Results\n\n") + f.write("| Scenario | Phase | Concurrency | Input | Output | Duration(s) | Success | Req/s | In tok/s | Out tok/s | Total tok/s | Mean TTFT(ms) | P95 TTFT(ms) | P99 TTFT(ms) | Mean TPOT(ms) | P95 TPOT(ms) | P99 TPOT(ms) | Mean E2E(ms) | P95 E2E(ms) | P99 E2E(ms) | SLO |\n") + f.write("|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|\n") + + for s in scenarios: + cfg = s["config"] + m = s["metrics"] + slo = s.get("slo_status", {}).get("overall", "") + f.write( + f"| {s['name']} | {cfg['phase']} | {cfg['concurrency']} | {cfg['input_len']} | {cfg['output_len']} | " + f"{m['duration_s']:.2f} | {m['success']} | {m['request_throughput']:.2f} | " + f"{m['input_token_throughput']:.2f} | {m['output_token_throughput']:.2f} | " + f"{m['total_token_throughput']:.2f} | " + f"{m['ttft_ms']['mean']:.2f} | {m['ttft_ms']['p95']:.2f} | {m['ttft_ms']['p99']:.2f} | " + f"{m['tpot_ms']['mean']:.2f} | {m['tpot_ms']['p95']:.2f} | {m['tpot_ms']['p99']:.2f} | " + f"{m['e2e_ms']['mean']:.2f} | {m['e2e_ms']['p95']:.2f} | {m['e2e_ms']['p99']:.2f} | {slo} |\n" + ) + f.write("\n") + f.write("SLO: S2 tier — TTFT P95 < 3000ms, TPOT mean < 50ms. ✅ pass, ⚠️ partial, ❌ fail.\n\n") + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("result_root", type=Path) + parser.add_argument("--backend", default=None, choices=["sglang", "vllm"]) + args = parser.parse_args() + + result_root = args.result_root + raw_dir = result_root / "raw_outputs" + results_json = result_root / "results.json" + + if not raw_dir.exists(): + raise SystemExit(f"raw_outputs directory not found: {raw_dir}") + + backend = args.backend + if backend is None: + # Infer from filenames if not provided. + for p in raw_dir.iterdir(): + if p.name.startswith("sglang_"): + backend = "sglang" + break + if p.name.startswith("vllm_"): + backend = "vllm" + break + if backend is None: + raise SystemExit("Could not infer backend from raw outputs") + + scenarios = [] + for jsonl_path in sorted(raw_dir.glob(f"{backend}_*.jsonl")): + parts = jsonl_path.stem.split("_") + if len(parts) < 5: + continue + + label = parts[1] + if label == "sharegpt": + phase = "sharegpt" + dataset = "sharegpt" + else: + phase = label + dataset = "random" + + # The last three numeric tokens are: concurrency, input_len/output_ctx, output_len. + try: + concurrency, input_len, output_len = int(parts[-3]), int(parts[-2]), int(parts[-1]) + except ValueError: + continue + + data = parse_jsonl(jsonl_path) + if data is None: + continue + + metrics = compute_metrics(data) + scenario = { + "name": scenario_name(concurrency, input_len, output_len), + "config": { + "phase": phase, + "concurrency": concurrency, + "input_len": input_len, + "output_len": output_len, + "dataset": dataset, + "num_prompts": metrics["success"] + metrics["failed"], + }, + "metrics": metrics, + "slo_status": slo_status(metrics), + "raw_file": str(jsonl_path), + } + scenarios.append(scenario) + + if not scenarios: + print("No benchmark outputs found to parse") + return + + if results_json.exists(): + with open(results_json, "r", encoding="utf-8") as f: + data = json.load(f) + data["scenarios"] = scenarios + with open(results_json, "w", encoding="utf-8") as f: + json.dump(data, f, indent=2, ensure_ascii=False) + + generate_report(result_root, backend, scenarios) + print(f"Parsed {len(scenarios)} scenarios into {result_root}/report.md") + + +if __name__ == "__main__": + main() diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/comparison.md b/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/comparison.md new file mode 100644 index 0000000..e73a8ce --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/comparison.md @@ -0,0 +1,23 @@ +# SGLang vs vLLM on DeepSeek-V4-Flash (H200, TP=8) + +## Summary + +- Model: `/data/models/DeepSeek-V4-Flash` +- Hardware: 8x NVIDIA H200 143GB +- Tensor Parallelism: 8 +- Benchmark client: `sglang.bench_serving` +- SLO reference: S2 tier (TTFT P95 < 3s, TPOT < 50ms) + +## Side-by-side results + +| Scenario | Backend | Conc | Input | Output | Req/s | OutTok/s | TTFT P95(ms) | TTFT P99(ms) | TPOT Mean(ms) | TPOT P95(ms) | TPOT P99(ms) | E2E P99(ms) | SLO | +|---|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:| +| c25_i65536_o256 | sglang | 25 | 65536 | 256 | 0.74 | 100.20 | 27897.01 | 33819.41 | 246.08 | 493.20 | 1511.47 | 81695.50 | ❌ | +| c25_i65536_o256 | vllm | 25 | 65536 | 256 | 0.90 | 121.20 | 23785.08 | 27991.50 | 160.93 | 269.48 | 274.21 | 64535.46 | ❌ | +| c8_i65536_o1024 | sglang | 8 | 65536 | 1024 | 0.56 | 296.33 | 4369.34 | 9023.87 | 21.75 | 31.44 | 33.87 | 30303.44 | ⚠️ | +| c8_i65536_o1024 | vllm | 8 | 65536 | 1024 | 0.61 | 320.37 | 4409.89 | 6408.20 | 20.22 | 29.04 | 30.73 | 27532.97 | ⚠️ | + +## Notes + +- SLO check uses TTFT P95 and TPOT mean (the same criteria as `scripts/SLO_STANDARDS.md`). +- A ⚠️ indicates one of the two metrics is out of target; ❌ indicates both are out. diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang/report.md b/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang/report.md new file mode 100644 index 0000000..828ed0c --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang/report.md @@ -0,0 +1,16 @@ +# H200 SGLANG Comparison Benchmark Report + +- Result root: `/data/user1/yy/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang` +- Model: `/data/models/DeepSeek-V4-Flash` +- Backend: SGLANG (TP=8) +- Benchmark client: `sglang.bench_serving --backend sglang` + +## Results + +| Scenario | Phase | Concurrency | Input | Output | Duration(s) | Success | Req/s | In tok/s | Out tok/s | Total tok/s | Mean TTFT(ms) | P95 TTFT(ms) | P99 TTFT(ms) | Mean TPOT(ms) | P95 TPOT(ms) | P99 TPOT(ms) | Mean E2E(ms) | P95 E2E(ms) | P99 E2E(ms) | SLO | +|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:| +| c25_i65536_o256 | 64k | 25 | 65536 | 256 | 134.58 | 100 | 0.74 | 24467.44 | 100.20 | 24567.63 | 6853.10 | 27897.01 | 33819.41 | 246.08 | 493.20 | 1511.47 | 33391.78 | 69227.68 | 81695.50 | ❌ | +| c8_i65536_o1024 | 64k | 8 | 65536 | 1024 | 177.68 | 100 | 0.56 | 18531.97 | 296.33 | 18828.31 | 1887.62 | 4369.34 | 9023.87 | 21.75 | 31.44 | 33.87 | 14046.05 | 28233.77 | 30303.44 | ⚠️ | + +SLO: S2 tier — TTFT P95 < 3000ms, TPOT mean < 50ms. ✅ pass, ⚠️ partial, ❌ fail. + diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang/results.json b/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang/results.json new file mode 100644 index 0000000..522b158 --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang/results.json @@ -0,0 +1,140 @@ +{ + "metadata": { + "experiment": "dsv4_h200_64k_sglang_vs_vllm_sglang", + "run_id": "20260708-092556", + "timestamp": "2026-07-08T09:26:00+00:00", + "model": "/data/models/DeepSeek-V4-Flash", + "backend": "sglang", + "engine": "sglang", + "hardware": "8x NVIDIA H200 143GB", + "accelerator": "NVIDIA H200", + "chip": "nvidia_h200", + "script": "experiments/dsv4_h200_64k_sglang_vs_vllm/run_bench.sh", + "env": "/data/user1/yy/envs/sglang", + "git_commit": "1c827a5", + "git_dirty": "dirty", + "description": "H200 64k-context sglang TP=8 comparison for DeepSeek-V4-Flash" + }, + "config": { + "tp": 8, + "cuda_visible_devices": "0,1,2,3,4,5,6,7", + "max_model_len": 70000, + "backend": "sglang", + "server_start_script": "experiments/dsv4_h200_64k_sglang_vs_vllm/start_sglang.sh", + "server_args": "sglang serve --trust-remote-code --model-path /data/models/DeepSeek-V4-Flash --tp 8 --moe-runner-backend marlin --context-length 70000 --max-running-requests 64 --mem-fraction-static 0.88 --host 0.0.0.0 --port 30006" + }, + "scenarios": [ + { + "name": "c25_i65536_o256", + "config": { + "phase": "64k", + "concurrency": 25, + "input_len": 65536, + "output_len": 256, + "dataset": "random", + "num_prompts": 100 + }, + "metrics": { + "success": 100, + "failed": 0, + "duration_s": 134.57641048599908, + "request_throughput": 0.7430722787067032, + "input_token_throughput": 24467.438149887097, + "output_token_throughput": 100.19586606081185, + "total_token_throughput": 24567.63401594791, + "total_input_tokens": 3292740, + "total_output_tokens": 13484, + "e2e_ms": { + "mean": 33391.77867801962, + "p50": 34128.26199350093, + "p90": 66453.93764560722, + "p95": 69227.67581411026, + "p99": 81695.49766001117 + }, + "ttft_ms": { + "mean": 6853.104796579864, + "p50": 3640.2646064962028, + "p90": 18856.929641509505, + "p95": 27897.01368195747, + "p99": 33819.410034179025 + }, + "tpot_ms": { + "mean": 246.08454978038876, + "p50": 202.58080690000497, + "p90": 353.2847035188578, + "p95": 493.2046375697246, + "p99": 1511.474447885656 + }, + "itl_ms": { + "mean": 198.28636850066192, + "p50": 9.821248000662308, + "p90": 107.71931030321866, + "p95": 191.68548340021457, + "p99": 4516.802086499083 + } + }, + "slo_status": { + "ttft_p95_ok": false, + "tpot_mean_ok": false, + "overall": "❌" + }, + "raw_file": "/data/user1/yy/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang/raw_outputs/sglang_64k_0708_25_65536_256.jsonl" + }, + { + "name": "c8_i65536_o1024", + "config": { + "phase": "64k", + "concurrency": 8, + "input_len": 65536, + "output_len": 1024, + "dataset": "random", + "num_prompts": 100 + }, + "metrics": { + "success": 100, + "failed": 0, + "duration_s": 177.67885851500614, + "request_throughput": 0.5628131609791626, + "input_token_throughput": 18531.974076825278, + "output_token_throughput": 296.33238551874865, + "total_token_throughput": 18828.306462344026, + "total_input_tokens": 3292740, + "total_output_tokens": 52652, + "e2e_ms": { + "mean": 14046.045127669786, + "p50": 14341.503521005507, + "p90": 25773.02649689955, + "p95": 28233.767246660136, + "p99": 30303.43603371788 + }, + "ttft_ms": { + "mean": 1887.6239945401903, + "p50": 1750.550087497686, + "p90": 2827.0532990034562, + "p95": 4369.34463484722, + "p99": 9023.866625820376 + }, + "tpot_ms": { + "mean": 21.749239020865442, + "p50": 23.1435364794262, + "p90": 29.76345981834677, + "p95": 31.438390882666276, + "p99": 33.87255562285786 + }, + "itl_ms": { + "mean": 23.135925370298736, + "p50": 8.084575507382397, + "p90": 8.361168300325517, + "p95": 8.722730595764, + "p99": 178.0338145857968 + } + }, + "slo_status": { + "ttft_p95_ok": false, + "tpot_mean_ok": true, + "overall": "⚠️" + }, + "raw_file": "/data/user1/yy/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/sglang/raw_outputs/sglang_64k_0708_8_65536_1024.jsonl" + } + ] +} \ No newline at end of file diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm/report.md b/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm/report.md new file mode 100644 index 0000000..a2347be --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm/report.md @@ -0,0 +1,16 @@ +# H200 VLLM Comparison Benchmark Report + +- Result root: `/data/user1/yy/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm` +- Model: `/data/models/DeepSeek-V4-Flash` +- Backend: VLLM (TP=8) +- Benchmark client: `sglang.bench_serving --backend vllm` + +## Results + +| Scenario | Phase | Concurrency | Input | Output | Duration(s) | Success | Req/s | In tok/s | Out tok/s | Total tok/s | Mean TTFT(ms) | P95 TTFT(ms) | P99 TTFT(ms) | Mean TPOT(ms) | P95 TPOT(ms) | P99 TPOT(ms) | Mean E2E(ms) | P95 E2E(ms) | P99 E2E(ms) | SLO | +|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:| +| c25_i65536_o256 | 64k | 25 | 65536 | 256 | 111.25 | 100 | 0.90 | 29597.12 | 121.20 | 29718.32 | 6836.31 | 23785.08 | 27991.50 | 160.93 | 269.48 | 274.21 | 27483.93 | 58658.69 | 64535.46 | ❌ | +| c8_i65536_o1024 | 64k | 8 | 65536 | 1024 | 164.35 | 100 | 0.61 | 20035.37 | 320.37 | 20355.75 | 1752.40 | 4409.89 | 6408.20 | 20.22 | 29.04 | 30.73 | 12987.19 | 24810.55 | 27532.97 | ⚠️ | + +SLO: S2 tier — TTFT P95 < 3000ms, TPOT mean < 50ms. ✅ pass, ⚠️ partial, ❌ fail. + diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm/results.json b/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm/results.json new file mode 100644 index 0000000..e631839 --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm/results.json @@ -0,0 +1,140 @@ +{ + "metadata": { + "experiment": "dsv4_h200_64k_sglang_vs_vllm_vllm", + "run_id": "20260708-092556", + "timestamp": "2026-07-08T09:26:00+00:00", + "model": "/data/models/DeepSeek-V4-Flash", + "backend": "vllm", + "engine": "vllm", + "hardware": "8x NVIDIA H200 143GB", + "accelerator": "NVIDIA H200", + "chip": "nvidia_h200", + "script": "experiments/dsv4_h200_64k_sglang_vs_vllm/run_bench.sh", + "env": "/data/user1/yy/envs/vllm", + "git_commit": "1c827a5", + "git_dirty": "dirty", + "description": "H200 64k-context vllm TP=8 comparison for DeepSeek-V4-Flash" + }, + "config": { + "tp": 8, + "cuda_visible_devices": "0,1,2,3,4,5,6,7", + "max_model_len": 70000, + "backend": "vllm", + "server_start_script": "experiments/dsv4_h200_64k_sglang_vs_vllm/start_vllm.sh", + "server_args": "vllm serve /data/models/DeepSeek-V4-Flash --trust-remote-code --tensor-parallel-size 8 --kv-cache-dtype fp8 --max-model-len 70000 --max-num-seqs 64 --block-size 256 --gpu-memory-utilization 0.90 --tokenizer-mode deepseek_v4 --reasoning-parser deepseek_v4 --no-disable-hybrid-kv-cache-manager --disable-uvicorn-access-log --port 30005" + }, + "scenarios": [ + { + "name": "c25_i65536_o256", + "config": { + "phase": "64k", + "concurrency": 25, + "input_len": 65536, + "output_len": 256, + "dataset": "random", + "num_prompts": 100 + }, + "metrics": { + "success": 100, + "failed": 0, + "duration_s": 111.25205481000012, + "request_throughput": 0.8988598023720394, + "input_token_throughput": 29597.11625662509, + "output_token_throughput": 121.2022557518458, + "total_token_throughput": 29718.318512376936, + "total_input_tokens": 3292740, + "total_output_tokens": 13484, + "e2e_ms": { + "mean": 27483.927442140266, + "p50": 25770.644948504923, + "p90": 52341.01150579518, + "p95": 58658.69059454852, + "p99": 64535.46363994714 + }, + "ttft_ms": { + "mean": 6836.30942699936, + "p50": 4711.959886997647, + "p90": 18929.24038059719, + "p95": 23785.08340215194, + "p99": 27991.500275050505 + }, + "tpot_ms": { + "mean": 160.9309907200676, + "p50": 175.74239191996944, + "p90": 258.3217706461885, + "p95": 269.4797523540955, + "p99": 274.21162041586393 + }, + "itl_ms": { + "mean": 158.6445563515138, + "p50": 219.7408100037137, + "p90": 294.3773947918089, + "p95": 308.6502741047297, + "p99": 345.6698693789076 + } + }, + "slo_status": { + "ttft_p95_ok": false, + "tpot_mean_ok": false, + "overall": "❌" + }, + "raw_file": "/data/user1/yy/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm/raw_outputs/vllm_64k_0708_25_65536_256.jsonl" + }, + { + "name": "c8_i65536_o1024", + "config": { + "phase": "64k", + "concurrency": 8, + "input_len": 65536, + "output_len": 1024, + "dataset": "random", + "num_prompts": 100 + }, + "metrics": { + "success": 100, + "failed": 0, + "duration_s": 164.34632389699982, + "request_throughput": 0.608471170080279, + "input_token_throughput": 20035.373605701378, + "output_token_throughput": 320.3722404706685, + "total_token_throughput": 20355.74584617205, + "total_input_tokens": 3292740, + "total_output_tokens": 52652, + "e2e_ms": { + "mean": 12987.194003800396, + "p50": 12404.946581998956, + "p90": 23933.946933604602, + "p95": 24810.553805056403, + "p99": 27532.967685358744 + }, + "ttft_ms": { + "mean": 1752.3960321310733, + "p50": 1674.458354995295, + "p90": 2854.2016049992526, + "p95": 4409.8852371978855, + "p99": 6408.1971558836885 + }, + "tpot_ms": { + "mean": 20.220859090868107, + "p50": 21.102676729652877, + "p90": 27.359026930714556, + "p95": 29.035320478265973, + "p99": 30.72600634009067 + }, + "itl_ms": { + "mean": 22.493866762442327, + "p50": 8.34087949624518, + "p90": 10.428507499455009, + "p95": 207.22519800256123, + "p99": 284.1898243590549 + } + }, + "slo_status": { + "ttft_p95_ok": false, + "tpot_mean_ok": true, + "overall": "⚠️" + }, + "raw_file": "/data/user1/yy/experiments/dsv4_h200_64k_sglang_vs_vllm/results/20260708-092556/vllm/raw_outputs/vllm_64k_0708_8_65536_1024.jsonl" + } + ] +} \ No newline at end of file diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/run_bench.sh b/experiments/dsv4_h200_64k_sglang_vs_vllm/run_bench.sh new file mode 100644 index 0000000..eaf41b5 --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/run_bench.sh @@ -0,0 +1,275 @@ +#!/usr/bin/env bash +# 64k-context SGLang vs vLLM comparison on H200. +set -Eeuo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +EXPERIMENT_NAME="$(basename "$SCRIPT_DIR")" + +# shellcheck source=/dev/null +source "${SCRIPT_DIR}/../../scripts/common/lib.sh" +# shellcheck source=/dev/null +source "${SCRIPT_DIR}/../../scripts/common/platform.sh" +# shellcheck source=/dev/null +source "${SCRIPT_DIR}/config.env" + +RUN_ID="${RUN_ID:-$(date '+%Y%m%d-%H%M%S')}" +RESULT_BASE="${SCRIPT_DIR}/results" + +log_dir_global="${RESULT_BASE}/${RUN_ID}/logs" +mkdir -p "$log_dir_global" +log_init "${log_dir_global}/orchestrator.log" + +log "experiment=${EXPERIMENT_NAME}" +log "run_id=${RUN_ID}" +log "platform=${PLATFORM}" +log "hardware=${HARDWARE}" +log "model=${MODEL_PATH}" +log "max_model_len=${MAX_MODEL_LEN}" +log "scenarios=${#SCENARIOS[@]}" + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +is_server_healthy() { + local port="$1" + curl --fail --silent --show-error --max-time 5 "http://127.0.0.1:${port}/health" >/dev/null 2>&1 +} + +stop_server() { + local backend="$1" + local pid_file="/data/user1/yy/dsv4_h200_64k_sglang_vs_vllm_${backend}.pid" + if [[ -f "$pid_file" ]]; then + local pid + pid="$(cat "$pid_file")" + if kill -0 "$pid" 2>/dev/null; then + log "stopping ${backend} server pid=${pid}" + kill "$pid" 2>/dev/null || true + sleep 5 + kill -9 "$pid" 2>/dev/null || true + fi + rm -f "$pid_file" + fi + # Fallback cleanup. + if [[ "$backend" == "sglang" ]]; then + pkill -9 -f "sglang serve.*DeepSeek-V4-Flash" 2>/dev/null || true + else + pkill -9 -f "vllm serve.*DeepSeek-V4-Flash" 2>/dev/null || true + fi + sleep 2 +} + +start_server() { + local backend="$1" + local start_script + if [[ "$backend" == "sglang" ]]; then + start_script="${SGLANG_START_SCRIPT}" + local port="$SGLANG_PORT" + else + start_script="${VLLM_START_SCRIPT}" + local port="$VLLM_PORT" + fi + + log "starting ${backend} server with ${start_script}" + bash "${start_script}" >> "${log_dir_global}/${backend}.server.outer.log" 2>&1 + + if ! is_server_healthy "$port"; then + log "error: ${backend} server failed to become healthy" + return 1 + fi + log "${backend} server is healthy on port ${port}" +} + +run_warmup() { + local backend="$1" + local port input_len output_len num + if [[ "$backend" == "sglang" ]]; then + port="$SGLANG_PORT" + else + port="$VLLM_PORT" + fi + + # Warmup with a moderate 64k request to exercise the long-context path. + input_len=65536 + output_len=256 + num=1 + + log "warming up ${backend} (input=${input_len}, output=${output_len}, num=${num})" + "${VENV_CLIENT}/bin/python" "${SCRIPT_DIR}/warmup.py" \ + --backend "$backend" \ + --port "$port" \ + --input-len "$input_len" \ + --output-len "$output_len" \ + --num "$num" \ + --env-python "${VENV_CLIENT}/bin/python" \ + >> "${log_dir_global}/${backend}.warmup.log" 2>&1 + log "warmup for ${backend} completed" +} + +scenario_already_completed() { + local output_file="$1" + local expected="$2" + [[ -s "$output_file" ]] || return 1 + local completed + completed="$(${VENV_CLIENT}/bin/python -c " +import json, sys +path = sys.argv[1] +try: + with open(path, 'r', encoding='utf-8') as f: + for line in f: + line = line.strip() + if line: + data = json.loads(line) + print(data.get('completed', 0)) + break +except Exception: + print(0) +" "$output_file")" + [[ "${completed:-0}" -ge "$expected" ]] +} + +run_benchmark() { + local backend="$1" + local result_root="${RESULT_BASE}/${RUN_ID}/${backend}" + local raw_dir="${result_root}/raw_outputs" + local phase_log_dir="${result_root}/logs" + + mkdir -p "$raw_dir" "$phase_log_dir" + + local port + if [[ "$backend" == "sglang" ]]; then + port="$SGLANG_PORT" + else + port="$VLLM_PORT" + fi + + log "===== ${backend} START (max_model_len=${MAX_MODEL_LEN}) =====" + + stop_server "$backend" + start_server "$backend" + run_warmup "$backend" + + for scenario in "${SCENARIOS[@]}"; do + read -r concurrency input_len output_len num_prompts <<< "$scenario" + output_file="${raw_dir}/${backend}_64k_$(date '+%m%d')_${concurrency}_${input_len}_${output_len}.jsonl" + detail_log="${phase_log_dir}/${backend}_64k_c${concurrency}_i${input_len}_o${output_len}.log" + + if scenario_already_completed "$output_file" "$num_prompts"; then + log "skipping already-completed ${backend} scenario: c=${concurrency} i=${input_len} o=${output_len}" + continue + fi + + log "running ${backend} scenario: c=${concurrency} i=${input_len} o=${output_len} n=${num_prompts}" + + "${VENV_CLIENT}/bin/python" -m sglang.bench_serving \ + --backend "$backend" \ + --host 127.0.0.1 \ + --port "$port" \ + --dataset-name random \ + --random-input-len "$input_len" \ + --random-output-len "$output_len" \ + --num-prompts "$num_prompts" \ + --max-concurrency "$concurrency" \ + --request-rate 10000 \ + --output-file "$output_file" \ + --output-details \ + > "$detail_log" 2>&1 || { + log "ERROR: ${backend} scenario c=${concurrency} i=${input_len} o=${output_len} failed; see ${detail_log}" + continue + } + + log "finished ${backend} scenario: output=${output_file}" + done + + stop_server "$backend" + log "===== ${backend} DONE =====" +} + +parse_backend() { + local backend="$1" + local result_root="${RESULT_BASE}/${RUN_ID}/${backend}" + log "parsing ${backend} results in ${result_root}" + "${VENV_CLIENT}/bin/python" "${SCRIPT_DIR}/parse_backend.py" "$result_root" --backend "$backend" \ + >> "${result_root}/logs/parse.log" 2>&1 || { + log "WARNING: parser failed for ${backend}; see ${result_root}/logs/parse.log" + } +} + +write_backend_metadata() { + local backend="$1" + local result_root="${RESULT_BASE}/${RUN_ID}/${backend}" + ensure_result_root "$result_root" + local meta_json="${result_root}/results.json" + + local env_path + if [[ "$backend" == "sglang" ]]; then + env_path="$VENV_SGLANG" + else + env_path="$VENV_VLLM" + fi + + # Capture the exact server launch args for reproducibility. + local server_args + if [[ "$backend" == "sglang" ]]; then + server_args="sglang serve --trust-remote-code --model-path $MODEL_PATH --tp $TP --moe-runner-backend marlin --context-length $MAX_MODEL_LEN --max-running-requests $MAX_RUNNING --mem-fraction-static 0.88 --host 0.0.0.0 --port $SGLANG_PORT" + else + server_args="vllm serve $MODEL_PATH --trust-remote-code --tensor-parallel-size $TP --kv-cache-dtype fp8 --max-model-len $MAX_MODEL_LEN --max-num-seqs $MAX_NUM_SEQS --block-size 256 --gpu-memory-utilization 0.90 --tokenizer-mode deepseek_v4 --reasoning-parser deepseek_v4 --no-disable-hybrid-kv-cache-manager --disable-uvicorn-access-log --port $VLLM_PORT" + fi + + write_metadata_json \ + "$meta_json" \ + "${EXPERIMENT_NAME}_${backend}" \ + "$RUN_ID" \ + "$MODEL_PATH" \ + "$backend" \ + "$backend" \ + "$HARDWARE" \ + "$ACCELERATOR" \ + "$CHIP" \ + "experiments/${EXPERIMENT_NAME}/run_bench.sh" \ + "$env_path" \ + "H200 64k-context ${backend} TP=8 comparison for DeepSeek-V4-Flash" + + # Embed config and server args. + jq --arg backend "$backend" \ + --arg server_args "$server_args" \ + '.config = { + "tp": 8, + "cuda_visible_devices": "0,1,2,3,4,5,6,7", + "max_model_len": 70000, + "backend": $backend, + "server_start_script": "experiments/dsv4_h200_64k_sglang_vs_vllm/start_\($backend).sh", + "server_args": $server_args + }' "$meta_json" > "${meta_json}.tmp" && mv "${meta_json}.tmp" "$meta_json" +} + +# --------------------------------------------------------------------------- +# Main +# --------------------------------------------------------------------------- + +# Cleanup any leftovers. +stop_server sglang +stop_server vllm + +write_backend_metadata sglang +write_backend_metadata vllm + +# Run SGLang. +run_benchmark sglang +parse_backend sglang + +# Run vLLM. +run_benchmark vllm +parse_backend vllm + +# Generate comparison. +log "generating comparison report" +"${VENV_CLIENT}/bin/python" "${SCRIPT_DIR}/compare.py" \ + --sglang "${RESULT_BASE}/${RUN_ID}/sglang" \ + --vllm "${RESULT_BASE}/${RUN_ID}/vllm" \ + --output "${RESULT_BASE}/${RUN_ID}/comparison.md" \ + >> "${log_dir_global}/compare.log" 2>&1 || { + log "WARNING: comparison script failed; see ${log_dir_global}/compare.log" + } + +log "all results saved to ${RESULT_BASE}/${RUN_ID}" diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/start_sglang.sh b/experiments/dsv4_h200_64k_sglang_vs_vllm/start_sglang.sh new file mode 100755 index 0000000..bfa03e2 --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/start_sglang.sh @@ -0,0 +1,62 @@ +#!/bin/bash +# Start SGLang server for the 64k-context comparison. +set -e + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +# shellcheck source=/dev/null +source "${SCRIPT_DIR}/config.env" + +cd /data/user1/yy +mkdir -p logs + +export PATH="${VENV_SGLANG}/bin:$PATH" +export PYTHONUNBUFFERED=1 +export SGLANG_LOG_LEVEL=info +export TMPDIR=/data/user1/yy/tmp +export CUDA_VISIBLE_DEVICES="${CUDA_VISIBLE_DEVICES}" + +LOG="/data/user1/yy/logs/dsv4_h200_64k_sglang_vs_vllm_sglang_$(date +%Y%m%d_%H%M%S).log" +PID_FILE="/data/user1/yy/dsv4_h200_64k_sglang_vs_vllm_sglang.pid" + +rm -f "$PID_FILE" + +echo "=== Starting SGLang server (TP=$TP, max_model_len=$MAX_MODEL_LEN, max_running=$MAX_RUNNING) ===" +echo "Model: $MODEL_PATH" +echo "Port: $SGLANG_PORT" +echo "Log: $LOG" + +nohup sglang serve \ + --trust-remote-code \ + --model-path "$MODEL_PATH" \ + --tp "$TP" \ + --moe-runner-backend marlin \ + --context-length "$MAX_MODEL_LEN" \ + --max-running-requests "$MAX_RUNNING" \ + --mem-fraction-static 0.88 \ + --host 0.0.0.0 \ + --port "$SGLANG_PORT" \ + > "$LOG" 2>&1 & + +PID=$! +echo $PID > "$PID_FILE" +echo "PID: $PID" +echo "Waiting for health..." + +for i in $(seq 1 240); do + if curl --fail --silent --show-error --max-time 5 "http://127.0.0.1:${SGLANG_PORT}/health" >/dev/null 2>&1; then + echo "SGLang server is ready at http://127.0.0.1:${SGLANG_PORT}" + echo "Log: $LOG" + exit 0 + fi + if ! kill -0 $PID 2>/dev/null; then + echo "ERROR: SGLang server exited early" + tail -200 "$LOG" + exit 1 + fi + echo "Waiting... ($i/240)" + sleep 5 +done + +echo "ERROR: SGLang server not healthy after 240 retries" +tail -200 "$LOG" +exit 1 diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/start_vllm.sh b/experiments/dsv4_h200_64k_sglang_vs_vllm/start_vllm.sh new file mode 100755 index 0000000..e915bae --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/start_vllm.sh @@ -0,0 +1,65 @@ +#!/bin/bash +# Start vLLM server for the 64k-context comparison. +set -e + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +# shellcheck source=/dev/null +source "${SCRIPT_DIR}/config.env" + +cd /data/user1/yy +mkdir -p logs + +VENV="${VENV_VLLM}" +export PATH="$VENV/bin:$PATH" +export PYTHONUNBUFFERED=1 +export TMPDIR=/data/user1/yy/tmp +export CUDA_VISIBLE_DEVICES="${CUDA_VISIBLE_DEVICES}" + +LOG="/data/user1/yy/logs/dsv4_h200_64k_sglang_vs_vllm_vllm_$(date +%Y%m%d_%H%M%S).log" +PID_FILE="/data/user1/yy/dsv4_h200_64k_sglang_vs_vllm_vllm.pid" + +rm -f "$PID_FILE" + +echo "=== Starting vLLM server (TP=$TP, max_model_len=$MAX_MODEL_LEN, max_num_seqs=$MAX_NUM_SEQS) ===" +echo "Model: $MODEL_PATH" +echo "Port: $VLLM_PORT" +echo "Log: $LOG" + +nohup vllm serve "$MODEL_PATH" \ + --trust-remote-code \ + --tensor-parallel-size "$TP" \ + --kv-cache-dtype fp8 \ + --max-model-len "$MAX_MODEL_LEN" \ + --max-num-seqs "$MAX_NUM_SEQS" \ + --block-size 256 \ + --gpu-memory-utilization 0.90 \ + --tokenizer-mode deepseek_v4 \ + --reasoning-parser deepseek_v4 \ + --no-disable-hybrid-kv-cache-manager \ + --disable-uvicorn-access-log \ + --port "$VLLM_PORT" \ + > "$LOG" 2>&1 & + +PID=$! +echo $PID > "$PID_FILE" +echo "PID: $PID" +echo "Waiting for health..." + +for i in $(seq 1 240); do + if curl --fail --silent --show-error --max-time 5 "http://127.0.0.1:${VLLM_PORT}/health" >/dev/null 2>&1; then + echo "vLLM server is ready at http://127.0.0.1:${VLLM_PORT}" + echo "Log: $LOG" + exit 0 + fi + if ! kill -0 $PID 2>/dev/null; then + echo "ERROR: vLLM server exited early" + tail -200 "$LOG" + exit 1 + fi + echo "Waiting... ($i/240)" + sleep 5 +done + +echo "ERROR: vLLM server not healthy after 240 retries" +tail -200 "$LOG" +exit 1 diff --git a/experiments/dsv4_h200_64k_sglang_vs_vllm/warmup.py b/experiments/dsv4_h200_64k_sglang_vs_vllm/warmup.py new file mode 100755 index 0000000..75dbe5f --- /dev/null +++ b/experiments/dsv4_h200_64k_sglang_vs_vllm/warmup.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +"""Send a small number of warmup requests to a running backend. + +Uses sglang.bench_serving with a single request so that the same code path +(prefill / decode kernels, CUDA graphs, etc.) is exercised before the real +benchmark begins. Discards the output. +""" +import argparse +import subprocess +import sys +import tempfile +from pathlib import Path + + +def run_warmup(backend: str, host: str, port: int, input_len: int, output_len: int, num: int, env_python: Path) -> None: + with tempfile.NamedTemporaryFile(mode="w", suffix=".jsonl", delete=True) as tmp: + cmd = [ + str(env_python), + "-m", + "sglang.bench_serving", + "--backend", + backend, + "--host", + host, + "--port", + str(port), + "--dataset-name", + "random", + "--random-input-len", + str(input_len), + "--random-output-len", + str(output_len), + "--num-prompts", + str(num), + "--max-concurrency", + "1", + "--request-rate", + "10000", + "--output-file", + tmp.name, + "--output-details", + ] + print(f"[warmup] {' '.join(cmd)}", flush=True) + result = subprocess.run(cmd, capture_output=True, text=True) + if result.returncode != 0: + print("[warmup] FAILED", file=sys.stderr) + print(result.stdout, file=sys.stderr) + print(result.stderr, file=sys.stderr) + sys.exit(1) + print(f"[warmup] OK: backend={backend} port={port} input={input_len} output={output_len} num={num}") + + +def main(): + parser = argparse.ArgumentParser(description="Warmup a serving backend.") + parser.add_argument("--backend", required=True, choices=["sglang", "vllm"]) + parser.add_argument("--port", type=int, required=True) + parser.add_argument("--input-len", type=int, required=True) + parser.add_argument("--output-len", type=int, required=True) + parser.add_argument("--num", type=int, default=1) + parser.add_argument("--host", default="127.0.0.1") + parser.add_argument("--env-python", default="/data/user1/yy/envs/sglang/bin/python") + args = parser.parse_args() + + run_warmup( + backend=args.backend, + host=args.host, + port=args.port, + input_len=args.input_len, + output_len=args.output_len, + num=args.num, + env_python=Path(args.env_python), + ) + + +if __name__ == "__main__": + main()