- Add dsv4_h20_vllm_tp_dp_matrix experiment (vLLM backend) - Add dsv4_h20_sglang_tp_dp_matrix experiment (SGLang backend) - Add nvidia_h20 platform config with H20-specific paths - Fix platform.sh auto-detection to distinguish H20 from H200 - Fix vLLM Docker startup: remove --entrypoint override for CDI compat - Fix vLLM 0.25.1 CLI args: remove redundant 'serve' from SERVER_ARGS - Download ShareGPT dataset to local datasets dir - Rename DSL to OSL across both experiments
71 lines
2.7 KiB
Bash
71 lines
2.7 KiB
Bash
# TP×DP matrix experiment for DeepSeek-V4-Flash on H20 (8 GPUs) using SGLang.
|
||
# Tests SGLang with three parallel configurations:
|
||
# TP=2, DP=4 -> 2 GPUs per replica, 4 replicas
|
||
# TP=4, DP=2 -> 4 GPUs per replica, 2 replicas
|
||
# TP=8, DP=1 -> 8 GPUs, no data parallelism
|
||
|
||
EXPERIMENT="dsv4_h20_sglang_tp_dp_matrix"
|
||
MODEL_NAME="DeepSeek-V4-Flash"
|
||
MODEL_PATH="/data1/hf_models/DeepSeek-V4-Flash"
|
||
SERVED_MODEL_NAME="deepseek-v4-flash"
|
||
|
||
SGLANG_PORT="${SGLANG_PORT:-30031}"
|
||
|
||
# Python interpreter for orchestration scripts (parse_backend.py, compare.py, etc.)
|
||
# and the benchmark client. Defaults to the system python3 if the sglang venv
|
||
# does not exist on the host.
|
||
VENV_CLIENT="${VENV_CLIENT:-/data1/yy/sskj/envs/sglang}"
|
||
|
||
# Run the benchmark client natively (0) or inside Docker (1).
|
||
USE_DOCKER_CLIENT="${USE_DOCKER_CLIENT:-1}"
|
||
|
||
export CUDA_VISIBLE_DEVICES="${CUDA_VISIBLE_DEVICES:-0,1,2,3,4,5,6,7}"
|
||
|
||
# Runtime working directory for logs, pid files, and tmp. Defaults to a local
|
||
# directory under this experiment so the benchmark is self-contained.
|
||
RUNTIME_BASE="${RUNTIME_BASE:-${SCRIPT_DIR}/runtime}"
|
||
|
||
# Parallel configurations to test. Format: "TP DP"
|
||
declare -a PARALLEL_CONFIGS=(
|
||
"2 4"
|
||
"4 2"
|
||
"8 1"
|
||
)
|
||
|
||
# SGLang server settings
|
||
CONTEXT_LENGTH="${CONTEXT_LENGTH:-1048576}"
|
||
MAX_RUNNING_REQUESTS="${MAX_RUNNING_REQUESTS:-128}"
|
||
MEM_FRACTION_STATIC="${MEM_FRACTION_STATIC:-0.88}"
|
||
MOE_RUNNER_BACKEND="${MOE_RUNNER_BACKEND:-marlin}"
|
||
|
||
# Deployment switch. 0 = native sglang venv, 1 = Docker.
|
||
USE_DOCKER="${USE_DOCKER:-1}"
|
||
DOCKER_IMAGE="${DOCKER_IMAGE:-lmsysorg/sglang:latest}"
|
||
|
||
# Dataset used by sglang.bench_serving --dataset-name random.
|
||
# The random sampler needs a ShareGPT-style JSON file locally; it falls back to
|
||
# downloading from HuggingFace, which usually fails on offline H20 nodes.
|
||
DATASET_PATH="${DATASET_PATH:-/data1/yy/sskj/datasets/ShareGPT_V3_unfiltered_cleaned_split.json}"
|
||
|
||
# Matrix and concurrency rules are defined in matrix.json by default.
|
||
MATRIX_FILE="${MATRIX_FILE:-${SCRIPT_DIR:-.}/matrix.json}"
|
||
MATRIX_MODE="${MATRIX_MODE:-Y}"
|
||
|
||
# Sampling density for concurrency.
|
||
# 0 = use the default heuristic in generate_scenarios.py (6-8 points).
|
||
# 2 = only test the low and high endpoints.
|
||
export CONCURRENCY_SAMPLES="${CONCURRENCY_SAMPLES:-2}"
|
||
|
||
# Per-scenario timeout to avoid hangs (seconds).
|
||
SCENARIO_TIMEOUT_S="${SCENARIO_TIMEOUT_S:-1800}"
|
||
|
||
# GPU memory sampling interval (seconds).
|
||
GPU_MEM_SAMPLE_INTERVAL_S="${GPU_MEM_SAMPLE_INTERVAL_S:-1}"
|
||
|
||
# Dry-run mode: if 1, only log the server args and scenario plan without starting
|
||
# any server or sending requests.
|
||
DRY_RUN="${DRY_RUN:-0}"
|
||
|
||
# Per-config scenario limit for quick smoke tests. 0 = run all generated scenarios.
|
||
GRID_LIMIT="${GRID_LIMIT:-0}"
|