44 lines
1.5 KiB
Bash
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# TP×DP matrix experiment for DeepSeek-V4-Flash on H200 (8 GPUs).
# Tests vLLM with three parallel configurations:
# TP=2, DP=4 -> 2 GPUs per replica, 4 replicas
# TP=4, DP=2 -> 4 GPUs per replica, 2 replicas
# TP=8, DP=1 -> 8 GPUs, no data parallelism
EXPERIMENT="dsv4_h200_vllm_tp_dp_matrix"
MODEL_NAME="DeepSeek-V4-Flash"
MODEL_PATH="/data/models/DeepSeek-V4-Flash"
SERVED_MODEL_NAME="deepseek-v4-flash"
VLLM_PORT="${VLLM_PORT:-30030}"
# vLLM 0.24.0 multi-port DP supervisor listens on this hard-coded port.
VLLM_DP_SUPERVISOR_PORT="${VLLM_DP_SUPERVISOR_PORT:-9256}"
VENV_VLLM="${VENV_VLLM:-/data/user1/yy/envs/vllm}"
VENV_SGLANG="${VENV_SGLANG:-/data/user1/yy/envs/sglang}"
# The benchmark client is sglang.bench_serving, even when the backend is vLLM.
VENV_CLIENT="${VENV_CLIENT:-$VENV_SGLANG}"
export CUDA_VISIBLE_DEVICES="${CUDA_VISIBLE_DEVICES:-0,1,2,3,4,5,6,7}"
# Parallel configurations to test. Format: "TP DP"
declare -a PARALLEL_CONFIGS=(
"2 4"
"4 2"
"8 1"
)
# Matrix and concurrency rules are defined in matrix.json.
# This file is consumed by generate_scenarios.py.
MATRIX_FILE="${SCRIPT_DIR:-.}/matrix.json"
MATRIX_MODE="${MATRIX_MODE:-Y}"
# Sampling density for concurrency. 0 means use the default heuristic in
# generate_scenarios.py (6-8 points, or all integers when range is small).
CONCURRENCY_SAMPLES="${CONCURRENCY_SAMPLES:-0}"
# Per-scenario timeout to avoid hangs (seconds).
SCENARIO_TIMEOUT_S="${SCENARIO_TIMEOUT_S:-1800}"
# GPU memory sampling interval (seconds).
GPU_MEM_SAMPLE_INTERVAL_S="${GPU_MEM_SAMPLE_INTERVAL_S:-1}"