79 lines
3.1 KiB
Bash
79 lines
3.1 KiB
Bash
# Tiny adaptive-concurrency experiment for DeepSeek-V4-Flash on RTX 6000D
|
|
# (8 GPUs) using vLLM. It fixes OSL=1024 and sweeps ISL=1K..128K.
|
|
# Tests vLLM with three parallel configurations:
|
|
# TP=2, DP=4 -> 2 GPUs per replica, 4 replicas
|
|
# TP=4, DP=2 -> 4 GPUs per replica, 2 replicas
|
|
# TP=8, DP=1 -> 8 GPUs, no data parallelism
|
|
|
|
EXPERIMENT="dsv4_pro6000_vllm_tiny_1k_output"
|
|
MODEL_NAME="DeepSeek-V4-Flash"
|
|
MODEL_PATH="/data/6000D/DeepSeek-V4-Flash"
|
|
SERVED_MODEL_NAME="deepseek-v4-flash"
|
|
|
|
VLLM_PORT="${VLLM_PORT:-30030}"
|
|
|
|
# Python interpreter for orchestration scripts (parse_backend.py, compare.py, etc.)
|
|
# and the benchmark client. Defaults to the system python3 if the sglang venv
|
|
# does not exist on the host.
|
|
VENV_CLIENT="${VENV_CLIENT:-/root/.miniconda3/envs/sglang}"
|
|
|
|
# Run the benchmark client natively (0) or inside Docker (1).
|
|
USE_DOCKER_CLIENT="${USE_DOCKER_CLIENT:-1}"
|
|
|
|
export CUDA_VISIBLE_DEVICES="${CUDA_VISIBLE_DEVICES:-0,1,2,3,4,5,6,7}"
|
|
|
|
# Runtime working directory for logs, pid files, and tmp. Defaults to a local
|
|
# directory under this experiment so the benchmark is self-contained.
|
|
RUNTIME_BASE="${RUNTIME_BASE:-${SCRIPT_DIR}/runtime}"
|
|
|
|
# Parallel configurations to test. Format: "TP DP"
|
|
declare -a PARALLEL_CONFIGS=(
|
|
"2 4"
|
|
"4 2"
|
|
"8 1"
|
|
)
|
|
|
|
# vLLM server settings, verified for this model and machine.
|
|
GPU_MEMORY_UTILIZATION="${GPU_MEMORY_UTILIZATION:-0.9}"
|
|
KV_CACHE_DTYPE="${KV_CACHE_DTYPE:-fp8}"
|
|
BLOCK_SIZE="${BLOCK_SIZE:-256}"
|
|
MAX_MODEL_LEN="${MAX_MODEL_LEN:-131072}"
|
|
MAX_NUM_SEQS="${MAX_NUM_SEQS:-128}"
|
|
|
|
# Deployment switch. 0 = native vllm venv, 1 = Docker.
|
|
USE_DOCKER="${USE_DOCKER:-1}"
|
|
DOCKER_IMAGE="${DOCKER_IMAGE:-vllm-sm120-dsv4:0.25.1-fi0.6.14}"
|
|
|
|
# Benchmark client Docker image. vLLM's image does not include the SGLang benchmark client,
|
|
# so the client runs inside the SGLang image and targets the vLLM backend via the
|
|
# OpenAI-compatible API.
|
|
DOCKER_CLIENT_IMAGE="${DOCKER_CLIENT_IMAGE:-sglang-sm120-dsv4:0.5.15.post1-fi0.6.14-sm120fix1}"
|
|
|
|
# To use ShareGPT, set BENCH_DATASET_NAME=random and DATASET_PATH explicitly.
|
|
BENCH_DATASET_NAME="${BENCH_DATASET_NAME:-random}"
|
|
DATASET_PATH="${DATASET_PATH:-${ROOT_DIR}/dataset/ShareGPT_V3_unfiltered_cleaned_split.json}"
|
|
SGLANG_BENCH_MODULE="${SGLANG_BENCH_MODULE:-sglang.benchmark.serving}"
|
|
CACHE_DIR="${CACHE_DIR:-${ROOT_DIR}/vllm_sm120_cache}"
|
|
|
|
# Matrix contains ISL=1K..128K with fixed OSL=1K.
|
|
MATRIX_FILE="${MATRIX_FILE:-${SCRIPT_DIR:-.}/matrix.json}"
|
|
MATRIX_MODE="${MATRIX_MODE:-Y}"
|
|
|
|
# Sampling density for concurrency.
|
|
# 0 = use the default heuristic in generate_scenarios.py (6-8 points).
|
|
# 2 = only test the low and high endpoints.
|
|
export CONCURRENCY_SAMPLES="${CONCURRENCY_SAMPLES:-2}"
|
|
|
|
# Per-scenario timeout to avoid hangs (seconds).
|
|
SCENARIO_TIMEOUT_S="${SCENARIO_TIMEOUT_S:-1800}"
|
|
|
|
# GPU memory sampling interval (seconds).
|
|
GPU_MEM_SAMPLE_INTERVAL_S="${GPU_MEM_SAMPLE_INTERVAL_S:-1}"
|
|
|
|
# Dry-run mode: if 1, only log the server args and scenario plan without starting
|
|
# any server or sending requests.
|
|
DRY_RUN="${DRY_RUN:-0}"
|
|
|
|
# Per-config scenario limit for quick smoke tests. 0 = run all generated scenarios.
|
|
GRID_LIMIT="${GRID_LIMIT:-0}"
|