dsv4_h200_vllm_tp_dp_matrix: use minimal official vLLM args, no EP, Y-only, exclude 1M

This commit is contained in:
yy-fighting 2026-07-09 10:01:51 +00:00
parent 04a384b265
commit 15b5489535
4 changed files with 222 additions and 52 deletions

View File

@ -0,0 +1,29 @@
{
"metadata": {
"experiment": "dsv4_h200_vllm_tp_dp_matrix_tp2_dp4",
"run_id": "20260709-095536_smoke",
"timestamp": "2026-07-09T09:55:43+00:00",
"model": "/data/models/DeepSeek-V4-Flash",
"backend": "vllm",
"engine": "vllm",
"hardware": "8x NVIDIA H200 143GB",
"accelerator": "NVIDIA H200",
"chip": "nvidia_h200",
"script": "experiments/dsv4_h200_vllm_tp_dp_matrix/run_bench.sh",
"env": "/data/user1/yy/envs/vllm",
"git_commit": "ec85ef1",
"git_dirty": "dirty",
"description": "H200 vLLM TP×DP matrix for DeepSeek-V4-Flash"
},
"config": {
"tp": 2,
"dp": 4,
"cuda_visible_devices": "0,1,2,3,4,5,6,7",
"backend": "vllm",
"server_start_script": "experiments/dsv4_h200_vllm_tp_dp_matrix/start_vllm_dp.sh",
"server_args_per_isl": {
"1024": "vllm serve /data/models/DeepSeek-V4-Flash --trust-remote-code --tensor-parallel-size 2 --kv-cache-dtype fp8 --max-model-len 5184 --max-num-seqs 136 --block-size 256 --gpu-memory-utilization 0.90 --tokenizer-mode deepseek_v4 --reasoning-parser deepseek_v4 --no-disable-hybrid-kv-cache-manager --disable-uvicorn-access-log --port 30030 --data-parallel-size 4 --data-parallel-size-local 4 --data-parallel-multi-port-external-lb "
}
},
"scenarios": []
}

View File

@ -0,0 +1,166 @@
mark input_len output_len concurrency num_prompts
Y 1024 128 1 5
Y 1024 128 22 110
Y 1024 128 43 215
Y 1024 128 65 325
Y 1024 128 86 430
Y 1024 128 107 535
Y 1024 128 128 640
Y 1024 256 1 5
Y 1024 256 22 110
Y 1024 256 43 215
Y 1024 256 65 325
Y 1024 256 86 430
Y 1024 256 107 535
Y 1024 256 128 640
Y 1024 512 1 5
Y 1024 512 22 110
Y 1024 512 43 215
Y 1024 512 65 325
Y 1024 512 86 430
Y 1024 512 107 535
Y 1024 512 128 640
Y 1024 1024 1 5
Y 1024 1024 22 110
Y 1024 1024 43 215
Y 1024 1024 65 325
Y 1024 1024 86 430
Y 1024 1024 107 535
Y 1024 1024 128 640
Y 1024 2048 1 5
Y 1024 2048 22 110
Y 1024 2048 43 215
Y 1024 2048 65 325
Y 1024 2048 86 430
Y 1024 2048 107 535
Y 1024 2048 128 640
Y 1024 4096 1 5
Y 1024 4096 22 110
Y 1024 4096 43 215
Y 1024 4096 65 325
Y 1024 4096 86 430
Y 1024 4096 107 535
Y 1024 4096 128 640
Y 4096 128 1 5
Y 4096 128 11 55
Y 4096 128 22 110
Y 4096 128 33 165
Y 4096 128 43 215
Y 4096 128 53 265
Y 4096 128 64 320
Y 4096 256 1 5
Y 4096 256 11 55
Y 4096 256 22 110
Y 4096 256 33 165
Y 4096 256 43 215
Y 4096 256 53 265
Y 4096 256 64 320
Y 4096 512 1 5
Y 4096 512 11 55
Y 4096 512 22 110
Y 4096 512 33 165
Y 4096 512 43 215
Y 4096 512 53 265
Y 4096 512 64 320
Y 4096 1024 1 5
Y 4096 1024 11 55
Y 4096 1024 22 110
Y 4096 1024 33 165
Y 4096 1024 43 215
Y 4096 1024 53 265
Y 4096 1024 64 320
Y 4096 2048 1 5
Y 4096 2048 11 55
Y 4096 2048 22 110
Y 4096 2048 33 165
Y 4096 2048 43 215
Y 4096 2048 53 265
Y 4096 2048 64 320
Y 4096 4096 1 5
Y 4096 4096 11 55
Y 4096 4096 22 110
Y 4096 4096 33 165
Y 4096 4096 43 215
Y 4096 4096 53 265
Y 4096 4096 64 320
Y 16384 128 1 5
Y 16384 128 6 30
Y 16384 128 11 55
Y 16384 128 17 85
Y 16384 128 22 110
Y 16384 128 27 135
Y 16384 128 32 160
Y 16384 256 1 5
Y 16384 256 6 30
Y 16384 256 11 55
Y 16384 256 17 85
Y 16384 256 22 110
Y 16384 256 27 135
Y 16384 256 32 160
Y 16384 512 1 5
Y 16384 512 6 30
Y 16384 512 11 55
Y 16384 512 17 85
Y 16384 512 22 110
Y 16384 512 27 135
Y 16384 512 32 160
Y 16384 1024 1 5
Y 16384 1024 6 30
Y 16384 1024 11 55
Y 16384 1024 17 85
Y 16384 1024 22 110
Y 16384 1024 27 135
Y 16384 1024 32 160
Y 16384 2048 1 5
Y 16384 2048 6 30
Y 16384 2048 11 55
Y 16384 2048 17 85
Y 16384 2048 22 110
Y 16384 2048 27 135
Y 16384 2048 32 160
Y 65536 128 1 5
Y 65536 128 2 10
Y 65536 128 3 15
Y 65536 128 5 25
Y 65536 128 6 30
Y 65536 128 7 35
Y 65536 128 8 40
Y 65536 256 1 5
Y 65536 256 2 10
Y 65536 256 3 15
Y 65536 256 5 25
Y 65536 256 6 30
Y 65536 256 7 35
Y 65536 256 8 40
Y 65536 512 1 5
Y 65536 512 2 10
Y 65536 512 3 15
Y 65536 512 5 25
Y 65536 512 6 30
Y 65536 512 7 35
Y 65536 512 8 40
Y 65536 1024 1 5
Y 65536 1024 2 10
Y 65536 1024 3 15
Y 65536 1024 5 25
Y 65536 1024 6 30
Y 65536 1024 7 35
Y 65536 1024 8 40
Y 131072 128 1 5
Y 131072 128 2 10
Y 131072 128 3 15
Y 131072 128 4 20
Y 131072 256 1 5
Y 131072 256 2 10
Y 131072 256 3 15
Y 131072 256 4 20
Y 131072 512 1 5
Y 131072 512 2 10
Y 131072 512 3 15
Y 131072 512 4 20
Y 262144 128 1 5
Y 262144 128 2 10
Y 262144 256 1 5
Y 262144 256 2 10
Y 524288 128 1 5
Y 524288 128 2 10
1 mark input_len output_len concurrency num_prompts
2 Y 1024 128 1 5
3 Y 1024 128 22 110
4 Y 1024 128 43 215
5 Y 1024 128 65 325
6 Y 1024 128 86 430
7 Y 1024 128 107 535
8 Y 1024 128 128 640
9 Y 1024 256 1 5
10 Y 1024 256 22 110
11 Y 1024 256 43 215
12 Y 1024 256 65 325
13 Y 1024 256 86 430
14 Y 1024 256 107 535
15 Y 1024 256 128 640
16 Y 1024 512 1 5
17 Y 1024 512 22 110
18 Y 1024 512 43 215
19 Y 1024 512 65 325
20 Y 1024 512 86 430
21 Y 1024 512 107 535
22 Y 1024 512 128 640
23 Y 1024 1024 1 5
24 Y 1024 1024 22 110
25 Y 1024 1024 43 215
26 Y 1024 1024 65 325
27 Y 1024 1024 86 430
28 Y 1024 1024 107 535
29 Y 1024 1024 128 640
30 Y 1024 2048 1 5
31 Y 1024 2048 22 110
32 Y 1024 2048 43 215
33 Y 1024 2048 65 325
34 Y 1024 2048 86 430
35 Y 1024 2048 107 535
36 Y 1024 2048 128 640
37 Y 1024 4096 1 5
38 Y 1024 4096 22 110
39 Y 1024 4096 43 215
40 Y 1024 4096 65 325
41 Y 1024 4096 86 430
42 Y 1024 4096 107 535
43 Y 1024 4096 128 640
44 Y 4096 128 1 5
45 Y 4096 128 11 55
46 Y 4096 128 22 110
47 Y 4096 128 33 165
48 Y 4096 128 43 215
49 Y 4096 128 53 265
50 Y 4096 128 64 320
51 Y 4096 256 1 5
52 Y 4096 256 11 55
53 Y 4096 256 22 110
54 Y 4096 256 33 165
55 Y 4096 256 43 215
56 Y 4096 256 53 265
57 Y 4096 256 64 320
58 Y 4096 512 1 5
59 Y 4096 512 11 55
60 Y 4096 512 22 110
61 Y 4096 512 33 165
62 Y 4096 512 43 215
63 Y 4096 512 53 265
64 Y 4096 512 64 320
65 Y 4096 1024 1 5
66 Y 4096 1024 11 55
67 Y 4096 1024 22 110
68 Y 4096 1024 33 165
69 Y 4096 1024 43 215
70 Y 4096 1024 53 265
71 Y 4096 1024 64 320
72 Y 4096 2048 1 5
73 Y 4096 2048 11 55
74 Y 4096 2048 22 110
75 Y 4096 2048 33 165
76 Y 4096 2048 43 215
77 Y 4096 2048 53 265
78 Y 4096 2048 64 320
79 Y 4096 4096 1 5
80 Y 4096 4096 11 55
81 Y 4096 4096 22 110
82 Y 4096 4096 33 165
83 Y 4096 4096 43 215
84 Y 4096 4096 53 265
85 Y 4096 4096 64 320
86 Y 16384 128 1 5
87 Y 16384 128 6 30
88 Y 16384 128 11 55
89 Y 16384 128 17 85
90 Y 16384 128 22 110
91 Y 16384 128 27 135
92 Y 16384 128 32 160
93 Y 16384 256 1 5
94 Y 16384 256 6 30
95 Y 16384 256 11 55
96 Y 16384 256 17 85
97 Y 16384 256 22 110
98 Y 16384 256 27 135
99 Y 16384 256 32 160
100 Y 16384 512 1 5
101 Y 16384 512 6 30
102 Y 16384 512 11 55
103 Y 16384 512 17 85
104 Y 16384 512 22 110
105 Y 16384 512 27 135
106 Y 16384 512 32 160
107 Y 16384 1024 1 5
108 Y 16384 1024 6 30
109 Y 16384 1024 11 55
110 Y 16384 1024 17 85
111 Y 16384 1024 22 110
112 Y 16384 1024 27 135
113 Y 16384 1024 32 160
114 Y 16384 2048 1 5
115 Y 16384 2048 6 30
116 Y 16384 2048 11 55
117 Y 16384 2048 17 85
118 Y 16384 2048 22 110
119 Y 16384 2048 27 135
120 Y 16384 2048 32 160
121 Y 65536 128 1 5
122 Y 65536 128 2 10
123 Y 65536 128 3 15
124 Y 65536 128 5 25
125 Y 65536 128 6 30
126 Y 65536 128 7 35
127 Y 65536 128 8 40
128 Y 65536 256 1 5
129 Y 65536 256 2 10
130 Y 65536 256 3 15
131 Y 65536 256 5 25
132 Y 65536 256 6 30
133 Y 65536 256 7 35
134 Y 65536 256 8 40
135 Y 65536 512 1 5
136 Y 65536 512 2 10
137 Y 65536 512 3 15
138 Y 65536 512 5 25
139 Y 65536 512 6 30
140 Y 65536 512 7 35
141 Y 65536 512 8 40
142 Y 65536 1024 1 5
143 Y 65536 1024 2 10
144 Y 65536 1024 3 15
145 Y 65536 1024 5 25
146 Y 65536 1024 6 30
147 Y 65536 1024 7 35
148 Y 65536 1024 8 40
149 Y 131072 128 1 5
150 Y 131072 128 2 10
151 Y 131072 128 3 15
152 Y 131072 128 4 20
153 Y 131072 256 1 5
154 Y 131072 256 2 10
155 Y 131072 256 3 15
156 Y 131072 256 4 20
157 Y 131072 512 1 5
158 Y 131072 512 2 10
159 Y 131072 512 3 15
160 Y 131072 512 4 20
161 Y 262144 128 1 5
162 Y 262144 128 2 10
163 Y 262144 256 1 5
164 Y 262144 256 2 10
165 Y 524288 128 1 5
166 Y 524288 128 2 10

View File

@ -15,7 +15,7 @@ source "${SCRIPT_DIR}/config.env"
RUN_ID="${RUN_ID:-$(date '+%Y%m%d-%H%M%S')}"
RESULT_BASE="${SCRIPT_DIR}/results"
MATRIX_FILE="${MATRIX_FILE:-${SCRIPT_DIR}/matrix.json}"
MATRIX_MODE="${MATRIX_MODE:-Y+P}"
MATRIX_MODE="${MATRIX_MODE:-Y}"
SCENARIO_TIMEOUT_S="${SCENARIO_TIMEOUT_S:-1800}"
GPU_MEM_SAMPLE_INTERVAL_S="${GPU_MEM_SAMPLE_INTERVAL_S:-1}"
@ -72,22 +72,14 @@ stop_server() {
build_server_args() {
local tp="$1"
local dp="$2"
local max_model_len="$3"
local max_num_seqs="$4"
local args=(
"vllm serve" "$MODEL_PATH"
--trust-remote-code
--tensor-parallel-size "$tp"
--kv-cache-dtype fp8
--max-model-len "$max_model_len"
--max-num-seqs "$max_num_seqs"
--block-size 256
--gpu-memory-utilization 0.90
--tokenizer-mode deepseek_v4
--reasoning-parser deepseek_v4
--no-disable-hybrid-kv-cache-manager
--disable-uvicorn-access-log
--tensor-parallel-size "$tp"
--no-enable-flashinfer-autotune
--port "$VLLM_PORT"
)
if [[ "$dp" -gt 1 ]]; then
@ -103,11 +95,9 @@ build_server_args() {
start_server() {
local tp="$1"
local dp="$2"
local max_model_len="$3"
local max_num_seqs="$4"
log "starting vllm server tp=${tp} dp=${dp} max_model_len=${max_model_len} max_num_seqs=${max_num_seqs}"
bash "${SCRIPT_DIR}/start_vllm_dp.sh" "$tp" "$dp" "$max_model_len" "$max_num_seqs" \
log "starting vllm server tp=${tp} dp=${dp}"
bash "${SCRIPT_DIR}/start_vllm_dp.sh" "$tp" "$dp" \
>> "${log_dir_global}/vllm_tp${tp}_dp${dp}.server.outer.log" 2>&1
local hport
@ -244,22 +234,23 @@ run_parallel_config() {
"$VENV_VLLM" \
"H200 vLLM TP×DP matrix for DeepSeek-V4-Flash"
# Embed static config now; server_args will be updated per ISL group.
jq --arg tp "$tp" --arg dp "$dp" --arg cuda "$CUDA_VISIBLE_DEVICES" \
# Embed static config and the exact server command line used.
local server_args_str
server_args_str="$(build_server_args "$tp" "$dp")"
jq --arg tp "$tp" --arg dp "$dp" --arg cuda "$CUDA_VISIBLE_DEVICES" --arg args "$server_args_str" \
'.config = {
"tp": ($tp | tonumber),
"dp": ($dp | tonumber),
"cuda_visible_devices": $cuda,
"backend": "vllm",
"server_start_script": "experiments/'${EXPERIMENT_NAME}'/start_vllm_dp.sh"
"server_start_script": "experiments/'${EXPERIMENT_NAME}'/start_vllm_dp.sh",
"server_args": $args
}' "${result_root}/results.json" > "${result_root}/results.json.tmp" && \
mv "${result_root}/results.json.tmp" "${result_root}/results.json"
# Group scenarios by input_len. We start one server per input_len with a
# max_model_len large enough for the biggest output_len in that group.
# Group scenarios by input_len. We restart the server for each ISL to keep
# memory state isolated and to warm up per context length.
local current_isl=""
local max_dsl_for_isl=0
local max_conc_for_isl=0
local group_started=false
tail -n +2 "$scenario_tsv" | while IFS=$'\t' read -r mark isl dsl conc num; do
@ -269,28 +260,15 @@ run_parallel_config() {
stop_server "$tp" "$dp"
fi
current_isl="$isl"
max_dsl_for_isl="$(awk -F'\t' -v isl="$isl" '$2==isl {if($3>max) max=$3} END{print max+0}' "$scenario_tsv")"
max_conc_for_isl="$(awk -F'\t' -v isl="$isl" '$2==isl {if($4>max) max=$4} END{print max+0}' "$scenario_tsv")"
local max_model_len=$((isl + max_dsl_for_isl + 64))
local max_num_seqs=$((max_conc_for_isl + 8))
stop_server "$tp" "$dp"
if ! start_server "$tp" "$dp" "$max_model_len" "$max_num_seqs"; then
if ! start_server "$tp" "$dp"; then
log "ERROR: ${config_label} failed to start for ISL=${isl}; skipping this group"
group_started=false
continue
fi
group_started=true
# Update metadata with the actual server args for this ISL group.
local server_args_str
server_args_str="$(build_server_args "$tp" "$dp" "$max_model_len" "$max_num_seqs")"
jq --arg isl "$isl" --arg args "$server_args_str" \
'.config.server_args_per_isl += {($isl): $args}' \
"${result_root}/results.json" > "${result_root}/results.json.tmp" && \
mv "${result_root}/results.json.tmp" "${result_root}/results.json"
# Warmup with the shortest output for this ISL.
run_warmup "$tp" "$dp" "$isl" 128
fi

View File

@ -1,12 +1,18 @@
#!/usr/bin/env bash
# Start vLLM server for a given TP×DP configuration.
# Usage: start_vllm_dp.sh <TP> <DP> <max_model_len> <max_num_seqs>
# Usage: start_vllm_dp.sh <TP> <DP>
#
# Uses the minimal argument set from the official DeepSeek-V4-Flash recipe:
# --trust-remote-code
# --kv-cache-dtype fp8
# --block-size 256
# --tensor-parallel-size <TP>
# --no-enable-flashinfer-autotune
# plus DP-related flags when DP > 1.
set -e
TP="${1}"
DP="${2}"
MAX_MODEL_LEN="${3}"
MAX_NUM_SEQS="${4}"
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# shellcheck source=/dev/null
@ -26,20 +32,13 @@ PID_FILE="/data/user1/yy/${EXPERIMENT}_vllm_tp${TP}_dp${DP}.pid"
rm -f "$PID_FILE"
# Build the base command as an array.
SERVER_ARGS=(
vllm serve "$MODEL_PATH"
--trust-remote-code
--tensor-parallel-size "$TP"
--kv-cache-dtype fp8
--max-model-len "$MAX_MODEL_LEN"
--max-num-seqs "$MAX_NUM_SEQS"
--block-size 256
--gpu-memory-utilization 0.90
--tokenizer-mode deepseek_v4
--reasoning-parser deepseek_v4
--no-disable-hybrid-kv-cache-manager
--disable-uvicorn-access-log
--tensor-parallel-size "$TP"
--no-enable-flashinfer-autotune
--port "$VLLM_PORT"
)
@ -49,9 +48,7 @@ if [[ "$DP" -gt 1 ]]; then
--data-parallel-size-local "$DP"
--data-parallel-multi-port-external-lb
)
# vLLM 0.24.0 hard-codes the DP supervisor port to 9256 in
# entrypoints/openai/cli_args.py. The bench client and health checks must
# target that port when DP > 1.
# vLLM 0.24.0 hard-codes the DP supervisor port to 9256.
HEALTH_PORT="$VLLM_DP_SUPERVISOR_PORT"
else
HEALTH_PORT="$VLLM_PORT"
@ -59,7 +56,7 @@ fi
SERVER_ARGS_STR="${SERVER_ARGS[*]}"
echo "=== Starting vLLM server (TP=${TP}, DP=${DP}, max_model_len=${MAX_MODEL_LEN}, max_num_seqs=${MAX_NUM_SEQS}) ==="
echo "=== Starting vLLM server (TP=${TP}, DP=${DP}) ==="
echo "Model: $MODEL_PATH"
echo "Health port: $HEALTH_PORT"
echo "Command: $SERVER_ARGS_STR"