dsv4实验发现DP副本绑定到同一组die的问题(99a22f0),glm52存在相同问题: 缺--max-num-batched-tokens和--api-server-count导致DP worker设备分配异常 (不加api-server-count时vllm为N个DP rank启动N个API server)。 对齐docs.vllm.ai GLM5.2 A3官方教程: - 加 --max-num-batched-tokens 8192 (官方值,影响DP调度) - 加 --api-server-count 1 (官方值,避免多API server干扰设备分配) - 去掉 --kv-cache-dtype fp8 (官方不指定,用默认bfloat16;且此镜像fp8本就未生效) - 保留 --trust-remote-code / --enable-expert-parallel / enable_dsa_cp (官方有)
120 lines
5.7 KiB
Bash
120 lines
5.7 KiB
Bash
# TP×DP matrix experiment for GLM-5.2 on Ascend 910C (8 NPUs / 16 dies) using vLLM-Ascend.
|
||
# Tests vLLM with three parallel configurations:
|
||
# TP=2, DP=4 -> 2 dies per replica, 4 replicas
|
||
# TP=4, DP=2 -> 4 dies per replica, 2 replicas
|
||
# TP=8, DP=1 -> 8 dies, no data parallelism
|
||
#
|
||
# Platform: ascend_910c (see platforms/ascend_910c.env).
|
||
# Host: 910c.1 / NPU-NODE61, openEuler 22.03 SP4 aarch64, driver 25.5.2, CANN 9.0.0.
|
||
# Image: vllm-ascend; load the tarball from /mnt/models first (see envs/ASCEND_910C_ENV_SETUP.md).
|
||
|
||
EXPERIMENT="glm52_910c_vllm_tp_dp_matrix"
|
||
MODEL_NAME="GLM-5.2"
|
||
# GLM-5.2 ships two quantized variants on this host; w4a8c8 is the default.
|
||
# Switch to /mnt/models/GLM-5.2-w8a8 by overriding MODEL_PATH if needed.
|
||
MODEL_PATH="${MODEL_PATH:-/mnt/models/GLM-5.2-w4a8c8}"
|
||
SERVED_MODEL_NAME="glm-5.2"
|
||
|
||
VLLM_PORT="${VLLM_PORT:-30050}"
|
||
|
||
# Dedicated container name so this experiment never touches other 910c runs.
|
||
CONTAINER_NAME="vllm-ascend-glm52-910c"
|
||
|
||
# Python interpreter for the benchmark client inside the vllm-ascend container.
|
||
CONTAINER_PYTHON="/usr/local/python3.12.13/bin/python3"
|
||
|
||
# vllm-ascend image. Override with the exact tag after `docker load`-ing one of:
|
||
# /mnt/models/vllm-ascend-glm5.2-a3-openeuler.tar (GLM5.2-tuned, recommended)
|
||
# /mnt/models/vllm-ascend-v0.23.0rc1-a3-openeuler.tar (general v0.23)
|
||
USE_DOCKER="${USE_DOCKER:-1}"
|
||
DOCKER_IMAGE="${DOCKER_IMAGE:-local/vllm-ascend:0.23-a3-20260718-sglang}"
|
||
|
||
# Benchmark client Docker image. vLLM's image does not include sglang.bench_serving;
|
||
# reuse the vllm-ascend container itself for the client via `docker exec` (see
|
||
# run_adaptive_concurrency_add16.sh), so this is only used if USE_DOCKER_CLIENT=1
|
||
# with an external sglang image. Default off on 910c.
|
||
DOCKER_CLIENT_IMAGE="${DOCKER_CLIENT_IMAGE:-lmsysorg/sglang:latest}"
|
||
USE_DOCKER_CLIENT="${USE_DOCKER_CLIENT:-0}"
|
||
|
||
# Device selection. ASCEND_VISIBLE_DEVICES selects NPU cards 0..7; the Ascend
|
||
# Docker Runtime (default runtime on this host) injects the matching dies.
|
||
export ASCEND_VISIBLE_DEVICES="${ASCEND_VISIBLE_DEVICES:-0,1,2,3,4,5,6,7}"
|
||
# Keep CUDA_VISIBLE_DEVICES for parity with the shared library; vllm-ascend
|
||
# ignores it on NPU but some helper code reads it.
|
||
export CUDA_VISIBLE_DEVICES="${CUDA_VISIBLE_DEVICES:-0,1,2,3,4,5,6,7}"
|
||
|
||
# Runtime working directory for logs, pid files, and tmp.
|
||
RUNTIME_BASE="${RUNTIME_BASE:-${SCRIPT_DIR}/runtime}"
|
||
|
||
# Parallel configurations to test. Format: "TP DP"
|
||
# A3 910C has 16 dies (8 cards x 2 dies/card). TP/DP address dies, so TP*DP=16
|
||
# uses all dies. The mapping to H20 (8 cards) for fair comparison is:
|
||
# A3 TP=4 DP=4 -> 4 dies/replica x 4 replicas == H20 TP=2 DP=4 (2 cards/replica)
|
||
# A3 TP=8 DP=2 -> 8 dies/replica x 2 replicas == H20 TP=4 DP=2 (4 cards/replica)
|
||
# A3 TP=16 DP=1 -> 16 dies/replica x 1 replica == H20 TP=8 DP=1 (8 cards/replica)
|
||
# Override via PARALLEL_CONFIGS_STR="4,4 8,2 16,1" (space-separated).
|
||
if [[ -n "${PARALLEL_CONFIGS_STR:-}" ]]; then
|
||
declare -a PARALLEL_CONFIGS=()
|
||
for pair in $PARALLEL_CONFIGS_STR; do
|
||
PARALLEL_CONFIGS+=("${pair//,/ }")
|
||
done
|
||
else
|
||
declare -a PARALLEL_CONFIGS=(
|
||
"4 4"
|
||
"8 2"
|
||
"16 1"
|
||
)
|
||
fi
|
||
|
||
# vLLM-Ascend server settings for GLM-5.2 (w4a8c8).
|
||
# Notes:
|
||
# - KV cache dtype fp8 is supported on 910C; fall back to fp16 if the image rejects it.
|
||
# - block-size 128 matches Ascend page semantics (P99 of H20 uses 256; 910C favors 128).
|
||
# - MAX_MODEL_LEN: GLM-5.2 supports up to 128K context; cap at 131072.
|
||
# - gpu-memory-utilization maps to NPU HBM fraction on vllm-ascend (0.9 mirrors H20).
|
||
GPU_MEMORY_UTILIZATION="${GPU_MEMORY_UTILIZATION:-0.95}"
|
||
KV_CACHE_DTYPE="${KV_CACHE_DTYPE:-}"
|
||
BLOCK_SIZE="${BLOCK_SIZE:-128}"
|
||
MAX_MODEL_LEN="${MAX_MODEL_LEN:-131072}"
|
||
MAX_NUM_SEQS="${MAX_NUM_SEQS:-256}"
|
||
# Official A3 tutorial params (docs.vllm.ai GLM5.2): max-num-batched-tokens=8192,
|
||
# api-server-count=1 (without it, vllm spawns N API servers for N DP ranks,
|
||
# disturbing DP worker device placement -- same issue as dsv4 fix 99a22f0).
|
||
MAX_NUM_BATCHED_TOKENS="${MAX_NUM_BATCHED_TOKENS:-8192}"
|
||
API_SERVER_COUNT="${API_SERVER_COUNT:-1}"
|
||
|
||
# Per-TP parameter overrides (applied in start_vllm_docker.sh via case $TP).
|
||
# GLM-5.2-w4a8c8 ~391 GiB total; with expert-parallel expert weights are sharded
|
||
# across TP dies but dense (attn) weights are replicated. Each die has 64GB HBM.
|
||
# TP=4: experts/4 + full dense; very tight KV cache -> cap context heavily
|
||
# TP=8: experts/8 + full dense; tight KV cache (verified: max_len=16384 fits)
|
||
# TP=16: experts/16 + full dense; ample KV cache (full 128K context)
|
||
TP4_GPU_MEMORY_UTILIZATION="${TP4_GPU_MEMORY_UTILIZATION:-0.97}"
|
||
TP4_MAX_MODEL_LEN="${TP4_MAX_MODEL_LEN:-4096}"
|
||
TP4_MAX_NUM_SEQS="${TP4_MAX_NUM_SEQS:-32}"
|
||
TP8_GPU_MEMORY_UTILIZATION="${TP8_GPU_MEMORY_UTILIZATION:-0.95}"
|
||
TP8_MAX_MODEL_LEN="${TP8_MAX_MODEL_LEN:-16384}"
|
||
TP8_MAX_NUM_SEQS="${TP8_MAX_NUM_SEQS:-64}"
|
||
TP16_GPU_MEMORY_UTILIZATION="${TP16_GPU_MEMORY_UTILIZATION:-0.92}"
|
||
TP16_MAX_MODEL_LEN="${TP16_MAX_MODEL_LEN:-131072}"
|
||
TP16_MAX_NUM_SEQS="${TP16_MAX_NUM_SEQS:-256}"
|
||
|
||
# vLLM-Ascend-specific launch flags injected by start_vllm_docker.sh.
|
||
# attention backend for 910C: use the fused/atb attention path. Adjust per image.
|
||
VLLM_ASCEND_ATTENTION_BACKEND="${VLLM_ASCEND_ATTENTION_BACKEND:-atb}"
|
||
|
||
# Dataset used by sglang.bench_serving --dataset-name random.
|
||
DATASET_PATH="${DATASET_PATH:-${ROOT_DIR}/datasets/ShareGPT_V3_unfiltered_cleaned_split.json}"
|
||
|
||
# Matrix and concurrency rules are defined in matrix.json by default.
|
||
MATRIX_FILE="${MATRIX_FILE:-${SCRIPT_DIR:-.}/matrix.json}"
|
||
MATRIX_MODE="${MATRIX_MODE:-Y}"
|
||
|
||
export CONCURRENCY_SAMPLES="${CONCURRENCY_SAMPLES:-2}"
|
||
|
||
SCENARIO_TIMEOUT_S="${SCENARIO_TIMEOUT_S:-1800}"
|
||
GPU_MEM_SAMPLE_INTERVAL_S="${GPU_MEM_SAMPLE_INTERVAL_S:-1}"
|
||
|
||
DRY_RUN="${DRY_RUN:-0}"
|
||
GRID_LIMIT="${GRID_LIMIT:-0}"
|