Add P800 long-context matrix experiment (64k/128k) for DeepSeek-V4-Flash-INT8
- New experiment: experiments/dsv4_p800_long_context_matrix/ - Single backend: sglang-xpu in Docker on Kunlun P800 - Context groups: 64k (max_context_len=80000, max_running=10) and 128k (max_context_len=140000, max_running=5) - Scenarios: concurrency sweeps x output lengths 256/1024/4096 - Parameterized start_sglang.sh enforces max_running >= 2 to avoid P800 INT8 req_to_token_pool allocation failures - Reuses proven P800 INT8 environment, NSA/EAGLE settings, and bench_serving invocations from dsv4_p800_max_context_length
This commit is contained in:
parent
b1b5f50f56
commit
90eecc4ede
33
experiments/dsv4_p800_long_context_matrix/README.md
Normal file
33
experiments/dsv4_p800_long_context_matrix/README.md
Normal file
@ -0,0 +1,33 @@
|
||||
# P800 Long-Context Matrix for DeepSeek-V4-Flash-INT8
|
||||
|
||||
Long-context throughput/latency matrix for SGLang-XPU on Kunlun P800 (8x XPU).
|
||||
|
||||
## Scope
|
||||
|
||||
- Model: `DeepSeek-V4-Flash-INT8` (W8A8 INT8)
|
||||
- Backend/engine: `sglang-xpu`
|
||||
- Hardware: 8x Kunlun P800 XPU
|
||||
- Input lengths: 64k, 128k
|
||||
- Output lengths: 256, 1024, 4096
|
||||
- Concurrency: sweeps per context group (see `config.env`)
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
cd /data1/yy/sskj/experiments/dsv4_p800_long_context_matrix
|
||||
bash run_bench.sh
|
||||
```
|
||||
|
||||
Optional environment overrides:
|
||||
|
||||
```bash
|
||||
PORT=30013 RUN_ID=myrun bash run_bench.sh
|
||||
```
|
||||
|
||||
## Outputs
|
||||
|
||||
Results are saved under `results/${RUN_ID}/`:
|
||||
|
||||
- `raw_outputs/`: raw `sglang.bench_serving` JSONL files
|
||||
- `report.md`: parsed per-scenario metrics
|
||||
- `results.json`: structured metadata + scenarios
|
||||
51
experiments/dsv4_p800_long_context_matrix/config.env
Normal file
51
experiments/dsv4_p800_long_context_matrix/config.env
Normal file
@ -0,0 +1,51 @@
|
||||
# Long-context matrix for SGLang-XPU on DeepSeek-V4-Flash-INT8 (P800, TP=8).
|
||||
# Input lengths: 64k, 128k (P800 INT8 max input length is ~131072 tokens).
|
||||
# Output lengths: 256, 1k, 4k
|
||||
# Concurrency: conservative to fit XPU memory.
|
||||
|
||||
EXPERIMENT="dsv4_p800_long_context_matrix"
|
||||
MODEL_NAME="DeepSeek-V4-Flash-INT8"
|
||||
MODEL_PATH="/data1/models/DeepSeek-V4-Flash-INT8"
|
||||
SERVED_MODEL_NAME="deepseek-v4-flash-int8"
|
||||
|
||||
PORT="${PORT:-30013}"
|
||||
|
||||
# Padding added to the longest sequence length when setting server context length.
|
||||
CONTEXT_PAD="${CONTEXT_PAD:-1024}"
|
||||
|
||||
# Per-context server settings. Each group is run with its own server start.
|
||||
# Format: "input_label max_context_len max_running"
|
||||
declare -a CONTEXT_GROUPS=(
|
||||
"64k 80000 10"
|
||||
"128k 140000 5"
|
||||
)
|
||||
|
||||
# Scenarios per group: "concurrency input_len output_len num_prompts"
|
||||
declare -a SCENARIOS_64K=(
|
||||
"10 65536 256 50"
|
||||
"5 65536 256 25"
|
||||
"2 65536 256 10"
|
||||
"10 65536 1024 50"
|
||||
"5 65536 1024 25"
|
||||
"2 65536 1024 10"
|
||||
"10 65536 4096 50"
|
||||
"5 65536 4096 25"
|
||||
"2 65536 4096 10"
|
||||
)
|
||||
|
||||
declare -a SCENARIOS_128K=(
|
||||
"5 131072 256 20"
|
||||
"2 131072 256 10"
|
||||
"1 131072 256 5"
|
||||
"5 131072 1024 20"
|
||||
"2 131072 1024 10"
|
||||
"1 131072 1024 5"
|
||||
"5 131072 4096 20"
|
||||
"2 131072 4096 10"
|
||||
"1 131072 4096 5"
|
||||
)
|
||||
|
||||
CONTAINER_NAME="${CONTAINER_NAME:-sglang-dsv4-flash}"
|
||||
CONTAINER_PYTHON="${CONTAINER_PYTHON:-/root/miniconda/envs/python310_torch25_cuda/bin/python}"
|
||||
DOCKER_IMAGE="${DOCKER_IMAGE:-iregistry.baidu-int.com/xpu/sglang-p800-pd-disagg-0510:20260511_4202}"
|
||||
PATCH_ROOT="${PATCH_ROOT:-${ROOT_DIR}/platforms/patches/kunlun_p800}"
|
||||
275
experiments/dsv4_p800_long_context_matrix/run_bench.sh
Executable file
275
experiments/dsv4_p800_long_context_matrix/run_bench.sh
Executable file
@ -0,0 +1,275 @@
|
||||
#!/usr/bin/env bash
|
||||
# Long-context matrix for P800 SGLang INT8.
|
||||
set -Eeuo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
EXPERIMENT_NAME="$(basename "$SCRIPT_DIR")"
|
||||
|
||||
# shellcheck source=/dev/null
|
||||
source "${SCRIPT_DIR}/../../scripts/common/lib.sh"
|
||||
# shellcheck source=/dev/null
|
||||
source "${SCRIPT_DIR}/../../scripts/common/platform.sh"
|
||||
# shellcheck source=/dev/null
|
||||
source "${SCRIPT_DIR}/config.env"
|
||||
|
||||
RUN_ID="${RUN_ID:-$(date '+%Y%m%d-%H%M%S')}"
|
||||
RUN_DATE="$(date '+%m%d')"
|
||||
RESULT_BASE="${SCRIPT_DIR}/results"
|
||||
RESULT_ROOT="${RESULT_BASE}/${RUN_ID}"
|
||||
LOG_DIR="${RESULT_ROOT}/logs"
|
||||
RAW_DIR="${RESULT_ROOT}/raw_outputs"
|
||||
CONTAINER_PYTHON="${CONTAINER_PYTHON:-/root/miniconda/envs/python310_torch25_cuda/bin/python}"
|
||||
|
||||
ensure_result_root "$RESULT_ROOT"
|
||||
log_init "${LOG_DIR}/orchestrator.log"
|
||||
|
||||
log "experiment=${EXPERIMENT_NAME}"
|
||||
log "run_id=${RUN_ID}"
|
||||
log "platform=${PLATFORM}"
|
||||
log "hardware=${HARDWARE}"
|
||||
log "model=${MODEL_PATH}"
|
||||
log "groups=${#CONTEXT_GROUPS[@]}"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
is_server_healthy() {
|
||||
curl --fail --silent --show-error --max-time 5 "http://127.0.0.1:${PORT}/health" >/dev/null 2>&1
|
||||
}
|
||||
|
||||
stop_server() {
|
||||
log "stopping container ${CONTAINER_NAME}"
|
||||
docker rm -f "$CONTAINER_NAME" 2>/dev/null || true
|
||||
pkill -9 -f 'sglang.launch_server' 2>/dev/null || true
|
||||
sleep 2
|
||||
}
|
||||
|
||||
start_server() {
|
||||
local max_context_len="$1"
|
||||
local max_running="$2"
|
||||
log "starting server (context-length=${max_context_len}, max-running=${max_running})"
|
||||
bash "${SCRIPT_DIR}/start_sglang.sh" "$max_context_len" "$max_running" >> "${LOG_DIR}/start_server_${max_context_len}.log" 2>&1
|
||||
}
|
||||
|
||||
run_warmup() {
|
||||
local input_len="$1"
|
||||
local output_len="${2:-256}"
|
||||
local container_output="/tmp/bench_outputs/sglang_warmup_${input_len}_${output_len}.jsonl"
|
||||
local detail_log="${LOG_DIR}/warmup_${input_len}_${output_len}.log"
|
||||
|
||||
log "warming up (input=${input_len}, output=${output_len}, num=1)"
|
||||
docker exec "$CONTAINER_NAME" mkdir -p "$(dirname "$container_output")"
|
||||
|
||||
docker exec "$CONTAINER_NAME" \
|
||||
env HF_HUB_OFFLINE=1 TRANSFORMERS_OFFLINE=1 HF_DATASETS_OFFLINE=1 \
|
||||
"${CONTAINER_PYTHON}" -m sglang.bench_serving \
|
||||
--backend sglang \
|
||||
--host 127.0.0.1 \
|
||||
--port "$PORT" \
|
||||
--model "$MODEL_PATH" \
|
||||
--dataset-name random \
|
||||
--dataset-path /workspace/dummy_sharegpt.json \
|
||||
--random-input-len "$input_len" \
|
||||
--random-output-len "$output_len" \
|
||||
--random-range-ratio 1.0 \
|
||||
--num-prompts 1 \
|
||||
--max-concurrency 1 \
|
||||
--request-rate 10000 \
|
||||
--output-file "$container_output" \
|
||||
--output-details \
|
||||
> "$detail_log" 2>&1 || {
|
||||
log "WARNING: warmup failed; see ${detail_log}"
|
||||
return 1
|
||||
}
|
||||
log "warmup completed"
|
||||
}
|
||||
|
||||
scenario_already_completed() {
|
||||
local output_file="$1"
|
||||
local expected="$2"
|
||||
[[ -s "$output_file" ]] || return 1
|
||||
local completed
|
||||
completed="$(python3 -c "
|
||||
import json, sys
|
||||
path = sys.argv[1]
|
||||
try:
|
||||
with open(path, 'r', encoding='utf-8') as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if line:
|
||||
data = json.loads(line)
|
||||
print(data.get('completed', 0))
|
||||
break
|
||||
except Exception:
|
||||
print(0)
|
||||
" "$output_file")"
|
||||
[[ "${completed:-0}" -ge "$expected" ]]
|
||||
}
|
||||
|
||||
run_group() {
|
||||
local group_label="$1"
|
||||
local max_context_len="$2"
|
||||
local max_running="$3"
|
||||
local -n scenarios_ref="$4"
|
||||
|
||||
mkdir -p "$RAW_DIR" "$LOG_DIR"
|
||||
|
||||
log "===== ${group_label} START (context-length=${max_context_len}, max-running=${max_running}) ====="
|
||||
|
||||
stop_server
|
||||
if ! start_server "$max_context_len" "$max_running"; then
|
||||
log "ERROR: failed to start server for ${group_label}; skipping group"
|
||||
stop_server
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Warmup with the shortest output for this input length to save time.
|
||||
local warmup_input="${scenarios_ref[0]}"
|
||||
local warmup_input_len
|
||||
warmup_input_len="$(echo "$warmup_input" | awk '{print $2}')"
|
||||
run_warmup "$warmup_input_len" 256
|
||||
|
||||
for scenario in "${scenarios_ref[@]}"; do
|
||||
read -r concurrency input_len output_len num_prompts <<< "$scenario"
|
||||
|
||||
local host_output_file="${RAW_DIR}/sglang_${group_label}_${RUN_DATE}_${concurrency}_${input_len}_${output_len}.jsonl"
|
||||
local container_output="/tmp/bench_outputs/sglang_${group_label}_${RUN_DATE}_${concurrency}_${input_len}_${output_len}.jsonl"
|
||||
local detail_log="${LOG_DIR}/sglang_${group_label}_c${concurrency}_i${input_len}_o${output_len}.log"
|
||||
|
||||
if scenario_already_completed "$host_output_file" "$num_prompts"; then
|
||||
log "skipping already-completed scenario: group=${group_label} c=${concurrency} i=${input_len} o=${output_len}"
|
||||
continue
|
||||
fi
|
||||
|
||||
log "running scenario: group=${group_label} c=${concurrency} i=${input_len} o=${output_len} n=${num_prompts}"
|
||||
|
||||
docker exec "$CONTAINER_NAME" mkdir -p "$(dirname "$container_output")"
|
||||
|
||||
local bench_rc=0
|
||||
docker exec "$CONTAINER_NAME" \
|
||||
env HF_HUB_OFFLINE=1 TRANSFORMERS_OFFLINE=1 HF_DATASETS_OFFLINE=1 \
|
||||
"${CONTAINER_PYTHON}" -m sglang.bench_serving \
|
||||
--backend sglang \
|
||||
--host 127.0.0.1 \
|
||||
--port "$PORT" \
|
||||
--model "$MODEL_PATH" \
|
||||
--dataset-name random \
|
||||
--dataset-path /workspace/dummy_sharegpt.json \
|
||||
--random-input-len "$input_len" \
|
||||
--random-output-len "$output_len" \
|
||||
--random-range-ratio 1.0 \
|
||||
--num-prompts "$num_prompts" \
|
||||
--max-concurrency "$concurrency" \
|
||||
--request-rate 10000 \
|
||||
--output-file "$container_output" \
|
||||
--output-details \
|
||||
> "$detail_log" 2>&1 || bench_rc=$?
|
||||
|
||||
if [[ "$bench_rc" -ne 0 ]]; then
|
||||
log "ERROR: scenario group=${group_label} c=${concurrency} i=${input_len} o=${output_len} failed; see ${detail_log}"
|
||||
continue
|
||||
fi
|
||||
|
||||
docker cp "${CONTAINER_NAME}:${container_output}" "$host_output_file" || {
|
||||
log "ERROR: failed to copy output from container for scenario group=${group_label} c=${concurrency} i=${input_len} o=${output_len}"
|
||||
continue
|
||||
}
|
||||
|
||||
# Verify the request actually completed.
|
||||
local completed
|
||||
completed="$(python3 -c "
|
||||
import json
|
||||
with open('${host_output_file}') as f:
|
||||
for line in f:
|
||||
data = json.loads(line)
|
||||
print(data.get('completed', 0))
|
||||
break
|
||||
")"
|
||||
if [[ "${completed:-0}" -lt "$num_prompts" ]]; then
|
||||
log "ERROR: scenario group=${group_label} c=${concurrency} i=${input_len} o=${output_len} only completed ${completed}/${num_prompts}"
|
||||
continue
|
||||
fi
|
||||
|
||||
log "finished scenario: output=${host_output_file}"
|
||||
done
|
||||
|
||||
stop_server
|
||||
log "===== ${group_label} DONE ====="
|
||||
}
|
||||
|
||||
parse_backend() {
|
||||
log "parsing results in ${RESULT_ROOT}"
|
||||
python3 "${SCRIPT_DIR}/../../scripts/common/parse_backend.py" "$RESULT_ROOT" --backend sglang \
|
||||
>> "${LOG_DIR}/parse.log" 2>&1 || {
|
||||
log "WARNING: parse_backend.py failed; see ${LOG_DIR}/parse.log"
|
||||
}
|
||||
}
|
||||
|
||||
write_metadata() {
|
||||
ensure_result_root "$RESULT_ROOT"
|
||||
local meta_json="${RESULT_ROOT}/results.json"
|
||||
|
||||
write_metadata_json \
|
||||
"$meta_json" \
|
||||
"$EXPERIMENT_NAME" \
|
||||
"$RUN_ID" \
|
||||
"$MODEL_PATH" \
|
||||
"sglang" \
|
||||
"sglang-xpu" \
|
||||
"$HARDWARE" \
|
||||
"$ACCELERATOR" \
|
||||
"$CHIP" \
|
||||
"experiments/${EXPERIMENT_NAME}/run_bench.sh" \
|
||||
"" \
|
||||
"P800 long-context matrix for DeepSeek-V4-Flash-INT8"
|
||||
|
||||
# Embed config.
|
||||
python3 - "$meta_json" <<'PY'
|
||||
import json
|
||||
import sys
|
||||
|
||||
path = sys.argv[1]
|
||||
with open(path, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
|
||||
data["config"] = {
|
||||
"tp": 8,
|
||||
"ep": 8,
|
||||
"xpu_visible_devices": "0,1,2,3,4,5,6,7",
|
||||
"platform": "kunlun_p800",
|
||||
"container_name": "sglang-dsv4-flash",
|
||||
"context_pad": 1024,
|
||||
"groups": [
|
||||
{"label": "64k", "max_context_len": 80000, "max_running": 10},
|
||||
{"label": "128k", "max_context_len": 140000, "max_running": 5}
|
||||
]
|
||||
}
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
json.dump(data, f, indent=2, ensure_ascii=False)
|
||||
PY
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Main
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
stop_server
|
||||
write_metadata
|
||||
|
||||
for group in "${CONTEXT_GROUPS[@]}"; do
|
||||
read -r group_label max_context_len max_running <<< "$group"
|
||||
|
||||
# Resolve scenario array for this group.
|
||||
scenario_array_name="SCENARIOS_${group_label^^}"
|
||||
if [[ -z "${!scenario_array_name:-}" ]]; then
|
||||
log "WARNING: no scenarios defined for group ${group_label}"
|
||||
continue
|
||||
fi
|
||||
|
||||
run_group "$group_label" "$max_context_len" "$max_running" "$scenario_array_name"
|
||||
done
|
||||
|
||||
parse_backend
|
||||
|
||||
log "all results saved to ${RESULT_ROOT}"
|
||||
133
experiments/dsv4_p800_long_context_matrix/start_sglang.sh
Executable file
133
experiments/dsv4_p800_long_context_matrix/start_sglang.sh
Executable file
@ -0,0 +1,133 @@
|
||||
#!/usr/bin/env bash
|
||||
# Start P800 SGLang INT8 server for the long-context matrix.
|
||||
# Usage: start_sglang.sh <max_context_len> <max_running>
|
||||
set -Eeuo pipefail
|
||||
|
||||
MAX_CONTEXT_LEN="${1:-}"
|
||||
MAX_RUNNING="${2:-}"
|
||||
if [[ -z "$MAX_CONTEXT_LEN" || -z "$MAX_RUNNING" ]]; then
|
||||
echo "Usage: $0 <max_context_len> <max_running>"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ "$MAX_RUNNING" -lt 2 ]]; then
|
||||
echo "ERROR: max_running must be >= 2 on P800 INT8 (1 causes req_to_token_pool allocation failures)"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
# shellcheck source=/dev/null
|
||||
source "${SCRIPT_DIR}/../../scripts/common/lib.sh"
|
||||
# shellcheck source=/dev/null
|
||||
source "${SCRIPT_DIR}/../../scripts/common/platform.sh"
|
||||
# shellcheck source=/dev/null
|
||||
source "${SCRIPT_DIR}/config.env"
|
||||
|
||||
RESULT_ROOT="${RESULT_ROOT:-/tmp/${EXPERIMENT}}"
|
||||
SERVER_LOG="${RESULT_ROOT}/logs/server_${MAX_CONTEXT_LEN}.outer.log"
|
||||
mkdir -p "$(dirname "$SERVER_LOG")"
|
||||
|
||||
log "starting P800 SGLang INT8 server (context-length=${MAX_CONTEXT_LEN}, max-running=${MAX_RUNNING})"
|
||||
log "model: ${MODEL_PATH}"
|
||||
log "port: ${PORT}"
|
||||
log "server log: ${SERVER_LOG}"
|
||||
|
||||
# Stop any existing container with the same name.
|
||||
docker rm -f "$CONTAINER_NAME" 2>/dev/null || true
|
||||
|
||||
# Build device args.
|
||||
device_args=""
|
||||
for i in 0 1 2 3 4 5 6 7; do
|
||||
device_args="${device_args} --device /dev/xpu${i}:/dev/xpu${i}"
|
||||
done
|
||||
device_args="${device_args} --device /dev/xpuctrl:/dev/xpuctrl"
|
||||
|
||||
# Environment variables required by the P800 SGLang INT8 image.
|
||||
env_args=(
|
||||
-e XPU_VISIBLE_DEVICES=0,1,2,3,4,5,6,7
|
||||
-e CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7
|
||||
-e CUDA_DEVICE_ORDER=OAM_ID
|
||||
-e SGLANG_USE_TRANSFORMERS_V5_TOKENIZER=1
|
||||
-e XMLIR_FORCE_USE_XPU_GRAPH=1
|
||||
-e SGLANG_DSV4_MODE=2604
|
||||
-e PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
|
||||
-e SGLANG_NSA_DUAL_STREAM=true
|
||||
-e SGLANG_NSA_QUANT_WQ_B_WK=false
|
||||
-e SGLANG_FP8_PAGED_MQA_LOGITS_TORCH=1
|
||||
-e SGLANG_OPT_DEEPGEMM_HC_PRENORM=false
|
||||
-e SGLANG_OPT_USE_TILELANG_MHC_PRE=1
|
||||
-e SGLANG_OPT_USE_TILELANG_MHC_POST=1
|
||||
-e SGLANG_CLEAN_REQUEST_WHEN_RETRACT=1
|
||||
-e SGLANG_SET_CPU_AFFINITY=1
|
||||
-e SGLANG_OPT_USE_KLX_TOPK_KERNEL=1
|
||||
-e XSGL_INTERTYPE_BFP16=1
|
||||
-e ENABLE_FAST_BFP16_ATTN=1
|
||||
-e XSGL_USE_DEEP_GEMM_BMM=1
|
||||
-e XSGL_XDNN_QUANT=1
|
||||
-e XSGL_FUSE_RMS_NORM_QUANT=1
|
||||
-e XSGL_TRANSPOSE_MATMUL_WEIGHT=1
|
||||
-e XINFER_QUANT_SDNN=1
|
||||
-e XSGL_USE_MOE_SIGMOID_GROUP_TOPK_NORM=1
|
||||
-e XSGL_EARLY_FIRST_TOKEN=1
|
||||
-e XSGL_ENABLE_TGEMM_FP16=1
|
||||
-e SGLANG_ENABLE_SPEC_V2=True
|
||||
-e SGLANG_ENABLE_OVERLAP_PLAN_STREAM=1
|
||||
-e PYTHONDONTWRITEBYTECODE=1
|
||||
-e XTORCH_OPS_LIB_DIR=/root/miniconda/envs/python310_torch25_cuda/lib/python3.10/site-packages/xtorch_ops
|
||||
-e XPU_RUNTIME_LIB_DIR=/root/miniconda/envs/python310_torch25_cuda/xcudart/lib
|
||||
-e BKCL_TREE_THRESHOLD=1048576
|
||||
-e CUDA_ENABLE_P2P_NO_UVA=1
|
||||
-e NCCL_IB_GID_INDEX=3
|
||||
-e IS_DSV4=1
|
||||
-e MC_CUSTOM_TOPO_JSON=/workspace/nic_priority_matrix_test.json
|
||||
# INT8 specific
|
||||
-e SGLANG_DSV4_FP4_EXPERTS=false
|
||||
-e SGLANG_APPLY_CONFIG_BACKUP=auto
|
||||
-e BKCL_ENABLE_XDR=1
|
||||
-e BKCL_RDMA_NICS=eth1,eth1,eth3,eth3,eth5,eth5,eth7,eth7
|
||||
-e BKCL_RDMA_VERBS=1
|
||||
-e XSGL_INT8_LM_HEAD=1
|
||||
-e SGLANG_P800_ALL_GATHER_FALLBACK=0
|
||||
)
|
||||
|
||||
# Launch args. Keep the proven P800 INT8 command from dsv4_p800_max_context_length,
|
||||
# but make --context-length and --max-running-requests dynamic per group.
|
||||
launch_args="--host 0.0.0.0 --port ${PORT} --model-path /models --attention-backend nsa --nsa-prefill klxdsa --nsa-decode klxdsa --trust-remote-code --disable-custom-all-reduce --chunked-prefill-size 8192 --page-size 64 --mem-fraction-static 0.8 --max-prefill-tokens ${MAX_CONTEXT_LEN} --max-running-requests ${MAX_RUNNING} --tensor-parallel-size 8 --ep-size 8 --disable-shared-experts-fusion --quantization w8a8_int8 --kv-cache-dtype float16 --disable-piecewise-cuda-graph --cuda-graph-max-bs 32 --watchdog-timeout 3000000 --tool-call-parser deepseekv4 --reasoning-parser deepseek-v4 --speculative-algorithm EAGLE --speculative-num-steps 3 --speculative-eagle-topk 1 --speculative-num-draft-tokens 4 --constrained-json-disable-any-whitespace --enable-metrics --enable-request-time-stats-logging --context-length ${MAX_CONTEXT_LEN}"
|
||||
|
||||
# Base64-encode the bootstrap command to avoid host-shell quoting issues.
|
||||
server_cmd=$(cat <<EOF
|
||||
cd /workspace
|
||||
find /root/miniconda/envs/python310_torch25_cuda/lib/python3.10/site-packages/sglang -type d -name __pycache__ -exec rm -rf {} + 2>/dev/null || true
|
||||
/root/miniconda/envs/python310_torch25_cuda/bin/pip install --upgrade safetensors -q
|
||||
/root/miniconda/envs/python310_torch25_cuda/bin/pip install https://files.pythonhosted.org/packages/14/8b/2a1333a6455c6fad401c2285dee6f58016c55b1cb44cae3a31f8a9cc7d83/apache_tvm_ffi-0.1.0b2-cp310-cp310-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl -q
|
||||
/root/miniconda/envs/python310_torch25_cuda/bin/python -c "import torch; torch.float8_e8m0fnu = torch.uint8; import runpy, sys; sys.argv[0] = 'sglang.launch_server'; runpy.run_module('sglang.launch_server', run_name='__main__')" ${launch_args}
|
||||
EOF
|
||||
)
|
||||
server_cmd_b64=$(printf '%s' "$server_cmd" | base64 -w0)
|
||||
|
||||
patch_mounts=(
|
||||
-v "${PATCH_ROOT}/nic_priority_matrix_test.json:/workspace/nic_priority_matrix_test.json:ro"
|
||||
-v "${PATCH_ROOT}/dummy_sharegpt.json:/workspace/dummy_sharegpt.json:ro"
|
||||
)
|
||||
|
||||
docker run -d \
|
||||
--name "${CONTAINER_NAME}" \
|
||||
--privileged \
|
||||
--network host \
|
||||
--ipc host \
|
||||
${device_args} \
|
||||
-v "${MODEL_PATH}:/models:ro" \
|
||||
-v "${MODEL_PATH}:${MODEL_PATH}:ro" \
|
||||
"${patch_mounts[@]}" \
|
||||
"${env_args[@]}" \
|
||||
"${DOCKER_IMAGE}" \
|
||||
bash -c "echo '${server_cmd_b64}' | base64 -d | bash" \
|
||||
>> "${SERVER_LOG}" 2>&1
|
||||
|
||||
log "container ${CONTAINER_NAME} started, waiting for health"
|
||||
if health_check 127.0.0.1 "$PORT" 600; then
|
||||
log "container ${CONTAINER_NAME} is healthy"
|
||||
else
|
||||
log "ERROR: container ${CONTAINER_NAME} failed health check"
|
||||
exit 1
|
||||
fi
|
||||
Loading…
x
Reference in New Issue
Block a user