diff --git a/experiments/dsv4_p800_long_context_matrix/README.md b/experiments/dsv4_p800_long_context_matrix/README.md index b8a77ce..c59e2e6 100644 --- a/experiments/dsv4_p800_long_context_matrix/README.md +++ b/experiments/dsv4_p800_long_context_matrix/README.md @@ -9,7 +9,7 @@ Long-context throughput/latency matrix for SGLang-XPU on Kunlun P800 (8x XPU). - Hardware: 8x Kunlun P800 XPU - Input lengths: 64k, 128k - Output lengths: 256, 1024, 4096 -- Concurrency: sweeps per context group (see `config.env`) +- Concurrency: **1** — P800 INT8 can only sustain one concurrent long-context request without OOM; `max_running=2` is kept because `max_running=1` triggers `req_to_token_pool` allocation failures in this image. ## Usage diff --git a/experiments/dsv4_p800_long_context_matrix/config.env b/experiments/dsv4_p800_long_context_matrix/config.env index 9d88604..0bde182 100644 --- a/experiments/dsv4_p800_long_context_matrix/config.env +++ b/experiments/dsv4_p800_long_context_matrix/config.env @@ -1,7 +1,9 @@ # Long-context matrix for SGLang-XPU on DeepSeek-V4-Flash-INT8 (P800, TP=8). # Input lengths: 64k, 128k (P800 INT8 max input length is ~131072 tokens). # Output lengths: 256, 1k, 4k -# Concurrency: conservative to fit XPU memory. +# Concurrency: P800 INT8 can only sustain 1 concurrent long-context request +# without OOM, so all scenarios use concurrency=1. max_running is kept at 2 +# because max_running=1 triggers req_to_token_pool allocation failures. EXPERIMENT="dsv4_p800_long_context_matrix" MODEL_NAME="DeepSeek-V4-Flash-INT8" @@ -16,33 +18,21 @@ CONTEXT_PAD="${CONTEXT_PAD:-1024}" # Per-context server settings. Each group is run with its own server start. # Format: "input_label max_context_len max_running" declare -a CONTEXT_GROUPS=( - "64k 80000 10" - "128k 140000 5" + "64k 72000 2" + "128k 140000 2" ) # Scenarios per group: "concurrency input_len output_len num_prompts" declare -a SCENARIOS_64K=( - "10 65536 256 50" - "5 65536 256 25" - "2 65536 256 10" - "10 65536 1024 50" - "5 65536 1024 25" - "2 65536 1024 10" - "10 65536 4096 50" - "5 65536 4096 25" - "2 65536 4096 10" + "1 65536 256 10" + "1 65536 1024 10" + "1 65536 4096 10" ) declare -a SCENARIOS_128K=( - "5 131072 256 20" - "2 131072 256 10" - "1 131072 256 5" - "5 131072 1024 20" - "2 131072 1024 10" - "1 131072 1024 5" - "5 131072 4096 20" - "2 131072 4096 10" - "1 131072 4096 5" + "1 131072 256 5" + "1 131072 1024 5" + "1 131072 4096 5" ) CONTAINER_NAME="${CONTAINER_NAME:-sglang-dsv4-flash}" diff --git a/experiments/dsv4_p800_long_context_matrix/run_bench.sh b/experiments/dsv4_p800_long_context_matrix/run_bench.sh index 61ddb86..4b4c7d2 100755 --- a/experiments/dsv4_p800_long_context_matrix/run_bench.sh +++ b/experiments/dsv4_p800_long_context_matrix/run_bench.sh @@ -38,6 +38,10 @@ is_server_healthy() { curl --fail --silent --show-error --max-time 5 "http://127.0.0.1:${PORT}/health" >/dev/null 2>&1 } +container_running() { + docker inspect "$CONTAINER_NAME" >/dev/null 2>&1 +} + stop_server() { log "stopping container ${CONTAINER_NAME}" docker rm -f "$CONTAINER_NAME" 2>/dev/null || true @@ -59,6 +63,10 @@ run_warmup() { local detail_log="${LOG_DIR}/warmup_${input_len}_${output_len}.log" log "warming up (input=${input_len}, output=${output_len}, num=1)" + if ! container_running; then + log "WARNING: container not running, skipping warmup" + return 1 + fi docker exec "$CONTAINER_NAME" mkdir -p "$(dirname "$container_output")" docker exec "$CONTAINER_NAME" \ @@ -142,6 +150,11 @@ run_group() { continue fi + if ! container_running; then + log "WARNING: container died, skipping remaining scenarios in group=${group_label}" + break + fi + log "running scenario: group=${group_label} c=${concurrency} i=${input_len} o=${output_len} n=${num_prompts}" docker exec "$CONTAINER_NAME" mkdir -p "$(dirname "$container_output")" @@ -168,6 +181,10 @@ run_group() { if [[ "$bench_rc" -ne 0 ]]; then log "ERROR: scenario group=${group_label} c=${concurrency} i=${input_len} o=${output_len} failed; see ${detail_log}" + if ! container_running; then + log "WARNING: container died, skipping remaining scenarios in group=${group_label}" + break + fi continue fi @@ -241,8 +258,8 @@ data["config"] = { "container_name": "sglang-dsv4-flash", "context_pad": 1024, "groups": [ - {"label": "64k", "max_context_len": 80000, "max_running": 10}, - {"label": "128k", "max_context_len": 140000, "max_running": 5} + {"label": "64k", "max_context_len": 72000, "max_running": 2}, + {"label": "128k", "max_context_len": 140000, "max_running": 2} ] } with open(path, "w", encoding="utf-8") as f: