[Feat] add Phase 3 SGLang timeline profiling

This commit is contained in:
Zhiyi Hong 2026-07-31 18:37:12 +08:00
parent 1dc5612e3a
commit 3b7291e8a0
4 changed files with 725 additions and 0 deletions

View File

@ -1,5 +1,9 @@
# sskj — 多平台大模型推理性能基准测试项目
> **更新2026-07-31 18:38:00 CST**
>
> 新增 DeepSeek-V4-Pro 双机 Pro6000D SGLang Phase 3 时间线分析唯一入口。代码提供双节点环境审计、PyTorch Profiler smoke、Nsight Systems smoke 与三段正式 capture range覆盖 Decode 对照、Decode 背景叠加 128K Prefill、独立 128K Prefill沿用 Phase 1 已验证的 TP16/EP2、CUDA Graph 与双 Rail NET/IB 配置,不重复 Phase 2 的硬件采样和通信微基准。阶段尚未产出正式结果,因此按档案门禁暂不创建 `phase3_exp.html``phase3_code.html`
>
> **更新2026-07-31 17:22:20 CST**
>
> 完成 DeepSeek-V4-Pro 双机 Pro6000D SGLang Phase 2 正式收口。最终 Run `dsv4pro-phase2-20260731-163620` 在 28 分 44 秒内完成 8/8 benchmark、8/8 精确测量窗口和 18/18 采集器启停;混合 Prefill 令 Decode Output TPS 下降 23.96%、TPOT P95 增加 66.75%。GPU/CPU/NUMA、双 Rail RDMA、PCIe P2P、8/16-GPU AllReduce 与 `NCCL_CROSS_NIC=0/1/2` 数据均已归档;证据排除原始 Rail 带宽饱和、整机 CPU 饱和、GPU 降频和 CROSS_NIC 选择作为首要原因。实验结束后两节点容器、端口和 16 张 GPU 均已清理。

View File

@ -0,0 +1,67 @@
# DSV4-Pro / Pro6000D / 2-Node / SGLang Timeline Profiling
Phase 3 captures short distributed timelines for the completed TP16 baseline.
It does not rerun Phase 2 hardware sampling or communication microbenchmarks.
## One entry point
Run only on Head `174.1.51.5`:
```bash
cd /data/hzy/sskj/experiments/pro6000/dsv4pro_pro6000d_2node_sglang_timeline_profiling
bash run_timeline_profiling.sh audit
DRY_RUN=1 RUN_ID=dsv4pro-phase3-dryrun-$(date +%Y%m%d-%H%M%S) \
bash run_timeline_profiling.sh all
RUN_ID=dsv4pro-phase3-torch-smoke-$(date +%Y%m%d-%H%M%S) \
bash run_timeline_profiling.sh torch-smoke
RUN_ID=dsv4pro-phase3-nsys-smoke-$(date +%Y%m%d-%H%M%S) \
bash run_timeline_profiling.sh nsys-smoke
```
After both smoke tests pass, run the three formal capture ranges in one model
service lifetime:
```bash
RUN_ID=dsv4pro-phase3-$(date +%Y%m%d-%H%M%S)
tmux new-session -d -s dsv4pro-phase3 \
"RUN_ID=${RUN_ID} bash run_timeline_profiling.sh all \
2>&1 | tee /data/hzy/${RUN_ID}.log"
tmux attach -t dsv4pro-phase3
```
The script starts and stops both nodes, triggers `/start_profile`, retrieves
Worker artifacts, exports basic `nsys stats`, and cleans containers and Worker
staging paths. No `source`, Conda activation, or command on Worker is required.
## Formal ranges
1. `decode_control`: `1K -> 256`, `C=32`.
2. `mixed_decode_with_128k_prefill`: active `1K -> 512`, `C=32` decode plus one `128K -> 1` injection.
3. `long_prefill`: isolated `128K -> 1`, `C=1`.
The primary run keeps CUDA Graph enabled. A later layer-wise NVTX run with
disabled graphs is allowed only if these production-faithful traces cannot map
the dominant kernels to model components.
## Outputs
```text
results/<RUN_ID>/
manifest.json
service/
cases/
profiles/head/
nsys/
torch/
profiles/worker/
nsys/
torch/
```
Profiler runs are diagnostic and must not replace the Phase 2 no-profiler TPS
baseline.

View File

@ -0,0 +1,50 @@
# Phase 3: short distributed timelines for the verified TP16 baseline.
EXPERIMENT="dsv4pro_pro6000d_2node_sglang_timeline_profiling"
# Reuse model, topology, Docker image, serving knobs and benchmark paths from
# the completed Phase 1 baseline. This file only defines profiling policy.
PHASE1_DIR="${PHASE1_DIR:-$(cd "$(dirname "${BASH_SOURCE[0]}")/../dsv4pro_pro6000d_2node_sglang_tp16_quick_map" && pwd)}"
PROFILE_PORT="${PROFILE_PORT:-30002}"
PROFILE_DIST_INIT_PORT="${PROFILE_DIST_INIT_PORT:-20002}"
PROFILE_ROOT_IN_CONTAINER="${PROFILE_ROOT_IN_CONTAINER:-/profiles}"
WORKER_STAGE_ROOT="${WORKER_STAGE_ROOT:-/data/hzy/dsv4pro_phase3_worker}"
# The primary run keeps CUDA Graph enabled. Layer-wise NVTX with disabled
# graphs is deliberately deferred until the production-faithful trace proves
# that kernel names are insufficient.
NSYS_TRACE="${NSYS_TRACE:-cuda,nvtx,nccl,osrt}"
NSYS_CUDA_GRAPH_TRACE="${NSYS_CUDA_GRAPH_TRACE:-node}"
NSYS_EXPORT_TIMEOUT_S="${NSYS_EXPORT_TIMEOUT_S:-300}"
NSYS_CONTROL_STEPS="${NSYS_CONTROL_STEPS:-16}"
NSYS_MIXED_STEPS="${NSYS_MIXED_STEPS:-32}"
NSYS_PREFILL_STEPS="${NSYS_PREFILL_STEPS:-16}"
NSYS_SMOKE_STEPS="${NSYS_SMOKE_STEPS:-5}"
TORCH_SMOKE_STEPS="${TORCH_SMOKE_STEPS:-3}"
# Short profiling workloads. They reproduce the Phase 2 scheduler shapes but
# avoid its full result matrix and long hardware sampling.
CONTROL_ISL="${CONTROL_ISL:-1024}"
CONTROL_OSL="${CONTROL_OSL:-256}"
CONTROL_CONCURRENCY="${CONTROL_CONCURRENCY:-32}"
CONTROL_PROMPTS="${CONTROL_PROMPTS:-32}"
MIXED_ISL="${MIXED_ISL:-1024}"
MIXED_OSL="${MIXED_OSL:-512}"
MIXED_CONCURRENCY="${MIXED_CONCURRENCY:-32}"
MIXED_PROMPTS="${MIXED_PROMPTS:-64}"
MIXED_INJECTION_ISL="${MIXED_INJECTION_ISL:-131072}"
MIXED_INJECTION_OSL="${MIXED_INJECTION_OSL:-1}"
PREFILL_ISL="${PREFILL_ISL:-131072}"
PREFILL_OSL="${PREFILL_OSL:-1}"
PROFILE_TIMEOUT_S="${PROFILE_TIMEOUT_S:-1800}"
PROFILE_STOP_WAIT_S="${PROFILE_STOP_WAIT_S:-180}"
HEALTH_CHECK_RETRIES="${HEALTH_CHECK_RETRIES:-600}"
HEALTH_CHECK_INTERVAL_S="${HEALTH_CHECK_INTERVAL_S:-5}"
RESULT_BASE="${RESULT_BASE:-$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/results}"
RUNTIME_BASE="${RUNTIME_BASE:-$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/runtime}"
DRY_RUN="${DRY_RUN:-0}"
ALLOW_BUSY_GPU="${ALLOW_BUSY_GPU:-0}"

View File

@ -0,0 +1,604 @@
#!/usr/bin/env bash
# Capture short PyTorch and Nsight timelines for the two-node DSV4-Pro TP16 service.
set -Eeuo pipefail
SELF_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# shellcheck source=/dev/null
source "${SELF_DIR}/config.env"
# shellcheck source=/dev/null
source "${PHASE1_DIR}/config.env"
# Phase 1 config defines SCRIPT_DIR/RESULT_BASE; restore Phase 3 ownership.
SCRIPT_DIR="${SELF_DIR}"
REPO_ROOT="$(cd "${SCRIPT_DIR}/../../.." && pwd)"
# shellcheck source=/dev/null
source "${SELF_DIR}/config.env"
ACTION="${1:-all}"
RUN_ID="${RUN_ID:-dsv4pro-phase3-$(date +%Y%m%d-%H%M%S)}"
RUN_STARTED_AT="$(date '+%Y-%m-%dT%H:%M:%S%z')"
RESULT_DIR="${RESULT_BASE}/${RUN_ID}"
WORKER_RUN_DIR="${WORKER_STAGE_ROOT}/${RUN_ID}"
HEAD_PROFILE_HOST="${RESULT_DIR}/profiles/head"
WORKER_PROFILE_HOST="${WORKER_RUN_DIR}/profiles/worker"
SERVICE_DIR="${RESULT_DIR}/service"
CASE_DIR="${RESULT_DIR}/cases"
HEAD_CONTAINER="${EXPERIMENT}_head"
WORKER_CONTAINER="${EXPERIMENT}_worker"
SGLANG_PORT="${PROFILE_PORT}"
DIST_INIT_PORT="${PROFILE_DIST_INIT_PORT}"
PROFILE_MODE=""
CAPTURE_RANGES=0
CLEANUP_ON_EXIT=0
declare -a DOCKER_CMD=()
declare -a BENCH_CMD=()
log() {
printf '[%s] %s\n' "$(date '+%Y-%m-%d %H:%M:%S')" "$*"
}
iso_now() {
date '+%Y-%m-%dT%H:%M:%S%z'
}
print_command() {
printf '%q ' "$@"
printf '\n'
}
node_is_local() {
local node="$1"
local local_ips
[[ "${node}" == "localhost" || "${node}" == "127.0.0.1" ]] && return 0
local_ips="$(hostname -I 2>/dev/null || true)"
[[ " ${local_ips} " == *" ${node} "* ]]
}
run_on_node() {
local node="$1"
local command="$2"
if node_is_local "${node}"; then
bash -lc "${command}"
else
ssh -o BatchMode=yes -o StrictHostKeyChecking=no "${node}" "${command}"
fi
}
service_is_healthy() {
curl --fail --silent --show-error --max-time 5 \
"http://${HEAD_IP}:${SGLANG_PORT}/health" >/dev/null 2>&1
}
remote_has_gpu_processes() {
local node="$1"
local output
output="$(run_on_node "${node}" \
"nvidia-smi --query-compute-apps=pid --format=csv,noheader 2>/dev/null" || true)"
[[ -n "${output//[[:space:]]/}" ]]
}
validate_network_config() {
local interface_spec="${NCCL_SOCKET_IFNAME#=}"
local hca_spec="${NCCL_IB_HCA#=}"
[[ "${interface_spec}" == "eth0" ]] || {
log "ERROR: Phase 3 requires the verified bootstrap interface eth0"
return 1
}
[[ "${hca_spec}" == "mlx5_0:1,mlx5_3:1" ]] || {
log "ERROR: Phase 3 requires the verified dual rails mlx5_0:1,mlx5_3:1"
return 1
}
}
preflight_node() {
local node="$1"
run_on_node "${node}" \
"test -d '${MODEL_PATH}' && docker image inspect '${DOCKER_IMAGE}' >/dev/null" \
|| {
log "ERROR: ${node} is missing model, dataset, Docker image, or Docker access"
return 1
}
local device
local -a devices=()
IFS=',' read -r -a devices <<< "${RDMA_DEVICE_PATHS}"
for device in "${devices[@]}"; do
run_on_node "${node}" "test -c '${device}'" || {
log "ERROR: ${node} is missing RDMA device ${device}"
return 1
}
done
if [[ "${ALLOW_BUSY_GPU}" != "1" ]] && remote_has_gpu_processes "${node}"; then
log "ERROR: ${node} has active GPU compute processes"
return 1
fi
run_on_node "${node}" \
"docker run --rm --entrypoint nsys '${DOCKER_IMAGE}' --version" >/dev/null \
|| {
log "ERROR: ${node} image does not provide nsys"
return 1
}
}
audit() {
validate_network_config
[[ -f "${DATASET_PATH}" ]] || {
log "ERROR: Head benchmark dataset is missing: ${DATASET_PATH}"
return 1
}
preflight_node "${HEAD_NODE}"
preflight_node "${WORKER_NODE}"
log "Audit passed: two nodes idle, RDMA devices present, image nsys available"
}
prepare_profile_dirs() {
mkdir -p "${HEAD_PROFILE_HOST}" "${SERVICE_DIR}" "${CASE_DIR}"
if [[ "${DRY_RUN}" == "1" ]]; then
log "[DRY] ${WORKER_NODE}: mkdir -p ${WORKER_PROFILE_HOST}"
return 0
fi
run_on_node "${WORKER_NODE}" "mkdir -p '${WORKER_PROFILE_HOST}'"
}
build_server_command() {
local node_rank="$1"
local container_name="$2"
local profile_host_dir="$3"
DOCKER_CMD=(
docker run -d
--name "${container_name}"
--gpus all
--network host
--ipc host
--shm-size 20g
--ulimit memlock=-1
--ulimit stack=67108864
-v "${MODEL_PATH}:${MODEL_PATH}:ro"
-v "${SGLANG_CACHE_DIR}:/root/.cache"
-v "${profile_host_dir}:${PROFILE_ROOT_IN_CONTAINER}"
-e "CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE_DEVICES}"
-e PYTHONUNBUFFERED=1
-e HF_HUB_OFFLINE=1
-e TRANSFORMERS_OFFLINE=1
-e PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
-e "NCCL_SOCKET_IFNAME=${NCCL_SOCKET_IFNAME}"
-e "NCCL_IB_HCA=${NCCL_IB_HCA}"
-e "NCCL_CROSS_NIC=${NCCL_CROSS_NIC}"
-e "NCCL_DEBUG=${NCCL_DEBUG}"
-e SGLANG_SHARED_EXPERT_TP1=1
)
local device
local -a devices=()
IFS=',' read -r -a devices <<< "${RDMA_DEVICE_PATHS}"
for device in "${devices[@]}"; do
DOCKER_CMD+=(--device "${device}")
done
if [[ "${PROFILE_MODE}" == "nsys" ]]; then
DOCKER_CMD+=(
--entrypoint nsys
"${DOCKER_IMAGE}"
profile
--force-overwrite=true
--trace="${NSYS_TRACE}"
--sample=none
--cpuctxsw=none
--trace-fork-before-exec=true
--cuda-graph-trace="${NSYS_CUDA_GRAPH_TRACE}"
--capture-range=cudaProfilerApi
--capture-range-end="repeat:${CAPTURE_RANGES}:defer"
--output="${PROFILE_ROOT_IN_CONTAINER}/nsys/timeline"
python3
)
else
DOCKER_CMD+=(--entrypoint python3 "${DOCKER_IMAGE}")
fi
DOCKER_CMD+=(
-m sglang.launch_server
--model-path "${MODEL_PATH}"
--tp-size "${TP_SIZE}"
--ep-size "${EP_SIZE}"
--nnodes "${NNODES}"
--node-rank "${node_rank}"
--dist-init-addr "${HEAD_IP}:${DIST_INIT_PORT}"
--trust-remote-code
--host 0.0.0.0
--port "${SGLANG_PORT}"
--mem-fraction-static "${MEM_FRACTION_STATIC}"
--cuda-graph-max-bs-decode "${CUDA_GRAPH_MAX_BS_DECODE}"
--max-running-requests "${MAX_RUNNING_REQUESTS}"
)
if [[ -n "${SGLANG_EXTRA_ARGS}" ]]; then
local -a extra_args=()
read -r -a extra_args <<< "${SGLANG_EXTRA_ARGS}"
DOCKER_CMD+=("${extra_args[@]}")
fi
}
start_service_node() {
local node="$1"
local node_rank="$2"
local container_name="$3"
local role="$4"
local profile_host_dir="$5"
build_server_command "${node_rank}" "${container_name}" "${profile_host_dir}"
print_command "${DOCKER_CMD[@]}" > "${SERVICE_DIR}/${role}_server_cmd.txt"
if [[ "${DRY_RUN}" == "1" ]]; then
printf '[DRY] %s (%s): ' "${role}" "${node}"
print_command "${DOCKER_CMD[@]}"
return 0
fi
run_on_node "${node}" "mkdir -p '${profile_host_dir}/nsys' '${profile_host_dir}/torch'"
run_on_node "${node}" "docker rm -f '${container_name}' >/dev/null 2>&1 || true"
local command
command="$(print_command "${DOCKER_CMD[@]}")"
run_on_node "${node}" "${command}" > "${SERVICE_DIR}/${role}_container_id.txt"
}
verify_nccl_transport_node() {
local node="$1"
local container_name="$2"
local role="$3"
local log_file="${SERVICE_DIR}/${role}_nccl_transport.log"
run_on_node "${node}" "docker logs '${container_name}' 2>&1" > "${log_file}"
grep -Eq 'NET/IB.*Using|Using network IB|via NET/IB' "${log_file}" || {
log "ERROR: ${role} log does not prove NET/IB"
return 1
}
grep -Eq 'mlx5_0' "${log_file}" && grep -Eq 'mlx5_3' "${log_file}" || {
log "ERROR: ${role} log does not prove both RDMA rails"
return 1
}
}
start_service() {
prepare_profile_dirs
if [[ "${DRY_RUN}" != "1" ]]; then
audit
fi
start_service_node "${WORKER_NODE}" 1 "${WORKER_CONTAINER}" worker "${WORKER_PROFILE_HOST}"
[[ "${DRY_RUN}" == "1" ]] || sleep 5
start_service_node "${HEAD_NODE}" 0 "${HEAD_CONTAINER}" head "${HEAD_PROFILE_HOST}"
if [[ "${DRY_RUN}" == "1" ]]; then
return 0
fi
local attempt
for (( attempt=1; attempt<=HEALTH_CHECK_RETRIES; attempt++ )); do
if service_is_healthy; then
log "SGLang healthy after ${attempt} checks"
verify_nccl_transport_node "${HEAD_NODE}" "${HEAD_CONTAINER}" head
verify_nccl_transport_node "${WORKER_NODE}" "${WORKER_CONTAINER}" worker
return 0
fi
sleep "${HEALTH_CHECK_INTERVAL_S}"
done
log "ERROR: SGLang health timeout"
return 1
}
stop_service_node() {
local node="$1"
local container_name="$2"
local role="$3"
if [[ "${DRY_RUN}" == "1" ]]; then
log "[DRY] ${node}: docker stop --time ${NSYS_EXPORT_TIMEOUT_S} ${container_name}"
return 0
fi
run_on_node "${node}" \
"docker stop --time '${NSYS_EXPORT_TIMEOUT_S}' '${container_name}' >/dev/null 2>&1 || true"
run_on_node "${node}" "docker logs '${container_name}' 2>&1" \
> "${SERVICE_DIR}/${role}_server.log" 2>&1 || true
run_on_node "${node}" "docker inspect '${container_name}' 2>/dev/null" \
> "${SERVICE_DIR}/${role}_inspect.json" 2>/dev/null || true
run_on_node "${node}" "docker rm -f '${container_name}' >/dev/null 2>&1 || true"
}
fetch_worker_profiles() {
[[ "${DRY_RUN}" == "1" ]] && return 0
mkdir -p "${RESULT_DIR}/profiles/worker"
scp -q -r "${WORKER_NODE}:${WORKER_PROFILE_HOST}/." \
"${RESULT_DIR}/profiles/worker/"
run_on_node "${WORKER_NODE}" "rm -rf '${WORKER_RUN_DIR}'"
}
stop_service() {
stop_service_node "${HEAD_NODE}" "${HEAD_CONTAINER}" head
stop_service_node "${WORKER_NODE}" "${WORKER_CONTAINER}" worker
fetch_worker_profiles || true
}
flush_cache() {
if [[ "${DRY_RUN}" == "1" ]]; then
log "[DRY] POST http://${HEAD_IP}:${SGLANG_PORT}/flush_cache"
return 0
fi
curl --fail --silent --show-error -X POST \
"http://${HEAD_IP}:${SGLANG_PORT}/flush_cache" >/dev/null
}
profile_done_count() {
run_on_node "${HEAD_NODE}" "docker logs '${HEAD_CONTAINER}' 2>&1" \
| grep -c 'Profiling done' || true
}
start_profile() {
local case_id="$1"
local activities_json="$2"
local steps="$3"
local output_dir="${PROFILE_ROOT_IN_CONTAINER}/torch/${case_id}"
local response_file="${CASE_DIR}/${case_id}/start_profile_response.json"
local payload
payload="{\"output_dir\":\"${output_dir}\",\"num_steps\":${steps},\"activities\":${activities_json},\"profile_id\":\"${RUN_ID}-${case_id}\",\"profile_prefix\":\"${case_id}\",\"merge_profiles\":false}"
mkdir -p "${CASE_DIR}/${case_id}"
printf '%s\n' "${payload}" > "${CASE_DIR}/${case_id}/start_profile_request.json"
if [[ "${DRY_RUN}" == "1" ]]; then
log "[DRY] POST /start_profile case=${case_id} steps=${steps} activities=${activities_json}"
return 0
fi
curl --fail --silent --show-error -X POST \
"http://${HEAD_IP}:${SGLANG_PORT}/start_profile" \
-H 'Content-Type: application/json' \
-d "${payload}" | tee "${response_file}"
}
wait_for_profile_stop() {
local before="$1"
local waited=0
while (( waited < PROFILE_STOP_WAIT_S )); do
if (( $(profile_done_count) > before )); then
return 0
fi
sleep 2
((waited+=2))
done
log "ERROR: profiler did not stop within ${PROFILE_STOP_WAIT_S}s"
return 1
}
prepare_bench_command() {
local output_file="$1"
local isl="$2"
local osl="$3"
local concurrency="$4"
local prompts="$5"
local seed="$6"
BENCH_CMD=(
docker run --rm
--network host
-v "${MODEL_PATH}:${MODEL_PATH}:ro"
-v "${DATASET_PATH}:${DATASET_PATH}:ro"
-v "$(dirname "${output_file}"):$(dirname "${output_file}")"
-e PYTHONUNBUFFERED=1
-e HF_HUB_OFFLINE=1
-e TRANSFORMERS_OFFLINE=1
--entrypoint python3
"${DOCKER_CLIENT_IMAGE}"
-m "${SGLANG_BENCH_MODULE}"
--backend sglang
--host "${HEAD_IP}"
--port "${SGLANG_PORT}"
--dataset-name "${BENCH_DATASET_NAME}"
--dataset-path "${DATASET_PATH}"
--random-input-len "${isl}"
--random-output-len "${osl}"
--random-range-ratio 1.0
--num-prompts "${prompts}"
--max-concurrency "${concurrency}"
--request-rate "${REQUEST_RATE}"
--output-file "${output_file}"
--output-details
--disable-tqdm
--warmup-requests 0
--seed "${seed}"
)
}
run_bench() {
local case_id="$1"
local isl="$2"
local osl="$3"
local concurrency="$4"
local prompts="$5"
local seed="$6"
local case_path="${CASE_DIR}/${case_id}"
mkdir -p "${case_path}"
prepare_bench_command "${case_path}/bench.jsonl" \
"${isl}" "${osl}" "${concurrency}" "${prompts}" "${seed}"
print_command timeout --signal=TERM --kill-after=30s \
"${PROFILE_TIMEOUT_S}s" "${BENCH_CMD[@]}" > "${case_path}/bench_cmd.txt"
if [[ "${DRY_RUN}" == "1" ]]; then
printf '[DRY] '
cat "${case_path}/bench_cmd.txt"
return 0
fi
timeout --signal=TERM --kill-after=30s "${PROFILE_TIMEOUT_S}s" \
"${BENCH_CMD[@]}" > "${case_path}/bench.log" 2>&1
}
run_profiled_case() {
local case_id="$1"
local activities_json="$2"
local steps="$3"
local isl="$4"
local osl="$5"
local concurrency="$6"
local prompts="$7"
local seed="$8"
local before=0
flush_cache
[[ "${DRY_RUN}" == "1" ]] || before="$(profile_done_count)"
start_profile "${case_id}" "${activities_json}" "${steps}"
run_bench "${case_id}" "${isl}" "${osl}" "${concurrency}" "${prompts}" "${seed}"
[[ "${DRY_RUN}" == "1" ]] || wait_for_profile_stop "${before}"
}
wait_for_bench_main() {
local bench_log="$1"
local bench_pid="$2"
local waited=0
while (( waited < 180 )); do
grep -Fq 'Starting main benchmark run' "${bench_log}" 2>/dev/null && return 0
kill -0 "${bench_pid}" 2>/dev/null || return 1
sleep 1
((waited+=1))
done
return 1
}
run_mixed_profile() {
local case_id="mixed_decode_with_128k_prefill"
local background_id="${case_id}/decode_background"
local injection_id="${case_id}/prefill_injection"
local background_path="${CASE_DIR}/${background_id}"
local before=0
flush_cache
mkdir -p "${background_path}"
prepare_bench_command "${background_path}/bench.jsonl" \
"${MIXED_ISL}" "${MIXED_OSL}" "${MIXED_CONCURRENCY}" "${MIXED_PROMPTS}" 5201
print_command timeout --signal=TERM --kill-after=30s \
"${PROFILE_TIMEOUT_S}s" "${BENCH_CMD[@]}" > "${background_path}/bench_cmd.txt"
if [[ "${DRY_RUN}" == "1" ]]; then
log "[DRY] start decode background, wait for main run, start profile, inject 128K prefill"
start_profile "${case_id}" '["CUDA_PROFILER"]' "${NSYS_MIXED_STEPS}"
run_bench "${injection_id}" "${MIXED_INJECTION_ISL}" \
"${MIXED_INJECTION_OSL}" 1 1 5202
return 0
fi
timeout --signal=TERM --kill-after=30s "${PROFILE_TIMEOUT_S}s" \
"${BENCH_CMD[@]}" > "${background_path}/bench.log" 2>&1 &
local background_pid=$!
wait_for_bench_main "${background_path}/bench.log" "${background_pid}" || {
kill "${background_pid}" 2>/dev/null || true
wait "${background_pid}" 2>/dev/null || true
log "ERROR: mixed decode background did not enter main benchmark"
return 1
}
before="$(profile_done_count)"
start_profile "${case_id}" '["CUDA_PROFILER"]' "${NSYS_MIXED_STEPS}"
run_bench "${injection_id}" "${MIXED_INJECTION_ISL}" \
"${MIXED_INJECTION_OSL}" 1 1 5202
wait_for_profile_stop "${before}"
wait "${background_pid}"
}
write_manifest() {
local status="$1"
local commit dirty
commit="$(git -C "${REPO_ROOT}" rev-parse HEAD 2>/dev/null || printf unknown)"
dirty=false
[[ -n "$(git -C "${REPO_ROOT}" status --porcelain 2>/dev/null || true)" ]] && dirty=true
cat > "${RESULT_DIR}/manifest.json" <<EOF
{
"run_id": "${RUN_ID}",
"status": "${status}",
"action": "${ACTION}",
"profile_mode": "${PROFILE_MODE}",
"started_at": "${RUN_STARTED_AT}",
"updated_at": "$(iso_now)",
"git_commit": "${commit}",
"git_dirty": ${dirty},
"head_node": "${HEAD_NODE}",
"worker_node": "${WORKER_NODE}",
"docker_image": "${DOCKER_IMAGE}",
"cuda_graph_enabled": true,
"tp_size": ${TP_SIZE},
"ep_size": ${EP_SIZE}
}
EOF
}
generate_nsys_stats() {
[[ "${DRY_RUN}" == "1" ]] && return 0
local report relative output
while IFS= read -r report; do
relative="${report#${RESULT_DIR}/}"
output="${report%.nsys-rep}.stats.txt"
docker run --rm \
-v "${RESULT_DIR}:/results:ro" \
--entrypoint nsys "${DOCKER_IMAGE}" stats \
--report cuda_gpu_kern_sum,cuda_api_sum,cuda_kern_exec_sum,nvtx_sum \
"/results/${relative}" > "${output}" 2>&1 || true
done < <(find "${RESULT_DIR}/profiles" -type f -name '*.nsys-rep' | sort)
}
cleanup() {
if [[ "${CLEANUP_ON_EXIT}" == "1" ]]; then
stop_service || true
fi
}
run_torch_smoke() {
PROFILE_MODE="torch"
CAPTURE_RANGES=0
CLEANUP_ON_EXIT=1
start_service
write_manifest RUNNING
run_profiled_case torch_decode_control '["CPU","GPU"]' \
"${TORCH_SMOKE_STEPS}" "${CONTROL_ISL}" 128 8 8 5101
stop_service
CLEANUP_ON_EXIT=0
write_manifest COMPLETED
}
run_nsys_smoke() {
PROFILE_MODE="nsys"
CAPTURE_RANGES=1
CLEANUP_ON_EXIT=1
start_service
write_manifest RUNNING
run_profiled_case nsys_decode_control '["CUDA_PROFILER"]' \
"${NSYS_SMOKE_STEPS}" "${CONTROL_ISL}" "${CONTROL_OSL}" \
"${CONTROL_CONCURRENCY}" "${CONTROL_PROMPTS}" 5102
stop_service
CLEANUP_ON_EXIT=0
generate_nsys_stats
write_manifest COMPLETED
}
run_all() {
PROFILE_MODE="nsys"
CAPTURE_RANGES=3
CLEANUP_ON_EXIT=1
start_service
write_manifest RUNNING
run_profiled_case decode_control '["CUDA_PROFILER"]' \
"${NSYS_CONTROL_STEPS}" "${CONTROL_ISL}" "${CONTROL_OSL}" \
"${CONTROL_CONCURRENCY}" "${CONTROL_PROMPTS}" 5301
run_mixed_profile
run_profiled_case long_prefill '["CUDA_PROFILER"]' \
"${NSYS_PREFILL_STEPS}" "${PREFILL_ISL}" "${PREFILL_OSL}" 1 1 5303
stop_service
CLEANUP_ON_EXIT=0
generate_nsys_stats
write_manifest COMPLETED
}
trap cleanup EXIT INT TERM
case "${ACTION}" in
audit)
audit
;;
torch-smoke)
run_torch_smoke
;;
nsys-smoke)
run_nsys_smoke
;;
all)
run_all
;;
stop)
CLEANUP_ON_EXIT=1
cleanup
CLEANUP_ON_EXIT=0
;;
*)
printf 'Usage: %s {audit|torch-smoke|nsys-smoke|all|stop}\n' "$0" >&2
exit 2
;;
esac
log "Phase 3 action=${ACTION} complete; results=${RESULT_DIR}"