[Feat] finalize Phase 2 hardware attribution pipeline
This commit is contained in:
parent
a583c337ba
commit
30664faa41
@ -0,0 +1,227 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Small PCIe P2P and NCCL AllReduce baseline for the Phase 2 entry."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import os
|
||||||
|
import statistics
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import torch
|
||||||
|
import torch.distributed as dist
|
||||||
|
|
||||||
|
|
||||||
|
RESULT_PREFIX = "COMM_RESULT "
|
||||||
|
|
||||||
|
|
||||||
|
def parse_size(value: str) -> int:
|
||||||
|
text = value.strip().upper()
|
||||||
|
units = {"K": 1 << 10, "M": 1 << 20, "G": 1 << 30}
|
||||||
|
if text[-1:] in units:
|
||||||
|
return int(float(text[:-1]) * units[text[-1]])
|
||||||
|
return int(text)
|
||||||
|
|
||||||
|
|
||||||
|
def percentile(values: list[float], quantile: float) -> float:
|
||||||
|
ordered = sorted(values)
|
||||||
|
if not ordered:
|
||||||
|
return math.nan
|
||||||
|
if len(ordered) == 1:
|
||||||
|
return ordered[0]
|
||||||
|
rank = (len(ordered) - 1) * quantile
|
||||||
|
lower = math.floor(rank)
|
||||||
|
upper = math.ceil(rank)
|
||||||
|
weight = rank - lower
|
||||||
|
return ordered[lower] * (1 - weight) + ordered[upper] * weight
|
||||||
|
|
||||||
|
|
||||||
|
def emit(value: dict[str, Any]) -> None:
|
||||||
|
print(RESULT_PREFIX + json.dumps(value, sort_keys=True), flush=True)
|
||||||
|
|
||||||
|
|
||||||
|
def run_p2p(args: argparse.Namespace) -> None:
|
||||||
|
device_count = torch.cuda.device_count()
|
||||||
|
size_bytes = parse_size(args.size)
|
||||||
|
elements = max(1, size_bytes // torch.tensor([], dtype=torch.float16).element_size())
|
||||||
|
for source in range(device_count):
|
||||||
|
for destination in range(device_count):
|
||||||
|
if source == destination:
|
||||||
|
continue
|
||||||
|
supported = torch.cuda.can_device_access_peer(source, destination)
|
||||||
|
if not supported:
|
||||||
|
emit(
|
||||||
|
{
|
||||||
|
"test": "p2p_copy",
|
||||||
|
"node": args.node,
|
||||||
|
"source_gpu": source,
|
||||||
|
"destination_gpu": destination,
|
||||||
|
"peer_access": False,
|
||||||
|
"size_bytes": size_bytes,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
with torch.cuda.device(source):
|
||||||
|
source_tensor = torch.ones(elements, dtype=torch.float16, device=source)
|
||||||
|
with torch.cuda.device(destination):
|
||||||
|
destination_tensor = torch.empty(
|
||||||
|
elements,
|
||||||
|
dtype=torch.float16,
|
||||||
|
device=destination,
|
||||||
|
)
|
||||||
|
for _ in range(args.warmup):
|
||||||
|
destination_tensor.copy_(source_tensor, non_blocking=True)
|
||||||
|
torch.cuda.synchronize(destination)
|
||||||
|
|
||||||
|
samples_ms: list[float] = []
|
||||||
|
for _ in range(args.iterations):
|
||||||
|
start = torch.cuda.Event(enable_timing=True)
|
||||||
|
end = torch.cuda.Event(enable_timing=True)
|
||||||
|
start.record()
|
||||||
|
destination_tensor.copy_(source_tensor, non_blocking=True)
|
||||||
|
end.record()
|
||||||
|
end.synchronize()
|
||||||
|
samples_ms.append(start.elapsed_time(end))
|
||||||
|
|
||||||
|
mean_ms = statistics.fmean(samples_ms)
|
||||||
|
emit(
|
||||||
|
{
|
||||||
|
"test": "p2p_copy",
|
||||||
|
"node": args.node,
|
||||||
|
"source_gpu": source,
|
||||||
|
"destination_gpu": destination,
|
||||||
|
"peer_access": True,
|
||||||
|
"size_bytes": size_bytes,
|
||||||
|
"iterations": args.iterations,
|
||||||
|
"mean_ms": mean_ms,
|
||||||
|
"p50_ms": percentile(samples_ms, 0.50),
|
||||||
|
"p95_ms": percentile(samples_ms, 0.95),
|
||||||
|
"bandwidth_GBps": size_bytes / (mean_ms / 1000.0) / 1e9,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def run_all_reduce(args: argparse.Namespace) -> None:
|
||||||
|
dist.init_process_group("nccl")
|
||||||
|
rank = dist.get_rank()
|
||||||
|
world_size = dist.get_world_size()
|
||||||
|
local_rank = int(os.environ["LOCAL_RANK"])
|
||||||
|
torch.cuda.set_device(local_rank)
|
||||||
|
|
||||||
|
for size_text in args.sizes.split(","):
|
||||||
|
size_bytes = parse_size(size_text)
|
||||||
|
elements = max(
|
||||||
|
1,
|
||||||
|
size_bytes // torch.tensor([], dtype=torch.float32).element_size(),
|
||||||
|
)
|
||||||
|
tensor = torch.empty(elements, dtype=torch.float32, device=local_rank)
|
||||||
|
expected = world_size * (world_size + 1) / 2
|
||||||
|
|
||||||
|
for repetition in range(1, args.repetitions + 1):
|
||||||
|
for _ in range(args.warmup):
|
||||||
|
tensor.fill_(rank + 1)
|
||||||
|
dist.all_reduce(tensor)
|
||||||
|
torch.cuda.synchronize(local_rank)
|
||||||
|
|
||||||
|
local_samples_ms: list[float] = []
|
||||||
|
for _ in range(args.iterations):
|
||||||
|
tensor.fill_(rank + 1)
|
||||||
|
torch.cuda.synchronize(local_rank)
|
||||||
|
dist.barrier()
|
||||||
|
start = torch.cuda.Event(enable_timing=True)
|
||||||
|
end = torch.cuda.Event(enable_timing=True)
|
||||||
|
start.record()
|
||||||
|
dist.all_reduce(tensor)
|
||||||
|
end.record()
|
||||||
|
end.synchronize()
|
||||||
|
local_samples_ms.append(start.elapsed_time(end))
|
||||||
|
|
||||||
|
wrong_values = int(
|
||||||
|
not torch.allclose(
|
||||||
|
tensor[0],
|
||||||
|
torch.tensor(expected, device=local_rank),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
gathered_samples: list[list[float] | None] = [None] * world_size
|
||||||
|
gathered_wrong: list[int | None] = [None] * world_size
|
||||||
|
dist.all_gather_object(gathered_samples, local_samples_ms)
|
||||||
|
dist.all_gather_object(gathered_wrong, wrong_values)
|
||||||
|
if rank == 0:
|
||||||
|
per_iteration_max = [
|
||||||
|
max(
|
||||||
|
float(samples[index])
|
||||||
|
for samples in gathered_samples
|
||||||
|
if samples is not None
|
||||||
|
)
|
||||||
|
for index in range(args.iterations)
|
||||||
|
]
|
||||||
|
mean_ms = statistics.fmean(per_iteration_max)
|
||||||
|
algbw = size_bytes / (mean_ms / 1000.0) / 1e9
|
||||||
|
emit(
|
||||||
|
{
|
||||||
|
"test": "all_reduce",
|
||||||
|
"scope": args.scope,
|
||||||
|
"world_size": world_size,
|
||||||
|
"size_bytes": size_bytes,
|
||||||
|
"repetition": repetition,
|
||||||
|
"iterations": args.iterations,
|
||||||
|
"mean_ms": mean_ms,
|
||||||
|
"p50_ms": percentile(per_iteration_max, 0.50),
|
||||||
|
"p95_ms": percentile(per_iteration_max, 0.95),
|
||||||
|
"algbw_GBps": algbw,
|
||||||
|
"busbw_GBps": algbw * 2 * (world_size - 1) / world_size,
|
||||||
|
"wrong_values": sum(
|
||||||
|
int(value or 0) for value in gathered_wrong
|
||||||
|
),
|
||||||
|
"nccl_cross_nic": os.environ.get(
|
||||||
|
"NCCL_CROSS_NIC",
|
||||||
|
"",
|
||||||
|
),
|
||||||
|
"nccl_socket_ifname": os.environ.get(
|
||||||
|
"NCCL_SOCKET_IFNAME",
|
||||||
|
"",
|
||||||
|
),
|
||||||
|
"nccl_ib_hca": os.environ.get("NCCL_IB_HCA", ""),
|
||||||
|
"torch_version": torch.__version__,
|
||||||
|
"cuda_version": torch.version.cuda,
|
||||||
|
"nccl_version": ".".join(
|
||||||
|
str(part) for part in torch.cuda.nccl.version()
|
||||||
|
),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
dist.destroy_process_group()
|
||||||
|
|
||||||
|
|
||||||
|
def build_parser() -> argparse.ArgumentParser:
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
subparsers = parser.add_subparsers(dest="command", required=True)
|
||||||
|
|
||||||
|
p2p = subparsers.add_parser("p2p")
|
||||||
|
p2p.add_argument("--node", required=True)
|
||||||
|
p2p.add_argument("--size", default="256M")
|
||||||
|
p2p.add_argument("--warmup", type=int, default=3)
|
||||||
|
p2p.add_argument("--iterations", type=int, default=10)
|
||||||
|
|
||||||
|
all_reduce = subparsers.add_parser("all-reduce")
|
||||||
|
all_reduce.add_argument("--scope", required=True)
|
||||||
|
all_reduce.add_argument("--sizes", default="1M,64M,1G")
|
||||||
|
all_reduce.add_argument("--repetitions", type=int, default=3)
|
||||||
|
all_reduce.add_argument("--warmup", type=int, default=5)
|
||||||
|
all_reduce.add_argument("--iterations", type=int, default=10)
|
||||||
|
return parser
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
args = build_parser().parse_args()
|
||||||
|
if args.command == "p2p":
|
||||||
|
run_p2p(args)
|
||||||
|
else:
|
||||||
|
run_all_reduce(args)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@ -21,6 +21,10 @@ RUN_MIXED_CASE="${RUN_MIXED_CASE:-1}"
|
|||||||
|
|
||||||
# Monitoring policy.
|
# Monitoring policy.
|
||||||
SAMPLE_INTERVAL_S="${SAMPLE_INTERVAL_S:-1}"
|
SAMPLE_INTERVAL_S="${SAMPLE_INTERVAL_S:-1}"
|
||||||
|
CPU_SAMPLE_INTERVAL_S="${CPU_SAMPLE_INTERVAL_S:-5}"
|
||||||
|
PROCESS_SAMPLE_INTERVAL_S="${PROCESS_SAMPLE_INTERVAL_S:-5}"
|
||||||
|
NET_SAMPLE_INTERVAL_S="${NET_SAMPLE_INTERVAL_S:-5}"
|
||||||
|
PERF_INTERVAL_MS="${PERF_INTERVAL_MS:-5000}"
|
||||||
IDLE_BASELINE_S="${IDLE_BASELINE_S:-15}"
|
IDLE_BASELINE_S="${IDLE_BASELINE_S:-15}"
|
||||||
POST_RUN_COOLDOWN_S="${POST_RUN_COOLDOWN_S:-15}"
|
POST_RUN_COOLDOWN_S="${POST_RUN_COOLDOWN_S:-15}"
|
||||||
CASE_COOLDOWN_S="${CASE_COOLDOWN_S:-5}"
|
CASE_COOLDOWN_S="${CASE_COOLDOWN_S:-5}"
|
||||||
@ -30,6 +34,23 @@ PERF_EVENTS="${PERF_EVENTS:-cycles,instructions,cache-misses,context-switches,cp
|
|||||||
RDMA_HCAS="${RDMA_HCAS:-mlx5_0 mlx5_3}"
|
RDMA_HCAS="${RDMA_HCAS:-mlx5_0 mlx5_3}"
|
||||||
NUMASTAT_INTERVAL_S="${NUMASTAT_INTERVAL_S:-5}"
|
NUMASTAT_INTERVAL_S="${NUMASTAT_INTERVAL_S:-5}"
|
||||||
CLOCK_SKEW_TOLERANCE_S="${CLOCK_SKEW_TOLERANCE_S:-2}"
|
CLOCK_SKEW_TOLERANCE_S="${CLOCK_SKEW_TOLERANCE_S:-2}"
|
||||||
|
REQUIRE_PRECISE_WINDOWS="${REQUIRE_PRECISE_WINDOWS:-1}"
|
||||||
|
|
||||||
|
# One-time communication baseline. It runs before the model service starts.
|
||||||
|
RUN_COMMUNICATION_BASELINE="${RUN_COMMUNICATION_BASELINE:-1}"
|
||||||
|
COMMUNICATION_IMAGE="${COMMUNICATION_IMAGE:-lmsysorg/sglang:nightly-dev-cu13-20260720-b3570a45}"
|
||||||
|
COMMUNICATION_MASTER_PORT="${COMMUNICATION_MASTER_PORT:-29620}"
|
||||||
|
COMMUNICATION_SIZES="${COMMUNICATION_SIZES:-1M,64M,1G}"
|
||||||
|
COMMUNICATION_REPETITIONS="${COMMUNICATION_REPETITIONS:-3}"
|
||||||
|
COMMUNICATION_WARMUP="${COMMUNICATION_WARMUP:-5}"
|
||||||
|
COMMUNICATION_ITERATIONS="${COMMUNICATION_ITERATIONS:-10}"
|
||||||
|
P2P_SIZE="${P2P_SIZE:-256M}"
|
||||||
|
P2P_WARMUP="${P2P_WARMUP:-3}"
|
||||||
|
P2P_ITERATIONS="${P2P_ITERATIONS:-10}"
|
||||||
|
CROSS_NIC_VALUES="${CROSS_NIC_VALUES:-0 1 2}"
|
||||||
|
NCCL_SOCKET_IFNAME="${NCCL_SOCKET_IFNAME:-=eth0}"
|
||||||
|
NCCL_IB_HCA="${NCCL_IB_HCA:-=mlx5_0:1,mlx5_3:1}"
|
||||||
|
RDMA_DEVICE_PATHS="${RDMA_DEVICE_PATHS:-/dev/infiniband/rdma_cm /dev/infiniband/uverbs0 /dev/infiniband/uverbs3}"
|
||||||
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
REPO_ROOT="$(cd "${SCRIPT_DIR}/../../.." && pwd)"
|
REPO_ROOT="$(cd "${SCRIPT_DIR}/../../.." && pwd)"
|
||||||
@ -37,4 +58,4 @@ RESULT_BASE="${RESULT_BASE:-${SCRIPT_DIR}/results}"
|
|||||||
RUNTIME_BASE="${RUNTIME_BASE:-${SCRIPT_DIR}/runtime}"
|
RUNTIME_BASE="${RUNTIME_BASE:-${SCRIPT_DIR}/runtime}"
|
||||||
|
|
||||||
DRY_RUN="${DRY_RUN:-0}"
|
DRY_RUN="${DRY_RUN:-0}"
|
||||||
ALLOW_PARTIAL_COLLECTORS="${ALLOW_PARTIAL_COLLECTORS:-1}"
|
ALLOW_PARTIAL_COLLECTORS="${ALLOW_PARTIAL_COLLECTORS:-0}"
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@ -11,6 +11,7 @@ ACTION="${1:-all}"
|
|||||||
RUN_ID="${RUN_ID:-dsv4pro-phase2-$(date +%Y%m%d-%H%M%S)}"
|
RUN_ID="${RUN_ID:-dsv4pro-phase2-$(date +%Y%m%d-%H%M%S)}"
|
||||||
RESULT_DIR="${RESULT_BASE}/${RUN_ID}"
|
RESULT_DIR="${RESULT_BASE}/${RUN_ID}"
|
||||||
RESULT_TOOL="${SCRIPT_DIR}/hardware_contention_attribution.py"
|
RESULT_TOOL="${SCRIPT_DIR}/hardware_contention_attribution.py"
|
||||||
|
COMMUNICATION_TOOL="${SCRIPT_DIR}/communication_baseline.py"
|
||||||
MARKERS_PATH="${RESULT_DIR}/markers.csv"
|
MARKERS_PATH="${RESULT_DIR}/markers.csv"
|
||||||
SERVICE_DIR="${RESULT_DIR}/service"
|
SERVICE_DIR="${RESULT_DIR}/service"
|
||||||
COMMAND_DIR="${RESULT_DIR}/commands"
|
COMMAND_DIR="${RESULT_DIR}/commands"
|
||||||
@ -72,6 +73,10 @@ validate_config() {
|
|||||||
log "ERROR: result tool is missing or not executable: ${RESULT_TOOL}"
|
log "ERROR: result tool is missing or not executable: ${RESULT_TOOL}"
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
[[ -f "${COMMUNICATION_TOOL}" ]] || {
|
||||||
|
log "ERROR: communication tool is missing: ${COMMUNICATION_TOOL}"
|
||||||
|
return 1
|
||||||
|
}
|
||||||
case "${RUN_MIXED_CASE}" in
|
case "${RUN_MIXED_CASE}" in
|
||||||
0|1) ;;
|
0|1) ;;
|
||||||
*)
|
*)
|
||||||
@ -86,6 +91,20 @@ validate_config() {
|
|||||||
return 1
|
return 1
|
||||||
;;
|
;;
|
||||||
esac
|
esac
|
||||||
|
case "${REQUIRE_PRECISE_WINDOWS}" in
|
||||||
|
0|1) ;;
|
||||||
|
*)
|
||||||
|
log "ERROR: REQUIRE_PRECISE_WINDOWS must be 0 or 1"
|
||||||
|
return 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
case "${RUN_COMMUNICATION_BASELINE}" in
|
||||||
|
0|1) ;;
|
||||||
|
*)
|
||||||
|
log "ERROR: RUN_COMMUNICATION_BASELINE must be 0 or 1"
|
||||||
|
return 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
[[ "${SAMPLE_INTERVAL_S}" =~ ^[1-9][0-9]*$ ]] || {
|
[[ "${SAMPLE_INTERVAL_S}" =~ ^[1-9][0-9]*$ ]] || {
|
||||||
log "ERROR: SAMPLE_INTERVAL_S must be a positive integer"
|
log "ERROR: SAMPLE_INTERVAL_S must be a positive integer"
|
||||||
return 1
|
return 1
|
||||||
@ -94,6 +113,14 @@ validate_config() {
|
|||||||
log "ERROR: NUMASTAT_INTERVAL_S must be a positive integer"
|
log "ERROR: NUMASTAT_INTERVAL_S must be a positive integer"
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
local interval_name
|
||||||
|
for interval_name in CPU_SAMPLE_INTERVAL_S PROCESS_SAMPLE_INTERVAL_S \
|
||||||
|
NET_SAMPLE_INTERVAL_S PERF_INTERVAL_MS; do
|
||||||
|
[[ "${!interval_name}" =~ ^[1-9][0-9]*$ ]] || {
|
||||||
|
log "ERROR: ${interval_name} must be a positive integer"
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
done
|
||||||
[[ "${CLOCK_SKEW_TOLERANCE_S}" =~ ^[0-9]+$ ]] || {
|
[[ "${CLOCK_SKEW_TOLERANCE_S}" =~ ^[0-9]+$ ]] || {
|
||||||
log "ERROR: CLOCK_SKEW_TOLERANCE_S must be a non-negative integer"
|
log "ERROR: CLOCK_SKEW_TOLERANCE_S must be a non-negative integer"
|
||||||
return 1
|
return 1
|
||||||
@ -121,6 +148,13 @@ preflight_node_tools() {
|
|||||||
log "ERROR: ${node} is missing required tool: dcgmi"
|
log "ERROR: ${node} is missing required tool: dcgmi"
|
||||||
return 1
|
return 1
|
||||||
fi
|
fi
|
||||||
|
elif ! run_on_node "${node}" "dcgmi discovery -l >/dev/null 2>&1"; then
|
||||||
|
if [[ "${ALLOW_PARTIAL_COLLECTORS}" == "1" ]]; then
|
||||||
|
log "WARN: ${node} DCGM Host Engine is unavailable"
|
||||||
|
else
|
||||||
|
log "ERROR: ${node} DCGM Host Engine is unavailable; run systemctl start nvidia-dcgm"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
fi
|
fi
|
||||||
local hca
|
local hca
|
||||||
for hca in ${RDMA_HCAS}; do
|
for hca in ${RDMA_HCAS}; do
|
||||||
@ -149,6 +183,20 @@ preflight_clock_sync() {
|
|||||||
log "node clock skew check passed: delta=${delta}s"
|
log "node clock skew check passed: delta=${delta}s"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
preflight_gpus_idle() {
|
||||||
|
local node="$1"
|
||||||
|
local active
|
||||||
|
active="$(
|
||||||
|
run_on_node "${node}" \
|
||||||
|
"nvidia-smi --query-compute-apps=pid --format=csv,noheader,nounits 2>/dev/null | sed '/^[[:space:]]*$/d'" \
|
||||||
|
|| true
|
||||||
|
)"
|
||||||
|
if [[ -n "${active}" ]]; then
|
||||||
|
log "ERROR: ${node} has active GPU compute processes: ${active//$'\n'/,}"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
write_manifest() {
|
write_manifest() {
|
||||||
local git_commit git_dirty
|
local git_commit git_dirty
|
||||||
git_commit="$(git -C "${REPO_ROOT}" rev-parse HEAD 2>/dev/null || printf unknown)"
|
git_commit="$(git -C "${REPO_ROOT}" rev-parse HEAD 2>/dev/null || printf unknown)"
|
||||||
@ -167,10 +215,16 @@ write_manifest() {
|
|||||||
--fixed-case-ids "${FIXED_CASE_IDS}" \
|
--fixed-case-ids "${FIXED_CASE_IDS}" \
|
||||||
--run-mixed-case "${RUN_MIXED_CASE}" \
|
--run-mixed-case "${RUN_MIXED_CASE}" \
|
||||||
--sample-interval-s "${SAMPLE_INTERVAL_S}" \
|
--sample-interval-s "${SAMPLE_INTERVAL_S}" \
|
||||||
|
--cpu-sample-interval-s "${CPU_SAMPLE_INTERVAL_S}" \
|
||||||
|
--process-sample-interval-s "${PROCESS_SAMPLE_INTERVAL_S}" \
|
||||||
|
--net-sample-interval-s "${NET_SAMPLE_INTERVAL_S}" \
|
||||||
|
--perf-interval-ms "${PERF_INTERVAL_MS}" \
|
||||||
--numastat-interval-s "${NUMASTAT_INTERVAL_S}" \
|
--numastat-interval-s "${NUMASTAT_INTERVAL_S}" \
|
||||||
--clock-skew-tolerance-s "${CLOCK_SKEW_TOLERANCE_S}" \
|
--clock-skew-tolerance-s "${CLOCK_SKEW_TOLERANCE_S}" \
|
||||||
--idle-baseline-s "${IDLE_BASELINE_S}" \
|
--idle-baseline-s "${IDLE_BASELINE_S}" \
|
||||||
--post-run-cooldown-s "${POST_RUN_COOLDOWN_S}" \
|
--post-run-cooldown-s "${POST_RUN_COOLDOWN_S}" \
|
||||||
|
--require-precise-windows "${REQUIRE_PRECISE_WINDOWS}" \
|
||||||
|
--run-communication-baseline "${RUN_COMMUNICATION_BASELINE}" \
|
||||||
--dry-run "${DRY_RUN}"
|
--dry-run "${DRY_RUN}"
|
||||||
}
|
}
|
||||||
|
|
||||||
@ -212,6 +266,185 @@ run_phase1_action() {
|
|||||||
"${command[@]}"
|
"${command[@]}"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
build_communication_docker_command() {
|
||||||
|
local output_name="$1"
|
||||||
|
local container_name="$2"
|
||||||
|
local entrypoint="$3"
|
||||||
|
local cross_nic="$4"
|
||||||
|
shift 4
|
||||||
|
local -a docker_cmd=(
|
||||||
|
docker run --rm
|
||||||
|
--name "${container_name}"
|
||||||
|
--gpus all
|
||||||
|
--network host
|
||||||
|
--ipc host
|
||||||
|
--shm-size 20g
|
||||||
|
--ulimit memlock=-1
|
||||||
|
--ulimit stack=67108864
|
||||||
|
-v "${REPO_ROOT}:${REPO_ROOT}:ro"
|
||||||
|
-e "NCCL_SOCKET_IFNAME=${NCCL_SOCKET_IFNAME}"
|
||||||
|
-e "NCCL_IB_HCA=${NCCL_IB_HCA}"
|
||||||
|
-e "NCCL_CROSS_NIC=${cross_nic}"
|
||||||
|
-e NCCL_DEBUG=INFO
|
||||||
|
-e NCCL_DEBUG_SUBSYS=INIT,NET,GRAPH,TUNING
|
||||||
|
-e TORCH_NCCL_ASYNC_ERROR_HANDLING=1
|
||||||
|
)
|
||||||
|
local device
|
||||||
|
for device in ${RDMA_DEVICE_PATHS}; do
|
||||||
|
docker_cmd+=(--device "${device}")
|
||||||
|
done
|
||||||
|
docker_cmd+=(--entrypoint "${entrypoint}" "${COMMUNICATION_IMAGE}" "$@")
|
||||||
|
printf -v "${output_name}" '%q ' "${docker_cmd[@]}"
|
||||||
|
}
|
||||||
|
|
||||||
|
run_p2p_baseline() {
|
||||||
|
local role="$1"
|
||||||
|
local node="$2"
|
||||||
|
local container="${EXPERIMENT}_comm_p2p_${role}_${RUN_ID}"
|
||||||
|
local command
|
||||||
|
build_communication_docker_command \
|
||||||
|
command "${container}" python3 2 \
|
||||||
|
"${COMMUNICATION_TOOL}" p2p \
|
||||||
|
--node "${role}" \
|
||||||
|
--size "${P2P_SIZE}" \
|
||||||
|
--warmup "${P2P_WARMUP}" \
|
||||||
|
--iterations "${P2P_ITERATIONS}"
|
||||||
|
printf '%s\n' "${command}" \
|
||||||
|
> "${COMMAND_DIR}/communication_p2p_${role}.cmd.txt"
|
||||||
|
if [[ "${DRY_RUN}" == "1" ]]; then
|
||||||
|
log "[DRY] P2P baseline role=${role}: ${command}"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
log "START P2P baseline role=${role}"
|
||||||
|
run_on_node "${node}" "${command}" \
|
||||||
|
> "${RESULT_DIR}/communication/p2p_${role}.log" 2>&1
|
||||||
|
log "DONE P2P baseline role=${role}"
|
||||||
|
}
|
||||||
|
|
||||||
|
run_single_node_allreduce() {
|
||||||
|
local role="$1"
|
||||||
|
local node="$2"
|
||||||
|
local master_port="$3"
|
||||||
|
local container="${EXPERIMENT}_comm_ar8_${role}_${RUN_ID}"
|
||||||
|
local command
|
||||||
|
build_communication_docker_command \
|
||||||
|
command "${container}" torchrun 2 \
|
||||||
|
--standalone \
|
||||||
|
--nnodes=1 \
|
||||||
|
--nproc-per-node=8 \
|
||||||
|
--master-port "${master_port}" \
|
||||||
|
"${COMMUNICATION_TOOL}" all-reduce \
|
||||||
|
--scope "${role}_8gpu" \
|
||||||
|
--sizes "${COMMUNICATION_SIZES}" \
|
||||||
|
--repetitions "${COMMUNICATION_REPETITIONS}" \
|
||||||
|
--warmup "${COMMUNICATION_WARMUP}" \
|
||||||
|
--iterations "${COMMUNICATION_ITERATIONS}"
|
||||||
|
printf '%s\n' "${command}" \
|
||||||
|
> "${COMMAND_DIR}/communication_allreduce_${role}_8gpu.cmd.txt"
|
||||||
|
if [[ "${DRY_RUN}" == "1" ]]; then
|
||||||
|
log "[DRY] AllReduce baseline scope=${role}_8gpu: ${command}"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
log "START AllReduce baseline scope=${role}_8gpu"
|
||||||
|
run_on_node "${node}" "${command}" \
|
||||||
|
> "${RESULT_DIR}/communication/allreduce_${role}_8gpu.log" 2>&1
|
||||||
|
log "DONE AllReduce baseline scope=${role}_8gpu"
|
||||||
|
}
|
||||||
|
|
||||||
|
run_two_node_allreduce() {
|
||||||
|
local cross_nic="$1"
|
||||||
|
local master_port="$(( COMMUNICATION_MASTER_PORT + cross_nic + 10 ))"
|
||||||
|
local head_container="${EXPERIMENT}_comm_ar16_head_x${cross_nic}_${RUN_ID}"
|
||||||
|
local worker_container="${EXPERIMENT}_comm_ar16_worker_x${cross_nic}_${RUN_ID}"
|
||||||
|
local head_command worker_command
|
||||||
|
local -a common_args=(
|
||||||
|
--nnodes=2
|
||||||
|
--nproc-per-node=8
|
||||||
|
--master-addr "${HEAD_NODE}"
|
||||||
|
--master-port "${master_port}"
|
||||||
|
)
|
||||||
|
build_communication_docker_command \
|
||||||
|
worker_command "${worker_container}" torchrun "${cross_nic}" \
|
||||||
|
"${common_args[@]}" \
|
||||||
|
--node-rank=1 \
|
||||||
|
"${COMMUNICATION_TOOL}" all-reduce \
|
||||||
|
--scope two_node_16gpu \
|
||||||
|
--sizes "${COMMUNICATION_SIZES}" \
|
||||||
|
--repetitions "${COMMUNICATION_REPETITIONS}" \
|
||||||
|
--warmup "${COMMUNICATION_WARMUP}" \
|
||||||
|
--iterations "${COMMUNICATION_ITERATIONS}"
|
||||||
|
build_communication_docker_command \
|
||||||
|
head_command "${head_container}" torchrun "${cross_nic}" \
|
||||||
|
"${common_args[@]}" \
|
||||||
|
--node-rank=0 \
|
||||||
|
"${COMMUNICATION_TOOL}" all-reduce \
|
||||||
|
--scope two_node_16gpu \
|
||||||
|
--sizes "${COMMUNICATION_SIZES}" \
|
||||||
|
--repetitions "${COMMUNICATION_REPETITIONS}" \
|
||||||
|
--warmup "${COMMUNICATION_WARMUP}" \
|
||||||
|
--iterations "${COMMUNICATION_ITERATIONS}"
|
||||||
|
printf '%s\n' "${head_command}" \
|
||||||
|
> "${COMMAND_DIR}/communication_allreduce_16gpu_crossnic${cross_nic}_head.cmd.txt"
|
||||||
|
printf '%s\n' "${worker_command}" \
|
||||||
|
> "${COMMAND_DIR}/communication_allreduce_16gpu_crossnic${cross_nic}_worker.cmd.txt"
|
||||||
|
if [[ "${DRY_RUN}" == "1" ]]; then
|
||||||
|
log "[DRY] AllReduce 16 GPU cross_nic=${cross_nic} Head: ${head_command}"
|
||||||
|
log "[DRY] AllReduce 16 GPU cross_nic=${cross_nic} Worker: ${worker_command}"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
log "START AllReduce baseline scope=two_node_16gpu cross_nic=${cross_nic}"
|
||||||
|
set +e
|
||||||
|
run_on_node "${WORKER_NODE}" "${worker_command}" \
|
||||||
|
> "${RESULT_DIR}/communication/allreduce_16gpu_crossnic${cross_nic}_worker.log" \
|
||||||
|
2>&1 &
|
||||||
|
local worker_pid=$!
|
||||||
|
sleep 3
|
||||||
|
run_on_node "${HEAD_NODE}" "${head_command}" \
|
||||||
|
> "${RESULT_DIR}/communication/allreduce_16gpu_crossnic${cross_nic}_head.log" \
|
||||||
|
2>&1
|
||||||
|
local head_rc=$?
|
||||||
|
wait "${worker_pid}"
|
||||||
|
local worker_rc=$?
|
||||||
|
set -e
|
||||||
|
if (( head_rc != 0 || worker_rc != 0 )); then
|
||||||
|
log "ERROR: AllReduce cross_nic=${cross_nic} head_rc=${head_rc} worker_rc=${worker_rc}"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
log "DONE AllReduce baseline scope=two_node_16gpu cross_nic=${cross_nic}"
|
||||||
|
}
|
||||||
|
|
||||||
|
cleanup_communication_containers() {
|
||||||
|
[[ "${DRY_RUN}" == "1" ]] && return 0
|
||||||
|
local node
|
||||||
|
for node in "${HEAD_NODE}" "${WORKER_NODE}"; do
|
||||||
|
run_on_node "${node}" \
|
||||||
|
"docker ps -aq --filter 'name=^/${EXPERIMENT}_comm_' | xargs -r docker rm -f >/dev/null 2>&1" \
|
||||||
|
|| true
|
||||||
|
done
|
||||||
|
}
|
||||||
|
|
||||||
|
run_communication_baseline() {
|
||||||
|
[[ "${RUN_COMMUNICATION_BASELINE}" == "1" ]] || {
|
||||||
|
log "SKIP communication baseline by configuration"
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
mkdir -p "${RESULT_DIR}/communication" "${COMMAND_DIR}"
|
||||||
|
if [[ "${DRY_RUN}" != "1" ]]; then
|
||||||
|
preflight_gpus_idle "${HEAD_NODE}"
|
||||||
|
preflight_gpus_idle "${WORKER_NODE}"
|
||||||
|
fi
|
||||||
|
run_p2p_baseline head "${HEAD_NODE}"
|
||||||
|
run_p2p_baseline worker "${WORKER_NODE}"
|
||||||
|
run_single_node_allreduce head "${HEAD_NODE}" "${COMMUNICATION_MASTER_PORT}"
|
||||||
|
run_single_node_allreduce worker "${WORKER_NODE}" "$(( COMMUNICATION_MASTER_PORT + 1 ))"
|
||||||
|
local cross_nic
|
||||||
|
for cross_nic in ${CROSS_NIC_VALUES}; do
|
||||||
|
run_two_node_allreduce "${cross_nic}"
|
||||||
|
done
|
||||||
|
cleanup_communication_containers
|
||||||
|
}
|
||||||
|
|
||||||
start_service() {
|
start_service() {
|
||||||
mark_event service_start
|
mark_event service_start
|
||||||
# Set this before launch so the EXIT trap also cleans a partially started pair.
|
# Set this before launch so the EXIT trap also cleans a partially started pair.
|
||||||
@ -364,7 +597,7 @@ trap 'exit 0' HUP TERM PIPE
|
|||||||
while :; do
|
while :; do
|
||||||
printf 'wall_time_ns=%s\\n' \"\$(date +%s%N)\"
|
printf 'wall_time_ns=%s\\n' \"\$(date +%s%N)\"
|
||||||
docker top '${container}' -eo pid,ppid,psr,pcpu,pmem,stat,comm,args
|
docker top '${container}' -eo pid,ppid,psr,pcpu,pmem,stat,comm,args
|
||||||
sleep '${SAMPLE_INTERVAL_S}'
|
sleep '${PROCESS_SAMPLE_INTERVAL_S}'
|
||||||
done
|
done
|
||||||
"
|
"
|
||||||
}
|
}
|
||||||
@ -383,20 +616,27 @@ fi
|
|||||||
}
|
}
|
||||||
|
|
||||||
numastat_command() {
|
numastat_command() {
|
||||||
local container="$1"
|
local role="$1"
|
||||||
|
local container="$2"
|
||||||
printf '%s' "
|
printf '%s' "
|
||||||
trap 'exit 0' HUP TERM PIPE
|
trap 'exit 0' HUP TERM PIPE
|
||||||
|
printf '%s\n' 'wall_time_ns,node,node0_mib,node1_mib,total_mib,processes'
|
||||||
while :; do
|
while :; do
|
||||||
printf 'wall_time_ns=%s\\n' \"\$(date +%s%N)\"
|
|
||||||
PIDS=\$(docker top '${container}' -eo pid 2>/dev/null |
|
PIDS=\$(docker top '${container}' -eo pid 2>/dev/null |
|
||||||
awk 'NR > 1 {print \$1}')
|
awk 'NR > 1 {print \$1}')
|
||||||
if [[ -z \"\${PIDS}\" ]]; then
|
if [[ -z \"\${PIDS}\" ]]; then
|
||||||
printf 'container has no visible processes: %s\\n' '${container}' >&2
|
printf 'container has no visible processes: %s\\n' '${container}' >&2
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
values=\$(
|
||||||
for pid in \${PIDS}; do
|
for pid in \${PIDS}; do
|
||||||
numastat -p \"\${pid}\"
|
numastat -p \"\${pid}\" 2>/dev/null |
|
||||||
done
|
awk '\$1 == \"Total\" {print \$2, \$3, \$4}'
|
||||||
|
done |
|
||||||
|
awk '{n0 += \$1; n1 += \$2; total += \$3; count += 1}
|
||||||
|
END {printf \"%.2f,%.2f,%.2f,%d\", n0, n1, total, count}'
|
||||||
|
)
|
||||||
|
printf '%s,%s,%s\\n' \"\$(date +%s%N)\" '${role}' \"\${values}\"
|
||||||
sleep '${NUMASTAT_INTERVAL_S}'
|
sleep '${NUMASTAT_INTERVAL_S}'
|
||||||
done
|
done
|
||||||
"
|
"
|
||||||
@ -416,23 +656,36 @@ dcgmi dmon -e '${DCGM_FIELD_IDS}' -d '$(( SAMPLE_INTERVAL_S * 1000 ))' |
|
|||||||
"
|
"
|
||||||
local mpstat_command="
|
local mpstat_command="
|
||||||
trap 'exit 0' HUP TERM PIPE
|
trap 'exit 0' HUP TERM PIPE
|
||||||
mpstat -P ALL '${SAMPLE_INTERVAL_S}'
|
LC_ALL=C stdbuf -oL -eL mpstat -P ALL '${CPU_SAMPLE_INTERVAL_S}' |
|
||||||
|
while IFS= read -r line; do
|
||||||
|
printf '%s\\t%s\\t%s\\n' \"\$(date +%s%N)\" '${role}' \"\${line}\"
|
||||||
|
done
|
||||||
"
|
"
|
||||||
local pidstat_command
|
local pidstat_command
|
||||||
pidstat_command="$(container_pid_preamble "${container}")"
|
pidstat_command="$(container_pid_preamble "${container}")"
|
||||||
pidstat_command+="
|
pidstat_command+="
|
||||||
trap 'exit 0' HUP TERM PIPE
|
trap 'exit 0' HUP TERM PIPE
|
||||||
pidstat -durwt -p \"\${PIDS}\" '${SAMPLE_INTERVAL_S}'
|
LC_ALL=C stdbuf -oL -eL pidstat -durw -p \"\${PIDS}\" '${PROCESS_SAMPLE_INTERVAL_S}' |
|
||||||
|
while IFS= read -r line; do
|
||||||
|
printf '%s\\t%s\\t%s\\n' \"\$(date +%s%N)\" '${role}' \"\${line}\"
|
||||||
|
done
|
||||||
"
|
"
|
||||||
local sar_command="
|
local sar_command="
|
||||||
trap 'exit 0' HUP TERM PIPE
|
trap 'exit 0' HUP TERM PIPE
|
||||||
sar -n DEV,EDEV '${SAMPLE_INTERVAL_S}'
|
LC_ALL=C stdbuf -oL -eL sar -n DEV,EDEV '${NET_SAMPLE_INTERVAL_S}' |
|
||||||
|
while IFS= read -r line; do
|
||||||
|
printf '%s\\t%s\\t%s\\n' \"\$(date +%s%N)\" '${role}' \"\${line}\"
|
||||||
|
done
|
||||||
"
|
"
|
||||||
local perf_command
|
local perf_command
|
||||||
perf_command="$(container_pid_preamble "${container}")"
|
perf_command="$(container_pid_preamble "${container}")"
|
||||||
perf_command+="
|
perf_command+="
|
||||||
trap 'exit 0' HUP TERM PIPE
|
trap 'exit 0' HUP TERM PIPE
|
||||||
perf stat -p \"\${PIDS}\" -I 1000 -e '${PERF_EVENTS}'
|
LC_ALL=C stdbuf -oL -eL perf stat -p \"\${PIDS}\" \
|
||||||
|
-I '${PERF_INTERVAL_MS}' -e '${PERF_EVENTS}' 2>&1 |
|
||||||
|
while IFS= read -r line; do
|
||||||
|
printf '%s\\t%s\\t%s\\n' \"\$(date +%s%N)\" '${role}' \"\${line}\"
|
||||||
|
done
|
||||||
"
|
"
|
||||||
|
|
||||||
start_stream_collector \
|
start_stream_collector \
|
||||||
@ -452,7 +705,7 @@ perf stat -p \"\${PIDS}\" -I 1000 -e '${PERF_EVENTS}'
|
|||||||
start_stream_collector \
|
start_stream_collector \
|
||||||
"${role}" "${node}" docker_top.log "$(docker_top_command "${container}")"
|
"${role}" "${node}" docker_top.log "$(docker_top_command "${container}")"
|
||||||
start_stream_collector \
|
start_stream_collector \
|
||||||
"${role}" "${node}" numastat.log "$(numastat_command "${container}")"
|
"${role}" "${node}" numa_samples.csv "$(numastat_command "${role}" "${container}")"
|
||||||
}
|
}
|
||||||
|
|
||||||
start_collectors() {
|
start_collectors() {
|
||||||
@ -581,7 +834,8 @@ run_mixed_case() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
summarize_results() {
|
summarize_results() {
|
||||||
python3 "${RESULT_TOOL}" summarize "${RESULT_DIR}"
|
python3 "${RESULT_TOOL}" summarize "${RESULT_DIR}" \
|
||||||
|
--require-precise-windows "${REQUIRE_PRECISE_WINDOWS}"
|
||||||
}
|
}
|
||||||
|
|
||||||
finish_manifest() {
|
finish_manifest() {
|
||||||
@ -595,6 +849,7 @@ cleanup() {
|
|||||||
local rc=$?
|
local rc=$?
|
||||||
stop_collectors || true
|
stop_collectors || true
|
||||||
stop_service || true
|
stop_service || true
|
||||||
|
cleanup_communication_containers || true
|
||||||
if (( rc != 0 )) && [[ -f "${RESULT_DIR}/manifest.json" ]]; then
|
if (( rc != 0 )) && [[ -f "${RESULT_DIR}/manifest.json" ]]; then
|
||||||
finish_manifest ABORTED || true
|
finish_manifest ABORTED || true
|
||||||
fi
|
fi
|
||||||
@ -620,6 +875,7 @@ run_all() {
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
trap cleanup EXIT INT TERM
|
trap cleanup EXIT INT TERM
|
||||||
|
run_communication_baseline
|
||||||
start_service
|
start_service
|
||||||
capture_static_snapshots before
|
capture_static_snapshots before
|
||||||
start_collectors
|
start_collectors
|
||||||
@ -652,6 +908,10 @@ run_all() {
|
|||||||
mark_event cooldown_start
|
mark_event cooldown_start
|
||||||
sleep_if_real "${POST_RUN_COOLDOWN_S}"
|
sleep_if_real "${POST_RUN_COOLDOWN_S}"
|
||||||
mark_event cooldown_end
|
mark_event cooldown_end
|
||||||
|
if ! check_collectors; then
|
||||||
|
log "ERROR: one or more required collectors exited during the run"
|
||||||
|
((failures+=1))
|
||||||
|
fi
|
||||||
stop_collectors
|
stop_collectors
|
||||||
capture_static_snapshots after
|
capture_static_snapshots after
|
||||||
stop_service
|
stop_service
|
||||||
@ -666,6 +926,22 @@ run_all() {
|
|||||||
(( failures == 0 ))
|
(( failures == 0 ))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
run_communication_only() {
|
||||||
|
validate_config
|
||||||
|
enable_result_logging
|
||||||
|
mkdir -p "${RESULT_DIR}" "${COMMAND_DIR}" "${RESULT_DIR}/communication"
|
||||||
|
if [[ "${DRY_RUN}" != "1" ]]; then
|
||||||
|
preflight_node_tools "${HEAD_NODE}"
|
||||||
|
preflight_node_tools "${WORKER_NODE}"
|
||||||
|
preflight_clock_sync
|
||||||
|
fi
|
||||||
|
trap cleanup_communication_containers EXIT INT TERM
|
||||||
|
run_communication_baseline
|
||||||
|
python3 "${RESULT_TOOL}" summarize-communication "${RESULT_DIR}"
|
||||||
|
trap - EXIT INT TERM
|
||||||
|
log "Communication baseline complete: result=${RESULT_DIR}"
|
||||||
|
}
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
case "${ACTION}" in
|
case "${ACTION}" in
|
||||||
all)
|
all)
|
||||||
@ -675,12 +951,15 @@ main() {
|
|||||||
enable_result_logging
|
enable_result_logging
|
||||||
summarize_results
|
summarize_results
|
||||||
;;
|
;;
|
||||||
|
communication)
|
||||||
|
run_communication_only
|
||||||
|
;;
|
||||||
stop)
|
stop)
|
||||||
enable_result_logging
|
enable_result_logging
|
||||||
stop_service
|
stop_service
|
||||||
;;
|
;;
|
||||||
*)
|
*)
|
||||||
printf 'Usage: %s {all|summarize|stop}\n' "$0" >&2
|
printf 'Usage: %s {all|communication|summarize|stop}\n' "$0" >&2
|
||||||
return 2
|
return 2
|
||||||
;;
|
;;
|
||||||
esac
|
esac
|
||||||
|
|||||||
@ -113,6 +113,10 @@ class HardwareContentionAttributionTest(unittest.TestCase):
|
|||||||
rows = attribution.read_csv_rows(result_dir / "bench_summary.csv")
|
rows = attribution.read_csv_rows(result_dir / "bench_summary.csv")
|
||||||
self.assertEqual(rows[0]["phase2_bench_run"], "prefill")
|
self.assertEqual(rows[0]["phase2_bench_run"], "prefill")
|
||||||
self.assertTrue((result_dir / "report.md").exists())
|
self.assertTrue((result_dir / "report.md").exists())
|
||||||
|
report = (result_dir / "report.md").read_text(encoding="utf-8")
|
||||||
|
self.assertIn("Samples CPU/process/perf", report)
|
||||||
|
self.assertIn("Samples/repetitions", report)
|
||||||
|
self.assertIn("missing samples are reported as `-`", report)
|
||||||
|
|
||||||
def test_case_hardware_is_cut_by_phase1_meta_window(self) -> None:
|
def test_case_hardware_is_cut_by_phase1_meta_window(self) -> None:
|
||||||
with tempfile.TemporaryDirectory() as temporary:
|
with tempfile.TemporaryDirectory() as temporary:
|
||||||
@ -179,6 +183,113 @@ class HardwareContentionAttributionTest(unittest.TestCase):
|
|||||||
"long_prefill_latency_128k_c1",
|
"long_prefill_latency_128k_c1",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def test_case_window_prefers_precise_main_benchmark_fields(self) -> None:
|
||||||
|
with tempfile.TemporaryDirectory() as temporary:
|
||||||
|
result_dir = Path(temporary)
|
||||||
|
meta_path = (
|
||||||
|
result_dir
|
||||||
|
/ "bench"
|
||||||
|
/ "decode"
|
||||||
|
/ "cases"
|
||||||
|
/ "decode_case"
|
||||||
|
/ "rep1"
|
||||||
|
/ "meta.json"
|
||||||
|
)
|
||||||
|
meta_path.parent.mkdir(parents=True)
|
||||||
|
meta_path.write_text(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"case_id": "decode_case",
|
||||||
|
"role": "",
|
||||||
|
"stage": "decode",
|
||||||
|
"repetition": 1,
|
||||||
|
"status": "COMPLETED",
|
||||||
|
"started_at": "2026-07-31T00:00:00+00:00",
|
||||||
|
"ended_at": "2026-07-31T00:01:00+00:00",
|
||||||
|
"measurement_started_at": "2026-07-31T00:00:10+00:00",
|
||||||
|
"measurement_ended_at": "2026-07-31T00:00:30+00:00",
|
||||||
|
"measurement_window_source": "bench_main_marker_plus_duration",
|
||||||
|
}
|
||||||
|
),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
|
||||||
|
windows = attribution.load_case_windows(result_dir)
|
||||||
|
|
||||||
|
self.assertEqual(len(windows), 1)
|
||||||
|
self.assertEqual(windows[0]["duration_s"], 20)
|
||||||
|
self.assertEqual(
|
||||||
|
windows[0]["window_source"],
|
||||||
|
"bench_main_marker_plus_duration",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_dcgm_and_timestamped_cpu_parsers(self) -> None:
|
||||||
|
with tempfile.TemporaryDirectory() as temporary:
|
||||||
|
root = Path(temporary)
|
||||||
|
dcgm = root / "dcgm.log"
|
||||||
|
dcgm.write_text(
|
||||||
|
"wall_time_ns,node,dcgm_output\n"
|
||||||
|
"1000000000,head,GPU 0 0.900 0.700 0.300 0.200 0.500 1000000000 2000000000\n",
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
mpstat = root / "mpstat.log"
|
||||||
|
mpstat.write_text(
|
||||||
|
"1000000000\thead\t12:00:00 PM all 10.00 0.00 5.00 1.00 "
|
||||||
|
"0.00 2.00 0.00 0.00 0.00 82.00\n",
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
|
||||||
|
dcgm_rows = attribution.parse_dcgm(dcgm)
|
||||||
|
cpu_rows = attribution.parse_mpstat(mpstat)
|
||||||
|
|
||||||
|
self.assertEqual(dcgm_rows[0]["sm_active"], 0.7)
|
||||||
|
self.assertEqual(dcgm_rows[0]["pcie_rx_bytes_per_s"], 2_000_000_000)
|
||||||
|
self.assertEqual(cpu_rows[0]["cpu_active_pct"], 18)
|
||||||
|
self.assertEqual(cpu_rows[0]["iowait_pct"], 1)
|
||||||
|
|
||||||
|
def test_communication_aggregate_separates_local_and_cross_numa(self) -> None:
|
||||||
|
rows = [
|
||||||
|
{
|
||||||
|
"test": "p2p_copy",
|
||||||
|
"node": "head",
|
||||||
|
"source_gpu": 0,
|
||||||
|
"destination_gpu": 1,
|
||||||
|
"size_bytes": 1024,
|
||||||
|
"bandwidth_GBps": 20,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"test": "p2p_copy",
|
||||||
|
"node": "head",
|
||||||
|
"source_gpu": 0,
|
||||||
|
"destination_gpu": 4,
|
||||||
|
"size_bytes": 1024,
|
||||||
|
"bandwidth_GBps": 10,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"test": "all_reduce",
|
||||||
|
"scope": "two_node_16gpu",
|
||||||
|
"nccl_cross_nic": "1",
|
||||||
|
"size_bytes": 1024,
|
||||||
|
"mean_ms": 1,
|
||||||
|
"algbw_GBps": 1,
|
||||||
|
"busbw_GBps": 1.875,
|
||||||
|
"wrong_values": 0,
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
aggregate = attribution.aggregate_communication_rows(rows)
|
||||||
|
|
||||||
|
self.assertEqual(len(aggregate), 3)
|
||||||
|
path_classes = {
|
||||||
|
row.get("path_class")
|
||||||
|
for row in aggregate
|
||||||
|
if row["test"] == "p2p_copy"
|
||||||
|
}
|
||||||
|
self.assertEqual(
|
||||||
|
path_classes,
|
||||||
|
{"same_pcie_switch", "cross_numa_sys"},
|
||||||
|
)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _write_csv(
|
def _write_csv(
|
||||||
path: Path,
|
path: Path,
|
||||||
|
|||||||
@ -9,8 +9,9 @@ import json
|
|||||||
import math
|
import math
|
||||||
import re
|
import re
|
||||||
import statistics
|
import statistics
|
||||||
|
import time
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from datetime import datetime
|
from datetime import datetime, timedelta
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
@ -33,6 +34,10 @@ SUMMARY_FIELDS = [
|
|||||||
"started_at",
|
"started_at",
|
||||||
"ended_at",
|
"ended_at",
|
||||||
"elapsed_s",
|
"elapsed_s",
|
||||||
|
"measurement_started_at",
|
||||||
|
"measurement_ended_at",
|
||||||
|
"measurement_duration_s",
|
||||||
|
"measurement_window_source",
|
||||||
"completed",
|
"completed",
|
||||||
"failed",
|
"failed",
|
||||||
"duration_s",
|
"duration_s",
|
||||||
@ -305,6 +310,7 @@ def parse_scenarios(path: Path) -> list[dict[str, Any]]:
|
|||||||
|
|
||||||
|
|
||||||
def write_case(args: argparse.Namespace) -> None:
|
def write_case(args: argparse.Namespace) -> None:
|
||||||
|
measurement = measurement_window(args.measurement_marker, args.bench_file)
|
||||||
value = {
|
value = {
|
||||||
"run_id": args.run_id,
|
"run_id": args.run_id,
|
||||||
"suite": args.suite,
|
"suite": args.suite,
|
||||||
@ -323,6 +329,7 @@ def write_case(args: argparse.Namespace) -> None:
|
|||||||
"started_at": args.started_at,
|
"started_at": args.started_at,
|
||||||
"ended_at": args.ended_at,
|
"ended_at": args.ended_at,
|
||||||
"elapsed_s": args.elapsed_s,
|
"elapsed_s": args.elapsed_s,
|
||||||
|
**measurement,
|
||||||
"bench_file": args.bench_file,
|
"bench_file": args.bench_file,
|
||||||
"bench_log": args.bench_log,
|
"bench_log": args.bench_log,
|
||||||
"note": args.note,
|
"note": args.note,
|
||||||
@ -330,6 +337,51 @@ def write_case(args: argparse.Namespace) -> None:
|
|||||||
write_json(args.path, value)
|
write_json(args.path, value)
|
||||||
|
|
||||||
|
|
||||||
|
def write_measurement_start(path: Path) -> None:
|
||||||
|
write_json(
|
||||||
|
path,
|
||||||
|
{
|
||||||
|
"recorded_at": datetime.now()
|
||||||
|
.astimezone()
|
||||||
|
.isoformat(timespec="microseconds"),
|
||||||
|
"wall_time_ns": time.time_ns(),
|
||||||
|
"source": "bench_log_main_marker",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def measurement_window(marker_path: str, bench_file: str) -> dict[str, Any]:
|
||||||
|
empty = {
|
||||||
|
"measurement_started_at": None,
|
||||||
|
"measurement_ended_at": None,
|
||||||
|
"measurement_duration_s": None,
|
||||||
|
"measurement_window_source": "unavailable",
|
||||||
|
}
|
||||||
|
marker = Path(marker_path) if marker_path else None
|
||||||
|
bench = Path(bench_file) if bench_file else None
|
||||||
|
if marker is None or bench is None or not marker.exists():
|
||||||
|
return empty
|
||||||
|
try:
|
||||||
|
marker_value = read_json(marker)
|
||||||
|
started_at = str(marker_value["recorded_at"])
|
||||||
|
started = datetime.fromisoformat(started_at)
|
||||||
|
except (OSError, KeyError, TypeError, ValueError, json.JSONDecodeError):
|
||||||
|
return empty
|
||||||
|
bench_value = read_bench_output(bench)
|
||||||
|
if bench_value is None:
|
||||||
|
return empty
|
||||||
|
duration_s = first_float(bench_value, "duration", "benchmark_duration")
|
||||||
|
if duration_s is None or duration_s <= 0:
|
||||||
|
return empty
|
||||||
|
ended = started + timedelta(seconds=duration_s)
|
||||||
|
return {
|
||||||
|
"measurement_started_at": started_at,
|
||||||
|
"measurement_ended_at": ended.isoformat(timespec="microseconds"),
|
||||||
|
"measurement_duration_s": duration_s,
|
||||||
|
"measurement_window_source": "bench_main_marker_plus_duration",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def write_manifest(args: argparse.Namespace) -> None:
|
def write_manifest(args: argparse.Namespace) -> None:
|
||||||
value: dict[str, Any] = {}
|
value: dict[str, Any] = {}
|
||||||
if args.path.exists():
|
if args.path.exists():
|
||||||
@ -618,6 +670,7 @@ def add_case_arguments(parser: argparse.ArgumentParser) -> None:
|
|||||||
parser.add_argument("--started-at", required=True)
|
parser.add_argument("--started-at", required=True)
|
||||||
parser.add_argument("--ended-at", required=True)
|
parser.add_argument("--ended-at", required=True)
|
||||||
parser.add_argument("--elapsed-s", type=float, required=True)
|
parser.add_argument("--elapsed-s", type=float, required=True)
|
||||||
|
parser.add_argument("--measurement-marker", default="")
|
||||||
parser.add_argument("--bench-file", required=True)
|
parser.add_argument("--bench-file", required=True)
|
||||||
parser.add_argument("--bench-log", required=True)
|
parser.add_argument("--bench-log", required=True)
|
||||||
parser.add_argument("--note", default="")
|
parser.add_argument("--note", default="")
|
||||||
@ -678,6 +731,9 @@ def main() -> None:
|
|||||||
mark_failed_parser.add_argument("--error-type", required=True)
|
mark_failed_parser.add_argument("--error-type", required=True)
|
||||||
mark_failed_parser.add_argument("--note", required=True)
|
mark_failed_parser.add_argument("--note", required=True)
|
||||||
|
|
||||||
|
measurement_start_parser = subparsers.add_parser("mark-measurement-start")
|
||||||
|
measurement_start_parser.add_argument("--path", type=Path, required=True)
|
||||||
|
|
||||||
summarize_parser = subparsers.add_parser("summarize")
|
summarize_parser = subparsers.add_parser("summarize")
|
||||||
summarize_parser.add_argument("result_dir", type=Path)
|
summarize_parser.add_argument("result_dir", type=Path)
|
||||||
|
|
||||||
@ -708,6 +764,8 @@ def main() -> None:
|
|||||||
complete_manifest(args.path, args.status)
|
complete_manifest(args.path, args.status)
|
||||||
elif args.command == "mark-case-failed":
|
elif args.command == "mark-case-failed":
|
||||||
mark_case_failed(args.path, args.error_type, args.note)
|
mark_case_failed(args.path, args.error_type, args.note)
|
||||||
|
elif args.command == "mark-measurement-start":
|
||||||
|
write_measurement_start(args.path)
|
||||||
elif args.command == "summarize":
|
elif args.command == "summarize":
|
||||||
summarize(args.result_dir)
|
summarize(args.result_dir)
|
||||||
|
|
||||||
|
|||||||
@ -483,6 +483,7 @@ write_case_meta() {
|
|||||||
local bench_file="${18}"
|
local bench_file="${18}"
|
||||||
local bench_log="${19}"
|
local bench_log="${19}"
|
||||||
local note="${20}"
|
local note="${20}"
|
||||||
|
local measurement_marker="${21}"
|
||||||
|
|
||||||
python3 "${RESULT_TOOL}" write-case \
|
python3 "${RESULT_TOOL}" write-case \
|
||||||
--path "${meta_path}" \
|
--path "${meta_path}" \
|
||||||
@ -503,6 +504,7 @@ write_case_meta() {
|
|||||||
--started-at "${started_at}" \
|
--started-at "${started_at}" \
|
||||||
--ended-at "${ended_at}" \
|
--ended-at "${ended_at}" \
|
||||||
--elapsed-s "${elapsed_s}" \
|
--elapsed-s "${elapsed_s}" \
|
||||||
|
--measurement-marker "${measurement_marker}" \
|
||||||
--bench-file "${bench_file}" \
|
--bench-file "${bench_file}" \
|
||||||
--bench-log "${bench_log}" \
|
--bench-log "${bench_log}" \
|
||||||
--note "${note}"
|
--note "${note}"
|
||||||
@ -542,6 +544,25 @@ detect_error_type() {
|
|||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
|
watch_bench_main_start() {
|
||||||
|
local bench_log="$1"
|
||||||
|
local bench_pid="$2"
|
||||||
|
local marker_path="$3"
|
||||||
|
|
||||||
|
while kill -0 "${bench_pid}" 2>/dev/null; do
|
||||||
|
if grep -Fq "Starting main benchmark run" "${bench_log}" 2>/dev/null; then
|
||||||
|
python3 "${RESULT_TOOL}" mark-measurement-start --path "${marker_path}"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
sleep 0.1
|
||||||
|
done
|
||||||
|
if grep -Fq "Starting main benchmark run" "${bench_log}" 2>/dev/null; then
|
||||||
|
python3 "${RESULT_TOOL}" mark-measurement-start --path "${marker_path}"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
run_bench_case() {
|
run_bench_case() {
|
||||||
local suite="$1"
|
local suite="$1"
|
||||||
local case_id="$2"
|
local case_id="$2"
|
||||||
@ -563,6 +584,7 @@ run_bench_case() {
|
|||||||
local bench_log="${case_dir}/bench.log"
|
local bench_log="${case_dir}/bench.log"
|
||||||
local meta_path="${case_dir}/meta.json"
|
local meta_path="${case_dir}/meta.json"
|
||||||
local command_file="${case_dir}/bench_cmd.txt"
|
local command_file="${case_dir}/bench_cmd.txt"
|
||||||
|
local measurement_marker="${case_dir}/measurement_start.json"
|
||||||
|
|
||||||
if case_already_completed "${meta_path}" "${bench_file}"; then
|
if case_already_completed "${meta_path}" "${bench_file}"; then
|
||||||
log "SKIP completed case=${case_id} rep=${repetition}"
|
log "SKIP completed case=${case_id} rep=${repetition}"
|
||||||
@ -591,14 +613,22 @@ run_bench_case() {
|
|||||||
"${BENCH_CMD[@]}" > "${command_file}"
|
"${BENCH_CMD[@]}" > "${command_file}"
|
||||||
|
|
||||||
local started_at start_epoch ended_at elapsed_s rc status error_type
|
local started_at start_epoch ended_at elapsed_s rc status error_type
|
||||||
|
local bench_pid marker_pid
|
||||||
started_at="$(iso_now)"
|
started_at="$(iso_now)"
|
||||||
start_epoch="$(date +%s)"
|
start_epoch="$(date +%s)"
|
||||||
log "START case=${case_id} rep=${repetition} isl=${isl} osl=${osl} c=${concurrency}"
|
log "START case=${case_id} rep=${repetition} isl=${isl} osl=${osl} c=${concurrency}"
|
||||||
|
rm -f "${measurement_marker}"
|
||||||
|
|
||||||
set +e
|
set +e
|
||||||
timeout --signal=TERM --kill-after=30s "${timeout_s}s" \
|
timeout --signal=TERM --kill-after=30s "${timeout_s}s" \
|
||||||
"${BENCH_CMD[@]}" > "${bench_log}" 2>&1
|
"${BENCH_CMD[@]}" > "${bench_log}" 2>&1 &
|
||||||
|
bench_pid=$!
|
||||||
|
watch_bench_main_start \
|
||||||
|
"${bench_log}" "${bench_pid}" "${measurement_marker}" &
|
||||||
|
marker_pid=$!
|
||||||
|
wait "${bench_pid}"
|
||||||
rc=$?
|
rc=$?
|
||||||
|
wait "${marker_pid}" >/dev/null 2>&1 || true
|
||||||
set -e
|
set -e
|
||||||
|
|
||||||
ended_at="$(iso_now)"
|
ended_at="$(iso_now)"
|
||||||
@ -618,7 +648,8 @@ run_bench_case() {
|
|||||||
"${meta_path}" "${suite}" "${case_id}" "${role}" "${stage}" \
|
"${meta_path}" "${suite}" "${case_id}" "${role}" "${stage}" \
|
||||||
"${repetition}" "${isl}" "${osl}" "${concurrency}" "${num_prompts}" \
|
"${repetition}" "${isl}" "${osl}" "${concurrency}" "${num_prompts}" \
|
||||||
"${warmup_requests}" "${status}" "${error_type}" "${rc}" "${started_at}" \
|
"${warmup_requests}" "${status}" "${error_type}" "${rc}" "${started_at}" \
|
||||||
"${ended_at}" "${elapsed_s}" "${bench_file}" "${bench_log}" "${note}"
|
"${ended_at}" "${elapsed_s}" "${bench_file}" "${bench_log}" "${note}" \
|
||||||
|
"${measurement_marker}"
|
||||||
|
|
||||||
LAST_CASE_STATUS="${status}"
|
LAST_CASE_STATUS="${status}"
|
||||||
LAST_CASE_ERROR="${error_type}"
|
LAST_CASE_ERROR="${error_type}"
|
||||||
|
|||||||
@ -4,6 +4,7 @@ import json
|
|||||||
import sys
|
import sys
|
||||||
import tempfile
|
import tempfile
|
||||||
import unittest
|
import unittest
|
||||||
|
from datetime import datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
@ -111,6 +112,35 @@ class QuickMapResultsTest(unittest.TestCase):
|
|||||||
self.assertIn("long_prefill_latency_128k_c1", report_text)
|
self.assertIn("long_prefill_latency_128k_c1", report_text)
|
||||||
self.assertIn("OOM", report_text)
|
self.assertIn("OOM", report_text)
|
||||||
|
|
||||||
|
def test_measurement_window_uses_main_marker_and_benchmark_duration(self) -> None:
|
||||||
|
with tempfile.TemporaryDirectory() as temporary:
|
||||||
|
root = Path(temporary)
|
||||||
|
marker = root / "measurement_start.json"
|
||||||
|
bench = root / "bench.jsonl"
|
||||||
|
marker.write_text(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"recorded_at": "2026-07-31T12:00:00.250000+08:00",
|
||||||
|
"wall_time_ns": 1,
|
||||||
|
}
|
||||||
|
),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
bench.write_text(json.dumps({"duration": 12.5}) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
window = quick_map_results.measurement_window(str(marker), str(bench))
|
||||||
|
|
||||||
|
self.assertEqual(window["measurement_duration_s"], 12.5)
|
||||||
|
self.assertEqual(
|
||||||
|
window["measurement_window_source"],
|
||||||
|
"bench_main_marker_plus_duration",
|
||||||
|
)
|
||||||
|
ended = datetime.fromisoformat(str(window["measurement_ended_at"]))
|
||||||
|
self.assertEqual(
|
||||||
|
ended,
|
||||||
|
datetime.fromisoformat("2026-07-31T12:00:12.750000+08:00"),
|
||||||
|
)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _write_meta(
|
def _write_meta(
|
||||||
path: Path,
|
path: Path,
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user