fix(deploy): DP_FLAG 未达 BOOTSTRAP 启动命令 + bench 的 model/tokenizer 分离

- profile.py: DP_FLAG 追加到 LAUNCH_ARGS 与 BOOTSTRAP(BOOTSTRAP 的 ${LAUNCH_ARGS}
  引用在 env 解析阶段已内联,必须直接 append 到 BOOTSTRAP 末尾),并改为用
  rendered 作模板展开源;修复 910c TP4/DP4 启动退化为 TP4 单 DP 布局导致
  专家权重不分片 OOM(61.3GB/die) 的问题(顺带修复 p800/pro6000 同类隐患)
- cli.py/runner.py: bench 的 API model 名改用 SERVED_MODEL_NAME(vLLM 严格校验),
  tokenizer 独立用 MODEL_PATH 路径并 --tokenizer 透传、docker client 挂载;
  修复 910c/vLLM 场景 404 Not Found
- dsv4 910c profile 恢复标准加载参数(prefetch+multithread);config.env per-TP
  max-model-len 默认改为已验证值(32768/65536/131072)
- 实测: 910c TP4/DP4 16-worker 布局启动健康, sskj.bench smoke 40/40
  (TTFT 1667ms, TPOT 35.4ms)
This commit is contained in:
shishi 2026-08-03 16:41:29 +08:00
parent ea8302561e
commit f33c5f1d3d
4 changed files with 34 additions and 12 deletions

View File

@ -90,18 +90,20 @@ API_SERVER_COUNT="${API_SERVER_COUNT:-1}"
# Per-TP parameter overrides (applied in start_vllm_docker.sh via case $TP). # Per-TP parameter overrides (applied in start_vllm_docker.sh via case $TP).
# DSV4-Flash w8a8 weight per die = ~280GiB / TP. Each die has 64GB HBM. # DSV4-Flash w8a8 weight per die = ~280GiB / TP. Each die has 64GB HBM.
# Defaults below are the verified-OK values (from the 2026-07-29..30 adaptive
# runs' server_cmd.txt); 1048576-context OOMs on TP4/TP8 (KV cache budget).
# TP=4: ~39 GiB/die weights -> 25 GiB for KV cache; cap context to avoid OOM # TP=4: ~39 GiB/die weights -> 25 GiB for KV cache; cap context to avoid OOM
# TP=8: ~35 GiB/die weights -> 29 GiB for KV cache; balanced # TP=8: ~35 GiB/die weights -> 29 GiB for KV cache; balanced
# TP=16: ~17.5 GiB/die weights -> 46 GiB for KV cache; full context # TP=16: ~17.5 GiB/die weights -> 46 GiB for KV cache; full context
TP4_GPU_MEMORY_UTILIZATION="${TP4_GPU_MEMORY_UTILIZATION:-0.9}" TP4_GPU_MEMORY_UTILIZATION="${TP4_GPU_MEMORY_UTILIZATION:-0.9}"
TP4_MAX_MODEL_LEN="${TP4_MAX_MODEL_LEN:-1048576}" TP4_MAX_MODEL_LEN="${TP4_MAX_MODEL_LEN:-32768}"
TP4_MAX_NUM_SEQS="${TP4_MAX_NUM_SEQS:-64}" TP4_MAX_NUM_SEQS="${TP4_MAX_NUM_SEQS:-128}"
TP8_GPU_MEMORY_UTILIZATION="${TP8_GPU_MEMORY_UTILIZATION:-0.9}" TP8_GPU_MEMORY_UTILIZATION="${TP8_GPU_MEMORY_UTILIZATION:-0.9}"
TP8_MAX_MODEL_LEN="${TP8_MAX_MODEL_LEN:-1048576}" TP8_MAX_MODEL_LEN="${TP8_MAX_MODEL_LEN:-65536}"
TP8_MAX_NUM_SEQS="${TP8_MAX_NUM_SEQS:-64}" TP8_MAX_NUM_SEQS="${TP8_MAX_NUM_SEQS:-256}"
TP16_GPU_MEMORY_UTILIZATION="${TP16_GPU_MEMORY_UTILIZATION:-0.9}" TP16_GPU_MEMORY_UTILIZATION="${TP16_GPU_MEMORY_UTILIZATION:-0.9}"
TP16_MAX_MODEL_LEN="${TP16_MAX_MODEL_LEN:-1048576}" TP16_MAX_MODEL_LEN="${TP16_MAX_MODEL_LEN:-131072}"
TP16_MAX_NUM_SEQS="${TP16_MAX_NUM_SEQS:-64}" TP16_MAX_NUM_SEQS="${TP16_MAX_NUM_SEQS:-256}"
# vLLM-Ascend-specific launch flags injected by start_vllm_docker.sh. # vLLM-Ascend-specific launch flags injected by start_vllm_docker.sh.
# attention backend for 910C: use the fused/atb attention path. Adjust per image. # attention backend for 910C: use the fused/atb attention path. Adjust per image.

View File

@ -87,13 +87,16 @@ def _cmd_run(args: argparse.Namespace, root: Path) -> int:
url = args.url or f"http://127.0.0.1:{default_port}" url = args.url or f"http://127.0.0.1:{default_port}"
base_url, host, port = _normalize_url(url) base_url, host, port = _normalize_url(url)
# API model name: SERVED_MODEL_NAME first (vLLM validates it strictly);
# tokenizer for prompt length counting stays on the local MODEL_PATH.
model = ( model = (
args.model args.model
or experiment_env.get("MODEL_PATH")
or experiment_env.get("SERVED_MODEL_NAME") or experiment_env.get("SERVED_MODEL_NAME")
or experiment_env.get("MODEL_PATH")
or experiment_env.get("MODEL_NAME") or experiment_env.get("MODEL_NAME")
or "model" or "model"
) )
tokenizer = args.model or experiment_env.get("MODEL_PATH")
dataset = args.dataset or experiment_env.get("DATASET_PATH") dataset = args.dataset or experiment_env.get("DATASET_PATH")
client_mode = bench_config.resolve_client_mode(args.client, args.platform, experiment_env) client_mode = bench_config.resolve_client_mode(args.client, args.platform, experiment_env)
@ -189,6 +192,7 @@ def _cmd_run(args: argparse.Namespace, root: Path) -> int:
host=host, host=host,
port=port, port=port,
model=model, model=model,
tokenizer=tokenizer,
dataset_name=dataset_name, dataset_name=dataset_name,
dataset_path=dataset, dataset_path=dataset,
container_dataset_path=experiment_env.get("CONTAINER_DATASET_PATH"), container_dataset_path=experiment_env.get("CONTAINER_DATASET_PATH"),

View File

@ -18,6 +18,7 @@ class BenchClientOptions:
host: str host: str
port: int port: int
model: str model: str
tokenizer: str | None = None
dataset_name: str = "random" dataset_name: str = "random"
dataset_path: str | None = None dataset_path: str | None = None
container_dataset_path: str | None = None container_dataset_path: str | None = None
@ -45,6 +46,10 @@ def _bench_args(scenario: dict[str, Any], options: BenchClientOptions) -> list[s
str(options.port), str(options.port),
"--model", "--model",
options.model, options.model,
]
if options.tokenizer:
args += ["--tokenizer", options.tokenizer]
args += [
"--dataset-name", "--dataset-name",
options.dataset_name, options.dataset_name,
"--random-input-len", "--random-input-len",
@ -107,8 +112,9 @@ def run_scenario(scenario: dict[str, Any], options: BenchClientOptions) -> int:
cmd = ["docker", "run", "--rm", "--network", "host"] cmd = ["docker", "run", "--rm", "--network", "host"]
if options.root: if options.root:
cmd += ["-v", f"{options.root}:{options.root}"] cmd += ["-v", f"{options.root}:{options.root}"]
if Path(options.model).exists(): for mount in (options.model, options.tokenizer):
cmd += ["-v", f"{options.model}:{options.model}:ro"] if mount and Path(mount).exists():
cmd += ["-v", f"{mount}:{mount}:ro"]
if options.dataset_path and Path(options.dataset_path).exists(): if options.dataset_path and Path(options.dataset_path).exists():
cmd += ["-v", f"{options.dataset_path}:{options.dataset_path}:ro"] cmd += ["-v", f"{options.dataset_path}:{options.dataset_path}:ro"]
if options.output_file: if options.output_file:

View File

@ -57,13 +57,23 @@ def render_profile(profile: dict[str, str], overrides: dict[str, str] | None = N
"PORT", "PORT",
] ]
rendered = dict(values) rendered = dict(values)
for key in template_keys:
if key in rendered:
rendered[key] = expand_template(str(rendered[key]), values)
# Append DP_FLAG to LAUNCH_ARGS and BOOTSTRAP BEFORE template expansion.
# BOOTSTRAP's ${LAUNCH_ARGS} reference is already inlined during env-file
# parsing (without DP_FLAG), so appending to BOOTSTRAP directly is what
# guarantees the data-parallel flag reaches the actual launch command.
dp = int(values["DP"]) dp = int(values["DP"])
if dp > 1 and rendered.get("DP_FLAG"): if dp > 1 and rendered.get("DP_FLAG"):
rendered["LAUNCH_ARGS"] = f"{rendered.get('LAUNCH_ARGS', '')} {rendered['DP_FLAG']}".strip() rendered["LAUNCH_ARGS"] = f"{rendered.get('LAUNCH_ARGS', '')} {rendered['DP_FLAG']}".strip()
if "BOOTSTRAP" in rendered:
rendered["BOOTSTRAP"] = f"{rendered['BOOTSTRAP']} {rendered['DP_FLAG']}".strip()
# Expand against `rendered` (not `values`) so later references such as
# BOOTSTRAP's ${LAUNCH_ARGS} pick up earlier mutations (e.g. DP_FLAG).
for key in template_keys:
if key in rendered:
rendered[key] = expand_template(str(rendered[key]), rendered)
if not rendered.get("RUNTIME"): if not rendered.get("RUNTIME"):
rendered["RUNTIME"] = "docker" if rendered.get("DOCKER_IMAGE") else "native" rendered["RUNTIME"] = "docker" if rendered.get("DOCKER_IMAGE") else "native"
return rendered return rendered