From df924159838c28b9481ae0c5c0de187ccbc407cc Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 01:21:50 +0000 Subject: [PATCH 01/31] fix: repair Kimi B300 Mooncake RDMA startup and recovery MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Discover active Mellanox RDMA rails by mlx5_core driver on DSXE, load the host mlx5 provider, backport Mooncake load-failure recovery, and raise the master lease to 60s so transferred keys survive the observed get tail. 修复 Kimi B300 上 Mooncake 的 RDMA 启动与加载恢复:按驱动选择活动网卡、 加载主机 mlx5 provider、回移植加载失败恢复,并将 master lease 提高到 60 秒。 Co-authored-by: Wenyao Gao --- inferencex-e2e/benchmarks/benchmark_lib.sh | 22 ++ .../configs/kimik3-b300-mooncake.sh | 43 ++-- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 9 +- inferencex-e2e/configs/runners.yaml | 5 + .../docs/configuration-procedures.md | 17 +- .../docs/configuration-procedures_zh.md | 15 +- inferencex-e2e/docs/waiver/3088.md | 78 ++++++ inferencex-e2e/docs/waiver/3088_zh.md | 58 +++++ inferencex-e2e/perf-changelog.yaml | 8 + inferencex-e2e/runners/__init__.py | 0 .../runners/patch_kimik3_mooncake_recovery.py | 228 ++++++++++++++++++ inferencex-e2e/runners/test_kimik3_b300.py | 52 ++++ .../runners/test_mooncake_rdma_device.py | 62 +++++ 13 files changed, 562 insertions(+), 35 deletions(-) create mode 100644 inferencex-e2e/docs/waiver/3088.md create mode 100644 inferencex-e2e/docs/waiver/3088_zh.md create mode 100644 inferencex-e2e/runners/__init__.py create mode 100644 inferencex-e2e/runners/patch_kimik3_mooncake_recovery.py create mode 100644 inferencex-e2e/runners/test_kimik3_b300.py create mode 100644 inferencex-e2e/runners/test_mooncake_rdma_device.py diff --git a/inferencex-e2e/benchmarks/benchmark_lib.sh b/inferencex-e2e/benchmarks/benchmark_lib.sh index 39a7e22df3..7a418cb37b 100644 --- a/inferencex-e2e/benchmarks/benchmark_lib.sh +++ b/inferencex-e2e/benchmarks/benchmark_lib.sh @@ -178,6 +178,28 @@ PYPORT export PORT } +select_mooncake_rdma_device() { + local sysfs_root="${1:-/sys/class/infiniband}" + local device + MOONCAKE_RAIL="" + for device in "$sysfs_root"/*; do + # DSXE has both EFA and Mellanox adapters. The latter may be renamed + # ibp*, so identify the driver rather than assuming an mlx5_* name. + [[ "$(readlink "$device/device/driver" 2>/dev/null)" == */mlx5_core ]] || continue + grep -qx '4: ACTIVE' "$device/ports/1/state" 2>/dev/null || continue + case "$(cat "$device/ports/1/link_layer" 2>/dev/null)" in + InfiniBand) MC_GID_INDEX=0 ;; + Ethernet) MC_GID_INDEX=3 ;; + *) continue ;; + esac + MOONCAKE_RAIL="${device##*/}" + export MC_GID_INDEX + return 0 + done + echo "Error: no active Mellanox RDMA rail; Mooncake cannot initialise" >&2 + return 1 +} + agentic_kv_offload_enabled() { if [[ -z "${KV_OFFLOADING+x}" || -z "$KV_OFFLOADING" ]]; then echo "Error: KV_OFFLOADING must be set for agentic benchmarks" >&2 diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh index 243b3d2111..34de47004c 100755 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh @@ -1,6 +1,12 @@ #!/usr/bin/env bash -# Pin the worker's Mooncake client and point its store at one active RDMA rail. +# Pin the worker's Mooncake client, backport load-failure recovery, and point +# the store at one active Mellanox RDMA rail (driver-selected, including ibp*). set -euo pipefail + +ws=/infmax-workspace +# Temporary upstream #55297 backport; docs/waiver/3088.md is pending review. +python3 "$ws/runners/patch_kimik3_mooncake_recovery.py" + pip_install=(python3 -m pip install) if python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages; then pip_install+=(--break-system-packages) @@ -10,25 +16,23 @@ fi python3 -c "from mooncake.store import MooncakeDistributedStore" >/dev/null # Rail-isolated nodes: two RNICs cannot reach each other even within a node, so -# every rank uses one rail. mlx5_0 is down on some nodes, and topology discovery -# then finds no HCA, so take the first active rail at runtime. DSXE nodes name -# their rails rdmap*. -rail="" -for device in mlx5_0 mlx5_1 mlx5_2 mlx5_3 mlx5_4 mlx5_5 mlx5_8 mlx5_9 \ - mlx5_10 mlx5_11 mlx5_16 mlx5_17 mlx5_20 mlx5_21 mlx5_22 mlx5_23 \ - $(ls /sys/class/infiniband 2>/dev/null | grep '^rdmap' | sort -V); do - if grep -q ACTIVE "/sys/class/infiniband/$device/ports/1/state" 2>/dev/null; then - rail="$device" - break - fi -done -if [[ -z "$rail" ]]; then - echo "Error: no active RDMA rail on $(hostname); Mooncake cannot initialise" >&2 - for state in /sys/class/infiniband/*/ports/*/state; do - echo "$state: $(cat "$state" 2>&1)" >&2 - done - exit 1 +# every rank uses one Mellanox rail. Identify by driver (mlx5_core) rather than +# assuming mlx5_* names — DSXE may rename them ibp*. EFA rails are skipped. +# shellcheck source=/dev/null +source "$ws/benchmarks/benchmark_lib.sh" --validation-only +select_mooncake_rdma_device +rail="$MOONCAKE_RAIL" +echo "Mooncake rail: $rail (MC_GID_INDEX=$MC_GID_INDEX)" + +# The enroot EFA hook binds the host libibverbs over the image's, and +# libibverbs only loads providers built against its own private ABI, so the +# image's mlx5 provider never loads and the Mellanox rail vanishes. +# runners.yaml mounts the host library directory at /host-usr-lib; libibverbs +# appends its own -rdmavNN suffix to an absolute RDMAV_DRIVERS entry. +if [[ -d /host-usr-lib/libibverbs ]]; then + export RDMAV_DRIVERS=/host-usr-lib/libibverbs/libmlx5 fi + config="${MOONCAKE_CONFIG_PATH:-/logs/mooncake_store_config.json}" python3 - "$config" "$rail" <<'PY' import json, sys @@ -39,4 +43,3 @@ config["device_name"] = rail with open(path, "w") as handle: json.dump(config, handle, indent=2) PY -echo "Mooncake rail: $rail" diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index e8d5e7039b..72e5ca7b01 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -26,9 +26,10 @@ base: interval_seconds: 10 max_attempts: 360 # Embedded Mooncake: each TP rank contributes TOTAL_CPU_DRAM_GB / 8 GB. The - # setup script pins the client to the master's version and fills in the - # node's active RDMA rail. DSXE rails are InfiniBand without a netdev, so the - # transfer engine picks its own GID (a RoCE v2 index 3 does not exist). + # setup script pins the client, backports load-failure recovery, selects one + # active Mellanox rail by driver (ibp* or mlx5_*), sets MC_GID_INDEX from the + # link layer, and loads the host mlx5 provider via RDMAV_DRIVERS when + # /host-usr-lib is mounted. Master lease is 60s (see mooncake-master args). setup_script: kimik3-b300-mooncake.sh services: - name: mooncake-master @@ -36,7 +37,7 @@ base: preamble: >- python3 -m pip install --break-system-packages --quiet --no-cache-dir --no-deps --force-reinstall mooncake-transfer-engine-cuda13==0.3.11.post1 - args: ["--eviction_high_watermark_ratio=0.95", "--eviction_ratio=0.10"] + args: ["--eviction_high_watermark_ratio=0.95", "--eviction_ratio=0.10", "--default_kv_lease_ttl=60000"] options: store_config: mode: embedded diff --git a/inferencex-e2e/configs/runners.yaml b/inferencex-e2e/configs/runners.yaml index 8036d3de69..c63ccdac8a 100644 --- a/inferencex-e2e/configs/runners.yaml +++ b/inferencex-e2e/configs/runners.yaml @@ -599,6 +599,11 @@ clusters: single-node-models: staged container-aliases: [dynamo-trtllm, dynamo-sglang, dynamo-vllm] nginx-aliases: [nginx-sqsh] + # Host libibverbs ABI for Mooncake RDMA: enroot EFA hook overlays the + # image provider, so mount the host library tree and load mlx5 via + # RDMAV_DRIVERS in kimik3-b300-mooncake.sh. + mounts: + /usr/lib/x86_64-linux-gnu: /host-usr-lib gb200-nv: gpus-per-node: 4 available-cpu-dram-mib: 860_160 diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index d2df756784..5afa7ee55c 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -275,7 +275,7 @@ All ten AgentX throughput points use DSpark K6 (target verification length 7) and the committed golden AL 3.77. C1/2/4/8/16 use TP8/EP1; C48/64/96/128/256 use TP8/DPA8/EP8 with native RCCL. Each point runs for 3600 seconds. The C256 full GSM8K eval omits forced acceptance. Keep the -pinned `rocm/atom-dev:nightly_202609291501` image and GPU-only KV. C1 through C16 use +pinned `rocm/atom-dev:nightly_202609161445` image and GPU-only KV. C1 through C16 use BF16 KV, while C48 and above retain FP8 KV; all points use the FP4 index cache, 8192-token checkpoints and DEP dense FULL graph ladder. Fixed q7 graphs are captured in each new server; confirm target and DSpark draft capture in @@ -288,12 +288,8 @@ record model/source identity and requested settings. Successful startup, graph capture and requests require runtime log evidence. The pinned image is the official ATOM nightly -`rocm/atom-dev:nightly_202609291501` (ATOM `0.1.7.dev46+g74fd942b0`, ROCm 7.2.4), -which includes the merged +`rocm/atom-dev:nightly_202609161445`, which includes the merged [ROCm/ATOM#2233](https://github.com/ROCm/ATOM/pull/2233) inference-mode fix. -From this image ATOM stores the checkpoint's `ue8m0` FP8 block scales as E8M0 -on gfx950 by default ([ROCm/ATOM#2419](https://github.com/ROCm/ATOM/pull/2419)); -the powers-of-two scales are represented exactly. The recipe does not patch AITER source at runtime; TP communication fusion, DSpark K6 and graph capture use the implementation shipped in the image. @@ -341,6 +337,15 @@ The H200 DSpark recipe uses the same minimum capture size and preserves the same B300 uses the same minimum capture size at c1/c2/c4. Its c1 CI comparison reduced request ITL P90/P99 from 38.74/41.42 ms to 2.62/3.45 ms; c2/c4 require CI confirmation. +Kimi-K3 on B300 selects one active Mellanox adapter by its sysfs driver, including +DSXE `ibp*` names; EFA devices are excluded from this RDMA recipe. The embedded +Mooncake ranks share that adapter. InfiniBand uses GID index 0 and RoCE retains +index 3. If no compatible active adapter exists, startup fails before serving. +On DSXE the container's libibverbs comes from the host through the enroot EFA +hook, so `configs/runners.yaml` mounts the host library directory at `/host-usr-lib` +and the setup script loads its mlx5 provider through `RDMAV_DRIVERS`. + + The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU `image` pinned in [`nvidia-master.yaml`](../configs/nvidia-master.yaml) (originally `vllm/vllm-openai:deepseekv41-flash-0909`, which B300 still uses) at TP4 on Blackwell SKUs with native five-token DSpark, probabilistic drafting. Throughput uses the [committed golden AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection and adaptive verification. `--engram-config '{"cpu_offload":true}'` diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 0d023deb1c..5abaa89cf4 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -255,7 +255,7 @@ DSpark Markov/confidence head、全部 66 个分片的 header 与 payload 边界 全部十个 AgentX 性能点使用 DSpark K6(target 验证长度为 7)和已提交的 golden AL 3.77。C1/2/4/8/16 使用 TP8/EP1;C48/64/96/128/256 使用 TP8/DPA8/EP8 原生 RCCL。每个性能点运行 3600 秒。C256 全量 GSM8K 不传强制 -接受率参数。保留固定的 `rocm/atom-dev:nightly_202609291501` 镜像和 GPU KV;C1 至 C16 +接受率参数。保留固定的 `rocm/atom-dev:nightly_202609161445` 镜像和 GPU KV;C1 至 C16 使用 BF16 KV,C48 及以上继续使用 FP8 KV,所有任务均使用 FP4 index cache、 8192-token checkpoint 和 DEP dense FULL graph 阶梯。每个新服务进程重新捕获固定 q7 图;必须从 `server.log` 确认 target 和 DSpark draft capture 完成。confidence @@ -266,11 +266,8 @@ schedule 和 ragged verification 保持关闭。 `runtime_manifest.json` 和 `server_command.txt` 保存模型/源码身份及请求的配置。 成功启动、graph capture 和请求执行仍需运行时日志证明。 -固定镜像为官方 ATOM nightly `rocm/atom-dev:nightly_202609291501`(ATOM `0.1.7.dev46+g74fd942b0`, -ROCm 7.2.4),已包含已合入的 +固定镜像为官方 ATOM nightly `rocm/atom-dev:nightly_202609161445`,已包含已合入的 [ROCm/ATOM#2233](https://github.com/ROCm/ATOM/pull/2233) inference-mode 修复。 -自该镜像起,ATOM 在 gfx950 上默认以 E8M0 存储检查点的 `ue8m0` FP8 block scale -([ROCm/ATOM#2419](https://github.com/ROCm/ATOM/pull/2419)),2 的幂次 scale 可被精确表示。 配方不再在运行时修改 AITER 源码;TP 通信融合、DSpark K6 和 graph capture 直接使用镜像内实现。 @@ -315,6 +312,14 @@ H200 的 DSpark 配方使用相同的最小捕获范围,并保持相同的工 B300 在 c1/c2/c4 使用相同的最小捕获范围。其 c1 CI 对比中,请求 ITL P90/P99 从 38.74/41.42 ms 降至 2.62/3.45 ms;c2/c4 仍需 CI 验证。 +B300 上的 Kimi-K3 按 sysfs 驱动选择一块活动的 Mellanox 网卡,包括 DSXE 的 +`ibp*` 命名;本 RDMA 配方排除 EFA。嵌入式 Mooncake 各 rank 共用该网卡。 +InfiniBand 使用 GID 索引 0,RoCE 保留索引 3。若没有可用的活动适配器,则在 +服务启动前失败。DSXE 上容器的 libibverbs 来自 enroot EFA hook 挂载的主机库, +因此 `configs/runners.yaml` 将主机库目录挂到 `/host-usr-lib`,setup 脚本通过 +`RDMAV_DRIVERS` 加载其 mlx5 provider。 + + 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 [`nvidia-master.yaml`](../configs/nvidia-master.yaml) 中按 SKU 固定的 `image`(最初为 `vllm/vllm-openai:deepseekv41-flash-0909`,B300 仍在使用),在 Blackwell SKU 上采用 TP4、原生五 token DSpark、 概率采样草稿。吞吐测试使用[已提交的黄金 AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样和自适应验证。 diff --git a/inferencex-e2e/docs/waiver/3088.md b/inferencex-e2e/docs/waiver/3088.md new file mode 100644 index 0000000000..5f694f0e91 --- /dev/null +++ b/inferencex-e2e/docs/waiver/3088.md @@ -0,0 +1,78 @@ +# Inference-engine patch waiver — PR #3088 + +**English** | [中文](./3088_zh.md) + +**Status: proposed; not approved.** The maintainer's sign-off must explicitly link +this waiver under the engine-patch item of the [review checklist](../PR_REVIEW_CHECKLIST.md). + +## Scope and provenance + +- Config: `kimik3-fp4-b300-vllm-agentic-dspark` in `configs/nvidia-master.yaml`. +- Image: `vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77`, vLLM source + `3696c772aae308f2420a8f307b0971e6986c4818`. +- Mooncake remains `mooncake-transfer-engine-cuda13==0.3.11.post1`. +- Entrypoint: `benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh` + (the `setup_script` of `benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml`) + invokes `runners/patch_kimik3_mooncake_recovery.py` before pinning Mooncake and + selecting the RDMA rail. Other hardware recipes do not invoke it. +- Backport: [vLLM #55297](https://github.com/vllm-project/vllm/pull/55297), pinned + commit `f3a831c2015d9eb6f7e600dbd2ef565166d64437`; the upstream PR is still open. + +## Why the shipped image needs this change + +[PR run 34818543261](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34818543261) +failed its C8 benchmark after repeated Mooncake `LEASE_EXPIRED` reads. The server +logged 688 core KV-load recoveries; one logged rank/key failed 126 times. At +08:36:37 UTC the reported get duration averaged 20.551 seconds, with 587/587 keys +failed, while the master lease was 5 seconds. The benchmark failed its unchanged +95% metric-coverage gate; valid partial power does not qualify that performance point. + +In this image, async first-block failure can reset a request to token zero, then +repeat the same external lookup on the next scheduling attempt. Upstream #55297 +addresses exactly that unbounded recovery path. Its author reports an integration +negative control and successful local recomputation after the fix. That upstream +evidence is not current-head B300 qualification. Existing logs establish repeated +key failures but omit request IDs and per-rank completion counters, so they do not +prove this is the only cause of the C8 stall. The independent C8 eval Slurm launch +socket timeout is outside this waiver. + +## Exact patch and limits + +The backport adds the upstream `on_load_failure` callback to the connector base, +MooncakeStore connector/scheduler, MultiConnector, and core scheduler. Only async +failures reset to zero request a bypass. That request skips external lookups until +local allocation succeeds; partial-prefix and synchronous recovery keep their +existing behavior. Global cache metadata and other requests remain available. + +Only two insertion contexts were adapted for the older source layout; the upstream +recovery additions are unchanged. The helper verifies all five pristine or patched +source hashes before any writes. Unknown and mixed states fail the launch. + +This does not change the pinned image, Mooncake version, transport, external KV +tier, workload, speculative-decoding settings, 11-point concurrency curve, or 95% +gate. The recipe separately raises the master lease TTL from 5 s to 60 s +(`--default_kv_lease_ttl=60000`); the patch itself does not touch leases. Recovery +changes serving behavior and must be visible in results provenance; patched results +must not be represented as an unmodified upstream image. + +## Validation and removal + +Eight local patch-application and fail-closed checks pass. In the cached pinned +image, [runtime run 34832901970](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34832901970) +reproduced the original scheduler regression (two passes, one expected failure). +The candidate passed 51 core/error/Store/Multi tests; one Store test failed because +its block fixture lacked the pinned API. After correcting only that fixture, +[its single-case run](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34833354601) +passed, giving 52 distinct passing candidate cases across the two runs. Both runs +verified the same five patched source hashes; no production patch changed between them. + +These are runtime CPU regression checks, without model loading. The affected full +PR sweep/evals and maintainer waiver approval remain required before sign-off; +the original three successful performance points do not qualify the changed runtime. + +After #55297 or an equivalent reviewed fix lands and ships in an official vLLM +image, update this config to that pinned image, remove the helper and invocation, +and retire this waiver and its translation in the same PR. Requalify the complete +existing B300 curve and applicable evals; do not replace the curve with a diagnostic +subset. If upstream changes the proposed behavior before merge, review and update +this exact backport before adopting those changes. diff --git a/inferencex-e2e/docs/waiver/3088_zh.md b/inferencex-e2e/docs/waiver/3088_zh.md new file mode 100644 index 0000000000..b7cd317a61 --- /dev/null +++ b/inferencex-e2e/docs/waiver/3088_zh.md @@ -0,0 +1,58 @@ +# 推理引擎补丁豁免 — PR #3088 + +[English](./3088.md) | **中文** + +**状态:待审,尚未批准。** 维护者签核时必须在[评审清单](../PR_REVIEW_CHECKLIST.md) +的引擎补丁项明确链接这份豁免。 + +## 范围与来源 + +- 配置:`configs/nvidia-master.yaml` 中的 `kimik3-fp4-b300-vllm-agentic-dspark`。 +- 镜像保持 `vllm/vllm-openai:nightly-dev-x86_64-cu13-3696c77`,vLLM 源码为 + `3696c772aae308f2420a8f307b0971e6986c4818`。 +- Mooncake 保持 `mooncake-transfer-engine-cuda13==0.3.11.post1`。 +- 入口:`benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh` + (`b300-fp4-mtp/agentic.yaml` 的 `setup_script`)在启动服务前调用 + `runners/patch_kimik3_mooncake_recovery.py`;其他硬件配方不调用。 +- 回移 [vLLM #55297](https://github.com/vllm-project/vllm/pull/55297),固定提交为 + `f3a831c2015d9eb6f7e600dbd2ef565166d64437`;该上游 PR 目前尚未合并。 + +## 原镜像的问题 + +[PR run 34818543261](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34818543261) +的 C8 benchmark 在反复出现 Mooncake `LEASE_EXPIRED` 后失败。日志有688次核心调度恢复, +一个已记录的 rank/key 失败126次。08:36:37 UTC 的 get 平均耗时20.551秒,587/587个 key +失败,而 master 租约只有5秒。性能未通过保持不变的95%信号覆盖门槛;部分功耗有效不能使该性能点合格。 + +此镜像在异步首块加载失败后可能将请求退回 token 0,再次命中同一外部缓存,形成无界恢复循环。 +上游 #55297 专门修复该路径,作者报告了集成负对照及修复后成功本地重算。这不等于本PR当前头 +已经通过 B300 验证。旧日志缺 request ID 和各 rank 完成计数,尚不能证明循环是 C8 停滞的唯一原因。 +C8 eval 的 Slurm 启动 socket timeout 是独立问题,不在本豁免范围内。 + +## 补丁与限制 + +回移上游 `on_load_failure` 回调,涉及 connector 基类、MooncakeStore connector/scheduler、 +MultiConnector 和核心 scheduler。只有异步失败且退回0的请求触发绕过;该请求暂时跳过外部查询, +直到成功分配本地重算资源。部分前缀与同步恢复保持原行为,全局缓存元数据和其他请求仍可使用。 + +仅为旧版本调整了两处方法插入上下文,上游恢复逻辑新增内容不变。写入前检查全部5个文件的原始或 +完整补丁 SHA;未知版本、部分已修改状态均停止启动。 + +镜像、Mooncake 版本、传输、外部 KV 层、工作负载、推测解码配置、11点并发曲线和95%门槛 +均不改变。配方另将 master 租约 TTL 从 5 秒提高到 60 秒(`--default_kv_lease_ttl=60000`),补丁本身不涉及租约。恢复行为属于运行时改动,结果来源必须明确包含此补丁,不能描述为未修改的上游镜像。 + +## 验证与移除 + +8 项本地补丁应用及拒绝未知状态的检查通过。固定缓存镜像中的 +[运行 34832901970](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34832901970) +复现原版调度器回归(2 项通过、1 项预期失败)。候选的核心、错误传播、Store 和 Multi 共 51 项通过; +另一个 Store 用例因 blocks 测试夹具不符合旧版 API 失败。仅修正该夹具后, +[单点运行](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34833354601)通过, +两个运行合计覆盖 52 个不同的候选用例。两次均验证同一组 5 个补丁源码 SHA,生产补丁未改变。 + +这些是未加载模型的运行时 CPU 回归。请求签核前仍需完整 PR sweep/eval 和维护者豁免批准; +原先三个成功性能点不代表修改后运行时已获资格。 + +待 #55297 或同等已评审修复合入并进入官方 vLLM 镜像后,在同一PR中更新配置镜像、移除补丁工具和调用、 +退役本豁免及翻译,再验证完整现有 B300 曲线及适用 eval。诊断子集不能替换完整曲线。若上游合入前修改 +方案,应先复核并更新这里固定的回移内容。 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9b9c48a0ef..e22c7e8c9a 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9168,3 +9168,11 @@ - "Relevant ATOM changes in the range: V4 decode reuses the sparse prefill ASM (ROCm/ATOM#2271), greedy sampler picks via aiter.topk_select (ROCm/ATOM#2244), and FP8 block scales declared scale_fmt ue8m0 are now stored as E8M0 on gfx950 by default (ROCm/ATOM#2419; previously FP32 unless ATOM_FP8_BLOCKSCALE_USE_E8M0_SCALE=1). The upstream DeepSeek-V4 recipes are unchanged across the range, and every recipe flag and choice (all2all-backend rccl, dp-load-balance least_tokens, moe-backend standard) remains valid." - "No data-type or precision change to the DeepSeek-V4-Pro-0813 DSpark draft: no online quantization is configured, so it keeps its checkpoint precision; the E8M0 scale storage represents the checkpoint's power-of-two block scales exactly. kv-cache-dtype and index-cache-dtype touch cache storage only." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3605 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + description: + - "Discover active Mellanox RDMA devices by driver on DSXE, select the link-layer GID, and load the host mlx5 provider so Mooncake sees the rail" + - "Backport vLLM #55297 for request-local recompute after failed asynchronous Mooncake loads on B300; see docs/waiver/3088.md" + - "Raise the Mooncake master lease from the 5s default to 60s (--default_kv_lease_ttl=60000): at concurrency >= 32 gets took 10-30s at p90, so 60-75% of transferred keys were discarded as LEASE_EXPIRED and recomputed; 60s matches the client's own 60s transfer cap. vLLM's execute_model timeout stays at its 300s default" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 diff --git a/inferencex-e2e/runners/__init__.py b/inferencex-e2e/runners/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/inferencex-e2e/runners/patch_kimik3_mooncake_recovery.py b/inferencex-e2e/runners/patch_kimik3_mooncake_recovery.py new file mode 100644 index 0000000000..0e40484069 --- /dev/null +++ b/inferencex-e2e/runners/patch_kimik3_mooncake_recovery.py @@ -0,0 +1,228 @@ +#!/usr/bin/env python3 +"""Backport vLLM #55297 to the pinned Kimi-K3 B300 Mooncake image. + +Upstream: f3a831c2015d9eb6f7e600dbd2ef565166d64437 +Base: 3696c772aae308f2420a8f307b0971e6986c4818 +Only two hook insertion contexts differ from upstream; recovery logic is unchanged. +See docs/waiver/3088.md. Unknown or partially patched sources stop the launch. +""" + +from __future__ import annotations + +import hashlib +import importlib.util +import sys +from pathlib import Path + +# Relative package path, pristine SHA256, patched SHA256, exact edits. +PATCHES = ( + ('distributed/kv_transfer/kv_connector/v1/base.py', + 'bc1965431087676876f58360cd9cc07ab6c06febe6d747695f10b051fd85c412', + 'cbce0b160ca2d14c477103cf3b7f3433b2533eaa631439a693d1ffd06501342f', ( + (''' """ + return + + def update_connector_output(self, connector_output: KVConnectorOutput): + """ + Update KVConnector state from worker-side connectors output. +''', ''' """ + return + + def on_load_failure(self, request_ids: set[str]) -> None: + """Notify the connector before failed external KV loads are looked up again. + + Connectors may use this callback to make a failed external cache hit a + request-local miss on the next scheduling attempt. The default is a + no-op because not every connector needs special handling before the + affected tokens are recomputed. + """ + return + + def update_connector_output(self, connector_output: KVConnectorOutput): + """ + Update KVConnector state from worker-side connectors output. +'''), + )), + ('distributed/kv_transfer/kv_connector/v1/mooncake/store/connector.py', + 'd18e207bfce93ab53b2156902c8bdf7b223e1441c5070eddb07b4db2cb7b54b9', + '44ea4cda5ef3dbc8c7a32b824dc43280e264f4f63e956a5197d716c2fbaf66a3', ( + (''' def take_events(self) -> Iterable[KVCacheEvent]: +''', ''' def on_load_failure(self, request_ids: set[str]) -> None: + if self.connector_scheduler is not None: + self.connector_scheduler.on_load_failure(request_ids) + + def take_events(self) -> Iterable[KVCacheEvent]: +'''), + )), + ('distributed/kv_transfer/kv_connector/v1/mooncake/store/scheduler.py', + 'c75fc95a585ee4390963b01618c5ece1b52b30f8da95ff1a00a850948a943544', + '3fbd7807ea1b1d55558b604c208e6158d932e57778e3d8a30705edae22bdf7bb', ( + (''' + # Per-request state + self.load_specs: dict[str, LoadSpec] = {} # to be loaded + self._request_trackers: dict[str, RequestTracker] = {} # scheduled new requests + self._unfinished_requests: dict[str, tuple[Request, tuple[list[int], ...]]] = {} + self._unfinished_request_ids: set[str] = set() +''', ''' + # Per-request state + self.load_specs: dict[str, LoadSpec] = {} # to be loaded + # A failed load can rewind a request to token zero. Bypass lookups + # until local allocation succeeds so stale metadata cannot cause a livelock. + self._load_failure_bypass_req_ids: set[str] = set() + self._request_trackers: dict[str, RequestTracker] = {} # scheduled new requests + self._unfinished_requests: dict[str, tuple[Request, tuple[list[int], ...]]] = {} + self._unfinished_request_ids: set[str] = set() +'''), + (''' Returns ``(None, False)`` when an async lookup is still in flight, + signaling the scheduler to retry this request on a later step. + """ + if not self.enable_lookup: + return 0, False + +''', ''' Returns ``(None, False)`` when an async lookup is still in flight, + signaling the scheduler to retry this request on a later step. + """ + if request.request_id in self._load_failure_bypass_req_ids: + self.load_specs.pop(request.request_id, None) + logger.info( + "Skipping Mooncake lookup for request %s after KV load failure", + request.request_id, + ) + return 0, False + + if not self.enable_lookup: + return 0, False + +'''), + (''' + self._unfinished_requests[request.request_id] = (request, local_block_ids) + self._unfinished_request_ids.add(request.request_id) + + if request.request_id not in self.load_specs: + return +''', ''' + self._unfinished_requests[request.request_id] = (request, local_block_ids) + self._unfinished_request_ids.add(request.request_id) + self._load_failure_bypass_req_ids.discard(request.request_id) + + if request.request_id not in self.load_specs: + return +'''), + (''' + for finished_req_id in scheduler_output.finished_req_ids: + self.client.discard(finished_req_id) + self.load_specs.pop(finished_req_id, None) + self._request_trackers.pop(finished_req_id, None) + self._unfinished_requests.pop(finished_req_id, None) +''', ''' + for finished_req_id in scheduler_output.finished_req_ids: + self.client.discard(finished_req_id) + self._load_failure_bypass_req_ids.discard(finished_req_id) + self.load_specs.pop(finished_req_id, None) + self._request_trackers.pop(finished_req_id, None) + self._unfinished_requests.pop(finished_req_id, None) +'''), + (''' def update_connector_output(self, connector_output: KVConnectorOutput) -> None: +''', ''' def on_load_failure(self, request_ids: set[str]) -> None: + """Skip external lookups until requests are allocated for recompute.""" + self._load_failure_bypass_req_ids.update(request_ids) + + def update_connector_output(self, connector_output: KVConnectorOutput) -> None: +'''), + )), + ('distributed/kv_transfer/kv_connector/v1/multi_connector.py', + 'aafafbf4b0e3a43e7c864270fe56ffaa4b0f3dc343e77a0871db25f075d984dd', + '6f4fad0450ef91a75717b9631d668379d905d07fd876325da8d72eae17cca17e', ( + (''' for c in self._connectors: + c.on_new_request(request) + + def build_connector_meta( + self, scheduler_output: SchedulerOutput + ) -> MultiKVConnectorMetadata: +''', ''' for c in self._connectors: + c.on_new_request(request) + + def on_load_failure(self, request_ids: set[str]) -> None: + for c in self._connectors: + c.on_load_failure(request_ids) + + def build_connector_meta( + self, scheduler_output: SchedulerOutput + ) -> MultiKVConnectorMetadata: +'''), + )), + ('v1/core/sched/scheduler.py', + '6ba2a83c7fb6078e4d1c2af7a9a2bf820f83b0570f9e1e1908294a981bf1bace', + '9266f66969dbc0a6474bf53b4b3835c77a961ad4ffb9398ebcb368e1cb2eafc7', ( + (''' total_failed_tokens, + ) + + # Mark async requests with KV load failures for retry once loading completes + self.failed_recving_kv_req_ids |= async_failed_req_ids + # Return sync affected IDs to skip in update_from_output +''', ''' total_failed_tokens, + ) + + # Only async requests rewound to zero return to the waiting queue and + # run the connector lookup again. Notify the connector so a persistent + # external hit cannot make that request repeat the same failed load. + async_relookup_req_ids = { + req_id + for req_id in async_failed_req_ids + if self.requests[req_id].num_computed_tokens == 0 + } + if self.connector is not None and async_relookup_req_ids: + self.connector.on_load_failure(async_relookup_req_ids) + + # Mark async requests with KV load failures for retry once loading completes + self.failed_recving_kv_req_ids |= async_failed_req_ids + # Return sync affected IDs to skip in update_from_output +'''), + )), +) + + +def patch_mooncake(package_root: Path) -> bool: + """Preflight every source before writing; return False for the complete patch.""" + pending: list[tuple[Path, str]] = [] + already_patched = 0 + for relative, pristine_sha, patched_sha, edits in PATCHES: + path = package_root / relative + source = path.read_text() + digest = hashlib.sha256(source.encode()).hexdigest() + if digest == patched_sha: + already_patched += 1 + continue + if digest != pristine_sha: + raise RuntimeError(f"unsupported vLLM source: {path} (SHA256 {digest})") + patched = source + for old, new in edits: + if patched.count(old) != 1: + raise RuntimeError(f"unexpected patch context: {path}") + patched = patched.replace(old, new) + if hashlib.sha256(patched.encode()).hexdigest() != patched_sha: + raise RuntimeError(f"unexpected patched source: {path}") + pending.append((path, patched)) + if already_patched and pending: + raise RuntimeError("partially patched vLLM Mooncake recovery; refusing to serve") + for path, patched in pending: + path.write_text(patched) + return bool(pending) + + +def main() -> int: + try: + spec = importlib.util.find_spec("vllm") + if spec is None or not spec.submodule_search_locations: + raise RuntimeError("vllm package is not installed") + root = Path(next(iter(spec.submodule_search_locations))) + changed = patch_mooncake(root) + except (OSError, RuntimeError) as error: + print(f"ERROR: Kimi B300 Mooncake recovery patch: {error}", file=sys.stderr) + return 1 + print("Applied" if changed else "Already applied", "vLLM #55297 Mooncake recovery") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/inferencex-e2e/runners/test_kimik3_b300.py b/inferencex-e2e/runners/test_kimik3_b300.py new file mode 100644 index 0000000000..342a41b535 --- /dev/null +++ b/inferencex-e2e/runners/test_kimik3_b300.py @@ -0,0 +1,52 @@ +"""Unit coverage for the B300 Mooncake recovery patch.""" +from __future__ import annotations + +from pathlib import Path + +import pytest + + +@pytest.fixture +def mooncake_patch_fixture(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): + """Small package sources exercise the patcher's file-write contract.""" + import hashlib + from runners import patch_kimik3_mooncake_recovery as recovery + + original = {"first.py": "value = 1\n", "second.py": "value = 2\n"} + updated = {"first.py": "value = 3\n", "second.py": "value = 4\n"} + patches = [] + for name, source in original.items(): + (tmp_path / name).write_text(source) + patches.append(( + name, hashlib.sha256(source.encode()).hexdigest(), + hashlib.sha256(updated[name].encode()).hexdigest(), + ((source, updated[name]),), + )) + monkeypatch.setattr(recovery, "PATCHES", patches) + return recovery, tmp_path, original, updated + + +def test_mooncake_patch_is_complete_and_idempotent(mooncake_patch_fixture): + recovery, root, _, expected = mooncake_patch_fixture + assert recovery.patch_mooncake(root) is True + assert {name: (root / name).read_text() for name in expected} == expected + assert recovery.patch_mooncake(root) is False + assert {name: (root / name).read_text() for name in expected} == expected + + +@pytest.mark.parametrize("state", ["unknown", "partial", "missing"]) +def test_mooncake_patch_preflights_all_sources_before_writing( + mooncake_patch_fixture, state: str, +): + recovery, root, _, updated = mooncake_patch_fixture + second = root / "second.py" + if state == "unknown": + second.write_text("different upstream revision\n") + elif state == "partial": + second.write_text(updated["second.py"]) + else: + second.unlink() + before = {path.name: path.read_bytes() for path in root.iterdir()} + with pytest.raises((RuntimeError, OSError)): + recovery.patch_mooncake(root) + assert {path.name: path.read_bytes() for path in root.iterdir()} == before diff --git a/inferencex-e2e/runners/test_mooncake_rdma_device.py b/inferencex-e2e/runners/test_mooncake_rdma_device.py new file mode 100644 index 0000000000..2c2a8ef2b6 --- /dev/null +++ b/inferencex-e2e/runners/test_mooncake_rdma_device.py @@ -0,0 +1,62 @@ +"""Exercise rail selection with Linux sysfs layouts, including DSXE names.""" +import subprocess +from pathlib import Path + +LIB = Path(__file__).resolve().parents[1] / "benchmarks/benchmark_lib.sh" + + +def add_device( + root: Path, name: str, *, driver: str = "mlx5_core", + state: str = "4: ACTIVE", layer: str = "InfiniBand", +) -> None: + device = root / name + port = device / "ports/1" + port.mkdir(parents=True) + (port / "state").write_text(state + "\n") + (port / "link_layer").write_text(layer + "\n") + (device / "device").mkdir() + (device / "device/driver").symlink_to("/sys/bus/pci/drivers/" + driver) + + +def select(root: Path) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [ + "bash", + "-ec", + ( + 'source "$1"; select_mooncake_rdma_device "$2"; ' + 'printf "%s %s\n" "$MOONCAKE_RAIL" "$MC_GID_INDEX"' + ), + "bash", + str(LIB), + str(root), + ], + text=True, + capture_output=True, + timeout=5, + ) + + +def test_renamed_infiniband_device_skips_efa_and_down_port(tmp_path: Path) -> None: + add_device(tmp_path, "a_efa", driver="efa", layer="Unknown") + add_device(tmp_path, "ibp198s0f0", state="1: DOWN") + add_device(tmp_path, "ibp199s0f0") + result = select(tmp_path) + assert result.returncode == 0, result.stderr + assert result.stdout == "ibp199s0f0 0\n" + + +def test_roce_keeps_gid_three(tmp_path: Path) -> None: + add_device(tmp_path, "mlx5_0", state="1: DOWN", layer="Ethernet") + add_device(tmp_path, "mlx5_1", layer="Ethernet") + result = select(tmp_path) + assert result.returncode == 0, result.stderr + assert result.stdout == "mlx5_1 3\n" + + +def test_no_usable_rail_fails(tmp_path: Path) -> None: + add_device(tmp_path, "mlx5_0", state="1: DOWN") + add_device(tmp_path, "rdmap86s0", driver="efa", layer="Unknown") + result = select(tmp_path) + assert result.returncode != 0 + assert result.stdout == "" From 22385631c6e6fdf0062e253256c52af34fc007b5 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 01:22:25 +0000 Subject: [PATCH 02/31] fix: expose Mooncake RDMA helper under --validation-only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit kimik3-b300-mooncake.sh sources benchmark_lib.sh with --validation-only, but select_mooncake_rdma_device lived below that early return, so the setup script failed with exit 127 and cancelled the fail-fast canary. 将 select_mooncake_rdma_device 移到 --validation-only 门禁之上,并让测试 以相同方式 source,避免再次出现 command not found。 Co-authored-by: Wenyao Gao --- inferencex-e2e/benchmarks/benchmark_lib.sh | 46 ++++++++++--------- inferencex-e2e/perf-changelog.yaml | 9 ++++ .../runners/test_mooncake_rdma_device.py | 3 +- 3 files changed, 35 insertions(+), 23 deletions(-) diff --git a/inferencex-e2e/benchmarks/benchmark_lib.sh b/inferencex-e2e/benchmarks/benchmark_lib.sh index 7a418cb37b..a052e69a35 100644 --- a/inferencex-e2e/benchmarks/benchmark_lib.sh +++ b/inferencex-e2e/benchmarks/benchmark_lib.sh @@ -135,6 +135,30 @@ run_amd_multinode_after_preflight() { --kill-on-bad-exit=1 --signal=TERM@30 --unbuffered "$@" } +# Setup scripts (for example kimik3-b300-mooncake.sh) source this library with +# --validation-only, so the Mooncake rail helper must stay above that gate. +select_mooncake_rdma_device() { + local sysfs_root="${1:-/sys/class/infiniband}" + local device + MOONCAKE_RAIL="" + for device in "$sysfs_root"/*; do + # DSXE has both EFA and Mellanox adapters. The latter may be renamed + # ibp*, so identify the driver rather than assuming an mlx5_* name. + [[ "$(readlink "$device/device/driver" 2>/dev/null)" == */mlx5_core ]] || continue + grep -qx '4: ACTIVE' "$device/ports/1/state" 2>/dev/null || continue + case "$(cat "$device/ports/1/link_layer" 2>/dev/null)" in + InfiniBand) MC_GID_INDEX=0 ;; + Ethernet) MC_GID_INDEX=3 ;; + *) continue ;; + esac + MOONCAKE_RAIL="${device##*/}" + export MC_GID_INDEX + return 0 + done + echo "Error: no active Mellanox RDMA rail; Mooncake cannot initialise" >&2 + return 1 +} + # Launchers may load only input validation, without benchmark initialization. if [[ "${1-}" == "--validation-only" ]]; then return 0 @@ -178,28 +202,6 @@ PYPORT export PORT } -select_mooncake_rdma_device() { - local sysfs_root="${1:-/sys/class/infiniband}" - local device - MOONCAKE_RAIL="" - for device in "$sysfs_root"/*; do - # DSXE has both EFA and Mellanox adapters. The latter may be renamed - # ibp*, so identify the driver rather than assuming an mlx5_* name. - [[ "$(readlink "$device/device/driver" 2>/dev/null)" == */mlx5_core ]] || continue - grep -qx '4: ACTIVE' "$device/ports/1/state" 2>/dev/null || continue - case "$(cat "$device/ports/1/link_layer" 2>/dev/null)" in - InfiniBand) MC_GID_INDEX=0 ;; - Ethernet) MC_GID_INDEX=3 ;; - *) continue ;; - esac - MOONCAKE_RAIL="${device##*/}" - export MC_GID_INDEX - return 0 - done - echo "Error: no active Mellanox RDMA rail; Mooncake cannot initialise" >&2 - return 1 -} - agentic_kv_offload_enabled() { if [[ -z "${KV_OFFLOADING+x}" || -z "$KV_OFFLOADING" ]]; then echo "Error: KV_OFFLOADING must be set for agentic benchmarks" >&2 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index e22c7e8c9a..9dab336b9e 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9176,3 +9176,12 @@ - "Backport vLLM #55297 for request-local recompute after failed asynchronous Mooncake loads on B300; see docs/waiver/3088.md" - "Raise the Mooncake master lease from the 5s default to 60s (--default_kv_lease_ttl=60000): at concurrency >= 32 gets took 10-30s at p90, so 60-75% of transferred keys were discarded as LEASE_EXPIRED and recomputed; 60s matches the client's own 60s transfer cap. vLLM's execute_model timeout stays at its 300s default" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Move select_mooncake_rdma_device above benchmark_lib.sh's --validation-only gate so kimik3-b300-mooncake.sh can resolve the helper (exit 127 / command not found previously cancelled the fail-fast canary)" + - "将 select_mooncake_rdma_device 移到 benchmark_lib.sh 的 --validation-only 门禁之上,使 kimik3-b300-mooncake.sh 能解析该辅助函数(此前 exit 127 / command not found 导致 fail-fast canary 取消)" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 diff --git a/inferencex-e2e/runners/test_mooncake_rdma_device.py b/inferencex-e2e/runners/test_mooncake_rdma_device.py index 2c2a8ef2b6..4fb6965de9 100644 --- a/inferencex-e2e/runners/test_mooncake_rdma_device.py +++ b/inferencex-e2e/runners/test_mooncake_rdma_device.py @@ -19,12 +19,13 @@ def add_device( def select(root: Path) -> subprocess.CompletedProcess[str]: + # Match kimik3-b300-mooncake.sh: the helper must resolve under --validation-only. return subprocess.run( [ "bash", "-ec", ( - 'source "$1"; select_mooncake_rdma_device "$2"; ' + 'source "$1" --validation-only; select_mooncake_rdma_device "$2"; ' 'printf "%s %s\n" "$MOONCAKE_RAIL" "$MC_GID_INDEX"' ), "bash", From f5de7a52d66a25f7bfcb40d3cee0b928a7d56d89 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Wed, 30 Sep 2026 22:34:54 -0700 Subject: [PATCH 03/31] fix: restore B300 Mooncake execute_model timeout to 1800s MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip fail-fast 36800796192 killed agentic c40 after ~279s of engine silence under the default 300s VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS cap (shm_broadcast waits, then EngineDead / ProfileAborted). Restore the supported 1800s knob used by sibling Kimi Mooncake recipes; keep the 60s Mooncake master lease. 将 B300 Mooncake AgentX 的 execute_model 超时恢复为 1800 秒,保留 60 秒 master 租约。 Co-authored-by: Wenyao Gao --- .../srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 72e5ca7b01..b7ded411b4 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -85,6 +85,9 @@ base: VLLM_USE_DIRECT_DCP_KV_GATHER: '1' VLLM_ENGINE_READY_TIMEOUT_S: '3600' VLLM_RPC_TIMEOUT: '600000' + # Mooncake loads can block inside execute_model past the 300s default; + # c40 on run 36800796192 went silent ~279s then EngineDead at the cap. + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: '1800' VLLM_PREFIX_CACHE_RETENTION_INTERVAL: '0' VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS: '0' PYTHONNOUSERSITE: '1' From 76c772e0ff326cfa3c114c68cd2f35d9553eedca Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Wed, 30 Sep 2026 22:35:48 -0700 Subject: [PATCH 04/31] docs: changelog for B300 Mooncake execute_model timeout restore MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Append-only perf-changelog entry for restoring VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 after tip fail-fast 36800796192 c40 EngineDead / ProfileAborted. 为恢复 execute_model 1800 秒超时追加 changelog。 Co-authored-by: Wenyao Gao --- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9dab336b9e..cd78467f83 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9185,3 +9185,12 @@ - "Move select_mooncake_rdma_device above benchmark_lib.sh's --validation-only gate so kimik3-b300-mooncake.sh can resolve the helper (exit 127 / command not found previously cancelled the fail-fast canary)" - "将 select_mooncake_rdma_device 移到 benchmark_lib.sh 的 --validation-only 门禁之上,使 kimik3-b300-mooncake.sh 能解析该辅助函数(此前 exit 127 / command not found 导致 fail-fast canary 取消)" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Restore VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 on the B300 Mooncake AgentX recipe. Tip fail-fast run 36800796192 killed agentic c40 after ~279s of engine silence (shm_broadcast 60s waits from 04:57:58-05:00:58 PT window, then EngineCore TimeoutError / EngineDeadError at 05:01:56) under the default 300s execute_model cap; AIPerf then ProfileAborted at 30/292 = 10.3% failed requests. Keep the existing 60s Mooncake master lease (--default_kv_lease_ttl=60000). Sibling Kimi Mooncake recipes already set this timeout." + - "将 B300 Mooncake AgentX 配方的 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS 恢复为 1800。tip fail-fast 运行 36800796192 在默认 300 秒 execute_model 上限下,于约 279 秒引擎静默后杀死 agentic c40(shm_broadcast 60 秒等待,随后 EngineCore TimeoutError / EngineDeadError);AIPerf 随即以 30/292 = 10.3% 失败请求触发 ProfileAborted。保留既有的 60 秒 Mooncake master 租约。同系列 Kimi Mooncake 配方已设置该超时。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From ee33eebaf104a3d0e24abd24d84d6cb200525559 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 08:45:59 +0000 Subject: [PATCH 05/31] fix: align B300 Mooncake DCP8 IO with GB300 and fail empty rail MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip c48 already had device_name=ibp198s0f0 after setup (Wrote line is pre-patch); crash was GPU memcpy / DCP multimem under load. Match proven GB300 DCP8 compact_group_io + max_load_batch_keys=2 and MC_MAX_MR_SIZE, and refuse empty device_name or missing host mlx5 provider. 将 B300 Mooncake DCP8 IO 与已验证的 GB300 设置对齐,并在空 rail / 缺少主机 mlx5 provider 时明确失败。c48 的 Wrote device_name='' 是 patch 前转储;真实失败是高负载下的 GPU memcpy / DCP multimem。 Co-authored-by: Cursor --- inferencex-e2e/benchmarks/benchmark_lib.sh | 5 ++ .../configs/kimik3-b300-mooncake.sh | 18 +++++- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 12 +++- .../docs/configuration-procedures.md | 14 +++-- .../docs/configuration-procedures_zh.md | 12 ++-- inferencex-e2e/perf-changelog.yaml | 9 +++ .../runners/test_mooncake_rdma_device.py | 63 +++++++++++++++++++ 7 files changed, 119 insertions(+), 14 deletions(-) diff --git a/inferencex-e2e/benchmarks/benchmark_lib.sh b/inferencex-e2e/benchmarks/benchmark_lib.sh index a052e69a35..58a636feb5 100644 --- a/inferencex-e2e/benchmarks/benchmark_lib.sh +++ b/inferencex-e2e/benchmarks/benchmark_lib.sh @@ -152,6 +152,11 @@ select_mooncake_rdma_device() { *) continue ;; esac MOONCAKE_RAIL="${device##*/}" + if [[ -z "$MOONCAKE_RAIL" || "$MOONCAKE_RAIL" == "*" ]]; then + echo "Error: resolved an empty Mooncake RDMA rail name from $device" >&2 + return 1 + fi + export MOONCAKE_RAIL export MC_GID_INDEX return 0 done diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh index 34de47004c..152b7e202e 100755 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/kimik3-b300-mooncake.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # Pin the worker's Mooncake client, backport load-failure recovery, and point # the store at one active Mellanox RDMA rail (driver-selected, including ibp*). -set -euo pipefail +set -eo pipefail ws=/infmax-workspace # Temporary upstream #55297 backport; docs/waiver/3088.md is pending review. @@ -22,6 +22,10 @@ python3 -c "from mooncake.store import MooncakeDistributedStore" >/dev/null source "$ws/benchmarks/benchmark_lib.sh" --validation-only select_mooncake_rdma_device rail="$MOONCAKE_RAIL" +if [[ -z "$rail" ]]; then + echo "Error: select_mooncake_rdma_device returned an empty rail" >&2 + exit 1 +fi echo "Mooncake rail: $rail (MC_GID_INDEX=$MC_GID_INDEX)" # The enroot EFA hook binds the host libibverbs over the image's, and @@ -29,17 +33,25 @@ echo "Mooncake rail: $rail (MC_GID_INDEX=$MC_GID_INDEX)" # image's mlx5 provider never loads and the Mellanox rail vanishes. # runners.yaml mounts the host library directory at /host-usr-lib; libibverbs # appends its own -rdmavNN suffix to an absolute RDMAV_DRIVERS entry. -if [[ -d /host-usr-lib/libibverbs ]]; then - export RDMAV_DRIVERS=/host-usr-lib/libibverbs/libmlx5 +if [[ ! -d /host-usr-lib/libibverbs ]]; then + echo "Error: /host-usr-lib/libibverbs missing; host mlx5 provider required" >&2 + exit 1 fi +export RDMAV_DRIVERS=/host-usr-lib/libibverbs/libmlx5 config="${MOONCAKE_CONFIG_PATH:-/logs/mooncake_store_config.json}" python3 - "$config" "$rail" <<'PY' import json, sys path, rail = sys.argv[1:] +if not rail.strip(): + raise SystemExit("Error: refusing to write empty Mooncake device_name") with open(path) as handle: config = json.load(handle) config["device_name"] = rail with open(path, "w") as handle: json.dump(config, handle, indent=2) +written = json.load(open(path)) +if not str(written.get("device_name") or "").strip(): + raise SystemExit(f"Error: mooncake store config still has empty device_name: {written!r}") +print(f"Patched mooncake_store_config device_name={written['device_name']!r}") PY diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index b7ded411b4..8373a78109 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -28,8 +28,10 @@ base: # Embedded Mooncake: each TP rank contributes TOTAL_CPU_DRAM_GB / 8 GB. The # setup script pins the client, backports load-failure recovery, selects one # active Mellanox rail by driver (ibp* or mlx5_*), sets MC_GID_INDEX from the - # link layer, and loads the host mlx5 provider via RDMAV_DRIVERS when - # /host-usr-lib is mounted. Master lease is 60s (see mooncake-master args). + # link layer, requires the host mlx5 provider via RDMAV_DRIVERS at + # /host-usr-lib, and refuses an empty device_name. Master lease is 60s. + # device_name starts empty; kimik3-b300-mooncake.sh patches it before engines + # start (srt-slurm's "Wrote mooncake_store_config" line is the pre-patch dump). setup_script: kimik3-b300-mooncake.sh services: - name: mooncake-master @@ -74,7 +76,10 @@ base: attention-backend: TOKENSPEED_MLA attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' disable-uvicorn-access-log: true - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"enable_offload":false}}' + # compact_group_io + max_load_batch_keys match proven GB300 DCP8 Mooncake + # recipes; without them tip c48 died in Mooncake GPU memcpy concurrent with + # direct DCP kv-gather multimem timeout under high KV pressure. + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_offload":false}}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' @@ -99,6 +104,7 @@ base: MC_STORE_MEMCPY: '1' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' MC_SLICE_SIZE: '1048576' + MC_MAX_MR_SIZE: '4294967296' MC_WORKERS_PER_CTX: '4' WITH_NVIDIA_PEERMEM: '0' VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 5afa7ee55c..7ae7b6b61b 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -340,10 +340,16 @@ B300 uses the same minimum capture size at c1/c2/c4. Its c1 CI comparison reduce Kimi-K3 on B300 selects one active Mellanox adapter by its sysfs driver, including DSXE `ibp*` names; EFA devices are excluded from this RDMA recipe. The embedded Mooncake ranks share that adapter. InfiniBand uses GID index 0 and RoCE retains -index 3. If no compatible active adapter exists, startup fails before serving. -On DSXE the container's libibverbs comes from the host through the enroot EFA -hook, so `configs/runners.yaml` mounts the host library directory at `/host-usr-lib` -and the setup script loads its mlx5 provider through `RDMAV_DRIVERS`. +index 3. If no compatible active adapter exists, or if the host mlx5 provider mount +is missing, startup fails before serving. The recipe YAML may leave `device_name` +empty; `kimik3-b300-mooncake.sh` patches a real rail into the store config and +refuses to continue with an empty name (srt-slurm's earlier "Wrote +mooncake_store_config" line is the pre-patch dump). On DSXE the container's +libibverbs comes from the host through the enroot EFA hook, so +`configs/runners.yaml` mounts the host library directory at `/host-usr-lib` and +the setup script requires `RDMAV_DRIVERS` to load its mlx5 provider. The DCP8 +connector also sets `compact_group_io` and `max_load_batch_keys: 2` like the +proven GB300 DCP8 Mooncake recipes. The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 5abaa89cf4..fcf256e97c 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -314,10 +314,14 @@ B300 在 c1/c2/c4 使用相同的最小捕获范围。其 c1 CI 对比中,请 B300 上的 Kimi-K3 按 sysfs 驱动选择一块活动的 Mellanox 网卡,包括 DSXE 的 `ibp*` 命名;本 RDMA 配方排除 EFA。嵌入式 Mooncake 各 rank 共用该网卡。 -InfiniBand 使用 GID 索引 0,RoCE 保留索引 3。若没有可用的活动适配器,则在 -服务启动前失败。DSXE 上容器的 libibverbs 来自 enroot EFA hook 挂载的主机库, -因此 `configs/runners.yaml` 将主机库目录挂到 `/host-usr-lib`,setup 脚本通过 -`RDMAV_DRIVERS` 加载其 mlx5 provider。 +InfiniBand 使用 GID 索引 0,RoCE 保留索引 3。若没有可用的活动适配器,或缺少 +主机 mlx5 provider 挂载,则在服务启动前失败。配方 YAML 可将 `device_name` 留空; +`kimik3-b300-mooncake.sh` 会在引擎启动前把真实 rail 写入 store config,并拒绝 +空名称(srt-slurm 较早的 `Wrote mooncake_store_config` 行是 patch 前的转储)。 +DSXE 上容器的 libibverbs 来自 enroot EFA hook 挂载的主机库,因此 +`configs/runners.yaml` 将主机库目录挂到 `/host-usr-lib`,setup 脚本通过 +`RDMAV_DRIVERS` 强制加载其 mlx5 provider。DCP8 连接器还设置与已验证的 GB300 +DCP8 Mooncake 配方相同的 `compact_group_io` 与 `max_load_batch_keys: 2`。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index cd78467f83..52b33ce1ef 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9194,3 +9194,12 @@ - "Restore VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 on the B300 Mooncake AgentX recipe. Tip fail-fast run 36800796192 killed agentic c40 after ~279s of engine silence (shm_broadcast 60s waits from 04:57:58-05:00:58 PT window, then EngineCore TimeoutError / EngineDeadError at 05:01:56) under the default 300s execute_model cap; AIPerf then ProfileAborted at 30/292 = 10.3% failed requests. Keep the existing 60s Mooncake master lease (--default_kv_lease_ttl=60000). Sibling Kimi Mooncake recipes already set this timeout." - "将 B300 Mooncake AgentX 配方的 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS 恢复为 1800。tip fail-fast 运行 36800796192 在默认 300 秒 execute_model 上限下,于约 279 秒引擎静默后杀死 agentic c40(shm_broadcast 60 秒等待,随后 EngineCore TimeoutError / EngineDeadError);AIPerf 随即以 30/292 = 10.3% 失败请求触发 ProfileAborted。保留既有的 60 秒 Mooncake master 租约。同系列 Kimi Mooncake 配方已设置该超时。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Align B300 DCP8 Mooncake connector with proven GB300 DCP8 settings: compact_group_io + max_load_batch_keys=2 and MC_MAX_MR_SIZE=4GiB. Tip fail-fast run 36820538765 agentic c48 (job 110258065670, Slurm 6544) had a real rail (setup patched device_name to ibp198s0f0; srt-slurm Wrote line is pre-patch) then died with GPU memcpy fail / direct DCP kv-gather multimem timeout under high KV pressure. Also fail loud if the host mlx5 provider mount is missing or device_name would stay empty." + - "将 B300 DCP8 Mooncake 连接器与已验证的 GB300 DCP8 设置对齐:compact_group_io、max_load_batch_keys=2 以及 MC_MAX_MR_SIZE=4GiB。tip fail-fast 运行 36820538765 的 agentic c48(job 110258065670,Slurm 6544)已有真实 rail(setup 将 device_name 写成 ibp198s0f0;srt-slurm 的 Wrote 行为 patch 前转储),随后在高 KV 压力下因 GPU memcpy 失败 / direct DCP kv-gather multimem timeout 崩溃。主机 mlx5 provider 挂载缺失或 device_name 仍为空时改为明确失败。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 diff --git a/inferencex-e2e/runners/test_mooncake_rdma_device.py b/inferencex-e2e/runners/test_mooncake_rdma_device.py index 4fb6965de9..be91705349 100644 --- a/inferencex-e2e/runners/test_mooncake_rdma_device.py +++ b/inferencex-e2e/runners/test_mooncake_rdma_device.py @@ -1,8 +1,24 @@ """Exercise rail selection with Linux sysfs layouts, including DSXE names.""" +import json import subprocess from pathlib import Path LIB = Path(__file__).resolve().parents[1] / "benchmarks/benchmark_lib.sh" +PATCH_CONFIG = r""" +import json, sys +path, rail = sys.argv[1:] +if not rail.strip(): + raise SystemExit("Error: refusing to write empty Mooncake device_name") +with open(path) as handle: + config = json.load(handle) +config["device_name"] = rail +with open(path, "w") as handle: + json.dump(config, handle, indent=2) +written = json.load(open(path)) +if not str(written.get("device_name") or "").strip(): + raise SystemExit(f"Error: mooncake store config still has empty device_name: {written!r}") +print(written["device_name"]) +""" def add_device( @@ -38,6 +54,17 @@ def select(root: Path) -> subprocess.CompletedProcess[str]: ) +def patch_config(path: Path, rail: str) -> subprocess.CompletedProcess[str]: + """Same write+verify contract as kimik3-b300-mooncake.sh.""" + return subprocess.run( + ["python3", "-", str(path), rail], + input=PATCH_CONFIG, + text=True, + capture_output=True, + timeout=5, + ) + + def test_renamed_infiniband_device_skips_efa_and_down_port(tmp_path: Path) -> None: add_device(tmp_path, "a_efa", driver="efa", layer="Unknown") add_device(tmp_path, "ibp198s0f0", state="1: DOWN") @@ -61,3 +88,39 @@ def test_no_usable_rail_fails(tmp_path: Path) -> None: result = select(tmp_path) assert result.returncode != 0 assert result.stdout == "" + + +def test_select_then_patch_replaces_empty_device_name(tmp_path: Path) -> None: + add_device(tmp_path, "ibp198s0f0") + config = tmp_path / "mooncake_store_config.json" + config.write_text(json.dumps({ + "mode": "embedded", + "protocol": "rdma", + "device_name": "", + "enable_offload": False, + })) + selected = select(tmp_path) + assert selected.returncode == 0, selected.stderr + rail, gid = selected.stdout.strip().split() + assert rail == "ibp198s0f0" + assert gid == "0" + patched = patch_config(config, rail) + assert patched.returncode == 0, patched.stderr + assert patched.stdout.strip() == "ibp198s0f0" + written = json.loads(config.read_text()) + assert written["device_name"] == "ibp198s0f0" + assert written["protocol"] == "rdma" + + +def test_empty_rail_cannot_silently_patch_device_name(tmp_path: Path) -> None: + config = tmp_path / "mooncake_store_config.json" + config.write_text(json.dumps({ + "mode": "embedded", + "protocol": "rdma", + "device_name": "", + })) + for empty in ("", " "): + result = patch_config(config, empty) + assert result.returncode != 0, result.stdout + assert "empty" in (result.stderr + result.stdout).lower() + assert json.loads(config.read_text())["device_name"] == "" From 15820fd2e111c2eda0698d444b538b05a7cc3202 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 11:03:25 +0000 Subject: [PATCH 06/31] fix: drop B300 Mooncake compact_group_io; disable direct KV gather MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit c8 on tip ee33eeba had a real rail (Patched device_name=ibp198s0f0) but compact_group_io stormed TRANSFER_FAIL on ~25MiB puts then hung at 0 tok/s. Revert compact_group_io, keep max_load_batch_keys=2, and set VLLM_USE_DIRECT_DCP_KV_GATHER=0 after the prior multimem/memcpy signature. 撤销 B300 Mooncake 的 compact_group_io,关闭 direct KV gather。c8 已有真实 rail,但 compact_group_io 对约 25MiB put 大量 TRANSFER_FAIL 后挂起。 Co-authored-by: Cursor --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 13 ++++++++----- inferencex-e2e/docs/configuration-procedures.md | 7 ++++--- inferencex-e2e/docs/configuration-procedures_zh.md | 6 ++++-- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 25 insertions(+), 10 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 8373a78109..f129dc8acb 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -76,10 +76,11 @@ base: attention-backend: TOKENSPEED_MLA attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}' disable-uvicorn-access-log: true - # compact_group_io + max_load_batch_keys match proven GB300 DCP8 Mooncake - # recipes; without them tip c48 died in Mooncake GPU memcpy concurrent with - # direct DCP kv-gather multimem timeout under high KV pressure. - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"compact_group_io":true,"max_load_batch_keys":2,"enable_offload":false}}' + # Do NOT enable compact_group_io on DSXE single-rail Mooncake: tip ee33eeba + # c8 (run 36838397438) enabled it and immediately stormed TRANSFER_FAIL on + # compact-group-io-v1 ~25MiB puts (7177 fails), then hung at 0 tok/s until + # the 1800s sample_tokens cap. Keep max_load_batch_keys=2 only. + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"max_load_batch_keys":2,"enable_offload":false}}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' @@ -87,7 +88,9 @@ base: # These default to auto; name them so the measured DCP a2a path runs. VLLM_USE_DIRECT_DCP_A2A: '1' VLLM_USE_DIRECT_DCP_Q_GATHER: '1' - VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + # Prior tip c48 hit direct DCP kv-gather multimem timeout then GPU memcpy + # failure; disable direct KV gather (same as GB200 DCP16 / MI355X lanes). + VLLM_USE_DIRECT_DCP_KV_GATHER: '0' VLLM_ENGINE_READY_TIMEOUT_S: '3600' VLLM_RPC_TIMEOUT: '600000' # Mooncake loads can block inside execute_model past the 300s default; diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 7ae7b6b61b..8ef046e1e0 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -347,9 +347,10 @@ refuses to continue with an empty name (srt-slurm's earlier "Wrote mooncake_store_config" line is the pre-patch dump). On DSXE the container's libibverbs comes from the host through the enroot EFA hook, so `configs/runners.yaml` mounts the host library directory at `/host-usr-lib` and -the setup script requires `RDMAV_DRIVERS` to load its mlx5 provider. The DCP8 -connector also sets `compact_group_io` and `max_load_batch_keys: 2` like the -proven GB300 DCP8 Mooncake recipes. +the setup script requires `RDMAV_DRIVERS` to load its mlx5 provider. Keep +`max_load_batch_keys: 2`, but do not enable `compact_group_io` on this DSXE +single-rail path (it storm-failed ~25 MiB compact-group puts at c8). Direct DCP +KV gather is off (`VLLM_USE_DIRECT_DCP_KV_GATHER=0`). The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index fcf256e97c..6f33549cd7 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -320,8 +320,10 @@ InfiniBand 使用 GID 索引 0,RoCE 保留索引 3。若没有可用的活动 空名称(srt-slurm 较早的 `Wrote mooncake_store_config` 行是 patch 前的转储)。 DSXE 上容器的 libibverbs 来自 enroot EFA hook 挂载的主机库,因此 `configs/runners.yaml` 将主机库目录挂到 `/host-usr-lib`,setup 脚本通过 -`RDMAV_DRIVERS` 强制加载其 mlx5 provider。DCP8 连接器还设置与已验证的 GB300 -DCP8 Mooncake 配方相同的 `compact_group_io` 与 `max_load_batch_keys: 2`。 +`RDMAV_DRIVERS` 强制加载其 mlx5 provider。保留 `max_load_batch_keys: 2`,但不要在 +该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 +compact-group put 产生大量失败)。Direct DCP KV gather 关闭 +(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`)。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 52b33ce1ef..ad3bfcbc91 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9203,3 +9203,12 @@ - "Align B300 DCP8 Mooncake connector with proven GB300 DCP8 settings: compact_group_io + max_load_batch_keys=2 and MC_MAX_MR_SIZE=4GiB. Tip fail-fast run 36820538765 agentic c48 (job 110258065670, Slurm 6544) had a real rail (setup patched device_name to ibp198s0f0; srt-slurm Wrote line is pre-patch) then died with GPU memcpy fail / direct DCP kv-gather multimem timeout under high KV pressure. Also fail loud if the host mlx5 provider mount is missing or device_name would stay empty." - "将 B300 DCP8 Mooncake 连接器与已验证的 GB300 DCP8 设置对齐:compact_group_io、max_load_batch_keys=2 以及 MC_MAX_MR_SIZE=4GiB。tip fail-fast 运行 36820538765 的 agentic c48(job 110258065670,Slurm 6544)已有真实 rail(setup 将 device_name 写成 ibp198s0f0;srt-slurm 的 Wrote 行为 patch 前转储),随后在高 KV 压力下因 GPU memcpy 失败 / direct DCP kv-gather multimem timeout 崩溃。主机 mlx5 provider 挂载缺失或 device_name 仍为空时改为明确失败。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Revert compact_group_io on B300 DSXE Mooncake (keep max_load_batch_keys=2) and set VLLM_USE_DIRECT_DCP_KV_GATHER=0. Tip ee33eeba fail-fast run 36838397438 agentic c8 patched device_name to ibp198s0f0 (GH Wrote line is pre-setup), then Compact Mooncake group I/O (~25MiB puts) stormed 7177 TRANSFER_FAIL / save_put_failed_keys up to 3782 and hung at 0 tok/s until the 1800s sample_tokens timeout. Prior tip c48 multimem/memcpy with KV_GATHER=1 motivates disabling direct KV gather." + - "撤销 B300 DSXE Mooncake 上的 compact_group_io(保留 max_load_batch_keys=2),并将 VLLM_USE_DIRECT_DCP_KV_GATHER 设为 0。tip ee33eeba fail-fast 运行 36838397438 的 agentic c8 已将 device_name 写成 ibp198s0f0(GH Wrote 行为 setup 前转储),随后 Compact Mooncake group I/O(约 25MiB put)触发 7177 次 TRANSFER_FAIL / save_put_failed_keys 最高 3782,并以 0 tok/s 挂起直至 1800 秒 sample_tokens 超时。此前 tip c48 在 KV_GATHER=1 下出现 multimem/memcpy,因此关闭 direct KV gather。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 1d9c54d647f0641bb50f91d3a8c0c0fa484a4adb Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 13:43:22 +0000 Subject: [PATCH 07/31] fix: drop B300 Mooncake cumem after register_buffer -600 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip c2 (15820fd2e / run 36852970338) had a real ibp198s0f0 rail, then register_buffer failed (-600) on the ~39GiB cuMem KV region and stormed AddressNotRegistered TRANSFER_FAIL until sample_tokens hang. 中文:tip c2 已有真实 rail,但 cuMem KV 的 register_buffer 失败(-600) 导致 AddressNotRegistered TRANSFER_FAIL,最终 sample_tokens 挂起;关闭 enable-cumem-allocator。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 6 +++++- inferencex-e2e/docs/configuration-procedures.md | 5 ++++- inferencex-e2e/docs/configuration-procedures_zh.md | 5 ++++- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 22 insertions(+), 3 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index f129dc8acb..33d3a51c5e 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -68,7 +68,11 @@ base: load-format: fastsafetensors moe-backend: auto no-enable-flashinfer-autotune: true - enable-cumem-allocator: true + # Do NOT enable cuMem/VMM on DSXE Mooncake RDMA: tip 15820fd2e c2 (and the + # passing c1 canary) logged register_buffer failed ... len ~39GiB: -600 on + # every rank, then AddressNotRegistered TRANSFER_FAIL on puts inside that + # region. c1 recomputed through it; c2 hung at 0 tok/s until sample_tokens. + # Same class as LMCache dropping cumem when VMM buffers cannot be exported. enable-prefix-caching: true prefix-match-unit: 128 kv-cache-dtype: fp8 diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 8ef046e1e0..b95f6152c5 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -350,7 +350,10 @@ libibverbs comes from the host through the enroot EFA hook, so the setup script requires `RDMAV_DRIVERS` to load its mlx5 provider. Keep `max_load_batch_keys: 2`, but do not enable `compact_group_io` on this DSXE single-rail path (it storm-failed ~25 MiB compact-group puts at c8). Direct DCP -KV gather is off (`VLLM_USE_DIRECT_DCP_KV_GATHER=0`). +KV gather is off (`VLLM_USE_DIRECT_DCP_KV_GATHER=0`). Do not enable +`enable-cumem-allocator`: tip c2 (run 36852970338) and the passing c1 canary both +saw `register_buffer failed ... -600` on the ~39 GiB KV region, then +`AddressNotRegistered` TRANSFER_FAIL on puts; c2 hung until `sample_tokens`. The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 6f33549cd7..898aa4e8c0 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -323,7 +323,10 @@ DSXE 上容器的 libibverbs 来自 enroot EFA hook 挂载的主机库,因此 `RDMAV_DRIVERS` 强制加载其 mlx5 provider。保留 `max_load_batch_keys: 2`,但不要在 该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。Direct DCP KV gather 关闭 -(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`)。 +(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`)。不要启用 `enable-cumem-allocator`:tip c2 +(运行 36852970338)与通过的 c1 canary 均在约 39 GiB 的 KV 区域上出现 +`register_buffer failed ... -600`,随后 put 触发 `AddressNotRegistered` +TRANSFER_FAIL;c2 挂起直至 `sample_tokens`。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index ad3bfcbc91..7a148ba020 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9212,3 +9212,12 @@ - "Revert compact_group_io on B300 DSXE Mooncake (keep max_load_batch_keys=2) and set VLLM_USE_DIRECT_DCP_KV_GATHER=0. Tip ee33eeba fail-fast run 36838397438 agentic c8 patched device_name to ibp198s0f0 (GH Wrote line is pre-setup), then Compact Mooncake group I/O (~25MiB puts) stormed 7177 TRANSFER_FAIL / save_put_failed_keys up to 3782 and hung at 0 tok/s until the 1800s sample_tokens timeout. Prior tip c48 multimem/memcpy with KV_GATHER=1 motivates disabling direct KV gather." - "撤销 B300 DSXE Mooncake 上的 compact_group_io(保留 max_load_batch_keys=2),并将 VLLM_USE_DIRECT_DCP_KV_GATHER 设为 0。tip ee33eeba fail-fast 运行 36838397438 的 agentic c8 已将 device_name 写成 ibp198s0f0(GH Wrote 行为 setup 前转储),随后 Compact Mooncake group I/O(约 25MiB put)触发 7177 次 TRANSFER_FAIL / save_put_failed_keys 最高 3782,并以 0 tok/s 挂起直至 1800 秒 sample_tokens 超时。此前 tip c48 在 KV_GATHER=1 下出现 multimem/memcpy,因此关闭 direct KV gather。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Drop enable-cumem-allocator on B300 DSXE Mooncake. Tip 15820fd2e fail-fast run 36852970338 agentic c2 (job 110367970533, Slurm 6595, gpu-04) patched device_name to ibp198s0f0 (GH Wrote line is pre-setup); every rank logged register_buffer failed for the ~39GiB KV region (-600), then AddressNotRegistered TRANSFER_FAIL on puts inside that span; after ~38m healthy serve the engine hung at 0 tok/s (Running: 2) until the 1800s sample_tokens timeout / EngineDead / ProfileAborted. nccl_error:16 was ibv_query_port_speed WARN only. Passing tip canary c1 had the same register_buffer -600 storm but survived via recompute. Same class as LMCache dropping cumem when VMM buffers cannot be registered/exported; keep max_load_batch_keys=2, no compact_group_io, KV_GATHER=0, MC_MAX_MR_SIZE=4GiB." + - "在 B300 DSXE Mooncake 上关闭 enable-cumem-allocator。tip 15820fd2e fail-fast 运行 36852970338 的 agentic c2(job 110367970533,Slurm 6595,gpu-04)已将 device_name 写成 ibp198s0f0(GH Wrote 行为 setup 前转储);各 rank 均对约 39GiB KV 区域报 register_buffer failed(-600),随后该地址范围内的 put 触发 AddressNotRegistered TRANSFER_FAIL;约 38 分钟健康服务后引擎以 0 tok/s(Running: 2)挂起直至 1800 秒 sample_tokens 超时 / EngineDead / ProfileAborted。nccl_error:16 仅为 ibv_query_port_speed WARN。通过的 tip canary c1 有相同的 register_buffer -600 风暴,但靠 recompute 存活。与 LMCache 在 VMM 缓冲无法注册/导出时关闭 cumem 同类;保留 max_load_batch_keys=2、不启用 compact_group_io、KV_GATHER=0、MC_MAX_MR_SIZE=4GiB。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From e55883edaba03192d15b2e40b7ef235f5c75759e Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 16:36:07 +0000 Subject: [PATCH 08/31] fix: drop B300 Mooncake MC_MAX_MR_SIZE after register -600 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip c32 (1d9c54d / run 36870807626) still had real ibp198s0f0 rail with cumem off, but MC_MAX_MR_SIZE=4GiB made register_buffer fail (-600) on the ~43GiB KV region → AddressNotRegistered storm → NCCL worker crash. Pre-MC_MAX_MR tips registered cleanly; remove the 4GiB cap on DSXE. 中文:c32 在 cumem 已关且 rail 真实时仍因 MC_MAX_MR_SIZE=4GiB 导致 register_buffer -600 / AddressNotRegistered,最终 NCCL 杀 worker;移除 该上限。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 13 +++++++------ inferencex-e2e/docs/configuration-procedures.md | 9 +++++---- inferencex-e2e/docs/configuration-procedures_zh.md | 8 ++++---- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 25 insertions(+), 14 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 33d3a51c5e..b7e4e93ca2 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -68,11 +68,10 @@ base: load-format: fastsafetensors moe-backend: auto no-enable-flashinfer-autotune: true - # Do NOT enable cuMem/VMM on DSXE Mooncake RDMA: tip 15820fd2e c2 (and the - # passing c1 canary) logged register_buffer failed ... len ~39GiB: -600 on - # every rank, then AddressNotRegistered TRANSFER_FAIL on puts inside that - # region. c1 recomputed through it; c2 hung at 0 tok/s until sample_tokens. - # Same class as LMCache dropping cumem when VMM buffers cannot be exported. + # Keep cuMem/VMM off on this DSXE Mooncake path (LMCache-style caution for + # VMM buffers). The register_buffer -600 / AddressNotRegistered storm on + # tip 15820fd2e/1d9c54d was driven by MC_MAX_MR_SIZE=4GiB (removed below), + # not by cumem alone — pre-MC_MAX_MR tips registered with cumem on. enable-prefix-caching: true prefix-match-unit: 128 kv-cache-dtype: fp8 @@ -111,7 +110,9 @@ base: MC_STORE_MEMCPY: '1' MC_ENABLE_DEST_DEVICE_AFFINITY: '1' MC_SLICE_SIZE: '1048576' - MC_MAX_MR_SIZE: '4294967296' + # Do NOT set MC_MAX_MR_SIZE on DSXE: tip ee33eeba+ (4GiB) made every rank + # register_buffer fail (-600) on the ~40GiB KV region, then AddressNotRegistered + # TRANSFER_FAIL. Pre-MC_MAX_MR tips (c40/c48) registered cleanly without it. MC_WORKERS_PER_CTX: '4' WITH_NVIDIA_PEERMEM: '0' VLLM_MOONCAKE_LOAD_RECV_THREADS: '4' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index b95f6152c5..80c8231212 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -350,10 +350,11 @@ libibverbs comes from the host through the enroot EFA hook, so the setup script requires `RDMAV_DRIVERS` to load its mlx5 provider. Keep `max_load_batch_keys: 2`, but do not enable `compact_group_io` on this DSXE single-rail path (it storm-failed ~25 MiB compact-group puts at c8). Direct DCP -KV gather is off (`VLLM_USE_DIRECT_DCP_KV_GATHER=0`). Do not enable -`enable-cumem-allocator`: tip c2 (run 36852970338) and the passing c1 canary both -saw `register_buffer failed ... -600` on the ~39 GiB KV region, then -`AddressNotRegistered` TRANSFER_FAIL on puts; c2 hung until `sample_tokens`. +KV gather is off (`VLLM_USE_DIRECT_DCP_KV_GATHER=0`). Do not set +`MC_MAX_MR_SIZE` here: with 4GiB every rank hit `register_buffer failed ... -600` +on the ~40 GiB KV region and stormed `AddressNotRegistered` TRANSFER_FAIL +(c2/c32); pre-`MC_MAX_MR` tips registered cleanly. Keep `enable-cumem-allocator` +off on this path. The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 898aa4e8c0..ad1e472b7e 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -323,10 +323,10 @@ DSXE 上容器的 libibverbs 来自 enroot EFA hook 挂载的主机库,因此 `RDMAV_DRIVERS` 强制加载其 mlx5 provider。保留 `max_load_batch_keys: 2`,但不要在 该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。Direct DCP KV gather 关闭 -(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`)。不要启用 `enable-cumem-allocator`:tip c2 -(运行 36852970338)与通过的 c1 canary 均在约 39 GiB 的 KV 区域上出现 -`register_buffer failed ... -600`,随后 put 触发 `AddressNotRegistered` -TRANSFER_FAIL;c2 挂起直至 `sample_tokens`。 +(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`)。不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB +时各 rank 对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 +`AddressNotRegistered` TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。 +此路径保持关闭 `enable-cumem-allocator`。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 7a148ba020..502a34ea2f 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9221,3 +9221,12 @@ - "Drop enable-cumem-allocator on B300 DSXE Mooncake. Tip 15820fd2e fail-fast run 36852970338 agentic c2 (job 110367970533, Slurm 6595, gpu-04) patched device_name to ibp198s0f0 (GH Wrote line is pre-setup); every rank logged register_buffer failed for the ~39GiB KV region (-600), then AddressNotRegistered TRANSFER_FAIL on puts inside that span; after ~38m healthy serve the engine hung at 0 tok/s (Running: 2) until the 1800s sample_tokens timeout / EngineDead / ProfileAborted. nccl_error:16 was ibv_query_port_speed WARN only. Passing tip canary c1 had the same register_buffer -600 storm but survived via recompute. Same class as LMCache dropping cumem when VMM buffers cannot be registered/exported; keep max_load_batch_keys=2, no compact_group_io, KV_GATHER=0, MC_MAX_MR_SIZE=4GiB." - "在 B300 DSXE Mooncake 上关闭 enable-cumem-allocator。tip 15820fd2e fail-fast 运行 36852970338 的 agentic c2(job 110367970533,Slurm 6595,gpu-04)已将 device_name 写成 ibp198s0f0(GH Wrote 行为 setup 前转储);各 rank 均对约 39GiB KV 区域报 register_buffer failed(-600),随后该地址范围内的 put 触发 AddressNotRegistered TRANSFER_FAIL;约 38 分钟健康服务后引擎以 0 tok/s(Running: 2)挂起直至 1800 秒 sample_tokens 超时 / EngineDead / ProfileAborted。nccl_error:16 仅为 ibv_query_port_speed WARN。通过的 tip canary c1 有相同的 register_buffer -600 风暴,但靠 recompute 存活。与 LMCache 在 VMM 缓冲无法注册/导出时关闭 cumem 同类;保留 max_load_batch_keys=2、不启用 compact_group_io、KV_GATHER=0、MC_MAX_MR_SIZE=4GiB。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Remove MC_MAX_MR_SIZE from B300 DSXE Mooncake. Tip 1d9c54d fail-fast run 36870807626 agentic c32 (job 110437058728, Slurm 6614, gpu-01) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; GH Wrote line is pre-setup) and cumem already off, but every rank still logged register_buffer failed for the ~43.7GiB KV region (-600) then ~128k AddressNotRegistered TRANSFER_FAIL; after ~40m of degraded serve an NCCL ALLGATHER watchdog (600s) killed VllmWorker-2 (worker_crash:8) → EngineDead / ProfileAborted. Pre-MC_MAX_MR tips (old c40/c48) had zero register_buffer -600; the 4GiB cap was added in ee33eeba GB300 alignment and is the verified driver. Keep no compact_group_io, max_load_batch_keys=2, KV_GATHER=0, cumem off." + - "从 B300 DSXE Mooncake 移除 MC_MAX_MR_SIZE。tip 1d9c54d fail-fast 运行 36870807626 的 agentic c32(job 110437058728,Slurm 6614,gpu-01)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched;GH Wrote 行为 setup 前转储)且 cumem 已关,但各 rank 仍对约 43.7GiB KV 区域报 register_buffer failed(-600),随后约 12.8 万次 AddressNotRegistered TRANSFER_FAIL;约 40 分钟降级服务后 NCCL ALLGATHER watchdog(600 秒)杀死 VllmWorker-2(worker_crash:8)→ EngineDead / ProfileAborted。加入 MC_MAX_MR 之前的 tip(旧 c40/c48)无 register_buffer -600;该 4GiB 上限来自 ee33eeba 的 GB300 对齐,是已验证根因。保留不启用 compact_group_io、max_load_batch_keys=2、KV_GATHER=0、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 97f83d115e0e16c18fafa1ce99c396ea86b011fc Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 16:41:19 +0000 Subject: [PATCH 09/31] chore: touch perf-changelog to retrigger fail-fast sweep MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit EOF blank line so pull_request synchronize matches run-sweep paths. 中文:在 perf-changelog 末尾追加空行,触发 synchronize 以重跑 fail-fast sweep。 Co-authored-by: Wenyao Gao --- inferencex-e2e/perf-changelog.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 502a34ea2f..72b3eba103 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9230,3 +9230,4 @@ - "Remove MC_MAX_MR_SIZE from B300 DSXE Mooncake. Tip 1d9c54d fail-fast run 36870807626 agentic c32 (job 110437058728, Slurm 6614, gpu-01) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; GH Wrote line is pre-setup) and cumem already off, but every rank still logged register_buffer failed for the ~43.7GiB KV region (-600) then ~128k AddressNotRegistered TRANSFER_FAIL; after ~40m of degraded serve an NCCL ALLGATHER watchdog (600s) killed VllmWorker-2 (worker_crash:8) → EngineDead / ProfileAborted. Pre-MC_MAX_MR tips (old c40/c48) had zero register_buffer -600; the 4GiB cap was added in ee33eeba GB300 alignment and is the verified driver. Keep no compact_group_io, max_load_batch_keys=2, KV_GATHER=0, cumem off." - "从 B300 DSXE Mooncake 移除 MC_MAX_MR_SIZE。tip 1d9c54d fail-fast 运行 36870807626 的 agentic c32(job 110437058728,Slurm 6614,gpu-01)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched;GH Wrote 行为 setup 前转储)且 cumem 已关,但各 rank 仍对约 43.7GiB KV 区域报 register_buffer failed(-600),随后约 12.8 万次 AddressNotRegistered TRANSFER_FAIL;约 40 分钟降级服务后 NCCL ALLGATHER watchdog(600 秒)杀死 VllmWorker-2(worker_crash:8)→ EngineDead / ProfileAborted。加入 MC_MAX_MR 之前的 tip(旧 c40/c48)无 register_buffer -600;该 4GiB 上限来自 ee33eeba 的 GB300 对齐,是已验证根因。保留不启用 compact_group_io、max_load_batch_keys=2、KV_GATHER=0、关闭 cumem。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + From e51c58f5c02daf55470c0497ce5d5447ad70b935 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 19:38:36 +0000 Subject: [PATCH 10/31] fix(kimik3-b300): cut Mooncake max_load_batch_keys to 1 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip 860c1ccf c48 hung in DCP PYNCCL ALLGATHER under ~100% KV with clean Mooncake metrics; reduce async load batch pressure. 将 tip 860c1ccf c48 在约 100% KV 下挂起于 DCP PYNCCL ALLGATHER(Mooncake 指标干净)的根因对应为异步 load 批压力,将 max_load_batch_keys 降为 1。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 6 ++++-- inferencex-e2e/docs/configuration-procedures.md | 15 ++++++++------- .../docs/configuration-procedures_zh.md | 16 +++++++++------- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 30 insertions(+), 16 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index b7e4e93ca2..796c020b57 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -82,8 +82,10 @@ base: # Do NOT enable compact_group_io on DSXE single-rail Mooncake: tip ee33eeba # c8 (run 36838397438) enabled it and immediately stormed TRANSFER_FAIL on # compact-group-io-v1 ~25MiB puts (7177 fails), then hung at 0 tok/s until - # the 1800s sample_tokens cap. Keep max_load_batch_keys=2 only. - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"max_load_batch_keys":2,"enable_offload":false}}' + # the 1800s sample_tokens cap. Tip 860c1ccf c48 (run 36894040300) had clean + # Mooncake metrics then hung in DCP PYNCCL _ALLGATHER_BASE (PG 3) under + # ~100% GPU KV with max_load_batch_keys=2 async loads; cut to 1. + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"max_load_batch_keys":1,"enable_offload":false}}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 80c8231212..3024bdac4c 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -348,13 +348,14 @@ mooncake_store_config" line is the pre-patch dump). On DSXE the container's libibverbs comes from the host through the enroot EFA hook, so `configs/runners.yaml` mounts the host library directory at `/host-usr-lib` and the setup script requires `RDMAV_DRIVERS` to load its mlx5 provider. Keep -`max_load_batch_keys: 2`, but do not enable `compact_group_io` on this DSXE -single-rail path (it storm-failed ~25 MiB compact-group puts at c8). Direct DCP -KV gather is off (`VLLM_USE_DIRECT_DCP_KV_GATHER=0`). Do not set -`MC_MAX_MR_SIZE` here: with 4GiB every rank hit `register_buffer failed ... -600` -on the ~40 GiB KV region and stormed `AddressNotRegistered` TRANSFER_FAIL -(c2/c32); pre-`MC_MAX_MR` tips registered cleanly. Keep `enable-cumem-allocator` -off on this path. +`max_load_batch_keys: 1` (tip 860c1ccf c48 hung in DCP PYNCCL `_ALLGATHER_BASE` +under ~100% GPU KV with batch keys=2 and clean Mooncake metrics), but do not +enable `compact_group_io` on this DSXE single-rail path (it storm-failed ~25 MiB +compact-group puts at c8). Direct DCP KV gather is off +(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`). Do not set `MC_MAX_MR_SIZE` here: with 4GiB +every rank hit `register_buffer failed ... -600` on the ~40 GiB KV region and +stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips +registered cleanly. Keep `enable-cumem-allocator` off on this path. The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index ad1e472b7e..dfb0d14cfb 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -320,13 +320,15 @@ InfiniBand 使用 GID 索引 0,RoCE 保留索引 3。若没有可用的活动 空名称(srt-slurm 较早的 `Wrote mooncake_store_config` 行是 patch 前的转储)。 DSXE 上容器的 libibverbs 来自 enroot EFA hook 挂载的主机库,因此 `configs/runners.yaml` 将主机库目录挂到 `/host-usr-lib`,setup 脚本通过 -`RDMAV_DRIVERS` 强制加载其 mlx5 provider。保留 `max_load_batch_keys: 2`,但不要在 -该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 -compact-group put 产生大量失败)。Direct DCP KV gather 关闭 -(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`)。不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB -时各 rank 对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 -`AddressNotRegistered` TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。 -此路径保持关闭 `enable-cumem-allocator`。 +`RDMAV_DRIVERS` 强制加载其 mlx5 provider。保留 `max_load_batch_keys: 1` +(tip 860c1ccf 的 c48 在 batch keys=2、Mooncake 指标干净时,于约 100% GPU KV +下挂起在 DCP PYNCCL `_ALLGATHER_BASE`),但不要在该 DSXE 单 rail 路径上启用 +`compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。 +Direct DCP KV gather 关闭(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`)。不要在此设置 +`MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 +`register_buffer failed ... -600`,并引发 `AddressNotRegistered` TRANSFER_FAIL +(c2/c32);加入该变量之前的 tip 注册正常。此路径保持关闭 +`enable-cumem-allocator`。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index ebcb75de57..5add77948e 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9242,3 +9242,12 @@ - "Update the SGLang ROCm image from lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260927 to lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260929 on this arm only. The other MI355X arms are left on their current tags." - "No serving flag outside the HiCache block changes, and no other config key is touched. Each of the 19 matrix points resolves to exactly one recipe override, with no override left unused." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3611 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cut Mooncake max_load_batch_keys from 2 to 1 on B300 DSXE. Tip 860c1ccf fail-fast run 36894040300 agentic c48 (job 110510865864, Slurm 6639, gpu-03) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; GH Wrote line is pre-setup), zero register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem, and clean Mooncake load/save metrics (failed_keys=0) through ~35m of serve; then under GPU KV ~100% the engine hung at 0 tok/s (Running: 4, Waiting: 46) and NCCL _ALLGATHER_BASE on PG ID 3 timed out after 600s (last started work: -1) → VllmWorker-4 died (worker_crash:8) → EngineDead / ProfileAborted. Keep no compact_group_io, no MC_MAX_MR_SIZE, KV_GATHER=0, cumem off, load_async true." + - "将 B300 DSXE 上的 Mooncake max_load_batch_keys 从 2 降为 1。tip 860c1ccf fail-fast 运行 36894040300 的 agentic c48(job 110510865864,Slurm 6639,gpu-03)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched;GH Wrote 行为 setup 前转储),无 register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem,且 Mooncake load/save 指标干净(failed_keys=0)约 35 分钟;随后在 GPU KV ~100% 下引擎以 0 tok/s 挂起(Running: 4,Waiting: 46),PG ID 3 上的 NCCL _ALLGATHER_BASE 在 600 秒后超时(last started work: -1)→ VllmWorker-4 死亡(worker_crash:8)→ EngineDead / ProfileAborted。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、KV_GATHER=0、关闭 cumem、load_async true。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 670fb9f06c37f0677e8ee831edebf819beb23d06 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 22:29:45 +0000 Subject: [PATCH 11/31] fix(kimik3-b300): disable Mooncake load_async on DSXE MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip e51c58f5 c32 repeated the DCP PYNCCL ALLGATHER hang under ~98% KV with clean Mooncake metrics; max_load_batch_keys=1 alone was insufficient. 将 tip e51c58f5 c32 在约 98% KV 下重复的 DCP PYNCCL ALLGATHER 挂起(Mooncake 指标干净;仅降 batch_keys 不够)对应为关闭 load_async。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 9 +++++---- inferencex-e2e/docs/configuration-procedures.md | 5 +++-- inferencex-e2e/docs/configuration-procedures_zh.md | 7 ++++--- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 21 insertions(+), 9 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 796c020b57..6d93f792fc 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -82,10 +82,11 @@ base: # Do NOT enable compact_group_io on DSXE single-rail Mooncake: tip ee33eeba # c8 (run 36838397438) enabled it and immediately stormed TRANSFER_FAIL on # compact-group-io-v1 ~25MiB puts (7177 fails), then hung at 0 tok/s until - # the 1800s sample_tokens cap. Tip 860c1ccf c48 (run 36894040300) had clean - # Mooncake metrics then hung in DCP PYNCCL _ALLGATHER_BASE (PG 3) under - # ~100% GPU KV with max_load_batch_keys=2 async loads; cut to 1. - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"max_load_batch_keys":1,"enable_offload":false}}' + # the 1800s sample_tokens cap. Tips 860c1ccf c48 / e51c58f5 c32 had clean + # Mooncake metrics then hung in DCP PYNCCL _ALLGATHER_BASE (PG 3, last + # started work: -1) under ~98-100% GPU KV with load_async=true; disable + # async loads (keep max_load_batch_keys=1, lookup_async). + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":false,"lookup_async":true,"max_load_batch_keys":1,"enable_offload":false}}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 3024bdac4c..a123ccd5f7 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -348,8 +348,9 @@ mooncake_store_config" line is the pre-patch dump). On DSXE the container's libibverbs comes from the host through the enroot EFA hook, so `configs/runners.yaml` mounts the host library directory at `/host-usr-lib` and the setup script requires `RDMAV_DRIVERS` to load its mlx5 provider. Keep -`max_load_batch_keys: 1` (tip 860c1ccf c48 hung in DCP PYNCCL `_ALLGATHER_BASE` -under ~100% GPU KV with batch keys=2 and clean Mooncake metrics), but do not +`max_load_batch_keys: 1` and `load_async: false` (tips 860c1ccf c48 / +e51c58f5 c32 hung in DCP PYNCCL `_ALLGATHER_BASE` under ~98–100% GPU KV with +async loads and clean Mooncake metrics; `last started work: -1`), but do not enable `compact_group_io` on this DSXE single-rail path (it storm-failed ~25 MiB compact-group puts at c8). Direct DCP KV gather is off (`VLLM_USE_DIRECT_DCP_KV_GATHER=0`). Do not set `MC_MAX_MR_SIZE` here: with 4GiB diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index dfb0d14cfb..7f1e5c9c67 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -320,9 +320,10 @@ InfiniBand 使用 GID 索引 0,RoCE 保留索引 3。若没有可用的活动 空名称(srt-slurm 较早的 `Wrote mooncake_store_config` 行是 patch 前的转储)。 DSXE 上容器的 libibverbs 来自 enroot EFA hook 挂载的主机库,因此 `configs/runners.yaml` 将主机库目录挂到 `/host-usr-lib`,setup 脚本通过 -`RDMAV_DRIVERS` 强制加载其 mlx5 provider。保留 `max_load_batch_keys: 1` -(tip 860c1ccf 的 c48 在 batch keys=2、Mooncake 指标干净时,于约 100% GPU KV -下挂起在 DCP PYNCCL `_ALLGATHER_BASE`),但不要在该 DSXE 单 rail 路径上启用 +`RDMAV_DRIVERS` 强制加载其 mlx5 provider。保留 `max_load_batch_keys: 1` 且 +`load_async: false`(tip 860c1ccf 的 c48 / tip e51c58f5 的 c32 在异步 load、 +Mooncake 指标干净时,于约 98–100% GPU KV 下挂起在 DCP PYNCCL +`_ALLGATHER_BASE`,`last started work: -1`),但不要在该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。 Direct DCP KV gather 关闭(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`)。不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 5add77948e..9ea9f0b87d 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9251,3 +9251,12 @@ - "Cut Mooncake max_load_batch_keys from 2 to 1 on B300 DSXE. Tip 860c1ccf fail-fast run 36894040300 agentic c48 (job 110510865864, Slurm 6639, gpu-03) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; GH Wrote line is pre-setup), zero register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem, and clean Mooncake load/save metrics (failed_keys=0) through ~35m of serve; then under GPU KV ~100% the engine hung at 0 tok/s (Running: 4, Waiting: 46) and NCCL _ALLGATHER_BASE on PG ID 3 timed out after 600s (last started work: -1) → VllmWorker-4 died (worker_crash:8) → EngineDead / ProfileAborted. Keep no compact_group_io, no MC_MAX_MR_SIZE, KV_GATHER=0, cumem off, load_async true." - "将 B300 DSXE 上的 Mooncake max_load_batch_keys 从 2 降为 1。tip 860c1ccf fail-fast 运行 36894040300 的 agentic c48(job 110510865864,Slurm 6639,gpu-03)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched;GH Wrote 行为 setup 前转储),无 register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem,且 Mooncake load/save 指标干净(failed_keys=0)约 35 分钟;随后在 GPU KV ~100% 下引擎以 0 tok/s 挂起(Running: 4,Waiting: 46),PG ID 3 上的 NCCL _ALLGATHER_BASE 在 600 秒后超时(last started work: -1)→ VllmWorker-4 死亡(worker_crash:8)→ EngineDead / ProfileAborted。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、KV_GATHER=0、关闭 cumem、load_async true。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Disable Mooncake load_async on B300 DSXE (keep max_load_batch_keys=1, lookup_async). Tip e51c58f5 fail-fast run 36915805279 agentic c32 (job 110582923073, Slurm 6677, gpu-07) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; GH Wrote line is pre-setup), zero register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem, and clean Mooncake metrics (failed_keys=0); under GPU KV ~98% the engine hung at 0 tok/s (Running: 10, Waiting: 23) and NCCL _ALLGATHER_BASE on PG ID 3 timed out after 600s (last started work: -1) → VllmWorker-1 died (worker_crash:8) → EngineDead / ProfileAborted. Same signature as tip 860c1ccf c48 with batch_keys=2; cutting batch_keys to 1 alone did not stop the hang. Keep no compact_group_io, no MC_MAX_MR_SIZE, KV_GATHER=0, cumem off." + - "在 B300 DSXE 上关闭 Mooncake load_async(保留 max_load_batch_keys=1、lookup_async)。tip e51c58f5 fail-fast 运行 36915805279 的 agentic c32(job 110582923073,Slurm 6677,gpu-07)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched;GH Wrote 行为 setup 前转储),无 register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem,且 Mooncake 指标干净(failed_keys=0);在 GPU KV ~98% 下引擎以 0 tok/s 挂起(Running: 10,Waiting: 23),PG ID 3 上的 NCCL _ALLGATHER_BASE 在 600 秒后超时(last started work: -1)→ VllmWorker-1 死亡(worker_crash:8)→ EngineDead / ProfileAborted。与 tip 860c1ccf 的 c48(batch_keys=2)签名相同;仅将 batch_keys 降为 1 未能阻止挂起。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、KV_GATHER=0、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From d7176e6ec6cfcf2a8720adcfb551f70160b0e200 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 15:53:59 -0700 Subject: [PATCH 12/31] fix(kimik3-b300): restore Mooncake load_async=true (connector assert) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip ea88d652 fail-fast 36935627690 canary c1 aborted warmup with AssertionError: load_async must be True for better performance in mooncake/store/worker.py get_finished. Tip 670fb9f0 disable is incompatible with this vLLM Mooncake path; keep max_load_batch_keys=1. 中文:tip ea88d652 canary c1 因 Mooncake 断言要求 load_async=true 而崩溃; 恢复 stock true,保留 max_load_batch_keys=1。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 12 +++++++----- inferencex-e2e/docs/configuration-procedures.md | 10 ++++++---- inferencex-e2e/docs/configuration-procedures_zh.md | 7 +++++-- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 27 insertions(+), 11 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 6d93f792fc..2e16b67889 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -82,11 +82,13 @@ base: # Do NOT enable compact_group_io on DSXE single-rail Mooncake: tip ee33eeba # c8 (run 36838397438) enabled it and immediately stormed TRANSFER_FAIL on # compact-group-io-v1 ~25MiB puts (7177 fails), then hung at 0 tok/s until - # the 1800s sample_tokens cap. Tips 860c1ccf c48 / e51c58f5 c32 had clean - # Mooncake metrics then hung in DCP PYNCCL _ALLGATHER_BASE (PG 3, last - # started work: -1) under ~98-100% GPU KV with load_async=true; disable - # async loads (keep max_load_batch_keys=1, lookup_async). - kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":false,"lookup_async":true,"max_load_batch_keys":1,"enable_offload":false}}' + # the 1800s sample_tokens cap. Keep max_load_batch_keys=1 after tips + # 860c1ccf c48 / e51c58f5 c32 DCP PYNCCL _ALLGATHER_BASE hangs under + # ~98-100% GPU KV. Tip ea88d652 / sweep 36935627690 canary c1 crashed + # immediately with AssertionError "load_async must be True for better + # performance" in mooncake store worker get_finished when load_async was + # false — restore load_async=true (required by this vLLM Mooncake path). + kv-transfer-config: '{"kv_connector":"MooncakeStoreConnector","kv_role":"kv_both","kv_load_failure_policy":"recompute","kv_connector_extra_config":{"load_async":true,"lookup_async":true,"max_load_batch_keys":1,"enable_offload":false}}' env: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index ed8e684240..86f8514932 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -340,11 +340,13 @@ mooncake_store_config" line is the pre-patch dump). On DSXE the container's libibverbs comes from the host through the enroot EFA hook, so `configs/runners.yaml` mounts the host library directory at `/host-usr-lib` and the setup script requires `RDMAV_DRIVERS` to load its mlx5 provider. Keep -`max_load_batch_keys: 1` and `load_async: false` (tips 860c1ccf c48 / +`max_load_batch_keys: 1` and `load_async: true` (tips 860c1ccf c48 / e51c58f5 c32 hung in DCP PYNCCL `_ALLGATHER_BASE` under ~98–100% GPU KV with -async loads and clean Mooncake metrics; `last started work: -1`), but do not -enable `compact_group_io` on this DSXE single-rail path (it storm-failed ~25 MiB -compact-group puts at c8). Direct DCP KV gather is off +async loads and clean Mooncake metrics; `last started work: -1`. Tip ea88d652 +canary c1 crashed with Mooncake `AssertionError: load_async must be True for +better performance` when `load_async` was set false, so restore the required +stock true), but do not enable `compact_group_io` on this DSXE single-rail path +(it storm-failed ~25 MiB compact-group puts at c8). Direct DCP KV gather is off (`VLLM_USE_DIRECT_DCP_KV_GATHER=0`). Do not set `MC_MAX_MR_SIZE` here: with 4GiB every rank hit `register_buffer failed ... -600` on the ~40 GiB KV region and stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 451e129482..89f1d3235f 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -313,9 +313,12 @@ InfiniBand 使用 GID 索引 0,RoCE 保留索引 3。若没有可用的活动 DSXE 上容器的 libibverbs 来自 enroot EFA hook 挂载的主机库,因此 `configs/runners.yaml` 将主机库目录挂到 `/host-usr-lib`,setup 脚本通过 `RDMAV_DRIVERS` 强制加载其 mlx5 provider。保留 `max_load_batch_keys: 1` 且 -`load_async: false`(tip 860c1ccf 的 c48 / tip e51c58f5 的 c32 在异步 load、 +`load_async: true`(tip 860c1ccf 的 c48 / tip e51c58f5 的 c32 在异步 load、 Mooncake 指标干净时,于约 98–100% GPU KV 下挂起在 DCP PYNCCL -`_ALLGATHER_BASE`,`last started work: -1`),但不要在该 DSXE 单 rail 路径上启用 +`_ALLGATHER_BASE`,`last started work: -1`;tip ea88d652 的 canary c1 在 +`load_async: false` 时于 Mooncake `get_finished` 直接 +`AssertionError: load_async must be True for better performance`,故恢复必需的 +stock true),但不要在该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。 Direct DCP KV gather 关闭(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`)。不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index a8d7409f5a..9fabbf44fb 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9282,3 +9282,12 @@ - "Disable Mooncake load_async on B300 DSXE (keep max_load_batch_keys=1, lookup_async). Tip e51c58f5 fail-fast run 36915805279 agentic c32 (job 110582923073, Slurm 6677, gpu-07) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; GH Wrote line is pre-setup), zero register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem, and clean Mooncake metrics (failed_keys=0); under GPU KV ~98% the engine hung at 0 tok/s (Running: 10, Waiting: 23) and NCCL _ALLGATHER_BASE on PG ID 3 timed out after 600s (last started work: -1) → VllmWorker-1 died (worker_crash:8) → EngineDead / ProfileAborted. Same signature as tip 860c1ccf c48 with batch_keys=2; cutting batch_keys to 1 alone did not stop the hang. Keep no compact_group_io, no MC_MAX_MR_SIZE, KV_GATHER=0, cumem off." - "在 B300 DSXE 上关闭 Mooncake load_async(保留 max_load_batch_keys=1、lookup_async)。tip e51c58f5 fail-fast 运行 36915805279 的 agentic c32(job 110582923073,Slurm 6677,gpu-07)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched;GH Wrote 行为 setup 前转储),无 register_buffer -600 / AddressNotRegistered / TRANSFER_FAIL / multimem,且 Mooncake 指标干净(failed_keys=0);在 GPU KV ~98% 下引擎以 0 tok/s 挂起(Running: 10,Waiting: 23),PG ID 3 上的 NCCL _ALLGATHER_BASE 在 600 秒后超时(last started work: -1)→ VllmWorker-1 死亡(worker_crash:8)→ EngineDead / ProfileAborted。与 tip 860c1ccf 的 c48(batch_keys=2)签名相同;仅将 batch_keys 降为 1 未能阻止挂起。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、KV_GATHER=0、关闭 cumem。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Restore Mooncake load_async=true on B300 DSXE (keep max_load_batch_keys=1). Tip ea88d652 fail-fast run 36935627690 canary c1 (job 110615439146, Slurm 6679, gpu-00) reached health OK then aborted warmup: WorkerProc AssertionError in mooncake/store/worker.py get_finished — 'load_async must be True for better performance' — then EngineCore Internal Server Error / ProfileAborted. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive), not runtime ALLGATHER. Tip 670fb9f0 disable load_async is incompatible with this vLLM Mooncake connector assert; restore stock true." + - "在 B300 DSXE 上恢复 Mooncake load_async=true(保留 max_load_batch_keys=1)。tip ea88d652 fail-fast 运行 36935627690 的 canary c1(job 110615439146,Slurm 6679,gpu-00)健康检查通过后 warmup 中止:WorkerProc 在 mooncake/store/worker.py get_finished 断言 AssertionError: load_async must be True for better performance,随后 EngineCore Internal Server Error / ProfileAborted。ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报),非运行时 ALLGATHER。tip 670fb9f0 关闭 load_async 与该 vLLM Mooncake 连接器断言不兼容;恢复 stock true。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 1f837c46bf34b5c54123308a585dfb8a5afece74 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 1 Oct 2026 22:55:21 +0000 Subject: [PATCH 13/31] fix(kimik3-b300): disable direct DCP Q gather on DSXE MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit After load_async restore, tip d7176e6 would re-hit the Q_GATHER=1 + KV_GATHER=0 ALLGATHER hang; align Q gather off with KV gather. 在恢复 load_async 后,tip d7176e6 会重触 Q_GATHER=1 + KV_GATHER=0 的 ALLGATHER 挂起;将 Q gather 与 KV gather 一并关闭。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 6 +++++- inferencex-e2e/docs/configuration-procedures.md | 12 +++++++----- inferencex-e2e/docs/configuration-procedures_zh.md | 4 +++- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 24 insertions(+), 7 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 2e16b67889..e34b656a48 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -95,7 +95,11 @@ base: VLLM_USE_V2_MODEL_RUNNER: '1' # These default to auto; name them so the measured DCP a2a path runs. VLLM_USE_DIRECT_DCP_A2A: '1' - VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + # Tips 860c1ccf c48 / e51c58f5 c32 hung in DCP PYNCCL _ALLGATHER_BASE + # under ~98-100% GPU KV with Q_GATHER=1 + KV_GATHER=0 mixed; align Q + # gather off with KV gather (MI355X-style). load_async must stay true + # (tip ea88d652 AssertionError). + VLLM_USE_DIRECT_DCP_Q_GATHER: '0' # Prior tip c48 hit direct DCP kv-gather multimem timeout then GPU memcpy # failure; disable direct KV gather (same as GB200 DCP16 / MI355X lanes). VLLM_USE_DIRECT_DCP_KV_GATHER: '0' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 86f8514932..8a25d4853b 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -346,11 +346,13 @@ async loads and clean Mooncake metrics; `last started work: -1`. Tip ea88d652 canary c1 crashed with Mooncake `AssertionError: load_async must be True for better performance` when `load_async` was set false, so restore the required stock true), but do not enable `compact_group_io` on this DSXE single-rail path -(it storm-failed ~25 MiB compact-group puts at c8). Direct DCP KV gather is off -(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`). Do not set `MC_MAX_MR_SIZE` here: with 4GiB -every rank hit `register_buffer failed ... -600` on the ~40 GiB KV region and -stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips -registered cleanly. Keep `enable-cumem-allocator` off on this path. +(it storm-failed ~25 MiB compact-group puts at c8). Direct DCP KV and Q gather +are off (`VLLM_USE_DIRECT_DCP_KV_GATHER=0`, `VLLM_USE_DIRECT_DCP_Q_GATHER=0`); +prior tips with Q_GATHER=1 + KV_GATHER=0 hung in DCP PYNCCL `_ALLGATHER_BASE` +under ~98–100% GPU KV. Do not set `MC_MAX_MR_SIZE` here: with 4GiB every rank +hit `register_buffer failed ... -600` on the ~40 GiB KV region and stormed +`AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips registered +cleanly. Keep `enable-cumem-allocator` off on this path. The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 89f1d3235f..27d3073347 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -320,7 +320,9 @@ Mooncake 指标干净时,于约 98–100% GPU KV 下挂起在 DCP PYNCCL `AssertionError: load_async must be True for better performance`,故恢复必需的 stock true),但不要在该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。 -Direct DCP KV gather 关闭(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`)。不要在此设置 +Direct DCP KV 与 Q gather 均关闭(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`、 +`VLLM_USE_DIRECT_DCP_Q_GATHER=0`);此前在 Q_GATHER=1 + KV_GATHER=0 下 tip 于约 +98–100% GPU KV 挂起在 DCP PYNCCL `_ALLGATHER_BASE`。不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 `AddressNotRegistered` TRANSFER_FAIL (c2/c32);加入该变量之前的 tip 注册正常。此路径保持关闭 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9fabbf44fb..5cb68131b7 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9291,3 +9291,12 @@ - "Restore Mooncake load_async=true on B300 DSXE (keep max_load_batch_keys=1). Tip ea88d652 fail-fast run 36935627690 canary c1 (job 110615439146, Slurm 6679, gpu-00) reached health OK then aborted warmup: WorkerProc AssertionError in mooncake/store/worker.py get_finished — 'load_async must be True for better performance' — then EngineCore Internal Server Error / ProfileAborted. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive), not runtime ALLGATHER. Tip 670fb9f0 disable load_async is incompatible with this vLLM Mooncake connector assert; restore stock true." - "在 B300 DSXE 上恢复 Mooncake load_async=true(保留 max_load_batch_keys=1)。tip ea88d652 fail-fast 运行 36935627690 的 canary c1(job 110615439146,Slurm 6679,gpu-00)健康检查通过后 warmup 中止:WorkerProc 在 mooncake/store/worker.py get_finished 断言 AssertionError: load_async must be True for better performance,随后 EngineCore Internal Server Error / ProfileAborted。ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报),非运行时 ALLGATHER。tip 670fb9f0 关闭 load_async 与该 vLLM Mooncake 连接器断言不兼容;恢复 stock true。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Set VLLM_USE_DIRECT_DCP_Q_GATHER=0 on B300 DSXE (keep load_async=true, max_load_batch_keys=1, KV_GATHER=0). After tip ea88d652b canary c1 AssertionError forced restoring load_async, tip d7176e6 would otherwise re-run the Q_GATHER=1 + KV_GATHER=0 mix that hung tips 860c1ccf c48 / e51c58f5 c32 in DCP PYNCCL _ALLGATHER_BASE under ~98-100% GPU KV (last started work: -1; Mooncake metrics clean). Align Q gather off with KV gather (MI355X-style). Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上将 VLLM_USE_DIRECT_DCP_Q_GATHER 设为 0(保留 load_async=true、max_load_batch_keys=1、KV_GATHER=0)。tip ea88d652b canary c1 的 AssertionError 已强制恢复 load_async 后,tip d7176e6 否则会重跑 tip 860c1ccf c48 / e51c58f5 c32 在约 98–100% GPU KV 下挂起于 DCP PYNCCL _ALLGATHER_BASE 的 Q_GATHER=1 + KV_GATHER=0 混用路径(last started work: -1;Mooncake 指标干净)。将 Q gather 与 KV gather 一并关闭(对齐 MI355X)。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 59bfac1632738246252bed57ecf59117653dec0a Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 2 Oct 2026 01:11:39 +0000 Subject: [PATCH 14/31] fix(kimik3-b300): restore DCP Q gather; disable A2A MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip 1f837c46 eval c8 hung after dcp:0 then failed EP ncclCommInitRank with Q_GATHER=0; restore Q gather and probe A2A=0 for prior ALLGATHER hangs. 将 tip 1f837c46 eval c8 在 Q_GATHER=0 下 dcp:0 后挂起并 EP ncclCommInitRank 失败对应为恢复 Q gather,并以 A2A=0 探测此前 ALLGATHER 挂起。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 14 +++++++------- inferencex-e2e/docs/configuration-procedures.md | 15 ++++++++------- .../docs/configuration-procedures_zh.md | 14 +++++++------- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 31 insertions(+), 21 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index e34b656a48..d799a9992c 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -93,13 +93,13 @@ base: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' VLLM_USE_V2_MODEL_RUNNER: '1' - # These default to auto; name them so the measured DCP a2a path runs. - VLLM_USE_DIRECT_DCP_A2A: '1' - # Tips 860c1ccf c48 / e51c58f5 c32 hung in DCP PYNCCL _ALLGATHER_BASE - # under ~98-100% GPU KV with Q_GATHER=1 + KV_GATHER=0 mixed; align Q - # gather off with KV gather (MI355X-style). load_async must stay true - # (tip ea88d652 AssertionError). - VLLM_USE_DIRECT_DCP_Q_GATHER: '0' + # Tip 1f837c46 eval-only c8 hung ~8.5m after dcp:0 then failed EP + # PyNccl ncclCommInitRank (remote process exited) with Q_GATHER=0; + # restore Q gather. Disable direct A2A instead as the next probe for + # prior DCP PYNCCL _ALLGATHER_BASE hangs under ~98% GPU KV + # (tips 860c1ccf/e51c58f5). load_async must stay true. + VLLM_USE_DIRECT_DCP_A2A: '0' + VLLM_USE_DIRECT_DCP_Q_GATHER: '1' # Prior tip c48 hit direct DCP kv-gather multimem timeout then GPU memcpy # failure; disable direct KV gather (same as GB200 DCP16 / MI355X lanes). VLLM_USE_DIRECT_DCP_KV_GATHER: '0' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 8a25d4853b..6c921129fc 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -346,13 +346,14 @@ async loads and clean Mooncake metrics; `last started work: -1`. Tip ea88d652 canary c1 crashed with Mooncake `AssertionError: load_async must be True for better performance` when `load_async` was set false, so restore the required stock true), but do not enable `compact_group_io` on this DSXE single-rail path -(it storm-failed ~25 MiB compact-group puts at c8). Direct DCP KV and Q gather -are off (`VLLM_USE_DIRECT_DCP_KV_GATHER=0`, `VLLM_USE_DIRECT_DCP_Q_GATHER=0`); -prior tips with Q_GATHER=1 + KV_GATHER=0 hung in DCP PYNCCL `_ALLGATHER_BASE` -under ~98–100% GPU KV. Do not set `MC_MAX_MR_SIZE` here: with 4GiB every rank -hit `register_buffer failed ... -600` on the ~40 GiB KV region and stormed -`AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips registered -cleanly. Keep `enable-cumem-allocator` off on this path. +(it storm-failed ~25 MiB compact-group puts at c8). Direct DCP KV gather is off +(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`) and direct DCP A2A is off +(`VLLM_USE_DIRECT_DCP_A2A=0`); keep `VLLM_USE_DIRECT_DCP_Q_GATHER=1` (tip +1f837c46 eval-only c8 hung ~8.5m after `dcp:0` then failed EP +`ncclCommInitRank` with Q gather off). Do not set `MC_MAX_MR_SIZE` here: with +4GiB every rank hit `register_buffer failed ... -600` on the ~40 GiB KV region +and stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips +registered cleanly. Keep `enable-cumem-allocator` off on this path. The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 27d3073347..21dffde16c 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -320,13 +320,13 @@ Mooncake 指标干净时,于约 98–100% GPU KV 下挂起在 DCP PYNCCL `AssertionError: load_async must be True for better performance`,故恢复必需的 stock true),但不要在该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。 -Direct DCP KV 与 Q gather 均关闭(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`、 -`VLLM_USE_DIRECT_DCP_Q_GATHER=0`);此前在 Q_GATHER=1 + KV_GATHER=0 下 tip 于约 -98–100% GPU KV 挂起在 DCP PYNCCL `_ALLGATHER_BASE`。不要在此设置 -`MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 -`register_buffer failed ... -600`,并引发 `AddressNotRegistered` TRANSFER_FAIL -(c2/c32);加入该变量之前的 tip 注册正常。此路径保持关闭 -`enable-cumem-allocator`。 +Direct DCP KV gather 关闭(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`),Direct DCP A2A +关闭(`VLLM_USE_DIRECT_DCP_A2A=0`);保持 `VLLM_USE_DIRECT_DCP_Q_GATHER=1` +(tip 1f837c46 的 eval-only c8 在关闭 Q gather 后于 `dcp:0` 之后挂起约 8.5 +分钟,随后 EP `ncclCommInitRank` 失败)。不要在此设置 `MC_MAX_MR_SIZE`:设为 +4GiB 时各 rank 对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 +`AddressNotRegistered` TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。 +此路径保持关闭 `enable-cumem-allocator`。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 5cb68131b7..d077ff79cf 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9300,3 +9300,12 @@ - "Set VLLM_USE_DIRECT_DCP_Q_GATHER=0 on B300 DSXE (keep load_async=true, max_load_batch_keys=1, KV_GATHER=0). After tip ea88d652b canary c1 AssertionError forced restoring load_async, tip d7176e6 would otherwise re-run the Q_GATHER=1 + KV_GATHER=0 mix that hung tips 860c1ccf c48 / e51c58f5 c32 in DCP PYNCCL _ALLGATHER_BASE under ~98-100% GPU KV (last started work: -1; Mooncake metrics clean). Align Q gather off with KV gather (MI355X-style). Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." - "在 B300 DSXE 上将 VLLM_USE_DIRECT_DCP_Q_GATHER 设为 0(保留 load_async=true、max_load_batch_keys=1、KV_GATHER=0)。tip ea88d652b canary c1 的 AssertionError 已强制恢复 load_async 后,tip d7176e6 否则会重跑 tip 860c1ccf c48 / e51c58f5 c32 在约 98–100% GPU KV 下挂起于 DCP PYNCCL _ALLGATHER_BASE 的 Q_GATHER=1 + KV_GATHER=0 混用路径(last started work: -1;Mooncake 指标干净)。将 Q gather 与 KV gather 一并关闭(对齐 MI355X)。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Restore VLLM_USE_DIRECT_DCP_Q_GATHER=1 and set VLLM_USE_DIRECT_DCP_A2A=0 on B300 DSXE. Tip 1f837c46 fail-fast run 36937918258 eval-only c8 (job 110646326761, Slurm 6692, gpu-09) had a real rail (Mooncake rail: ibp198s0f0 / Patched) then hung ~8.5m after logging dcp:0 PYNCCL and failed WorkerProc init on EP group PyNcclCommunicator ncclCommInitRank with RuntimeError: NCCL error: remote process exited or there was a network error → Engine core initialization failed. Same tip canary c1 completed ep:0 in ~2s; Q_GATHER=0 is the new tip knob vs prior green evals. Keep load_async=true, max_load_batch_keys=1, KV_GATHER=0, no compact_group_io, no MC_MAX_MR_SIZE, cumem off. A2A=0 is the next probe for prior high-conc DCP PYNCCL _ALLGATHER_BASE hangs." + - "在 B300 DSXE 上恢复 VLLM_USE_DIRECT_DCP_Q_GATHER=1,并将 VLLM_USE_DIRECT_DCP_A2A 设为 0。tip 1f837c46 fail-fast 运行 36937918258 的 eval-only c8(job 110646326761,Slurm 6692,gpu-09)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),在记录 dcp:0 PYNCCL 后挂起约 8.5 分钟,随后 EP 组 PyNcclCommunicator ncclCommInitRank 报 RuntimeError: NCCL error: remote process exited or there was a network error → Engine core initialization failed。同 tip 的 canary c1 约 2 秒内完成 ep:0;相对此前全绿 eval,新 tip 旋钮为 Q_GATHER=0。保留 load_async=true、max_load_batch_keys=1、KV_GATHER=0、不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。A2A=0 作为此前高并发 DCP PYNCCL _ALLGATHER_BASE 挂起的下一探针。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 21dac229766172ae1960bdf9da76f111b0aa6ed6 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 2 Oct 2026 01:39:19 +0000 Subject: [PATCH 15/31] fix(changelog): restore dsv41flash-fp4-mi355x entry from main MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit check-changelog rejected a deletion of the #3571 config-keys line lost during the main merge; restore that append-only entry intact. check-changelog 拒绝 main 合入时丢失的 #3571 config-keys 行删除;完整恢复该 append-only 条目。 Co-authored-by: Wenyao Gao --- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index d077ff79cf..3a1a925e7d 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9265,6 +9265,15 @@ - "The preset supplies a 900000 ms TCP user timeout when unset; explicit AIPERF environment settings take precedence over preset defaults." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3646 +- config-keys: + - dsv41flash-fp4-mi355x-vllm-agentic-dspark + description: + - "Repin to vllm/vllm-openai-rocm:nightly-rocm100-ac9126e58aa7bbab1856ba6593ba4d5003fea516, which includes vllm-project/vllm#58671, vllm-project/vllm#58655 and vllm-project/vllm#53492." + - "Set VLLM_ROCM_USE_AITER_TRITON_SPARSE_MLA=True to enable the vllm-project/vllm#53492 gfx950 Gluon sparse-MLA kernel." + - "Add --attention-config '{\"indexer_kv_dtype\":\"mxfp4\",\"indexer_sparse_logits\":true}' and --block-size 128 to enable the vllm-project/vllm#58671 ROCm paged MXFP4 sparse-logits indexer, replacing the dense fp8 indexer path. A live A/B test (TP2 c16, matched 900s window, vllm-project/vllm#58208 reverted via vllm-project/vllm#59125 so the dense fallback doesn't crash) measured +14.7/+14.9% p50/p90 interactivity and -9.5/-10.8% p50/p90 e2e latency over the dense path, with throughput/GPU unchanged." + - "Drop c128 from both TP2 and TP4. Neither c128 point was on the Pareto frontier in #3555's run 36528242520: TP2 c64 dominated both (P90 E2EL 58 s against 185 s and 85 s, at 111k against 85k and 79k total tok/s/GPU)." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3571 + - config-keys: - kimik3-fp4-b300-vllm-agentic-dspark scenario-type: From bed9f1ce6cf91fd151f24d65d6ada91747b53cdc Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 2 Oct 2026 04:10:01 +0000 Subject: [PATCH 16/31] fix(kimik3-b300): lower high-conc gpu-memory-utilization to 0.85 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cut CONC 24+ util from 0.92/0.9 to 0.85 after tip b70e4260a c48 serve-time flashinfer FP4 MoE CUDA OOM with KV over-allocated under ESTIMATE_CUDAGRAPHS=0. 将 CONC 24+ 的 gpu-memory-utilization 从 0.92/0.9 降至 0.85,以修复 tip b70e4260a c48 在 ESTIMATE_CUDAGRAPHS=0 下 KV 过量分配导致的 flashinfer FP4 MoE 服务期 CUDA OOM。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 20 +++++++++++-------- .../docs/configuration-procedures.md | 7 ++++++- .../docs/configuration-procedures_zh.md | 6 +++++- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 32 insertions(+), 10 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index d799a9992c..39201b5ae4 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -135,8 +135,12 @@ base: KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2249' -# One variant per point. Admission is 2x CONC; CONC 56 and 70 keep more memory -# headroom. Graphs capture (1 + drafts) x 1..min(2x CONC, 128) tokens, then the +# One variant per point. Admission is 2x CONC. CONC 24+ use gpu-memory-utilization +# 0.85: with VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 +# (Slurm 6737) over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once +# CUDA graphs are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB +# needed, ~2.3 GiB free) ~4m after Application startup. Keep 0.92 on speculative +# c1–c16. Graphs capture (1 + drafts) x 1..min(2x CONC, 128) tokens, then the # larger powers of two to 8192. Throughput runs switch DSpark to synthetic # rejection at the golden acceptance length; points above CONC 16 do not draft # and keep the matrix's mtp label. @@ -205,7 +209,7 @@ override_c24: agg: args: max-num-seqs: 48 - gpu-memory-utilization: 0.92 + gpu-memory-utilization: 0.85 compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: @@ -217,7 +221,7 @@ override_c32: agg: args: max-num-seqs: 64 - gpu-memory-utilization: 0.92 + gpu-memory-utilization: 0.85 compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: @@ -229,7 +233,7 @@ override_c40: agg: args: max-num-seqs: 80 - gpu-memory-utilization: 0.92 + gpu-memory-utilization: 0.85 compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,128,256,512,1024,2048,4096,8192]}' benchmark: env: @@ -241,7 +245,7 @@ override_c48: agg: args: max-num-seqs: 96 - gpu-memory-utilization: 0.92 + gpu-memory-utilization: 0.85 compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,128,256,512,1024,2048,4096,8192]}' benchmark: env: @@ -253,7 +257,7 @@ override_c56: agg: args: max-num-seqs: 112 - gpu-memory-utilization: 0.9 + gpu-memory-utilization: 0.85 compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,128,256,512,1024,2048,4096,8192]}' benchmark: env: @@ -265,7 +269,7 @@ override_c70: agg: args: max-num-seqs: 140 - gpu-memory-utilization: 0.9 + gpu-memory-utilization: 0.85 compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119,120,121,122,123,124,125,126,127,128,256,512,1024,2048,4096,8192]}' benchmark: env: diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 8abfeddbbf..9aa4d28447 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -353,7 +353,12 @@ stock true), but do not enable `compact_group_io` on this DSXE single-rail path `ncclCommInitRank` with Q gather off). Do not set `MC_MAX_MR_SIZE` here: with 4GiB every rank hit `register_buffer failed ... -600` on the ~40 GiB KV region and stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips -registered cleanly. Keep `enable-cumem-allocator` off on this path. +registered cleanly. Keep `enable-cumem-allocator` off on this path. Keep +`gpu-memory-utilization` at 0.85 for CONC 24+ (tip b70e4260a c48 reached +Application startup with KV 45.2 GiB at 0.92 under +`VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0`, then OOMed in flashinfer FP4 MoE +`prepare_moe` allocating ~2.89 GiB with ~2.3 GiB free; vLLM suggested ~36.78 GiB +KV once CUDA graphs are counted). The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 4907f93236..6df06e0414 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -326,7 +326,11 @@ Direct DCP KV gather 关闭(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`),Direct DCP 分钟,随后 EP `ncclCommInitRank` 失败)。不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 `AddressNotRegistered` TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。 -此路径保持关闭 `enable-cumem-allocator`。 +此路径保持关闭 `enable-cumem-allocator`。CONC 24+ 将 `gpu-memory-utilization` +保持为 0.85(tip b70e4260a 的 c48 在 `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0` +与 0.92 下以 45.2 GiB KV 完成 Application startup,随后在 flashinfer FP4 MoE +`prepare_moe` 申请约 2.89 GiB 时仅剩约 2.3 GiB 空闲而 OOM;vLLM 在计入 CUDA +graph 后建议约 36.78 GiB KV)。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 2b2a7dfe28..32107bb33e 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9315,3 +9315,12 @@ - "Restore VLLM_USE_DIRECT_DCP_Q_GATHER=1 and set VLLM_USE_DIRECT_DCP_A2A=0 on B300 DSXE. Tip 1f837c46 fail-fast run 36937918258 eval-only c8 (job 110646326761, Slurm 6692, gpu-09) had a real rail (Mooncake rail: ibp198s0f0 / Patched) then hung ~8.5m after logging dcp:0 PYNCCL and failed WorkerProc init on EP group PyNcclCommunicator ncclCommInitRank with RuntimeError: NCCL error: remote process exited or there was a network error → Engine core initialization failed. Same tip canary c1 completed ep:0 in ~2s; Q_GATHER=0 is the new tip knob vs prior green evals. Keep load_async=true, max_load_batch_keys=1, KV_GATHER=0, no compact_group_io, no MC_MAX_MR_SIZE, cumem off. A2A=0 is the next probe for prior high-conc DCP PYNCCL _ALLGATHER_BASE hangs." - "在 B300 DSXE 上恢复 VLLM_USE_DIRECT_DCP_Q_GATHER=1,并将 VLLM_USE_DIRECT_DCP_A2A 设为 0。tip 1f837c46 fail-fast 运行 36937918258 的 eval-only c8(job 110646326761,Slurm 6692,gpu-09)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),在记录 dcp:0 PYNCCL 后挂起约 8.5 分钟,随后 EP 组 PyNcclCommunicator ncclCommInitRank 报 RuntimeError: NCCL error: remote process exited or there was a network error → Engine core initialization failed。同 tip 的 canary c1 约 2 秒内完成 ep:0;相对此前全绿 eval,新 tip 旋钮为 Q_GATHER=0。保留 load_async=true、max_load_batch_keys=1、KV_GATHER=0、不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。A2A=0 作为此前高并发 DCP PYNCCL _ALLGATHER_BASE 挂起的下一探针。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Lower gpu-memory-utilization to 0.85 on B300 DSXE CONC 24+ (keep load_async=true, max_load_batch_keys=1, KV_GATHER=0, Q_GATHER=1, A2A=0, ESTIMATE_CUDAGRAPHS=0). Tip b70e4260a fail-fast run 36952119306 agentic c48 (job 110686724979, Slurm 6737, gpu-05) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name), Application startup complete, then ~4m into AgentX serve CUDA OOM in flashinfer FP4 MoE prepare_moe (tried 2.89 GiB, ~2.3 GiB free; GPU KV 45.2 GiB at util 0.92; vLLM suggested ~36.78 GiB once CUDA graphs counted) → EngineDead / ProfileAborted. Zero register_buffer -600 / TRANSFER_FAIL / ALLGATHER. GHA 'engine never ready' was wrapper noise after serve-time OOM. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "将 B300 DSXE 上 CONC 24+ 的 gpu-memory-utilization 降至 0.85(保留 load_async=true、max_load_batch_keys=1、KV_GATHER=0、Q_GATHER=1、A2A=0、ESTIMATE_CUDAGRAPHS=0)。tip b70e4260a fail-fast 运行 36952119306 的 agentic c48(job 110686724979,Slurm 6737,gpu-05)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,随后 AgentX 服务约 4 分钟时在 flashinfer FP4 MoE prepare_moe 发生 CUDA OOM(申请 2.89 GiB,空闲约 2.3 GiB;util 0.92 下 GPU KV 45.2 GiB;计入 CUDA graph 后 vLLM 建议约 36.78 GiB)→ EngineDead / ProfileAborted。无 register_buffer -600 / TRANSFER_FAIL / ALLGATHER。GHA 的 engine never ready 为服务期 OOM 后的包装噪声。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 68cdcc58a76566e9dd04c07ff1be69f73fc2b4f2 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 2 Oct 2026 07:03:26 +0000 Subject: [PATCH 17/31] fix(kimik3-b300): re-enable direct DCP KV gather MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip bed9f1ce c40 hung in PyNCCL kv_gather _ALLGATHER_BASE (last started work: -1) with KV_GATHER=0; A2A=0 and util 0.85 did not clear it. Switch KV gather to the direct path already used for Q gather / GB300 DCP8. tip bed9f1ce 的 c40 在 KV_GATHER=0 下挂起于 PyNCCL kv_gather _ALLGATHER_BASE (last started work: -1);A2A=0 与 util 0.85 未能消除。将 KV gather 切回 Q gather / GB300 DCP8 已使用的 direct 路径。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 15 +++++++++------ inferencex-e2e/docs/configuration-procedures.md | 15 ++++++++------- .../docs/configuration-procedures_zh.md | 10 ++++++---- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 32 insertions(+), 17 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 39201b5ae4..fbcb3d9a9f 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -95,14 +95,17 @@ base: VLLM_USE_V2_MODEL_RUNNER: '1' # Tip 1f837c46 eval-only c8 hung ~8.5m after dcp:0 then failed EP # PyNccl ncclCommInitRank (remote process exited) with Q_GATHER=0; - # restore Q gather. Disable direct A2A instead as the next probe for - # prior DCP PYNCCL _ALLGATHER_BASE hangs under ~98% GPU KV - # (tips 860c1ccf/e51c58f5). load_async must stay true. + # restore Q gather. Keep A2A off (dcp allreduce stays PYNCCL). load_async + # must stay true. VLLM_USE_DIRECT_DCP_A2A: '0' VLLM_USE_DIRECT_DCP_Q_GATHER: '1' - # Prior tip c48 hit direct DCP kv-gather multimem timeout then GPU memcpy - # failure; disable direct KV gather (same as GB200 DCP16 / MI355X lanes). - VLLM_USE_DIRECT_DCP_KV_GATHER: '0' + # Tip bed9f1ce c40 (Slurm 6765): with KV_GATHER=0 the hang is PyNCCL + # kv_gather _ALLGATHER_BASE (dcp.py:1413; last started work: -1) after + # ~42m serve / GPU KV ~86-100%, Mooncake failed_keys=0. A2A=0 and util + # 0.85 did not clear it. Re-enable direct KV gather (GB300 DCP8 / + # B200 Mooncake stock); prior multimem/memcpy with KV_GATHER=1 was under + # util 0.92 before the 0.85 headroom cut. + VLLM_USE_DIRECT_DCP_KV_GATHER: '1' VLLM_ENGINE_READY_TIMEOUT_S: '3600' VLLM_RPC_TIMEOUT: '600000' # Mooncake loads can block inside execute_model past the 300s default; diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 9aa4d28447..4dbddef3f1 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -346,14 +346,15 @@ async loads and clean Mooncake metrics; `last started work: -1`. Tip ea88d652 canary c1 crashed with Mooncake `AssertionError: load_async must be True for better performance` when `load_async` was set false, so restore the required stock true), but do not enable `compact_group_io` on this DSXE single-rail path -(it storm-failed ~25 MiB compact-group puts at c8). Direct DCP KV gather is off -(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`) and direct DCP A2A is off -(`VLLM_USE_DIRECT_DCP_A2A=0`); keep `VLLM_USE_DIRECT_DCP_Q_GATHER=1` (tip +(it storm-failed ~25 MiB compact-group puts at c8). Direct DCP A2A is off (`VLLM_USE_DIRECT_DCP_A2A=0`); keep +`VLLM_USE_DIRECT_DCP_Q_GATHER=1` and `VLLM_USE_DIRECT_DCP_KV_GATHER=1` (tip 1f837c46 eval-only c8 hung ~8.5m after `dcp:0` then failed EP -`ncclCommInitRank` with Q gather off). Do not set `MC_MAX_MR_SIZE` here: with -4GiB every rank hit `register_buffer failed ... -600` on the ~40 GiB KV region -and stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips -registered cleanly. Keep `enable-cumem-allocator` off on this path. Keep +`ncclCommInitRank` with Q gather off; tip bed9f1ce c40 hung in PyNCCL +`kv_gather` `_ALLGATHER_BASE` with `last started work: -1` when KV gather was +off — A2A=0 and util 0.85 did not clear it). Do not set `MC_MAX_MR_SIZE` here: +with 4GiB every rank hit `register_buffer failed ... -600` on the ~40 GiB KV +region and stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` +tips registered cleanly. Keep `enable-cumem-allocator` off on this path. Keep `gpu-memory-utilization` at 0.85 for CONC 24+ (tip b70e4260a c48 reached Application startup with KV 45.2 GiB at 0.92 under `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0`, then OOMed in flashinfer FP4 MoE diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 6df06e0414..3548975194 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -320,11 +320,13 @@ Mooncake 指标干净时,于约 98–100% GPU KV 下挂起在 DCP PYNCCL `AssertionError: load_async must be True for better performance`,故恢复必需的 stock true),但不要在该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。 -Direct DCP KV gather 关闭(`VLLM_USE_DIRECT_DCP_KV_GATHER=0`),Direct DCP A2A -关闭(`VLLM_USE_DIRECT_DCP_A2A=0`);保持 `VLLM_USE_DIRECT_DCP_Q_GATHER=1` +Direct DCP A2A 关闭(`VLLM_USE_DIRECT_DCP_A2A=0`);保持 +`VLLM_USE_DIRECT_DCP_Q_GATHER=1` 与 `VLLM_USE_DIRECT_DCP_KV_GATHER=1` (tip 1f837c46 的 eval-only c8 在关闭 Q gather 后于 `dcp:0` 之后挂起约 8.5 -分钟,随后 EP `ncclCommInitRank` 失败)。不要在此设置 `MC_MAX_MR_SIZE`:设为 -4GiB 时各 rank 对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 +分钟,随后 EP `ncclCommInitRank` 失败;tip bed9f1ce 的 c40 在关闭 KV gather +时于 PyNCCL `kv_gather` `_ALLGATHER_BASE` 挂起,`last started work: -1`,A2A=0 +与 util 0.85 未能消除)。不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank +对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 `AddressNotRegistered` TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。 此路径保持关闭 `enable-cumem-allocator`。CONC 24+ 将 `gpu-memory-utilization` 保持为 0.85(tip b70e4260a 的 c48 在 `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0` diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 32107bb33e..f92064844c 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9324,3 +9324,12 @@ - "Lower gpu-memory-utilization to 0.85 on B300 DSXE CONC 24+ (keep load_async=true, max_load_batch_keys=1, KV_GATHER=0, Q_GATHER=1, A2A=0, ESTIMATE_CUDAGRAPHS=0). Tip b70e4260a fail-fast run 36952119306 agentic c48 (job 110686724979, Slurm 6737, gpu-05) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name), Application startup complete, then ~4m into AgentX serve CUDA OOM in flashinfer FP4 MoE prepare_moe (tried 2.89 GiB, ~2.3 GiB free; GPU KV 45.2 GiB at util 0.92; vLLM suggested ~36.78 GiB once CUDA graphs counted) → EngineDead / ProfileAborted. Zero register_buffer -600 / TRANSFER_FAIL / ALLGATHER. GHA 'engine never ready' was wrapper noise after serve-time OOM. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." - "将 B300 DSXE 上 CONC 24+ 的 gpu-memory-utilization 降至 0.85(保留 load_async=true、max_load_batch_keys=1、KV_GATHER=0、Q_GATHER=1、A2A=0、ESTIMATE_CUDAGRAPHS=0)。tip b70e4260a fail-fast 运行 36952119306 的 agentic c48(job 110686724979,Slurm 6737,gpu-05)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,随后 AgentX 服务约 4 分钟时在 flashinfer FP4 MoE prepare_moe 发生 CUDA OOM(申请 2.89 GiB,空闲约 2.3 GiB;util 0.92 下 GPU KV 45.2 GiB;计入 CUDA graph 后 vLLM 建议约 36.78 GiB)→ EngineDead / ProfileAborted。无 register_buffer -600 / TRANSFER_FAIL / ALLGATHER。GHA 的 engine never ready 为服务期 OOM 后的包装噪声。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Re-enable VLLM_USE_DIRECT_DCP_KV_GATHER=1 on B300 DSXE (keep A2A=0, Q_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 24+ util 0.85). Tip bed9f1ce fail-fast run 36963354550 agentic c40 (job 110720603462, Slurm 6765, gpu-13) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, Mooncake failed_keys=0; after ~42m serve under GPU KV ~86-100% hung at 0 tok/s (Running: 29, Waiting: 10) then PG ID 3 NCCL _ALLGATHER_BASE in dcp.py kv_gather timed out 600s (last started work: -1) → VllmWorker-3 died (worker_crash:8) → EngineDead / ProfileAborted. Stack is the PyNCCL path forced by KV_GATHER=0; Q gather already logged direct symmetric-memory. Prior multimem/memcpy with KV_GATHER=1 was under util 0.92 before the 0.85 headroom cut. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上重新启用 VLLM_USE_DIRECT_DCP_KV_GATHER=1(保留 A2A=0、Q_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 24+ util 0.85)。tip bed9f1ce fail-fast 运行 36963354550 的 agentic c40(job 110720603462,Slurm 6765,gpu-13)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,Mooncake failed_keys=0;约 42 分钟服务后在 GPU KV ~86–100% 下以 0 tok/s 挂起(Running: 29,Waiting: 10),随后 PG ID 3 上 dcp.py kv_gather 的 NCCL _ALLGATHER_BASE 超时 600 秒(last started work: -1)→ VllmWorker-3 死亡(worker_crash:8)→ EngineDead / ProfileAborted。堆栈为 KV_GATHER=0 强制的 PyNCCL 路径;Q gather 已记录使用 direct symmetric-memory。此前 KV_GATHER=1 的 multimem/memcpy 发生在 util 0.92、尚未降至 0.85 时。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 031de17bf5ee9f3206791ef1c98a55baab565c13 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 2 Oct 2026 13:05:40 +0000 Subject: [PATCH 18/31] fix(kimik3-b300): enable direct DCP A2A MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip 68cdcc58 c24 hung in PyNCCL ALLTOALL_BASE inside dcp_a2a_lse_reduce with A2A=0 while direct Q/KV gathers were already active. Enable direct A2A to match GB300 DCP8 stock; keep KV_GATHER=1. tip 68cdcc58 的 c24 在 A2A=0 且 Q/KV direct gather 已生效时,挂起于 dcp_a2a_lse_reduce 的 PyNCCL ALLTOALL_BASE。启用 direct A2A 以对齐 GB300 DCP8 stock;保留 KV_GATHER=1。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 17 +++++++------- .../docs/configuration-procedures.md | 20 ++++++++-------- .../docs/configuration-procedures_zh.md | 23 ++++++++++--------- inferencex-e2e/perf-changelog.yaml | 9 ++++++++ 4 files changed, 41 insertions(+), 28 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index fbcb3d9a9f..174b7fa894 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -94,17 +94,18 @@ base: VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' VLLM_USE_V2_MODEL_RUNNER: '1' # Tip 1f837c46 eval-only c8 hung ~8.5m after dcp:0 then failed EP - # PyNccl ncclCommInitRank (remote process exited) with Q_GATHER=0; - # restore Q gather. Keep A2A off (dcp allreduce stays PYNCCL). load_async - # must stay true. - VLLM_USE_DIRECT_DCP_A2A: '0' + # PyNccl ncclCommInitRank with Q_GATHER=0; keep Q gather on. Tip + # 68cdcc58 c24 (Slurm 6814): with A2A=0 the hang moved to PyNCCL + # ALLTOALL_BASE in dcp_a2a_lse_reduce (dcp.py:765; last started work: + # -1) after ~43m serve — direct Q/KV gathers were already active. + # Enable direct A2A (GB300 DCP8 / B200 stock). load_async must stay true. + VLLM_USE_DIRECT_DCP_A2A: '1' VLLM_USE_DIRECT_DCP_Q_GATHER: '1' # Tip bed9f1ce c40 (Slurm 6765): with KV_GATHER=0 the hang is PyNCCL # kv_gather _ALLGATHER_BASE (dcp.py:1413; last started work: -1) after - # ~42m serve / GPU KV ~86-100%, Mooncake failed_keys=0. A2A=0 and util - # 0.85 did not clear it. Re-enable direct KV gather (GB300 DCP8 / - # B200 Mooncake stock); prior multimem/memcpy with KV_GATHER=1 was under - # util 0.92 before the 0.85 headroom cut. + # ~42m serve / GPU KV ~86-100%, Mooncake failed_keys=0. Re-enable direct + # KV gather (GB300 DCP8 / B200 Mooncake stock); tip 68cdcc58 confirmed + # "Using direct symmetric-memory DCP chunked-context KV gather". VLLM_USE_DIRECT_DCP_KV_GATHER: '1' VLLM_ENGINE_READY_TIMEOUT_S: '3600' VLLM_RPC_TIMEOUT: '600000' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 4dbddef3f1..f42bc99cb7 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -346,15 +346,17 @@ async loads and clean Mooncake metrics; `last started work: -1`. Tip ea88d652 canary c1 crashed with Mooncake `AssertionError: load_async must be True for better performance` when `load_async` was set false, so restore the required stock true), but do not enable `compact_group_io` on this DSXE single-rail path -(it storm-failed ~25 MiB compact-group puts at c8). Direct DCP A2A is off (`VLLM_USE_DIRECT_DCP_A2A=0`); keep -`VLLM_USE_DIRECT_DCP_Q_GATHER=1` and `VLLM_USE_DIRECT_DCP_KV_GATHER=1` (tip -1f837c46 eval-only c8 hung ~8.5m after `dcp:0` then failed EP -`ncclCommInitRank` with Q gather off; tip bed9f1ce c40 hung in PyNCCL -`kv_gather` `_ALLGATHER_BASE` with `last started work: -1` when KV gather was -off — A2A=0 and util 0.85 did not clear it). Do not set `MC_MAX_MR_SIZE` here: -with 4GiB every rank hit `register_buffer failed ... -600` on the ~40 GiB KV -region and stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` -tips registered cleanly. Keep `enable-cumem-allocator` off on this path. Keep +(it storm-failed ~25 MiB compact-group puts at c8). Keep +`VLLM_USE_DIRECT_DCP_A2A=1`, `VLLM_USE_DIRECT_DCP_Q_GATHER=1`, and +`VLLM_USE_DIRECT_DCP_KV_GATHER=1` (tip 1f837c46 eval-only c8 hung ~8.5m after +`dcp:0` then failed EP `ncclCommInitRank` with Q gather off; tip bed9f1ce c40 +hung in PyNCCL `kv_gather` `_ALLGATHER_BASE` with `last started work: -1` when +KV gather was off; tip 68cdcc58 c24 then hung in PyNCCL `ALLTOALL_BASE` inside +`dcp_a2a_lse_reduce` with A2A off while direct Q/KV gathers were already active). +Do not set `MC_MAX_MR_SIZE` here: with 4GiB every rank hit +`register_buffer failed ... -600` on the ~40 GiB KV region and stormed +`AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips registered +cleanly. Keep `enable-cumem-allocator` off on this path. Keep `gpu-memory-utilization` at 0.85 for CONC 24+ (tip b70e4260a c48 reached Application startup with KV 45.2 GiB at 0.92 under `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0`, then OOMed in flashinfer FP4 MoE diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 3548975194..313e400eda 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -320,17 +320,18 @@ Mooncake 指标干净时,于约 98–100% GPU KV 下挂起在 DCP PYNCCL `AssertionError: load_async must be True for better performance`,故恢复必需的 stock true),但不要在该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。 -Direct DCP A2A 关闭(`VLLM_USE_DIRECT_DCP_A2A=0`);保持 -`VLLM_USE_DIRECT_DCP_Q_GATHER=1` 与 `VLLM_USE_DIRECT_DCP_KV_GATHER=1` -(tip 1f837c46 的 eval-only c8 在关闭 Q gather 后于 `dcp:0` 之后挂起约 8.5 -分钟,随后 EP `ncclCommInitRank` 失败;tip bed9f1ce 的 c40 在关闭 KV gather -时于 PyNCCL `kv_gather` `_ALLGATHER_BASE` 挂起,`last started work: -1`,A2A=0 -与 util 0.85 未能消除)。不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank -对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 -`AddressNotRegistered` TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。 -此路径保持关闭 `enable-cumem-allocator`。CONC 24+ 将 `gpu-memory-utilization` -保持为 0.85(tip b70e4260a 的 c48 在 `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0` -与 0.92 下以 45.2 GiB KV 完成 Application startup,随后在 flashinfer FP4 MoE +保持 `VLLM_USE_DIRECT_DCP_A2A=1`、`VLLM_USE_DIRECT_DCP_Q_GATHER=1` 与 +`VLLM_USE_DIRECT_DCP_KV_GATHER=1`(tip 1f837c46 的 eval-only c8 在关闭 Q gather +后于 `dcp:0` 之后挂起约 8.5 分钟,随后 EP `ncclCommInitRank` 失败;tip +bed9f1ce 的 c40 在关闭 KV gather 时于 PyNCCL `kv_gather` `_ALLGATHER_BASE` +挂起,`last started work: -1`;tip 68cdcc58 的 c24 在关闭 A2A 且 Q/KV direct +gather 已生效时,于 PyNCCL `ALLTOALL_BASE` / `dcp_a2a_lse_reduce` 挂起)。 +不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 +`register_buffer failed ... -600`,并引发 `AddressNotRegistered` +TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。此路径保持关闭 +`enable-cumem-allocator`。CONC 24+ 将 `gpu-memory-utilization` 保持为 0.85 +(tip b70e4260a 的 c48 在 `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0` 与 0.92 +下以 45.2 GiB KV 完成 Application startup,随后在 flashinfer FP4 MoE `prepare_moe` 申请约 2.89 GiB 时仅剩约 2.3 GiB 空闲而 OOM;vLLM 在计入 CUDA graph 后建议约 36.78 GiB KV)。 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index f92064844c..dc21a52f2c 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9333,3 +9333,12 @@ - "Re-enable VLLM_USE_DIRECT_DCP_KV_GATHER=1 on B300 DSXE (keep A2A=0, Q_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 24+ util 0.85). Tip bed9f1ce fail-fast run 36963354550 agentic c40 (job 110720603462, Slurm 6765, gpu-13) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, Mooncake failed_keys=0; after ~42m serve under GPU KV ~86-100% hung at 0 tok/s (Running: 29, Waiting: 10) then PG ID 3 NCCL _ALLGATHER_BASE in dcp.py kv_gather timed out 600s (last started work: -1) → VllmWorker-3 died (worker_crash:8) → EngineDead / ProfileAborted. Stack is the PyNCCL path forced by KV_GATHER=0; Q gather already logged direct symmetric-memory. Prior multimem/memcpy with KV_GATHER=1 was under util 0.92 before the 0.85 headroom cut. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." - "在 B300 DSXE 上重新启用 VLLM_USE_DIRECT_DCP_KV_GATHER=1(保留 A2A=0、Q_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 24+ util 0.85)。tip bed9f1ce fail-fast 运行 36963354550 的 agentic c40(job 110720603462,Slurm 6765,gpu-13)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,Mooncake failed_keys=0;约 42 分钟服务后在 GPU KV ~86–100% 下以 0 tok/s 挂起(Running: 29,Waiting: 10),随后 PG ID 3 上 dcp.py kv_gather 的 NCCL _ALLGATHER_BASE 超时 600 秒(last started work: -1)→ VllmWorker-3 死亡(worker_crash:8)→ EngineDead / ProfileAborted。堆栈为 KV_GATHER=0 强制的 PyNCCL 路径;Q gather 已记录使用 direct symmetric-memory。此前 KV_GATHER=1 的 multimem/memcpy 发生在 util 0.92、尚未降至 0.85 时。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Enable VLLM_USE_DIRECT_DCP_A2A=1 on B300 DSXE (keep Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 24+ util 0.85). Tip 68cdcc58 fail-fast run 36976589313 agentic c24 (job 110806435815, Slurm 6814, gpu-12) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, and logged both direct Q gather and direct chunked-context KV gather; Mooncake failed_keys=0. After ~43m serve hung at 0 tok/s (Running: 23, Waiting: 0, GPU KV ~68%) then PG ID 3 NCCL ALLTOALL_BASE in dcp.py dcp_a2a_lse_reduce timed out 600s (last started work: -1) → worker_crash:8 → EngineDead / ProfileAborted. Distinct from tip bed9f1ce c40 kv_gather _ALLGATHER_BASE (fixed by KV_GATHER=1). A2A=0 forced the PyNCCL ALLTOALL LSE-reduce path; enable direct A2A to match GB300 DCP8 / B200 stock. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上启用 VLLM_USE_DIRECT_DCP_A2A=1(保留 Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 24+ util 0.85)。tip 68cdcc58 fail-fast 运行 36976589313 的 agentic c24(job 110806435815,Slurm 6814,gpu-12)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,并已记录 direct Q gather 与 direct chunked-context KV gather;Mooncake failed_keys=0。约 43 分钟服务后以 0 tok/s 挂起(Running: 23,Waiting: 0,GPU KV ~68%),随后 PG ID 3 上 dcp.py dcp_a2a_lse_reduce 的 NCCL ALLTOALL_BASE 超时 600 秒(last started work: -1)→ worker_crash:8 → EngineDead / ProfileAborted。与 tip bed9f1ce c40 的 kv_gather _ALLGATHER_BASE(已由 KV_GATHER=1 修复)不同。A2A=0 强制走 PyNCCL ALLTOALL LSE-reduce 路径;启用 direct A2A 以对齐 GB300 DCP8 / B200 stock。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From a7c8fb8b7a3770f85e633bce94312b137bf31c61 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 2 Oct 2026 16:18:11 +0000 Subject: [PATCH 19/31] fix(kimik3-b300): cap high-conc max-num-seqs at 1x CONC MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cap CONC 48+ admission to 1x CONC after tip 031de17bf c56 hung in sample_tokens at ~99.7% GPU KV with all three direct DCP gathers on. 将 CONC 48+ 的 max-num-seqs 限制为 1×CONC:tip 031de17bf 的 c56 在三种 direct DCP gather 均已开启时于约 99.7% GPU KV 下挂起于 sample_tokens。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 38 +++++++++++-------- .../docs/configuration-procedures.md | 6 ++- .../docs/configuration-procedures_zh.md | 6 ++- inferencex-e2e/perf-changelog.yaml | 9 +++++ 4 files changed, 42 insertions(+), 17 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 174b7fa894..f043c912d7 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -139,15 +139,23 @@ base: KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2249' -# One variant per point. Admission is 2x CONC. CONC 24+ use gpu-memory-utilization -# 0.85: with VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 -# (Slurm 6737) over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once -# CUDA graphs are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB -# needed, ~2.3 GiB free) ~4m after Application startup. Keep 0.92 on speculative -# c1–c16. Graphs capture (1 + drafts) x 1..min(2x CONC, 128) tokens, then the -# larger powers of two to 8192. Throughput runs switch DSpark to synthetic -# rejection at the golden acceptance length; points above CONC 16 do not draft -# and keep the matrix's mtp label. +# One variant per point. Admission is 2x CONC through CONC 40; CONC 48+ use 1x +# CONC. Tip 031de17bf c56 (Slurm 6848) with A2A/Q/KV all direct, util 0.85, and +# max-num-seqs=112 packed GPU KV to ~99.7% (Running: 1, Waiting: 12, Deferred: +# 12) then hung workers for the full VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 — +# shm_broadcast starvation from 15:28, then TimeoutError: RPC call to +# sample_tokens timed out (no Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest +# nccl_error:16 was init-only ibv_query_port_speed WARN). Cap high-conc +# admission so prefix+deferred loads cannot pin the pool before decode. CONC +# 24+ use gpu-memory-utilization 0.85: with +# VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 (Slurm 6737) +# over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once CUDA graphs +# are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB needed, +# ~2.3 GiB free) ~4m after Application startup. Keep 0.92 on speculative +# c1–c16. Graphs capture every size up to max-num-seqs, then the larger powers +# of two to 8192. Throughput runs switch DSpark to synthetic rejection at the +# golden acceptance length; points above CONC 16 do not draft and keep the +# matrix's mtp label. override_c1: roles: agg: @@ -248,9 +256,9 @@ override_c48: roles: agg: args: - max-num-seqs: 96 + max-num-seqs: 48 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '48' @@ -260,9 +268,9 @@ override_c56: roles: agg: args: - max-num-seqs: 112 + max-num-seqs: 56 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '56' @@ -272,9 +280,9 @@ override_c70: roles: agg: args: - max-num-seqs: 140 + max-num-seqs: 70 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119,120,121,122,123,124,125,126,127,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '70' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index f42bc99cb7..49b8164b44 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -361,7 +361,11 @@ cleanly. Keep `enable-cumem-allocator` off on this path. Keep Application startup with KV 45.2 GiB at 0.92 under `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0`, then OOMed in flashinfer FP4 MoE `prepare_moe` allocating ~2.89 GiB with ~2.3 GiB free; vLLM suggested ~36.78 GiB -KV once CUDA graphs are counted). +KV once CUDA graphs are counted). Cap `max-num-seqs` at 1×CONC for CONC 48+ +(tip 031de17bf c56 with A2A/Q/KV all direct and util 0.85 packed GPU KV to +~99.7% under 2× admission, then hung workers through the 1800s +`sample_tokens` RPC timeout with no Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; +ingest `nccl_error:16` was init-only `ibv_query_port_speed` WARN). The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 313e400eda..ec2692b711 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -333,7 +333,11 @@ TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。此路 (tip b70e4260a 的 c48 在 `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0` 与 0.92 下以 45.2 GiB KV 完成 Application startup,随后在 flashinfer FP4 MoE `prepare_moe` 申请约 2.89 GiB 时仅剩约 2.3 GiB 空闲而 OOM;vLLM 在计入 CUDA -graph 后建议约 36.78 GiB KV)。 +graph 后建议约 36.78 GiB KV)。CONC 48+ 将 `max-num-seqs` 限制为 1×CONC +(tip 031de17bf 的 c56 在 A2A/Q/KV 均已 direct 且 util 0.85 时,2× 准入把 +GPU KV 堆到约 99.7%,随后 worker 挂满 1800 秒 `sample_tokens` RPC 超时,无 +Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 `nccl_error:16` 仅为 +初始化期 `ibv_query_port_speed` WARN)。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index dc21a52f2c..cd4e51a152 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9342,3 +9342,12 @@ - "Enable VLLM_USE_DIRECT_DCP_A2A=1 on B300 DSXE (keep Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 24+ util 0.85). Tip 68cdcc58 fail-fast run 36976589313 agentic c24 (job 110806435815, Slurm 6814, gpu-12) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, and logged both direct Q gather and direct chunked-context KV gather; Mooncake failed_keys=0. After ~43m serve hung at 0 tok/s (Running: 23, Waiting: 0, GPU KV ~68%) then PG ID 3 NCCL ALLTOALL_BASE in dcp.py dcp_a2a_lse_reduce timed out 600s (last started work: -1) → worker_crash:8 → EngineDead / ProfileAborted. Distinct from tip bed9f1ce c40 kv_gather _ALLGATHER_BASE (fixed by KV_GATHER=1). A2A=0 forced the PyNCCL ALLTOALL LSE-reduce path; enable direct A2A to match GB300 DCP8 / B200 stock. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." - "在 B300 DSXE 上启用 VLLM_USE_DIRECT_DCP_A2A=1(保留 Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 24+ util 0.85)。tip 68cdcc58 fail-fast 运行 36976589313 的 agentic c24(job 110806435815,Slurm 6814,gpu-12)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,并已记录 direct Q gather 与 direct chunked-context KV gather;Mooncake failed_keys=0。约 43 分钟服务后以 0 tok/s 挂起(Running: 23,Waiting: 0,GPU KV ~68%),随后 PG ID 3 上 dcp.py dcp_a2a_lse_reduce 的 NCCL ALLTOALL_BASE 超时 600 秒(last started work: -1)→ worker_crash:8 → EngineDead / ProfileAborted。与 tip bed9f1ce c40 的 kv_gather _ALLGATHER_BASE(已由 KV_GATHER=1 修复)不同。A2A=0 强制走 PyNCCL ALLTOALL LSE-reduce 路径;启用 direct A2A 以对齐 GB300 DCP8 / B200 stock。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 1x CONC for CONC 48+ on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 24+ util 0.85). Tip 031de17bf fail-fast run 37010740215 agentic c56 (job 110882921963, Slurm 6848, gpu-12) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, all three direct DCP gathers logged, Mooncake failed_keys=0. Measured warmup packed GPU KV to ~99.7% (Running: 1, Waiting: 12, Deferred: 12) then hung workers: shm_broadcast starvation from 15:28 for the full VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted. No Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Distinct from tip b70e4260a c48 OOM (util 0.92), tip bed9f1ce c40 kv_gather ALLGATHER (KV_GATHER=0), and tip 68cdcc58 c24 ALLTOALL dcp_a2a_lse_reduce (A2A=0). With all three directs already on, high-conc 2x admission was the remaining lever that pinned KV before decode. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上将 CONC 48+ 的 max-num-seqs 限制为 1×CONC(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 24+ util 0.85)。tip 031de17bf fail-fast 运行 37010740215 的 agentic c56(job 110882921963,Slurm 6848,gpu-12)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,三种 direct DCP gather 均已记录,Mooncake failed_keys=0。实测 warmup 将 GPU KV 堆到约 99.7%(Running: 1,Waiting: 12,Deferred: 12)后 worker 挂起:自 15:28 起 shm_broadcast 饥饿直至满 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted。无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。有别于 tip b70e4260a c48 OOM(util 0.92)、tip bed9f1ce c40 kv_gather ALLGATHER(KV_GATHER=0)、tip 68cdcc58 c24 ALLTOALL dcp_a2a_lse_reduce(A2A=0)。三种 direct 已全部开启时,高并发 2× 准入是仍会在 decode 前钉死 KV 的剩余杠杆。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 861a151253b502f8d8384678047d9c6d7e79254f Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 2 Oct 2026 18:43:19 +0000 Subject: [PATCH 20/31] fix(kimik3-b300): extend util 0.85 to CONC 8+ MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extend gpu-memory-utilization 0.85 from CONC 24+ to CONC 8+ (c8/c16); keep 0.92 only on c1–c4. Tip 5ab41690 fail-fast c8 soft-OOMed at util 0.92 during warmup (CUDACachingAllocator ~3.03 GiB / ~1.16 GiB free) → empty streams → ProfileAborted; nccl_error:16 was init-only WARN. 将 gpu-memory-utilization 0.85 从 CONC 24+ 扩展到 CONC 8+(c8/c16); 仅 c1–c4 保留 0.92。tip 5ab41690 fail-fast 的 c8 在 util 0.92 的 warmup 中软 OOM(CUDACachingAllocator 约 3.03 GiB / 约 1.16 GiB 空闲) → 空流 → ProfileAborted;nccl_error:16 仅为初始化期 WARN。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 20 +++++++++++-------- .../docs/configuration-procedures.md | 7 +++++-- .../docs/configuration-procedures_zh.md | 14 +++++++------ inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 34 insertions(+), 16 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index f043c912d7..00bece38cd 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -147,15 +147,19 @@ base: # sample_tokens timed out (no Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest # nccl_error:16 was init-only ibv_query_port_speed WARN). Cap high-conc # admission so prefix+deferred loads cannot pin the pool before decode. CONC -# 24+ use gpu-memory-utilization 0.85: with +# 8+ use gpu-memory-utilization 0.85: with # VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 (Slurm 6737) # over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once CUDA graphs # are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB needed, -# ~2.3 GiB free) ~4m after Application startup. Keep 0.92 on speculative -# c1–c16. Graphs capture every size up to max-num-seqs, then the larger powers -# of two to 8192. Throughput runs switch DSpark to synthetic rejection at the -# golden acceptance length; points above CONC 16 do not draft and keep the -# matrix's mtp label. +# ~2.3 GiB free) ~4m after Application startup. Tip 5ab41690 c8 (Slurm 6877) +# then soft-OOMed at util 0.92 during warmup (CUDACachingAllocator ~3.03 GiB +# alloc with ~1.16 GiB free on ranks 1–7) → empty streamed responses → +# InvalidInferenceResultError / ProfileAborted at 2/11 > 10% (engine stayed +# up; ingest nccl_error:16 again init-only WARN). Keep 0.92 only on c1–c4. +# Graphs capture every size up to max-num-seqs, then the larger powers of two +# to 8192. Throughput runs switch DSpark to synthetic rejection at the golden +# acceptance length; points above CONC 16 do not draft and keep the matrix's +# mtp label. override_c1: roles: agg: @@ -197,7 +201,7 @@ override_c8: agg: args: max-num-seqs: 16 - gpu-memory-utilization: 0.92 + gpu-memory-utilization: 0.85 compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[8,16,24,32,40,48,56,64,72,80,88,96,104,112,120,128,256,512,1024,2048,4096,8192]}' speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":7,"attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' benchmark: @@ -209,7 +213,7 @@ override_c16: agg: args: max-num-seqs: 32 - gpu-memory-utilization: 0.92 + gpu-memory-utilization: 0.85 compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,256,512,1024,2048,4096,8192]}' speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":3,"attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' benchmark: diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 49b8164b44..ba452d26dc 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -357,11 +357,14 @@ Do not set `MC_MAX_MR_SIZE` here: with 4GiB every rank hit `register_buffer failed ... -600` on the ~40 GiB KV region and stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips registered cleanly. Keep `enable-cumem-allocator` off on this path. Keep -`gpu-memory-utilization` at 0.85 for CONC 24+ (tip b70e4260a c48 reached +`gpu-memory-utilization` at 0.85 for CONC 8+ (tip b70e4260a c48 reached Application startup with KV 45.2 GiB at 0.92 under `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0`, then OOMed in flashinfer FP4 MoE `prepare_moe` allocating ~2.89 GiB with ~2.3 GiB free; vLLM suggested ~36.78 GiB -KV once CUDA graphs are counted). Cap `max-num-seqs` at 1×CONC for CONC 48+ +KV once CUDA graphs are counted. Tip 5ab41690 c8 then soft-OOMed at util 0.92 +during warmup — CUDACachingAllocator failed a ~3.03 GiB alloc with ~1.16 GiB +free — yielding empty streams and ProfileAborted at 2/11 > 10%; keep 0.92 only +on c1–c4). Cap `max-num-seqs` at 1×CONC for CONC 48+ (tip 031de17bf c56 with A2A/Q/KV all direct and util 0.85 packed GPU KV to ~99.7% under 2× admission, then hung workers through the 1800s `sample_tokens` RPC timeout with no Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index ec2692b711..7003b31712 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -329,15 +329,17 @@ gather 已生效时,于 PyNCCL `ALLTOALL_BASE` / `dcp_a2a_lse_reduce` 挂起 不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 `AddressNotRegistered` TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。此路径保持关闭 -`enable-cumem-allocator`。CONC 24+ 将 `gpu-memory-utilization` 保持为 0.85 +`enable-cumem-allocator`。CONC 8+ 将 `gpu-memory-utilization` 保持为 0.85 (tip b70e4260a 的 c48 在 `VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0` 与 0.92 下以 45.2 GiB KV 完成 Application startup,随后在 flashinfer FP4 MoE `prepare_moe` 申请约 2.89 GiB 时仅剩约 2.3 GiB 空闲而 OOM;vLLM 在计入 CUDA -graph 后建议约 36.78 GiB KV)。CONC 48+ 将 `max-num-seqs` 限制为 1×CONC -(tip 031de17bf 的 c56 在 A2A/Q/KV 均已 direct 且 util 0.85 时,2× 准入把 -GPU KV 堆到约 99.7%,随后 worker 挂满 1800 秒 `sample_tokens` RPC 超时,无 -Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 `nccl_error:16` 仅为 -初始化期 `ibv_query_port_speed` WARN)。 +graph 后建议约 36.78 GiB KV。tip 5ab41690 的 c8 在 util 0.92 的 warmup 中软 +OOM——CUDACachingAllocator 申请约 3.03 GiB 时仅剩约 1.16 GiB——导致空流与 +ProfileAborted(2/11 > 10%);仅 c1–c4 保留 0.92)。CONC 48+ 将 +`max-num-seqs` 限制为 1×CONC(tip 031de17bf 的 c56 在 A2A/Q/KV 均已 direct +且 util 0.85 时,2× 准入把 GPU KV 堆到约 99.7%,随后 worker 挂满 1800 秒 +`sample_tokens` RPC 超时,无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; +ingest 的 `nccl_error:16` 仅为初始化期 `ibv_query_port_speed` WARN)。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 37a00d0f74..9150dbcd4e 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9378,3 +9378,12 @@ - "Cap max-num-seqs at 1x CONC for CONC 48+ on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 24+ util 0.85). Tip 031de17bf fail-fast run 37010740215 agentic c56 (job 110882921963, Slurm 6848, gpu-12) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, all three direct DCP gathers logged, Mooncake failed_keys=0. Measured warmup packed GPU KV to ~99.7% (Running: 1, Waiting: 12, Deferred: 12) then hung workers: shm_broadcast starvation from 15:28 for the full VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted. No Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Distinct from tip b70e4260a c48 OOM (util 0.92), tip bed9f1ce c40 kv_gather ALLGATHER (KV_GATHER=0), and tip 68cdcc58 c24 ALLTOALL dcp_a2a_lse_reduce (A2A=0). With all three directs already on, high-conc 2x admission was the remaining lever that pinned KV before decode. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." - "在 B300 DSXE 上将 CONC 48+ 的 max-num-seqs 限制为 1×CONC(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 24+ util 0.85)。tip 031de17bf fail-fast 运行 37010740215 的 agentic c56(job 110882921963,Slurm 6848,gpu-12)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,三种 direct DCP gather 均已记录,Mooncake failed_keys=0。实测 warmup 将 GPU KV 堆到约 99.7%(Running: 1,Waiting: 12,Deferred: 12)后 worker 挂起:自 15:28 起 shm_broadcast 饥饿直至满 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted。无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。有别于 tip b70e4260a c48 OOM(util 0.92)、tip bed9f1ce c40 kv_gather ALLGATHER(KV_GATHER=0)、tip 68cdcc58 c24 ALLTOALL dcp_a2a_lse_reduce(A2A=0)。三种 direct 已全部开启时,高并发 2× 准入是仍会在 decode 前钉死 KV 的剩余杠杆。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Extend gpu-memory-utilization 0.85 to CONC 8+ on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 48+ max-num-seqs 1x). Tip 5ab41690 fail-fast run 37033506119 agentic c8 (job 110957197772, Slurm 6877, gpu-04) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, KV 38.89 GiB at util 0.92 under VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0; during warmup CUDACachingAllocator soft-OOMed on ranks 0-7 trying to allocate ~3.03 GiB with ~1.16 GiB free → empty streamed responses → InvalidInferenceResultError / ProfileAborted at 2/11 > 10% (engine stayed up; no Watchdog / ALLGATHER / ALLTOALL / hard EngineDead). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Same soft-OOM family as tip b70e4260a c48 at util 0.92; keep 0.92 only on c1-c4. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "将 B300 DSXE 上 gpu-memory-utilization 0.85 扩展到 CONC 8+(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 48+ max-num-seqs 1×)。tip 5ab41690 fail-fast 运行 37033506119 的 agentic c8(job 110957197772,Slurm 6877,gpu-04)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,在 VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0 与 util 0.92 下 KV 为 38.89 GiB;warmup 中 ranks 0–7 的 CUDACachingAllocator 软 OOM(申请约 3.03 GiB,空闲约 1.16 GiB)→ 空流 → InvalidInferenceResultError / ProfileAborted(2/11 > 10%;引擎未死;无 Watchdog / ALLGATHER / ALLTOALL / 硬 EngineDead)。ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。与 tip b70e4260a c48 在 util 0.92 下的软 OOM 同族;仅 c1–c4 保留 0.92。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 64232c0ea443aee781e2526d9f96cabca5509f8c Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 2 Oct 2026 22:11:07 +0000 Subject: [PATCH 21/31] fix(kimik3-b300): cap CONC 32+ max-num-seqs at 1x MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extend the 1x CONC admission cap from CONC 48+ to CONC 32+ (c32/c40); keep 2x only through CONC 24. Tip 861a1512 fail-fast c32 pinned GPU KV ~99–100% under max-num-seqs=64, then hung with shm_broadcast starvation into sample_tokens TimeoutError → EngineDead / ProfileAborted. 将 1×CONC 准入上限从 CONC 48+ 下扩到 CONC 32+(c32/c40);仅 CONC 24 及以下保留 2×。tip 861a1512 fail-fast 的 c32 在 max-num-seqs=64 下将 GPU KV 钉在约 99–100%,随后 shm_broadcast 饥饿直至 sample_tokens TimeoutError → EngineDead / ProfileAborted。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 21 ++++++++++++------- .../docs/configuration-procedures.md | 8 +++++-- .../docs/configuration-procedures_zh.md | 8 +++++-- inferencex-e2e/perf-changelog.yaml | 9 ++++++++ 4 files changed, 34 insertions(+), 12 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 00bece38cd..feacc5364a 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -139,15 +139,20 @@ base: KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2249' -# One variant per point. Admission is 2x CONC through CONC 40; CONC 48+ use 1x +# One variant per point. Admission is 2x CONC through CONC 24; CONC 32+ use 1x # CONC. Tip 031de17bf c56 (Slurm 6848) with A2A/Q/KV all direct, util 0.85, and # max-num-seqs=112 packed GPU KV to ~99.7% (Running: 1, Waiting: 12, Deferred: # 12) then hung workers for the full VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 — # shm_broadcast starvation from 15:28, then TimeoutError: RPC call to # sample_tokens timed out (no Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest -# nccl_error:16 was init-only ibv_query_port_speed WARN). Cap high-conc -# admission so prefix+deferred loads cannot pin the pool before decode. CONC -# 8+ use gpu-memory-utilization 0.85: with +# nccl_error:16 was init-only ibv_query_port_speed WARN). Tip 861a1512 c32 +# (Slurm 6916) repeated the same kill under max-num-seqs=64 (2x): GPU KV pinned +# ~99–100% for tens of minutes, then hung at 0 tok/s (Running: 23, Waiting: 11, +# Deferred: 11) with shm_broadcast starvation from 21:03 through the 1800s +# sample_tokens timeout → EngineDead / ProfileAborted (20 empty streams + 62 +# ClientConnectorError after shutdown; ingest nccl_error:16 again init-only). +# Cap CONC 32+ admission so prefix+deferred loads cannot pin the pool before +# decode. CONC 8+ use gpu-memory-utilization 0.85: with # VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 (Slurm 6737) # over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once CUDA graphs # are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB needed, @@ -236,9 +241,9 @@ override_c32: roles: agg: args: - max-num-seqs: 64 + max-num-seqs: 32 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '32' @@ -248,9 +253,9 @@ override_c40: roles: agg: args: - max-num-seqs: 80 + max-num-seqs: 40 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '40' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index ba452d26dc..660107c5c8 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -364,11 +364,15 @@ Application startup with KV 45.2 GiB at 0.92 under KV once CUDA graphs are counted. Tip 5ab41690 c8 then soft-OOMed at util 0.92 during warmup — CUDACachingAllocator failed a ~3.03 GiB alloc with ~1.16 GiB free — yielding empty streams and ProfileAborted at 2/11 > 10%; keep 0.92 only -on c1–c4). Cap `max-num-seqs` at 1×CONC for CONC 48+ +on c1–c4). Cap `max-num-seqs` at 1×CONC for CONC 32+ (tip 031de17bf c56 with A2A/Q/KV all direct and util 0.85 packed GPU KV to ~99.7% under 2× admission, then hung workers through the 1800s `sample_tokens` RPC timeout with no Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; -ingest `nccl_error:16` was init-only `ibv_query_port_speed` WARN). +ingest `nccl_error:16` was init-only `ibv_query_port_speed` WARN. Tip 861a1512 +c32 repeated the same kill under `max-num-seqs=64`: GPU KV pinned ~99–100%, +then hung at 0 tok/s with shm_broadcast starvation from 21:03 through the +1800s `sample_tokens` timeout → EngineDead / ProfileAborted; keep 2× only +through CONC 24). The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 7003b31712..e101206a18 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -335,11 +335,15 @@ TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。此路 `prepare_moe` 申请约 2.89 GiB 时仅剩约 2.3 GiB 空闲而 OOM;vLLM 在计入 CUDA graph 后建议约 36.78 GiB KV。tip 5ab41690 的 c8 在 util 0.92 的 warmup 中软 OOM——CUDACachingAllocator 申请约 3.03 GiB 时仅剩约 1.16 GiB——导致空流与 -ProfileAborted(2/11 > 10%);仅 c1–c4 保留 0.92)。CONC 48+ 将 +ProfileAborted(2/11 > 10%);仅 c1–c4 保留 0.92)。CONC 32+ 将 `max-num-seqs` 限制为 1×CONC(tip 031de17bf 的 c56 在 A2A/Q/KV 均已 direct 且 util 0.85 时,2× 准入把 GPU KV 堆到约 99.7%,随后 worker 挂满 1800 秒 `sample_tokens` RPC 超时,无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; -ingest 的 `nccl_error:16` 仅为初始化期 `ibv_query_port_speed` WARN)。 +ingest 的 `nccl_error:16` 仅为初始化期 `ibv_query_port_speed` WARN。tip +861a1512 的 c32 在 `max-num-seqs=64` 下重现同一杀伤:GPU KV 钉在约 99–100%, +随后以 0 tok/s 挂起,自 21:03 起 shm_broadcast 饥饿直至满 1800 秒 +`sample_tokens` 超时 → EngineDead / ProfileAborted;仅 CONC 24 及以下保留 +2×)。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9150dbcd4e..cfc69961aa 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9387,3 +9387,12 @@ - "Extend gpu-memory-utilization 0.85 to CONC 8+ on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 48+ max-num-seqs 1x). Tip 5ab41690 fail-fast run 37033506119 agentic c8 (job 110957197772, Slurm 6877, gpu-04) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, KV 38.89 GiB at util 0.92 under VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0; during warmup CUDACachingAllocator soft-OOMed on ranks 0-7 trying to allocate ~3.03 GiB with ~1.16 GiB free → empty streamed responses → InvalidInferenceResultError / ProfileAborted at 2/11 > 10% (engine stayed up; no Watchdog / ALLGATHER / ALLTOALL / hard EngineDead). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Same soft-OOM family as tip b70e4260a c48 at util 0.92; keep 0.92 only on c1-c4. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." - "将 B300 DSXE 上 gpu-memory-utilization 0.85 扩展到 CONC 8+(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 48+ max-num-seqs 1×)。tip 5ab41690 fail-fast 运行 37033506119 的 agentic c8(job 110957197772,Slurm 6877,gpu-04)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,在 VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0 与 util 0.92 下 KV 为 38.89 GiB;warmup 中 ranks 0–7 的 CUDACachingAllocator 软 OOM(申请约 3.03 GiB,空闲约 1.16 GiB)→ 空流 → InvalidInferenceResultError / ProfileAborted(2/11 > 10%;引擎未死;无 Watchdog / ALLGATHER / ALLTOALL / 硬 EngineDead)。ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。与 tip b70e4260a c48 在 util 0.92 下的软 OOM 同族;仅 c1–c4 保留 0.92。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 1x CONC for CONC 32+ on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85). Tip 861a1512 fail-fast run 37049360028 agentic c32 (job 111008983824, Slurm 6916, gpu-00) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, all three direct DCP gathers logged, Mooncake failed_keys=0. Under max-num-seqs=64 (2x) GPU KV pinned ~99-100% for tens of minutes, then hung at 0 tok/s (Running: 23, Waiting: 11, Deferred: 11, GPU KV 71.8%) with shm_broadcast starvation from 21:03 for the full VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted (20 InvalidInferenceResultError empty streams + 62 ClientConnectorError after shutdown). No Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Same admission-pin family as tip 031de17bf c56; extend 1x CONC from CONC 48+ down through c32/c40 and keep 2x only through CONC 24. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上将 CONC 32+ 的 max-num-seqs 限制为 1×CONC(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 8+ util 0.85)。tip 861a1512 fail-fast 运行 37049360028 的 agentic c32(job 111008983824,Slurm 6916,gpu-00)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,三种 direct DCP gather 均已记录,Mooncake failed_keys=0。在 max-num-seqs=64(2×)下 GPU KV 钉在约 99–100% 数十分钟,随后以 0 tok/s 挂起(Running: 23,Waiting: 11,Deferred: 11,GPU KV 71.8%),自 21:03 起 shm_broadcast 饥饿直至满 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted(关机后 20 个空流 InvalidInferenceResultError + 62 个 ClientConnectorError)。无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。与 tip 031de17bf c56 同属准入钉死族;将 1×CONC 从 CONC 48+ 下扩到 c32/c40,仅 CONC 24 及以下保留 2×。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From c138710c25bb58e7525376146ac9c78a63a0a377 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sat, 3 Oct 2026 01:37:57 +0000 Subject: [PATCH 22/31] fix(kimik3-b300): cap c70 max-num-seqs at 48 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip 1d937c73e agentic c70 hung under 1x=70 with Running≈0 / Waiting≈66 / Deferred≈50–67 / KV≈86–91% then sample_tokens timeout. Drop c70 below 1x to leave decode headroom. 将 tip 1d937c73e agentic c70 在 1×=70 下 Running≈0 / Waiting≈66 / Deferred≈50–67 / KV≈86–91% 后 sample_tokens 超时的问题,通过把 c70 降至 48 以保留 decode 余量来修复。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 7 +++++-- .../docs/configuration-procedures.md | 6 ++++-- .../docs/configuration-procedures_zh.md | 20 ++++++++++--------- inferencex-e2e/perf-changelog.yaml | 9 +++++++++ 4 files changed, 29 insertions(+), 13 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index feacc5364a..ecd990430d 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -289,9 +289,12 @@ override_c70: roles: agg: args: - max-num-seqs: 70 + # Tip 1d937c73e c70 under 1x (70) still deferred-admission starved: + # Running≈0 / Waiting≈66 / Deferred≈50–67 / KV≈86–91% for ~30m then + # sample_tokens 1800s timeout. Cap below 1x to leave decode headroom. + max-num-seqs: 48 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '70' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 6b4f39aff1..b16dafc76b 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -364,7 +364,7 @@ Application startup with KV 45.2 GiB at 0.92 under KV once CUDA graphs are counted. Tip 5ab41690 c8 then soft-OOMed at util 0.92 during warmup — CUDACachingAllocator failed a ~3.03 GiB alloc with ~1.16 GiB free — yielding empty streams and ProfileAborted at 2/11 > 10%; keep 0.92 only -on c1–c4). Cap `max-num-seqs` at 1×CONC for CONC 32+ +on c1–c4). Cap `max-num-seqs` at 1×CONC for CONC 32–56 and at 48 for CONC 70 (tip 031de17bf c56 with A2A/Q/KV all direct and util 0.85 packed GPU KV to ~99.7% under 2× admission, then hung workers through the 1800s `sample_tokens` RPC timeout with no Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; @@ -372,7 +372,9 @@ ingest `nccl_error:16` was init-only `ibv_query_port_speed` WARN. Tip 861a1512 c32 repeated the same kill under `max-num-seqs=64`: GPU KV pinned ~99–100%, then hung at 0 tok/s with shm_broadcast starvation from 21:03 through the 1800s `sample_tokens` timeout → EngineDead / ProfileAborted; keep 2× only -through CONC 24). +through CONC 24. Tip 1d937c73e c70 still starved under 1×=`70`: Running≈0 / +Waiting≈66 / Deferred≈50–67 / KV≈86–91% for ~30m then the same 1800s +`sample_tokens` timeout; drop c70 to `max-num-seqs=48`). The AgentX-only `dsv41flash-fp4--vllm-agentic-dspark` recipes use the per-SKU diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index ea26c73211..fd1f28fa2a 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -335,15 +335,17 @@ TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。此路 `prepare_moe` 申请约 2.89 GiB 时仅剩约 2.3 GiB 空闲而 OOM;vLLM 在计入 CUDA graph 后建议约 36.78 GiB KV。tip 5ab41690 的 c8 在 util 0.92 的 warmup 中软 OOM——CUDACachingAllocator 申请约 3.03 GiB 时仅剩约 1.16 GiB——导致空流与 -ProfileAborted(2/11 > 10%);仅 c1–c4 保留 0.92)。CONC 32+ 将 -`max-num-seqs` 限制为 1×CONC(tip 031de17bf 的 c56 在 A2A/Q/KV 均已 direct -且 util 0.85 时,2× 准入把 GPU KV 堆到约 99.7%,随后 worker 挂满 1800 秒 -`sample_tokens` RPC 超时,无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; -ingest 的 `nccl_error:16` 仅为初始化期 `ibv_query_port_speed` WARN。tip -861a1512 的 c32 在 `max-num-seqs=64` 下重现同一杀伤:GPU KV 钉在约 99–100%, -随后以 0 tok/s 挂起,自 21:03 起 shm_broadcast 饥饿直至满 1800 秒 -`sample_tokens` 超时 → EngineDead / ProfileAborted;仅 CONC 24 及以下保留 -2×)。 +ProfileAborted(2/11 > 10%);仅 c1–c4 保留 0.92)。CONC 32–56 将 +`max-num-seqs` 限制为 1×CONC,CONC 70 限制为 48(tip 031de17bf 的 c56 在 +A2A/Q/KV 均已 direct 且 util 0.85 时,2× 准入把 GPU KV 堆到约 99.7%,随后 +worker 挂满 1800 秒 `sample_tokens` RPC 超时,无 Watchdog / ALLGATHER / +ALLTOALL / CUDA OOM;ingest 的 `nccl_error:16` 仅为初始化期 +`ibv_query_port_speed` WARN。tip 861a1512 的 c32 在 `max-num-seqs=64` 下重现 +同一杀伤:GPU KV 钉在约 99–100%,随后以 0 tok/s 挂起,自 21:03 起 +shm_broadcast 饥饿直至满 1800 秒 `sample_tokens` 超时 → EngineDead / +ProfileAborted;仅 CONC 24 及以下保留 2×。tip 1d937c73e 的 c70 在 1×=`70` +下仍饥饿:Running≈0 / Waiting≈66 / Deferred≈50–67 / KV≈86–91% 约 30 分钟后 +同样满 1800 秒 `sample_tokens` 超时;将 c70 降为 `max-num-seqs=48`)。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 5c89248bde..bde03a33d1 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9436,3 +9436,12 @@ - "Cap max-num-seqs at 1x CONC for CONC 32+ on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85). Tip 861a1512 fail-fast run 37049360028 agentic c32 (job 111008983824, Slurm 6916, gpu-00) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, all three direct DCP gathers logged, Mooncake failed_keys=0. Under max-num-seqs=64 (2x) GPU KV pinned ~99-100% for tens of minutes, then hung at 0 tok/s (Running: 23, Waiting: 11, Deferred: 11, GPU KV 71.8%) with shm_broadcast starvation from 21:03 for the full VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted (20 InvalidInferenceResultError empty streams + 62 ClientConnectorError after shutdown). No Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Same admission-pin family as tip 031de17bf c56; extend 1x CONC from CONC 48+ down through c32/c40 and keep 2x only through CONC 24. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." - "在 B300 DSXE 上将 CONC 32+ 的 max-num-seqs 限制为 1×CONC(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 8+ util 0.85)。tip 861a1512 fail-fast 运行 37049360028 的 agentic c32(job 111008983824,Slurm 6916,gpu-00)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,三种 direct DCP gather 均已记录,Mooncake failed_keys=0。在 max-num-seqs=64(2×)下 GPU KV 钉在约 99–100% 数十分钟,随后以 0 tok/s 挂起(Running: 23,Waiting: 11,Deferred: 11,GPU KV 71.8%),自 21:03 起 shm_broadcast 饥饿直至满 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted(关机后 20 个空流 InvalidInferenceResultError + 62 个 ClientConnectorError)。无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。与 tip 031de17bf c56 同属准入钉死族;将 1×CONC 从 CONC 48+ 下扩到 c32/c40,仅 CONC 24 及以下保留 2×。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 48 for CONC 70 on B300 DSXE (keep A2A=1, Q_GATHER=1, KV_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-56 max-num-seqs 1x). Tip 1d937c73e fail-fast run 37071755348 agentic c70 (job 111074417068, Slurm 6950, gpu-11) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, all three direct DCP gathers logged, Mooncake failed_keys=0. Under max-num-seqs=70 (1x) warmup packed Waiting≈63-70 / Deferred≈50-67 with Running≈0 and GPU KV ≈86-91% for ~30m, then hung with shm_broadcast starvation through VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted (771 warmup dropped / 0 kept; time_to_ready=None). No Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest nccl_error:16 was init-only ibv_query_port_speed WARN (false positive). Same deferred-admission family as tip 861a1512 c32 / 031de17bf c56; 1x insufficient at c70 so drop below 1x to 48. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + - "在 B300 DSXE 上将 CONC 70 的 max-num-seqs 限制为 48(保留 A2A=1、Q_GATHER=1、KV_GATHER=1、load_async=true、max_load_batch_keys=1、CONC 8+ util 0.85、CONC 32–56 max-num-seqs 1×)。tip 1d937c73e fail-fast 运行 37071755348 的 agentic c70(job 111074417068,Slurm 6950,gpu-11)已有真实 rail(Mooncake rail: ibp198s0f0 / Patched),Application startup 完成,三种 direct DCP gather 均已记录,Mooncake failed_keys=0。在 max-num-seqs=70(1×)下 warmup 将 Waiting≈63–70 / Deferred≈50–67 堆满且 Running≈0、GPU KV ≈86–91% 约 30 分钟,随后 shm_broadcast 饥饿直至满 VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 → TimeoutError: RPC call to sample_tokens timed out → EngineDead / ProfileAborted(丢弃 771 warmup / 保留 0;time_to_ready=None)。无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 nccl_error:16 仅为初始化期 ibv_query_port_speed WARN(误报)。与 tip 861a1512 c32 / 031de17bf c56 同属 deferred 准入族;c70 上 1× 不足,降至 48。保留不启用 compact_group_io、不设 MC_MAX_MR_SIZE、关闭 cumem。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 7230f68f24ff8b808ca6633f772c23930bf6e1a0 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sat, 3 Oct 2026 08:19:42 +0000 Subject: [PATCH 23/31] fix(kimik3-b300): disable direct DCP KV gather MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip f7be12ed agentic c56 trapped in the direct KV-gather multimem kernel (timeout source=1 epoch=494017) then CUDA unspecified launch failure. Leave that stock path; keep A2A and Q gather on. tip f7be12ed 的 agentic c56 在 direct KV-gather multimem kernel 超时(source=1 epoch=494017)后触发 CUDA unspecified launch failure。关闭该 stock 路径;保留 A2A 与 Q gather。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 14 ++++++++------ inferencex-e2e/docs/configuration-procedures.md | 16 ++++++++++------ .../docs/configuration-procedures_zh.md | 16 ++++++++++------ inferencex-e2e/perf-changelog.yaml | 8 ++++++++ 4 files changed, 36 insertions(+), 18 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index ecd990430d..0095a44128 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -101,12 +101,14 @@ base: # Enable direct A2A (GB300 DCP8 / B200 stock). load_async must stay true. VLLM_USE_DIRECT_DCP_A2A: '1' VLLM_USE_DIRECT_DCP_Q_GATHER: '1' - # Tip bed9f1ce c40 (Slurm 6765): with KV_GATHER=0 the hang is PyNCCL - # kv_gather _ALLGATHER_BASE (dcp.py:1413; last started work: -1) after - # ~42m serve / GPU KV ~86-100%, Mooncake failed_keys=0. Re-enable direct - # KV gather (GB300 DCP8 / B200 Mooncake stock); tip 68cdcc58 confirmed - # "Using direct symmetric-memory DCP chunked-context KV gather". - VLLM_USE_DIRECT_DCP_KV_GATHER: '1' + # Tip bed9f1ce c40 hung in PyNCCL kv_gather _ALLGATHER_BASE when this + # was 0, so it was restored. Tip f7be12ed c56 (run 37086876838, Slurm + # 6989) with it on died in the direct kernel itself: "direct DCP + # kv-gather multimem timeout source=1 epoch=494017" then asm trap → + # CUDA unspecified launch failure (Mooncake memcpy -800 is the dead + # device, not the first boundary). Stock gate off selects + # all_gather_into_tensor. Keep A2A and Q gather on. + VLLM_USE_DIRECT_DCP_KV_GATHER: '0' VLLM_ENGINE_READY_TIMEOUT_S: '3600' VLLM_RPC_TIMEOUT: '600000' # Mooncake loads can block inside execute_model past the 300s default; diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index b16dafc76b..62efae4612 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -347,12 +347,16 @@ canary c1 crashed with Mooncake `AssertionError: load_async must be True for better performance` when `load_async` was set false, so restore the required stock true), but do not enable `compact_group_io` on this DSXE single-rail path (it storm-failed ~25 MiB compact-group puts at c8). Keep -`VLLM_USE_DIRECT_DCP_A2A=1`, `VLLM_USE_DIRECT_DCP_Q_GATHER=1`, and -`VLLM_USE_DIRECT_DCP_KV_GATHER=1` (tip 1f837c46 eval-only c8 hung ~8.5m after -`dcp:0` then failed EP `ncclCommInitRank` with Q gather off; tip bed9f1ce c40 -hung in PyNCCL `kv_gather` `_ALLGATHER_BASE` with `last started work: -1` when -KV gather was off; tip 68cdcc58 c24 then hung in PyNCCL `ALLTOALL_BASE` inside -`dcp_a2a_lse_reduce` with A2A off while direct Q/KV gathers were already active). +`VLLM_USE_DIRECT_DCP_A2A=1` and `VLLM_USE_DIRECT_DCP_Q_GATHER=1`, and set +`VLLM_USE_DIRECT_DCP_KV_GATHER=0` (tip 1f837c46 eval-only c8 hung ~8.5m after +`dcp:0` then failed EP `ncclCommInitRank` with Q gather off; tip 68cdcc58 c24 +hung in PyNCCL `ALLTOALL_BASE` inside `dcp_a2a_lse_reduce` with A2A off. Tip +bed9f1ce c40 hung in PyNCCL `kv_gather` `_ALLGATHER_BASE` when KV gather was +off, which is why the gate was restored; tip f7be12ed c56, run 37086876838, +then died on the direct kernel itself — `direct DCP kv-gather multimem timeout +source=1 epoch=494017` followed by an `asm trap` that is the CUDA unspecified +launch failure — so the stock gate is off again. Mooncake GPU memcpy `-800` +was the poisoned device, not the first boundary.). Do not set `MC_MAX_MR_SIZE` here: with 4GiB every rank hit `register_buffer failed ... -600` on the ~40 GiB KV region and stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips registered diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index fd1f28fa2a..4350245270 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -320,12 +320,16 @@ Mooncake 指标干净时,于约 98–100% GPU KV 下挂起在 DCP PYNCCL `AssertionError: load_async must be True for better performance`,故恢复必需的 stock true),但不要在该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。 -保持 `VLLM_USE_DIRECT_DCP_A2A=1`、`VLLM_USE_DIRECT_DCP_Q_GATHER=1` 与 -`VLLM_USE_DIRECT_DCP_KV_GATHER=1`(tip 1f837c46 的 eval-only c8 在关闭 Q gather -后于 `dcp:0` 之后挂起约 8.5 分钟,随后 EP `ncclCommInitRank` 失败;tip -bed9f1ce 的 c40 在关闭 KV gather 时于 PyNCCL `kv_gather` `_ALLGATHER_BASE` -挂起,`last started work: -1`;tip 68cdcc58 的 c24 在关闭 A2A 且 Q/KV direct -gather 已生效时,于 PyNCCL `ALLTOALL_BASE` / `dcp_a2a_lse_reduce` 挂起)。 +保持 `VLLM_USE_DIRECT_DCP_A2A=1` 与 `VLLM_USE_DIRECT_DCP_Q_GATHER=1`,并将 +`VLLM_USE_DIRECT_DCP_KV_GATHER` 设为 0(tip 1f837c46 的 eval-only c8 在关闭 Q +gather 后于 `dcp:0` 之后挂起约 8.5 分钟,随后 EP `ncclCommInitRank` 失败;tip +68cdcc58 的 c24 在关闭 A2A 时于 PyNCCL `ALLTOALL_BASE` / `dcp_a2a_lse_reduce` +挂起。tip bed9f1ce 的 c40 在关闭 KV gather 时于 PyNCCL `kv_gather` +`_ALLGATHER_BASE` 挂起,因此该开关曾被重新打开;tip f7be12ed 的 c56(运行 +37086876838)随后死在 direct kernel 本身——`direct DCP kv-gather multimem +timeout source=1 epoch=494017`,接着 `asm trap`,即 CUDA unspecified launch +failure——故该 stock 开关再次关闭。Mooncake GPU memcpy `-800` 是设备已损坏后的 +后果,不是第一故障边界。)。 不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 `AddressNotRegistered` TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。此路径保持关闭 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index cb9f80095e..b8ca371328 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9450,3 +9450,11 @@ description: - "Update the GLM-5.2 MXFP4 MI355X SGLang AgentX image from lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260924 to lmsysorg/sglang-rocm:v0.5.20-rocm720-mi35x-20260930 (digest sha256:ac30a0fe98fb3f6e258d302280110f1be44c83edc54b914037db57c82cc20828). The image is built from SGLang commit d26ec4690d10d4c69890280b313d0409c8265feb, which contains the merge commit for sgl-project/sglang#38583." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3616 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Disable VLLM_USE_DIRECT_DCP_KV_GATHER on B300 DSXE (keep A2A=1, Q_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-56 max-num-seqs 1x, c70 max-num-seqs 48). Tip f7be12ed fail-fast run 37086876838 agentic c56 (job 111113692599, Slurm 6989, gpu-01) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; the srt-slurm Wrote line is the pre-setup dump), Application startup complete, and profiling kept 1429 requests. At 07:39:41 the direct kernel logged direct DCP kv-gather multimem timeout source=1 epoch=494017 then trapped the device (dcp_direct_kv_gather.cu wait_for_epoch plus asm trap) which is the CUDA unspecified launch failure observed in KVCacheStoreSendingThread, then GPU memcpy failed src_dev=-1 dst_dev=2 size=884736 / TRANSFER_FAIL -800, VllmWorker-2 died, EngineDead / ProfileAborted (143/1429 = 10.007%). Mooncake failed_keys stayed 0 until that instant. Not deferred-admission starvation: the prior 10s window was 347 tok/s with Running: 9 and there was no sample_tokens timeout. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Distinct from tip 1d937c73e c70 (Running about 0, 1800s sample_tokens, 0 kept). The stock gate leaves the trapping multimem kernel for all_gather_into_tensor. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From fd61acb055c10264fbb8f1bcad88229b7be82a99 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sat, 3 Oct 2026 11:08:48 +0000 Subject: [PATCH 24/31] fix(kimik3-b300): disable direct DCP Q gather MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip db6caecc agentic c24 trapped in the direct Q-gather multimem kernel (timeout source=7 epoch=1107073) then CUDA unspecified launch failure. Leave that stock path; keep A2A on and KV gather off. CAUTION: tip 1f837c46 / sweep 36937918258 previously hung at EP ncclCommInitRank after ~8.5m with Q_GATHER=0 — watch eval init. tip db6caecc 的 agentic c24 在 direct Q-gather multimem kernel 超时(source=7 epoch=1107073)后触发 CUDA unspecified launch failure。关闭该 stock 路径;保留 A2A,KV gather 保持关闭。 注意:tip 1f837c46 / sweep 36937918258 曾在 Q_GATHER=0 下于约 8.5 分钟后挂在 EP ncclCommInitRank — 需关注 eval 初始化。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 20 +++++++++------ .../docs/configuration-procedures.md | 25 +++++++++++-------- .../docs/configuration-procedures_zh.md | 24 ++++++++++-------- inferencex-e2e/perf-changelog.yaml | 8 ++++++ 4 files changed, 50 insertions(+), 27 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 0095a44128..978f5bbfc4 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -93,21 +93,27 @@ base: VLLM_ALLREDUCE_USE_FLASHINFER: '1' VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: '1' VLLM_USE_V2_MODEL_RUNNER: '1' - # Tip 1f837c46 eval-only c8 hung ~8.5m after dcp:0 then failed EP - # PyNccl ncclCommInitRank with Q_GATHER=0; keep Q gather on. Tip - # 68cdcc58 c24 (Slurm 6814): with A2A=0 the hang moved to PyNCCL + # Tip 68cdcc58 c24 (Slurm 6814): with A2A=0 the hang was PyNCCL # ALLTOALL_BASE in dcp_a2a_lse_reduce (dcp.py:765; last started work: - # -1) after ~43m serve — direct Q/KV gathers were already active. - # Enable direct A2A (GB300 DCP8 / B200 stock). load_async must stay true. + # -1) after ~43m serve. Enable direct A2A (GB300 DCP8 / B200 stock). + # load_async must stay true. VLLM_USE_DIRECT_DCP_A2A: '1' - VLLM_USE_DIRECT_DCP_Q_GATHER: '1' + # Tip 1f837c46 eval-only c8 hung ~8.5m after dcp:0 then failed EP + # PyNccl ncclCommInitRank with Q_GATHER=0, so Q was restored. Tip + # db6caecc c24 (run 37109541419, Slurm 7019) then died in the direct + # Q kernel itself: "direct DCP q-gather multimem timeout source=7 + # epoch=1107073" → CUDA unspecified launch failure in + # KVCacheStoreSendingThread → EngineDead / ProfileAborted (105/1041). + # Mooncake failed_keys=0; GPU KV ~28–42%; not admission starvation. + # Stock gate off leaves the trapping multimem path. Keep A2A on. + VLLM_USE_DIRECT_DCP_Q_GATHER: '0' # Tip bed9f1ce c40 hung in PyNCCL kv_gather _ALLGATHER_BASE when this # was 0, so it was restored. Tip f7be12ed c56 (run 37086876838, Slurm # 6989) with it on died in the direct kernel itself: "direct DCP # kv-gather multimem timeout source=1 epoch=494017" then asm trap → # CUDA unspecified launch failure (Mooncake memcpy -800 is the dead # device, not the first boundary). Stock gate off selects - # all_gather_into_tensor. Keep A2A and Q gather on. + # all_gather_into_tensor. Keep A2A on; Q gather also gated off above. VLLM_USE_DIRECT_DCP_KV_GATHER: '0' VLLM_ENGINE_READY_TIMEOUT_S: '3600' VLLM_RPC_TIMEOUT: '600000' diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index c73d965128..52891147fd 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -347,16 +347,21 @@ canary c1 crashed with Mooncake `AssertionError: load_async must be True for better performance` when `load_async` was set false, so restore the required stock true), but do not enable `compact_group_io` on this DSXE single-rail path (it storm-failed ~25 MiB compact-group puts at c8). Keep -`VLLM_USE_DIRECT_DCP_A2A=1` and `VLLM_USE_DIRECT_DCP_Q_GATHER=1`, and set -`VLLM_USE_DIRECT_DCP_KV_GATHER=0` (tip 1f837c46 eval-only c8 hung ~8.5m after -`dcp:0` then failed EP `ncclCommInitRank` with Q gather off; tip 68cdcc58 c24 -hung in PyNCCL `ALLTOALL_BASE` inside `dcp_a2a_lse_reduce` with A2A off. Tip -bed9f1ce c40 hung in PyNCCL `kv_gather` `_ALLGATHER_BASE` when KV gather was -off, which is why the gate was restored; tip f7be12ed c56, run 37086876838, -then died on the direct kernel itself — `direct DCP kv-gather multimem timeout -source=1 epoch=494017` followed by an `asm trap` that is the CUDA unspecified -launch failure — so the stock gate is off again. Mooncake GPU memcpy `-800` -was the poisoned device, not the first boundary.). +`VLLM_USE_DIRECT_DCP_A2A=1`, and set both `VLLM_USE_DIRECT_DCP_Q_GATHER=0` and +`VLLM_USE_DIRECT_DCP_KV_GATHER=0` (tip 68cdcc58 c24 hung in PyNCCL +`ALLTOALL_BASE` inside `dcp_a2a_lse_reduce` with A2A off. Tip bed9f1ce c40 +hung in PyNCCL `kv_gather` `_ALLGATHER_BASE` when KV gather was off, which is +why that gate was restored; tip f7be12ed c56, run 37086876838, then died on +the direct KV kernel itself — `direct DCP kv-gather multimem timeout source=1 +epoch=494017` followed by an `asm trap` that is the CUDA unspecified launch +failure — so KV gather is gated off again. Tip db6caecc c24, run 37109541419, +then died on the direct Q kernel — `direct DCP q-gather multimem timeout +source=7 epoch=1107073` → CUDA unspecified launch failure in +`KVCacheStoreSendingThread` → EngineDead / ProfileAborted — so Q gather is +gated off too. Tip 1f837c46 eval-only c8 previously hung ~8.5m after `dcp:0` +then failed EP `ncclCommInitRank` with Q gather off; watch eval init on this +gate. Mooncake GPU memcpy `-800` was the poisoned device, not the first +boundary.). Do not set `MC_MAX_MR_SIZE` here: with 4GiB every rank hit `register_buffer failed ... -600` on the ~40 GiB KV region and stormed `AddressNotRegistered` TRANSFER_FAIL (c2/c32); pre-`MC_MAX_MR` tips registered diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 21c13761ef..9931bf3657 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -326,16 +326,20 @@ Mooncake 指标干净时,于约 98–100% GPU KV 下挂起在 DCP PYNCCL `AssertionError: load_async must be True for better performance`,故恢复必需的 stock true),但不要在该 DSXE 单 rail 路径上启用 `compact_group_io`(c8 上曾对约 25 MiB 的 compact-group put 产生大量失败)。 -保持 `VLLM_USE_DIRECT_DCP_A2A=1` 与 `VLLM_USE_DIRECT_DCP_Q_GATHER=1`,并将 -`VLLM_USE_DIRECT_DCP_KV_GATHER` 设为 0(tip 1f837c46 的 eval-only c8 在关闭 Q -gather 后于 `dcp:0` 之后挂起约 8.5 分钟,随后 EP `ncclCommInitRank` 失败;tip -68cdcc58 的 c24 在关闭 A2A 时于 PyNCCL `ALLTOALL_BASE` / `dcp_a2a_lse_reduce` -挂起。tip bed9f1ce 的 c40 在关闭 KV gather 时于 PyNCCL `kv_gather` -`_ALLGATHER_BASE` 挂起,因此该开关曾被重新打开;tip f7be12ed 的 c56(运行 -37086876838)随后死在 direct kernel 本身——`direct DCP kv-gather multimem -timeout source=1 epoch=494017`,接着 `asm trap`,即 CUDA unspecified launch -failure——故该 stock 开关再次关闭。Mooncake GPU memcpy `-800` 是设备已损坏后的 -后果,不是第一故障边界。)。 +保持 `VLLM_USE_DIRECT_DCP_A2A=1`,并将 `VLLM_USE_DIRECT_DCP_Q_GATHER` 与 +`VLLM_USE_DIRECT_DCP_KV_GATHER` 均设为 0(tip 68cdcc58 的 c24 在关闭 A2A 时于 +PyNCCL `ALLTOALL_BASE` / `dcp_a2a_lse_reduce` 挂起。tip bed9f1ce 的 c40 在关闭 +KV gather 时于 PyNCCL `kv_gather` `_ALLGATHER_BASE` 挂起,因此该开关曾被重新 +打开;tip f7be12ed 的 c56(运行 37086876838)随后死在 direct KV kernel +本身——`direct DCP kv-gather multimem timeout source=1 epoch=494017`,接着 +`asm trap`,即 CUDA unspecified launch failure——故 KV gather 再次关闭。tip +db6caecc 的 c24(运行 37109541419)随后死在 direct Q kernel——`direct DCP +q-gather multimem timeout source=7 epoch=1107073` → +`KVCacheStoreSendingThread` 中 CUDA unspecified launch failure → EngineDead / +ProfileAborted——故 Q gather 一并关闭。tip 1f837c46 的 eval-only c8 曾在关闭 Q +gather 后于 `dcp:0` 之后挂起约 8.5 分钟并导致 EP `ncclCommInitRank` 失败;此门 +禁下需关注 eval 初始化。Mooncake GPU memcpy `-800` 是设备已损坏后的后果,不是 +第一故障边界。)。 不要在此设置 `MC_MAX_MR_SIZE`:设为 4GiB 时各 rank 对约 40 GiB KV 区域报 `register_buffer failed ... -600`,并引发 `AddressNotRegistered` TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。此路径保持关闭 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index edbed126b8..a608284ea5 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9467,3 +9467,11 @@ description: - "Disable VLLM_USE_DIRECT_DCP_KV_GATHER on B300 DSXE (keep A2A=1, Q_GATHER=1, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-56 max-num-seqs 1x, c70 max-num-seqs 48). Tip f7be12ed fail-fast run 37086876838 agentic c56 (job 111113692599, Slurm 6989, gpu-01) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; the srt-slurm Wrote line is the pre-setup dump), Application startup complete, and profiling kept 1429 requests. At 07:39:41 the direct kernel logged direct DCP kv-gather multimem timeout source=1 epoch=494017 then trapped the device (dcp_direct_kv_gather.cu wait_for_epoch plus asm trap) which is the CUDA unspecified launch failure observed in KVCacheStoreSendingThread, then GPU memcpy failed src_dev=-1 dst_dev=2 size=884736 / TRANSFER_FAIL -800, VllmWorker-2 died, EngineDead / ProfileAborted (143/1429 = 10.007%). Mooncake failed_keys stayed 0 until that instant. Not deferred-admission starvation: the prior 10s window was 347 tok/s with Running: 9 and there was no sample_tokens timeout. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Distinct from tip 1d937c73e c70 (Running about 0, 1800s sample_tokens, 0 kept). The stock gate leaves the trapping multimem kernel for all_gather_into_tensor. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Disable VLLM_USE_DIRECT_DCP_Q_GATHER on B300 DSXE (keep A2A=1, KV_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-56 max-num-seqs 1x, c70 max-num-seqs 48). Tip db6caecc fail-fast run 37109541419 agentic c24 (job 111177660118, Slurm 7019, gpu-17) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; the srt-slurm Wrote line is the pre-setup dump), Application startup complete, and profiling kept 1041 requests (dropped 265 warmup). At 10:41:47 the direct kernel logged direct DCP q-gather multimem timeout source=7 epoch=1107073 then CUDA unspecified launch failure in KVCacheStoreSendingThread across ranks, EngineDead / ProfileAborted (105/1041 = 10.086%). Mooncake failed_keys stayed 0; prior 10s windows were healthy (Running ~17-25, GPU KV ~28-42%, generation hundreds of tok/s). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same multimem-trap family as tip f7be12ed c56 kv-gather (KV_GATHER already 0); not deferred-admission starvation. CAUTION: tip 1f837c46 / sweep 36937918258 previously tried Q_GATHER=0 and hung ~8.5m after dcp:0 on agentic-eval c8 at EP PyNcclCommunicator ncclCommInitRank (rail ibp198s0f0) — Q was restored and A2A briefly disabled then restored; watch eval init under this gate with A2A=1. Stock gate leaves the trapping q-gather multimem kernel. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From ebc4eb499051074ca0cb2f5fc41e224bad8248d1 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sat, 3 Oct 2026 15:43:33 +0000 Subject: [PATCH 25/31] fix(kimik3-b300): cap c56 max-num-seqs at 48 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip fd61acb0 agentic c56 packed GPU KV to 98.2% (Running: 8, Waiting: 53, Deferred: 38) then hung PyNCCL kv_gather _ALLGATHER_BASE (dcp.py:1413, last started work: -1) for 600s. Drop c56 below 1x to 48 like c70. Keep A2A on; do not re-enable KV/Q gather. 将 tip fd61acb0 agentic c56 在 GPU KV 98.2%(Running: 8, Waiting: 53, Deferred: 38)后挂起 PyNCCL kv_gather _ALLGATHER_BASE(dcp.py:1413, last started work: -1)600 秒的 问题,通过把 c56 降至 48(与 c70 一致)来修复。保留 A2A; 不重新打开 KV/Q gather。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 12 +++++++++--- inferencex-e2e/perf-changelog.yaml | 8 ++++++++ 2 files changed, 17 insertions(+), 3 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 978f5bbfc4..a2834a7a75 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -160,7 +160,10 @@ base: # sample_tokens timeout → EngineDead / ProfileAborted (20 empty streams + 62 # ClientConnectorError after shutdown; ingest nccl_error:16 again init-only). # Cap CONC 32+ admission so prefix+deferred loads cannot pin the pool before -# decode. CONC 8+ use gpu-memory-utilization 0.85: with +# decode. Tip fd61acb0 c56 (run 37118728189, Slurm 7040) still packed GPU KV +# to 98.2% at 1x=56 (Running: 8, Waiting: 53, Deferred: 38) then hung PyNCCL +# kv_gather _ALLGATHER_BASE (dcp.py:1413, last started work: -1) for 600s. +# Cap c56 at 48 like c70. CONC 8+ use gpu-memory-utilization 0.85: with # VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 (Slurm 6737) # over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once CUDA graphs # are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB needed, @@ -285,9 +288,12 @@ override_c56: roles: agg: args: - max-num-seqs: 56 + # Tip fd61acb0 c56 (Slurm 7040) under 1x (56) packed GPU KV to 98.2% + # (Running: 8, Waiting: 53, Deferred: 38) then hung kv_gather + # _ALLGATHER_BASE 600s (last started work: -1). Cap at 48 like c70. + max-num-seqs: 48 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,64,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '56' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index a608284ea5..9e4432eabe 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9475,3 +9475,11 @@ description: - "Disable VLLM_USE_DIRECT_DCP_Q_GATHER on B300 DSXE (keep A2A=1, KV_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-56 max-num-seqs 1x, c70 max-num-seqs 48). Tip db6caecc fail-fast run 37109541419 agentic c24 (job 111177660118, Slurm 7019, gpu-17) had a real rail (Mooncake rail: ibp198s0f0 / Patched device_name; the srt-slurm Wrote line is the pre-setup dump), Application startup complete, and profiling kept 1041 requests (dropped 265 warmup). At 10:41:47 the direct kernel logged direct DCP q-gather multimem timeout source=7 epoch=1107073 then CUDA unspecified launch failure in KVCacheStoreSendingThread across ranks, EngineDead / ProfileAborted (105/1041 = 10.086%). Mooncake failed_keys stayed 0; prior 10s windows were healthy (Running ~17-25, GPU KV ~28-42%, generation hundreds of tok/s). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same multimem-trap family as tip f7be12ed c56 kv-gather (KV_GATHER already 0); not deferred-admission starvation. CAUTION: tip 1f837c46 / sweep 36937918258 previously tried Q_GATHER=0 and hung ~8.5m after dcp:0 on agentic-eval c8 at EP PyNcclCommunicator ncclCommInitRank (rail ibp198s0f0) — Q was restored and A2A briefly disabled then restored; watch eval init under this gate with A2A=1. Stock gate leaves the trapping q-gather multimem kernel. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 48 for CONC 56 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-48 max-num-seqs 1x, c70 max-num-seqs 48). Tip fd61acb0 fail-fast run 37118728189 agentic c56 (job 111203259062, Slurm 7040, gpu-10) had a real rail, Application startup complete, Mooncake failed_keys=0, and warmup completed 619/619 with 0 errors. Profiling packed GPU KV to 98.2% (Running: 8, Waiting: 53, Deferred: 38) then hung at 0 tok/s at 15:08:49; first causal line is NCCL watchdog WorkNCCL SeqNum=342641 OpType=_ALLGATHER_BASE Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 15:18:37 → DistBackendError / EngineDead / ProfileAborted (62/392 = 15.816%). Direct A2A remained on (dcp.py:1283). Not A2A multimem/ALLTOALL and not a direct-gather multimem trap. Same packed-KV PyNCCL kv_gather family as tip bed9f1ce c40 / 860c1ccf c48; do not re-enable KV/Q gather (multimem traps on f7be12ed/db6caecc). 1x=56 still insufficient at c56 so drop below 1x to 48 like c70. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From c9ffe2b12d30d909b0e3ef97cdb0010ce6cb38d2 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sat, 3 Oct 2026 18:10:12 +0000 Subject: [PATCH 26/31] fix(kimik3-b300): cap c16 max-num-seqs at 1x MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip ebc4eb499 agentic c16 packed GPU KV to ~99% during warmup (Running: 3, Waiting: 7, Deferred: 7) then hung PyNCCL _ALLGATHER_BASE 600s (SeqNum=89905, last started work: -1). Drop c16 from 2x=32 to 1x=16. Keep A2A on; do not re-enable KV/Q gather. 将 tip ebc4eb499 agentic c16 在 warmup 把 GPU KV 堆到约 99% (Running: 3, Waiting: 7, Deferred: 7)后挂起 PyNCCL _ALLGATHER_BASE 600 秒(SeqNum=89905, last started work: -1) 的问题,通过把 c16 从 2×=32 降至 1×=16 来修复。保留 A2A; 不重新打开 KV/Q gather。 Co-authored-by: Wenyao Gao --- .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 16 ++++++++++++---- inferencex-e2e/docs/configuration-procedures.md | 16 +++++++++++----- .../docs/configuration-procedures_zh.md | 14 ++++++++++---- inferencex-e2e/perf-changelog.yaml | 8 ++++++++ 4 files changed, 41 insertions(+), 13 deletions(-) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index a2834a7a75..cc6f742e48 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -147,8 +147,14 @@ base: KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2249' -# One variant per point. Admission is 2x CONC through CONC 24; CONC 32+ use 1x -# CONC. Tip 031de17bf c56 (Slurm 6848) with A2A/Q/KV all direct, util 0.85, and +# One variant per point. Admission is 2x CONC through CONC 8 and CONC 24; CONC +# 16 and CONC 32–48 use 1x CONC; c56 and c70 cap at 48. Tip ebc4eb499 c16 +# (run 37134266476, Slurm 7062) under 2x=32 packed GPU KV to ~98–99.7% during +# AgentX warmup (Running: 3, Waiting: 7, Deferred: 7 at 17:44:47) then hung +# PyNCCL _ALLGATHER_BASE 600s (SeqNum=89905, last started work: -1, PG ID 3) +# → DistBackendError / VllmWorker-7 died / EngineDead / ProfileAborted +# (176 warmup dropped / 0 kept). Cap c16 at 1x=16. Tip 031de17bf c56 +# (Slurm 6848) with A2A/Q/KV all direct, util 0.85, and # max-num-seqs=112 packed GPU KV to ~99.7% (Running: 1, Waiting: 12, Deferred: # 12) then hung workers for the full VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 — # shm_broadcast starvation from 15:28, then TimeoutError: RPC call to @@ -228,9 +234,11 @@ override_c16: roles: agg: args: - max-num-seqs: 32 + # Tip ebc4eb499 c16 (Slurm 7062) under 2x=32 packed GPU KV ~99% in + # warmup then hung PyNCCL _ALLGATHER_BASE 600s (last started work: -1). + max-num-seqs: 16 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,128,256,512,1024,2048,4096,8192]}' speculative-config: '{"method":"dspark","model":"Inferact/Kimi-K3-DSpark","num_speculative_tokens":3,"attention_backend":"TOKENSPEED_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"block"}' benchmark: env: diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 52891147fd..09f6d8d6c5 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -373,17 +373,23 @@ Application startup with KV 45.2 GiB at 0.92 under KV once CUDA graphs are counted. Tip 5ab41690 c8 then soft-OOMed at util 0.92 during warmup — CUDACachingAllocator failed a ~3.03 GiB alloc with ~1.16 GiB free — yielding empty streams and ProfileAborted at 2/11 > 10%; keep 0.92 only -on c1–c4). Cap `max-num-seqs` at 1×CONC for CONC 32–56 and at 48 for CONC 70 -(tip 031de17bf c56 with A2A/Q/KV all direct and util 0.85 packed GPU KV to +on c1–c4). Cap `max-num-seqs` at 1×CONC for CONC 16 and CONC 32–48, and at 48 +for CONC 56 and CONC 70 (tip 031de17bf c56 with A2A/Q/KV all direct and util 0.85 packed GPU KV to ~99.7% under 2× admission, then hung workers through the 1800s `sample_tokens` RPC timeout with no Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest `nccl_error:16` was init-only `ibv_query_port_speed` WARN. Tip 861a1512 c32 repeated the same kill under `max-num-seqs=64`: GPU KV pinned ~99–100%, then hung at 0 tok/s with shm_broadcast starvation from 21:03 through the -1800s `sample_tokens` timeout → EngineDead / ProfileAborted; keep 2× only -through CONC 24. Tip 1d937c73e c70 still starved under 1×=`70`: Running≈0 / +1800s `sample_tokens` timeout → EngineDead / ProfileAborted; keep 2× on CONC +8 and CONC 24. Tip ebc4eb499 c16 under 2×=`32` packed GPU KV to ~99% during +warmup (Running: 3, Waiting: 7, Deferred: 7) then hung PyNCCL +`_ALLGATHER_BASE` 600s (`last started work: -1`) → DistBackendError / +EngineDead / ProfileAborted with 176 warmup dropped / 0 kept; drop c16 to +1×=`16`. Tip 1d937c73e c70 still starved under 1×=`70`: Running≈0 / Waiting≈66 / Deferred≈50–67 / KV≈86–91% for ~30m then the same 1800s -`sample_tokens` timeout; drop c70 to `max-num-seqs=48`). +`sample_tokens` timeout; drop c70 to `max-num-seqs=48`. Tip fd61acb0 c56 +under 1×=`56` packed GPU KV to 98.2% then hung the same `_ALLGATHER_BASE` +watchdog; drop c56 to `max-num-seqs=48`). The B200 entry uses `vllm/vllm-openai:nightly-dev-x86_64-cu130-ac9126e58aa7` with FlashInfer autotuning. TP4 covers concurrency 1–128. DEP2 (TP1 x DP2 + EP2, DeepGEMM MegaMoE) covers 8–32 and diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index 9931bf3657..c6bbb92729 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -349,17 +349,23 @@ TRANSFER_FAIL(c2/c32);加入该变量之前的 tip 注册正常。此路 `prepare_moe` 申请约 2.89 GiB 时仅剩约 2.3 GiB 空闲而 OOM;vLLM 在计入 CUDA graph 后建议约 36.78 GiB KV。tip 5ab41690 的 c8 在 util 0.92 的 warmup 中软 OOM——CUDACachingAllocator 申请约 3.03 GiB 时仅剩约 1.16 GiB——导致空流与 -ProfileAborted(2/11 > 10%);仅 c1–c4 保留 0.92)。CONC 32–56 将 -`max-num-seqs` 限制为 1×CONC,CONC 70 限制为 48(tip 031de17bf 的 c56 在 +ProfileAborted(2/11 > 10%);仅 c1–c4 保留 0.92)。CONC 16 与 CONC 32–48 将 +`max-num-seqs` 限制为 1×CONC,CONC 56 与 CONC 70 限制为 48(tip 031de17bf 的 c56 在 A2A/Q/KV 均已 direct 且 util 0.85 时,2× 准入把 GPU KV 堆到约 99.7%,随后 worker 挂满 1800 秒 `sample_tokens` RPC 超时,无 Watchdog / ALLGATHER / ALLTOALL / CUDA OOM;ingest 的 `nccl_error:16` 仅为初始化期 `ibv_query_port_speed` WARN。tip 861a1512 的 c32 在 `max-num-seqs=64` 下重现 同一杀伤:GPU KV 钉在约 99–100%,随后以 0 tok/s 挂起,自 21:03 起 shm_broadcast 饥饿直至满 1800 秒 `sample_tokens` 超时 → EngineDead / -ProfileAborted;仅 CONC 24 及以下保留 2×。tip 1d937c73e 的 c70 在 1×=`70` +ProfileAborted;仅 CONC 8 与 CONC 24 保留 2×。tip ebc4eb499 的 c16 在 2×=`32` +下 warmup 把 GPU KV 堆到约 99%(Running: 3,Waiting: 7,Deferred: 7),随后 +PyNCCL `_ALLGATHER_BASE` 挂起 600 秒(`last started work: -1`)→ +DistBackendError / EngineDead / ProfileAborted(176 warmup 丢弃 / 0 保留); +将 c16 降为 1×=`16`。tip 1d937c73e 的 c70 在 1×=`70` 下仍饥饿:Running≈0 / Waiting≈66 / Deferred≈50–67 / KV≈86–91% 约 30 分钟后 -同样满 1800 秒 `sample_tokens` 超时;将 c70 降为 `max-num-seqs=48`)。 +同样满 1800 秒 `sample_tokens` 超时;将 c70 降为 `max-num-seqs=48`。tip +fd61acb0 的 c56 在 1×=`56` 下把 GPU KV 堆到 98.2% 后同样挂起 `_ALLGATHER_BASE` +watchdog;将 c56 降为 `max-num-seqs=48`)。 仅运行 AgentX 的 `dsv41flash-fp4--vllm-agentic-dspark` 配方使用 diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 9e4432eabe..6c77920732 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9483,3 +9483,11 @@ description: - "Cap max-num-seqs at 48 for CONC 56 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-48 max-num-seqs 1x, c70 max-num-seqs 48). Tip fd61acb0 fail-fast run 37118728189 agentic c56 (job 111203259062, Slurm 7040, gpu-10) had a real rail, Application startup complete, Mooncake failed_keys=0, and warmup completed 619/619 with 0 errors. Profiling packed GPU KV to 98.2% (Running: 8, Waiting: 53, Deferred: 38) then hung at 0 tok/s at 15:08:49; first causal line is NCCL watchdog WorkNCCL SeqNum=342641 OpType=_ALLGATHER_BASE Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 15:18:37 → DistBackendError / EngineDead / ProfileAborted (62/392 = 15.816%). Direct A2A remained on (dcp.py:1283). Not A2A multimem/ALLTOALL and not a direct-gather multimem trap. Same packed-KV PyNCCL kv_gather family as tip bed9f1ce c40 / 860c1ccf c48; do not re-enable KV/Q gather (multimem traps on f7be12ed/db6caecc). 1x=56 still insufficient at c56 so drop below 1x to 48 like c70. Keep no compact_group_io, no MC_MAX_MR_SIZE, cumem off." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 1x CONC (16) for CONC 16 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-48 max-num-seqs 1x, c56/c70 max-num-seqs 48). Tip ebc4eb499 fail-fast run 37134266476 agentic c16 (job 111250499353, Slurm 7062, gpu-13) had a real rail, Application startup complete at 17:29:18Z, Mooncake failed_keys=0, and warmup returned 140/177 with 10 in flight from 17:44:47 at 0 tok/s while GPU KV stayed ~98-99.7% (Running: 3, Waiting: 7, Deferred: 7). First causal line is NCCL watchdog WorkNCCL SeqNum=89905 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 on PG ID 3 Rank 3 at 17:54:46 → DistBackendError / VllmWorker-7 died / EngineDead / ProfileAborted (176 warmup dropped / 0 kept; profiling never started). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same packed-KV PyNCCL _ALLGATHER_BASE family as tip fd61acb0 c56; not a direct-gather multimem trap and not setup. 2x=32 at c16 is the remaining 2x cell besides c8/c24; drop c16 to 1x=16. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From f3997a93aaea2663ead720910a9fbb178d006c64 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sat, 3 Oct 2026 21:35:20 +0000 Subject: [PATCH 27/31] fix(kimik3-b300): cap c70 max-num-seqs at 32 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip c9ffe2b12 fail-fast 37143203751 agentic c70 hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE under max-num-seqs=48; drop admission to 32. Keep gathers off. 将 B300 DSXE 上 CONC 70 的 max-num-seqs 从 48 降到 32。tip c9ffe2b12 fail-fast 运行 37143203751 的 agentic c70 在 profiling 中挂死于 PyNCCL kv_gather _ALLGATHER_BASE;保持 KV/Q gather 关闭。 Co-authored-by: Wenyao Gao --- failure-recovery-37143203751-c70.md | 63 +++++++++++++++++++ .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 42 +++++++------ inferencex-e2e/perf-changelog.yaml | 8 +++ 3 files changed, 95 insertions(+), 18 deletions(-) create mode 100644 failure-recovery-37143203751-c70.md diff --git a/failure-recovery-37143203751-c70.md b/failure-recovery-37143203751-c70.md new file mode 100644 index 0000000000..d87ddcd9cd --- /dev/null +++ b/failure-recovery-37143203751-c70.md @@ -0,0 +1,63 @@ +# Failure recovery — Run Sweep 37143203751 agentic c70 + +## Class + +**recipe** (admission): `override_c70` `max-num-seqs` too high for DCP PyNCCL `kv_gather` under AgentX c70 on B300 DSXE. + +## Evidence + +| Field | Value | +| --- | --- | +| Tip SHA | `c9ffe2b12d30d909b0e3ef97cdb0010ce6cb38d2` | +| Run | [37143203751](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37143203751) attempt 1 RED | +| Job | `111276873706` (agentic c70) | +| Slurm | `7087` on `b300-dsxe_07` / `dsxe-sa-b300-prd0-gpu-15` | +| server_logs artifact | `11285248802` (`server_logs_kimik3_tp8_conc70_...`) | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=48`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileAborted`, `worker_crash:8`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `20:01:07` (false positive). +- Real rail: `Mooncake rail: ibp198s0f0` / `Patched device_name='ibp198s0f0'`. +- `Application startup complete`; Mooncake `failed_keys=0` through serve. +- Warmup completed `772/774`; profiling started `21:06:27` and kept `115` requests (not warmup-only die). + +### First real kill boundary + +**Watchdog `_ALLGATHER_BASE` in `kv_gather` (dcp.py:1413)** at `21:17:09`: + +```text +[Rank 7] Watchdog caught collective operation timeout: +WorkNCCL(SeqNum=353066, OpType=_ALLGATHER_BASE, NumelIn=3142656, +NumelOut=25141248, Timeout(ms)=600000) ran for 600016 ms +PG ID 3: last enqueued work: 353081, last started work: -1, +last completed work: 353065 +stack: all_gather_into_tensor → kv_gather (dcp.py:1413) → +_context_parallel_compute_prefill_context → _forward_prefill_fused +→ DistBackendError / terminate / EngineDead / ProfileAborted +``` + +Hang window starts ~`21:07:09` (600s before watchdog). Last healthy engine line then stall: + +- `21:07:12` Running: 10, Waiting: 2, Deferred: 2, GPU KV: **21.3%**, gen 232 tok/s +- `21:07:22` Running: 10, Waiting: 2, Deferred: 2, GPU KV: 21.3%, gen **0.0** tok/s + +No multimem timeout, no CUDA OOM, no `sample_tokens` timeout. + +### Same-run packing (context, not the kill line) + +Earlier in the same profile, admission still packed the pool: max GPU KV **99.6%** at `21:00:12` (Running: 0, Waiting: 48, Deferred: 46); 99 windows with Running≈0 and KV≥85% from `20:38:42`–`21:01:32`. Engine recovered and resumed generating before the ALLGATHER hang. Distinct from tip `1d937c73e` c70 deferred-admission starve under `max-num-seqs=70` (0 kept, `sample_tokens` 1800s). + +Same PyNCCL `kv_gather` family as tip `fd61acb0` c56 / `ebc4eb499` c16. + +## Fix + +Smallest supported knob only: + +- `override_c70.max-num-seqs`: **48 → 32** +- Match `cudagraph_capture_sizes` to max 32 (same shape as `override_c32`) +- Do **not** re-enable `KV_GATHER` / `Q_GATHER` +- Do **not** set `compact_group_io`, `MC_MAX_MR_SIZE`, or `load_async=false` +- Do **not** touch #3632 + +`origin/main` already merged (0 behind); no merge required for this push. diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index cc6f742e48..3101195a7d 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -148,11 +148,11 @@ base: TOTAL_CPU_DRAM_GB: '2249' # One variant per point. Admission is 2x CONC through CONC 8 and CONC 24; CONC -# 16 and CONC 32–48 use 1x CONC; c56 and c70 cap at 48. Tip ebc4eb499 c16 -# (run 37134266476, Slurm 7062) under 2x=32 packed GPU KV to ~98–99.7% during -# AgentX warmup (Running: 3, Waiting: 7, Deferred: 7 at 17:44:47) then hung -# PyNCCL _ALLGATHER_BASE 600s (SeqNum=89905, last started work: -1, PG ID 3) -# → DistBackendError / VllmWorker-7 died / EngineDead / ProfileAborted +# 16 and CONC 32–48 use 1x CONC; c56 caps at 48; c70 caps at 32. Tip ebc4eb499 +# c16 (run 37134266476, Slurm 7062) under 2x=32 packed GPU KV to ~98–99.7% +# during AgentX warmup (Running: 3, Waiting: 7, Deferred: 7 at 17:44:47) then +# hung PyNCCL _ALLGATHER_BASE 600s (SeqNum=89905, last started work: -1, PG ID +# 3) → DistBackendError / VllmWorker-7 died / EngineDead / ProfileAborted # (176 warmup dropped / 0 kept). Cap c16 at 1x=16. Tip 031de17bf c56 # (Slurm 6848) with A2A/Q/KV all direct, util 0.85, and # max-num-seqs=112 packed GPU KV to ~99.7% (Running: 1, Waiting: 12, Deferred: @@ -169,13 +169,16 @@ base: # decode. Tip fd61acb0 c56 (run 37118728189, Slurm 7040) still packed GPU KV # to 98.2% at 1x=56 (Running: 8, Waiting: 53, Deferred: 38) then hung PyNCCL # kv_gather _ALLGATHER_BASE (dcp.py:1413, last started work: -1) for 600s. -# Cap c56 at 48 like c70. CONC 8+ use gpu-memory-utilization 0.85: with -# VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 (Slurm 6737) -# over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once CUDA graphs -# are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB needed, -# ~2.3 GiB free) ~4m after Application startup. Tip 5ab41690 c8 (Slurm 6877) -# then soft-OOMed at util 0.92 during warmup (CUDACachingAllocator ~3.03 GiB -# alloc with ~1.16 GiB free on ranks 1–7) → empty streamed responses → +# Cap c56 at 48. Tip c9ffe2b12 c70 (run 37143203751, Slurm 7087) under +# max-num-seqs=48 still hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE +# (SeqNum=353066, last started work: -1) after packing GPU KV to 99.6% earlier +# in the same profile; cap c70 at 32. CONC 8+ use gpu-memory-utilization 0.85: +# with VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 (Slurm +# 6737) over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once CUDA +# graphs are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB +# needed, ~2.3 GiB free) ~4m after Application startup. Tip 5ab41690 c8 +# (Slurm 6877) then soft-OOMed at util 0.92 during warmup (CUDACachingAllocator +# ~3.03 GiB alloc with ~1.16 GiB free on ranks 1–7) → empty streamed responses → # InvalidInferenceResultError / ProfileAborted at 2/11 > 10% (engine stayed # up; ingest nccl_error:16 again init-only WARN). Keep 0.92 only on c1–c4. # Graphs capture every size up to max-num-seqs, then the larger powers of two @@ -298,7 +301,7 @@ override_c56: args: # Tip fd61acb0 c56 (Slurm 7040) under 1x (56) packed GPU KV to 98.2% # (Running: 8, Waiting: 53, Deferred: 38) then hung kv_gather - # _ALLGATHER_BASE 600s (last started work: -1). Cap at 48 like c70. + # _ALLGATHER_BASE 600s (last started work: -1). Cap at 48. max-num-seqs: 48 gpu-memory-utilization: 0.85 compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,64,128,256,512,1024,2048,4096,8192]}' @@ -311,12 +314,15 @@ override_c70: roles: agg: args: - # Tip 1d937c73e c70 under 1x (70) still deferred-admission starved: - # Running≈0 / Waiting≈66 / Deferred≈50–67 / KV≈86–91% for ~30m then - # sample_tokens 1800s timeout. Cap below 1x to leave decode headroom. - max-num-seqs: 48 + # Tip 1d937c73e c70 under 1x (70) deferred-admission starved + # (Running≈0 / Waiting≈66 / Deferred≈50–67 / KV≈86–91% → sample_tokens + # 1800s). Cap to 48 left decode headroom but tip c9ffe2b12 c70 (run + # 37143203751, Slurm 7087) still hung mid-profile in PyNCCL kv_gather + # _ALLGATHER_BASE (dcp.py:1413, SeqNum=353066, last started work: -1) + # after earlier packing GPU KV to 99.6%. Cap further to 32. + max-num-seqs: 32 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,64,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '70' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 6c77920732..2db6713f69 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9491,3 +9491,11 @@ description: - "Cap max-num-seqs at 1x CONC (16) for CONC 16 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 32-48 max-num-seqs 1x, c56/c70 max-num-seqs 48). Tip ebc4eb499 fail-fast run 37134266476 agentic c16 (job 111250499353, Slurm 7062, gpu-13) had a real rail, Application startup complete at 17:29:18Z, Mooncake failed_keys=0, and warmup returned 140/177 with 10 in flight from 17:44:47 at 0 tok/s while GPU KV stayed ~98-99.7% (Running: 3, Waiting: 7, Deferred: 7). First causal line is NCCL watchdog WorkNCCL SeqNum=89905 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 on PG ID 3 Rank 3 at 17:54:46 → DistBackendError / VllmWorker-7 died / EngineDead / ProfileAborted (176 warmup dropped / 0 kept; profiling never started). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same packed-KV PyNCCL _ALLGATHER_BASE family as tip fd61acb0 c56; not a direct-gather multimem trap and not setup. 2x=32 at c16 is the remaining 2x cell besides c8/c24; drop c16 to 1x=16. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 32 for CONC 70 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/32-48 max-num-seqs 1x, c56 max-num-seqs 48). Tip c9ffe2b12 fail-fast run 37143203751 agentic c70 (job 111276873706, Slurm 7087, gpu-15) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, Mooncake failed_keys=0, warmup completed 772/774, and profiling kept 115 requests. During profile GPU KV packed to 99.6% (Running: 0, Waiting: 48, Deferred: 46 at 21:00:12); after recovery the first causal kill is NCCL watchdog WorkNCCL SeqNum=353066 OpType=_ALLGATHER_BASE Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 21:17:09 (hang from ~21:07:09 with Running: 10, Waiting: 2, Deferred: 2, GPU KV 21.3%) → DistBackendError / worker_crash:8 / EngineDead / ProfileAborted (82/115 InvalidInferenceResultError). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same PyNCCL kv_gather family as tip fd61acb0 c56 / ebc4eb499 c16; not multimem trap, not sample_tokens deferred-admission starve (distinct from tip 1d937c73e c70 at max-num-seqs=70). Cap 48 still insufficient at c70 so drop to 32. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 71080871410ca24104005a5c4df313430726c1b5 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sun, 4 Oct 2026 00:12:16 +0000 Subject: [PATCH 28/31] fix(kimik3-b300): cap c24 max-num-seqs at 1x MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip f3997a93 fail-fast 37155636981 agentic c24 hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE under max-num-seqs=48; drop admission to 24. Keep gathers off. 将 B300 DSXE 上 CONC 24 的 max-num-seqs 从 48 降到 24。tip f3997a93 fail-fast 运行 37155636981 的 agentic c24 在 profiling 中挂死于 PyNCCL kv_gather _ALLGATHER_BASE;保持 KV/Q gather 关闭。 Co-authored-by: Wenyao Gao --- failure-recovery-37155636981-c24.md | 66 ++++++++++++++++ .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 75 ++++++++++--------- inferencex-e2e/perf-changelog.yaml | 8 ++ 3 files changed, 115 insertions(+), 34 deletions(-) create mode 100644 failure-recovery-37155636981-c24.md diff --git a/failure-recovery-37155636981-c24.md b/failure-recovery-37155636981-c24.md new file mode 100644 index 0000000000..8344683347 --- /dev/null +++ b/failure-recovery-37155636981-c24.md @@ -0,0 +1,66 @@ +# Failure recovery — Run Sweep 37155636981 agentic c24 + +## Class + +**recipe** (admission): `override_c24` `max-num-seqs` too high (2×=48) for DCP PyNCCL `kv_gather` under AgentX c24 on B300 DSXE. + +## Evidence + +| Field | Value | +| --- | --- | +| Tip SHA | `f3997a93aaea2663ead720910a9fbb178d006c64` | +| Run | [37155636981](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37155636981) attempt 1 RED | +| Job | `111311962571` (agentic c24) | +| Slurm | `7107` on `b300-dsxe_07` / `dsxe-sa-b300-prd0-gpu-12` | +| server_logs artifact | `11288576572` (`server_logs_kimik3_tp8_conc24_...`) | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=48`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileAborted`, `worker_crash:8`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `23:05:03` (false positive). +- Real rail / Mooncake path healthy through serve; `failed_keys=0` in KV transfer metrics. +- `Application startup complete`; startup args confirm `max_num_seqs: 48`. +- Profiling kept `730/1077` (265 warmup, 288 error dropped) before kill; not warmup-only die. + +### First real kill boundary + +**Watchdog `_ALLGATHER_BASE` in `kv_gather` (dcp.py:1413)** at `23:56:57`: + +```text +[Rank 1] Watchdog caught collective operation timeout: +WorkNCCL(SeqNum=224109, OpType=_ALLGATHER_BASE, NumelIn=4755456, +NumelOut=38043648, Timeout(ms)=600000) ran for 600006 ms +PG ID 3: last enqueued work: 224148, last started work: -1, +last completed work: 224108 +stack: all_gather_into_tensor → kv_gather (dcp.py:1413) → +_context_parallel_compute_prefill_context → _forward_prefill_fused +→ DistBackendError / terminate / worker_crash:8 / EngineDead / +ProfileAborted (82/812 = 10.099%) +``` + +Hang window starts ~`23:46:57` (600s before watchdog). Last healthy then stall: + +- `23:46:50` Running: 22, Waiting: 0, GPU KV: **78.8%**, gen 202 tok/s +- `23:47:00` Running: 17, Waiting: 0, GPU KV: 75.7%, gen 402 tok/s +- `23:47:10` Running: 17, Waiting: 0, GPU KV: 75.7%, gen **0.0** tok/s + +No multimem timeout, no CUDA OOM, no `sample_tokens` timeout. Do not re-enable KV_GATHER or Q_GATHER. + +### Same-run packing (context) + +Earlier in the same profile, peak GPU KV **91.1%** at `23:41:20` (Running: 19, Waiting: 1, Deferred: 1). Aggregate result also reported GPU KV usage **93.3%**. c24 was the last mid-conc still at 2× admission. + +Same PyNCCL `kv_gather` family as tip `c9ffe2b12` c70 / `fd61acb0` c56 / `ebc4eb499` c16. + +## Fix + +Smallest supported knob only: + +- `override_c24.max-num-seqs`: **48 → 24** +- Match `cudagraph_capture_sizes` to max 24 +- Header comment: admission is 2× through CONC 8 only; CONC 16–48 use 1× +- Keep A2A=1; KV_GATHER=0; Q_GATHER=0; no compact_group_io / MC_MAX_MR_SIZE + +## New tip + +(filled after push) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index 3101195a7d..a6908f82a7 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -147,38 +147,42 @@ base: KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2249' -# One variant per point. Admission is 2x CONC through CONC 8 and CONC 24; CONC -# 16 and CONC 32–48 use 1x CONC; c56 caps at 48; c70 caps at 32. Tip ebc4eb499 -# c16 (run 37134266476, Slurm 7062) under 2x=32 packed GPU KV to ~98–99.7% -# during AgentX warmup (Running: 3, Waiting: 7, Deferred: 7 at 17:44:47) then -# hung PyNCCL _ALLGATHER_BASE 600s (SeqNum=89905, last started work: -1, PG ID -# 3) → DistBackendError / VllmWorker-7 died / EngineDead / ProfileAborted -# (176 warmup dropped / 0 kept). Cap c16 at 1x=16. Tip 031de17bf c56 -# (Slurm 6848) with A2A/Q/KV all direct, util 0.85, and -# max-num-seqs=112 packed GPU KV to ~99.7% (Running: 1, Waiting: 12, Deferred: -# 12) then hung workers for the full VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 — -# shm_broadcast starvation from 15:28, then TimeoutError: RPC call to -# sample_tokens timed out (no Watchdog / ALLGATHER / ALLTOALL / CUDA OOM; ingest -# nccl_error:16 was init-only ibv_query_port_speed WARN). Tip 861a1512 c32 -# (Slurm 6916) repeated the same kill under max-num-seqs=64 (2x): GPU KV pinned -# ~99–100% for tens of minutes, then hung at 0 tok/s (Running: 23, Waiting: 11, -# Deferred: 11) with shm_broadcast starvation from 21:03 through the 1800s -# sample_tokens timeout → EngineDead / ProfileAborted (20 empty streams + 62 -# ClientConnectorError after shutdown; ingest nccl_error:16 again init-only). -# Cap CONC 32+ admission so prefix+deferred loads cannot pin the pool before -# decode. Tip fd61acb0 c56 (run 37118728189, Slurm 7040) still packed GPU KV -# to 98.2% at 1x=56 (Running: 8, Waiting: 53, Deferred: 38) then hung PyNCCL -# kv_gather _ALLGATHER_BASE (dcp.py:1413, last started work: -1) for 600s. -# Cap c56 at 48. Tip c9ffe2b12 c70 (run 37143203751, Slurm 7087) under -# max-num-seqs=48 still hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE -# (SeqNum=353066, last started work: -1) after packing GPU KV to 99.6% earlier -# in the same profile; cap c70 at 32. CONC 8+ use gpu-memory-utilization 0.85: -# with VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 (Slurm -# 6737) over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once CUDA -# graphs are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB -# needed, ~2.3 GiB free) ~4m after Application startup. Tip 5ab41690 c8 -# (Slurm 6877) then soft-OOMed at util 0.92 during warmup (CUDACachingAllocator -# ~3.03 GiB alloc with ~1.16 GiB free on ranks 1–7) → empty streamed responses → +# One variant per point. Admission is 2x CONC through CONC 8; CONC 16–48 use +# 1x CONC; c56 caps at 48; c70 caps at 32. Tip ebc4eb499 c16 (run 37134266476, +# Slurm 7062) under 2x=32 packed GPU KV to ~98–99.7% during AgentX warmup +# (Running: 3, Waiting: 7, Deferred: 7 at 17:44:47) then hung PyNCCL +# _ALLGATHER_BASE 600s (SeqNum=89905, last started work: -1, PG ID 3) → +# DistBackendError / VllmWorker-7 died / EngineDead / ProfileAborted (176 +# warmup dropped / 0 kept). Cap c16 at 1x=16. Tip f3997a93 c24 (run +# 37155636981, Slurm 7107) under 2x=48 hung mid-profile in PyNCCL kv_gather +# _ALLGATHER_BASE (dcp.py:1413, SeqNum=224109, last started work: -1) after +# peak GPU KV 91.1% earlier in the same profile (Running: 17–25 near hang); +# cap c24 at 1x=24. Tip 031de17bf c56 (Slurm 6848) with A2A/Q/KV all direct, +# util 0.85, and max-num-seqs=112 packed GPU KV to ~99.7% (Running: 1, +# Waiting: 12, Deferred: 12) then hung workers for the full +# VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 — shm_broadcast starvation from +# 15:28, then TimeoutError: RPC call to sample_tokens timed out (no Watchdog / +# ALLGATHER / ALLTOALL / CUDA OOM; ingest nccl_error:16 was init-only +# ibv_query_port_speed WARN). Tip 861a1512 c32 (Slurm 6916) repeated the same +# kill under max-num-seqs=64 (2x): GPU KV pinned ~99–100% for tens of minutes, +# then hung at 0 tok/s (Running: 23, Waiting: 11, Deferred: 11) with +# shm_broadcast starvation from 21:03 through the 1800s sample_tokens timeout +# → EngineDead / ProfileAborted (20 empty streams + 62 ClientConnectorError +# after shutdown; ingest nccl_error:16 again init-only). Cap CONC 32+ admission +# so prefix+deferred loads cannot pin the pool before decode. Tip fd61acb0 c56 +# (run 37118728189, Slurm 7040) still packed GPU KV to 98.2% at 1x=56 +# (Running: 8, Waiting: 53, Deferred: 38) then hung PyNCCL kv_gather +# _ALLGATHER_BASE (dcp.py:1413, last started work: -1) for 600s. Cap c56 at +# 48. Tip c9ffe2b12 c70 (run 37143203751, Slurm 7087) under max-num-seqs=48 +# still hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE (SeqNum=353066, +# last started work: -1) after packing GPU KV to 99.6% earlier in the same +# profile; cap c70 at 32. CONC 8+ use gpu-memory-utilization 0.85: with +# VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0, tip b70e4260a c48 (Slurm 6737) +# over-allocated GPU KV to 45.2 GiB (vLLM suggested 36.78 GiB once CUDA graphs +# are counted) and OOMed in flashinfer FP4 MoE prepare_moe (~2.89 GiB needed, +# ~2.3 GiB free) ~4m after Application startup. Tip 5ab41690 c8 (Slurm 6877) +# then soft-OOMed at util 0.92 during warmup (CUDACachingAllocator ~3.03 GiB +# alloc with ~1.16 GiB free on ranks 1–7) → empty streamed responses → # InvalidInferenceResultError / ProfileAborted at 2/11 > 10% (engine stayed # up; ingest nccl_error:16 again init-only WARN). Keep 0.92 only on c1–c4. # Graphs capture every size up to max-num-seqs, then the larger powers of two @@ -251,9 +255,12 @@ override_c24: roles: agg: args: - max-num-seqs: 48 + # Tip f3997a93 c24 (Slurm 7107) under 2x=48 hung mid-profile in + # PyNCCL kv_gather _ALLGATHER_BASE 600s (dcp.py:1413, SeqNum=224109, + # last started work: -1) after peak GPU KV 91.1% earlier. + max-num-seqs: 24 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,64,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '24' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 2db6713f69..4c48992323 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9499,3 +9499,11 @@ description: - "Cap max-num-seqs at 32 for CONC 70 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/32-48 max-num-seqs 1x, c56 max-num-seqs 48). Tip c9ffe2b12 fail-fast run 37143203751 agentic c70 (job 111276873706, Slurm 7087, gpu-15) had a real rail (Mooncake rail: ibp198s0f0 / Patched), Application startup complete, Mooncake failed_keys=0, warmup completed 772/774, and profiling kept 115 requests. During profile GPU KV packed to 99.6% (Running: 0, Waiting: 48, Deferred: 46 at 21:00:12); after recovery the first causal kill is NCCL watchdog WorkNCCL SeqNum=353066 OpType=_ALLGATHER_BASE Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 21:17:09 (hang from ~21:07:09 with Running: 10, Waiting: 2, Deferred: 2, GPU KV 21.3%) → DistBackendError / worker_crash:8 / EngineDead / ProfileAborted (82/115 InvalidInferenceResultError). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same PyNCCL kv_gather family as tip fd61acb0 c56 / ebc4eb499 c16; not multimem trap, not sample_tokens deferred-admission starve (distinct from tip 1d937c73e c70 at max-num-seqs=70). Cap 48 still insufficient at c70 so drop to 32. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 1x CONC (24) for CONC 24 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/32-48 max-num-seqs 1x, c56 max-num-seqs 48, c70 max-num-seqs 32). Tip f3997a93 fail-fast run 37155636981 agentic c24 (job 111311962571, Slurm 7107, gpu-12) had a real rail, Application startup complete, Mooncake failed_keys=0, warmup progressed, and profiling kept 730/1077 (265 warmup, 288 error dropped). Peak GPU KV 91.1% at 23:41:20 (Running: 19, Waiting: 1, Deferred: 1). First causal kill is NCCL watchdog WorkNCCL SeqNum=224109 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 23:56:57 → DistBackendError / worker_crash:8 / EngineDead / ProfileAborted (82/812 = 10.099%). Hang from ~23:46:57 with Running: 17–25 and GPU KV ~76–84%. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same PyNCCL kv_gather family as tip c9ffe2b12 c70 / fd61acb0 c56 / ebc4eb499 c16; c24 was the last mid-conc still at 2x. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 From 767f0b3b2eefc9c1058c52c87156fd4be7678882 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sat, 3 Oct 2026 20:06:36 -0700 Subject: [PATCH 29/31] fix(kimik3-b300): cap c40 max-num-seqs at 32 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip 71080871 fail-fast 37164230642 agentic c40 hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE under max-num-seqs=40; drop admission to 32. Keep gathers off. 将 B300 DSXE 上 CONC 40 的 max-num-seqs 从 40 降到 32。tip 71080871 fail-fast 运行 37164230642 的 agentic c40 在 profiling 中挂死于 PyNCCL kv_gather _ALLGATHER_BASE;保持 KV/Q gather 关闭。 Co-authored-by: Wenyao Gao --- failure-recovery-37164230642-c40.md | 63 +++++++++++++++++++ .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 18 ++++-- inferencex-e2e/perf-changelog.yaml | 9 +++ 3 files changed, 85 insertions(+), 5 deletions(-) create mode 100644 failure-recovery-37164230642-c40.md diff --git a/failure-recovery-37164230642-c40.md b/failure-recovery-37164230642-c40.md new file mode 100644 index 0000000000..93268ab59a --- /dev/null +++ b/failure-recovery-37164230642-c40.md @@ -0,0 +1,63 @@ +# Failure recovery — Run Sweep 37164230642 agentic c40 + +## Class + +**recipe** (admission): `override_c40` `max-num-seqs` too high (1×=40) for DCP PyNCCL `kv_gather` under AgentX c40 on B300 DSXE. + +## Evidence + +| Field | Value | +| --- | --- | +| Tip SHA | `71080871410ca24104005a5c4df313430726c1b5` | +| Run | [37164230642](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37164230642) attempt 1 RED | +| Job | `111336354457` (agentic c40) | +| Slurm | `7123` on `b300-dsxe_06` / `dsxe-sa-b300-prd0-gpu-09` | +| server_logs artifact | `11292012067` (`server_logs_kimik3_tp8_conc40_...`) | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=40`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileAborted`, `worker_crash:8`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `01:41:33` (false positive). +- Real rail / Mooncake path healthy through serve. +- `Application startup complete`; runtime args confirm `--max-num-seqs 40`. +- Profiling kept `838/1376` (444 warmup, 449 error dropped) before kill; not warmup-only die. +- Canary + all agentic evals SUCCESS on this tip; sole agentic fail = c40 (fail-fast cancelled siblings). + +### First real kill boundary + +**Watchdog `_ALLGATHER_BASE` in `kv_gather` (dcp.py:1413)** at `02:49:41`: + +```text +[Rank 7] Watchdog caught collective operation timeout: +WorkNCCL(SeqNum=345118, OpType=_ALLGATHER_BASE, NumelIn=4755456, +NumelOut=38043648, Timeout(ms)=600000) ran for 600000 ms +PG ID 3: last enqueued work: 345144, last started work: -1, +last completed work: 345117 +stack: all_gather_into_tensor → kv_gather (dcp.py:1413) → +DistBackendError / terminate / worker_crash:8 / ProfileAborted +(94/932 = 10.086%) +``` + +Hang window starts ~`02:39:41` (600s before watchdog). Last healthy then stall: + +- `02:39:48` Running: 17, Waiting: 20, Deferred: 20, GPU KV: **71.5%**, gen 200 tok/s +- `02:39:58` Running: 17, Waiting: 20, Deferred: 20, GPU KV: 71.5%, gen **0.0** tok/s + +Earlier packing in the same profile: peak GPU KV **~99.9%** with elevated Waiting/Deferred (e.g. Running: 2 / Waiting: 35 / Deferred: 21 at `02:33:28`; Running: 7 / Waiting: 27 / Deferred: 23 at `02:38:38`). Aggregate result also reported GPU KV usage **100.0%**. + +No multimem timeout, no CUDA OOM, no `sample_tokens` timeout. Do not re-enable KV_GATHER or Q_GATHER. + +Same PyNCCL `kv_gather` family as tip `f3997a93` c24 / `c9ffe2b12` c70 / `fd61acb0` c56 / `ebc4eb499` c16. Tip already at 1× CONC for c40; further admission cut required (same pattern as c56 56→48 and c70 48→32). + +## Fix + +Smallest supported knob only: + +- `override_c40.max-num-seqs`: **40 → 32** +- Match `cudagraph_capture_sizes` to max 32 +- Header comment: note c40 caps at 32 (not 1×) +- Keep A2A=1; KV_GATHER=0; Q_GATHER=0; no compact_group_io / MC_MAX_MR_SIZE + +## New tip + +(filled after push) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index a6908f82a7..e0313956cb 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -147,8 +147,8 @@ base: KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2249' -# One variant per point. Admission is 2x CONC through CONC 8; CONC 16–48 use -# 1x CONC; c56 caps at 48; c70 caps at 32. Tip ebc4eb499 c16 (run 37134266476, +# One variant per point. Admission is 2x CONC through CONC 8; CONC 16/24/32/48 +# use 1x CONC; c40 caps at 32; c56 caps at 48; c70 caps at 32. Tip ebc4eb499 c16 (run 37134266476, # Slurm 7062) under 2x=32 packed GPU KV to ~98–99.7% during AgentX warmup # (Running: 3, Waiting: 7, Deferred: 7 at 17:44:47) then hung PyNCCL # _ALLGATHER_BASE 600s (SeqNum=89905, last started work: -1, PG ID 3) → @@ -157,7 +157,12 @@ base: # 37155636981, Slurm 7107) under 2x=48 hung mid-profile in PyNCCL kv_gather # _ALLGATHER_BASE (dcp.py:1413, SeqNum=224109, last started work: -1) after # peak GPU KV 91.1% earlier in the same profile (Running: 17–25 near hang); -# cap c24 at 1x=24. Tip 031de17bf c56 (Slurm 6848) with A2A/Q/KV all direct, +# cap c24 at 1x=24. Tip 71080871 c40 (run 37164230642, Slurm 7123) under +# 1x=40 packed GPU KV to ~99.9% (Running: 2–30, Waiting: 16–35, Deferred: +# 11–34) then hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE +# (dcp.py:1413, SeqNum=345118, last started work: -1) → DistBackendError / +# worker_crash:8 / ProfileAborted (94/932 = 10.086%); cap c40 at 32. Tip +# 031de17bf c56 (Slurm 6848) with A2A/Q/KV all direct, # util 0.85, and max-num-seqs=112 packed GPU KV to ~99.7% (Running: 1, # Waiting: 12, Deferred: 12) then hung workers for the full # VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS=1800 — shm_broadcast starvation from @@ -282,9 +287,12 @@ override_c40: roles: agg: args: - max-num-seqs: 40 + # Tip 71080871 c40 (Slurm 7123) under 1x=40 packed GPU KV to ~99.9% + # then hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE 600s + # (dcp.py:1413, SeqNum=345118, last started work: -1). Cap at 32. + max-num-seqs: 32 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,64,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '40' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 4c48992323..296f63ace5 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9507,3 +9507,12 @@ description: - "Cap max-num-seqs at 1x CONC (24) for CONC 24 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/32-48 max-num-seqs 1x, c56 max-num-seqs 48, c70 max-num-seqs 32). Tip f3997a93 fail-fast run 37155636981 agentic c24 (job 111311962571, Slurm 7107, gpu-12) had a real rail, Application startup complete, Mooncake failed_keys=0, warmup progressed, and profiling kept 730/1077 (265 warmup, 288 error dropped). Peak GPU KV 91.1% at 23:41:20 (Running: 19, Waiting: 1, Deferred: 1). First causal kill is NCCL watchdog WorkNCCL SeqNum=224109 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 23:56:57 → DistBackendError / worker_crash:8 / EngineDead / ProfileAborted (82/812 = 10.099%). Hang from ~23:46:57 with Running: 17–25 and GPU KV ~76–84%. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same PyNCCL kv_gather family as tip c9ffe2b12 c70 / fd61acb0 c56 / ebc4eb499 c16; c24 was the last mid-conc still at 2x. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 32 for CONC 40 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/24/32/48 max-num-seqs 1x, c56 max-num-seqs 48, c70 max-num-seqs 32). Tip 71080871 fail-fast run 37164230642 agentic c40 (job 111336354457, Slurm 7123, b300-dsxe_06 / gpu-09) had a real rail, Application startup complete, Mooncake path healthy, warmup progressed, and profiling kept 838/1376 (444 warmup, 449 error dropped). Peak GPU KV ~99.9% with Waiting/Deferred queues elevated (e.g. Running: 2, Waiting: 35, Deferred: 21 at 02:33:28; Running: 7, Waiting: 27, Deferred: 23 at 02:38:38). First causal kill is NCCL watchdog WorkNCCL SeqNum=345118 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 02:49:41 → DistBackendError / worker_crash:8 / ProfileAborted (94/932 = 10.086%). Hang from ~02:39:41 with Running: 17, Waiting: 20, Deferred: 20, gen 0 tok/s. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same PyNCCL kv_gather family as tip f3997a93 c24 / c9ffe2b12 c70 / fd61acb0 c56 / ebc4eb499 c16; 1x=40 still insufficient at c40 so drop to 32. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + From cefbd21826df0eee931eeaf8664660b642388f7e Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sun, 4 Oct 2026 05:58:46 +0000 Subject: [PATCH 30/31] fix(kimik3-b300): cap c48 max-num-seqs at 32 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tip 767f0b3b2 fail-fast 37173109032 agentic c48 hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE under max-num-seqs=48; drop admission to 32. Keep gathers off. 将 B300 DSXE 上 CONC 48 的 max-num-seqs 从 48 降到 32。tip 767f0b3b2 fail-fast 运行 37173109032 的 agentic c48 在 profiling 中挂死于 PyNCCL kv_gather _ALLGATHER_BASE;保持 KV/Q gather 关闭。 Co-authored-by: Wenyao Gao --- failure-recovery-37173109032-c48.md | 65 +++++++++++++++++++ .../kimik3/vllm/b300-fp4-mtp/agentic.yaml | 15 +++-- inferencex-e2e/perf-changelog.yaml | 8 +++ 3 files changed, 84 insertions(+), 4 deletions(-) create mode 100644 failure-recovery-37173109032-c48.md diff --git a/failure-recovery-37173109032-c48.md b/failure-recovery-37173109032-c48.md new file mode 100644 index 0000000000..f6b749cbb8 --- /dev/null +++ b/failure-recovery-37173109032-c48.md @@ -0,0 +1,65 @@ +# Failure recovery — Run Sweep 37173109032 agentic c48 + +## Class + +**recipe** (admission): `override_c48` `max-num-seqs` too high (1×=48) for DCP PyNCCL `kv_gather` under AgentX c48 on B300 DSXE. + +## Evidence + +| Field | Value | +| --- | --- | +| Tip SHA | `767f0b3b2eefc9c1058c52c87156fd4be7678882` | +| Run | [37173109032](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/37173109032) attempt 1 RED | +| Job | `111362024256` (agentic c48) | +| Slurm | `7147` on `b300-dsxe_04` / `dsxe-sa-b300-prd0-gpu-10` | +| server_logs artifact | `11295365223` | +| Knobs on tip | A2A=1, KV_GATHER=0, Q_GATHER=0, `max-num-seqs=48`, util 0.85, `load_async=true`, `lookup_async=true`, `max_load_batch_keys=1` | + +Wrapper signals (`ProfileAborted`, `worker_crash:8`, `nccl_error:16`) are insufficient alone: + +- `nccl_error:16` is init-only `ibv_query_port_speed` WARN at `04:35:49` (`first_ts=last_ts`; false positive for the kill). +- Real rail / Mooncake path healthy through serve. +- `Application startup complete`; runtime args confirm `--max-num-seqs 48`. +- Profiling kept `452/1033` (531 warmup, 467 error dropped) before kill; not warmup-only die. +- SIGTERM only at shutdown after DistBackend (`05:45:00`), not the first boundary. + +### First real kill boundary + +**Watchdog `_ALLGATHER_BASE` in `kv_gather` (dcp.py:1413)** at `05:43:54`: + +```text +[Rank 3] Watchdog caught collective operation timeout: +WorkNCCL(SeqNum=334427, OpType=_ALLGATHER_BASE, NumelIn=4755456, +NumelOut=38043648, Timeout(ms)=600000) ran for 600017 ms +PG ID 3: last enqueued work: 334452, last started work: -1, +last completed work: 334426 +stack: all_gather_into_tensor → kv_gather (dcp.py:1413) → +DistBackendError / terminate / worker_crash:8 / EngineDead / +ProfileAborted (50/499 = 10.0%) +``` + +Hang window starts ~`05:33:54` (600s before watchdog). Last live then stall: + +- `05:33:18` Running: 10, Waiting: 39, Deferred: 28, GPU KV: **100.0%** +- `05:33:58` Running: 13, Waiting: 38, Deferred: 35, GPU KV: **98.1%**, gen 145.5 tok/s +- `05:34:08` Running: 13, Waiting: 38, Deferred: 35, GPU KV: 98.1%, gen **0.0** tok/s +- `05:34:56` first mid-profile `shm_broadcast` starvation line + +Earlier packing in the same profile also hit GPU KV **99.8%** / **99.7%**. No multimem timeout, no CUDA OOM, no `sample_tokens` timeout. Do not re-enable KV_GATHER or Q_GATHER. + +Same PyNCCL `kv_gather` family as tip `71080871` c40 / `f3997a93` c24 / `c9ffe2b12` c70 / `fd61acb0` c56 / `ebc4eb499` c16. Tip already at 1× CONC for c48; further admission cut required (same pattern as c40 40→32 and c70 48→32). + +Eval c32 xgrammar ISE on this ledger remains infra/flake (prior dig); left alone. + +## Fix + +Smallest supported knob only: + +- `override_c48.max-num-seqs`: **48 → 32** +- Match `cudagraph_capture_sizes` to max 32 +- Header comment: note c48 caps at 32 (not 1×) +- Keep A2A=1; KV_GATHER=0; Q_GATHER=0; no compact_group_io / MC_MAX_MR_SIZE + +## New tip + +(filled after push) diff --git a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml index e0313956cb..121b637bc2 100644 --- a/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml +++ b/inferencex-e2e/benchmarks/single_node/srt-slurm-recipes/kimik3/vllm/b300-fp4-mtp/agentic.yaml @@ -147,8 +147,8 @@ base: KV_OFFLOADING: dram TOTAL_CPU_DRAM_GB: '2249' -# One variant per point. Admission is 2x CONC through CONC 8; CONC 16/24/32/48 -# use 1x CONC; c40 caps at 32; c56 caps at 48; c70 caps at 32. Tip ebc4eb499 c16 (run 37134266476, +# One variant per point. Admission is 2x CONC through CONC 8; CONC 16/24/32 +# use 1x CONC; c40/c48/c70 cap at 32; c56 caps at 48. Tip ebc4eb499 c16 (run 37134266476, # Slurm 7062) under 2x=32 packed GPU KV to ~98–99.7% during AgentX warmup # (Running: 3, Waiting: 7, Deferred: 7 at 17:44:47) then hung PyNCCL # _ALLGATHER_BASE 600s (SeqNum=89905, last started work: -1, PG ID 3) → @@ -162,6 +162,10 @@ base: # 11–34) then hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE # (dcp.py:1413, SeqNum=345118, last started work: -1) → DistBackendError / # worker_crash:8 / ProfileAborted (94/932 = 10.086%); cap c40 at 32. Tip +# 767f0b3b2 c48 (run 37173109032, Slurm 7147) under 1x=48 packed GPU KV to +# 100.0% then hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE +# (dcp.py:1413, SeqNum=334427, last started work: -1) → DistBackendError / +# worker_crash:8 / ProfileAborted (50/499 = 10.0%); cap c48 at 32. Tip # 031de17bf c56 (Slurm 6848) with A2A/Q/KV all direct, # util 0.85, and max-num-seqs=112 packed GPU KV to ~99.7% (Running: 1, # Waiting: 12, Deferred: 12) then hung workers for the full @@ -302,9 +306,12 @@ override_c48: roles: agg: args: - max-num-seqs: 48 + # Tip 767f0b3b2 c48 (Slurm 7147) under 1x=48 packed GPU KV to 100.0% + # then hung mid-profile in PyNCCL kv_gather _ALLGATHER_BASE 600s + # (dcp.py:1413, SeqNum=334427, last started work: -1). Cap at 32. + max-num-seqs: 32 gpu-memory-utilization: 0.85 - compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,64,128,256,512,1024,2048,4096,8192]}' + compilation-config: '{"cudagraph_mode":"FULL_AND_PIECEWISE","cudagraph_capture_sizes":[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,64,128,256,512,1024,2048,4096,8192]}' benchmark: env: CONC: '48' diff --git a/inferencex-e2e/perf-changelog.yaml b/inferencex-e2e/perf-changelog.yaml index 296f63ace5..cc95c53ad0 100644 --- a/inferencex-e2e/perf-changelog.yaml +++ b/inferencex-e2e/perf-changelog.yaml @@ -9516,3 +9516,11 @@ - "Cap max-num-seqs at 32 for CONC 40 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/24/32/48 max-num-seqs 1x, c56 max-num-seqs 48, c70 max-num-seqs 32). Tip 71080871 fail-fast run 37164230642 agentic c40 (job 111336354457, Slurm 7123, b300-dsxe_06 / gpu-09) had a real rail, Application startup complete, Mooncake path healthy, warmup progressed, and profiling kept 838/1376 (444 warmup, 449 error dropped). Peak GPU KV ~99.9% with Waiting/Deferred queues elevated (e.g. Running: 2, Waiting: 35, Deferred: 21 at 02:33:28; Running: 7, Waiting: 27, Deferred: 23 at 02:38:38). First causal kill is NCCL watchdog WorkNCCL SeqNum=345118 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 02:49:41 → DistBackendError / worker_crash:8 / ProfileAborted (94/932 = 10.086%). Hang from ~02:39:41 with Running: 17, Waiting: 20, Deferred: 20, gen 0 tok/s. Ingest nccl_error:16 was init-only ibv_query_port_speed WARN. Same PyNCCL kv_gather family as tip f3997a93 c24 / c9ffe2b12 c70 / fd61acb0 c56 / ebc4eb499 c16; 1x=40 still insufficient at c40 so drop to 32. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 +- config-keys: + - kimik3-fp4-b300-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Cap max-num-seqs at 32 for CONC 48 on B300 DSXE (keep A2A=1, KV_GATHER=0, Q_GATHER=0, load_async=true, max_load_batch_keys=1, CONC 8+ util 0.85, CONC 16/24/32 max-num-seqs 1x, c40/c70 max-num-seqs 32, c56 max-num-seqs 48). Tip 767f0b3b2 fail-fast run 37173109032 agentic c48 (job 111362024256, Slurm 7147, b300-dsxe_04 / gpu-10) had a real rail, Application startup complete, Mooncake path healthy, warmup progressed, and profiling kept 452/1033 (531 warmup, 467 error dropped). Peak GPU KV 100.0% at 05:33:18 (Running: 10, Waiting: 39, Deferred: 28); hang onset ~05:33:54 with last live metrics at 05:33:58 (Running: 13, Waiting: 38, Deferred: 35, GPU KV 98.1%) then 0 tok/s. First causal kill is NCCL watchdog WorkNCCL SeqNum=334427 OpType=_ALLGATHER_BASE NumelIn=4755456 Timeout(ms)=600000 last started work: -1 in kv_gather (dcp.py:1413 all_gather_into_tensor) at 05:43:54 → DistBackendError / worker_crash:8 / EngineDead / ProfileAborted (50/499 = 10.0%). Ingest nccl_error:16 was init-only ibv_query_port_speed WARN (first_ts=last_ts=04:35:49). SIGTERM only at shutdown after DistBackend. Same PyNCCL kv_gather family as tip 71080871 c40 / f3997a93 c24 / c9ffe2b12 c70 / fd61acb0 c56 / ebc4eb499 c16; 1x=48 still insufficient at c48 so drop to 32. Do not re-enable KV/Q gather, compact_group_io, or MC_MAX_MR_SIZE. Keep cumem off." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3088 + From 141fbb8567c8dfc14616512fd74bc97806400ada Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sun, 4 Oct 2026 05:58:54 +0000 Subject: [PATCH 31/31] docs(kimik3-b300): record c48 recovery tip SHA MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fill failure-recovery-37173109032-c48.md with pushed tip cefbd21826df0eee931eeaf8664660b642388f7e. 在 failure-recovery-37173109032-c48.md 中记录已推送 tip SHA。 Co-authored-by: Wenyao Gao --- failure-recovery-37173109032-c48.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/failure-recovery-37173109032-c48.md b/failure-recovery-37173109032-c48.md index f6b749cbb8..9a6f600ada 100644 --- a/failure-recovery-37173109032-c48.md +++ b/failure-recovery-37173109032-c48.md @@ -62,4 +62,4 @@ Smallest supported knob only: ## New tip -(filled after push) +`cefbd21826df0eee931eeaf8664660b642388f7e` — `override_c48` `max-num-seqs` 48→32; gathers remain gated.